diff --git a/.github/workflows/main.yml b/.github/workflows/main.yml index dd0083fa..7068d60c 100644 --- a/.github/workflows/main.yml +++ b/.github/workflows/main.yml @@ -3,6 +3,10 @@ run-name: Native build by ${{ github.actor }} on: workflow_dispatch: workflow_call: + + # Runs on pushes to the default branch and on pull requests targeting it. + # A topic branch is gated through its pull request, so nothing beyond the + # default branch is enumerated here. pull_request: branches: - main diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index dc17c0d1..92a88214 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -3,6 +3,10 @@ run-name: CI by ${{ github.actor }} on: workflow_dispatch: + + # Runs on pushes to the default branch and on pull requests targeting it. + # A topic branch is gated through its pull request, so nothing beyond the + # default branch is enumerated here. pull_request: branches: - main @@ -67,6 +71,21 @@ jobs: cmake -B BUILD -DCMAKE_BUILD_TYPE=Release cmake --build BUILD --parallel $BUILD_JOBS --config Release + # The C++ suite. This is the only `ctest` invocation under + # .github/workflows/, so it is what makes the tree's add_test() + # registrations gate anything. + # + # This runner (ubuntu-24.04, GitHub-hosted) has NO NVIDIA GPU, so the + # tests are split by label: the device-free set must be green here and + # is the real gate, and the gpu set reports SKIPPED rather than FAILED + # when there is no device -- so it is green here too, and becomes a + # genuine gate the day this job moves to a GPU runner. The script also + # fails if any registered test carries neither label, and fails if a + # label selects zero tests, so the gate cannot quietly become a no-op. + - name: Run CTest suite + run: | + ./scripts/run_ctest_ci.sh BUILD + - name: Set up Python uses: actions/setup-python@v5 with: diff --git a/.gitignore b/.gitignore index 0c2d87b1..bb0069bd 100644 --- a/.gitignore +++ b/.gitignore @@ -49,6 +49,26 @@ BUILD* tests/resources/ tests/results/ +# Encoder bitstream output. A run that supplies no output path gets the +# encoder's default name -- out.264, out.265 or out.ivf by codec -- in the +# directory it was run from, so a binary invoked by hand from the tree root +# leaves one here. Test binaries run under CTest do not: each runs in its own +# directory inside the build tree, which BUILD* above already covers. +# +# The three default names, anchored to the root, and NOT the extensions: an +# elementary stream in these formats is also the shape a reference or golden +# bitstream takes, and one added under docs/ or tests/ is content the +# repository is meant to carry rather than an artifact of a build. +/out.264 +/out.265 +/out.ivf + # Generated into the build tree by cmake/VulkanDispatchTable.cmake common/libs/VkCodecUtils/HelpersDispatchTable.h common/libs/VkCodecUtils/HelpersDispatchTable.cpp + +# IDE / editor project state. Machine-local and not part of the source. +.settings/ +.vscode/ +.idea/ +*.code-workspace diff --git a/CMakeLists.txt b/CMakeLists.txt index ac987d40..91b4a6a0 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -25,6 +25,34 @@ set(CMAKE_POSITION_INDEPENDENT_CODE ON) set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} "${CMAKE_CURRENT_SOURCE_DIR}/cmake") set(SCRIPTS_DIR "${CMAKE_CURRENT_SOURCE_DIR}/scripts") +# vvs_add_validation_gated_test(), which registers a test whose +# validation-layer output is gating. Included here rather than per directory +# so that every subdirectory reaches one definition and one gate script. +include(VvsValidationGate) + +# CTest aggregation. Every add_test() in this tree lives in a subdirectory, +# and CMake writes a CTestTestfile.cmake at the build ROOT only when testing is +# enabled HERE -- so without this line `cd build && ctest` reports +# "Total Tests: 0" and exits 0 having run nothing, which is a CI step that can +# never fail. The subdirectories' own enable_testing() calls make them +# runnable when configured standalone; this one makes them runnable from the +# top, which is how CI configures the project. +# +# LABEL CONTRACT -- every add_test() in this tree MUST carry exactly one of: +# +# device-free Needs no Vulkan device. Runs and must pass on any host, +# including the GPU-less GitHub runner. This is the gating set. +# gpu Needs a real GPU. MUST also declare how it skips when no +# device is available, so that a GPU-less host reports SKIPPED +# and never FAILED: SKIP_RETURN_CODE for a plain add_test(), +# or vvs_add_validation_gated_test(), which declares the +# equivalent SKIP_REGULAR_EXPRESSION itself. A CI job red on +# arrival is a job people learn to ignore. +# +# scripts/run_ctest_ci.sh enforces the contract: it fails if any registered +# test is in neither set, and CI calls it. Add a test, add its label. +enable_testing() + # Common options option(BUILD_VIDEO_PARSER "Build the video parser" ON) option(BUILD_DECODER "Build the video decoder" ON) @@ -148,14 +176,55 @@ add_definitions( -DVK_USE_VIDEO_QUEUE -DVK_USE_VIDEO_DECODE_QUEUE -DVK_USE_VIDEO_ENCODE_QUEUE - # The encoder's preprocess compute filter (VulkanFilterYuvCompute) is built - # and enabled by default. Embedders that gate it - the Chromium vendored - # copy does - key off this macro, so define it here to keep the standalone - # build and those consumers in agreement. - -DVK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED -DAPI_NAME="${API_NAME}" ) +# Encoder preprocess compute filter (VulkanFilterYuvCompute). +# +# ON by default because upstream has no gate at all: the filter path there is +# unconditional and EncoderConfig::enablePreprocessComputeFilter defaults to +# true. Leaving the guard undefined silently drops the filter from +# vk-video-enc-test too, which is not what the guard is for. +# +# The guard is for embedders that cannot supply a GLSL compiler, which the +# filter needs to compile its generated GLSL at runtime +# (VulkanShaderCompiler.cpp). No such embedder remains: standalone builds get a +# backend from cmake/VulkanShaderCompilerBackend.cmake (glslang by default, +# shaderc via -DVK_VIDEO_SAMPLES_SHADER_BACKEND=shaderc), and Chromium defines +# the macro too, in third_party/vulkan_video_samples/BUILD.gn +# config("private_includes"). +# +# A COMPILE-TIME switch, and a real one. VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED +# is consulted by guards in VkEncoderConfig.h, VkVideoEncoder.{h,cpp}, +# vulkan_video_encoder_ext.cpp and the encoder-ext filter test. OFF drops the +# preprocess compute filter's code, and with it the EncoderConfig fields +# filterType and enablePreprocessComputeFilter -- which is what lets an embedder +# with no GLSL compiler backend build the encoder at all. +# +# Separately there is a RUNTIME flag of the same name, +# EncoderConfig::enablePreprocessComputeFilter (default true), which +# VkVideoEncoder::InitEncoder branches on. It exists only in an ON build; +# nothing in-tree clears it and no CLI or ext-API exposes it. +# +# Running without the filter is a supported configuration: the input image is +# then the encode source, so InitEncoder refuses any input format the driver does +# not advertise for the profile, with VK_ERROR_FORMAT_NOT_SUPPORTED. That refusal +# is unconditional in an OFF build and gated on the runtime flag in an ON one. +# Nothing that would need converting is accepted, which includes the default +# EncoderConfig::input.vkFormat. +# +# The four encoder-ext tests that require the filter define the macro on their +# own targets, so this option does not reach them. +# +# OFF IS NOT EXERCISED BY THE DEFAULT BUILD. It was broken -- one use of the +# runtime flag sat outside the guard declaring it -- until that was fixed and +# both settings were built. Configure it explicitly after touching filter code. +option(BUILD_ENCODER_COMPUTE_FILTER + "Build the encoder's preprocess compute filter (requires a GLSL compiler backend)" ON) +if(BUILD_ENCODER_COMPUTE_FILTER) + add_definitions(-DVK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED=1) +endif() + # Common include directories include_directories( ${CMAKE_SOURCE_DIR}/common/include @@ -236,7 +305,7 @@ include(${CMAKE_CURRENT_SOURCE_DIR}/cmake/VulkanDispatchTable.cmake) # For Windows, we use a static lib because the Windows loader has a fairly restrictive loader search # path that can't be easily modified to point it to the same directory that contains the layers. set(VKVIDEO_UTILS_VLF_SOURCES - ../common/layers/vk_format_utils.cpp + common/layers/vk_format_utils.cpp ) if (WIN32) @@ -488,6 +557,14 @@ if(BUILD_ENCODER) MESSAGE(STATUS "VULKAN_VIDEO_ENCODER_INCLUDE path is not set. Setting the default path location to ${CMAKE_CURRENT_SOURCE_DIR}/include") set(VULKAN_VIDEO_ENCODER_INCLUDE "${CMAKE_CURRENT_SOURCE_DIR}/vk_video_encoder/include" CACHE PATH "Path to Vulkan Video Encoder include directory" FORCE) endif() + +# The internal header set. NOT exported by any target and NOT installed: it is +# reachable from the encoder library and from the in-tree tests that include it, +# and from nothing a consumer links or installs. Keeping it out of +# VULKAN_VIDEO_ENCODER_INCLUDE is what makes "internal" a build fact rather than +# a filename convention. + set(VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE + "${CMAKE_CURRENT_SOURCE_DIR}/vk_video_encoder/internal") if (DEFINED ENV{VULKAN_VIDEO_ENCODER_LIB_DIR_PATH}) MESSAGE(STATUS "VULKAN_VIDEO_ENCODER_LIB_DIR_PATH ENV VAR is set to $ENV{VULKAN_VIDEO_ENCODER_LIB_DIR_PATH}") set(VULKAN_VIDEO_ENCODER_LIB_PATH "$ENV{VULKAN_VIDEO_ENCODER_LIB_DIR_PATH}" CACHE PATH "Path to Vulkan Video Encoder library directory" FORCE) @@ -516,6 +593,333 @@ if(BUILD_ENCODER) if(BUILD_TESTS AND NOT DEFINED DEQP_TARGET) add_subdirectory(vk_video_encoder/test/vulkan-video-enc) + # Device-free coverage of the ext API's chained-descriptor walk. Links + # the static encoder archive for the internal header's test seams. + add_subdirectory(vk_video_encoder/test/encoder-ext-sync) + # Device-free coverage of which VkVideoEncoderConfig members + # Reconfigure carries, which it refuses, and which it still + # ignores on purpose. Links the static encoder archive for the + # internal header's null-backend and capture-push seams, which + # are what let a Reconfigure call be reached without a device. + add_subdirectory(vk_video_encoder/test/encoder-ext-reconfigure) + # Real-device coverage of the per-frame RELEASE fence export. Skips + # (CTest SKIP, exit 77) on a host with no encode-capable GPU. + # LINUX-ONLY TESTS, guarded individually below. + # + # Each reads a POSIX facility as the SUBJECT of the test, not as a + # convenience: dlopen for loader adoption, /proc fd enumeration and + # dirent for dma-buf fd ownership, unistd for the fd lifetime + # assertions. There is no Windows equivalent to assert about, so they + # are Linux-only by construction rather than merely unported. + # + # THE SAME LIST IS GUARDED IN vk_video_encoder/CMakeLists.txt. Both + # files register these directories -- this one is what a top-level + # configure uses -- so a guard added to only one of them is no guard + # at all. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-release-fence) + endif() + # Real-device coverage of the per-frame ACQUIRE fence's fd OWNERSHIP: + # that every refusal exit consumes the caller's sync_fd, and that the + # success path still consumes it exactly once. Needs a GPU for the + # same reason its sibling does -- a real sync_fd needs a real queue + # signal to export from, and the success case needs a real import. + # Skips (CTest SKIP, exit 77) without one. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-acquire-fd) + endif() + # Device-free coverage of the input-format taxonomy (which rung of the + # adaptation ladder a given format needs), of the preprocess-conversion + # decision the binder derives from it, and of the transfer-function + # declaration. All are pure functions of their arguments; what the + # filter then DOES with the pixels needs a GPU and lives in + # vk_filter_test. + add_subdirectory(vk_video_encoder/test/encoder-ext-filter) + # Real-device REGRESSION test for a defect this guards against: + # DrainPendingFrames() permanently disabled async assembly, and the + # synchronous fallback publishes no completion record, so every frame + # submitted after that call was encoded and then never became + # acquirable. DrainAndRestartThreads() closed it. This is a plain + # gating test -- it exits 0 and carries NO WILL_FAIL; see the note in + # the test's own CMakeLists.txt for why leaving WILL_FAIL on would + # have reported the working fix as a failure. Skips (CTest SKIP, + # exit 77) without a GPU. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-drain-assembly) + endif() + # Device-free coverage of the CONTRACT the test above rests on: that + # VkVideoEncoder::AssembleBitstreamData -- the synchronous assembly + # path -- publishes NO completion record, in either output mode. Seven + # sites in the tree are built on that premise, including the guard in + # ProcessOrderedFrames and the WriteDataToFile contract comment, and + # this entry is the ONLY thing that executes the function: the ext + # surface cannot reach it (asyncAssembly is pinned on, a subscriber is + # always registered, and ProcessOrderedFrames refuses the sync + # fallback whenever one exists) and the file-based CLI that can reach + # it exposes no VkVideoEncoder to observe. Subclasses the encoder with + # a null device context, so there is no GPU, driver or display + # dependence and nothing to skip. + add_subdirectory(vk_video_encoder/test/encoder-sync-assembly) + # Real-device coverage of the DIRECT submit's FIXED 8-slot wait + # array. The direct-encode path assembles waits into + # VkSemaphoreSubmitInfoKHR[8] and its fill loop is bounded by that + # capacity, so surplus waits were DISCARDED with no log, no assert and + # a SUCCESS status -- and because the ext layer appends the imported + # acquire-fence semaphore LAST, that producer fence is the first + # casualty. Two entries: the DIRECT subject, and a STAGED control that + # runs the identical case against the non-truncating vector assembly, + # which is what makes a green subject arm distinguishable from a test + # that cannot fail. Skips (exit 77) without an encode-capable GPU. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-direct-wait-capacity) + endif() + # Real-device coverage of the ADOPT-mode SESSION: the embedder + # owns the VkInstance and names the VkPhysicalDevice, and the + # LIBRARY creates its own VkDevice on them and encodes. That + # combination -- externalInstance + externalPhysicalDevice with + # externalDevice left NULL -- is the supported embedding shape, + # and this suite is the only thing in the tree that exercises + # externalInstance, externalPhysicalDevice or externalDevice at + # all: the only caller was the out-of-tree Chromium embedder. + # Four entries: the ADOPT subject, an OWN control that makes a + # red subject attributable to the adopted handles rather than to + # the host, the physical-device pin, and config.validate over a + # borrowed instance. Skips (exit 77) without an encode-capable + # GPU. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-adopt-device) + endif() + # Real-device coverage of INPUT RESIDENCY on an OS-HANDLE + # registration. Every other handleType under + # vk_video_encoder/test was ..._VK_IMAGE -- eight registration sites, + # zero OPAQUE_FD -- so the entire OS-handle import path, and every + # rule the ext layer applies ONLY to an OS handle, had no test of any + # kind. The residency read in SubmitRegisteredFrame is exactly such a + # rule, and a suite that only registers VK_IMAGE takes its other arm + # every time. + # + # The registration is a SELF-IMPORT: the test exports OPAQUE_FD from + # the LIBRARY's own VkDevice and hands the fd straight back to it, + # which is what Chromium's shipping CPU staging tier does and which + # nothing in this tree did. + # + # It asserts on a library-side counter + # (VkVideoEncoderInputResidencyInfo, added for it), NOT on a + # validation-error count, and that is forced rather than preferred: + # the two barrier programs the residency decision selects between are + # both spec-clean and both leave the image in the same layout, so a + # layer-based assertion would be green either way -- a test that + # cannot fail. Five entries: the OPAQUE_FD + explicit-LOCAL subject, + # a VK_IMAGE control over the IDENTICAL image that makes a red + # subject attributable to handleType alone, an explicit FOREIGN and + # an AUTO pin that together stop the fix degenerating into "never + # acquire", and a mirror of the Chromium tier-3 layout declaration + # which PASSES -- the design registered it WILL_FAIL and the hardware + # disagreed, so the registration was corrected rather than the + # measurement explained away; the suite's own CMakeLists records the + # reasoning. Skips (exit 77) without an encode-capable GPU. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-input-residency) + endif() + # Real-device INPUT FORMAT MATRIX. VkEncClassifyInputFormat names + # nine formats; whether one REGISTERS, and onto which input path, + # is a fact about a device, a session and a descriptor that the + # taxonomy table cannot see. This walks all nine plus two excluded + # controls against the LIBRARY-OWNED device and asserts both halves + # per arm -- the registration status AND the slot's resolved + # inputPath, read from VkEncProbeResource rather than inferred from + # the config flag, which is equally true of a registration that + # resolved to STAGED. Each arm also runs a NEGATIVE control with + # its own declaration withheld, so a fix that merely deleted a + # refusal fails here. Skips (exit 77) without an encode-capable GPU. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-format-matrix) + endif() + add_subdirectory(vk_video_encoder/test/encoder-ext-input-format-query) + # The v3 interface, driven the way a host drives it: Result and Expected + # semantics, role discovery, the aliasing reference that keeps a session alive + # behind a role, and the two configuration rules the library owns -- which + # input path a format resolves to, and which codecs carry HDR metadata. The + # checks that need no device always run; the rest report SKIP without an + # encode-capable GPU rather than passing vacuously. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-interface) + endif() + add_subdirectory(vk_video_encoder/test/encoder-drain) + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-release-fence) + endif() + + # The ENCODE half of the same matrix. The routing test above proves a + # descriptor registers and a slot resolves; it contains no encode call + # at all. This one fills each format with a known primaries pattern, + # submits real frames and writes the bitstream out for an INDEPENDENT + # decoder to judge -- and abandons a 3-plane row that does not resolve + # to FILTER before submitting, because staging one is a measured hang. + if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(vk_video_encoder/test/encoder-ext-format-encode) + endif() + + # Real-device coverage of the NVIDIA 615.06 dma-buf IMPORT-ORDINAL + # GUARD, which shipped with a coverage floor of exactly zero: no + # test, CMake entry or CI file referenced it, and + # VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF reached + # VkEncImportExternalImage from NO encoder-ext test at all -- every + # registration in the tree was VK_IMAGE or OPAQUE_FD, on both of + # which the guard returns at its second line. The suite was green + # with the workaround deleted. + # + # THE GUARD IS RETIRED. kVkEncImportOrdinalGuardCount is 0 at the + # shipped default, so it takes no sacrificial imports and retains + # nothing. Do not read the paragraph below as a description of a + # working mitigation; it never was one. + # + # WHAT WAS MEASURED, on an A4000 against NVIDIA 615.06. An import is + # damaged iff its process-global dma-buf import ordinal falls on one + # of two residues mod 32. The guard consumes ordinals ahead of the + # caller, so it moves WHICH imports are damaged and never HOW MANY: + # the damage rate is 12.5% at K = 0, 1, 2, 3 and 4 alike. It is a + # phase shift wearing the costume of a fix. + # + # It is also not recoverable downstream. Writing through a damaged + # import is swallowed, and an immediate re-import at ordinal+1 + # rescued 0 of 8 -- once a buffer is damaged it stays damaged. The + # damage splits in two: on one population the bytes ARE in the + # buffer and only our Vulkan mapping is dead (an EGL import and a + # plain CPU mmap of the same fd both read the frame perfectly); on + # the other the producer's write never landed and the pixels are + # gone for every reader. A phase-shifted control moved both + # populations exactly as predicted, 13/13. + # + # WHAT THIS SUITE IS FOR NOW. It exports a REAL dma-buf from a real + # NVIDIA VkDevice and imports it, with counting thunks in the device + # dispatch table, and asserts the retirement is REAL: at K = 0 the + # guard must take zero imports, retain nothing, and report DISABLED + # -- while the same harness, given an explicit non-zero count, still + # holds the armed build to its original claim. That second arm is + # what keeps the retirement honest instead of merely quiet, and it + # is why the mechanism was left in place rather than deleted. + # + # THE KILL SWITCH NO LONGER DISCRIMINATES, and must not be sold as + # a mutation proof: at K = 0 the subject is already inert, so + # VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD=1 leaves it green (0/21). + # The live mutation proof is the explicit --guard-count=2 arm, which + # goes red 11/21 against the retired library. + # + # Skips (exit 77) with no GPU, no dma-buf export, or a non-NVIDIA + # GPU, where the guard refuses to run by design. + add_subdirectory(vk_video_encoder/test/encoder-ext-import-ordinal-guard) + + # The same guard's REPORTING CHANNEL, device-free. The suite above + # asks whether the retirement is real on hardware; this one asks + # whether anything OUTSIDE the library can find out. The guard + # used to answer only on stderr, which the shipping Chromium + # configuration discards wholesale (silenceStdio), so both of its + # failure paths were silent in the one deployment it exists for. + # This gates the carrier that replaced that narration: + # VkVideoEncoderImportGuardInfo, chainable onto the registration + # status echo and onto the completion snapshot. It covers the + # CARRIER only. At the shipped default a real dma-buf import on an + # NVIDIA device reports DISABLED; COMPLETE and INCOMPLETE are + # reachable only on a re-armed build, and belong to the suite above + # and to hardware. + add_subdirectory(vk_video_encoder/test/encoder-ext-import-guard) + + # The dma-buf import CONTENT probe. The two suites above are about a + # RETIRED workaround and what it reported; this one is about a + # DETECTOR -- what the driver actually put in the buffer -- and the + # distinction decides what a green here means. + # + # Device-free, and it covers the DECISION: the predicate over every + # (Y,U,V) liveness quadrant, the scorer over synthetic NV12 buffers + # laid out with the plane offsets, padded row pitch and skipped rows a + # real HOST_VISIBLE LINEAR image has, the once-per-REGISTRATION latch, + # the oldest-damaged-first report ordering, the drain on retirement, + # and VkVideoEncoderImportContentInfo on both of its chain points. + # + # It does NOT cover whether the probe fires on a buffer NVIDIA 615.06 + # really poisoned; that needs the broken driver and a real GBM dma-buf + # import, and is proven on hardware or not at all. What it buys is + # that the part which would otherwise be exercised only on the one + # host with the broken driver is regression-tested everywhere. The + # predicate is asserted quadrant by quadrant rather than spot-checked + # because the CHROMA-ONLY version of it has produced a wrong + # conclusion on this defect twice, and a fully legal black frame + # (Y=16, U=V=128) is asserted CLEAN because that is the false positive + # that would reroute working buffers. + add_subdirectory(vk_video_encoder/test/encoder-ext-import-content) + + # H.264 Baseline must not emit CABAC. profile_idc 66 covers Baseline + # and Constrained Baseline alike and neither admits CABAC: + # entropy_coding_mode_flag must be 0 there. Device-free. + # + # The entropy coder is chosen by nobody: there is no command-line + # switch for it, it is taken from the device + # preferredStdEntropyCodingModeFlag, which is CABAC on this vendor + # hardware. So an explicit --profile baseline emitted profile_idc 66 + # with entropy_coding_mode_flag 1 -- a non-conformant bitstream, and + # reachable from the WebRTC default profile. + # + # Three parts: the rule over the whole profile x entropy matrix, the + # emitted SPS/PPS pair, and the composition with the profile + # auto-upgrade in InitProfileLevel(). The third is the one worth + # having. That upgrade does NOT cover this case -- it is guarded by + # profileIdc == INVALID and so never runs on an explicit request -- + # and the suite fails if someone later widens it to fire on requests + # and assumes it subsumes the clamp. + add_subdirectory(vk_video_encoder/test/encoder-h264-baseline-entropy) + + # GOP sequencing, device-free. The GOP machine decides encode order, + # B-frame counts and B-frame positions before any picture exists, and + # nothing else in this tree asserts those tables -- the sibling suites + # set gopLength and then encode, which cannot tell a correct sequence + # from a plausible one. Carried as its own subdirectory because it + # compiles VkVideoGopStructure.cpp directly and links nothing else. + add_subdirectory(vk_video_encoder/test/gop-structure) + + # Encoder quality round trip: encode, decode with ffmpeg, and compare + # per-frame PSNR against the source. This is the only round-trip + # quality measurement in the tree -- the encoder's own --psnr option + # compares input against RECONSTRUCTED, inside the encoder, and so + # cannot see a stream that decodes wrongly. + # + # gpu-labelled, and it needs ffmpeg and ffprobe besides. It exits 77 + # when the encoder binary or either tool is missing, which is the skip + # this label requires, so a host that cannot run it reports SKIPPED + # rather than red. + # + # --encoder is passed explicitly rather than letting the script derive + # it from --samples-root: that derivation spells the build directory + # in lower case and this tree builds into BUILD. + add_test(NAME EncoderAv1QualityRoundTrip + COMMAND ${Python3_EXECUTABLE} + ${CMAKE_CURRENT_SOURCE_DIR}/vk_video_encoder/test/av1_encoder_quality_test.py + --encoder $ + --output-dir ${CMAKE_CURRENT_BINARY_DIR}/av1-quality + --num-frames 30 + --codecs av1) + set_tests_properties(EncoderAv1QualityRoundTrip PROPERTIES + SKIP_RETURN_CODE 77 LABELS "gpu") + + # The gate's own classification, device-free. The round trip above can + # only exercise these decisions on a host that has an encoder, ffmpeg + # and content -- and both of them are about reporting a run as + # something it was not, so the hosts that cannot run it are exactly + # the ones where a misclassification would go unnoticed. + # + # device-free, and with no SKIP_RETURN_CODE: there is nothing here to + # skip for, and a skip is the failure mode under test. + # + # THE LABEL IS NOT OPTIONAL, and "not gpu" is not a label. Every + # registered test must carry 'device-free' or 'gpu': those two name the + # CI sets, and run_ctest_ci.sh audits the registry against them, so a + # test with neither is in no set, gates nothing, and fails the audit + # rather than quietly running nowhere. + add_test(NAME EncoderAv1QualityGateClassification + COMMAND ${Python3_EXECUTABLE} + ${CMAKE_CURRENT_SOURCE_DIR}/vk_video_encoder/test/test_av1_quality_gate.py) + set_tests_properties(EncoderAv1QualityGateClassification PROPERTIES + LABELS "device-free") endif() if(BUILD_DEMOS AND NOT DEFINED DEQP_TARGET) diff --git a/cmake/VulkanShaderCompilerBackend.cmake b/cmake/VulkanShaderCompilerBackend.cmake index e0a659bc..c31d43ed 100644 --- a/cmake/VulkanShaderCompilerBackend.cmake +++ b/cmake/VulkanShaderCompilerBackend.cmake @@ -73,17 +73,23 @@ if(VK_VIDEO_SAMPLES_SHADER_BACKEND STREQUAL "glslang") find_library(GLSLANG_LIBRARY_DEBUG NAMES glslangd HINTS ${GLSLANG_SDK_LIB_HINTS}) find_library(GLSLANG_RESOURCE_LIMITS_LIBRARY_DEBUG NAMES glslang-default-resource-limitsd HINTS ${GLSLANG_SDK_LIB_HINTS}) - - # The SDK's glslang is a static library built with ENABLE_OPT=ON, so - # it carries unresolved SPIRV-Tools references (spvContextCreate, - # spvValidatorOptions*) that the linker must satisfy even though we - # never enable the optimizer. The Linux shared library resolves them - # internally and needs none of this. - find_library(SPIRV_TOOLS_LIBRARY NAMES SPIRV-Tools HINTS ${GLSLANG_SDK_LIB_HINTS}) - find_library(SPIRV_TOOLS_OPT_LIBRARY NAMES SPIRV-Tools-opt HINTS ${GLSLANG_SDK_LIB_HINTS}) find_library(SPIRV_TOOLS_LIBRARY_DEBUG NAMES SPIRV-Toolsd HINTS ${GLSLANG_SDK_LIB_HINTS}) find_library(SPIRV_TOOLS_OPT_LIBRARY_DEBUG NAMES SPIRV-Tools-optd HINTS ${GLSLANG_SDK_LIB_HINTS}) endif() + + # A glslang built with ENABLE_OPT=ON carries unresolved SPIRV-Tools + # references (spvContextCreate, spvValidatorOptions*, and the + # spvtools::Create*Pass factories) that the linker must satisfy even + # though we never enable the optimizer. This is NOT MSVC-specific: it + # applies whenever the glslang we found is a static archive, which is + # what Debian/Ubuntu's libglslang-dev ships. Leaving them unresolved + # still links a shared library, but dlopen() of it then fails with + # undefined symbol: _ZN8spvtools29CreateLocalMultiStoreElimPassEv + # so the search has to run on every platform. Where glslang is a shared + # library that resolves them internally, these archives contribute + # nothing and are simply not pulled in. + find_library(SPIRV_TOOLS_LIBRARY NAMES SPIRV-Tools HINTS ${GLSLANG_SDK_LIB_HINTS}) + find_library(SPIRV_TOOLS_OPT_LIBRARY NAMES SPIRV-Tools-opt HINTS ${GLSLANG_SDK_LIB_HINTS}) endif() if(GLSLANG_LIBRARY AND GLSLANG_RESOURCE_LIMITS_LIBRARY AND GLSLANG_INCLUDE_DIR) @@ -108,6 +114,11 @@ if(VK_VIDEO_SAMPLES_SHADER_BACKEND STREQUAL "glslang") message(STATUS "Found glslang: ${GLSLANG_LIBRARY} (debug: ${GLSLANG_LIBRARY_DEBUG})") else() set(VK_SHADER_COMPILER_LIBS ${GLSLANG_LIBRARY} ${GLSLANG_RESOURCE_LIMITS_LIBRARY}) + if(SPIRV_TOOLS_LIBRARY AND SPIRV_TOOLS_OPT_LIBRARY) + list(APPEND VK_SHADER_COMPILER_LIBS + ${SPIRV_TOOLS_OPT_LIBRARY} ${SPIRV_TOOLS_LIBRARY}) + message(STATUS "Found SPIRV-Tools for static glslang: ${SPIRV_TOOLS_LIBRARY}") + endif() message(STATUS "Found glslang: ${GLSLANG_LIBRARY}") endif() # Both the include root and its glslang/ subdirectory: the sources spell diff --git a/cmake/VvsValidationGate.cmake b/cmake/VvsValidationGate.cmake new file mode 100644 index 00000000..47519435 --- /dev/null +++ b/cmake/VvsValidationGate.cmake @@ -0,0 +1,110 @@ +# Register a CTest entry whose validation-layer output is gating. +# +# THE NAME IS DELIBERATE. Two encoder test directories carry their own gate +# script and define a function of their own, vvs_add_gated_test(). A CMake +# function is global from the point it is defined, so a shared definition +# sharing that name would be replaced by whichever directory the configure +# reached last and every directory after it would silently get the other +# implementation. The two names are distinct so that neither can shadow the +# other. +# +# A test added with add_test() alone reports only its exit code, so a run that +# emitted validation errors and still exited 0 is recorded as PASSED and the +# errors are visible to nobody. vvs_add_validation_gated_test() runs the same +# binary through cmake/validation_gate.cmake, which reads the exit code and +# the layer output together and fails the test that produced a message. +# +# vvs_add_validation_gated_test( +# TARGET +# [ARGS ...] +# [EXPECT =...]) +# +# EXPECT declares a CEILING per VUID, and it is the only way a message is +# tolerated. A VUID that is not named has a ceiling of zero, so a message the +# tree has never seen fails on its first occurrence; a VUID that is named +# fails on the first occurrence past its count. Both facts are printed on +# every run and appended to validation_gate_summary.txt in the build root, so +# an allowance is never silent. An EXPECT entry records a defect that the tree +# already carries and that the entry does not excuse -- name the defect where +# the entry is written. +# +# WHAT PINS WHAT. VK_LAYER_SETTINGS_PATH controls HOW MANY messages print; +# the layer path controls WHETHER ANY DO. VVS_VALIDATION_LAYER_PATH is empty +# by default, and while it is empty the ambient environment decides which +# layer the loader finds -- so set it to make a build self-contained, and set +# VVS_REQUIRE_VALIDATION_LAYER so that a host without a layer reports a +# failure rather than an ungated run. +# +# A GATED TEST STILL RUNS WHERE NO LAYER IS PRESENT, and its own exit code +# still decides its verdict; what a missing layer removes is the validation +# check and nothing else. That is what makes gating a test never weaker than +# leaving it on a plain add_test(): this adds a way to fail and takes none +# away. + +set(VVS_VALIDATION_ALLOW_VUIDS "unknown VkStructureType" CACHE STRING + "Regex matched against a validation message BODY; matches are tolerated") +option(VVS_REQUIRE_VALIDATION_LAYER + "Fail, rather than skip, a gated test when no validation layer is installed" + OFF) +set(VVS_VALIDATION_LAYER_PATH "" CACHE PATH + "Directory holding a validation-layer manifest; becomes the tests VK_LAYER_PATH") +set(VVS_VALIDATION_LAYER_LIBDIR "" CACHE PATH + "Directory added to the tests LD_LIBRARY_PATH. A Vulkan SDK layer manifest names a bare soname, so without this the layer loads only if its library is already on the default search path") + +set(VVS_VALIDATION_GATE_SCRIPT "${CMAKE_CURRENT_LIST_DIR}/validation_gate.cmake") +set(VVS_VALIDATION_LAYER_SETTINGS "${CMAKE_CURRENT_LIST_DIR}/vk_layer_settings.txt") + +function(vvs_add_validation_gated_test _name) + cmake_parse_arguments(_g "" "TARGET" "ARGS;EXPECT" ${ARGN}) + if(NOT _g_TARGET) + message(FATAL_ERROR "vvs_add_validation_gated_test(${_name}): TARGET is required") + endif() + + # SPACE-separated. validation_gate.cmake documents TEST_ARGS as one argument + # or a space-separated string and splits it on whitespace; a CMake list + # interpolates semicolon-separated, which would reach the binary as a single + # unrecognised argument and make it print its usage and exit non-zero. + string(JOIN " " _gated_args ${_g_ARGS}) + # Comma-separated for the same reason, and VUID names carry no commas. + string(JOIN "," _gated_expect ${_g_EXPECT}) + + set(_env + "VK_LAYER_SETTINGS_PATH=${VVS_VALIDATION_LAYER_SETTINGS}" + # Forced through the loader rather than left to the binary: a binary + # that enables the layer only under a verbose flag would otherwise be + # gated by a layer that never loaded. + "VK_LOADER_LAYERS_ENABLE=VK_LAYER_KHRONOS_validation" + # Makes the loader name the layers it inserts. The gate requires that + # line before it reads a count, so a run with no layer is reported as + # such instead of as a clean one. + "VK_LOADER_DEBUG=layer") + if(VVS_VALIDATION_LAYER_PATH) + list(APPEND _env "VK_LAYER_PATH=${VVS_VALIDATION_LAYER_PATH}") + endif() + if(VVS_VALIDATION_LAYER_LIBDIR) + # Replaces rather than prepends. Deliberate, and opt-in: a test whose + # loader path half-comes from the ambient shell is not reproducible. + list(APPEND _env "LD_LIBRARY_PATH=${VVS_VALIDATION_LAYER_LIBDIR}") + endif() + + add_test(NAME ${_name} + COMMAND ${CMAKE_COMMAND} + -DTEST_EXE=$ + "-DTEST_ARGS=${_gated_args}" + -DREQUIRE_LAYER=${VVS_REQUIRE_VALIDATION_LAYER} + "-DALLOW_VUIDS=${VVS_VALIDATION_ALLOW_VUIDS}" + "-DEXPECT_VUIDS=${_gated_expect}" + "-DGATE_SUMMARY=${CMAKE_BINARY_DIR}/validation_gate_summary.txt" + -P "${VVS_VALIDATION_GATE_SCRIPT}") + # SKIP_REGULAR_EXPRESSION, not SKIP_RETURN_CODE: SKIP_RETURN_CODE outranks + # every other verdict in CTest, so a run that emitted validation errors on + # its way to a skip exit would be recorded as Skipped. The gate emits the + # token below only on a path that has already established there were none, + # and only for the test's own exit 77 -- the same condition SKIP_RETURN_CODE + # 77 named on these tests before they were gated. SKIP_RETURN_CODE cannot be + # carried alongside it in any case: the registered command is the cmake -P + # wrapper, whose exit code is 0 or 1 and never 77. + set_tests_properties(${_name} PROPERTIES + SKIP_REGULAR_EXPRESSION "VALIDATION_GATE_RESULT=SKIP" + ENVIRONMENT "${_env}") +endfunction() diff --git a/cmake/validation_gate.cmake b/cmake/validation_gate.cmake new file mode 100644 index 00000000..79f6a1d3 --- /dev/null +++ b/cmake/validation_gate.cmake @@ -0,0 +1,275 @@ +# Run a test binary and decide pass / fail / skip on the exit code AND the +# validation-layer output together. +# +# WHY A WRAPPER RATHER THAN set_tests_properties(FAIL_REGULAR_EXPRESSION): +# SKIP_RETURN_CODE outranks FAIL_REGULAR_EXPRESSION in CTest, so a run that +# emits validation errors on its way to a skip exit is recorded as Skipped and +# the errors are lost. A property-only gate cannot see both signals. +# +# THE ECHO IS SANITISED. CTest decides SKIP by regex-matching the whole +# captured output, and this script replays the child stdout and stderr into +# that same space. Every occurrence of the skip token in the child text is +# defanged before it is printed, so the token appears only where THIS script +# writes it, on a path that has already established there were no unexpected +# validation messages. +# +# WHAT IS COUNTED: MESSAGES, NOT VUID TOKENS. A binary that installs its own +# debug-utils messenger prints a "Validation Error: [ VUID-x ]" header AND a +# "The Vulkan spec states: ...(VUID-x)" trailer, two tokens per message; the +# layer default reporter prints the trailer only. Counting tokens therefore +# doubles under a reporter swap, which cannot distinguish a removed defect +# from a changed message format. What both reporters emit exactly once per +# message is the trailer, so the trailer closes a block and the block carries +# the body and the VUID name. +# +# THE LAYER IS FORCED, AND ITS INSERTION IS WITNESSED. A gated test that runs +# without the validation layer loaded is a check that cannot fail, and it +# looks exactly like a clean run. The layer is forced on through the loader +# rather than left to the binary, so a binary that enables validation only +# under a verbose flag is gated too; and the loader is asked to name the +# layers it inserts, so "no messages" is told apart from "no layer". +# +# WHAT A MISSING LAYER COSTS, AND WHAT IT DOES NOT. It costs the validation +# verdict, and only that: the test still runs and its own exit code still +# decides pass or fail. Reporting a skip instead would make this script +# STRICTLY WEAKER than the plain add_test() it replaces on every host with no +# installed layer -- a check that cannot fail, which is the failure this +# script exists to prevent, turned on the script itself. REQUIRE_LAYER is how +# a fleet makes the missing layer a failure in its own right. +# +# CONTRACT +# -DTEST_EXE= required +# -DTEST_ARGS= optional, one argument or a space-separated string +# -DREQUIRE_LAYER= optional. ON: a missing layer is a failure rather +# than an ungated run, so a fleet cannot sit green +# purely because the layer was absent everywhere. +# OFF, the default, still runs the test and still +# reports its exit code -- a missing layer removes +# the validation verdict and nothing else. +# -DALLOW_VUIDS= optional, matched against the message BODY. +# -DEXPECT_VUIDS= optional, comma-separated = ceilings. +# -DGATE_SUMMARY= optional, a file every run appends its tally to. +# -DRUN_TIMEOUT= optional, default 600. + +if(NOT DEFINED TEST_EXE) + message(FATAL_ERROR "validation_gate: TEST_EXE is required") +endif() +if(NOT DEFINED RUN_TIMEOUT OR RUN_TIMEOUT STREQUAL "") + set(RUN_TIMEOUT 600) +endif() +if(NOT DEFINED ALLOW_VUIDS OR ALLOW_VUIDS STREQUAL "") + # Matched against the BODY, deliberately, because the VUID NAME cannot carry + # this distinction. VUID-Vk-pNext-pNext fires both for a struct type the + # layer does not recognise -- header and layer built against different + # versions, benign -- and for a struct the layer knows perfectly well but + # which is not permitted in that chain, which is a real defect and one a + # video-encode library that chains many extension structures is exposed to. + # Only the body separates them. + set(ALLOW_VUIDS "unknown VkStructureType") +endif() + +set(_args "") +if(DEFINED TEST_ARGS AND NOT TEST_ARGS STREQUAL "") + # Accept both spellings. A caller that forwards a CMake list hands over + # "--arm;--validate", which carries no whitespace for separate_arguments to + # split on and would reach the binary as one unrecognised argument. + string(REPLACE ";" " " _test_args_norm "${TEST_ARGS}") + separate_arguments(_args NATIVE_COMMAND "${_test_args_norm}") +endif() + +# The declared ceilings, one CMake variable per VUID. +set(_expected_names "") +if(DEFINED EXPECT_VUIDS AND NOT EXPECT_VUIDS STREQUAL "") + string(REPLACE "," ";" _expect_list "${EXPECT_VUIDS}") + foreach(_entry IN LISTS _expect_list) + string(STRIP "${_entry}" _entry) + if(NOT _entry STREQUAL "") + if(NOT _entry MATCHES "^(VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+)=([0-9]+)$") + message(FATAL_ERROR + "validation_gate: EXPECT_VUIDS entry is not =: ${_entry}") + endif() + set(_exp_${CMAKE_MATCH_1} "${CMAKE_MATCH_2}") + list(APPEND _expected_names "${CMAKE_MATCH_1}") + endif() + endforeach() +endif() + +# TIMEOUT rather than streaming. execute_process buffers, so a hang would +# otherwise reach the CTest timeout and take every captured byte with it. +# Streaming would put the child raw text back into the CTest match space and +# re-open the skip-token collision described above. +execute_process( + COMMAND "${TEST_EXE}" ${_args} + TIMEOUT ${RUN_TIMEOUT} + RESULT_VARIABLE _rc + OUTPUT_VARIABLE _out + ERROR_VARIABLE _err) + +set(_all "${_out}${_err}") +set(_safe "${_all}") +string(REPLACE "VALIDATION_GATE_RESULT" "VALIDATION_GATE_RESULT_FROM_CHILD" _safe "${_safe}") +message("${_safe}") + +# The layer witness. The loader names each layer it inserts when it is asked +# to, so this line is present on every run the layer joined and absent on +# every run it did not, which is what separates a clean run from a blind one. +string(REGEX MATCHALL "Insert[a-z]* instance layer .VK_LAYER_KHRONOS_validation." + _layer_witness "${_all}") +list(LENGTH _layer_witness _n_witness) +set(_have_layer TRUE) +if(_n_witness EQUAL 0) + if(REQUIRE_LAYER) + message(FATAL_ERROR + "validation_gate: the validation layer was never inserted, so this" + " run gated nothing. Point VVS_VALIDATION_LAYER_PATH (and" + " VVS_VALIDATION_LAYER_LIBDIR) at an installed layer, or configure" + " with -DVVS_REQUIRE_VALIDATION_LAYER=OFF to allow an ungated" + " run.") + endif() + # A MISSING LAYER IS NOT A SKIP. The child has already run, above, and its + # exit code is the verdict it was written to give -- the same verdict it + # gave before it was gated. Returning here would discard that and report + # Skipped, which would make gating a test STRICTLY WEAKER than leaving it on + # a plain add_test(): on every runner without an installed layer, and that + # is the default runner, a gated test would report Skipped whether it passed + # or failed. So this arm records what was not checked and falls through to + # the exit-code check at the end. The only skip this script emits is the + # child's own exit 77. + set(_have_layer FALSE) + message("VALIDATION_GATE: no validation layer was inserted; nothing was" + " gated, and this run stands on the test's own exit code") +endif() + +# Block-based counting. A line-based count is wrong for one of the two +# reporters: the default reporter carries the VUID token on the trailer only, +# so skipping trailers discards every message it emits, while counting every +# token double-counts the messenger reporter. The trailer closes the block, +# and the block carries the body the allowlist is matched against. +string(REPLACE ";" "\\;" _lines "${_all}") +string(REPLACE "\n" ";" _lines "${_lines}") +set(_real 0) +set(_allowed 0) +set(_seen_names "") +set(_block "") + +macro(_vvs_tally _text) + if("${_text}" MATCHES "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+") + set(_vuid "${CMAKE_MATCH_0}") + if("${_text}" MATCHES "${ALLOW_VUIDS}") + math(EXPR _allowed "${_allowed}+1") + else() + math(EXPR _real "${_real}+1") + if(NOT DEFINED _seen_${_vuid}) + set(_seen_${_vuid} 0) + list(APPEND _seen_names "${_vuid}") + endif() + math(EXPR _seen_${_vuid} "${_seen_${_vuid}}+1") + endif() + endif() +endmacro() + +foreach(_line IN LISTS _lines) + set(_block "${_block}\n${_line}") + if(_line MATCHES "The Vulkan spec states") + _vvs_tally("${_block}") + set(_block "") + endif() +endforeach() +# A layer that emits a VUID with no trailer at all leaves a dangling block; +# count it rather than lose it. +_vvs_tally("${_block}") + +# Sort the tally into what the ceilings cover and what they do not. A VUID +# with no ceiling fails on its first message, and a VUID with one fails on the +# first message past it, so neither a new VUID nor a growing one can hide +# behind an allowance. +set(_over "") +set(_tolerated "") +set(_under "") +foreach(_vuid IN LISTS _seen_names) + set(_n "${_seen_${_vuid}}") + set(_limit 0) + if(DEFINED _exp_${_vuid}) + set(_limit "${_exp_${_vuid}}") + endif() + if(_n GREATER _limit) + list(APPEND _over "${_vuid} ${_n}/${_limit}") + else() + list(APPEND _tolerated "${_vuid}=${_n}/${_limit}") + endif() +endforeach() +foreach(_vuid IN LISTS _expected_names) + if(NOT DEFINED _seen_${_vuid}) + list(APPEND _under "${_vuid}=0/${_exp_${_vuid}}") + endif() +endforeach() + +string(REPLACE ";" " " _tolerated_txt "${_tolerated}") +string(REPLACE ";" ", " _over_txt "${_over}") +string(REPLACE ";" " " _under_txt "${_under}") + +# A tally taken with no layer is zero for the reason zero always looks like on +# this instrument -- nothing was watching -- so it is reported as unmeasured +# rather than as clean, here and in the summary file. +string(CONCAT _gate_tally "messages=${_real} skew-allowlisted=${_allowed}" + " tolerated=[${_tolerated_txt}]") +if(NOT _have_layer) + set(_gate_tally "messages=NOT-MEASURED (no validation layer was inserted)") +endif() +message("VALIDATION_GATE: exit=${_rc} ${_gate_tally}") +if(_under AND _have_layer) + # An enumerated ceiling that is no longer reached is stale. It is reported + # rather than enforced downwards, because a count that moves with the driver + # would otherwise turn the gate red for an improvement. + message("VALIDATION_GATE: ceiling no longer reached, lower or remove it:" + " ${_under_txt}") +endif() + +# Skew-allowlisted and tolerated messages are never silent. On a green run +# CTest shows no output at all, so an accumulation of them would otherwise be +# invisible; this line survives the run in a file. +if(DEFINED GATE_SUMMARY AND NOT GATE_SUMMARY STREQUAL "") + file(APPEND "${GATE_SUMMARY}" + "${TEST_EXE} ${TEST_ARGS}: exit=${_rc} ${_gate_tally}\n") +endif() + +# THE ENCODE-SOURCE LAYOUT, WHICH NO VALIDATION MESSAGE CARRIES. +# VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 is checked against the image +# layout map of the command buffer the encode is recorded into. A staged input +# has its barriers recorded into a DIFFERENT command buffer, so that map holds +# no entry for the encode-source image and the check returns true without +# comparing anything; the submit-time sweep reads the same registry and is +# blind in the same way. The encoder emits its own diagnostic when the staging +# arm leaves the image in anything other than VIDEO_ENCODE_SRC_KHR, and this +# promotes that from loud to gating. +string(REGEX MATCHALL "staged encode-source image is in layout" + _layout_hits "${_all}") +list(LENGTH _layout_hits _n_layout) +if(_n_layout GREATER 0) + message(FATAL_ERROR + "validation_gate: ${_n_layout} frame(s) reached vkCmdEncodeVideoKHR" + " with the staged encode-source image in the wrong layout" + " (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811). A staging arm is" + " missing its hand-off barrier to VIDEO_ENCODE_SRC_KHR. The" + " validation layer cannot see this, so this check is the only thing" + " that reports it.") +endif() + +if(_over) + message(FATAL_ERROR + "validation_gate: validation message(s) past their declared ceiling:" + " ${_over_txt}. A VUID with no declared ceiling has a ceiling of zero." + " (exit=${_rc}, ${_allowed} skew-allowlisted by body /${ALLOW_VUIDS}/)") +endif() + +if(_rc EQUAL 77) + message("VALIDATION_GATE_RESULT=SKIP") + return() +endif() + +# _rc is a STRING on abnormal exit ("Segmentation fault", "Process terminated +# due to timeout", "no such file or directory"). EQUAL comparisons are +# correctly false for those, so they fall through to here and fail. +if(NOT _rc EQUAL 0) + message(FATAL_ERROR "validation_gate: test did not exit cleanly: ${_rc}") +endif() diff --git a/cmake/vk_layer_settings.txt b/cmake/vk_layer_settings.txt new file mode 100644 index 00000000..63814b01 --- /dev/null +++ b/cmake/vk_layer_settings.txt @@ -0,0 +1,17 @@ +# Layer settings shared by every test registered through vvs_add_gated_test(). +# +# duplicate_message_limit = 0 means "report every message". The layer default +# is 10, and the cap is keyed on the VUID string alone with no per-object +# component, so a single repeated VUID truncates and a run understates itself +# -- a gate built on a truncated count cannot tell a shrinking defect from a +# capped one. +# +# The layer also searches the CURRENT WORKING DIRECTORY for a file of this +# name, so a run that merely unsets VK_LAYER_SETTINGS_PATH while sitting in a +# directory that holds one is still uncapped. +# +# This pins HOW MANY messages print. It does not pin WHICH: different +# validation-layer builds report different totals and different sets of +# distinct VUIDs for the same binary on the same host, so a count is only +# comparable against another count taken with the same layer. +khronos_validation.duplicate_message_limit = 0 diff --git a/common/include/nvidia_utils/vulkan/ycbcrvkinfo.h b/common/include/nvidia_utils/vulkan/ycbcrvkinfo.h index 01192f76..5b29c78a 100644 --- a/common/include/nvidia_utils/vulkan/ycbcrvkinfo.h +++ b/common/include/nvidia_utils/vulkan/ycbcrvkinfo.h @@ -34,6 +34,12 @@ typedef struct VkMpFormatInfo { VkFormat vkPlaneFormat[VK_MAX_NUM_IMAGE_PLANES_EXT]; // VkFormats for the corresponding plane. } VkMpFormatInfo; +// How many rows vkMpFormatInfo[] has. A compile-time bound, so an array sized +// from a walk of the table can be sized at compile time too. nvVkFormats.cpp +// static_asserts it against the table itself, which is where the table is +// visible -- so this cannot drift from it without failing the build. +#define YCBCR_VK_FORMAT_INFO_TABLE_SIZE 38 + typedef struct VkFormatDesc { VkFormat format; uint8_t numberOfChannels; @@ -104,6 +110,23 @@ extern "C" { */ const VkMpFormatInfo * YcbcrVkFormatInfo(const VkFormat format); +/** + * @brief YcbcrVkFormatInfoByIndex WALKS the multi-planar table. + * + * YcbcrVkFormatInfo() answers about a format a caller already holds, which is + * the only question this table could previously be asked. A caller that wants + * to know WHICH formats it describes -- to derive a routable set, a conversion + * target or a capability list from the table rather than restate it beside one + * -- had no way to enumerate it: the table is file-static in ycbcrinfotbl.h, + * and its two dense VkFormat ranges are spelled only in macros that header + * does not export. + * + * @param index 0 .. YCBCR_VK_FORMAT_INFO_TABLE_SIZE - 1, in table order. + * @retval pointer to the row, or NULL once |index| is past the end -- so a + * walk terminates on the return value and needs no size of its own. + */ +const VkMpFormatInfo * YcbcrVkFormatInfoByIndex(uint32_t index); + #ifdef __cplusplus } #endif diff --git a/common/libs/VkCodecUtils/VkEncoderRenderFrame.cpp b/common/libs/VkCodecUtils/VkEncoderRenderFrame.cpp deleted file mode 100644 index e7a4c027..00000000 --- a/common/libs/VkCodecUtils/VkEncoderRenderFrame.cpp +++ /dev/null @@ -1,29 +0,0 @@ -/* -* Copyright 2024 NVIDIA Corporation. -* -* Licensed under the Apache License, Version 2.0 (the "License"); -* you may not use this file except in compliance with the License. -* You may obtain a copy of the License at -* -* http://www.apache.org/licenses/LICENSE-2.0 -* -* Unless required by applicable law or agreed to in writing, software -* distributed under the License is distributed on an "AS IS" BASIS, -* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -* See the License for the specific language governing permissions and -* limitations under the License. -*/ - -#include "VkEncoderRenderFrame.h" - -VkEncoderRenderFrame::VkEncoderRenderFrame() -{ - // TODO Auto-generated constructor stub - -} - -VkEncoderRenderFrame::~VkEncoderRenderFrame() -{ - // TODO Auto-generated destructor stub -} - diff --git a/common/libs/VkCodecUtils/VkEncoderRenderFrame.h b/common/libs/VkCodecUtils/VkEncoderRenderFrame.h deleted file mode 100644 index ad2a6301..00000000 --- a/common/libs/VkCodecUtils/VkEncoderRenderFrame.h +++ /dev/null @@ -1,27 +0,0 @@ -/* -* Copyright 2024 NVIDIA Corporation. -* -* Licensed under the Apache License, Version 2.0 (the "License"); -* you may not use this file except in compliance with the License. -* You may obtain a copy of the License at -* -* http://www.apache.org/licenses/LICENSE-2.0 -* -* Unless required by applicable law or agreed to in writing, software -* distributed under the License is distributed on an "AS IS" BASIS, -* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. -* See the License for the specific language governing permissions and -* limitations under the License. -*/ - -#ifndef _VKCODECUTILS_VKENCODERRENDERFRAME_H_ -#define _VKCODECUTILS_VKENCODERRENDERFRAME_H_ - -class VkEncoderRenderFrame -{ -public: - VkEncoderRenderFrame(); - virtual ~VkEncoderRenderFrame(); -}; - -#endif /* _VKCODECUTILS_VKENCODERRENDERFRAME_H_ */ diff --git a/common/libs/VkCodecUtils/VkEncoderStdioLatch.cpp b/common/libs/VkCodecUtils/VkEncoderStdioLatch.cpp new file mode 100644 index 00000000..1e9b5870 --- /dev/null +++ b/common/libs/VkCodecUtils/VkEncoderStdioLatch.cpp @@ -0,0 +1,42 @@ +/* +* Copyright 2026 NVIDIA Corporation. +* +* Licensed under the Apache License, Version 2.0 (the "License"); +* you may not use this file except in compliance with the License. +* You may obtain a copy of the License at +* +* http://www.apache.org/licenses/LICENSE-2.0 +* +* Unless required by applicable law or agreed to in writing, software +* distributed under the License is distributed on an "AS IS" BASIS, +* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +* See the License for the specific language governing permissions and +* limitations under the License. +*/ + +#include "VkCodecUtils/VkEncoderStdioLatch.h" + +//============================================================================= +// The one definition of the stdio-silence counter. +// +// IT LIVES ALONE, and that is the point. It used to sit in +// VulkanDeviceContext.cpp, which meant every target compiling any file that +// routes output through the gate had to compile the device context too. The +// standalone decoder and demo targets pick individual VkCodecUtils sources +// and do not, so they failed to link the moment those sources started using +// the gate. +// +// This unit includes the latch header and nothing else -- no Vulkan, no +// device, no library link closure -- so a target can add it without taking on +// anything it was deliberately avoiding. +// +// Still exactly ONE definition, for the reason the header spells out: a +// function-local static behind an internal-linkage accessor would give every +// translation unit its own counter, and a silence request made in one would +// be invisible to the rest. +//============================================================================= +std::atomic& VkEncoderStdioSilenceCountRef() +{ + static std::atomic activeSilenceRequests{0}; + return activeSilenceRequests; +} diff --git a/common/libs/VkCodecUtils/VkEncoderStdioLatch.h b/common/libs/VkCodecUtils/VkEncoderStdioLatch.h new file mode 100644 index 00000000..85e30717 --- /dev/null +++ b/common/libs/VkCodecUtils/VkEncoderStdioLatch.h @@ -0,0 +1,266 @@ +/* +* Copyright 2026 NVIDIA Corporation. +* +* Licensed under the Apache License, Version 2.0 (the "License"); +* you may not use this file except in compliance with the License. +* You may obtain a copy of the License at +* +* http://www.apache.org/licenses/LICENSE-2.0 +* +* Unless required by applicable law or agreed to in writing, software +* distributed under the License is distributed on an "AS IS" BASIS, +* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +* See the License for the specific language governing permissions and +* limitations under the License. +*/ + +#ifndef _VKCODECUTILS_VKENCODERSTDIOLATCH_H_ +#define _VKCODECUTILS_VKENCODERSTDIOLATCH_H_ + +#include +#include +#include +#include +#include +#include +#include +#include + +//============================================================================= +// Process-wide stdio-silence gate. +// +// Chromium runs the encoder inside the sandboxed GPU process, where writes to +// stdout/stderr are at best lost and at worst trip sandbox diagnostics. The +// public VkVideoEncoderConfig::silenceStdio flag lets the caller suppress the +// library's info prints and error messages -- specifically, everything routed +// through the VkEncOut() / VkEncErr() wrappers below, which is where the +// library's std::cout / std::cerr output goes. +// +// SCOPE, precisely: the latch covers everything routed through the wrappers +// in this header -- the two gated streams via VkEncOut() / VkEncErr(), and +// gated formatted output via VkEncPrintfOut() / VkEncPrintfErr() / +// VkEncVPrintf(). A printf or fprintf written directly, bypassing those, is +// unaffected unless its own call site tests IsVkEncoderStdioSilenced(), which +// exactly one does: the argv-parsing failure print in +// vulkan_video_encoder_argv.cpp. +// +// EncoderConfig::ParseArguments() IS silenced. VkEncoderConfig.cpp routes its +// usage text and its per-option diagnostics through the gated wrappers and +// contains no direct printf or fprintf. +// +// The policy is process-wide rather than per-encoder state because the gated +// streams are used from translation units that have no handle to the +// VkVideoEncoderConfig. What is process-wide is a COUNT OF ACTIVE SILENCE +// REQUESTS, not a bool anyone may assign: +// +// * silence holds while ANY live platform, context, session or scoped query +// is requesting it; +// * a caller that wants audible output holds no token -- it does not, and +// cannot, turn another owner's silence off; +// * silence ends when the last such owner is gone. +// +// A bool could not express that. The previous form was assigned by whoever +// initialized last, so a second session created while a first was running +// overwrote the first's decision, and the format enumeration's +// save/set/restore could restore a value another thread had changed in +// between. Simultaneous callers genuinely cannot each choose: this is a +// process-wide effect and one owner requiring silence wins. +// +// ODR constraint: the accessor must NOT be `static inline` with a +// function-local static. `static` gives the function INTERNAL linkage, so +// every translation unit that includes this header would get its OWN copy +// of the function -- and therefore its OWN counter -- and a request made in +// one TU would be invisible to every other TU. The storage has exactly ONE +// definition, in VkEncoderStdioLatch.cpp, and this header only declares it. +// The thin wrappers are plain `inline` (external linkage, ODR-merged) so +// call sites are unchanged. +// +// WHY ITS OWN FILE. The definition used to sit in VulkanDeviceContext.cpp, +// on the reasoning that VkCodecUtils is the bottom-most target every encoder +// consumer links. That reasoning was wrong: the standalone decoder, demo and +// test targets pick individual VkCodecUtils sources and do not compile the +// device context, so they failed to link the moment those sources started +// routing output through this gate. A target that needs the gate adds +// VkEncoderStdioLatch.cpp, which pulls in no Vulkan and no device. +//============================================================================= +// Single definition: VkEncoderStdioLatch.cpp. +std::atomic& VkEncoderStdioSilenceCountRef(); + +// Reading the policy is a relaxed load: the counter arbitrates emission and +// publishes no other state, so no call site needs to see anything else that +// an owner did before requesting silence. +inline bool IsVkEncoderStdioSilenced() +{ + return VkEncoderStdioSilenceCountRef().load(std::memory_order_relaxed) > 0; +} + +// One request. Construction takes it, destruction gives it back, and a move +// transfers it -- so an owner's silence lasts exactly as long as the owner, +// which is the property a bool assignment could not express. +// +// An aliasing role that keeps an encoder alive must keep its logging policy +// alive too, which means holding a token of its own. Two tokens held by one +// object are fine: correctness here is balanced lifetime, not a single +// designated owner. +class VkEncoderStdioSilenceScope { +public: + // Requests nothing. This is what an audible owner holds. + VkEncoderStdioSilenceScope() : m_held(false) { } + + explicit VkEncoderStdioSilenceScope(bool requestSilence) + : m_held(requestSilence) + { + if (m_held) { + VkEncoderStdioSilenceCountRef().fetch_add( + 1, std::memory_order_relaxed); + } + } + + ~VkEncoderStdioSilenceScope() { Release(); } + + VkEncoderStdioSilenceScope(const VkEncoderStdioSilenceScope&) = delete; + VkEncoderStdioSilenceScope& operator=( + const VkEncoderStdioSilenceScope&) = delete; + + VkEncoderStdioSilenceScope(VkEncoderStdioSilenceScope&& other) noexcept + : m_held(other.m_held) + { + other.m_held = false; + } + + VkEncoderStdioSilenceScope& operator=( + VkEncoderStdioSilenceScope&& other) noexcept + { + if (this != &other) { + Release(); + m_held = other.m_held; + other.m_held = false; + } + return *this; + } + + // Idempotent: a token releases its one request and never a second. + void Release() + { + if (!m_held) { + return; + } + m_held = false; + const int previous = VkEncoderStdioSilenceCountRef().fetch_sub( + 1, std::memory_order_relaxed); + // Underflow means a request was released twice or never taken, which + // would leave the process audible while an owner still needs silence. + assert(previous > 0); + (void)previous; + } + + bool RequestsSilence() const { return m_held; } + +private: + bool m_held; +}; + +// A std::ostream backed by a no-op streambuf. VkEncOut()/VkEncErr() return a +// reference to either the real stream or this sink depending on the gate, so a +// call site only needs "std::cout" -> "VkEncOut()" (the trailing "<< ... << +// std::endl" chain is unchanged and simply discarded when silenced). +class VkEncoderNullStreambuf : public std::streambuf { +protected: + int overflow(int c) override { return c; } // swallow every character +}; + +// Plain `inline` (NOT `static inline`) so this is one ODR-merged function +// rather than one per translation unit. +// +// The objects it returns are THREAD-LOCAL. Discarding every character does +// not make an ostream safe to format into from two threads at once: width, +// fill, precision and the sentry all live in the stream object, and two +// silenced workers writing through one shared instance are a data race on +// them. One sink per thread costs nothing and removes it. +inline std::ostream& VkEncoderNullStream() +{ + static thread_local VkEncoderNullStreambuf nullBuf; + static thread_local std::ostream nullStream(&nullBuf); + return nullStream; +} + +inline std::ostream& VkEncOut() +{ + return IsVkEncoderStdioSilenced() ? VkEncoderNullStream() : std::cout; +} + +inline std::ostream& VkEncErr() +{ + return IsVkEncoderStdioSilenced() ? VkEncoderNullStream() : std::cerr; +} + + +// Printf-style diagnostics, routed through the same gate. +// +// The library's diagnostics are overwhelmingly printf-style. Rewriting every +// one of them into a stream expression would be a much larger and more +// error-prone change than giving the gate a formatting entry point, so this is +// the seam those call sites move to. +// +// SCOPE, and it is the same distinction the streams draw: these are for +// DIAGNOSTICS. Output written to a file the caller asked for is the library's +// product and never comes through here -- fprintf/fwrite to an output FILE* +// stays exactly as it is. +// +// Formatting happens only when the output will actually be emitted, so a +// silenced process pays nothing for a message it discards. +inline void VkEncVPrintf(std::ostream& out, const char* format, va_list args) +{ + char stack[1024]; + va_list retry; + va_copy(retry, args); + const int needed = std::vsnprintf(stack, sizeof(stack), format, args); + if (needed < 0) { + va_end(retry); + return; // encoding error in the format; nothing useful to say + } + if ((size_t)needed < sizeof(stack)) { + out << stack; + va_end(retry); + return; + } + // Rare: a message longer than the stack buffer. Heap only for that case. + std::string heap((size_t)needed + 1, '\0'); + std::vsnprintf(&heap[0], heap.size(), format, retry); + va_end(retry); + heap.resize((size_t)needed); + out << heap; +} + +#if defined(__GNUC__) +#define VK_ENC_PRINTF_LIKE(fmtIndex, firstArg) \ + __attribute__((format(printf, fmtIndex, firstArg))) +#else +#define VK_ENC_PRINTF_LIKE(fmtIndex, firstArg) +#endif + +VK_ENC_PRINTF_LIKE(1, 2) +inline void VkEncPrintfOut(const char* format, ...) +{ + if (IsVkEncoderStdioSilenced()) { + return; + } + va_list args; + va_start(args, format); + VkEncVPrintf(std::cout, format, args); + va_end(args); +} + +VK_ENC_PRINTF_LIKE(1, 2) +inline void VkEncPrintfErr(const char* format, ...) +{ + if (IsVkEncoderStdioSilenced()) { + return; + } + va_list args; + va_start(args, format); + VkEncVPrintf(std::cerr, format, args); + va_end(args); +} + +#endif /* _VKCODECUTILS_VKENCODERSTDIOLATCH_H_ */ diff --git a/common/libs/VkCodecUtils/VkImageResource.cpp b/common/libs/VkCodecUtils/VkImageResource.cpp index d79084ba..a0c70ca4 100644 --- a/common/libs/VkCodecUtils/VkImageResource.cpp +++ b/common/libs/VkCodecUtils/VkImageResource.cpp @@ -15,6 +15,7 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "VkCodecUtils/HelpersDispatchTable.h" #include "VkCodecUtils/Helpers.h" #include "VkCodecUtils/VulkanDeviceContext.h" @@ -336,7 +337,7 @@ VkResult VkImageResource::CreateExportable(const VulkanDeviceContext* vkDevCtx, do { result = vkDevCtx->CreateImage(device, &modifiedImageInfo, nullptr, &image); if (result != VK_SUCCESS) { - std::cerr << "[VkImageResource] CreateImage FAILED: result=" << result + VkEncErr() << "[VkImageResource] CreateImage FAILED: result=" << result << " format=" << modifiedImageInfo.format << " extent=" << modifiedImageInfo.extent.width << "x" << modifiedImageInfo.extent.height << " tiling=" << modifiedImageInfo.tiling @@ -362,7 +363,7 @@ VkResult VkImageResource::CreateExportable(const VulkanDeviceContext* vkDevCtx, actualDrmModifier = modProps.drmFormatModifier; // Warn if the driver returns DRM_FORMAT_MOD_INVALID — indicates a driver bug if (actualDrmModifier == ((1ULL << 56) - 1)) { - std::cerr << "[VkImageResource] WARNING: vkGetImageDrmFormatModifierPropertiesEXT " + VkEncErr() << "[VkImageResource] WARNING: vkGetImageDrmFormatModifierPropertiesEXT " << "returned DRM_FORMAT_MOD_INVALID — using requested modifier 0x" << std::hex << drmFormatModifier << std::dec << std::endl; actualDrmModifier = drmFormatModifier; @@ -659,6 +660,28 @@ VkResult VkImageResourceView::Create(const VulkanDeviceContext* vkDevCtx, usageCreateInfo.pNext = nullptr; usageCreateInfo.usage = planeUsageOverride; + // A view may only be created over an image whose usage includes at least + // one view-compatible bit (VUID-VkImageViewCreateInfo-image-04441). + // TRANSFER_SRC/DST are NOT among them, so a transfer-only image -- e.g. a + // staging copy source supplied by a caller, consumed solely through its + // raw VkImage by vkCmdCopyImage and barriers -- gets no view at all + // rather than an invalid one. Images this library allocates itself always + // carry a view-compatible usage, which is why this never fired before + // callers began handing in externally created images. + static const VkImageUsageFlags kViewCompatibleUsage = + VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_STORAGE_BIT | + VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | + VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT | + VK_IMAGE_USAGE_TRANSIENT_ATTACHMENT_BIT | + VK_IMAGE_USAGE_INPUT_ATTACHMENT_BIT | + VK_IMAGE_USAGE_VIDEO_DECODE_DST_BIT_KHR | + VK_IMAGE_USAGE_VIDEO_DECODE_DPB_BIT_KHR | + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_VIDEO_ENCODE_DPB_BIT_KHR; + if ((imageCreateInfo.usage & kViewCompatibleUsage) == 0) { + skipCombinedView = true; + } + if (!skipCombinedView) { if (mpInfo) { VkImageUsageFlags combinedUsage = imageCreateInfo.usage; @@ -707,8 +730,19 @@ VkResult VkImageResourceView::Create(const VulkanDeviceContext* vkDevCtx, } planeUsageCreateInfo.pNext = nullptr; + // A per-plane view reinterprets the image as a different format + // (R8, R8G8, ...), which the image must have been created mutable to + // permit (VUID-VkImageViewCreateInfo-image-01762). For an image this + // library allocated that always holds; for one handed in by a caller + // -- a dma_buf import, where the EXPORTER chose the create flags -- + // it usually does not, and requesting plane views would be invalid. + // Skip them and keep the combined view, which is all the encode path + // reads anyway. + const bool planeViewsPermitted = + (imageCreateInfo.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0; + // Skip per-plane views when usage is zero (video-only images like DPB) - if (planeUsageCreateInfo.usage != 0) { + if ((planeUsageCreateInfo.usage != 0) && planeViewsPermitted) { viewInfo.pNext = &planeUsageCreateInfo; // Per-plane views are bound as storage images to the YCbCr compute filter, @@ -881,8 +915,21 @@ VkResult VkImageResourceView::Create(const VulkanDeviceContext* vkDevCtx, } numViews++; - // Now create per-plane views for compute storage - if (mpInfo) { + // Now create per-plane views for compute storage. + // + // A per-plane view reinterprets the image as a different format (R8, + // R8G8, ...), which the image must have been created MUTABLE_FORMAT to + // permit (VUID-VkImageViewCreateInfo-image-01762). For an image this + // library allocated that always holds; for one handed in by a caller -- + // a dma_buf import, where the EXPORTER chose the create flags -- it may + // not, and this overload used to assume it did. Test the flag, exactly + // as the planeUsageOverride overload above already does: the rule + // belongs in Create(), where it holds for every caller, rather than in + // a private entry point each caller has to remember to pick. + const bool planeViewsPermitted = + (imageResource->GetImageCreateInfo().flags & + VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0; + if (mpInfo && planeViewsPermitted) { viewInfo.pNext = nullptr; // These views are bound as storage images to the YCbCr compute filter, whose // generated GLSL declares them as image2DArray and addresses them with diff --git a/common/libs/VkCodecUtils/VkImageResource.h b/common/libs/VkCodecUtils/VkImageResource.h index 54d2b9c6..6f47121f 100644 --- a/common/libs/VkCodecUtils/VkImageResource.h +++ b/common/libs/VkCodecUtils/VkImageResource.h @@ -304,6 +304,37 @@ class VkImageResourceView : public VkVideoRefCountBase VkImageUsageFlags combinedViewUsage, VkSharedBaseObj& imageResourceView); + /** + * @brief Per-plane views are a PROPERTY OF THE IMAGE, not of the caller. + * + * Both Create() overloads that can build per-plane views test + * VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT on the image's own create info + * first, and build none without it: a per-plane view reinterprets the + * image as R8 / R8G8 / ..., which only a mutable-format image permits + * (VUID-VkImageViewCreateInfo-image-01762). For an image this library + * allocated the flag is always set; for a registered EXTERNAL image the + * EXPORTER chose the create flags, so the answer is whatever the + * descriptor declared -- which is exactly why the encoder's registration + * path may ask for plane views without having to know, per registration, + * whether they are legal. + * + * When the flag is absent the wrapper still carries its combined view and + * reports GetNumberOfPlanes() == 0. That is NOT a graceful degradation for + * every consumer, and the difference matters to callers: VulkanFilter- + * YuvCompute trims its plane bindings by that count + * (UpdateImageDescriptorSets), but for a MULTI-PLANAR input it has no + * combined-view binding to fall back to -- ShaderGenerateImagePlane- + * Descriptors overwrites m_inputImageAspects with the PLANE bits and + * clears VK_IMAGE_ASPECT_COLOR_BIT -- so trimming to zero planes writes + * ZERO input descriptors and the dispatch reads unbound STORAGE_IMAGE + * bindings. Bad handles are what this avoids; a consumer that needs plane + * views must CHECK GetNumberOfPlanes() and refuse, which is what + * VkVideoEncoder::StageInputFrame does for external input. + * + * |combinedViewUsage| must be a non-zero subset of the image's usage; for + * a multi-planar format it must not include STORAGE or SAMPLED + * (VUID-VkImageViewCreateInfo-pNext-02662 / -usage-06415). + */ operator VkImageView() const { // Fall back to first plane view if combined view is null (storage-only case) diff --git a/common/libs/VkCodecUtils/VkThreadPool.h b/common/libs/VkCodecUtils/VkThreadPool.h index b9d5a508..fb3d1496 100644 --- a/common/libs/VkCodecUtils/VkThreadPool.h +++ b/common/libs/VkCodecUtils/VkThreadPool.h @@ -44,19 +44,30 @@ class VkThreadPool task = std::move(this->tasks.front()); this->tasks.pop(); } +#if defined(__cpp_exceptions) + // Exceptions-enabled consumers (the CLI apps): log and + // keep the worker alive; an escaping exception would + // otherwise terminate the process from a detached + // worker thread with no actionable context. try { task(); } catch (const std::exception& e) { - std::cerr << "Task threw an exception: " << e.what() << std::endl; + std::cerr << "VkThreadPool task threw: " << e.what() + << std::endl; } +#else + // Chromium builds with -fno-exceptions: nothing can be + // caught here; an escaping exception is process-fatal. + task(); +#endif } }); } template auto enqueue(F&& f, Args&&... args) - -> std::future::type> { - using return_type = typename std::result_of::type; + -> std::future::type> { + using return_type = typename std::invoke_result::type; auto task = std::make_shared< std::packaged_task >( std::bind(std::forward(f), std::forward(args)...) diff --git a/common/libs/VkCodecUtils/VkThreadSafeQueue.h b/common/libs/VkCodecUtils/VkThreadSafeQueue.h index 0b1cc0e2..d12610d7 100644 --- a/common/libs/VkCodecUtils/VkThreadSafeQueue.h +++ b/common/libs/VkCodecUtils/VkThreadSafeQueue.h @@ -17,6 +17,8 @@ #ifndef _VKCODECUTILS_VKTHREADSAFEQUEUE_H_ #define _VKCODECUTILS_VKTHREADSAFEQUEUE_H_ +#include // chromium: needed for uint32_t under -fmodules + #include #include #include @@ -40,8 +42,15 @@ class VkThreadSafeQueue { return false; } - // Wait for the consumer to consume the previous node item(s) - m_condProducer.wait(lock, [this]{ return (!m_queueIsFlushing && (m_queue.size() < m_maxPendingQueueNodes)); }); + // Wait for the consumer to consume the previous node item(s). The + // predicate must also wake on the flush latch: it is sticky and never + // resets, so a producer waiting only for space would block forever if + // the flush lands mid-wait -- the entry check above cannot see a latch + // raised after it ran. + m_condProducer.wait(lock, [this]{ return (m_queueIsFlushing || (m_queue.size() < m_maxPendingQueueNodes)); }); + if (m_queueIsFlushing) { + return false; + } m_queue.push(node); m_condConsumer.notify_one(); @@ -79,7 +88,7 @@ class VkThreadSafeQueue { return m_queue.empty(); } - bool Size() const { + size_t Size() const { std::lock_guard lock(m_mutex); return m_queue.size(); } @@ -99,6 +108,32 @@ class VkThreadSafeQueue { return ((m_queueIsFlushing == true) && m_queue.empty()); } + // Clear the flush latch so a DRAINED queue can be used again. + // + // SetFlushAndExit() is deliberately sticky -- every consumer above treats + // it as "this queue is dead" -- which makes it correct for teardown and + // wrong for a NON-TERMINAL drain, where the whole point is that the + // session continues. This is the only way back, and it is guarded: + // + // * REFUSES on a non-empty queue. Un-latching with items still in + // flight would let a producer push behind items no consumer is + // committed to draining. + // * The caller must have JOINED every consumer first. That cannot be + // checked here -- the queue does not own the threads -- so it is + // stated: clearing the latch under a live consumer re-arms its + // WaitAndPop against a queue whose ordering counters the caller is + // about to touch. + // + // Returns false, changing nothing, when the queue is not drained. + bool ClearFlushAndReuse() { + std::unique_lock lock(m_mutex); + if (!m_queue.empty()) { + return false; + } + m_queueIsFlushing = false; + return true; + } + private: bool TryPopNoLock(QueueNodeType& node) { diff --git a/common/libs/VkCodecUtils/VkVideoCrc.cpp b/common/libs/VkCodecUtils/VkVideoCrc.cpp index fa696014..012e70ee 100644 --- a/common/libs/VkCodecUtils/VkVideoCrc.cpp +++ b/common/libs/VkCodecUtils/VkVideoCrc.cpp @@ -15,6 +15,7 @@ */ #include "VkCodecUtils/VkVideoCrc.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include @@ -132,7 +133,7 @@ bool VkVideoCrc::BeginCrcCalculation(const std::vector& crcInitValue, if (!crcOutputFileName.empty()) { m_file = fopen(crcOutputFileName.c_str(), "w"); if (m_file == nullptr) { - fprintf(stderr, "\nWarning: Failed to open CRC output file '%s'.\n", crcOutputFileName.c_str()); + VkEncPrintfErr("\nWarning: Failed to open CRC output file '%s'.\n", crcOutputFileName.c_str()); m_initValue.clear(); m_accumulatedCrc.clear(); m_currentFrameCrc.clear(); diff --git a/common/libs/VkCodecUtils/VulkanBistreamBufferImpl.cpp b/common/libs/VkCodecUtils/VulkanBistreamBufferImpl.cpp index 8674acbc..2754c4d3 100644 --- a/common/libs/VkCodecUtils/VulkanBistreamBufferImpl.cpp +++ b/common/libs/VkCodecUtils/VulkanBistreamBufferImpl.cpp @@ -15,6 +15,7 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "VkCodecUtils/VulkanBistreamBufferImpl.h" #include "VkCodecUtils/Helpers.h" @@ -231,7 +232,7 @@ VkDeviceSize VulkanBitstreamBufferImpl::Resize(VkDeviceSize newSize, VkDeviceSiz return m_bufferSize; } - std::cout << " ======= Req resize old " << m_bufferSize << " -> new " << newSize << " ====== " << std::endl; + VkEncOut() << " ======= Req resize old " << m_bufferSize << " -> new " << newSize << " ====== " << std::endl; VkBuffer newBuffer = VK_NULL_HANDLE; VkDeviceSize newBufferOffset = 0; diff --git a/common/libs/VkCodecUtils/VulkanCommandBufferPool.h b/common/libs/VkCodecUtils/VulkanCommandBufferPool.h index 56ea0c4e..32131005 100644 --- a/common/libs/VkCodecUtils/VulkanCommandBufferPool.h +++ b/common/libs/VkCodecUtils/VulkanCommandBufferPool.h @@ -127,6 +127,22 @@ class VulkanCommandBufferPool : public VkVideoRefCountBase, return true; } + // Has this node's command buffer been handed to vkQueueSubmit yet? + // + // Read by the encoder ext layer's per-frame release fence. Exporting + // a SYNC_FD requires the semaphore to be signalled or to have a + // signal operation PENDING EXECUTION, so the export is only legal + // once the batch carrying that signal has been submitted -- and the + // encode submit is NOT always issued inline with the frame that + // produced it: under B-frame reordering the frame sits in the + // deferred queue and its submit happens on a later call. Asking the + // node is the only answer that cannot drift from that scheduling + // decision, which is why this is a query and not a config-derived + // prediction. + bool IsCommandBufferSubmitted() const { + return (m_cmdBufState == CmdBufStateSubmitted); + } + VkFence GetFence() const { if ((m_parent == nullptr) || (m_parentIndex < 0)) { assert(!"Invalid PoolNode state!"); diff --git a/common/libs/VkCodecUtils/VulkanComputePipeline.cpp b/common/libs/VkCodecUtils/VulkanComputePipeline.cpp index 911d468c..3b9f41db 100644 --- a/common/libs/VkCodecUtils/VulkanComputePipeline.cpp +++ b/common/libs/VkCodecUtils/VulkanComputePipeline.cpp @@ -15,6 +15,7 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include "VulkanComputePipeline.h" @@ -46,7 +47,7 @@ VkResult VulkanComputePipeline::CreatePipeline(const VulkanDeviceContext* vkDevC const bool verbose = false; - if (verbose) printf("\nCompute shader code:\n %s", shaderCode); + if (verbose) VkEncPrintfOut("\nCompute shader code:\n %s", shaderCode); DestroyShaderModule(); m_shaderModule = shaderCompiler.BuildGlslShader(shaderCode, @@ -61,7 +62,7 @@ VkResult VulkanComputePipeline::CreatePipeline(const VulkanDeviceContext* vkDevC // generator that emits invalid GLSL is a bug worth seeing, so it has to surface as // an error. This guard covers every compute filter, not only the generators known // to be able to emit invalid GLSL. - std::cerr << "VulkanComputePipeline: shader failed to compile; " + VkEncErr() << "VulkanComputePipeline: shader failed to compile; " "see the compiler diagnostics above. Pipeline not created." << std::endl; return VK_ERROR_INITIALIZATION_FAILED; diff --git a/common/libs/VkCodecUtils/VulkanDescriptorSetLayout.cpp b/common/libs/VkCodecUtils/VulkanDescriptorSetLayout.cpp index 79096569..984f18b7 100644 --- a/common/libs/VkCodecUtils/VulkanDescriptorSetLayout.cpp +++ b/common/libs/VkCodecUtils/VulkanDescriptorSetLayout.cpp @@ -15,6 +15,7 @@ */ #include "VulkanDescriptorSetLayout.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "VkCodecUtils/Helpers.h" // for alignedSize VkResult VulkanDescriptorSetLayout::CreateFragmentShaderLayouts(const uint32_t* setIds, uint32_t numSets, std::stringstream& imageFss) @@ -58,7 +59,7 @@ VkResult VulkanDescriptorSetLayout::CreateFragmentShaderLayouts(const uint32_t* } } } - // printf("\nFragment shader layout code:\n %s", imageFss.str().c_str()); + // VkEncPrintfOut("\nFragment shader layout code:\n %s", imageFss.str().c_str()); return VK_SUCCESS; } @@ -215,7 +216,7 @@ VkResult VulkanDescriptorSetLayout::CreateFragmentShaderOutput(VkDescriptorType break; } - // printf("\nFragment shader output code:\n %s", imageFss.str().c_str()); + // VkEncPrintfOut("\nFragment shader output code:\n %s", imageFss.str().c_str()); return VK_SUCCESS; } diff --git a/common/libs/VkCodecUtils/VulkanDeviceContext.cpp b/common/libs/VkCodecUtils/VulkanDeviceContext.cpp index 74f82056..dfc46fcd 100644 --- a/common/libs/VkCodecUtils/VulkanDeviceContext.cpp +++ b/common/libs/VkCodecUtils/VulkanDeviceContext.cpp @@ -30,10 +30,18 @@ #include // std::find_if #include // validation-error counter #include "VkCodecUtils/VulkanDeviceContext.h" +// Gated-logging latch (VkEncOut()/IsVkEncoderStdioSilenced()): declares the +// shared process-wide latch; lives in VkCodecUtils so this common code +// carries no include-path dependency on vk_video_encoder. +#include "VkCodecUtils/VkEncoderStdioLatch.h" #ifdef VIDEO_DISPLAY_QUEUE_SUPPORT #include "VkShell/Shell.h" #endif // VIDEO_DISPLAY_QUEUE_SUPPORT +// The silence counter's definition moved to VkEncoderStdioLatch.cpp, so a +// target can route output through the gate without compiling the whole +// device context. See that file for why there is exactly one. + #if !defined(VK_USE_PLATFORM_WIN32_KHR) PFN_vkGetInstanceProcAddr VulkanDeviceContext::LoadVk(VulkanLibraryHandleType &vulkanLibHandle, const char * pCustomLoader) @@ -127,23 +135,23 @@ VkResult VulkanDeviceContext::CheckAllInstanceLayers(bool verbose) std::vector layers; vk::enumerate(this, layers); - if (verbose) std::cout << "Enumerating instance layers:" << std::endl; + if (verbose) VkEncOut() << "Enumerating instance layers:" << std::endl; std::set layer_names; for (const auto &layer : layers) { layer_names.insert(layer.layerName); - if (verbose ) std::cout << '\t' << layer.layerName << std::endl; + if (verbose ) VkEncOut() << '\t' << layer.layerName << std::endl; } // all listed instance layers are required - if (verbose) std::cout << "Looking for instance layers:" << std::endl; + if (verbose) VkEncOut() << "Looking for instance layers:" << std::endl; for (uint32_t i = 0; i < m_reqInstanceLayers.size(); i++) { const char* name = m_reqInstanceLayers[i]; if (name == nullptr) { break; } - std::cout << '\t' << name << std::endl; + VkEncOut() << '\t' << name << std::endl; if (layer_names.find(name) == layer_names.end()) { - std::cerr << "AssertAllInstanceLayers() ERROR: requested instance layer" + VkEncErr() << "AssertAllInstanceLayers() ERROR: requested instance layer" << name << " is missing!" << std::endl << std::flush; return VK_ERROR_LAYER_NOT_PRESENT; } @@ -181,23 +189,23 @@ VkResult VulkanDeviceContext::CheckAllInstanceExtensions(bool verbose) std::vector exts; vk::enumerate(this, nullptr, exts); - if (verbose) std::cout << "Enumerating instance extensions:" << std::endl; + if (verbose) VkEncOut() << "Enumerating instance extensions:" << std::endl; std::set ext_names; for (const auto &ext : exts) { ext_names.insert(ext.extensionName); - if (verbose) std::cout << '\t' << ext.extensionName << std::endl; + if (verbose) VkEncOut() << '\t' << ext.extensionName << std::endl; } // all listed instance extensions are required - if (verbose) std::cout << "Looking for instance extensions:" << std::endl; + if (verbose) VkEncOut() << "Looking for instance extensions:" << std::endl; for (uint32_t i = 0; i < m_reqInstanceExtensions.size(); i++) { const char* name = m_reqInstanceExtensions[i]; if (name == nullptr) { break; } - if (verbose) std::cout << '\t' << name << std::endl; + if (verbose) VkEncOut() << '\t' << name << std::endl; if (ext_names.find(name) == ext_names.end()) { - std::cerr << "AssertAllInstanceExtensions() ERROR: requested instance extension " + VkEncErr() << "AssertAllInstanceExtensions() ERROR: requested instance extension " << name << " is missing!" << std::endl << std::flush; return VK_ERROR_EXTENSION_NOT_PRESENT; } @@ -215,7 +223,7 @@ VkResult VulkanDeviceContext::AddReqDeviceExtensions(const char* const* required } m_requestedDeviceExtensions.push_back(name); if (verbose) { - std::cout << "Added required device extension: " << name << std::endl; + VkEncOut() << "Added required device extension: " << name << std::endl; } } @@ -227,7 +235,7 @@ VkResult VulkanDeviceContext::AddReqDeviceExtension(const char* requiredDeviceEx if (requiredDeviceExtension) { m_requestedDeviceExtensions.push_back(requiredDeviceExtension); if (verbose) { - std::cout << "Added required device extension: " << requiredDeviceExtension << std::endl; + VkEncOut() << "Added required device extension: " << requiredDeviceExtension << std::endl; } } @@ -245,7 +253,7 @@ VkResult VulkanDeviceContext::AddOptDeviceExtensions(const char* const* optional } m_optDeviceExtensions.push_back(name); if (verbose) { - std::cout << "Added optional device extension: " << name << std::endl; + VkEncOut() << "Added optional device extension: " << name << std::endl; } } @@ -274,7 +282,7 @@ bool VulkanDeviceContext::HasAllDeviceExtensions(VkPhysicalDevice physDevice, co if (ext_names.find(name) == ext_names.end()) { hasAllRequiredExtensions = false; if (printMissingDeviceExt) { - std::cerr << __FUNCTION__ + VkEncErr() << __FUNCTION__ << ": ERROR: required device extension " << name << " is missing for device with name: " << printMissingDeviceExt << std::endl << std::flush; @@ -294,7 +302,7 @@ bool VulkanDeviceContext::HasAllDeviceExtensions(VkPhysicalDevice physDevice, co } if (ext_names.find(name) == ext_names.end()) { if (printMissingDeviceExt) { - std::cout << __FUNCTION__ + VkEncOut() << __FUNCTION__ << " : WARNING: requested optional device extension " << name << " is missing for device with name: " << printMissingDeviceExt << std::endl << std::flush; @@ -322,7 +330,7 @@ static int DumpSoLibs() auto* map = reinterpret_cast(p->ptr); while (map) { - std::cout << map->l_name << std::endl; + VkEncOut() << map->l_name << std::endl; // do something with |map| like with handle, returned by |dlopen()|. map = map->l_next; } @@ -333,13 +341,36 @@ static int DumpSoLibs() VkResult VulkanDeviceContext::InitVkInstance(const char * pAppName, VkInstance vkInstance, bool verbose) { - VkResult result = CheckAllInstanceLayers(verbose); - if (result != VK_SUCCESS) { - return result; - } - result = CheckAllInstanceExtensions(verbose); - if (result != VK_SUCCESS) { - return result; + VkResult result = VK_SUCCESS; + + // THESE TWO CHECKS INTERROGATE THE SYSTEM LOADER, NOT AN INSTANCE. + // + // They exist to answer one question -- "will the vkCreateInstance below + // succeed with m_reqInstanceLayers / m_reqInstanceExtensions?" -- and that + // question only arises on the path that actually creates an instance. + // + // On the ADOPT path (an instance the EMBEDDER created and this library + // merely borrows) they answered a different question and answered it + // confidently. Whether the loader OFFERS VK_LAYER_KHRONOS_validation or + // VK_EXT_debug_report says nothing about whether the embedder ENABLED + // either of them on the instance being handed over -- and Vulkan provides + // no way to ask an existing VkInstance what it was created with. A + // VK_SUCCESS here was then read downstream as permission to use the + // extension, which is how InitDebugReport() came to call a null dispatch + // entry and take the process down with it. See the note there. + // + // So: run them when creating, skip them when importing. Skipping is not a + // loss of coverage -- the checks never covered the imported instance in + // the first place; they only looked as though they did. + if (vkInstance == VK_NULL_HANDLE) { + result = CheckAllInstanceLayers(verbose); + if (result != VK_SUCCESS) { + return result; + } + result = CheckAllInstanceExtensions(verbose); + if (result != VK_SUCCESS) { + return result; + } } VkApplicationInfo app_info = {}; @@ -512,7 +543,7 @@ bool VulkanDeviceContext::DebugReportCallback(VkDebugReportFlagsEXT flags, VkDeb std::stringstream ss; ss << layer_prefix << ": " << msg; - std::ostream &st = (prio >= LOG_ERR) ? std::cerr : std::cout; + std::ostream &st = (prio >= LOG_ERR) ? VkEncErr() : VkEncOut(); st << msg << "\n"; return false; @@ -550,7 +581,7 @@ VKAPI_ATTR VkBool32 VKAPI_CALL VulkanDeviceContext::DebugUtilsMessengerCallback( (messageSeverity & VK_DEBUG_UTILS_MESSAGE_SEVERITY_WARNING_BIT_EXT) ? "Warning" : (messageSeverity & VK_DEBUG_UTILS_MESSAGE_SEVERITY_INFO_BIT_EXT) ? "Info" : "Debug"; - std::ostream &st = (messageSeverity & VK_DEBUG_UTILS_MESSAGE_SEVERITY_ERROR_BIT_EXT) ? std::cerr : std::cout; + std::ostream &st = (messageSeverity & VK_DEBUG_UTILS_MESSAGE_SEVERITY_ERROR_BIT_EXT) ? VkEncErr() : VkEncOut(); st << "Validation " << severity << ": [ " << (pCallbackData->pMessageIdName ? pCallbackData->pMessageIdName : "") << " ] | MessageID = 0x" << std::hex << pCallbackData->messageIdNumber << std::dec << "\n" << pCallbackData->pMessage << "\n" << std::endl; @@ -564,6 +595,51 @@ VkResult VulkanDeviceContext::InitDebugReport(bool validate, bool validateVerbos return VK_SUCCESS; } + // AN IMPORTED (ADOPT-MODE) INSTANCE GETS NO CALLBACK OF OURS, and that is + // a refusal rather than an oversight. + // + // A debug callback is an INSTANCE-level object: creating one requires the + // instance to have been created with VK_EXT_debug_utils or + // VK_EXT_debug_report ENABLED. On an adopted instance this library did not + // call vkCreateInstance, and Vulkan offers no way to ask an existing + // VkInstance which extensions it carries. The two things that look like an + // answer are both wrong: + // * the loader's instance-extension list (CheckAllInstanceExtensions) + // describes the LOADER, not the instance -- which is why that check is + // now confined to the create path; + // * GetInstanceProcAddr hands back a non-null trampoline for an instance + // extension whenever any layer or ICD implements it, enabled on THIS + // instance or not. + // + // Guessing wrong is not a status code. With the validation layer + // present in the loader but NOT enabled on the borrowed instance, the + // debug-utils probe below comes back null, control falls through to + // CreateDebugReportCallbackEXT, and that dispatch-table entry is null + // too -- a call through a null pointer inside this function, before the + // session has created anything at all. + // + // There is a second and independent reason, and it is the ADOPT lifetime + // rule rather than a crash. A callback we create is OURS to destroy, on an + // instance whose lifetime belongs to the embedder; ~VulkanDeviceContext + // destroys the messenger BEFORE it declines to destroy the imported + // instance, so an embedder that dropped its instance first would have us + // calling into a dead one. ADOPT retains and destroys nothing of the + // caller's, and a debug callback is not an exception to that. + // + // Validation itself still works over an adopted instance: the embedder's + // layer and the embedder's callback are what report. All that is skipped + // here is this library adding a second, redundant reporting channel to an + // object it does not own. + if (m_importedInstanceHandle) { + VkEncOut() << "VulkanDeviceContext: validation was requested on an " + "IMPORTED VkInstance -- not attaching a debug callback. " + "The instance belongs to the embedder, which is the only " + "party that knows which debug extensions it enabled and " + "the only one whose reporting may outlive this session." + << std::endl << std::flush; + return VK_SUCCESS; + } + // Prefer VK_EXT_debug_utils over VK_EXT_debug_report. // debug_utils provides messageIdNumber for reliable VUID filtering // and is the non-deprecated API. Load extension procs via GetInstanceProcAddr. @@ -608,6 +684,22 @@ VkResult VulkanDeviceContext::InitDebugReport(bool validate, bool validateVerbos debug_report_info.pfnCallback = debugReportCallback; debug_report_info.pUserData = reinterpret_cast(this); + // A null dispatch entry means "this instance has no VK_EXT_debug_report", + // which is a diagnostic we do without -- not a reason to jump to address + // 0. The imported-instance case above is what made this reachable, but the + // guard is deliberately unconditional: the table is filled by resolving + // names through the loader (HelpersDispatchTable.cpp), so any entry in it + // can be null for reasons this code does not control, and the one thing + // that must never happen is that a request for DIAGNOSTICS kills the + // process it was meant to diagnose. + if (CreateDebugReportCallbackEXT == nullptr) { + VkEncOut() << "VulkanDeviceContext: neither VK_EXT_debug_utils nor " + "VK_EXT_debug_report resolved on this instance; " + "continuing without a library debug callback." + << std::endl << std::flush; + return VK_SUCCESS; + } + return CreateDebugReportCallbackEXT(m_instance, &debug_report_info, nullptr, &m_debugReport); } @@ -632,7 +724,20 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev } m_physDevice = VK_NULL_HANDLE; + + // Device-extension accumulation is per candidate. HasAllDeviceExtensions() + // below calls AddRequiredDeviceExtension() for every required AND optional + // extension it finds on the candidate it is inspecting, and that function is + // a bare push_back with no de-duplication. Because the selection block + // returns VK_SUCCESS as soon as a device is chosen, anything a REJECTED + // candidate left behind would still be on the list handed to vkCreateDevice + // for the device actually selected. On a multi-GPU host that means requesting + // an optional extension only the rejected GPU advertised, which fails device + // creation with VK_ERROR_EXTENSION_NOT_PRESENT -- and duplicate names besides. + // Reset to the caller-supplied baseline before inspecting each candidate. + const size_t reqDeviceExtensionsBaseline = m_reqDeviceExtensions.size(); for (auto physicalDevice : availablePhysicalDevices) { + m_reqDeviceExtensions.resize(reqDeviceExtensionsBaseline); // Get Vulkan 1.1 specific properties which include deviceUUID @@ -654,7 +759,7 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev if (!deviceUuid.Compare( deviceVulkan11Properties.deviceUUID)) { vk::DeviceUuidUtils deviceUuid(deviceVulkan11Properties.deviceUUID); - std::cout << "*** Skipping vulkan physical device with NOT matching UUID: " + VkEncOut() << "*** Skipping vulkan physical device with NOT matching UUID: " << "Device Name: " << devProp2.properties.deviceName << std::hex << ", vendor ID: " << devProp2.properties.vendorID << ", device UUID: " << deviceUuid.ToString() @@ -667,7 +772,7 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev } if (!HasAllDeviceExtensions(physicalDevice, devProp2.properties.deviceName)) { - std::cerr << "ERROR: Found physical device with name: " << devProp2.properties.deviceName << std::hex + VkEncErr() << "ERROR: Found physical device with name: " << devProp2.properties.deviceName << std::hex << ", vendor ID: " << devProp2.properties.vendorID << ", and device ID: " << devProp2.properties.deviceID << std::dec << " NOT having the required extensions!" << std::endl << std::flush; @@ -723,19 +828,19 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev videoDecodeQueueFamily = i; videoDecodeQueueCount = queue.queueFamilyProperties.queueCount; - if (dumpQueues) std::cout << "\t Found video decode only queue family " << i << + if (dumpQueues) VkEncOut() << "\t Found video decode only queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; // Does the video decode queue also support transfer operations? if (queueFamilyFlags & VK_QUEUE_TRANSFER_BIT) { - if (dumpQueues) std::cout << "\t\t Video decode queue " << i << + if (dumpQueues) VkEncOut() << "\t\t Video decode queue " << i << " supports transfer operations" << std::endl; } // Does the video decode queue also support compute operations? if (queueFamilyFlags & VK_QUEUE_COMPUTE_BIT) { - if (dumpQueues) std::cout << "\t\t Video decode queue " << i << + if (dumpQueues) VkEncOut() << "\t\t Video decode queue " << i << " supports compute operations" << std::endl; } @@ -751,19 +856,19 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev videoEncodeQueueFamily = i; videoEncodeQueueCount = queue.queueFamilyProperties.queueCount; - if (dumpQueues) std::cout << "\t Found video encode only queue family " << i << + if (dumpQueues) VkEncOut() << "\t Found video encode only queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; // Does the video encode queue also support transfer operations? if (queueFamilyFlags & VK_QUEUE_TRANSFER_BIT) { - if (dumpQueues) std::cout << "\t\t Video encode queue " << i << + if (dumpQueues) VkEncOut() << "\t\t Video encode queue " << i << " supports transfer operations" << std::endl; } // Does the video encode queue also support compute operations? if (queueFamilyFlags & VK_QUEUE_COMPUTE_BIT) { - if (dumpQueues) std::cout << "\t\t Video encode queue " << i << + if (dumpQueues) VkEncOut() << "\t\t Video encode queue " << i << " supports compute operations" << std::endl; } @@ -782,7 +887,7 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev transferQueueFamily = i; } foundQueueTypes |= queueFamilyFlags; - if (dumpQueues) std::cout << "\t Found graphics queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; + if (dumpQueues) VkEncOut() << "\t Found graphics queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; } else if ((requestQueueTypes & VK_QUEUE_COMPUTE_BIT) && (computeQueueFamilyOnly < 0) && ((VK_QUEUE_COMPUTE_BIT | VK_QUEUE_TRANSFER_BIT) == (queueFamilyFlags & (VK_QUEUE_COMPUTE_BIT | VK_QUEUE_TRANSFER_BIT)))) { computeQueueFamilyOnly = i; @@ -790,12 +895,12 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev if ((transferQueueFamily < 0) && !!(queueFamilyFlags & VK_QUEUE_TRANSFER_BIT)) { transferQueueFamily = i; } - if (dumpQueues) std::cout << "\t Found compute only queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; + if (dumpQueues) VkEncOut() << "\t Found compute only queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; } else if ((requestQueueTypes & VK_QUEUE_TRANSFER_BIT) && (transferQueueFamilyOnly < 0) && (VK_QUEUE_TRANSFER_BIT == (queueFamilyFlags & VK_QUEUE_TRANSFER_BIT))) { transferQueueFamilyOnly = i; foundQueueTypes |= queueFamilyFlags; - if (dumpQueues) std::cout << "\t Found transfer only queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; + if (dumpQueues) VkEncOut() << "\t Found transfer only queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; } // requires only COMPUTE for frameProcessor queues @@ -803,13 +908,13 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev (queueFamilyFlags & VK_QUEUE_COMPUTE_BIT)) { computeQueueFamily = i; foundQueueTypes |= queueFamilyFlags; - if (dumpQueues) std::cout << "\t Found compute queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; + if (dumpQueues) VkEncOut() << "\t Found compute queue family " << i << " with " << queue.queueFamilyProperties.queueCount << " max num of queues." << std::endl; } // present queue must support the surface if ((pWsiDisplay != nullptr) && (presentQueueFamily < 0) && pWsiDisplay->PhysDeviceCanPresent(physicalDevice, i)) { - if (dumpQueues) std::cout << "\t Found present queue family " << i << "." << std::endl; + if (dumpQueues) VkEncOut() << "\t Found present queue family " << i << "." << std::endl; presentQueueFamily = i; } @@ -842,7 +947,7 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev if (true) { vk::DeviceUuidUtils deviceUuid(deviceVulkan11Properties.deviceUUID); - std::cout << "*** Selected Vulkan physical device with name: " << devProp2.properties.deviceName << std::hex + VkEncOut() << "*** Selected Vulkan physical device with name: " << devProp2.properties.deviceName << std::hex << ", vendor ID: " << devProp2.properties.vendorID << ", device UUID: " << deviceUuid.ToString() << ", and device ID: " << devProp2.properties.deviceID << std::dec @@ -854,7 +959,7 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev return VK_SUCCESS; } } - std::cerr << "ERROR: Found physical device with name: " << devProp2.properties.deviceName << std::hex + VkEncErr() << "ERROR: Found physical device with name: " << devProp2.properties.deviceName << std::hex << ", vendor ID: " << devProp2.properties.vendorID << ", and device ID: " << devProp2.properties.deviceID << std::dec << " NOT having the required queue families!" << std::endl << std::flush; @@ -863,6 +968,108 @@ VkResult VulkanDeviceContext::InitPhysicalDevice(int32_t deviceId, const vk::Dev return (m_physDevice != VK_NULL_HANDLE) ? VK_SUCCESS : VK_ERROR_FEATURE_NOT_PRESENT; } +VkResult VulkanDeviceContext::OverrideImportedQueueFamilies( + uint32_t videoEncodeQueueFamilyIndex, + VkVideoCodecOperationFlagsKHR videoEncodeQueueOperations, + uint32_t computeQueueFamilyIndex) +{ + if (m_physDevice == VK_NULL_HANDLE) { + assert(!"OverrideImportedQueueFamilies requires InitPhysicalDevice() first"); + return VK_ERROR_INITIALIZATION_FAILED; + } + + if ((videoEncodeQueueFamilyIndex == UINT32_MAX) && + (computeQueueFamilyIndex == UINT32_MAX)) { + return VK_SUCCESS; // nothing to override + } + + // Re-query the family table (with the video/query-status pNext chains) + // so the override can validate flags and refresh the cached + // per-family metadata the encoder relies on. + std::vector queues; + std::vector videoQueues; + std::vector queryResultStatus; + vk::get(this, m_physDevice, queues, videoQueues, queryResultStatus); + + // A10.3: the three arrays are filled in parallel, one entry per queue + // family, and the bounds check below validates an index against |queues| + // ONLY -- then uses it to index |videoQueues|. If they ever disagree that + // is an out-of-bounds read past a check that appeared to cover it. + // Require the invariant instead of assuming it. + if ((videoQueues.size() != queues.size()) || + (queryResultStatus.size() != queues.size())) { + VkEncErr() << "[VulkanDeviceContext] queue-family property arrays " + << "disagree in length (queues=" << queues.size() + << ", video=" << videoQueues.size() + << ", queryStatus=" << queryResultStatus.size() + << "); refusing to index them against one another" + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + if (videoEncodeQueueFamilyIndex != UINT32_MAX) { + if (videoEncodeQueueFamilyIndex >= queues.size()) { + VkEncErr() << "[VulkanDeviceContext] caller-provided encode " + << "queue family " << videoEncodeQueueFamilyIndex + << " out of range (device has " << queues.size() + << " families)" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + const VkQueueFlags familyFlags = + queues[videoEncodeQueueFamilyIndex].queueFamilyProperties.queueFlags; + const VkVideoCodecOperationFlagsKHR familyOps = + videoQueues[videoEncodeQueueFamilyIndex].videoCodecOperations; + if (((familyFlags & VK_QUEUE_VIDEO_ENCODE_BIT_KHR) == 0) || + ((videoEncodeQueueOperations != 0) && + ((familyOps & videoEncodeQueueOperations) == 0))) { + VkEncErr() << "[VulkanDeviceContext] caller-provided encode " + << "queue family " << videoEncodeQueueFamilyIndex + << " lacks VIDEO_ENCODE support for the requested codec " + << "(flags=0x" << std::hex << familyFlags + << ", codecOps=0x" << familyOps << std::dec << ")" + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + m_videoEncodeQueueFamily = (int32_t)videoEncodeQueueFamilyIndex; + m_videoEncodeNumQueues = (int32_t)queues[videoEncodeQueueFamilyIndex] + .queueFamilyProperties.queueCount; + m_videoEncodeQueueFlags = familyFlags & + (VK_QUEUE_GRAPHICS_BIT | VK_QUEUE_COMPUTE_BIT | + VK_QUEUE_TRANSFER_BIT | VK_QUEUE_VIDEO_DECODE_BIT_KHR | + VK_QUEUE_VIDEO_ENCODE_BIT_KHR); + m_videoEncodeQueryResultStatusSupport = + queryResultStatus[videoEncodeQueueFamilyIndex].queryResultStatusSupport; + VkEncOut() << "VulkanDeviceContext: using caller-provided encode " + << "queue family " << videoEncodeQueueFamilyIndex + << " for the imported VkDevice" << std::endl; + } + + if (computeQueueFamilyIndex != UINT32_MAX) { + if (computeQueueFamilyIndex >= queues.size()) { + VkEncErr() << "[VulkanDeviceContext] caller-provided compute " + << "queue family " << computeQueueFamilyIndex + << " out of range (device has " << queues.size() + << " families)" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + const VkQueueFlags familyFlags = + queues[computeQueueFamilyIndex].queueFamilyProperties.queueFlags; + if ((familyFlags & VK_QUEUE_COMPUTE_BIT) == 0) { + VkEncErr() << "[VulkanDeviceContext] caller-provided compute " + << "queue family " << computeQueueFamilyIndex + << " lacks VK_QUEUE_COMPUTE_BIT (flags=0x" << std::hex + << familyFlags << std::dec << ")" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + m_computeQueueFamily = (int32_t)computeQueueFamilyIndex; + VkEncOut() << "VulkanDeviceContext: using caller-provided compute " + << "queue family " << computeQueueFamilyIndex + << " for the imported VkDevice" << std::endl; + } + + return VK_SUCCESS; +} + VkResult VulkanDeviceContext::InitVulkanDevice(const char * pAppName, VkInstance vkInstance, bool verbose, @@ -892,7 +1099,6 @@ VkResult VulkanDeviceContext::CreateVulkanDevice(int32_t numDecodeQueues, VkDevice vkDevice) { if (vkDevice == VK_NULL_HANDLE) { - std::unordered_set uniqueQueueFamilies; VkDeviceCreateInfo devInfo = {}; devInfo.sType = VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO; devInfo.pNext = nullptr; @@ -916,28 +1122,59 @@ VkResult VulkanDeviceContext::CreateVulkanDevice(int32_t numDecodeQueues, // each ask for queueCount = 1 regardless of how many VIDEO queues were requested. const std::vector queuePriorities(std::max(1, (int)maxQueueInstances), 0.0f); std::array queueInfo = {}; - const bool isUnique = uniqueQueueFamilies.insert(m_gfxQueueFamily).second; - assert(isUnique); - if (!isUnique) { - return VK_ERROR_INITIALIZATION_FAILED; - } - if (createGraphicsQueue) { - queueInfo[devInfo.queueCreateInfoCount].sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; - queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = m_gfxQueueFamily; - queueInfo[devInfo.queueCreateInfoCount].queueCount = 1; - queueInfo[devInfo.queueCreateInfoCount].pQueuePriorities = queuePriorities.data(); - devInfo.queueCreateInfoCount++; - } - if (createPresentQueue && - !(m_presentQueueFamily != -1) && - uniqueQueueFamilies.insert(m_presentQueueFamily).second) { + // ONE ENTRY PER FAMILY, SIZED BY THE LARGEST ROLE THAT ASKED FOR IT. + // + // The bookkeeping this replaces had three separate faults. It + // reserved the graphics family in the uniqueness set unconditionally, + // so when graphics was NOT requested any other role sharing that + // family was silently dropped -- and then retrieved anyway. The + // present guard read !(m_presentQueueFamily != -1), which built the + // entry exactly when the family was INVALID and assigned -1 to a + // uint32_t queueFamilyIndex. And a family claimed by two roles kept + // whichever queueCount was written first, so a present or compute + // request could shrink a video family's count to one. + struct RequestedQueue { + int32_t family = -1; + uint32_t count = 0; + }; + std::array requested = {}; + uint32_t requestedFamilies = 0; + + // Records one role. |required| roles refuse an invalid family instead + // of being skipped: the caller asked for something no family can + // serve, and building a device without it would leave the retrieval + // below asking for a queue that does not exist. + auto requestQueue = [&](bool wanted, int32_t family, uint32_t count, + bool required) -> bool { + if (!wanted) { + return true; // not requested: reserve nothing + } + if (family < 0) { + return !required; + } + for (uint32_t i = 0; i < requestedFamilies; ++i) { + if (requested[i].family == family) { + requested[i].count = std::max(requested[i].count, count); + return true; + } + } + if (requestedFamilies >= requested.size()) { + return false; + } + requested[requestedFamilies].family = family; + requested[requestedFamilies].count = count; + requestedFamilies++; + return true; + }; - queueInfo[devInfo.queueCreateInfoCount].sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; - queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = m_presentQueueFamily; - queueInfo[devInfo.queueCreateInfoCount].queueCount = 1; - queueInfo[devInfo.queueCreateInfoCount].pQueuePriorities = queuePriorities.data(); - devInfo.queueCreateInfoCount++; + if (!requestQueue(createGraphicsQueue, m_gfxQueueFamily, 1, true)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + // Present always uses index 0 of its family, so it adds no count -- + // only the requirement that the family exist. + if (!requestQueue(createPresentQueue, m_presentQueueFamily, 1, true)) { + return VK_ERROR_INITIALIZATION_FAILED; } VkPhysicalDeviceVideoDecodeVP9FeaturesKHR videoDecodeVP9Feature { VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VIDEO_DECODE_VP9_FEATURES_KHR, @@ -1005,53 +1242,34 @@ VkResult VulkanDeviceContext::CreateVulkanDevice(int32_t numDecodeQueues, assert(synchronization2Features.synchronization2); if ((videoCodecs & VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) && !videoEncodeAV1Feature.videoEncodeAV1) { - std::cerr << "ERROR: AV1 encode requested but videoEncodeAV1 feature not supported" << std::endl; + VkEncErr() << "ERROR: AV1 encode requested but videoEncodeAV1 feature not supported" << std::endl; return VK_ERROR_FEATURE_NOT_PRESENT; } if ((videoCodecs & VK_VIDEO_CODEC_OPERATION_DECODE_VP9_BIT_KHR) && !videoDecodeVP9Feature.videoDecodeVP9) { - std::cerr << "ERROR: VP9 decode requested but videoDecodeVP9 feature not supported" << std::endl; + VkEncErr() << "ERROR: VP9 decode requested but videoDecodeVP9 feature not supported" << std::endl; return VK_ERROR_FEATURE_NOT_PRESENT; } devInfo.pNext = &deviceFeatures; - if ((numDecodeQueues > 0) && - (m_videoDecodeQueueFamily != -1) && - uniqueQueueFamilies.insert(m_videoDecodeQueueFamily).second) { - queueInfo[devInfo.queueCreateInfoCount].sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; - queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = m_videoDecodeQueueFamily; - queueInfo[devInfo.queueCreateInfoCount].queueCount = numDecodeQueues; - queueInfo[devInfo.queueCreateInfoCount].pQueuePriorities = queuePriorities.data(); - devInfo.queueCreateInfoCount++; - } - - if ((numEncodeQueues > 0) && - (m_videoEncodeQueueFamily != -1) && - uniqueQueueFamilies.insert(m_videoEncodeQueueFamily).second) { - queueInfo[devInfo.queueCreateInfoCount].sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; - queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = m_videoEncodeQueueFamily; - queueInfo[devInfo.queueCreateInfoCount].queueCount = numEncodeQueues; - queueInfo[devInfo.queueCreateInfoCount].pQueuePriorities = queuePriorities.data(); - devInfo.queueCreateInfoCount++; - } - - if (createComputeQueue && - (m_computeQueueFamily != -1) && - uniqueQueueFamilies.insert(m_computeQueueFamily).second) { - queueInfo[devInfo.queueCreateInfoCount].sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; - queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = m_computeQueueFamily; - queueInfo[devInfo.queueCreateInfoCount].queueCount = 1; - queueInfo[devInfo.queueCreateInfoCount].pQueuePriorities = queuePriorities.data(); - devInfo.queueCreateInfoCount++; + // Video/compute/transfer keep the existing tolerance: a role with no + // family is a device that cannot do it, and the caller finds out when + // it asks for the queue rather than at device creation. + if (!requestQueue(numDecodeQueues > 0, m_videoDecodeQueueFamily, + (uint32_t)numDecodeQueues, false) || + !requestQueue(numEncodeQueues > 0, m_videoEncodeQueueFamily, + (uint32_t)numEncodeQueues, false) || + !requestQueue(createComputeQueue, m_computeQueueFamily, 1, false) || + !requestQueue(createTransferQueue, m_transferQueueFamily, 1, false)) { + return VK_ERROR_INITIALIZATION_FAILED; } - if (createTransferQueue && - (m_transferQueueFamily != -1) && - uniqueQueueFamilies.insert(m_transferQueueFamily).second) { + for (uint32_t i = 0; i < requestedFamilies; ++i) { queueInfo[devInfo.queueCreateInfoCount].sType = VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO; - queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = m_transferQueueFamily; - queueInfo[devInfo.queueCreateInfoCount].queueCount = 1; + queueInfo[devInfo.queueCreateInfoCount].queueFamilyIndex = + (uint32_t)requested[i].family; + queueInfo[devInfo.queueCreateInfoCount].queueCount = requested[i].count; queueInfo[devInfo.queueCreateInfoCount].pQueuePriorities = queuePriorities.data(); devInfo.queueCreateInfoCount++; } @@ -1077,20 +1295,31 @@ VkResult VulkanDeviceContext::CreateVulkanDevice(int32_t numDecodeQueues, m_device = vkDevice; m_importedDeviceHandle = true; + // Log the imported-VkDevice path so an embedder can verify it is + // active. Gated on the silenceStdio latch via VkEncOut() so it + // stays quiet inside the sandboxed GPU process. + VkEncOut() << "VulkanDeviceContext: using caller-imported " + << "VkDevice (m_importedDeviceHandle=true)" + << std::endl; } vk::InitDispatchTableBottom(m_instance, m_device, this); - if (createGraphicsQueue) { + // Retrieve only what a family was actually selected for. An index of -1 + // reaching GetDeviceQueue as a uint32_t asks the driver for family + // 0xFFFFFFFF. On the created path the builder above guarantees an entry + // exists for each of these; on the imported path the caller's contract is + // that the queues exist, and this is the one part of it we can check. + if (createGraphicsQueue && (GetGfxQueueFamilyIdx() >= 0)) { GetDeviceQueue(m_device, GetGfxQueueFamilyIdx() , 0, &m_gfxQueue); } - if (createComputeQueue) { + if (createComputeQueue && (GetComputeQueueFamilyIdx() >= 0)) { GetDeviceQueue(m_device, GetComputeQueueFamilyIdx(), 0, &m_computeQueue); } - if (createPresentQueue) { + if (createPresentQueue && (GetPresentQueueFamilyIdx() >= 0)) { GetDeviceQueue(m_device, GetPresentQueueFamilyIdx(), 0, &m_presentQueue); } - if (createTransferQueue) { + if (createTransferQueue && (GetTransferQueueFamilyIdx() >= 0)) { GetDeviceQueue(m_device, GetTransferQueueFamilyIdx(), 0, &m_trasferQueue); } if (numDecodeQueues) { @@ -1117,7 +1346,18 @@ VkResult VulkanDeviceContext::CreateVulkanDevice(int32_t numDecodeQueues, } VulkanDeviceContext::VulkanDeviceContext() - : m_libHandle() + // NAMING THE BASE IS LOAD-BEARING. vk::VkInterfaceFunctions is a plain + // aggregate of function pointers with no constructor and no default + // member initializers, and this class has a user-provided constructor, so + // leaving the base out of this list DEFAULT-initializes it: every entry + // holds an indeterminate value until InitDispatchTable* runs. Reading one + // is UB, and GetVkGetInstanceProcAddr() promises callers nullptr before + // initialization -- a promise that was false for any embedder that took + // it at its word. The realistic non-null case is a recycled allocation: + // destroy one encoder, construct the next into the same chunk, and the + // stale table reads as live. {} value-initializes the whole base. + : vk::VkInterfaceFunctions{} + , m_libHandle() , m_instance() , m_physDevice() , m_gfxQueueFamily(-1) @@ -1136,6 +1376,7 @@ VulkanDeviceContext::VulkanDeviceContext() , m_videoDecodeQueueFlags(0) , m_videoEncodeQueueFlags(0) , m_importedInstanceHandle(false) + , m_retainedLibHandle(false) , m_importedDeviceHandle(false) , m_videoDecodeQueryResultStatusSupport(false) , m_videoEncodeQueryResultStatusSupport(false) @@ -1153,9 +1394,12 @@ VulkanDeviceContext::VulkanDeviceContext() } -void VulkanDeviceContext::DeviceWaitIdle() const +VkResult VulkanDeviceContext::DeviceWaitIdle() const { - vk::VkInterfaceFunctions::DeviceWaitIdle(m_device); + if (m_device == VK_NULL_HANDLE) { + return VK_SUCCESS; // nothing was ever created; nothing can be busy + } + return vk::VkInterfaceFunctions::DeviceWaitIdle(m_device); } VulkanDeviceContext::~VulkanDeviceContext() { @@ -1206,12 +1450,17 @@ VulkanDeviceContext::~VulkanDeviceContext() { m_importedDeviceHandle = false; + // RetainLoaderHandle() suppresses the unload. Anything that resolved + // Vulkan entry points out of this same shared object -- an embedder's + // own function-pointer table, another VulkanDeviceContext -- keeps + // holding pointers into it after this object dies, and unloading it + // underneath them is a crash rather than a leak avoided. #if !defined(VK_USE_PLATFORM_WIN32_KHR) - if (m_libHandle) { + if (m_libHandle && !m_retainedLibHandle) { dlclose(m_libHandle); } #else // defined(VK_USE_PLATFORM_WIN32_KHR) - if (m_libHandle) { + if (m_libHandle && !m_retainedLibHandle) { FreeLibrary(m_libHandle); } #endif // defined(VK_USE_PLATFORM_WIN32_KHR) @@ -1246,9 +1495,9 @@ const char * VulkanDeviceContext::FindRequiredDeviceExtension(const char* name) void VulkanDeviceContext::PrintExtensions(bool deviceExt) const { const std::vector& extensions = deviceExt ? m_deviceExtensions : m_instanceExtensions; - std::cout << "###### List of " << (deviceExt ? "Device" : "Instance") << " Extensions: ######" << std::endl; + VkEncOut() << "###### List of " << (deviceExt ? "Device" : "Instance") << " Extensions: ######" << std::endl; for (const auto& e : extensions) { - std::cout << "\t " << e.extensionName << "(v." << e.specVersion << ")\n"; + VkEncOut() << "\t " << e.extensionName << "(v." << e.specVersion << ")\n"; } } @@ -1257,30 +1506,41 @@ VkResult VulkanDeviceContext::PopulateInstanceExtensions() uint32_t extensionsCount = 0; VkResult result = EnumerateInstanceExtensionProperties( nullptr, &extensionsCount, nullptr ); if ((result != VK_SUCCESS) || (extensionsCount == 0)) { - std::cout << "Could not get the number of instance extensions." << std::endl; + VkEncOut() << "Could not get the number of instance extensions." << std::endl; return result; } m_instanceExtensions.resize( extensionsCount ); result = EnumerateInstanceExtensionProperties( nullptr, &extensionsCount, m_instanceExtensions.data() ); if ((result != VK_SUCCESS) || (extensionsCount == 0)) { - std::cout << "Could not enumerate instance extensions." << std::endl; + VkEncOut() << "Could not enumerate instance extensions." << std::endl; return result; } return result; } +VkResult VulkanDeviceContext::AdoptPhysicalDevice(VkPhysicalDevice physicalDevice) +{ + if (physicalDevice == VK_NULL_HANDLE) { + return VK_ERROR_INITIALIZATION_FAILED; + } + m_physDevice = physicalDevice; + // The queue families stay at their initialised values: this context is for + // physical-device-level queries only and never reaches vkCreateDevice. + return PopulateDeviceExtensions(); +} + VkResult VulkanDeviceContext::PopulateDeviceExtensions() { uint32_t extensions_count = 0; VkResult result = EnumerateDeviceExtensionProperties( m_physDevice, nullptr, &extensions_count, nullptr ); if ((result != VK_SUCCESS) || (extensions_count == 0)) { - std::cout << "Could not get the number of device extensions." << std::endl; + VkEncOut() << "Could not get the number of device extensions." << std::endl; return result; } m_deviceExtensions.resize( extensions_count ); result = EnumerateDeviceExtensionProperties( m_physDevice, nullptr, &extensions_count, m_deviceExtensions.data() ); if ((result != VK_SUCCESS) || (extensions_count == 0)) { - std::cout << "Could not enumerate device extensions." << std::endl; + VkEncOut() << "Could not enumerate device extensions." << std::endl; return result; } return result; @@ -1402,7 +1662,7 @@ VkResult VulkanDeviceContext::InitVulkanDecoderDevice(const char * pAppName, VkResult result = InitVulkanDevice(pAppName, vkInstance, enbaleVerboseDump); if (result != VK_SUCCESS) { - printf("Could not initialize the Vulkan device!\n"); + VkEncPrintfOut("Could not initialize the Vulkan device!\n"); return result; } diff --git a/common/libs/VkCodecUtils/VulkanDeviceContext.h b/common/libs/VkCodecUtils/VulkanDeviceContext.h index 22732090..fcc03816 100644 --- a/common/libs/VkCodecUtils/VulkanDeviceContext.h +++ b/common/libs/VkCodecUtils/VulkanDeviceContext.h @@ -214,7 +214,11 @@ class VulkanDeviceContext : public vk::VkInterfaceFunctions { operator VkDevice() const { return m_device; } - void DeviceWaitIdle() const; + // Returns the driver's verdict rather than discarding it. A wait that + // does not return VK_SUCCESS has not proved the device is idle, and a + // shutdown that treats it as if it had lets a caller reclaim memory the + // GPU is still reading. + VkResult DeviceWaitIdle() const; ~VulkanDeviceContext(); @@ -234,6 +238,24 @@ class VulkanDeviceContext : public vk::VkInterfaceFunctions { void PrintExtensions(bool deviceExt = false) const; + // Keep the Vulkan loader (libvulkan.so.1 / vulkan-1.dll) mapped for + // the process lifetime: this context's destructor will NOT + // dlclose()/FreeLibrary() it. + // + // Closing it is right for a standalone sample that owns its process. + // It is wrong wherever anything ELSE in the process has resolved + // Vulkan entry points out of the same shared object -- Chromium's + // gpu::VulkanFunctionPointers are bound to exactly this library -- + // because the unload invalidates their function pointers while they + // still hold them. The encoder context sets this in both of its + // modes, including for a context it creates and releases inside one + // call; nothing else in the tree does. + // + // One-way: there is deliberately no way to un-retain. A second owner + // asking for retention must not be able to have it revoked by the + // first. + void RetainLoaderHandle() { m_retainedLibHandle = true; } + #if !defined(VK_USE_PLATFORM_WIN32_KHR) typedef void* VulkanLibraryHandleType; #else @@ -276,6 +298,23 @@ class VulkanDeviceContext : public vk::VkInterfaceFunctions { const VkDebugUtilsMessengerCallbackDataEXT* pCallbackData, void* pUserData); + // Adopt a caller-supplied physical device WITHOUT queue selection. + // + // For capability probing on an ADOPT-mode context. The queries + // it enables -- vkGetPhysicalDeviceVideoCapabilitiesKHR and + // vkGetPhysicalDeviceVideoFormatPropertiesKHR -- are physical-device-level + // and need no queue family and no VkDevice, so running the full candidate + // scan is both unnecessary and wrong here: the caller has already chosen + // the device, and asking InitPhysicalDevice to re-derive it forces a queue + // request the probe has no use for. + // + // Populates the device-extension list, so callers do not have to re-query + // it themselves the way the capability path used to. + // + // Does NOT set any queue family. Anything that creates a VkDevice must go + // through InitPhysicalDevice() instead. + VkResult AdoptPhysicalDevice(VkPhysicalDevice physicalDevice); + VkResult InitPhysicalDevice(int32_t deviceId, const vk::DeviceUuidUtils& deviceUuid, const VkQueueFlags requestQueueTypes = (VK_QUEUE_GRAPHICS_BIT | /* VK_QUEUE_COMPUTE_BIT | */ @@ -301,6 +340,27 @@ class VulkanDeviceContext : public vk::VkInterfaceFunctions { bool createPresentQueue = false, bool createComputeQueue = false, VkDevice vkDevice = VK_NULL_HANDLE); + + // Imported-device support: override the queue families the + // internal InitPhysicalDevice() probe selected with the families the + // CALLER created queues for on its imported VkDevice. Must be called + // after InitPhysicalDevice() and before CreateVulkanDevice(). Passing + // UINT32_MAX for an index keeps the probed family (no override). + // + // Rationale: for an imported VkDevice the library never creates queues; + // it only vkGetDeviceQueue()s them. The probe picks families purely from + // the physical device's properties, which may legally differ from the + // families the caller's vkCreateDevice actually requested queues on -- + // and vkGetDeviceQueue on a family the device was not created with is + // undefined behavior. The override is validated against the physical + // device: the family index must exist, the encode family must expose + // VK_QUEUE_VIDEO_ENCODE_BIT_KHR and support videoEncodeQueueOperations, + // and the compute family must expose VK_QUEUE_COMPUTE_BIT. + VkResult OverrideImportedQueueFamilies( + uint32_t videoEncodeQueueFamilyIndex, + VkVideoCodecOperationFlagsKHR videoEncodeQueueOperations, + uint32_t computeQueueFamilyIndex); + VkResult InitDebugReport(bool validate = false, bool validateVerbose = false); // Validation-error accounting. @@ -344,6 +404,8 @@ class VulkanDeviceContext : public vk::VkInterfaceFunctions { VkQueueFlags m_videoDecodeQueueFlags; VkQueueFlags m_videoEncodeQueueFlags; uint32_t m_importedInstanceHandle : 1; + // See RetainLoaderHandle(): suppresses the destructor's unload. + uint32_t m_retainedLibHandle : 1; uint32_t m_importedDeviceHandle : 1; uint32_t m_videoDecodeQueryResultStatusSupport : 1; uint32_t m_videoEncodeQueryResultStatusSupport : 1; diff --git a/common/libs/VkCodecUtils/VulkanFilterYuvCompute.cpp b/common/libs/VkCodecUtils/VulkanFilterYuvCompute.cpp index 59fa5ec6..ee0e91df 100644 --- a/common/libs/VkCodecUtils/VulkanFilterYuvCompute.cpp +++ b/common/libs/VkCodecUtils/VulkanFilterYuvCompute.cpp @@ -15,6 +15,7 @@ */ #include "VulkanFilterYuvCompute.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "nvidia_utils/vulkan/ycbcrvkinfo.h" #include @@ -193,9 +194,10 @@ VkResult VulkanFilterYuvCompute::Create(const VulkanDeviceContext* vkDevCtx, // // Channel indices are into the vec4 returned by imageLoad: 0=r 1=g 2=b 3=a. // The packed 4:4:4 descriptor lives in nvidia_utils/vulkan/ycbcrvkinfo.h so that the -// filter, the encoder's staging copy and the format-selection code all read one table. -// Keep it that way: a channel order written out by hand at each site is how the sites -// drift apart, and a transposed Cb/Cr is a plausible picture rather than an error. +// filter, the encoder's staging copy and the format-selection code all read ONE table. +// A hand-written per-format switch here is what transposes Cb and Cr on one format +// while leaving its siblings correct, which is why the channel order is looked up +// rather than spelled out. typedef VkPackedYcbcrFormatDesc VkPackedYcbcrFormatInfo; static const VkPackedYcbcrFormatInfo* PackedYcbcrFormatInfo(VkFormat format) @@ -259,6 +261,12 @@ VkResult VulkanFilterYuvCompute::Init(const VkSamplerYcbcrConversionCreateInfo* true, // createSemaphores true // createFences ); + // Configure()'s result was dropped here: the next statement overwrote + // |result| unconditionally, so a failed pool/queue/semaphore setup was + // reported as whatever the sampler creation happened to return. + if (result != VK_SUCCESS) { + return result; + } if (pYcbcrConversionCreateInfo) { result = m_samplerYcbcrConversion.CreateVulkanSampler(m_vkDevCtx, @@ -303,8 +311,6 @@ VkResult VulkanFilterYuvCompute::Init(const VkSamplerYcbcrConversionCreateInfo* "main", m_workgroupSizeX, m_workgroupSizeY, &m_descriptorSetLayout); - - return VK_ERROR_LAYER_NOT_PRESENT; } VkResult VulkanFilterYuvCompute::InitDescriptorSetLayout(uint32_t maxNumFrames) @@ -393,13 +399,37 @@ VkResult VulkanFilterYuvCompute::InitDescriptorSetLayout(uint32_t maxNumFrames) pushConstantRange.offset = 0; pushConstantRange.size = sizeof(PushConstants); // Size matches the actual PushConstants struct + // autoSelect (last argument) rather than pinning push descriptors. + // + // The flags argument below is only a preference now: with autoSelect set, + // CreateDescriptorSet picks push descriptors when VK_KHR_push_descriptor is + // enabled, VK_EXT_descriptor_buffer when it is not and that one is, and + // plain descriptor sets when neither is. Pinning + // PUSH_DESCRIPTOR_BIT unconditionally discarded a descriptor-buffer arm + // that is fully implemented on both sides -- the layout/buffer half in + // VulkanDescriptorSetLayout and the bind half in this file's + // RecordCommandBuffer switch -- and that produces byte-identical output to + // the push-descriptor arm on hardware that has both. + // + // WHAT THIS DOES NOT FIX, recorded so the next reader does not re-derive + // it: this does not make the filter work on a device that enables neither + // extension. The third arm is a stub -- nothing in this file ever calls + // VulkanDescriptorSetLayout::WriteDescriptorSet, and the pool it allocates + // is typed COMBINED_IMAGE_SAMPLER while these bindings are storage + // images/buffers -- so it fails at Create with VK_ERROR_OUT_OF_POOL_MEMORY. + // That is a loud failure the encoder already treats as fatal, not silent + // corruption, but it is still a failure. In particular a caller-supplied + // VkDevice lands there twice over: VulkanDeviceContext's imported path + // never populates m_reqDeviceExtensions, so FindRequiredDeviceExtension + // answers "no" for every extension regardless of what the embedder actually + // enabled. return m_descriptorSetLayout.CreateDescriptorSet(m_vkDevCtx, setLayoutBindings, VK_DESCRIPTOR_SET_LAYOUT_CREATE_PUSH_DESCRIPTOR_BIT_KHR, 1, &pushConstantRange, &m_samplerYcbcrConversion, maxNumFrames, - false); + true); } static YcbcrBtStandard GetYcbcrPrimariesConstantsId(VkSamplerYcbcrModelConversion modelConversion) @@ -490,6 +520,31 @@ static void GenHeaderAndPushConst(std::stringstream& shaderStr) * @param set Descriptor set number * @param imageArray Whether the image should be declared as image2DArray instead of image2D */ +// The GLSL image-format qualifier to declare a binding of |format| with. +// +// vkFormatLookUp() answers nullptr for a format its table does not hold, so +// no caller below may dereference that answer for ->name. The table is not +// exhaustive over VkFormat and cannot be, so a routing change that brings a +// new single-plane format into shader generation must fail diagnosably here +// rather than on a null pointer -- the plane arms are safe only because every +// per-plane format the multi-planar table names is held, which is a property +// of two tables agreeing and not a guarantee. +// +// A format with no qualifier yields one that cannot compile, deliberately. +// CreatePipeline() already reports VK_ERROR_INITIALIZATION_FAILED when +// BuildGlslShader() rejects the generated GLSL, so an unknown format now +// fails at Init(), through the path a shader that will not compile already +// takes, and with a status the caller can act on. +static const char* GlslImageFormatQualifier(VkFormat format) +{ + const VkFormatDesc* formatDesc = vkFormatLookUp(format); + if ((formatDesc != nullptr) && (formatDesc->name != nullptr)) { + return formatDesc->name; + } + assert(!"no GLSL image format qualifier for this VkFormat"); + return "NO_GLSL_IMAGE_FORMAT_QUALIFIER_FOR_THIS_VKFORMAT"; +} + static void GenImageIoBindingLayout(std::stringstream& shaderStr, const char *imageName, const char *imageSubName, @@ -618,6 +673,9 @@ static void GenHandleSourcePositionWithReplicate(std::stringstream& shaderStr, b * @param chromaHorzRatio Horizontal chroma subsampling (1, 2, or 4) * @param chromaVertRatio Vertical chroma subsampling (1 or 2) * @param enableReplication Whether to enable edge pixel replication + * @param inputChromaHorzRatio Horizontal chroma subsampling of the INPUT, or 0 + * for "same as the block" (same-subsampling copy) + * @param inputChromaVertRatio Vertical chroma subsampling of the INPUT, or 0 */ static void GenBlockCoordinates(std::stringstream& shaderStr, uint32_t chromaHorzRatio, @@ -742,13 +800,19 @@ static void GenReadYCbCrBlock(std::stringstream& shaderStr, << " float cr = 0.0;\n"; for (uint32_t y = 0; y < chromaVertRatio; y++) { for (uint32_t x = 0; x < chromaHorzRatio; x++) { - shaderStr << " vec4 pk" << x << y << " = imageLoad(inputImageRGB, ivec3("; + // ivec2, not ivec3: inputImageRGB is the single COLOR_BIT + // binding, which ShaderGenerateImagePlaneDescriptors declares + // image2D because it is bound from the combined + // VK_IMAGE_VIEW_TYPE_2D view. pushConstants.srcLayer has nothing + // to select -- a 2D view carries the one layer, which is the + // layer the ivec3 form was reading anyway. + shaderStr << " vec4 pk" << x << y << " = imageLoad(inputImageRGB, "; if (enableReplication) { shaderStr << "min(lumaPos + ivec2(" << x << ", " << y << "), inputLumaMax)"; } else { shaderStr << "lumaPos + ivec2(" << x << ", " << y << ")"; } - shaderStr << ", pushConstants.srcLayer));\n" + shaderStr << ");\n" << " float y" << x << y << " = pk" << x << y << "." << PackedChan(packedInput->yChannel) << ";\n" << " cb += pk" << x << y << "." << PackedChan(packedInput->cbChannel) << ";\n" @@ -802,9 +866,12 @@ static void GenReadYCbCrBlock(std::stringstream& shaderStr, shaderStr << " cr += cbcr" << x << y << ".g;\n"; } } - float scale = 1.0f / (chromaHorzRatio * chromaVertRatio); - shaderStr << " cb *= " << scale << ";\n"; - shaderStr << " cr *= " << scale << ";\n"; + // A 1x1 block is one sample, not an average. + if ((chromaHorzRatio * chromaVertRatio) > 1) { + float scale = 1.0f / (chromaHorzRatio * chromaVertRatio); + shaderStr << " cb *= " << scale << ";\n"; + shaderStr << " cr *= " << scale << ";\n"; + } } else { // 3-plane format: average all Cb and Cr samples shaderStr << " float cb = 0.0;\n" @@ -829,9 +896,13 @@ static void GenReadYCbCrBlock(std::stringstream& shaderStr, shaderStr << " cr += cr" << x << y << ";\n"; } } - float scale = 1.0f / (chromaHorzRatio * chromaVertRatio); - shaderStr << " cb *= " << scale << ";\n"; - shaderStr << " cr *= " << scale << ";\n"; + // A 1x1 block is one sample, not an average -- emitting + // "cb *= 1" there would be noise in the generated shader. + if ((chromaHorzRatio * chromaVertRatio) > 1) { + float scale = 1.0f / (chromaHorzRatio * chromaVertRatio); + shaderStr << " cb *= " << scale << ";\n"; + shaderStr << " cr *= " << scale << ";\n"; + } } } else { // Subsampled input (4:2:0, 4:2:2). The block covers this many input chroma @@ -932,6 +1003,71 @@ static void GenConvertYCbCrBlock(std::stringstream& shaderStr, shaderStr << " \n"; } +/** + * @brief Generates code to apply MSB-to-LSB bit shift for block INPUT + * + * The mirror of GenApplyBlockOutputShift, and the reason that one is safe. + * + * The block path reads its samples with imageLoad through per-plane + * VK_FORMAT_R16_UNORM views. An X6/X4-packed source stores a 10/12-bit sample + * MSB-aligned in a 16-bit word, so that load already returns a correctly + * normalized value -- (pixel << 6) / 65535 ~= pixel / 1023 at 10-bit. The + * output stage then multiplies by the same factor to re-align for the write. + * With only the output half present the sample is scaled by 64 (or 16) with + * nothing undoing it, and every channel saturates. + + * + * Applied immediately after the read and before any conversion, so the + * conversion stage sees LSB-aligned values, which is what its normalization + * helpers assume. + * + * @param shaderStr Output stringstream + * @param chromaHorzRatio Horizontal block size + * @param chromaVertRatio Vertical block size + * @param inputBitDepth Input bit depth (8, 10, 12, or 16) + * @param enableShift Whether MSB-to-LSB shift is enabled + * @param hasChroma Whether the input carries chroma + */ +static void GenApplyBlockInputShift(std::stringstream& shaderStr, + uint32_t chromaHorzRatio, + uint32_t chromaVertRatio, + uint32_t inputBitDepth, + bool enableShift, + bool hasChroma = true) +{ + if (!enableShift) { + return; + } + + // Exact reciprocals of the output stage's factors: both are powers of two, + // so the round trip is an identity in float rather than an approximation. + float shiftRecip = 0.0f; + if (inputBitDepth == 10) { + shiftRecip = 1.0f / 64.0f; // 1 >> (16 - 10) + } else if (inputBitDepth == 12) { + shiftRecip = 1.0f / 16.0f; // 1 >> (16 - 12) + } else { + return; // No shift for 8 or 16-bit + } + + shaderStr << " // Apply MSB-to-LSB shift for " << inputBitDepth << "-bit input\n"; + + for (uint32_t y = 0; y < chromaVertRatio; y++) { + for (uint32_t x = 0; x < chromaHorzRatio; x++) { + shaderStr << " y" << x << y << " *= " << shiftRecip << ";\n"; + } + } + if (hasChroma) { + // cb/cr are already the averaged samples where the input needed + // averaging; scaling is linear, so applying it here is equivalent to + // scaling each contributor before the sum. + shaderStr << " cb *= " << shiftRecip << ";\n"; + shaderStr << " cr *= " << shiftRecip << ";\n"; + } + + shaderStr << " \n"; +} + /** * @brief Generates code to apply LSB-to-MSB bit shift for block output * @@ -991,14 +1127,25 @@ static void GenApplyBlockOutputShift(std::stringstream& shaderStr, * @param isOutputTwoPlane Whether output is 2-plane or 3-plane * @param hasOutputChroma Whether output has chroma planes */ +// The index an OUTPUT imageStore takes: ivec2 for an image2D binding, and +// ivec3(..., dstLayer) for an image2DArray one. Emitting the wrong one is +// VUID-vkCmdDispatch-viewType-07752 in one direction, or a shader that does not +// compile in the other, so every output store derives its form from the same +// predicate -- VulkanFilterYuvCompute::OutputDescriptorIsArray(). +static std::string OutStoreIndex(const std::string& xy, bool isArray) +{ + return isArray ? ("ivec3(" + xy + ", pushConstants.dstLayer)") : xy; +} + static void GenWriteYCbCrBlock(std::stringstream& shaderStr, - uint32_t lumaBlockHorzRatio, - uint32_t lumaBlockVertRatio, + uint32_t ySubsampleHorzRatio, + uint32_t ySubsampleVertRatio, bool isOutputTwoPlane, bool hasOutputChroma, uint32_t outputChromaHorzSubsampling = 2, uint32_t outputChromaVertSubsampling = 2, - const VkPackedYcbcrFormatInfo* packedOutput = nullptr) + const VkPackedYcbcrFormatInfo* packedOutput = nullptr, + bool outputIsArray = false) { // Packed (single-plane) output: one imageStore per luma pixel into the single // COLOR_BIT image declared as RGB, with Y/Cb/Cr placed in the channels the @@ -1006,10 +1153,10 @@ static void GenWriteYCbCrBlock(std::stringstream& shaderStr, // chroma sample per pixel. Mirrors the packed read arm in GenReadYCbCrBlock; without // it the same 'undeclared identifier' fires on outputImageY. if (packedOutput != nullptr) { - shaderStr << " // Write " << lumaBlockHorzRatio << "x" << lumaBlockVertRatio + shaderStr << " // Write " << ySubsampleHorzRatio << "x" << ySubsampleVertRatio << " packed " << packedOutput->debugName << " pixels\n"; - for (uint32_t y = 0; y < lumaBlockVertRatio; y++) { - for (uint32_t x = 0; x < lumaBlockHorzRatio; x++) { + for (uint32_t y = 0; y < ySubsampleVertRatio; y++) { + for (uint32_t x = 0; x < ySubsampleHorzRatio; x++) { // Build the vec4 by channel index so the order comes from the table, not // from a per-format hand-written store. const char* comp[4] = { "0.0", "0.0", "0.0", "1.0" }; @@ -1017,8 +1164,14 @@ static void GenWriteYCbCrBlock(std::stringstream& shaderStr, comp[packedOutput->yChannel & 3] = yExpr.c_str(); comp[packedOutput->cbChannel & 3] = hasOutputChroma ? "cbcrOut.x" : "0.5"; comp[packedOutput->crChannel & 3] = hasOutputChroma ? "cbcrOut.y" : "0.5"; - shaderStr << " imageStore(outputImageRGB, ivec3(lumaPos + ivec2(" - << x << ", " << y << "), pushConstants.dstLayer), vec4(" + // ivec2, not ivec3 -- the mirror of the packed read arm's + // index form, and for the same reason: outputImageRGB is the + // combined VK_IMAGE_VIEW_TYPE_2D view's binding, declared + // image2D. + shaderStr << " imageStore(outputImageRGB, " + << OutStoreIndex("lumaPos + ivec2(" + std::to_string(x) + ", " + + std::to_string(y) + ")", outputIsArray) + << ", vec4(" << comp[0] << ", " << comp[1] << ", " << comp[2] << ", " << comp[3] << "));\n"; } } @@ -1026,11 +1179,11 @@ static void GenWriteYCbCrBlock(std::stringstream& shaderStr, return; } - shaderStr << " // Write " << lumaBlockHorzRatio << "x" << lumaBlockVertRatio << " Y pixels\n"; + shaderStr << " // Write " << ySubsampleHorzRatio << "x" << ySubsampleVertRatio << " Y pixels\n"; // Write all Y pixels - for (uint32_t y = 0; y < lumaBlockVertRatio; y++) { - for (uint32_t x = 0; x < lumaBlockHorzRatio; x++) { + for (uint32_t y = 0; y < ySubsampleVertRatio; y++) { + for (uint32_t x = 0; x < ySubsampleHorzRatio; x++) { shaderStr << " imageStore(outputImageY, ivec3(lumaPos + ivec2(" << x << ", " << y << "), pushConstants.dstLayer), " "vec4(yOut" << x << y << ", 0, 0, 1));\n"; @@ -1049,9 +1202,9 @@ static void GenWriteYCbCrBlock(std::stringstream& shaderStr, if (isOutput444) { // Write chroma at each Y pixel position - shaderStr << " // Write " << lumaBlockHorzRatio << "x" << lumaBlockVertRatio << " chroma pixels (4:4:4 format)\n"; - for (uint32_t y = 0; y < lumaBlockVertRatio; y++) { - for (uint32_t x = 0; x < lumaBlockHorzRatio; x++) { + shaderStr << " // Write " << ySubsampleHorzRatio << "x" << ySubsampleVertRatio << " chroma pixels (4:4:4 format)\n"; + for (uint32_t y = 0; y < ySubsampleVertRatio; y++) { + for (uint32_t x = 0; x < ySubsampleHorzRatio; x++) { if (isOutputTwoPlane) { shaderStr << " imageStore(outputImageCbCr, ivec3(lumaPos + ivec2(" << x << ", " << y << "), pushConstants.dstLayer), " @@ -1756,7 +1909,7 @@ uint32_t VulkanFilterYuvCompute::ShaderGenerateImagePlaneDescriptors(std::string imageAspects = VK_IMAGE_ASPECT_PLANE_0_BIT; GenImageIoBindingLayout(shaderStr, imageName, "Y", - vkFormatLookUp(imageFormat)->name, + GlslImageFormatQualifier(imageFormat), isInput, ++startBinding, set, @@ -1769,7 +1922,7 @@ uint32_t VulkanFilterYuvCompute::ShaderGenerateImagePlaneDescriptors(std::string imageAspects = VK_IMAGE_ASPECT_PLANE_1_BIT | VK_IMAGE_ASPECT_PLANE_2_BIT; GenImageIoBindingLayout(shaderStr, imageName, "CbCr", - vkFormatLookUp(imageFormat)->name, + GlslImageFormatQualifier(imageFormat), isInput, ++startBinding, set, @@ -1791,7 +1944,7 @@ uint32_t VulkanFilterYuvCompute::ShaderGenerateImagePlaneDescriptors(std::string if (inputMpInfo) { GenImageIoBindingLayout(shaderStr, imageName, "Y", - vkFormatLookUp(inputMpInfo->vkPlaneFormat[0])->name, + GlslImageFormatQualifier(inputMpInfo->vkPlaneFormat[0]), isInput, ++startBinding, set, @@ -1802,7 +1955,7 @@ uint32_t VulkanFilterYuvCompute::ShaderGenerateImagePlaneDescriptors(std::string imageAspects = VK_IMAGE_ASPECT_PLANE_0_BIT | VK_IMAGE_ASPECT_PLANE_1_BIT; GenImageIoBindingLayout(shaderStr, imageName, "CbCr", - vkFormatLookUp(inputMpInfo->vkPlaneFormat[1])->name, + GlslImageFormatQualifier(inputMpInfo->vkPlaneFormat[1]), isInput, ++startBinding, set, @@ -1814,14 +1967,14 @@ uint32_t VulkanFilterYuvCompute::ShaderGenerateImagePlaneDescriptors(std::string VK_IMAGE_ASPECT_PLANE_2_BIT; GenImageIoBindingLayout(shaderStr, imageName, "Cb", - vkFormatLookUp(inputMpInfo->vkPlaneFormat[1])->name, + GlslImageFormatQualifier(inputMpInfo->vkPlaneFormat[1]), isInput, ++startBinding, set, imageArray); GenImageIoBindingLayout(shaderStr, imageName, "Cr", - vkFormatLookUp(inputMpInfo->vkPlaneFormat[2])->name, + GlslImageFormatQualifier(inputMpInfo->vkPlaneFormat[2]), isInput, ++startBinding, set, @@ -1831,12 +1984,40 @@ uint32_t VulkanFilterYuvCompute::ShaderGenerateImagePlaneDescriptors(std::string imageAspects = VK_IMAGE_ASPECT_COLOR_BIT; + // NOT |imageArray|, and this is the one arm where a per-call answer + // cannot be right. A COLOR_BIT binding is the SINGLE-PLANE arm, and + // UpdateImageDescriptorSets binds every COLOR_BIT descriptor from + // GetImageView() -- the COMBINED view, which follows the image's own + // layerCount. Declaring image2DArray over a VK_IMAGE_VIEW_TYPE_2D view + // is VUID-vkCmdDispatch-viewType-07752 and the access is UNDEFINED; + // this driver returns layer 0, which is why the picture was right and + // the fault survived. The plane arms above keep the caller's request, + // and must: they bind GetPlaneImageView(), which Create() declares + // VK_IMAGE_VIEW_TYPE_2D_ARRAY unconditionally. + // + // Deciding it HERE rather than at each call site is the point. This + // function chooses the arm, so only it knows which one was taken; a + // caller can only re-derive that from the format, and two places + // answering the same question independently is what went wrong -- + // InitYCBCRCOPY asked for an array on both of its sides, which is right + // for a planar side and wrong for a packed one. + // + // OutputDescriptorIsArray() IS THAT ONE PLACE, and the OUTPUT side must + // use it rather than a constant: every output imageStore already + // derives its index form from it through OutStoreIndex(). Hardcoding + // false here meant that a caller passing FLAG_OUTPUT_IMAGE_ARRAY got + // ivec3 stores against an image2D declaration -- GLSL that does not + // compile. With the flag unset the predicate is false and this arm + // declares image2D exactly as before, which is every caller in tree. + // + // The INPUT side keeps ivec2: there is no input-side flag, so its + // combined view is single-layer by construction. GenImageIoBindingLayout(shaderStr, imageName, "RGB", - vkFormatLookUp(imageFormat)->name, + GlslImageFormatQualifier(imageFormat), isInput, startBinding++, set, - imageArray); + isInput ? false : OutputDescriptorIsArray()); } return startBinding; @@ -1980,6 +2161,49 @@ uint32_t VulkanFilterYuvCompute::ShaderGeneratePlaneDescriptors(std::stringstrea * @param isLimitedRange Whether values are limited range (true) or full range (false) * @param hasChroma Whether to include chroma normalization functions */ +/** + * @brief The ITU-R code levels a normalized [0,1] Y'CbCr value maps onto + * + * One definition shared by everything that has to agree on what "narrow range" means. + * The limited-range levels are the 8-bit ITU-R definition (Y[16,235], Cb/Cr[16,240]) + * scaled by 2^(bitDepth-8), which reproduces the published constants exactly: + * 10-bit Y[64,940] C[64,960], 12-bit Y[256,3760] C[256,3840], 16-bit Y[4096,60160]. + */ +struct YCbCrRangeLevels { + double maxValue; // full-scale code value for the bit depth + double yBlack; // luma code for black + double yWhite; // luma code for white + double cZero; // chroma code at the bottom of the chroma excursion + double cScale; // chroma excursion, i.e. (cMax - cZero) +}; + +static YCbCrRangeLevels GetYCbCrRangeLevels(uint32_t bitDepth, bool isLimitedRange) +{ + YCbCrRangeLevels levels = {}; + levels.maxValue = (double)((1ULL << bitDepth) - 1ULL); + + if (!isLimitedRange) { + levels.yBlack = 0.0; + levels.yWhite = levels.maxValue; + levels.cZero = 0.0; + levels.cScale = levels.maxValue; + return levels; + } + + // Only 8/10/12/16 are real Y'CbCr depths; anything else falls back to the 8-bit + // levels rather than shifting by a negative amount. + const bool depthIsSupported = (bitDepth == 8) || (bitDepth == 10) || + (bitDepth == 12) || (bitDepth == 16); + assert(depthIsSupported); + const double scale = depthIsSupported ? (double)(1ULL << (bitDepth - 8)) : 1.0; + + levels.yBlack = 16.0 * scale; + levels.yWhite = 235.0 * scale; + levels.cZero = 16.0 * scale; + levels.cScale = 224.0 * scale; + return levels; +} + static void GenYCbCrNormalizationFuncs(std::stringstream& shaderStr, uint32_t bitDepth = 8, bool isLimitedRange = true, @@ -1988,53 +2212,15 @@ static void GenYCbCrNormalizationFuncs(std::stringstream& shaderStr, // STEP 1: Calculate normalization parameters based on bit depth and range // =========================================================================== - // Use double precision for calculations to maintain precision - double maxValue = (1ULL << bitDepth) - 1.0; // Max value for the given bit depth - - // Limited range values for different bit depths - double yBlack, yWhite, cZero, cScale; - - if (isLimitedRange) { - // Step 1.1: Calculate limited range (aka TV/Video range) values - // Use standard-compliant values for different bit depths - switch (bitDepth) { - case 10: - // 10-bit limited range: Y[64,940], C[64,960] - yBlack = 64.0; - yWhite = 940.0; - cZero = 64.0; - cScale = 896.0; // 960 - 64 - break; - case 12: - // 12-bit limited range: Y[256,3760], C[256,3840] - yBlack = 256.0; - yWhite = 3760.0; - cZero = 256.0; - cScale = 3584.0; // 3840 - 256 - break; - case 16: - // 16-bit limited range: scale 8-bit values by 2^8 - yBlack = 16.0 * 256.0; - yWhite = 235.0 * 256.0; - cZero = 16.0 * 256.0; - cScale = 224.0 * 256.0; - break; - case 8: - default: - // 8-bit limited range: Y[16,235], C[16,240] - yBlack = 16.0; - yWhite = 235.0; - cZero = 16.0; - cScale = 224.0; - break; - } - } else { - // Step 1.2: Calculate full range values (same for all bit depths, just scaled) - yBlack = 0.0; - yWhite = maxValue; - cZero = 0.0; - cScale = maxValue; - } + // Use double precision for calculations to maintain precision. + // Values come from the shared table so the generators and the RGBA->YCbCr + // range mapping cannot disagree. + const YCbCrRangeLevels rangeC = GetYCbCrRangeLevels(bitDepth, isLimitedRange); + const double maxValue = rangeC.maxValue; + const double yBlack = rangeC.yBlack; + const double yWhite = rangeC.yWhite; + const double cZero = rangeC.cZero; + const double cScale = rangeC.cScale; // Step 1.3: Calculate normalization factors with double precision double yRange = yWhite - yBlack; @@ -2188,48 +2374,6 @@ static void GenYCbCrNormalizationFuncs(std::stringstream& shaderStr, } } -/** - * @brief The ITU-R code levels a normalized [0,1] Y'CbCr value maps onto - * - * One definition shared by everything that has to agree on what "narrow range" means. - * The limited-range levels are the 8-bit ITU-R definition (Y[16,235], Cb/Cr[16,240]) - * scaled by 2^(bitDepth-8), which reproduces the published constants exactly: - * 10-bit Y[64,940] C[64,960], 12-bit Y[256,3760] C[256,3840], 16-bit Y[4096,60160]. - */ -struct YCbCrRangeLevels { - double maxValue; // full-scale code value for the bit depth - double yBlack; // luma code for black - double yWhite; // luma code for white - double cZero; // chroma code at the bottom of the chroma excursion - double cScale; // chroma excursion, i.e. (cMax - cZero) -}; - -static YCbCrRangeLevels GetYCbCrRangeLevels(uint32_t bitDepth, bool isLimitedRange) -{ - YCbCrRangeLevels levels = {}; - levels.maxValue = (double)((1ULL << bitDepth) - 1ULL); - - if (!isLimitedRange) { - levels.yBlack = 0.0; - levels.yWhite = levels.maxValue; - levels.cZero = 0.0; - levels.cScale = levels.maxValue; - return levels; - } - - // Only 8/10/12/16 are real Y'CbCr depths; anything else falls back to the 8-bit - // levels rather than shifting by a negative amount. - const bool depthIsSupported = (bitDepth == 8) || (bitDepth == 10) || - (bitDepth == 12) || (bitDepth == 16); - assert(depthIsSupported); - const double scale = depthIsSupported ? (double)(1ULL << (bitDepth - 8)) : 1.0; - - levels.yBlack = 16.0 * scale; - levels.yWhite = 235.0 * scale; - levels.cZero = 16.0 * scale; - levels.cScale = 224.0 * scale; - return levels; -} /** * @brief Generates convertYCbCrFormat() for the image->image path, in normalized space @@ -2557,10 +2701,26 @@ VkFormat VulkanFilterYuvCompute::GetOutputFormat(FilterType filterType, VkFormat } // DEPRECATED -- see the YCBCR2RGBA note in VulkanFilterYuvCompute.h. No production path -// uses this generator and its test family is disabled. It assumes a 2-plane input: a -// 3-plane format emits GLSL that does not compile. Deriving the chroma ratios from the -// input format, below, is necessary for a correct conversion but is not on its own enough -// to make this path usable. +// uses this generator, its test family is disabled, and it still assumes a 2-plane input +// (a 3-plane format emits GLSL that does not compile). The chroma-ratio fix below is +// correct as far as it goes but does not make the path usable on its own. +bool VulkanFilterYuvCompute::OutputDescriptorIsArray() const +{ + // A MULTI-PLANAR output binds per-plane views. VkImageResourceView::Create + // types those VK_IMAGE_VIEW_TYPE_2D_ARRAY unconditionally, in both of its + // overloads, so the declaration is an array image whatever the layer count. + // + // A SINGLE-PLANE output -- packed YCbCr (AYUV, Y410) or RGBA -- has no + // per-plane view to bind, so it binds the COMBINED view, and that one + // follows the image's layerCount. The caller declares it with + // FLAG_OUTPUT_IMAGE_ARRAY, because the shader is generated before any image + // is bound and the format alone cannot answer it. + const VkMpFormatInfo* mpOutputInfo = YcbcrVkFormatInfo(m_outputFormat); + const bool bindsPerPlaneViews = + (mpOutputInfo != nullptr) && (m_outputPackedYcbcr == nullptr); + return bindsPerPlaneViews ? true : (m_outputImageArray != 0); +} + size_t VulkanFilterYuvCompute::InitYCBCR2RGBA(std::string& computeShader) { // The compute filter uses two or three input images as separate planes @@ -2584,6 +2744,15 @@ size_t VulkanFilterYuvCompute::InitYCBCR2RGBA(std::string& computeShader) true, // isInput 0, // startBinding 0, // set + // imageArray: TRUE, and CORRECT -- do not + // "fix" this one to match the output below. + // The input is YCbCr by construction here + // (GetOutputFormat returns UNDEFINED + // otherwise), so it binds PLANE views, and + // VkImageResourceView::Create declares plane + // views VK_IMAGE_VIEW_TYPE_2D_ARRAY + // unconditionally. The loads below index them + // with ivec3 to match. true, VK_DESCRIPTOR_TYPE_STORAGE_BUFFER); @@ -2593,7 +2762,24 @@ size_t VulkanFilterYuvCompute::InitYCBCR2RGBA(std::string& computeShader) false, // isInput 4, // startBinding 0, // set - true, // imageArray + // GetOutputFormat forces this filter's output + // to R8G8B8A8_UNORM or R16G16B16A16_UNORM, so + // YcbcrVkFormatInfo is null here and + // ShaderGenerateImagePlaneDescriptors always + // takes the PACKED arm: one COLOR binding, + // bound from GetImageView(). That view is + // VK_IMAGE_VIEW_TYPE_2D for a single-layer + // image and VK_IMAGE_VIEW_TYPE_2D_ARRAY for a + // layered one, so it follows layerCount via + // the caller's FLAG_OUTPUT_IMAGE_ARRAY rather + // than being hardcoded either way -- this + // filter is selectable by VkEncDeriveFilterType + // and is LIVE. + // + // Passing it is belt and braces: the packed + // arm reads OutputDescriptorIsArray() itself, + // because it has to agree with the stores. + OutputDescriptorIsArray(), VK_DESCRIPTOR_TYPE_STORAGE_BUFFER); // Get format information to determine bit depth @@ -2611,7 +2797,24 @@ size_t VulkanFilterYuvCompute::InitYCBCR2RGBA(std::string& computeShader) GenYCbCrNormalizationFuncs(shaderStr, bitDepth, isLimitedRange, true); // 4. Generate YCbCr to RGB conversion function - const unsigned int bpp = (8 + mpInfo->planesLayout.bpp * 2); + // + // mpInfo is NULL for any format the multi-planar table does not hold, and + // this arm is reachable with one: GetOutputFormat() answers + // VK_FORMAT_UNDEFINED for exactly that case, but nothing reads that answer + // to refuse construction -- it is stored into m_outputFormat and never + // compared -- so a packed 4:4:4 input such as Y410 arrives here with a + // null mpInfo and used to dereference it. The line two above already + // guards the same pointer for bitDepth; this one did not. + // + // Falling back to 8 matches that guard rather than inventing a second + // answer. It does not make this arm serve a packed input: generation + // continues and emits the planar identifiers inputImageY/inputImageCbCr, + // which a packed side never declares, so the shader fails to compile and + // CreatePipeline reports VK_ERROR_INITIALIZATION_FAILED. That is the + // arm's real limitation, and it is now what a caller meets instead of a + // segmentation fault inside shader generation. + const unsigned int bpp = + (mpInfo != nullptr) ? (8 + mpInfo->planesLayout.bpp * 2) : 8; const YcbcrBtStandard btStandard = GetYcbcrPrimariesConstantsId(samplerYcbcrConversionCreateInfo.ycbcrModel); const YcbcrPrimariesConstants primariesConstants = GetYcbcrPrimariesConstants(btStandard); @@ -2649,13 +2852,14 @@ size_t VulkanFilterYuvCompute::InitYCBCR2RGBA(std::string& computeShader) "{\n"; GenHandleImagePosition(shaderStr); GenHandleSourcePositionWithReplicate(shaderStr, m_enableRowAndColumnReplication); - // Chroma is fetched at the INPUT format's real subsampling ratios. A fixed `srcPos/2` - // halves BOTH axes: right for 4:2:0, but wrong for 4:2:2 (chroma is full height, only - // the x axis is halved) and wrong for 4:4:4 (no subsampling at all). Reading chroma - // from the wrong rows is a large, chroma-dominated error that leaves luma nearly - // intact, which reads as a colour problem rather than an indexing one. The ratios come - // from the same YcbcrVkFormatInfo the rest of the plane maths uses, so they cannot - // drift from the descriptors that were generated above. + // Chroma is fetched at the INPUT format's real subsampling ratios, NOT at a + // hardcoded `srcPos/2`. Halving both axes is right only for 4:2:0: 4:2:2 chroma + // is full height and only the x axis is halved, and 4:4:4 is not subsampled at + // all, so a fixed /2 reads chroma from the wrong rows -- a large, + // chroma-dominated error that leaves luma nearly intact and so reads as a vague + // "shader generation bug" rather than as a subsampling one. The ratios come from + // the same YcbcrVkFormatInfo the rest of the plane maths uses, so they cannot + // drift from the descriptors generated above. const uint32_t chromaHorzRatio = (mpInfo != nullptr) ? (1u << mpInfo->planesLayout.secondaryPlaneSubsampledX) : 1u; const uint32_t chromaVertRatio = @@ -2673,12 +2877,15 @@ size_t VulkanFilterYuvCompute::InitYCBCR2RGBA(std::string& computeShader) " vec3 ycbcr = shiftCbCr(normalizeYCbCr(vec3(Y, CbCr)));\n" " vec4 rgba = vec4(convertYCbCrToRgb(ycbcr),1.0);\n" " // Store it back.\n" - " imageStore(outputImageRGB, ivec3(pos, pushConstants.dstLayer), rgba);\n" + " // The index form follows the binding declared above.\n"; + shaderStr << + " imageStore(outputImageRGB, " + << OutStoreIndex("pos", OutputDescriptorIsArray()) << ", rgba);\n" "}\n"; computeShader = shaderStr.str(); if (dumpShaders) - std::cout << "\nCompute Shader:\n" << computeShader; + VkEncOut() << "\nCompute Shader:\n" << computeShader; return computeShader.size(); } @@ -2761,12 +2968,24 @@ size_t VulkanFilterYuvCompute::InitYCBCRCOPY(std::string& computeShader) GenHeaderAndPushConst(shaderStr); // 2. Generate IO bindings + // + // imageArray is TRUE on both sides here and reads as unconditional, but it + // is a request about the PLANE bindings only: either side of this filter + // may be packed single-plane (AYUV / Y410), and + // ShaderGenerateImagePlaneDescriptors ignores this argument on that arm. + // The input side is image2D by construction; the output side comes from + // OutputDescriptorIsArray(), which is false here because this generator's + // callers do not set FLAG_OUTPUT_IMAGE_ARRAY -- so the arm still declares + // image2D, matching the combined VK_IMAGE_VIEW_TYPE_2D view it binds. + // Do not re-derive packed-ness here to pass a different flag -- that + // duplicate is what let the declaration and the binding disagree. + // // Input Descriptors ShaderGeneratePlaneDescriptors(shaderStr, true, // isInput 0, // startBinding 0, // set - true, + true, // imageArray (plane bindings) VK_DESCRIPTOR_TYPE_STORAGE_BUFFER); // Output Descriptors @@ -2774,12 +2993,12 @@ size_t VulkanFilterYuvCompute::InitYCBCRCOPY(std::string& computeShader) false, // isInput 4, // startBinding 0, // set - true, // imageArray + true, // imageArray (plane bindings) VK_DESCRIPTOR_TYPE_STORAGE_BUFFER); // Binding 9: Subsampled Y output image (optional, only if enabled) if (m_enableYSubsampling) { - std::cout << "[VulkanFilterYuvCompute] Generating binding 9 for subsampledImageY" << std::endl; + VkEncOut() << "[VulkanFilterYuvCompute] Generating binding 9 for subsampledImageY" << std::endl; shaderStr << "// Binding 9: Subsampled Y output (2x2 downsampled luma for AQ)\n"; // Use a separate variable for subsampledImage aspects to avoid modifying m_outputImageAspects // ShaderGenerateImagePlaneDescriptors modifies the imageAspects parameter for R8/R16 formats @@ -2892,6 +3111,11 @@ size_t VulkanFilterYuvCompute::InitYCBCRCOPY(std::string& computeShader) GenReadYCbCrBlock(shaderStr, ySubsampleHorzRatio, ySubsampleVertRatio, isInputTwoPlane, hasInputChroma, m_enableRowAndColumnReplication, inputChromaHorzRatio, inputChromaVertRatio, m_inputPackedYcbcr); + // 11b. Undo the source's MSB alignment BEFORE anything reads the values. + // Pairs with step 13; see GenApplyBlockInputShift for why the output + // half alone saturates an X6/X4-packed source. + GenApplyBlockInputShift(shaderStr, ySubsampleHorzRatio, ySubsampleVertRatio, inputBitDepth, m_inputEnableMsbToLsbShift, hasInputChroma); + // 12. Convert block (if needed) // cbcrOut is DECLARED here but REFERENCED by GenWriteYCbCrBlock, which gates on // hasOutputChroma. Gating the declaration on hasInputChroma alone leaves it undeclared @@ -2904,9 +3128,14 @@ size_t VulkanFilterYuvCompute::InitYCBCRCOPY(std::string& computeShader) // 14. Write YCbCr block (handles both 4:2:0 and 4:4:4 output) GenWriteYCbCrBlock(shaderStr, ySubsampleHorzRatio, ySubsampleVertRatio, isOutputTwoPlane, hasOutputChroma, - outputChromaHorzRatio, outputChromaVertRatio, m_outputPackedYcbcr); - - // 15. Compute and write subsampled Y (only if enabled for AQ) + outputChromaHorzRatio, outputChromaVertRatio, m_outputPackedYcbcr, OutputDescriptorIsArray()); + + // 15. Compute and write subsampled Y (only if enabled for AQ). + // The box filter spans the invocation's luma block, which is 2x2 for the + // 4:2:0 output that is the only configuration enabling this: the encoder + // sets FLAG_ENABLE_Y_SUBSAMPLING only alongside its NV12 encode input + // (VkVideoEncoder.cpp). A non-4:2:0 output would size the "subsampled" Y + // differently; that combination has no caller and is not addressed here. if (m_enableYSubsampling) { GenComputeAndWriteSubsampledY(shaderStr, ySubsampleHorzRatio, ySubsampleVertRatio); } @@ -2916,7 +3145,7 @@ size_t VulkanFilterYuvCompute::InitYCBCRCOPY(std::string& computeShader) computeShader = shaderStr.str(); if (dumpShaders) - std::cout << "\nCompute Shader:\n" << computeShader; + VkEncOut() << "\nCompute Shader:\n" << computeShader; return computeShader.size(); } @@ -3008,7 +3237,7 @@ size_t VulkanFilterYuvCompute::InitYCBCRCLEAR(std::string& computeShader) computeShader = shaderStr.str(); if (dumpShaders) - std::cout << "\nCompute Shader:\n" << computeShader; + VkEncOut() << "\nCompute Shader:\n" << computeShader; return computeShader.size(); } @@ -3107,11 +3336,19 @@ static void GenReadRgbBlock(std::stringstream& shaderStr, uint32_t chromaVertRatio, bool enableReplication) { - shaderStr << " // Read " << chromaHorzRatio << "x" << chromaVertRatio << " RGB block\n"; + shaderStr << " // Read " << chromaHorzRatio << "x" << chromaVertRatio + << " RGB block (imageLoad, STORAGE_IMAGE)\n"; for (uint32_t y = 0; y < chromaVertRatio; y++) { for (uint32_t x = 0; x < chromaHorzRatio; x++) { - shaderStr << " vec3 rgb" << x << y << " = imageLoad(inputImageRGBA, "; + // imageLoad returns the vec4 in RGBA order regardless of the byte + // order the VkFormat names -- that is a property of the format, not + // of the access -- so .rgb is the correct swizzle. The declared + // `rgba8` qualifier is what the storage form is told the layout is, + // and it is the only 8-bit four-component qualifier GLSL has. + shaderStr << " vec3 rgb" << x << y << " = " + << "imageLoad" + << "(inputImageRGBA, "; if (enableReplication) { shaderStr << "min(lumaPos + ivec2(" << x << ", " << y << "), inputLumaMax)"; } else { @@ -3304,6 +3541,40 @@ size_t VulkanFilterYuvCompute::InitRGBA2YCBCR(std::string& computeShader) GenHeaderAndPushConst(shaderStr); // 2. Input RGBA image binding (binding 0) + // + // THE COLOUR CONTRACT, stated here because it is decided by the pair + // (view format, access form) and by nothing else: + // + // * The view must be a _UNORM view, never a *_SRGB one, and two + // independent facts say so. + // + // WHY _UNORM. This filter applies the colour MATRIX ONLY -- + // GenRgbToYCbCrConversion emits the Rec.601/709/2020 matrix and an + // optional range compression, and no transfer function anywhere + // (see also ycbcr_utils.h, "No gamma correction yet"). That is + // CORRECT, because Rec.709/601 Y'CbCr is defined on GAMMA-ENCODED + // R'G'B'. The _UNORM view hands the stored code values over + // untouched, which is what the matrix wants. + // + // WHY A *_SRGB VIEW CANNOT REACH THIS BINDING AT ALL. The binding + // declared below is a VK_DESCRIPTOR_TYPE_STORAGE_IMAGE read with + // imageLoad, and no *_SRGB format carries + // VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT -- sRGB is a sampled-image + // feature. A view whose usage includes VK_IMAGE_USAGE_STORAGE_BIT + // must carry that feature itself + // (VUID-VkImageViewCreateInfo-usage-02275), so an sRGB view can + // never become this descriptor. A storage read applies no transfer + // function of its own either, which is why the requirement above is + // about the CODE VALUES the view holds and not about the fetch. + // * This is a property of the VIEW, so it is the caller's to honour; + // what is guaranteed here is that nothing on this path introduces a + // transfer function of its own. A _UNORM view in gives R'G'B' in. + // + // The component order is a property of the VIEW. The declaration below + // names the `rgba8` qualifier because GLSL offers no other for an 8-bit + // four-component storage image -- there is no `bgra8` -- so a + // B8G8R8A8_UNORM view is read through a declaration that does not spell + // its layout, and the order it yields is the implementation's to decide. shaderStr << " // Input RGBA image binding\n"; GenImageIoBindingLayout(shaderStr, "inputImage", "RGBA", "rgba8", true, 0, 0, false); @@ -3320,12 +3591,13 @@ size_t VulkanFilterYuvCompute::InitRGBA2YCBCR(std::string& computeShader) // packed -> SINGLE-plane, so there is no per-plane // view; it binds the combined view, which // follows layerCount and is 2D -> image2D - // Both directions are real errors, so this cannot be a - // constant: an ivec2 store into an image2DArray does not - // compile, and an image2DArray declaration against the - // packed combined view trips the same VUID the other way - // ("view is 2D but OpTypeImage Arrayed = 1"). - (m_outputPackedYcbcr == nullptr), + // Hardcoding true breaks every packed 4:4:4 format (AYUV, Y410): the + // generated store then fails to compile, and forcing that store to ivec3 + // instead trips the VUID from the other direction -- "view is 2D but + // OpTypeImage Arrayed = 1". (m_outputPackedYcbcr == nullptr) gets the + // planar/packed half of that right but assumes the packed arm is + // single-layer; OutputDescriptorIsArray() carries both halves. + OutputDescriptorIsArray(), VK_DESCRIPTOR_TYPE_STORAGE_BUFFER); // 3b. Binding 9: Subsampled Y output image (optional, only if enabled) @@ -3393,7 +3665,8 @@ size_t VulkanFilterYuvCompute::InitRGBA2YCBCR(std::string& computeShader) } // 8. Read RGB block - GenReadRgbBlock(shaderStr, chromaHorzRatio, chromaVertRatio, m_enableRowAndColumnReplication); + GenReadRgbBlock(shaderStr, chromaHorzRatio, chromaVertRatio, + m_enableRowAndColumnReplication); // 9. Convert RGB to YCbCr GenConvertRgbBlockToYCbCr(shaderStr, chromaHorzRatio, chromaVertRatio); @@ -3439,7 +3712,8 @@ size_t VulkanFilterYuvCompute::InitRGBA2YCBCR(std::string& computeShader) // which follows layerCount and is VK_IMAGE_VIEW_TYPE_2D -- hence the // matching image2D declaration in InitRGBA2YCBCR. The planar branch // below is the one that needs dstLayer. - << " imageStore(outputImageRGB, lumaPos, vec4(" + << " imageStore(outputImageRGB, " + << OutStoreIndex("lumaPos", OutputDescriptorIsArray()) << ", vec4(" << component[0] << ", " << component[1] << ", " << component[2] << ", " << component[3] << "));\n" << " \n"; @@ -3466,7 +3740,7 @@ size_t VulkanFilterYuvCompute::InitRGBA2YCBCR(std::string& computeShader) computeShader = shaderStr.str(); if (dumpShaders) - std::cout << "\nRGBA2YCBCR Compute Shader:\n" << computeShader; + VkEncOut() << "\nRGBA2YCBCR Compute Shader:\n" << computeShader; return computeShader.size(); } @@ -3585,9 +3859,11 @@ uint32_t VulkanFilterYuvCompute::UpdateImageDescriptorSets( writeDescriptorSets[descrIndex].dstSet = VK_NULL_HANDLE; writeDescriptorSets[descrIndex].dstBinding = dstBinding; writeDescriptorSets[descrIndex].descriptorCount = 1; - writeDescriptorSets[descrIndex].descriptorType = (ccSampler != VK_NULL_HANDLE) ? - VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER : - descriptorType; + const VkDescriptorType effectiveDescriptorType = + (ccSampler != VK_NULL_HANDLE) ? + VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER : + descriptorType; + writeDescriptorSets[descrIndex].descriptorType = effectiveDescriptorType; imageDescriptors[descrIndex].sampler = ccSampler; imageDescriptors[descrIndex].imageView = (curImageAspect == 0) ? imageView->GetImageView() : @@ -4822,9 +5098,10 @@ VkResult VulkanFilterYuvCompute::RecordComputeDispatch(VkCommandBuffer cmdBuf, uint32_t bufferIdx, const FilterExecutionDesc& execDesc) { // Delegate to the image->image overload, which owns the descriptor binding and the - // dispatch. Nothing here may report VK_SUCCESS without recording a dispatch: that - // turns RecordCommandBuffer(execDesc) into a transfers-only path that still reports - // success, handing the caller its staging copies and an output image nothing wrote. + // dispatch. Returning VK_SUCCESS here without recording anything would make + // RecordCommandBuffer(execDesc) a transfers-only path that still reports + // success: the caller gets its staging copies and an output image nothing + // ever wrote. Refuse instead. if (execDesc.numInputs == 0 || execDesc.numOutputs == 0) { return VK_ERROR_INVALID_EXTERNAL_HANDLE; } diff --git a/common/libs/VkCodecUtils/VulkanFilterYuvCompute.h b/common/libs/VkCodecUtils/VulkanFilterYuvCompute.h index a7b1de73..4c055d8e 100644 --- a/common/libs/VkCodecUtils/VulkanFilterYuvCompute.h +++ b/common/libs/VkCodecUtils/VulkanFilterYuvCompute.h @@ -345,9 +345,10 @@ struct FilterIOSlot { /// /// A TransferResource carries a raw VkImage, which is enough for vkCmdCopy* but not /// for the compute descriptors -- those bind image views, and for multi-planar - /// formats a per-plane view set. Leaving this null is an error, not a way to ask for - /// the transfers alone: RecordComputeDispatch() refuses it rather than reporting - /// success over a filter execution that never happened. + /// formats a per-plane view set. Leaving it null is refused rather than + /// tolerated: RecordComputeDispatch() returns VK_ERROR_INVALID_EXTERNAL_HANDLE, + /// because dispatching nothing and reporting success hands the caller its pre- + /// and post-transfers plus an output image the filter never wrote. const VkImageResourceView* primaryView{nullptr}; /// Optional pre-transfer: source data to stage into primary before compute @@ -720,7 +721,22 @@ class VulkanFilterYuvCompute : public VulkanFilter /// Replicate edge pixels for all out-of-bounds reads FLAG_ENABLE_ROW_COLUMN_REPLICATION_ALL = (1 << 4), - + + /// The OUTPUT image the caller will bind is layered (layerCount > 1). + /// + /// A single-plane output -- packed YCbCr or RGBA -- is bound from the + /// COMBINED view, and VkImageResourceView::Create types that view + /// VK_IMAGE_VIEW_TYPE_2D_ARRAY when layerCount > 1 and + /// VK_IMAGE_VIEW_TYPE_2D otherwise. The shader's OpTypeImage Arrayed + /// operand must match the view it is bound with + /// (VUID-vkCmdDispatch-viewType-07752), and the shader is generated at + /// Create() time, before any image is bound -- so the caller has to say. + /// + /// Default (unset) means single-layer, which is what every current + /// caller uses. A MULTI-PLANAR output is unaffected: it binds per-plane + /// views, which are always array views regardless of this flag. + FLAG_OUTPUT_IMAGE_ARRAY = (1 << 5), + // Transfer operation flags FLAG_PRE_TRANSFER_ENABLED = (1 << 8), ///< Enable pre-transfer stage FLAG_POST_TRANSFER_ENABLED = (1 << 9), ///< Enable post-transfer stage @@ -731,6 +747,17 @@ class VulkanFilterYuvCompute : public VulkanFilter FLAG_POST_TRANSFER = FLAG_POST_TRANSFER_ENABLED, }; + // Whether the OUTPUT descriptor must be declared as an array image. + // + // It follows the VIEW the descriptor binds, which is decided by the output + // format, not by a constant: + // multi-planar output -> per-plane views, always VK_IMAGE_VIEW_TYPE_2D_ARRAY + // single-plane output -> the combined view, which follows layerCount + // Hardcoding either answer has broken this filter in both directions -- + // once as a packed 4:4:4 store that would not compile, once as a stray + // layer index on a 2D view. + bool OutputDescriptorIsArray() const; + static constexpr uint32_t maxNumComputeDescr = 10; static constexpr VkImageAspectFlags validPlaneAspects = VK_IMAGE_ASPECT_PLANE_0_BIT | @@ -814,6 +841,10 @@ class VulkanFilterYuvCompute : public VulkanFilter , m_workgroupSizeX(16) , m_workgroupSizeY(16) , m_maxNumFrames(maxNumFrames) + , m_inputPackedYcbcr(nullptr) + , m_outputPackedYcbcr(nullptr) + , m_blockHorzRatio(2) + , m_blockVertRatio(2) , m_ycbcrPrimariesConstants (pYcbcrPrimariesConstants ? *pYcbcrPrimariesConstants : YcbcrPrimariesConstants{0.0, 0.0}) @@ -831,10 +862,7 @@ class VulkanFilterYuvCompute : public VulkanFilter , m_enableRowAndColumnReplication((filterFlags & (FLAG_ENABLE_ROW_COLUMN_REPLICATION_ONE | FLAG_ENABLE_ROW_COLUMN_REPLICATION_ALL)) != 0) , m_inputIsBuffer(false) , m_outputIsBuffer(false) - , m_inputPackedYcbcr(nullptr) - , m_outputPackedYcbcr(nullptr) - , m_blockHorzRatio(2) - , m_blockVertRatio(2) + , m_outputImageArray((filterFlags & FLAG_OUTPUT_IMAGE_ARRAY) != 0) , m_enableYSubsampling((filterFlags & FLAG_ENABLE_Y_SUBSAMPLING) != 0) , m_skipCompute((filterFlags & FLAG_SKIP_COMPUTE) != 0 || filterType == XFER_IMAGE_TO_BUFFER || @@ -1171,7 +1199,10 @@ class VulkanFilterYuvCompute : public VulkanFilter * @param isInput Whether this is an input or output resource * @param startBinding Starting binding number in the descriptor set * @param set Descriptor set number - * @param imageArray Whether to use image2DArray or image2D + * @param imageArray Whether the PLANE bindings use image2DArray or + * image2D. It does not reach the single-plane (COLOR_BIT) arm, + * which is always image2D because it is bound from the combined + * VK_IMAGE_VIEW_TYPE_2D view; see that arm for why. * @return The next available binding number after all descriptors are created */ uint32_t ShaderGenerateImagePlaneDescriptors(std::stringstream& computeShader, @@ -1218,7 +1249,8 @@ class VulkanFilterYuvCompute : public VulkanFilter * @param isInput Whether this is an input or output resource * @param startBinding Starting binding number in the descriptor set * @param set Descriptor set number - * @param imageArray Whether to use image2DArray or image2D (for image resources) + * @param imageArray Whether the PLANE bindings use image2DArray or image2D + * (image resources only; the single-plane arm is always image2D) * @param bufferType The Vulkan descriptor type to use for buffer resources * @return The next available binding number after all descriptors are created */ @@ -1340,6 +1372,7 @@ class VulkanFilterYuvCompute : public VulkanFilter uint32_t m_enableRowAndColumnReplication : 1; uint32_t m_inputIsBuffer : 1; uint32_t m_outputIsBuffer : 1; + uint32_t m_outputImageArray : 1; // FLAG_OUTPUT_IMAGE_ARRAY uint32_t m_enableYSubsampling : 1; // Enable 2x2 Y subsampling output uint32_t m_skipCompute : 1; // Skip compute (transfer-only mode) diff --git a/common/libs/VkCodecUtils/VulkanShaderCompiler.cpp b/common/libs/VkCodecUtils/VulkanShaderCompiler.cpp index 246b504a..586d0b53 100644 --- a/common/libs/VkCodecUtils/VulkanShaderCompiler.cpp +++ b/common/libs/VkCodecUtils/VulkanShaderCompiler.cpp @@ -15,6 +15,7 @@ */ #include "assert.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include @@ -41,7 +42,7 @@ void* VulkanShaderCompiler::GetSharedCompiler() { // First instance - create the shared backend g_sharedBackend = VulkanShaderCompilerBackend::Create(); if (g_sharedBackend == nullptr) { - std::cerr << "VulkanShaderCompiler: Failed to initialize the shader compiler backend!" << std::endl; + VkEncErr() << "VulkanShaderCompiler: Failed to initialize the shader compiler backend!" << std::endl; return nullptr; } } @@ -104,7 +105,7 @@ VkShaderModule VulkanShaderCompiler::BuildGlslShader(const char *shaderCode, siz VkResult result = vkDevCtx->CreateShaderModule(*vkDevCtx, &shaderModuleCreateInfo, nullptr, &shaderModule); assert(result == VK_SUCCESS); if (result != VK_SUCCESS) { - std::cerr << "Failed to create shader module" << std::endl; + VkEncErr() << "Failed to create shader module" << std::endl; return VK_NULL_HANDLE; } diff --git a/common/libs/VkCodecUtils/VulkanShaderCompilerGlslang.cpp b/common/libs/VkCodecUtils/VulkanShaderCompilerGlslang.cpp index 6bd11598..d60e05f8 100644 --- a/common/libs/VkCodecUtils/VulkanShaderCompilerGlslang.cpp +++ b/common/libs/VkCodecUtils/VulkanShaderCompilerGlslang.cpp @@ -20,6 +20,7 @@ // dependency without changing what compiles the shaders. #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include // The SPIRV headers are included WITHOUT a glslang/ prefix on purpose. An @@ -57,7 +58,7 @@ EShLanguage getGlslangShaderStage(VkShaderStageFlagBits type) case VK_SHADER_STAGE_COMPUTE_BIT: return EShLangCompute; default: - std::cerr << "VulkanShaderCompiler: invalid VkShaderStageFlagBits type = " + VkEncErr() << "VulkanShaderCompiler: invalid VkShaderStageFlagBits type = " << type << std::endl; } return EShLangCount; @@ -112,7 +113,7 @@ class GlslangCompilerBackend : public VulkanShaderCompilerBackend { if (!shader.parse(GetDefaultResources(), 450, ENoProfile, false /* forceDefaultVersionAndProfile */, false /* forwardCompatible */, messages)) { - std::cerr << "Compilation error: \n" + VkEncErr() << "Compilation error: \n" << shader.getInfoLog() << "\n" << shader.getInfoDebugLog() << std::endl; return false; @@ -121,7 +122,7 @@ class GlslangCompilerBackend : public VulkanShaderCompilerBackend { glslang::TProgram program; program.addShader(&shader); if (!program.link(messages)) { - std::cerr << "Link error: \n" << program.getInfoLog() << std::endl; + VkEncErr() << "Link error: \n" << program.getInfoLog() << std::endl; return false; } @@ -137,11 +138,11 @@ class GlslangCompilerBackend : public VulkanShaderCompilerBackend { const std::string spvMessages = logger.getAllMessages(); if (!spvMessages.empty()) { - std::cerr << "SPIR-V generation: " << spvMessages << std::endl; + VkEncErr() << "SPIR-V generation: " << spvMessages << std::endl; } if (spirv.empty()) { - std::cerr << "SPIR-V generation produced no code" << std::endl; + VkEncErr() << "SPIR-V generation produced no code" << std::endl; return false; } @@ -159,7 +160,7 @@ VulkanShaderCompilerBackend* VulkanShaderCompilerBackend::Create() { // Process-global; must be balanced by FinalizeProcess() in Destroy(). if (!glslang::InitializeProcess()) { - std::cerr << "VulkanShaderCompiler: glslang::InitializeProcess() failed!" << std::endl; + VkEncErr() << "VulkanShaderCompiler: glslang::InitializeProcess() failed!" << std::endl; return nullptr; } diff --git a/common/libs/VkCodecUtils/VulkanVideoImagePool.cpp b/common/libs/VkCodecUtils/VulkanVideoImagePool.cpp index 9c97b889..0b67127d 100644 --- a/common/libs/VkCodecUtils/VulkanVideoImagePool.cpp +++ b/common/libs/VkCodecUtils/VulkanVideoImagePool.cpp @@ -316,8 +316,36 @@ VkResult VulkanVideoImagePool::Configure(const VulkanDeviceContext* vkDevCtx, bool useImageArray, bool useImageViewArray, bool useLinearImage, - uint64_t drmFormatModifier) + uint64_t drmFormatModifier, + VkImageLayout initialLayout) { + // One family is the ordinary case and stays EXCLUSIVE. + return Configure(vkDevCtx, numImages, imageFormat, maxImageExtent, + imageUsage, std::vector{queueFamilyIndex}, + requiredMemProps, pVideoProfile, aspectMask, + useImageArray, useImageViewArray, useLinearImage, + drmFormatModifier, initialLayout); +} + +VkResult VulkanVideoImagePool::Configure(const VulkanDeviceContext* vkDevCtx, + uint32_t numImages, + VkFormat imageFormat, + const VkExtent2D& maxImageExtent, + VkImageUsageFlags imageUsage, + const std::vector& queueFamilyIndices, + VkMemoryPropertyFlags requiredMemProps, + const VkVideoProfileInfoKHR* pVideoProfile, + VkImageAspectFlags aspectMask, + bool useImageArray, + bool useImageViewArray, + bool useLinearImage, + uint64_t drmFormatModifier, + VkImageLayout initialLayout) +{ + if (queueFamilyIndices.empty()) { + assert(!"A pool needs at least one queue family"); + return VK_ERROR_INITIALIZATION_FAILED; + } std::lock_guard lock(m_queueMutex); if (numImages > m_imageResources.size()) { assert(!"Number of requested images exceeds the max size of the image array"); @@ -343,7 +371,17 @@ VkResult VulkanVideoImagePool::Configure(const VulkanDeviceContext* vkDevCtx, m_videoProfile.InitFromProfile(pVideoProfile); } - m_queueFamilyIndex = queueFamilyIndex; + // Copy and deduplicate. The create info points into this vector, so it + // has to outlive the call, and a duplicate entry in a CONCURRENT list is + // VUID-VkImageCreateInfo-sharingMode-01420. + m_queueFamilyIndices.clear(); + for (uint32_t family : queueFamilyIndices) { + if (std::find(m_queueFamilyIndices.begin(), m_queueFamilyIndices.end(), + family) == m_queueFamilyIndices.end()) { + m_queueFamilyIndices.push_back(family); + } + } + m_queueFamilyIndex = m_queueFamilyIndices[0]; m_requiredMemProps = requiredMemProps; // Image create info for the images @@ -361,10 +399,19 @@ VkResult VulkanVideoImagePool::Configure(const VulkanDeviceContext* vkDevCtx, m_imageCreateInfo.tiling = useLinearImage ? VK_IMAGE_TILING_LINEAR : VK_IMAGE_TILING_OPTIMAL; } m_imageCreateInfo.usage = imageUsage; - m_imageCreateInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE; - m_imageCreateInfo.queueFamilyIndexCount = 1; - m_imageCreateInfo.pQueueFamilyIndices = &m_queueFamilyIndex; - m_imageCreateInfo.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + if (m_queueFamilyIndices.size() > 1) { + // Written by one family and read by another. CONCURRENT is what makes + // that legal without the release/acquire pair no caller issues here. + m_imageCreateInfo.sharingMode = VK_SHARING_MODE_CONCURRENT; + m_imageCreateInfo.queueFamilyIndexCount = + (uint32_t)m_queueFamilyIndices.size(); + m_imageCreateInfo.pQueueFamilyIndices = m_queueFamilyIndices.data(); + } else { + m_imageCreateInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + m_imageCreateInfo.queueFamilyIndexCount = 1; + m_imageCreateInfo.pQueueFamilyIndices = &m_queueFamilyIndex; + } + m_imageCreateInfo.initialLayout = initialLayout; m_imageCreateInfo.flags = VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT; const bool hasVideoUsage = (imageUsage & (VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | @@ -376,6 +423,37 @@ VkResult VulkanVideoImagePool::Configure(const VulkanDeviceContext* vkDevCtx, | VK_IMAGE_CREATE_VIDEO_PROFILE_INDEPENDENT_BIT_KHR; } + // STORAGE ON A MULTI-PLANAR FORMAT IS A PER-PLANE-VIEW USAGE, so it has to + // be validated against the PLANE format and not against the YCbCr format. + // + // Neither VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM nor + // VK_FORMAT_G8_B8R8_2PLANE_420_UNORM advertises + // VK_FORMAT_FEATURE_2_STORAGE_IMAGE_BIT in either linear or optimal + // tiling, while the plane formats VK_FORMAT_R8_UNORM and + // VK_FORMAT_R8G8_UNORM advertise it in both. So a create that names + // VK_IMAGE_USAGE_STORAGE_BIT on the YCbCr format without + // VK_IMAGE_CREATE_EXTENDED_USAGE_BIT is checked against the image format, + // vkGetPhysicalDeviceImageFormatProperties2 answers + // VK_ERROR_FORMAT_NOT_SUPPORTED, and vkCreateImage is + // VUID-VkImageCreateInfo-imageCreateMaxMipLevels-02251. + // + // WHY ONLY ONE POOL EVER TRIPPED IT. Every pool with a video usage bit + // already gets EXTENDED_USAGE from the branch above. The odd one out is + // VkVideoEncoder::InitEncoder's LINEAR STAGING pool + // (m_linearInputImagePool), which asks for SAMPLED|STORAGE|TRANSFER_SRC and + // no video usage at all -- so it got MUTABLE_FORMAT alone and raised 02251 + // once per pool image at session init (24 per session on the file-input + // lane, in both the 2-plane and the 3-plane shape). + // + // MUTABLE_FORMAT, unconditional above, is the other half this needs: + // EXTENDED_USAGE relaxes the usage check to the VIEW format, and it is + // MUTABLE_FORMAT that makes an R8/R8G8 per-plane view of a YCbCr image + // legal to create at all (VUID-VkImageViewCreateInfo-image-01762). + if ((YcbcrVkFormatInfo(imageFormat) != nullptr) && + ((imageUsage & VK_IMAGE_USAGE_STORAGE_BIT) != 0)) { + m_imageCreateInfo.flags |= VK_IMAGE_CREATE_EXTENDED_USAGE_BIT; + } + // DRM format modifier pNext chain (must persist through image creation below) VkImageDrmFormatModifierListCreateInfoEXT drmModListInfo{ VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_LIST_CREATE_INFO_EXT}; diff --git a/common/libs/VkCodecUtils/VulkanVideoImagePool.h b/common/libs/VkCodecUtils/VulkanVideoImagePool.h index cd64611d..542c81c0 100644 --- a/common/libs/VkCodecUtils/VulkanVideoImagePool.h +++ b/common/libs/VkCodecUtils/VulkanVideoImagePool.h @@ -36,6 +36,7 @@ class VulkanVideoImagePoolNode : public VkVideoRefCountBase { VulkanVideoImagePoolNode() : m_vkDevCtx() , m_currentImageLayout(VK_IMAGE_LAYOUT_UNDEFINED) + , m_stagedInputResidualLayout(VK_IMAGE_LAYOUT_MAX_ENUM) , m_pictureResourceInfo() , m_imageResourceView() , m_parent(nullptr) @@ -97,9 +98,16 @@ class VulkanVideoImagePoolNode : public VkVideoRefCountBase { return false; } + // Whether this node holds an image -- not whether that image has a view. + // The two coincide for pool-allocated nodes, which always create image and + // view together, but not for an external wrapper (CreateExternal): a + // transfer-only staging import carries no view-compatible usage, so it has + // an image and deliberately no view. Testing the view there would report a + // node that plainly holds an image as empty. bool ImageExist() { - return (!!m_imageResourceView && (m_imageResourceView->GetImageView() != VK_NULL_HANDLE)); + return (!!m_imageResourceView && !!m_imageResourceView->GetImageResource() && + (m_imageResourceView->GetImageResource()->GetImage() != VK_NULL_HANDLE)); } bool RecreateImage() { @@ -121,6 +129,41 @@ class VulkanVideoImagePoolNode : public VkVideoRefCountBase { return true; } + // === STAGED-INPUT RESIDUAL LAYOUT === + // + // The layout VkVideoEncoder::StageInputFrame last LEFT this image in, + // as a fact about work the library itself recorded -- not a + // declaration. + // + // DELIBERATELY NOT m_currentImageLayout. That field is written by three + // unrelated producers (CreateImage from a create-info, CreateExternal + // from a caller-supplied declaration, and the pool handout in + // GetImageSetNewLayout from a REQUESTED layout), so it answers "what did + // somebody say" and not "what did we do". Reading it is what forced an + // earlier attempt at this fix to be reverted, and the reason is that a + // declaration cannot be a truth source about a barrier the library + // recorded. This field has exactly ONE writer -- the handback at the end + // of StageInputFrame, which sets it from the value that handback used as + // its barrier newLayout -- and exactly ONE reader, the acquire at the top + // of the NEXT StageInputFrame for the same registration. + // + // VK_IMAGE_LAYOUT_MAX_ENUM means "the library has not moved this image + // yet", i.e. the caller's declaration is still the only fact available. + // Every node starts there, including the per-frame wrapper the LEGACY + // SubmitExternalFrame lane builds fresh on every call (VkVideoEncoder:: + // WrapExternalImage) -- which is why that lane keeps its pre-existing + // behaviour byte for byte: a node that lives for one frame can never + // carry a residual into a second one. + bool HasStagedInputResidualLayout() const { + return m_stagedInputResidualLayout != VK_IMAGE_LAYOUT_MAX_ENUM; + } + VkImageLayout GetStagedInputResidualLayout() const { + return m_stagedInputResidualLayout; + } + void SetStagedInputResidualLayout(VkImageLayout residualLayout) { + m_stagedInputResidualLayout = residualLayout; + } + VkVideoPictureResourceInfoKHR* GetPictureResourceInfo() { return &m_pictureResourceInfo; } int32_t GetImageIndex() { return m_parentIndex; } @@ -156,6 +199,8 @@ class VulkanVideoImagePoolNode : public VkVideoRefCountBase { private: const VulkanDeviceContext* m_vkDevCtx; VkImageLayout m_currentImageLayout; + // See the accessors above. MAX_ENUM == "the library has not moved it". + VkImageLayout m_stagedInputResidualLayout; VkVideoPictureResourceInfoKHR m_pictureResourceInfo; VkSharedBaseObj m_imageResourceView; VkSharedBaseObj m_parent; @@ -192,6 +237,32 @@ class VulkanVideoImagePool : public VkVideoRefCountBase, static VkResult Create(const VulkanDeviceContext* vkDevCtx, VkSharedBaseObj& imagePool); + // For a pool whose images are written on one queue family and read on + // another. + // + // Ownership of a VK_SHARING_MODE_EXCLUSIVE image does not move between + // families by itself: without the release/acquire barrier pair, what the + // reading family sees is undefined. Naming both families here creates the + // images CONCURRENT instead. The list is copied -- the create info holds a + // pointer into pool storage, not into the caller's -- and deduplicated, so + // a session whose two families are the same still gets EXCLUSIVE, which is + // what it should have. + VkResult Configure(const VulkanDeviceContext* vkDevCtx, + uint32_t numImages, + VkFormat imageFormat, + const VkExtent2D& maxImageExtent, + VkImageUsageFlags imageUsage, + const std::vector& queueFamilyIndices, + VkMemoryPropertyFlags requiredMemProps, + const VkVideoProfileInfoKHR* pVideoProfile, + VkImageAspectFlags aspectMask, + bool useImageArray, + bool useImageViewArray, + bool useLinear, + uint64_t drmFormatModifier = 0, + VkImageLayout initialLayout = + VK_IMAGE_LAYOUT_UNDEFINED); + VkResult Configure(const VulkanDeviceContext* vkDevCtx, uint32_t numImages, VkFormat imageFormat, @@ -204,7 +275,16 @@ class VulkanVideoImagePool : public VkVideoRefCountBase, bool useImageArray, bool useImageViewArray, bool useLinear, - uint64_t drmFormatModifier = 0); + uint64_t drmFormatModifier = 0, + // The layout the images are CREATED in. UNDEFINED is + // right for every pool whose first writer is the + // device. A pool whose first writer is the HOST, through + // a persistent mapping of a LINEAR image, must say + // PREINITIALIZED instead: it is the only initial layout + // under which host writes performed before any barrier + // are preserved, and the only one a first acquire can + // legally name as its oldLayout. + VkImageLayout initialLayout = VK_IMAGE_LAYOUT_UNDEFINED); void Deinit(); @@ -237,6 +317,9 @@ class VulkanVideoImagePool : public VkVideoRefCountBase, const VulkanDeviceContext* m_vkDevCtx; std::mutex m_queueMutex; uint32_t m_queueFamilyIndex; + // Stable storage for the create info's pQueueFamilyIndices when the pool + // is shared across families. Deduplicated; a single entry means EXCLUSIVE. + std::vector m_queueFamilyIndices; VkVideoCoreProfile m_videoProfile; VkImageCreateInfo m_imageCreateInfo; VkMemoryPropertyFlags m_requiredMemProps; diff --git a/common/libs/VkCodecUtils/VulkanVideoUtils.cpp b/common/libs/VkCodecUtils/VulkanVideoUtils.cpp index 5ef194bc..1d6e9547 100644 --- a/common/libs/VkCodecUtils/VulkanVideoUtils.cpp +++ b/common/libs/VkCodecUtils/VulkanVideoUtils.cpp @@ -512,6 +512,11 @@ VkResult VulkanGraphicsPipeline::CreatePipeline(const VulkanDeviceContext* vkDev m_vertexShaderCache = m_vulkanShaderCompiler.BuildGlslShader(vss, strlen(vss), VK_SHADER_STAGE_VERTEX_BIT, m_vkDevCtx); + // Same unchecked producer as the compute twin: an unguarded + // VK_NULL_HANDLE would reach VkPipelineShaderStageCreateInfo::module. + if (m_vertexShaderCache == VK_NULL_HANDLE) { + return VK_ERROR_INVALID_SHADER_NV; + } } if (m_fssCache.str() != imageFss.str()) { @@ -519,6 +524,14 @@ VkResult VulkanGraphicsPipeline::CreatePipeline(const VulkanDeviceContext* vkDev m_fragmentShaderCache = m_vulkanShaderCompiler.BuildGlslShader(imageFss.str().c_str(), strlen(imageFss.str().c_str()), VK_SHADER_STAGE_FRAGMENT_BIT, m_vkDevCtx); + if (m_fragmentShaderCache == VK_NULL_HANDLE) { + // BEFORE the swap, deliberately. Caching the source that just + // failed to compile would make this the "current" fragment shader + // and the `m_fssCache.str() != imageFss.str()` test above would + // then skip the rebuild forever -- one bad compile would be + // permanent for the life of the object. + return VK_ERROR_INVALID_SHADER_NV; + } m_fssCache.swap(imageFss); if (verbose) printf("\nFragment shader cache output code:\n %s", m_fssCache.str().c_str()); diff --git a/common/libs/VkCodecUtils/nvVkFormats.cpp b/common/libs/VkCodecUtils/nvVkFormats.cpp index c919953f..876fbbdd 100644 --- a/common/libs/VkCodecUtils/nvVkFormats.cpp +++ b/common/libs/VkCodecUtils/nvVkFormats.cpp @@ -24,6 +24,15 @@ const VkFormatDesc vkFormatInfo[] = { { VK_FORMAT_R8G8_UNORM, 2, 2, "rg8", }, { VK_FORMAT_R8G8B8_UNORM, 3, 3, "rgb8", }, { VK_FORMAT_R8G8B8A8_UNORM, 4, 4, "rgba8", }, + // The same texel class as R8G8B8A8_UNORM, and both carried here for the + // same reason: the encoder accepts all three 8-bit RGBA spellings as + // filter inputs, so all three can reach shader generation, and a format + // this table does not hold has no image-format qualifier to declare its + // binding with. GLSL spells the qualifier by texel class rather than by + // component order -- there is no "bgra8" -- so the qualifier is "rgba8" + // for each, and the component order stays a property of the view. + { VK_FORMAT_B8G8R8A8_UNORM, 4, 4, "rgba8", }, + { VK_FORMAT_A8B8G8R8_UNORM_PACK32, 4, 4, "rgba8", }, { VK_FORMAT_R32G32B32A32_SFLOAT, 4, 16, "rgba32f", }, { VK_FORMAT_R16G16B16A16_SFLOAT, 4, 8, "rgba16f", }, { VK_FORMAT_R32G32_SFLOAT, 2, 8, "rg32f", }, @@ -81,6 +90,20 @@ const VkFormatDesc* vkFormatLookUp(VkFormat format) return pVkFormatDesc; } +static_assert((sizeof(vkMpFormatInfo) / sizeof(vkMpFormatInfo[0])) == + YCBCR_VK_FORMAT_INFO_TABLE_SIZE, + "YCBCR_VK_FORMAT_INFO_TABLE_SIZE is what a caller walking this " + "table sizes its array from, so it has to be this table's own " + "length rather than a number kept beside it"); + +const VkMpFormatInfo* YcbcrVkFormatInfoByIndex(uint32_t index) +{ + if (index >= (sizeof(vkMpFormatInfo) / sizeof(vkMpFormatInfo[0]))) { + return NULL; + } + return &vkMpFormatInfo[index]; +} + const VkMpFormatInfo* YcbcrVkFormatInfo(const VkFormat format) { return __ycbcrVkFormatInfo(format); diff --git a/common/libs/VkCodecUtils/pattern.cpp b/common/libs/VkCodecUtils/pattern.cpp index ddf0cc31..dab859be 100644 --- a/common/libs/VkCodecUtils/pattern.cpp +++ b/common/libs/VkCodecUtils/pattern.cpp @@ -15,6 +15,7 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "Helpers.h" #include "VkCodecUtils/VulkanDeviceContext.h" #include "pattern.h" @@ -149,7 +150,7 @@ void generateColorPatternRgba16161616( #include #define ABORT_IF_TRUE(cond) \ if (cond) { \ - printf("condition at %s %d failed, aborting\n", __FILE__, __LINE__); \ + VkEncPrintfOut("condition at %s %d failed, aborting\n", __FILE__, __LINE__); \ return; \ } diff --git a/common/libs/tests/CMakeLists.txt b/common/libs/tests/CMakeLists.txt index a343af93..8e2b0ff1 100644 --- a/common/libs/tests/CMakeLists.txt +++ b/common/libs/tests/CMakeLists.txt @@ -38,6 +38,7 @@ set(TEST_HEADERS set(VKCODECUTILS_DIR "${CMAKE_CURRENT_SOURCE_DIR}/../VkCodecUtils") set(VKCODECUTILS_SOURCES ${VKCODECUTILS_DIR}/VulkanDeviceContext.cpp + ${VKCODECUTILS_DIR}/VkEncoderStdioLatch.cpp ${VKCODECUTILS_DIR}/VulkanDeviceMemoryImpl.cpp ${VKCODECUTILS_DIR}/VkImageResource.cpp ${VKCODECUTILS_DIR}/VkBufferResource.cpp @@ -54,6 +55,7 @@ set(VKCODECUTILS_SOURCES ${VKCODECUTILS_DIR}/VulkanFenceSet.cpp ${VKCODECUTILS_DIR}/Helpers.cpp ${VK_DISPATCH_TABLE_SOURCE} + ${VK_DISPATCH_TABLE_HEADER} ${VKCODECUTILS_DIR}/nvVkFormats.cpp ) @@ -111,8 +113,54 @@ install(TARGETS ${PROJECT_NAME} # Add test enable_testing() -add_test(NAME FilterSmokeTest COMMAND ${PROJECT_NAME} --smoke) -add_test(NAME FilterAllTests COMMAND ${PROJECT_NAME} --all) + +# Both arms dispatch compute shaders and therefore need a real device. main() +# exits 77 when Vulkan init fails for a device-availability reason, so a +# GPU-less host reports SKIPPED rather than FAILED; every other init failure +# exits 1, so a genuine regression cannot hide in the skip arm. +# +# GATED ON VALIDATION. These arms build every image, view, descriptor and +# barrier the compute filter consumes, which is the widest Vulkan surface in +# the tree; and the binary asks for the validation layer only under its +# verbose flag, so under CTest it runs with no layer at all and its exit code +# is the whole verdict. vvs_add_validation_gated_test() forces the layer on, +# witnesses that the loader inserted it, and fails the arm that produced a +# message. +# +# THE CEILINGS BELOW NAME DEFECTS, THEY DO NOT EXCUSE THEM. Each is a +# violation this suite reports today, held at exactly the count it reports so +# that it cannot grow and nothing new can hide behind it: +# +# VUID-VkDeviceQueueCreateInfo-pQueuePriorities-parameter +# vkCreateDevice is called with pQueueCreateInfos[0].pQueuePriorities +# NULL. It is raised once per device creation, by every binary in the +# tree that stands up a device through VulkanDeviceContext directly. +# VUID-vkCmdDispatch-None-08114 +# the triple-output case dispatches with binding 9, the subsampled Y +# output, never written by vkUpdateDescriptorSets. +set(_filter_gate_expect + VUID-VkDeviceQueueCreateInfo-pQueuePriorities-parameter=1) + +vvs_add_validation_gated_test(FilterSmokeTest + TARGET ${PROJECT_NAME} + ARGS --smoke + EXPECT ${_filter_gate_expect}) +vvs_add_validation_gated_test(FilterAllTests + TARGET ${PROJECT_NAME} + ARGS --all + EXPECT ${_filter_gate_expect} + VUID-vkCmdDispatch-None-08114=1) +set_tests_properties(FilterSmokeTest FilterAllTests PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +# --list touches no Vulkan entry point at all: it builds the standard test +# table and prints it, and returns before app.init(). That makes it the one +# arm of this binary that belongs in the gating device-free set -- it proves +# the test-case table still constructs on every push. +add_test(NAME FilterListTests COMMAND ${PROJECT_NAME} --list) +set_tests_properties(FilterListTests PROPERTIES + LABELS "device-free") # Add DRM Format Modifier test (Linux only) if(UNIX AND NOT APPLE) @@ -121,6 +169,15 @@ if(UNIX AND NOT APPLE) endif() endif() +# Add Linux dma-buf SECOND-DEVICE import test (Linux only). +# The counterpart of win32_opaque_import: it proves an image exported by one +# VkDevice can be imported AND USED by a different VkDevice on the same GPU. +if(UNIX AND NOT APPLE) + if(EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/linux_dmabuf_import") + add_subdirectory(linux_dmabuf_import) + endif() +endif() + # Add Win32 Opaque Import test (Windows only) if(WIN32) if(EXISTS "${CMAKE_CURRENT_SOURCE_DIR}/win32_opaque_import") diff --git a/common/libs/tests/README.md b/common/libs/tests/README.md index e62819ac..5b13194e 100644 --- a/common/libs/tests/README.md +++ b/common/libs/tests/README.md @@ -10,7 +10,7 @@ Tests the `VulkanFilterYuvCompute` class directly, independent of any applicatio **Build:** ```bash -cd /data/nvidia/android-extra/video-apps/vulkan-video-samples +cd mkdir -p build && cd build cmake .. -DBUILD_TESTS=ON make -j$(nproc) vk_filter_test @@ -41,7 +41,7 @@ Tests the filter as integrated into the `ThreadedRenderingVk` application, inclu **Run:** ```bash -cd /data/nvidia/vulkan/samples/ThreadedRenderingVk_Standalone +cd ./scripts/test_dump_formats.sh ``` @@ -106,9 +106,31 @@ The `YCBCR2RGBA` filter mode has shader generation issues: ### Y410 Packed Format -Y410 is a packed format requiring special shader handling not yet implemented. - -**Workaround:** Tests `TC008_RGBA_to_Y410` and `TC017_Y410_to_RGBA` are disabled. +Packed 4:4:4 handling is implemented on the `YCBCRCOPY` arm, in both +directions: a packed input is bound as a single storage image and read with +`imageLoad()`, and a packed output is written the same way. That is the arm the +encoder builds, so Y410 is not unimplemented in general. + +It is **not** implemented on either of the two arms these tests exercise, and +both are shader-generation defects rather than missing features: + +- `RGBA2YCBCR` with a packed **output** declares `outputImageRGB` as an + `image2DArray` but stores into it with an `ivec2`, so the generated GLSL does + not compile. This is `TC008_RGBA_to_Y410`'s path. +- `YCBCR2RGBA` has two faults, and the ORDER matters to anyone fixing it. + Generation reaches neither GLSL statement first: it derives a bit depth from + `YcbcrVkFormatInfo(...)`, which answers NULL for `A2B10G10R10_UNORM_PACK32` + because that enumerant is outside both ranges the multi-planar table covers. + That dereference was unguarded and crashed the process; it is now guarded, so + the arm reaches its second fault -- it names `inputImageY` and + `inputImageCbCr` unconditionally, and a packed input declares neither, so the + shader does not compile. This is `TC017_Y410_to_RGBA`'s path. Fixing only the + identifier emission would not have made the case run. The arm is also marked + deprecated. + +**Workaround:** Tests `TC008_RGBA_to_Y410` and `TC017_Y410_to_RGBA` remain +disabled. Re-enabling either needs its arm fixed first -- the disable records a +live defect and is not stale. ### Buffer I/O diff --git a/common/libs/tests/drm_format_mod/CMakeLists.txt b/common/libs/tests/drm_format_mod/CMakeLists.txt index 2017f34c..a3a1caf2 100644 --- a/common/libs/tests/drm_format_mod/CMakeLists.txt +++ b/common/libs/tests/drm_format_mod/CMakeLists.txt @@ -39,6 +39,7 @@ set(TEST_HEADERS set(VKCODECUTILS_DIR "${CMAKE_CURRENT_SOURCE_DIR}/../../VkCodecUtils") set(VKCODECUTILS_SOURCES ${VKCODECUTILS_DIR}/VulkanDeviceContext.cpp + ${VKCODECUTILS_DIR}/VkEncoderStdioLatch.cpp ${VKCODECUTILS_DIR}/VulkanDeviceMemoryImpl.cpp ${VKCODECUTILS_DIR}/VkImageResource.cpp ${VKCODECUTILS_DIR}/VkBufferResource.cpp @@ -53,6 +54,7 @@ set(VKCODECUTILS_SOURCES ${VKCODECUTILS_DIR}/VulkanFenceSet.cpp ${VKCODECUTILS_DIR}/Helpers.cpp ${VK_DISPATCH_TABLE_SOURCE} + ${VK_DISPATCH_TABLE_HEADER} ${VKCODECUTILS_DIR}/nvVkFormats.cpp ) @@ -111,6 +113,15 @@ enable_testing() add_test(NAME DrmFormatModSmokeTest COMMAND ${PROJECT_NAME}) add_test(NAME DrmFormatModListFormats COMMAND ${PROJECT_NAME} --list-formats) add_test(NAME DrmFormatModAllTests COMMAND ${PROJECT_NAME} --all) +# All three enumerate real driver-reported DRM modifiers, so all three need a +# device -- including --list-formats, which queries the physical device. main() +# exits 77 when Vulkan init fails for a device-availability reason; any other +# init failure still exits 1. +set_tests_properties(DrmFormatModSmokeTest + DrmFormatModListFormats + DrmFormatModAllTests PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu") # Print status message(STATUS "drm_format_mod_test: Configured for Linux") diff --git a/common/libs/tests/drm_format_mod/drm-video-issue-readme.txt b/common/libs/tests/drm_format_mod/drm-video-issue-readme.txt index bff6a54c..944d6022 100644 --- a/common/libs/tests/drm_format_mod/drm-video-issue-readme.txt +++ b/common/libs/tests/drm_format_mod/drm-video-issue-readme.txt @@ -61,7 +61,7 @@ does not appear to handle this split correctly. How to reproduce ---------------- Build: - cd /data/nvidia/android-extra/video-apps/vulkan-video-samples/build + cd /build cmake --build . -j4 -- drm_format_mod_test Run (VIDEO_ENCODE_SRC): @@ -118,7 +118,7 @@ Debugging with GDB ------------------- Break on the vkCreateImage failure: - cd /data/nvidia/android-extra/video-apps/vulkan-video-samples/build + cd /build gdb --args ./bin/drm_format_mod_test --video-encode --format NV12 -v (gdb) break vkCreateImage diff --git a/common/libs/tests/drm_format_mod/run_tests_with_patched_layer.sh b/common/libs/tests/drm_format_mod/run_tests_with_patched_layer.sh index 5a601625..c2e4ff23 100644 --- a/common/libs/tests/drm_format_mod/run_tests_with_patched_layer.sh +++ b/common/libs/tests/drm_format_mod/run_tests_with_patched_layer.sh @@ -20,7 +20,7 @@ # ./run_tests_with_patched_layer.sh --ycbcr-only # ./run_tests_with_patched_layer.sh --report # ./run_tests_with_patched_layer.sh --gdb --ycbcr-only -# ./run_tests_with_patched_layer.sh --ssh-remote tzlatinski@192.168.122.216 --ycbcr-only +# ./run_tests_with_patched_layer.sh --ssh-remote user@host.example --ycbcr-only # SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" @@ -50,7 +50,7 @@ while [[ $# -gt 0 ]]; do --ssh-remote) if [[ -z "$2" || "$2" == --* ]]; then echo "ERROR: --ssh-remote requires a user@host argument" - echo "Example: --ssh-remote tzlatinski@192.168.122.216" + echo "Example: --ssh-remote user@host.example" exit 1 fi SSH_REMOTE="$2" @@ -89,13 +89,15 @@ fi # Source the driver development environment (sets up driver build paths) # Redirect output to /dev/null to avoid cluttering test output -DEV_ENV_SCRIPT="/data/nvidia-linux/dev_a/set_dev_env-dev.sh" +# Developer-tree locations. Both are machine-specific: override in the +# environment rather than editing this file. +DEV_ENV_SCRIPT="${DEV_ENV_SCRIPT:-/set_dev_env-dev.sh}" if [[ "$SKIP_DEV_ENV" -eq 0 ]] && [[ -f "$DEV_ENV_SCRIPT" ]]; then source "$DEV_ENV_SCRIPT" > /dev/null 2>&1 fi # Path to the patched validation layer build -PATCHED_LAYER_PATH="/data/nvidia/vulkan/validation-layers-build/Vulkan-ValidationLayers/build-vm/layers" +PATCHED_LAYER_PATH="${PATCHED_LAYER_PATH:-/layers}" # Path to the test binary (check both possible locations) if [[ -x "$REPO_ROOT/build/bin/drm_format_mod_test" ]]; then diff --git a/common/libs/tests/drm_format_mod/src/main.cpp b/common/libs/tests/drm_format_mod/src/main.cpp index 4797bed7..72a0f0ca 100644 --- a/common/libs/tests/drm_format_mod/src/main.cpp +++ b/common/libs/tests/drm_format_mod/src/main.cpp @@ -196,6 +196,32 @@ static bool parseArgs(int argc, char* argv[], TestConfig& config) { // Main //============================================================================= + +// --------------------------------------------------------------------------- +// CTest skip semantics. +// +// A host with no usable Vulkan device has proved nothing about this test's +// subject, so it must report SKIPPED (CTest's conventional 77), not a pass and +// not a failure. Only device-AVAILABILITY results map to 77. Every other +// VkResult -- VK_ERROR_DEVICE_LOST, VK_ERROR_OUT_OF_*_MEMORY, anything else -- +// stays a hard exit 1, so a real regression inside init() cannot hide behind +// this arm and get itself reported as "skipped". +// --------------------------------------------------------------------------- +static const int kCTestSkipExitCode = 77; + +static bool IsNoUsableDeviceResult(VkResult result) { + switch (result) { + case VK_ERROR_INCOMPATIBLE_DRIVER: // -9: no ICD the loader can use + case VK_ERROR_INITIALIZATION_FAILED: // -3: loader/ICD refused to come up + case VK_ERROR_EXTENSION_NOT_PRESENT: // -7: no device offers what we need + case VK_ERROR_LAYER_NOT_PRESENT: // -6 + case VK_ERROR_FEATURE_NOT_PRESENT: // -8: device lacks a required feature + return true; + default: + return false; + } +} + int main(int argc, char* argv[]) { std::cout << "======================================" << std::endl; std::cout << " DRM Format Modifier Test Suite" << std::endl; @@ -238,6 +264,12 @@ int main(int argc, char* argv[]) { VkResult result = testApp.init(config); if (result != VK_SUCCESS) { std::cerr << "Failed to initialize test application: " << result << std::endl; + if (IsNoUsableDeviceResult(result)) { + std::cerr << "No usable Vulkan device on this host: reporting SKIPPED (" + << kCTestSkipExitCode << "). Nothing was proved either way." + << std::endl; + return kCTestSkipExitCode; + } return 1; } diff --git a/common/libs/tests/inc/ColorConversion.h b/common/libs/tests/inc/ColorConversion.h index b5d6575d..ef8e4c79 100644 --- a/common/libs/tests/inc/ColorConversion.h +++ b/common/libs/tests/inc/ColorConversion.h @@ -334,6 +334,30 @@ enum class TestPatternType { Ramp, // Full ramp (all values) Solid, // Solid color (for specific color testing) Random, // Pseudo-random pattern + /** + * @brief Four pure-primary quadrants: R, G / B, white. + * + * THE PATTERN FOR COLOUR BUGS, and the reason it is not a gradient or a + * random field. The two failures that matter on an RGB->YCbCr path are + * both invisible to the obvious checks: + * + * - A RED/BLUE component swap (the exact failure a `rgba8` storage + * qualifier on a BGRA view produces) PERMUTES pixels between + * quadrants. It leaves the byte HISTOGRAM of the image unchanged, so + * no checksum, no size check and no histogram test can see it. + * - A Cb/Cr (U/V) plane swap likewise conserves the histogram. + * + * Saturated primaries separate these: red and blue sit at opposite + * extremes of BOTH chroma axes, so either swap moves a quadrant's (Cb,Cr) + * pair to a value no correct conversion could produce there. A grey ramp + * or a checkerboard has Cb = Cr = neutral everywhere and would pass + * happily under both faults. + * + * White is the fourth quadrant rather than black because black is what a + * dead conversion produces, and a quadrant that is CORRECT when zeroed + * cannot witness that failure. + */ + PurePrimaryQuadrants, }; /** diff --git a/common/libs/tests/inc/FilterTestApp.h b/common/libs/tests/inc/FilterTestApp.h index 8482d519..508f9b07 100644 --- a/common/libs/tests/inc/FilterTestApp.h +++ b/common/libs/tests/inc/FilterTestApp.h @@ -21,6 +21,7 @@ #include #include #include +#include #include "VkCodecUtils/VulkanDeviceContext.h" #include "VkCodecUtils/VkImageResource.h" @@ -28,6 +29,10 @@ #include "VkCodecUtils/VulkanFilterYuvCompute.h" #include "VkCodecUtils/VulkanCommandBufferPool.h" +// TestIOSlot::pattern is a TestPatternType, which this header therefore has to +// see rather than forward-declare (it is a scoped enum used by value). +#include "ColorConversion.h" + namespace vkfilter_test { /** @@ -111,6 +116,29 @@ struct TestIOSlot { uint32_t width{1920}; uint32_t height{1080}; + /// Which pattern to fill an INPUT slot with. Only consulted for the RGBA + /// family today; the YCbCr generators still synthesize from colour bars. + /// Defaults to ColorBars so every pre-existing case is unchanged. + TestPatternType pattern{TestPatternType::ColorBars}; + + /** + * @brief BGRA8 slots only: write the pattern in B,G,R,A byte order (true, + * the default) or leave it in logical R,G,B,A order (false). + * + * THIS EXISTS TO BE SET FALSE ONCE, as the control that proves the BGRA + * cases are testing anything at all. + * + * The BGRA cases stage byte-swapped pixels and validate against a + * reference built from the logical colours, so they PASS when the pipeline + * honours the VkFormat. But they would also pass if the swap never + * happened AND the format were ignored -- two mistakes cancelling. Setting + * this false breaks the symmetry deliberately: a BGRA image holding + * unswapped RGBA bytes MUST come out red/blue-exchanged, so the case must + * FAIL against the reference. If it passes, the format is not reaching the + * GPU and every other BGRA result here is vacuous. + */ + bool bgraStageSwap{true}; + // For validation bool generateTestPattern{true}; // Generate test pattern for inputs bool validateOutput{true}; // Validate output against reference @@ -130,6 +158,25 @@ struct TestCaseConfig { float tolerance{0.02f}; // Validation tolerance (0.0-1.0) uint32_t filterFlags{0}; // VulkanFilterYuvCompute::FilterFlags + + /** + * @brief The output is REQUIRED to disagree with the reference. + * + * Distinct from kKnownFilterDefects, which records a defect somebody + * intends to fix. This records a property that is correct and permanent: + * a configuration that CANNOT be right, asserted to be wrong so that the + * claim is checked by the machine instead of resting in a comment. + * + * The case it exists for: a BGRA8 source whose bytes are staged in logical + * R,G,B,A order rather than in the B,G,R,A order the format names (see + * TestIOSlot::bgraStageSwap). The picture in memory is then red/blue + * exchanged relative to the reference, so a pipeline that honours the + * VkFormat cannot reproduce it. If that case ever MATCHES the reference, + * the format is not reaching the GPU and every other component-order + * result is vacuous -- which is exactly when a silent pass would be most + * misleading. + */ + bool expectReferenceMismatch{false}; }; /** @@ -214,6 +261,16 @@ class FilterTestApp { // Command buffer pool for test execution VkCommandPool m_commandPool{VK_NULL_HANDLE}; + + // Optimal-tiled images this run has host-uploaded a pattern into, via + // uploadPatternToOptimalImage(). Those images are left in + // VK_IMAGE_LAYOUT_GENERAL with their contents live, so runTest()'s + // to-GENERAL barrier must name GENERAL as oldLayout for them -- naming + // the create-info UNDEFINED would be legal and would DISCARD the pattern + // that was just uploaded. Cleared per test: VkImage handles are + // recycled once a test's resources are released, so a stale handle here + // would alias an unrelated image in a later test. + std::set m_hostUploadedOptimal; /** * @brief Create test input resource @@ -242,15 +299,24 @@ class FilterTestApp { VkBufferUsageFlags usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT); /** - * @brief Generate test pattern in image/buffer + * @brief Generate a test pattern in an image or buffer. + * + * Both out-parameters describe the picture this call staged, never a second + * independently generated one: a reference built from a regenerated pattern + * silently tests nothing. + * + * @param pOutPatternData Receives the bytes AS STAGED, which is what the + * upload path copies into the resource. + * @param pOutReferencePattern Receives the same picture in logical R,G,B,A order, + * which is what the CPU reference generators expect. + * The two differ only for a BGRA8 slot that stages its + * bytes in the format's own order. */ - // pOutPatternData, when non-null, receives the exact bytes written into the input - // resource. The reference model must be derived from those bytes and not from a - // second, independently generated pattern, or the comparison silently tests nothing. VkResult generateTestPattern(const TestIOSlot& slot, VkSharedBaseObj& image, VkSharedBaseObj& buffer, - std::vector* pOutPatternData = nullptr); + std::vector* pOutPatternData = nullptr, + std::vector* pOutReferencePattern = nullptr); /** * @brief Validate output against expected result @@ -262,9 +328,16 @@ class FilterTestApp { const std::vector& referenceData); /** - * @brief Copy image to staging buffer for CPU readback + * @brief Read an OPTIMAL-tiled image back into a host-visible buffer. + * + * Takes the slot because the copy needs the format's plane geometry; + * the image itself does not carry it in a form this harness can use. + * The buffer comes back TIGHTLY PACKED -- plane after plane, no row + * padding -- so it lines up with calculateImageSize() and with the CPU + * reference without any further repacking. */ - VkResult copyImageToStagingBuffer(VkSharedBaseObj& image, + VkResult copyImageToStagingBuffer(const TestIOSlot& slot, + VkSharedBaseObj& image, VkSharedBaseObj& stagingBuffer); /** diff --git a/common/libs/tests/inc/TestCases.h b/common/libs/tests/inc/TestCases.h index 9af67b85..cd29fa0c 100644 --- a/common/libs/tests/inc/TestCases.h +++ b/common/libs/tests/inc/TestCases.h @@ -55,6 +55,7 @@ namespace TestCases { // 4:2:0 formats TestCaseConfig TC001_RGBA_to_NV12(); // 8-bit 4:2:0 2-plane + TestCaseConfig TC002_RGBA_to_P010(); // 10-bit 4:2:0 2-plane TestCaseConfig TC003_RGBA_to_P012(); // 12-bit 4:2:0 2-plane TestCaseConfig TC004_RGBA_to_I420(); // 8-bit 4:2:0 3-plane @@ -141,6 +142,20 @@ TestCaseConfig TC015b_P212_to_RGBA(); // 12-bit 4:2:2 (P212) to RGBA TestCaseConfig TC084_RGBA_to_P212_Linear(); // 12-bit 4:2:2 LINEAR (TRV repro shape) TestCaseConfig TC085_RGBA_to_P210_Linear(); // 10-bit 4:2:2 LINEAR (control) +// ============================================================================= +// RGBA/BGRA component-order tests +// ============================================================================= +// +// One picture of saturated primaries, staged in each format's own byte order and +// validated against one shared reference built from the logical colours. TC133 is +// the control that licenses reading the other two as results. See the block +// comment in TestCases.cpp. + +TestCaseConfig TC130_RGBA_to_NV12_PurePrimaries_Storage(); // must MATCH +TestCaseConfig TC132_BGRA_to_NV12_PurePrimaries_Storage(); // must MATCH +TestCaseConfig TC133_BGRA_NoSwapControl_MustDiffer(); // must DIFFER + + // ============================================================================= // Buffer I/O Tests // ============================================================================= @@ -163,6 +178,13 @@ TestCaseConfig TC076_P010Buffer_to_RGBABuffer(); // ============================================================================= TestCaseConfig TC080_RGBA_to_NV12_Linear(); +// Linear-tiled colour variants. Linear so the output can actually be read +// back and compared -- the optimal-tiled TC02x/TC03x cases cannot be. +TestCaseConfig TC092_RGBA_to_NV12_Linear_LimitedRange(); +TestCaseConfig TC093_RGBA_to_NV12_Linear_BT601(); +TestCaseConfig TC094_RGBA_to_NV12_Linear_BT2020(); +TestCaseConfig TC095_RGBA_to_P010_Linear_LimitedRange(); +TestCaseConfig TC096_RGBA_to_NV12_Linear_BT709_FullRange(); TestCaseConfig TC081_RGBA_to_P010_Linear(); TestCaseConfig TC082_Linear_NV12_to_Optimal_NV12(); TestCaseConfig TC083_Optimal_NV12_to_Linear_NV12(); @@ -183,6 +205,8 @@ TestCaseConfig TC101_Unaligned_Resolution_1922x1082(); TestCaseConfig TC102_4K_Resolution_3840x2160(); TestCaseConfig TC103_8K_Resolution_7680x4320(); TestCaseConfig TC104_Minimum_Resolution_2x2(); +TestCaseConfig TC105_Odd_Resolution_65x33_YUV444(); +TestCaseConfig TC106_Odd_Height_66x33_NV16(); // ============================================================================= // Transfer Operation Tests (Pre/Post Transfer scenarios) diff --git a/common/libs/tests/linux_dmabuf_import/CMakeLists.txt b/common/libs/tests/linux_dmabuf_import/CMakeLists.txt new file mode 100644 index 00000000..b534783f --- /dev/null +++ b/common/libs/tests/linux_dmabuf_import/CMakeLists.txt @@ -0,0 +1,149 @@ +# Copyright 2024-2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(linux_dmabuf_import_test LANGUAGES CXX) + +# Only build on Linux/Unix (dma-buf is a Linux kernel object) +if(NOT UNIX OR APPLE) + message(STATUS "linux_dmabuf_import_test: Skipping (Linux-only)") + return() +endif() + +set(CMAKE_CXX_STANDARD 17) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +# Test application source files +set(TEST_SOURCES + src/main.cpp + src/LinuxDmaBufImportTest.cpp +) + +# Test application header files +set(TEST_HEADERS + inc/LinuxDmaBufImportTest.h +) + +# VkCodecUtils source files (from parent directory) - same set the sibling +# drm_format_mod and win32_opaque_import tests link, so VulkanDeviceContext's +# dependencies are satisfied identically. +set(VKCODECUTILS_DIR "${CMAKE_CURRENT_SOURCE_DIR}/../../VkCodecUtils") +set(VKCODECUTILS_SOURCES + ${VKCODECUTILS_DIR}/VulkanDeviceContext.cpp + ${VKCODECUTILS_DIR}/VkEncoderStdioLatch.cpp + ${VKCODECUTILS_DIR}/VulkanDeviceMemoryImpl.cpp + ${VKCODECUTILS_DIR}/VkImageResource.cpp + ${VKCODECUTILS_DIR}/VkBufferResource.cpp + ${VKCODECUTILS_DIR}/VulkanShaderCompiler.cpp + ${VKCODECUTILS_DIR}/${VK_SHADER_COMPILER_BACKEND_SOURCE} + ${VKCODECUTILS_DIR}/VulkanComputePipeline.cpp + ${VKCODECUTILS_DIR}/VulkanDescriptorSetLayout.cpp + ${VKCODECUTILS_DIR}/VulkanSamplerYcbcrConversion.cpp + ${VKCODECUTILS_DIR}/VulkanCommandBufferPool.cpp + ${VKCODECUTILS_DIR}/VulkanCommandBuffersSet.cpp + ${VKCODECUTILS_DIR}/VulkanSemaphoreSet.cpp + ${VKCODECUTILS_DIR}/VulkanFenceSet.cpp + ${VKCODECUTILS_DIR}/Helpers.cpp + ${VK_DISPATCH_TABLE_SOURCE} + ${VK_DISPATCH_TABLE_HEADER} + ${VKCODECUTILS_DIR}/nvVkFormats.cpp +) + +# Create executable +add_executable(${PROJECT_NAME} + ${TEST_SOURCES} + ${TEST_HEADERS} + ${VKCODECUTILS_SOURCES} +) + +# The dispatch table is generated into the build tree, not checked in. +# Listing the generated source gives THIS .cpp a file-level dependency, +# but every other TU here reaches the generated HEADER through +# inc/LinuxDmaBufImportTest.h -> VkCodecUtils/VulkanDeviceContext.h, and +# nothing orders those objects after the generator without this. +add_dependencies(${PROJECT_NAME} GenerateDispatchTables) + +# Include directories +target_include_directories(${PROJECT_NAME} PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/inc + ${CMAKE_CURRENT_SOURCE_DIR}/../../ # For VkCodecUtils headers + ${CMAKE_CURRENT_SOURCE_DIR}/../../../include # For vulkan_interfaces.h + ${CMAKE_CURRENT_SOURCE_DIR}/../../../.. # For nvidia_utils + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so we only need headers. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +target_link_libraries(${PROJECT_NAME} PRIVATE ${VK_SHADER_COMPILER_LIBS}) + +# Platform-specific settings (Linux only at this point) +target_link_libraries(${PROJECT_NAME} PRIVATE + pthread + dl +) + +# Ensure POSIX functions are available (dup/close) +target_compile_definitions(${PROJECT_NAME} PRIVATE _POSIX_C_SOURCE=200809L) + +# Compile definitions +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +# Installation +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# NOTE ON CTest SEMANTICS: this executable returns 0 only when the +# second-device import AND the content round-trip both succeeded. Exit code 2 +# means "could not run" (no GPU, missing extension, nothing proved) and is +# deliberately a CTest FAILURE, not a skip - a host that cannot run this test +# must not report the device-independence claim as verified. +enable_testing() +add_test(NAME LinuxDmaBufSecondDeviceImport + COMMAND ${PROJECT_NAME}) +add_test(NAME LinuxDmaBufSecondDeviceImportLinear + COMMAND ${PROJECT_NAME} --linear-only --verbose) +add_test(NAME LinuxDmaBufSecondDeviceImportRGBA + COMMAND ${PROJECT_NAME} --format rgba8 --no-video-usage) + +# Exit 2 ("COULD NOT RUN - THIS IS NOT A PASS") is mapped to CTest SKIP rather +# than to FAIL. That does not weaken the note above: a CTest SKIP is not a +# pass, so the device-independence claim is still not reported as verified on +# a host that cannot run this. It only stops a GPU-less CI runner from being +# permanently red, which is the state that teaches people to ignore the job. +# Exit 1 -- the import ran and the round-trip did not hold -- remains a FAIL. +set_tests_properties(LinuxDmaBufSecondDeviceImport + LinuxDmaBufSecondDeviceImportLinear + LinuxDmaBufSecondDeviceImportRGBA PROPERTIES + SKIP_RETURN_CODE 2 + LABELS "gpu") + +# Print status +message(STATUS "linux_dmabuf_import_test: Configured for Linux") diff --git a/common/libs/tests/linux_dmabuf_import/inc/LinuxDmaBufImportTest.h b/common/libs/tests/linux_dmabuf_import/inc/LinuxDmaBufImportTest.h new file mode 100644 index 00000000..9c120d7e --- /dev/null +++ b/common/libs/tests/linux_dmabuf_import/inc/LinuxDmaBufImportTest.h @@ -0,0 +1,400 @@ +/* + * Copyright 2024-2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +//============================================================================= +// Linux dma-buf SECOND-DEVICE import test. +// +// What this proves (and nothing else does today): +// +// An image allocated and exported by one VkDevice can be imported, bound +// and READ by a DIFFERENT VkDevice on the SAME VkPhysicalDevice. +// +// That claim is the load-bearing premise of the whole "library creates its +// own device" direction: a different logical device on the same physical GPU +// has to work, or the library cannot own its VkDevice and still import a +// producer's buffers. +// +// The Windows sibling creates the second device but stops at +// vkBindImageMemory - it never touches the memory. This test goes further: +// it writes a known pattern through device A, hands the dma-buf to device B, +// and reads it back through device B. "Import returned VK_SUCCESS" is not +// accepted as proof that the memory is usable. +// +// Two arms run per modifier, through the SAME code path, differing only in +// which VkDevice performs the import: +// +// "second-device" import on VkDevice B <- the claim under test +// "same-device" import on VkDevice A <- control +// +// The control exists so a failure is attributable. second-device FAIL + +// same-device PASS is a device-independence defect. Both FAIL is an +// export/environment defect and says nothing about device independence. +// +// The IMPORT image's usage is separable from the exporter's, because that is +// the only half a Chromium-shaped consumer controls: the buffer is exported by +// GBM/Ozone and the VulkanVideoEncodeAccelerator picks imageUsage for its own +// VkImage from a vkGetPhysicalDeviceImageFormatProperties2 answer. Use +// --import-no-video-usage (or the raw --import-usage/--import-flags) to hold +// the exporter fixed and move only that half. +// +// WHAT THIS TEST IS LOOKING FOR, on an NV12 buffer: +// +// a block-linear modifier imported WITHOUT +// VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR can lose the ENTIRE chroma +// plane. Luma compares clean, the first bad byte is at the chroma plane +// offset, and CbCr reads back all zeros. Every Vulkan call returns +// VK_SUCCESS, so only a content compare catches it. +// +// It is the IMPORT image's usage that decides, and one bit of it. Holding the +// modifier and the plane layouts fixed: +// +// export usage import usage result +// ---------------------------------------------------------- +// 0x4007 (encode) 0x4007 / 0x4001 (encode) PASS +// 0x4007 (encode) 0x0007 / 0x0001 (no encode) CHROMA LOST +// 0x0007 (no encode) 0x4001 (encode) PASS +// 0x0007 (no encode) 0x0007 (no encode) CHROMA LOST +// +// The create flags are not implicated: every import-flag combination +// gives the same verdict for a given usage. Not every modifier is +// affected, and LINEAR is clean without the bit -- it cannot carry it, +// since vkCreateImage returns VK_ERROR_FORMAT_NOT_SUPPORTED for LINEAR +// NV12 + VIDEO_ENCODE_SRC. +// +// Deliberately NOT covered: +// - 3-plane I420 (VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM). That path hangs +// the GPU on hardware; this test must not +// be the thing that wedges a test machine. +// - Cross-PHYSICAL-device import. The plan (section 3.2) says that is a +// hard refusal by design (DEVICE_MISMATCH), so there is no claim to test. +// - Semaphore/fence handle exchange. Synchronisation here is host-side +// (fence wait + vkQueueWaitIdle) on purpose, so that a failure is a +// memory-import failure and not a sync-import failure. +//============================================================================= + +#pragma once + +#if defined(__linux__) + +#include + +#include +#include +#include + +#include "VkCodecUtils/VulkanDeviceContext.h" + +namespace linux_dmabuf_import_test { + +#ifndef DRM_FORMAT_MOD_LINEAR +#define DRM_FORMAT_MOD_LINEAR 0ULL +#endif +#ifndef DRM_FORMAT_MOD_INVALID +#define DRM_FORMAT_MOD_INVALID ((1ULL << 56) - 1) +#endif + +//============================================================================= +// Pipeline steps. Every failure is reported as "the run stopped at ", +// so a red result names the operation that broke rather than a bare VkResult. +//============================================================================= + +enum class Step { + NotStarted = 0, + ExportDeviceInit, // VulkanDeviceContext: instance + phys dev + device A + ImportDeviceCreate, // vkCreateDevice for device B on the same phys dev + ImportDeviceDispatch, // vkGetDeviceProcAddr for every entry point B needs + ModifierEnumerate, // vkGetPhysicalDeviceFormatProperties2 modifier list + ModifierCapability, // vkGetPhysicalDeviceImageFormatProperties2 (exportable/importable) + ExportImageCreate, // vkCreateImage on A, DRM modifier list + external memory + ExportModifierReadback, // vkGetImageDrmFormatModifierPropertiesEXT on A + ExportMemoryAllocate, // vkAllocateMemory on A, export + dedicated + ExportBindMemory, // vkBindImageMemory on A + ExportPlaneLayouts, // vkGetImageSubresourceLayout, MEMORY_PLANE_i aspects + ExportFd, // vkGetMemoryFdKHR, handleType = DMA_BUF + UploadPattern, // staging buffer -> image on A + ReleaseToForeign, // A's queue family -> VK_QUEUE_FAMILY_FOREIGN_EXT + ImportImageCreate, // vkCreateImage on B, explicit modifier + plane layouts + ImportFdMemoryTypes, // vkGetMemoryFdPropertiesKHR on B for THIS fd + ImportMemoryTypeSelect, // intersect fd mask with B's image requirements + ImportMemoryAllocate, // vkAllocateMemory on B with VkImportMemoryFdInfoKHR + ImportBindMemory, // vkBindImageMemory on B + AcquireFromForeign, // VK_QUEUE_FAMILY_FOREIGN_EXT -> B's queue family + Readback, // image -> staging buffer on B + ContentCompare, // byte compare against what A wrote + Complete +}; + +const char* stepName(Step s); +const char* vkResultName(VkResult r); + +//============================================================================= +// Configuration +//============================================================================= + +struct TestConfig { + bool verbose{false}; + bool validation{false}; + uint32_t width{1920}; + uint32_t height{1080}; + VkFormat format{VK_FORMAT_G8_B8R8_2PLANE_420_UNORM}; // NV12 + bool videoUsage{true}; // VIDEO_ENCODE_SRC + the flags that go with it + // Import-side override for videoUsage. -1 follows videoUsage; 0 or 1 + // forces the IMPORT image's usage/flags independently of the exporter's. + // + // This exists because Chromium's VulkanVideoEncodeAccelerator is only + // ever the importer: the buffer is exported by GBM/Ozone, and the VEA + // chooses imageUsage for its own VkImage from + // VveaModifierSupportsEncodeSrc(). Moving both halves together cannot + // tell us whether a chroma loss is attributable to the half Chromium + // controls. + int importVideoUsage{-1}; + // Raw overrides, applied after importVideoUsage. UINT32_MAX means + // "not set". These exist to separate the usage bit from the create + // flags, which otherwise move together. + uint32_t importUsageRaw{UINT32_MAX}; + uint32_t importFlagsRaw{UINT32_MAX}; + bool linearOnly{false}; // only DRM_FORMAT_MOD_LINEAR + bool contentCheck{true}; // write on A, read back on B, compare + bool sameDeviceControl{true};// also run the single-device control arm + uint64_t onlyModifier{DRM_FORMAT_MOD_INVALID}; // restrict to one modifier +}; + +//============================================================================= +// Per-device function table. +// +// Device B is created with raw vkCreateDevice, so it has no VulkanDeviceContext +// behind it and needs its own dispatch. Device A gets an identical table loaded +// the same way, which is what lets both arms run the SAME code: if the two arms +// disagree, the difference is the VkDevice and not the code path. +//============================================================================= + +struct DeviceFns { + PFN_vkGetDeviceQueue GetDeviceQueue{nullptr}; + PFN_vkDestroyDevice DestroyDevice{nullptr}; + PFN_vkCreateImage CreateImage{nullptr}; + PFN_vkDestroyImage DestroyImage{nullptr}; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements{nullptr}; + PFN_vkGetImageSubresourceLayout GetImageSubresourceLayout{nullptr}; + PFN_vkAllocateMemory AllocateMemory{nullptr}; + PFN_vkFreeMemory FreeMemory{nullptr}; + PFN_vkBindImageMemory BindImageMemory{nullptr}; + PFN_vkCreateBuffer CreateBuffer{nullptr}; + PFN_vkDestroyBuffer DestroyBuffer{nullptr}; + PFN_vkGetBufferMemoryRequirements GetBufferMemoryRequirements{nullptr}; + PFN_vkBindBufferMemory BindBufferMemory{nullptr}; + PFN_vkMapMemory MapMemory{nullptr}; + PFN_vkUnmapMemory UnmapMemory{nullptr}; + PFN_vkFlushMappedMemoryRanges FlushMappedMemoryRanges{nullptr}; + PFN_vkInvalidateMappedMemoryRanges InvalidateMappedMemoryRanges{nullptr}; + PFN_vkCreateCommandPool CreateCommandPool{nullptr}; + PFN_vkDestroyCommandPool DestroyCommandPool{nullptr}; + PFN_vkAllocateCommandBuffers AllocateCommandBuffers{nullptr}; + PFN_vkFreeCommandBuffers FreeCommandBuffers{nullptr}; + PFN_vkBeginCommandBuffer BeginCommandBuffer{nullptr}; + PFN_vkEndCommandBuffer EndCommandBuffer{nullptr}; + PFN_vkCmdPipelineBarrier CmdPipelineBarrier{nullptr}; + PFN_vkCmdCopyBufferToImage CmdCopyBufferToImage{nullptr}; + PFN_vkCmdCopyImageToBuffer CmdCopyImageToBuffer{nullptr}; + PFN_vkQueueSubmit QueueSubmit{nullptr}; + PFN_vkQueueWaitIdle QueueWaitIdle{nullptr}; + PFN_vkDeviceWaitIdle DeviceWaitIdle{nullptr}; + PFN_vkCreateFence CreateFence{nullptr}; + PFN_vkDestroyFence DestroyFence{nullptr}; + PFN_vkWaitForFences WaitForFences{nullptr}; + PFN_vkGetMemoryFdKHR GetMemoryFdKHR{nullptr}; + PFN_vkGetMemoryFdPropertiesKHR GetMemoryFdPropertiesKHR{nullptr}; + PFN_vkGetImageDrmFormatModifierPropertiesEXT + GetImageDrmFormatModifierPropertiesEXT{nullptr}; + + // Loads every pointer above. On the first null it returns false and puts + // the entry-point name in missingOut - a missing extension shows up as a + // named symbol rather than as a segfault on first use. + bool loadAll(PFN_vkGetDeviceProcAddr gdpa, VkDevice device, + std::string& missingOut); +}; + +//============================================================================= +// A device participating in the test (A = exporter/control, B = importer). +//============================================================================= + +struct TestDevice { + std::string label; + VkDevice device{VK_NULL_HANDLE}; + VkQueue queue{VK_NULL_HANDLE}; + uint32_t queueFamily{UINT32_MAX}; + VkCommandPool cmdPool{VK_NULL_HANDLE}; + DeviceFns fn; + bool ownsDevice{false}; // true only for B; A belongs to VulkanDeviceContext +}; + +//============================================================================= +// Format description used for the content round-trip. +//============================================================================= + +struct PlaneDesc { + VkImageAspectFlagBits aspect; + uint32_t widthDiv; + uint32_t heightDiv; + uint32_t texelBytes; +}; + +struct FormatDesc { + bool known{false}; + std::vector planes; +}; + +FormatDesc describeFormat(VkFormat format); +const char* formatName(VkFormat format); + +//============================================================================= +// One modifier's capability, from the PHYSICAL-device-scoped query. The plan +// (section 3.2) leans on that query being physical-device scoped, which is +// exactly why a second logical device is expected to be able to import. +//============================================================================= + +struct ModifierCandidate { + uint64_t modifier{DRM_FORMAT_MOD_INVALID}; + uint32_t memoryPlaneCount{0}; + VkFormatFeatureFlags tilingFeatures{0}; + bool queried{false}; + VkResult queryResult{VK_SUCCESS}; + bool exportable{false}; + bool importable{false}; + bool usable{false}; // exportable && importable && image format supported + std::string note; +}; + +//============================================================================= +// One end-to-end export/import cycle. +//============================================================================= + +struct ArmResult { + std::string arm; // "second-device" | "same-device" + uint64_t modifier{DRM_FORMAT_MOD_INVALID}; + + Step lastStepStarted{Step::NotStarted}; + Step failedAt{Step::NotStarted}; + VkResult vkResult{VK_SUCCESS}; + std::string detail; + + // Evidence that the two devices really are two devices on one GPU. + uint64_t exportDeviceHandle{0}; + uint64_t importDeviceHandle{0}; + + VkDeviceSize exportAllocSize{0}; + uint32_t exportMemTypeIndex{UINT32_MAX}; + VkDeviceSize importMemReqSize{0}; + uint32_t importMemTypeBits{0}; + uint32_t fdMemTypeBits{0}; + uint32_t importMemTypeIndex{UINT32_MAX}; + uint64_t readbackModifier{DRM_FORMAT_MOD_INVALID}; + + bool importSucceeded{false}; // create + alloc + bind all OK on the import device + bool contentAttempted{false}; + bool contentVerified{false}; + uint64_t firstMismatchOffset{UINT64_MAX}; + uint32_t expectedByte{0}; + uint32_t actualByte{0}; + uint64_t comparedBytes{0}; + + bool ok() const { return failedAt == Step::NotStarted; } +}; + +//============================================================================= +// Overall verdict. Deliberately tri-state: "the test did not run" must never +// be reportable as "the claim holds". +//============================================================================= + +enum class Verdict { + ClaimVerified, // >= 1 modifier: second-device import AND content round-trip OK + ClaimFailed, // a second-device arm failed (or every arm did) + CouldNotRun // no GPU / no extension / nothing to test - NOT a pass +}; + +class LinuxDmaBufImportTest { +public: + LinuxDmaBufImportTest() = default; + ~LinuxDmaBufImportTest(); + + // Brings up device A, then device B. A non-VK_SUCCESS return means the + // test could not run at all; the caller must report CouldNotRun, not pass. + VkResult init(const TestConfig& config); + + // Runs every usable modifier x {second-device, same-device}. + std::vector run(); + + const std::vector& modifiers() const { return m_modifiers; } + const std::string& initFailureDetail() const { return m_initFailureDetail; } + Step initFailedAt() const { return m_initFailedAt; } + + void printResults(const std::vector& results) const; + Verdict verdict(const std::vector& results, std::string& reasonOut) const; + +private: + VkResult createImportDevice(); + VkResult setUpDevice(TestDevice& dev); // command pool + queue + void tearDownDevice(TestDevice& dev); + + VkResult enumerateModifiers(); // fills m_modifiers + + ArmResult runCycle(const char* armLabel, + TestDevice& importDev, + const ModifierCandidate& mod); + + // Records + submits a one-shot command buffer, waits on a fence. + VkResult submitOneShot(TestDevice& dev, + VkCommandBuffer cmd, + const char*& failedCall); + + uint32_t chooseMemoryType(uint32_t candidateMask, + VkMemoryPropertyFlags required) const; + + VkImageUsageFlags exportUsage() const; + VkImageCreateFlags exportFlags() const; + // The import image's usage/flags. Equal to the export pair unless + // TestConfig::importVideoUsage overrides it. + bool importVideoUsage() const; + VkImageUsageFlags importUsage() const; + VkImageCreateFlags importFlags() const; + + // Host-visible staging buffer helper. Returns the buffer + memory + mapped + // pointer; caller destroys via destroyStaging(). + struct Staging { + VkBuffer buffer{VK_NULL_HANDLE}; + VkDeviceMemory memory{VK_NULL_HANDLE}; + void* mapped{nullptr}; + VkDeviceSize size{0}; + bool coherent{false}; + }; + VkResult createStaging(TestDevice& dev, VkDeviceSize size, + VkBufferUsageFlags usage, Staging& out); + void destroyStaging(TestDevice& dev, Staging& s); + + TestConfig m_config; + VulkanDeviceContext m_vkDevCtx; // owns instance + device A + TestDevice m_devA; // export / control device + TestDevice m_devB; // second device under test + VkPhysicalDeviceMemoryProperties m_memProps{}; + std::vector m_modifiers; + FormatDesc m_formatDesc; + std::string m_initFailureDetail; + Step m_initFailedAt{Step::NotStarted}; + std::string m_deviceName; +}; + +} // namespace linux_dmabuf_import_test + +#endif // __linux__ diff --git a/common/libs/tests/linux_dmabuf_import/src/LinuxDmaBufImportTest.cpp b/common/libs/tests/linux_dmabuf_import/src/LinuxDmaBufImportTest.cpp new file mode 100644 index 00000000..1490fb3d --- /dev/null +++ b/common/libs/tests/linux_dmabuf_import/src/LinuxDmaBufImportTest.cpp @@ -0,0 +1,1762 @@ +/* + * Copyright 2024-2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#if defined(__linux__) + +#include "LinuxDmaBufImportTest.h" + +#include + +#include +#include +#include +#include +#include +#include + +namespace linux_dmabuf_import_test { + +//============================================================================= +// Naming helpers - a failure has to be readable without a debugger +//============================================================================= + +const char* stepName(Step s) { + switch (s) { + case Step::NotStarted: return "-"; + case Step::ExportDeviceInit: return "ExportDeviceInit"; + case Step::ImportDeviceCreate: return "ImportDeviceCreate"; + case Step::ImportDeviceDispatch: return "ImportDeviceDispatch"; + case Step::ModifierEnumerate: return "ModifierEnumerate"; + case Step::ModifierCapability: return "ModifierCapability"; + case Step::ExportImageCreate: return "ExportImageCreate"; + case Step::ExportModifierReadback: return "ExportModifierReadback"; + case Step::ExportMemoryAllocate: return "ExportMemoryAllocate"; + case Step::ExportBindMemory: return "ExportBindMemory"; + case Step::ExportPlaneLayouts: return "ExportPlaneLayouts"; + case Step::ExportFd: return "ExportFd"; + case Step::UploadPattern: return "UploadPattern"; + case Step::ReleaseToForeign: return "ReleaseToForeign"; + case Step::ImportImageCreate: return "ImportImageCreate"; + case Step::ImportFdMemoryTypes: return "ImportFdMemoryTypes"; + case Step::ImportMemoryTypeSelect: return "ImportMemoryTypeSelect"; + case Step::ImportMemoryAllocate: return "ImportMemoryAllocate"; + case Step::ImportBindMemory: return "ImportBindMemory"; + case Step::AcquireFromForeign: return "AcquireFromForeign"; + case Step::Readback: return "Readback"; + case Step::ContentCompare: return "ContentCompare"; + case Step::Complete: return "Complete"; + } + return "?"; +} + +const char* vkResultName(VkResult r) { + switch (r) { + case VK_SUCCESS: return "VK_SUCCESS"; + case VK_NOT_READY: return "VK_NOT_READY"; + case VK_TIMEOUT: return "VK_TIMEOUT"; + case VK_INCOMPLETE: return "VK_INCOMPLETE"; + case VK_ERROR_OUT_OF_HOST_MEMORY: return "VK_ERROR_OUT_OF_HOST_MEMORY"; + case VK_ERROR_OUT_OF_DEVICE_MEMORY: return "VK_ERROR_OUT_OF_DEVICE_MEMORY"; + case VK_ERROR_INITIALIZATION_FAILED: return "VK_ERROR_INITIALIZATION_FAILED"; + case VK_ERROR_DEVICE_LOST: return "VK_ERROR_DEVICE_LOST"; + case VK_ERROR_MEMORY_MAP_FAILED: return "VK_ERROR_MEMORY_MAP_FAILED"; + case VK_ERROR_LAYER_NOT_PRESENT: return "VK_ERROR_LAYER_NOT_PRESENT"; + case VK_ERROR_EXTENSION_NOT_PRESENT: return "VK_ERROR_EXTENSION_NOT_PRESENT"; + case VK_ERROR_FEATURE_NOT_PRESENT: return "VK_ERROR_FEATURE_NOT_PRESENT"; + case VK_ERROR_INCOMPATIBLE_DRIVER: return "VK_ERROR_INCOMPATIBLE_DRIVER"; + case VK_ERROR_TOO_MANY_OBJECTS: return "VK_ERROR_TOO_MANY_OBJECTS"; + case VK_ERROR_FORMAT_NOT_SUPPORTED: return "VK_ERROR_FORMAT_NOT_SUPPORTED"; + case VK_ERROR_FRAGMENTED_POOL: return "VK_ERROR_FRAGMENTED_POOL"; + case VK_ERROR_UNKNOWN: return "VK_ERROR_UNKNOWN"; + case VK_ERROR_OUT_OF_POOL_MEMORY: return "VK_ERROR_OUT_OF_POOL_MEMORY"; + case VK_ERROR_INVALID_EXTERNAL_HANDLE: return "VK_ERROR_INVALID_EXTERNAL_HANDLE"; + case VK_ERROR_INVALID_DRM_FORMAT_MODIFIER_PLANE_LAYOUT_EXT: + return "VK_ERROR_INVALID_DRM_FORMAT_MODIFIER_PLANE_LAYOUT_EXT"; + default: { + static thread_local char buf[32]; + snprintf(buf, sizeof(buf), "VkResult(%d)", (int)r); + return buf; + } + } +} + +const char* formatName(VkFormat format) { + switch (format) { + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: return "NV12"; + case VK_FORMAT_G8_B8R8_2PLANE_422_UNORM: return "NV16"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16: return "P010"; + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16: return "P012"; + case VK_FORMAT_R8G8B8A8_UNORM: return "RGBA8"; + case VK_FORMAT_B8G8R8A8_UNORM: return "BGRA8"; + default: { + static thread_local char buf[32]; + snprintf(buf, sizeof(buf), "VkFormat(%d)", (int)format); + return buf; + } + } +} + +// Plane geometry for the content round-trip. +// +// Deliberately excludes VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM (I420): the +// device-independence plan's section 8.1 records that path hanging the GPU on +// hardware, and a test that wedges the machine it runs on is worse than no +// test. An unknown format is reported as "content check unavailable", never +// silently treated as verified. +FormatDesc describeFormat(VkFormat format) { + FormatDesc d; + switch (format) { + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: + d.known = true; + d.planes = {{VK_IMAGE_ASPECT_PLANE_0_BIT, 1, 1, 1}, + {VK_IMAGE_ASPECT_PLANE_1_BIT, 2, 2, 2}}; + break; + case VK_FORMAT_G8_B8R8_2PLANE_422_UNORM: + d.known = true; + d.planes = {{VK_IMAGE_ASPECT_PLANE_0_BIT, 1, 1, 1}, + {VK_IMAGE_ASPECT_PLANE_1_BIT, 2, 1, 2}}; + break; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16: + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16: + d.known = true; + d.planes = {{VK_IMAGE_ASPECT_PLANE_0_BIT, 1, 1, 2}, + {VK_IMAGE_ASPECT_PLANE_1_BIT, 2, 2, 4}}; + break; + case VK_FORMAT_R8G8B8A8_UNORM: + case VK_FORMAT_B8G8R8A8_UNORM: + d.known = true; + d.planes = {{VK_IMAGE_ASPECT_COLOR_BIT, 1, 1, 4}}; + break; + default: + break; + } + return d; +} + +static std::string modifierToString(uint64_t mod) { + if (mod == DRM_FORMAT_MOD_LINEAR) return "LINEAR"; + if (mod == DRM_FORMAT_MOD_INVALID) return "INVALID"; + std::ostringstream os; + os << "0x" << std::hex << mod; + return os.str(); +} + +// Deterministic, position-dependent, and not a constant: a copy that never ran +// cannot pass the comparison by accident. +static inline uint8_t patternByte(uint64_t i) { + return static_cast(((i * 131u) + 17u) ^ (i >> 8)); +} + +//============================================================================= +// DeviceFns +//============================================================================= + +bool DeviceFns::loadAll(PFN_vkGetDeviceProcAddr gdpa, VkDevice device, + std::string& missingOut) +{ + missingOut.clear(); + if ((gdpa == nullptr) || (device == VK_NULL_HANDLE)) { + missingOut = "vkGetDeviceProcAddr"; + return false; + } + + bool ok = true; + auto get = [&](const char* name) -> PFN_vkVoidFunction { + PFN_vkVoidFunction p = gdpa(device, name); + if ((p == nullptr) && ok) { + ok = false; + missingOut = name; + } + return p; + }; + + GetDeviceQueue = (PFN_vkGetDeviceQueue)get("vkGetDeviceQueue"); + DestroyDevice = (PFN_vkDestroyDevice)get("vkDestroyDevice"); + CreateImage = (PFN_vkCreateImage)get("vkCreateImage"); + DestroyImage = (PFN_vkDestroyImage)get("vkDestroyImage"); + GetImageMemoryRequirements = (PFN_vkGetImageMemoryRequirements)get("vkGetImageMemoryRequirements"); + GetImageSubresourceLayout = (PFN_vkGetImageSubresourceLayout)get("vkGetImageSubresourceLayout"); + AllocateMemory = (PFN_vkAllocateMemory)get("vkAllocateMemory"); + FreeMemory = (PFN_vkFreeMemory)get("vkFreeMemory"); + BindImageMemory = (PFN_vkBindImageMemory)get("vkBindImageMemory"); + CreateBuffer = (PFN_vkCreateBuffer)get("vkCreateBuffer"); + DestroyBuffer = (PFN_vkDestroyBuffer)get("vkDestroyBuffer"); + GetBufferMemoryRequirements = (PFN_vkGetBufferMemoryRequirements)get("vkGetBufferMemoryRequirements"); + BindBufferMemory = (PFN_vkBindBufferMemory)get("vkBindBufferMemory"); + MapMemory = (PFN_vkMapMemory)get("vkMapMemory"); + UnmapMemory = (PFN_vkUnmapMemory)get("vkUnmapMemory"); + FlushMappedMemoryRanges = (PFN_vkFlushMappedMemoryRanges)get("vkFlushMappedMemoryRanges"); + InvalidateMappedMemoryRanges= (PFN_vkInvalidateMappedMemoryRanges)get("vkInvalidateMappedMemoryRanges"); + CreateCommandPool = (PFN_vkCreateCommandPool)get("vkCreateCommandPool"); + DestroyCommandPool = (PFN_vkDestroyCommandPool)get("vkDestroyCommandPool"); + AllocateCommandBuffers = (PFN_vkAllocateCommandBuffers)get("vkAllocateCommandBuffers"); + FreeCommandBuffers = (PFN_vkFreeCommandBuffers)get("vkFreeCommandBuffers"); + BeginCommandBuffer = (PFN_vkBeginCommandBuffer)get("vkBeginCommandBuffer"); + EndCommandBuffer = (PFN_vkEndCommandBuffer)get("vkEndCommandBuffer"); + CmdPipelineBarrier = (PFN_vkCmdPipelineBarrier)get("vkCmdPipelineBarrier"); + CmdCopyBufferToImage = (PFN_vkCmdCopyBufferToImage)get("vkCmdCopyBufferToImage"); + CmdCopyImageToBuffer = (PFN_vkCmdCopyImageToBuffer)get("vkCmdCopyImageToBuffer"); + QueueSubmit = (PFN_vkQueueSubmit)get("vkQueueSubmit"); + QueueWaitIdle = (PFN_vkQueueWaitIdle)get("vkQueueWaitIdle"); + DeviceWaitIdle = (PFN_vkDeviceWaitIdle)get("vkDeviceWaitIdle"); + CreateFence = (PFN_vkCreateFence)get("vkCreateFence"); + DestroyFence = (PFN_vkDestroyFence)get("vkDestroyFence"); + WaitForFences = (PFN_vkWaitForFences)get("vkWaitForFences"); + + // These three are the extension surface the import path is built on. A + // null here means the extension was not ENABLED on this device (not merely + // absent from the physical device), which is precisely the failure mode + // the plan's section 3.2 calls out as EXTENSION_MISSING. + GetMemoryFdKHR = (PFN_vkGetMemoryFdKHR)get("vkGetMemoryFdKHR"); + GetMemoryFdPropertiesKHR = (PFN_vkGetMemoryFdPropertiesKHR)get("vkGetMemoryFdPropertiesKHR"); + GetImageDrmFormatModifierPropertiesEXT = + (PFN_vkGetImageDrmFormatModifierPropertiesEXT)get("vkGetImageDrmFormatModifierPropertiesEXT"); + + return ok; +} + +//============================================================================= +// Lifecycle +//============================================================================= + +LinuxDmaBufImportTest::~LinuxDmaBufImportTest() { + tearDownDevice(m_devB); + tearDownDevice(m_devA); +} + +void LinuxDmaBufImportTest::tearDownDevice(TestDevice& dev) { + if (dev.device == VK_NULL_HANDLE) { + return; + } + if (dev.fn.DeviceWaitIdle != nullptr) { + dev.fn.DeviceWaitIdle(dev.device); + } + if ((dev.cmdPool != VK_NULL_HANDLE) && (dev.fn.DestroyCommandPool != nullptr)) { + dev.fn.DestroyCommandPool(dev.device, dev.cmdPool, nullptr); + dev.cmdPool = VK_NULL_HANDLE; + } + if (dev.ownsDevice && (dev.fn.DestroyDevice != nullptr)) { + dev.fn.DestroyDevice(dev.device, nullptr); + } + dev.device = VK_NULL_HANDLE; +} + +//============================================================================= +// init +//============================================================================= + +VkResult LinuxDmaBufImportTest::init(const TestConfig& config) { + m_config = config; + m_initFailedAt = Step::ExportDeviceInit; + + // 4:2:0 needs even dimensions; say so rather than failing obscurely later. + if ((m_config.width & 1u) || (m_config.height & 1u)) { + std::cout << "[INFO] rounding " << m_config.width << "x" << m_config.height + << " down to even dimensions for chroma subsampling\n"; + m_config.width &= ~1u; + m_config.height &= ~1u; + } + if ((m_config.width == 0) || (m_config.height == 0)) { + m_initFailureDetail = "width/height must be non-zero"; + return VK_ERROR_INITIALIZATION_FAILED; + } + + m_formatDesc = describeFormat(m_config.format); + + static const char* const instanceExtensions[] = { + VK_KHR_GET_PHYSICAL_DEVICE_PROPERTIES_2_EXTENSION_NAME, + VK_KHR_EXTERNAL_MEMORY_CAPABILITIES_EXTENSION_NAME, + nullptr + }; + m_vkDevCtx.AddReqInstanceExtensions(instanceExtensions); + + if (m_config.validation) { + static const char* const layers[] = { "VK_LAYER_KHRONOS_validation", nullptr }; + static const char* const debugExt[] = { VK_EXT_DEBUG_REPORT_EXTENSION_NAME, nullptr }; + m_vkDevCtx.AddReqInstanceLayers(layers); + m_vkDevCtx.AddReqInstanceExtensions(debugExt); + std::cout << "[INFO] Validation layers enabled\n"; + } + + // The first four are exactly the precondition set the library's own + // logical device must enable -- VK_KHR_external_memory_fd, + // VK_EXT_external_memory_dma_buf, VK_EXT_queue_family_foreign and + // VK_EXT_image_drm_format_modifier -- or a registration is refused with + // EXTENSION_MISSING. The library gates on the same set. + // Requiring them here means a host that cannot possibly support the design + // fails at init with a name, instead of failing an import later. + static const char* const requiredDeviceExtensions[] = { + VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME, + VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME, + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME, + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME, + VK_KHR_EXTERNAL_MEMORY_EXTENSION_NAME, + VK_KHR_IMAGE_FORMAT_LIST_EXTENSION_NAME, + VK_KHR_BIND_MEMORY_2_EXTENSION_NAME, + VK_KHR_SAMPLER_YCBCR_CONVERSION_EXTENSION_NAME, + VK_KHR_MAINTENANCE_1_EXTENSION_NAME, + VK_KHR_GET_MEMORY_REQUIREMENTS_2_EXTENSION_NAME, + VK_KHR_DEDICATED_ALLOCATION_EXTENSION_NAME, + nullptr + }; + m_vkDevCtx.AddReqDeviceExtensions(requiredDeviceExtensions, m_config.verbose); + + static const char* const videoExtensions[] = { + VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME, + VK_KHR_TIMELINE_SEMAPHORE_EXTENSION_NAME, + VK_KHR_VIDEO_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, + nullptr + }; + if (m_config.videoUsage) { + m_vkDevCtx.AddOptDeviceExtensions(videoExtensions, m_config.verbose); + } + + VkResult result = m_vkDevCtx.InitVulkanDevice("LinuxDmaBufImportTest", + VK_NULL_HANDLE, m_config.verbose); + if (result != VK_SUCCESS) { + m_initFailureDetail = "InitVulkanDevice failed (no Vulkan loader/ICD?)"; + return result; + } + + if (m_config.validation) { + m_vkDevCtx.InitDebugReport(true, true); + } + + vk::DeviceUuidUtils deviceUuid; + result = m_vkDevCtx.InitPhysicalDevice( + -1, deviceUuid, + VK_QUEUE_TRANSFER_BIT | VK_QUEUE_COMPUTE_BIT, + nullptr, + 0, VK_VIDEO_CODEC_OPERATION_NONE_KHR, + 0, VK_VIDEO_CODEC_OPERATION_NONE_KHR); + if (result != VK_SUCCESS) { + m_initFailureDetail = + "InitPhysicalDevice failed - no physical device offering the " + "required dma-buf/DRM-modifier extension set"; + return result; + } + + VkPhysicalDeviceProperties props{}; + m_vkDevCtx.GetPhysicalDeviceProperties(m_vkDevCtx.getPhysicalDevice(), &props); + m_deviceName = props.deviceName; + std::cout << "[INFO] Physical device: " << props.deviceName + << " driver=" << VK_VERSION_MAJOR(props.driverVersion) << "." + << VK_VERSION_MINOR(props.driverVersion) << "." + << VK_VERSION_PATCH(props.driverVersion) << "\n"; + + // Re-check enablement by name. AddReqDeviceExtensions influences device + // selection, but the point of the check is to be able to SAY which + // extension is missing when the run cannot proceed. + for (const char* const* p = requiredDeviceExtensions; *p != nullptr; ++p) { + if (m_vkDevCtx.FindRequiredDeviceExtension(*p) == nullptr) { + m_initFailureDetail = std::string("required device extension not available: ") + *p; + return VK_ERROR_EXTENSION_NOT_PRESENT; + } + } + + result = m_vkDevCtx.CreateVulkanDevice( + 0, // numDecodeQueues + 0, // numEncodeQueues + VK_VIDEO_CODEC_OPERATION_NONE_KHR, + true, // transfer queue + false, // graphics queue + false, // present queue + true); // compute queue + if (result != VK_SUCCESS) { + m_initFailureDetail = "CreateVulkanDevice (device A) failed"; + return result; + } + + m_devA.label = "A(export)"; + m_devA.device = m_vkDevCtx.getDevice(); + m_devA.queue = m_vkDevCtx.GetComputeQueue(); + m_devA.queueFamily = static_cast(m_vkDevCtx.GetComputeQueueFamilyIdx()); + if (m_devA.queue == VK_NULL_HANDLE) { + m_devA.queue = m_vkDevCtx.GetTransferQueue(); + m_devA.queueFamily = static_cast(m_vkDevCtx.GetTransferQueueFamilyIdx()); + } + if ((m_devA.queue == VK_NULL_HANDLE) || (m_devA.queueFamily == UINT32_MAX)) { + m_initFailureDetail = "device A has no compute or transfer queue"; + return VK_ERROR_INITIALIZATION_FAILED; + } + m_devA.ownsDevice = false; // VulkanDeviceContext owns it + + std::string missing; + if (!m_devA.fn.loadAll(m_vkDevCtx.GetDeviceProcAddr, m_devA.device, missing)) { + m_initFailedAt = Step::ExportDeviceInit; + m_initFailureDetail = "device A is missing entry point " + missing; + return VK_ERROR_EXTENSION_NOT_PRESENT; + } + + m_vkDevCtx.GetPhysicalDeviceMemoryProperties(m_vkDevCtx.getPhysicalDevice(), &m_memProps); + + m_initFailedAt = Step::ImportDeviceCreate; + result = createImportDevice(); + if (result != VK_SUCCESS) { + return result; + } + + m_initFailedAt = Step::ExportDeviceInit; + result = setUpDevice(m_devA); + if (result != VK_SUCCESS) { + m_initFailureDetail = "command pool creation failed on device A"; + return result; + } + m_initFailedAt = Step::ImportDeviceCreate; + result = setUpDevice(m_devB); + if (result != VK_SUCCESS) { + m_initFailureDetail = "command pool creation failed on device B"; + return result; + } + + // The whole test is meaningless if these are the same VkDevice. + if (m_devA.device == m_devB.device) { + m_initFailureDetail = "device A and device B are the same VkDevice handle"; + return VK_ERROR_INITIALIZATION_FAILED; + } + + std::cout << "[INFO] VkDevice A = 0x" << std::hex + << (uint64_t)(uintptr_t)m_devA.device + << " VkDevice B = 0x" << (uint64_t)(uintptr_t)m_devB.device + << std::dec << " (one VkPhysicalDevice, queue family " + << m_devA.queueFamily << ")\n"; + + m_initFailedAt = Step::ModifierEnumerate; + result = enumerateModifiers(); + if (result != VK_SUCCESS) { + return result; + } + + m_initFailedAt = Step::NotStarted; + return VK_SUCCESS; +} + +//============================================================================= +// Second VkDevice on the SAME VkPhysicalDevice. +// +// This is the Linux counterpart of Win32OpaqueImportTest::createImportDevice() +// (win32_opaque_import/src/Win32OpaqueImportTest.cpp:172-176). Same idea, and +// the same physical device by construction: m_vkDevCtx.getPhysicalDevice() is +// used verbatim, so a "different GPU" confound is impossible. +//============================================================================= + +VkResult LinuxDmaBufImportTest::createImportDevice() { + VkPhysicalDevice physDev = m_vkDevCtx.getPhysicalDevice(); + + const char* wanted[] = { + VK_KHR_EXTERNAL_MEMORY_EXTENSION_NAME, + VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME, + VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME, + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME, + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME, + VK_KHR_IMAGE_FORMAT_LIST_EXTENSION_NAME, + VK_KHR_BIND_MEMORY_2_EXTENSION_NAME, + VK_KHR_SAMPLER_YCBCR_CONVERSION_EXTENSION_NAME, + VK_KHR_MAINTENANCE_1_EXTENSION_NAME, + VK_KHR_GET_MEMORY_REQUIREMENTS_2_EXTENSION_NAME, + VK_KHR_DEDICATED_ALLOCATION_EXTENSION_NAME, + VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME, + VK_KHR_TIMELINE_SEMAPHORE_EXTENSION_NAME, + VK_KHR_VIDEO_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, + }; + const uint32_t wantedCount = sizeof(wanted) / sizeof(wanted[0]); + + // Everything from this index on is only wanted when the image carries video + // usage, and is optional even then. Everything BEFORE it is the import + // contract itself and its absence is fatal. Index 11 is + // VK_KHR_synchronization2 - keep this in step with the array above. + const uint32_t firstVideoIdx = 11; + + uint32_t availCount = 0; + m_vkDevCtx.EnumerateDeviceExtensionProperties(physDev, nullptr, &availCount, nullptr); + std::vector avail(availCount); + if (availCount != 0) { + m_vkDevCtx.EnumerateDeviceExtensionProperties(physDev, nullptr, &availCount, avail.data()); + } + + auto haveExt = [&](const char* name) { + for (const auto& a : avail) { + if (strcmp(name, a.extensionName) == 0) return true; + } + return false; + }; + + std::vector enabled; + for (uint32_t i = 0; i < wantedCount; ++i) { + if ((i >= firstVideoIdx) && !m_config.videoUsage) { + continue; + } + if (haveExt(wanted[i])) { + enabled.push_back(wanted[i]); + } else if (i < firstVideoIdx) { + // Non-negotiable: without these the import cannot be attempted at all. + m_initFailureDetail = + std::string("device B cannot enable required extension: ") + wanted[i]; + return VK_ERROR_EXTENSION_NOT_PRESENT; + } + } + const bool haveVideoMaintenance1 = + m_config.videoUsage && haveExt(VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME); + const bool haveSync2 = haveExt(VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME) && m_config.videoUsage; + const bool haveTimeline = haveExt(VK_KHR_TIMELINE_SEMAPHORE_EXTENSION_NAME) && m_config.videoUsage; + + // Which of the features the physical device actually offers. + VkPhysicalDeviceVideoMaintenance1FeaturesKHR vm1F{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VIDEO_MAINTENANCE_1_FEATURES_KHR}; + VkPhysicalDeviceSynchronization2Features sync2F{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SYNCHRONIZATION_2_FEATURES}; + VkPhysicalDeviceTimelineSemaphoreFeatures timelineF{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_TIMELINE_SEMAPHORE_FEATURES}; + VkPhysicalDeviceSamplerYcbcrConversionFeatures ycbcrF{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_SAMPLER_YCBCR_CONVERSION_FEATURES}; + + ycbcrF.pNext = &timelineF; + timelineF.pNext = &sync2F; + sync2F.pNext = haveVideoMaintenance1 ? (void*)&vm1F : nullptr; + + VkPhysicalDeviceFeatures2 query{VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2}; + query.pNext = &ycbcrF; + m_vkDevCtx.GetPhysicalDeviceFeatures2(physDev, &query); + + // Rebuild the chain from scratch, enabling only what is both supported and + // backed by an enabled extension. VUID-VkDeviceCreateInfo-pNext-pNext. + void* featureChain = nullptr; + ycbcrF.pNext = nullptr; timelineF.pNext = nullptr; + sync2F.pNext = nullptr; vm1F.pNext = nullptr; + + if (haveVideoMaintenance1 && (vm1F.videoMaintenance1 == VK_TRUE)) { + vm1F.pNext = featureChain; + featureChain = &vm1F; + } + if (haveSync2 && (sync2F.synchronization2 == VK_TRUE)) { + sync2F.pNext = featureChain; + featureChain = &sync2F; + } + if (haveTimeline && (timelineF.timelineSemaphore == VK_TRUE)) { + timelineF.pNext = featureChain; + featureChain = &timelineF; + } + if (ycbcrF.samplerYcbcrConversion == VK_TRUE) { + ycbcrF.pNext = featureChain; + featureChain = &ycbcrF; + } + + // VIDEO_PROFILE_INDEPENDENT is only legal with videoMaintenance1 enabled; + // if the device cannot give us that, drop video usage rather than create + // images that are spec-invalid on B but not on A (which would make the two + // arms incomparable - the exact confound this test exists to avoid). + if (m_config.videoUsage && + (!haveVideoMaintenance1 || (vm1F.videoMaintenance1 != VK_TRUE))) { + std::cout << "[WARN] videoMaintenance1 unavailable on this device: " + "dropping VIDEO_ENCODE_SRC / VIDEO_PROFILE_INDEPENDENT " + "from the tested image (reported, not silent)\n"; + m_config.videoUsage = false; + } + + VkPhysicalDeviceFeatures2 enabledFeatures{VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FEATURES_2}; + enabledFeatures.pNext = featureChain; // features{} left zeroed on purpose + + // Same queue family as A: the only difference between the two arms must be + // the VkDevice, not the queue family. + float priority = 1.0f; + VkDeviceQueueCreateInfo queueCI{VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO}; + queueCI.queueFamilyIndex = m_devA.queueFamily; + queueCI.queueCount = 1; + queueCI.pQueuePriorities = &priority; + + VkDeviceCreateInfo devCI{VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO}; + devCI.pNext = &enabledFeatures; + devCI.queueCreateInfoCount = 1; + devCI.pQueueCreateInfos = &queueCI; + devCI.enabledExtensionCount = static_cast(enabled.size()); + devCI.ppEnabledExtensionNames = enabled.data(); + + PFN_vkCreateDevice pfnCreateDevice = (PFN_vkCreateDevice) + m_vkDevCtx.GetInstanceProcAddr(m_vkDevCtx.getInstance(), "vkCreateDevice"); + if (pfnCreateDevice == nullptr) { + m_initFailureDetail = "vkCreateDevice entry point not found"; + return VK_ERROR_INITIALIZATION_FAILED; + } + + VkDevice device = VK_NULL_HANDLE; + VkResult result = pfnCreateDevice(physDev, &devCI, nullptr, &device); + if (result != VK_SUCCESS) { + m_initFailureDetail = std::string("vkCreateDevice for device B failed: ") + + vkResultName(result); + return result; + } + + m_devB.label = "B(import)"; + m_devB.device = device; + m_devB.queueFamily = m_devA.queueFamily; + m_devB.ownsDevice = true; + + std::string missing; + if (!m_devB.fn.loadAll(m_vkDevCtx.GetDeviceProcAddr, m_devB.device, missing)) { + m_initFailedAt = Step::ImportDeviceDispatch; + m_initFailureDetail = "device B is missing entry point " + missing; + return VK_ERROR_EXTENSION_NOT_PRESENT; + } + + m_devB.fn.GetDeviceQueue(m_devB.device, m_devB.queueFamily, 0, &m_devB.queue); + if (m_devB.queue == VK_NULL_HANDLE) { + m_initFailureDetail = "vkGetDeviceQueue returned NULL on device B"; + return VK_ERROR_INITIALIZATION_FAILED; + } + + std::cout << "[INFO] Import device created: a second VkDevice on the same " + "VkPhysicalDevice (" << enabled.size() << " extensions)\n"; + return VK_SUCCESS; +} + +VkResult LinuxDmaBufImportTest::setUpDevice(TestDevice& dev) { + VkCommandPoolCreateInfo poolCI{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; + poolCI.queueFamilyIndex = dev.queueFamily; + poolCI.flags = VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT; + return dev.fn.CreateCommandPool(dev.device, &poolCI, nullptr, &dev.cmdPool); +} + +//============================================================================= +// Image usage / flags under test +//============================================================================= + +VkImageUsageFlags LinuxDmaBufImportTest::exportUsage() const { + VkImageUsageFlags usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | + VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_SAMPLED_BIT; + if (m_config.videoUsage) { + usage |= VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + } + return usage; +} + +bool LinuxDmaBufImportTest::importVideoUsage() const { + return (m_config.importVideoUsage < 0) ? m_config.videoUsage + : (m_config.importVideoUsage != 0); +} + +VkImageUsageFlags LinuxDmaBufImportTest::importUsage() const { + VkImageUsageFlags usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | + VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_SAMPLED_BIT; + if (importVideoUsage()) { + usage |= VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + } + if (m_config.importUsageRaw != UINT32_MAX) { + usage = (VkImageUsageFlags)m_config.importUsageRaw; + } + return usage; +} + +VkImageCreateFlags LinuxDmaBufImportTest::importFlags() const { + if (m_config.importFlagsRaw != UINT32_MAX) { + return (VkImageCreateFlags)m_config.importFlagsRaw; + } + if (!importVideoUsage()) { + return 0; + } + return VK_IMAGE_CREATE_EXTENDED_USAGE_BIT | + VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT | + VK_IMAGE_CREATE_VIDEO_PROFILE_INDEPENDENT_BIT_KHR; +} + +VkImageCreateFlags LinuxDmaBufImportTest::exportFlags() const { + if (!m_config.videoUsage) { + return 0; + } + // The triple a video-usage exporter sets, matching the sibling + // drm_format_mod test (DrmFormatModTest.cpp:828-830): EXTENDED_USAGE + + // MUTABLE_FORMAT for per-plane views, VIDEO_PROFILE_INDEPENDENT so no + // VkVideoProfileListInfoKHR is needed at image-creation time. The + // library does not invent these - it copies whatever the exporter + // declared (imageCI.flags = desc.imageFlags, + // vulkan_video_encoder_ext.cpp:3274) - so the exporter is where they + // have to be right. + return VK_IMAGE_CREATE_EXTENDED_USAGE_BIT | + VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT | + VK_IMAGE_CREATE_VIDEO_PROFILE_INDEPENDENT_BIT_KHR; +} + +//============================================================================= +// Modifier enumeration + PHYSICAL-device-scoped capability query. +// +// Section 3.2 of the plan rests on this query being physical-device scoped +// ("The modifier-import capability query is physical-device-scoped"), which is +// the spec reason a second LOGICAL device is expected to be able to import. +// Asking it here means the run knows, before it starts, which modifiers the +// driver claims are both EXPORTABLE and IMPORTABLE - so an import failure on a +// modifier the driver advertised is unambiguously a defect and not a +// misconfiguration by the test. +//============================================================================= + +VkResult LinuxDmaBufImportTest::enumerateModifiers() { + VkPhysicalDevice physDev = m_vkDevCtx.getPhysicalDevice(); + + VkDrmFormatModifierPropertiesListEXT list{ + VK_STRUCTURE_TYPE_DRM_FORMAT_MODIFIER_PROPERTIES_LIST_EXT}; + VkFormatProperties2 fmtProps{VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_2}; + fmtProps.pNext = &list; + + m_vkDevCtx.GetPhysicalDeviceFormatProperties2(physDev, m_config.format, &fmtProps); + const uint32_t count = list.drmFormatModifierCount; + if (count == 0) { + m_initFailureDetail = + std::string("the driver reports no DRM format modifiers for ") + + formatName(m_config.format); + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + std::vector props(count); + list.pDrmFormatModifierProperties = props.data(); + m_vkDevCtx.GetPhysicalDeviceFormatProperties2(physDev, m_config.format, &fmtProps); + + VkFormat viewFormat = m_config.format; + const VkImageUsageFlags usage = exportUsage(); + const VkImageCreateFlags flags = exportFlags(); + + for (uint32_t i = 0; i < count; ++i) { + ModifierCandidate c; + c.modifier = props[i].drmFormatModifier; + c.memoryPlaneCount = props[i].drmFormatModifierPlaneCount; + c.tilingFeatures = props[i].drmFormatModifierTilingFeatures; + + if (m_config.linearOnly && (c.modifier != DRM_FORMAT_MOD_LINEAR)) { + c.note = "skipped (--linear-only)"; + m_modifiers.push_back(c); + continue; + } + if ((m_config.onlyModifier != DRM_FORMAT_MOD_INVALID) && + (c.modifier != m_config.onlyModifier)) { + c.note = "skipped (--modifier)"; + m_modifiers.push_back(c); + continue; + } + + VkPhysicalDeviceExternalImageFormatInfo extInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_IMAGE_FORMAT_INFO}; + extInfo.handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + VkPhysicalDeviceImageDrmFormatModifierInfoEXT modInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_DRM_FORMAT_MODIFIER_INFO_EXT}; + modInfo.pNext = &extInfo; + modInfo.drmFormatModifier = c.modifier; + modInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + + VkImageFormatListCreateInfo formatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + formatListCI.pNext = &modInfo; + formatListCI.viewFormatCount = 1; + formatListCI.pViewFormats = &viewFormat; + + VkPhysicalDeviceImageFormatInfo2 info{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_FORMAT_INFO_2}; + info.pNext = (flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) + ? (void*)&formatListCI : (void*)&modInfo; + info.format = m_config.format; + info.type = VK_IMAGE_TYPE_2D; + info.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + info.usage = usage; + info.flags = flags; + + VkExternalImageFormatProperties extProps{ + VK_STRUCTURE_TYPE_EXTERNAL_IMAGE_FORMAT_PROPERTIES}; + VkImageFormatProperties2 out{VK_STRUCTURE_TYPE_IMAGE_FORMAT_PROPERTIES_2}; + out.pNext = &extProps; + + c.queried = true; + c.queryResult = m_vkDevCtx.GetPhysicalDeviceImageFormatProperties2( + physDev, &info, &out); + + if (c.queryResult == VK_SUCCESS) { + const VkExternalMemoryFeatureFlags f = + extProps.externalMemoryProperties.externalMemoryFeatures; + c.exportable = (f & VK_EXTERNAL_MEMORY_FEATURE_EXPORTABLE_BIT) != 0; + c.importable = (f & VK_EXTERNAL_MEMORY_FEATURE_IMPORTABLE_BIT) != 0; + c.usable = c.exportable && c.importable; + if (!c.usable) { + c.note = "driver does not advertise EXPORTABLE+IMPORTABLE for dma-buf"; + } + } else { + c.note = std::string("image format unsupported: ") + vkResultName(c.queryResult); + } + + if (c.usable && (c.memoryPlaneCount == 0)) { + c.usable = false; + c.note = "drmFormatModifierPlaneCount is 0"; + } + + m_modifiers.push_back(c); + } + + bool anyUsable = false; + for (const auto& c : m_modifiers) { + if (c.usable) { anyUsable = true; break; } + } + if (!anyUsable) { + m_initFailedAt = Step::ModifierCapability; + m_initFailureDetail = + std::string("no DRM modifier for ") + formatName(m_config.format) + + " is advertised as both EXPORTABLE and IMPORTABLE for dma-buf with " + "the requested usage/flags - there is nothing to test on this host"; + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + return VK_SUCCESS; +} + +//============================================================================= +// Small helpers +//============================================================================= + +uint32_t LinuxDmaBufImportTest::chooseMemoryType(uint32_t candidateMask, + VkMemoryPropertyFlags required) const +{ + for (uint32_t i = 0; i < m_memProps.memoryTypeCount; ++i) { + if ((candidateMask & (1u << i)) && + ((m_memProps.memoryTypes[i].propertyFlags & required) == required)) { + return i; + } + } + return UINT32_MAX; +} + +VkResult LinuxDmaBufImportTest::createStaging(TestDevice& dev, VkDeviceSize size, + VkBufferUsageFlags usage, Staging& out) +{ + out = Staging{}; + out.size = size; + + VkBufferCreateInfo bufCI{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + bufCI.size = size; + bufCI.usage = usage; + bufCI.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + + VkResult r = dev.fn.CreateBuffer(dev.device, &bufCI, nullptr, &out.buffer); + if (r != VK_SUCCESS) { + return r; + } + + VkMemoryRequirements req{}; + dev.fn.GetBufferMemoryRequirements(dev.device, out.buffer, &req); + + uint32_t typeIdx = chooseMemoryType(req.memoryTypeBits, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + out.coherent = (typeIdx != UINT32_MAX); + if (typeIdx == UINT32_MAX) { + typeIdx = chooseMemoryType(req.memoryTypeBits, VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT); + } + if (typeIdx == UINT32_MAX) { + dev.fn.DestroyBuffer(dev.device, out.buffer, nullptr); + out.buffer = VK_NULL_HANDLE; + return VK_ERROR_OUT_OF_DEVICE_MEMORY; + } + + VkMemoryAllocateInfo allocInfo{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocInfo.allocationSize = req.size; + allocInfo.memoryTypeIndex = typeIdx; + + r = dev.fn.AllocateMemory(dev.device, &allocInfo, nullptr, &out.memory); + if (r != VK_SUCCESS) { + dev.fn.DestroyBuffer(dev.device, out.buffer, nullptr); + out.buffer = VK_NULL_HANDLE; + return r; + } + + r = dev.fn.BindBufferMemory(dev.device, out.buffer, out.memory, 0); + if (r == VK_SUCCESS) { + r = dev.fn.MapMemory(dev.device, out.memory, 0, VK_WHOLE_SIZE, 0, &out.mapped); + } + if (r != VK_SUCCESS) { + destroyStaging(dev, out); + } + return r; +} + +void LinuxDmaBufImportTest::destroyStaging(TestDevice& dev, Staging& s) { + if (s.mapped != nullptr) { + dev.fn.UnmapMemory(dev.device, s.memory); + s.mapped = nullptr; + } + if (s.buffer != VK_NULL_HANDLE) { + dev.fn.DestroyBuffer(dev.device, s.buffer, nullptr); + s.buffer = VK_NULL_HANDLE; + } + if (s.memory != VK_NULL_HANDLE) { + dev.fn.FreeMemory(dev.device, s.memory, nullptr); + s.memory = VK_NULL_HANDLE; + } +} + +VkResult LinuxDmaBufImportTest::submitOneShot(TestDevice& dev, VkCommandBuffer cmd, + const char*& failedCall) +{ + failedCall = nullptr; + + VkResult r = dev.fn.EndCommandBuffer(cmd); + if (r != VK_SUCCESS) { failedCall = "vkEndCommandBuffer"; return r; } + + VkFenceCreateInfo fenceCI{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + VkFence fence = VK_NULL_HANDLE; + r = dev.fn.CreateFence(dev.device, &fenceCI, nullptr, &fence); + if (r != VK_SUCCESS) { failedCall = "vkCreateFence"; return r; } + + VkSubmitInfo submit{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submit.commandBufferCount = 1; + submit.pCommandBuffers = &cmd; + + r = dev.fn.QueueSubmit(dev.queue, 1, &submit, fence); + if (r != VK_SUCCESS) { + failedCall = "vkQueueSubmit"; + dev.fn.DestroyFence(dev.device, fence, nullptr); + return r; + } + + // 5 s: a cross-device copy that has not landed by then is hung, and a hang + // must be reported as a failure rather than waited on for ever. + r = dev.fn.WaitForFences(dev.device, 1, &fence, VK_TRUE, 5ull * 1000 * 1000 * 1000); + dev.fn.DestroyFence(dev.device, fence, nullptr); + if (r != VK_SUCCESS) { failedCall = "vkWaitForFences"; return r; } + + r = dev.fn.QueueWaitIdle(dev.queue); + if (r != VK_SUCCESS) { failedCall = "vkQueueWaitIdle"; } + return r; +} + +//============================================================================= +// One end-to-end cycle: export on A, import on `importDev`, read back. +//============================================================================= + +ArmResult LinuxDmaBufImportTest::runCycle(const char* armLabel, + TestDevice& importDev, + const ModifierCandidate& mod) +{ + ArmResult res; + res.arm = armLabel; + res.modifier = mod.modifier; + res.exportDeviceHandle = (uint64_t)(uintptr_t)m_devA.device; + res.importDeviceHandle = (uint64_t)(uintptr_t)importDev.device; + + VkImage expImage = VK_NULL_HANDLE; + VkDeviceMemory expMem = VK_NULL_HANDLE; + VkImage impImage = VK_NULL_HANDLE; + VkDeviceMemory impMem = VK_NULL_HANDLE; + int exportFd = -1; + int importFd = -1; + Staging upload{}; + Staging download{}; + VkCommandBuffer cmdA = VK_NULL_HANDLE; + VkCommandBuffer cmdB = VK_NULL_HANDLE; + + std::vector memPlaneLayouts; + std::vector regions; + std::vector> planeRanges; // offset, size + VkDeviceSize totalStagingBytes = 0; + + auto fail = [&](Step s, VkResult r, const std::string& d) { + res.failedAt = s; + res.vkResult = r; + res.detail = d; + }; + + // The content round-trip needs the modifier's tiling to support transfers + // in both directions. When it does not, that is REPORTED, and the arm can + // no longer contribute a "claim verified" - it is not quietly downgraded. + const bool tilingCanTransfer = + ((mod.tilingFeatures & VK_FORMAT_FEATURE_TRANSFER_DST_BIT) != 0) && + ((mod.tilingFeatures & VK_FORMAT_FEATURE_TRANSFER_SRC_BIT) != 0); + const bool doContent = + m_config.contentCheck && m_formatDesc.known && tilingCanTransfer; + + auto body = [&]() { + //-------------------------------------------------------------------- + // Guard the premise of the whole test. + //-------------------------------------------------------------------- + if ((strcmp(armLabel, "second-device") == 0) && + (importDev.device == m_devA.device)) { + fail(Step::ImportDeviceCreate, VK_ERROR_INITIALIZATION_FAILED, + "the 'second-device' arm was handed device A - the test would " + "prove nothing; refusing to report a result"); + return; + } + + //-------------------------------------------------------------------- + // 1. Exportable image on device A. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ExportImageCreate; + + VkFormat viewFormat = m_config.format; + uint64_t modifier = mod.modifier; + + VkExternalMemoryImageCreateInfo extMemCI{ + VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_IMAGE_CREATE_INFO}; + extMemCI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + VkImageDrmFormatModifierListCreateInfoEXT drmList{ + VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_LIST_CREATE_INFO_EXT}; + drmList.pNext = &extMemCI; + drmList.drmFormatModifierCount = 1; + drmList.pDrmFormatModifiers = &modifier; + + VkImageFormatListCreateInfo formatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + formatListCI.pNext = &drmList; + formatListCI.viewFormatCount = 1; + formatListCI.pViewFormats = &viewFormat; + + VkImageCreateInfo imageCI{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + // MUTABLE_FORMAT on a DRM-modifier image obliges a view-format list + // (VUID-VkImageCreateInfo-tiling-02353). + imageCI.pNext = (exportFlags() & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) + ? (void*)&formatListCI : (void*)&drmList; + imageCI.flags = exportFlags(); + imageCI.imageType = VK_IMAGE_TYPE_2D; + imageCI.format = m_config.format; + imageCI.extent = {m_config.width, m_config.height, 1}; + imageCI.mipLevels = 1; + imageCI.arrayLayers = 1; + imageCI.samples = VK_SAMPLE_COUNT_1_BIT; + imageCI.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + imageCI.usage = exportUsage(); + imageCI.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + imageCI.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + VkResult r = m_devA.fn.CreateImage(m_devA.device, &imageCI, nullptr, &expImage); + if (r != VK_SUCCESS) { + fail(Step::ExportImageCreate, r, + "vkCreateImage on the export device rejected the modifier the " + "driver had just advertised"); + return; + } + + //-------------------------------------------------------------------- + // 2. Which modifier did we actually get? + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ExportModifierReadback; + VkImageDrmFormatModifierPropertiesEXT gotMod{ + VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_PROPERTIES_EXT}; + r = m_devA.fn.GetImageDrmFormatModifierPropertiesEXT(m_devA.device, expImage, &gotMod); + if (r != VK_SUCCESS) { + fail(Step::ExportModifierReadback, r, + "vkGetImageDrmFormatModifierPropertiesEXT failed - the importer " + "cannot be told which layout to expect"); + return; + } + res.readbackModifier = gotMod.drmFormatModifier; + if (res.readbackModifier != mod.modifier) { + fail(Step::ExportModifierReadback, VK_ERROR_INITIALIZATION_FAILED, + "driver chose modifier " + modifierToString(res.readbackModifier) + + " from a single-entry list containing only " + + modifierToString(mod.modifier)); + return; + } + + //-------------------------------------------------------------------- + // 3. Exportable, dedicated allocation on A. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ExportMemoryAllocate; + VkMemoryRequirements memReqs{}; + m_devA.fn.GetImageMemoryRequirements(m_devA.device, expImage, &memReqs); + + VkMemoryDedicatedAllocateInfo dedicated{ + VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO}; + dedicated.image = expImage; + + VkExportMemoryAllocateInfo exportAI{ + VK_STRUCTURE_TYPE_EXPORT_MEMORY_ALLOCATE_INFO}; + exportAI.pNext = &dedicated; + exportAI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + VkMemoryAllocateInfo allocInfo{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocInfo.pNext = &exportAI; + allocInfo.allocationSize = memReqs.size; + allocInfo.memoryTypeIndex = chooseMemoryType(memReqs.memoryTypeBits, + VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + if (allocInfo.memoryTypeIndex == UINT32_MAX) { + fail(Step::ExportMemoryAllocate, VK_ERROR_OUT_OF_DEVICE_MEMORY, + "no DEVICE_LOCAL memory type in the export image's requirements mask"); + return; + } + res.exportAllocSize = memReqs.size; + res.exportMemTypeIndex = allocInfo.memoryTypeIndex; + + r = m_devA.fn.AllocateMemory(m_devA.device, &allocInfo, nullptr, &expMem); + if (r != VK_SUCCESS) { + fail(Step::ExportMemoryAllocate, r, "exportable allocation failed on device A"); + return; + } + + res.lastStepStarted = Step::ExportBindMemory; + r = m_devA.fn.BindImageMemory(m_devA.device, expImage, expMem, 0); + if (r != VK_SUCCESS) { + fail(Step::ExportBindMemory, r, "vkBindImageMemory failed on device A"); + return; + } + + //-------------------------------------------------------------------- + // 4. Memory-plane layouts. For a DRM-modifier image these come from the + // MEMORY_PLANE aspects, not the colour PLANE aspects, and the count + // is the modifier's drmFormatModifierPlaneCount. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ExportPlaneLayouts; + memPlaneLayouts.resize(mod.memoryPlaneCount); + for (uint32_t p = 0; p < mod.memoryPlaneCount; ++p) { + VkImageSubresource sub{}; + sub.aspectMask = (VkImageAspectFlags)(VK_IMAGE_ASPECT_MEMORY_PLANE_0_BIT_EXT << p); + sub.mipLevel = 0; + sub.arrayLayer = 0; + m_devA.fn.GetImageSubresourceLayout(m_devA.device, expImage, &sub, + &memPlaneLayouts[p]); + } + + //-------------------------------------------------------------------- + // 5. Export the dma-buf fd. From here on the fd is a resource with an + // owner, and every exit below accounts for it. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ExportFd; + VkMemoryGetFdInfoKHR getFd{VK_STRUCTURE_TYPE_MEMORY_GET_FD_INFO_KHR}; + getFd.memory = expMem; + getFd.handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + r = m_devA.fn.GetMemoryFdKHR(m_devA.device, &getFd, &exportFd); + if ((r != VK_SUCCESS) || (exportFd < 0)) { + fail(Step::ExportFd, r, "vkGetMemoryFdKHR (DMA_BUF) failed on device A"); + return; + } + + //-------------------------------------------------------------------- + // 6. Write a known pattern through device A and release the image to + // VK_QUEUE_FAMILY_FOREIGN_EXT - the same handshake the encoder does + // for an imported frame (VkVideoEncoder.cpp StageInputFrame). + //-------------------------------------------------------------------- + if (doContent) { + res.lastStepStarted = Step::UploadPattern; + res.contentAttempted = true; + + uint64_t offset = 0; + for (const auto& pd : m_formatDesc.planes) { + const uint32_t pw = m_config.width / pd.widthDiv; + const uint32_t ph = m_config.height / pd.heightDiv; + const uint64_t bytes = (uint64_t)pw * ph * pd.texelBytes; + + VkBufferImageCopy region{}; + region.bufferOffset = offset; + region.bufferRowLength = 0; // tightly packed + region.bufferImageHeight = 0; + region.imageSubresource.aspectMask = pd.aspect; + region.imageSubresource.mipLevel = 0; + region.imageSubresource.baseArrayLayer = 0; + region.imageSubresource.layerCount = 1; + region.imageOffset = {0, 0, 0}; + region.imageExtent = {pw, ph, 1}; + regions.push_back(region); + planeRanges.emplace_back(offset, bytes); + + // 16-byte aligned plane starts keep bufferOffset legal for every + // texel-block size in play (VUID-VkBufferImageCopy-bufferOffset-00193). + offset += (bytes + 15u) & ~15ull; + } + totalStagingBytes = offset; + + r = createStaging(m_devA, totalStagingBytes, + VK_BUFFER_USAGE_TRANSFER_SRC_BIT, upload); + if (r != VK_SUCCESS) { + fail(Step::UploadPattern, r, "host-visible upload buffer failed on device A"); + return; + } + uint8_t* src = static_cast(upload.mapped); + for (const auto& pr : planeRanges) { + for (uint64_t i = 0; i < pr.second; ++i) { + src[pr.first + i] = patternByte(pr.first + i); + } + } + if (!upload.coherent) { + VkMappedMemoryRange range{VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE}; + range.memory = upload.memory; + range.offset = 0; + range.size = VK_WHOLE_SIZE; + m_devA.fn.FlushMappedMemoryRanges(m_devA.device, 1, &range); + } + + VkCommandBufferAllocateInfo cbAI{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cbAI.commandPool = m_devA.cmdPool; + cbAI.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; + cbAI.commandBufferCount = 1; + r = m_devA.fn.AllocateCommandBuffers(m_devA.device, &cbAI, &cmdA); + if (r != VK_SUCCESS) { + fail(Step::UploadPattern, r, "vkAllocateCommandBuffers failed on device A"); + return; + } + + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + r = m_devA.fn.BeginCommandBuffer(cmdA, &begin); + if (r != VK_SUCCESS) { + fail(Step::UploadPattern, r, "vkBeginCommandBuffer failed on device A"); + return; + } + + VkImageMemoryBarrier toDst{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + toDst.srcAccessMask = 0; + toDst.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + toDst.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + toDst.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + toDst.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + toDst.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + toDst.image = expImage; + toDst.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + m_devA.fn.CmdPipelineBarrier(cmdA, + VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &toDst); + + m_devA.fn.CmdCopyBufferToImage(cmdA, upload.buffer, expImage, + VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + (uint32_t)regions.size(), regions.data()); + + // Release to FOREIGN. The acquire on the import device must repeat + // these same oldLayout/newLayout values; the transition happens once. + res.lastStepStarted = Step::ReleaseToForeign; + VkImageMemoryBarrier release{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + release.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + release.dstAccessMask = 0; + release.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + release.newLayout = VK_IMAGE_LAYOUT_GENERAL; + release.srcQueueFamilyIndex = m_devA.queueFamily; + release.dstQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT; + release.image = expImage; + release.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + m_devA.fn.CmdPipelineBarrier(cmdA, + VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, + 0, 0, nullptr, 0, nullptr, 1, &release); + + const char* failedCall = nullptr; + r = submitOneShot(m_devA, cmdA, failedCall); + if (r != VK_SUCCESS) { + fail(Step::ReleaseToForeign, r, + std::string(failedCall ? failedCall : "submit") + + " failed on device A while writing the pattern"); + return; + } + } + + //-------------------------------------------------------------------- + // 7. Import on the target device. Explicit modifier + the exporter's + // memory-plane layouts, exactly as the library does at + // vulkan_video_encoder_ext.cpp:3286-3303. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ImportImageCreate; + + VkExternalMemoryImageCreateInfo impExtMemCI{ + VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_IMAGE_CREATE_INFO}; + impExtMemCI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + std::vector importLayouts(memPlaneLayouts.size()); + for (size_t p = 0; p < memPlaneLayouts.size(); ++p) { + importLayouts[p].offset = memPlaneLayouts[p].offset; + importLayouts[p].size = 0; // VUID-...-size-02267 + importLayouts[p].rowPitch = memPlaneLayouts[p].rowPitch; + // arrayLayers is 1 and extent.depth is 1, so these MUST be 0 + // (VUID-...-arrayPitch-02268 / -depthPitch-02269), whatever the + // exporter reported. + importLayouts[p].arrayPitch = 0; + importLayouts[p].depthPitch = 0; + } + + VkImageDrmFormatModifierExplicitCreateInfoEXT drmExplicit{ + VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_EXPLICIT_CREATE_INFO_EXT}; + drmExplicit.pNext = &impExtMemCI; + drmExplicit.drmFormatModifier = res.readbackModifier; + drmExplicit.drmFormatModifierPlaneCount = (uint32_t)importLayouts.size(); + drmExplicit.pPlaneLayouts = importLayouts.data(); + + VkImageFormatListCreateInfo impFormatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + impFormatListCI.pNext = &drmExplicit; + impFormatListCI.viewFormatCount = 1; + impFormatListCI.pViewFormats = &viewFormat; + + VkImageCreateInfo impCI{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + impCI.pNext = (importFlags() & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) + ? (void*)&impFormatListCI : (void*)&drmExplicit; + impCI.flags = importFlags(); + impCI.imageType = VK_IMAGE_TYPE_2D; + impCI.format = m_config.format; + impCI.extent = {m_config.width, m_config.height, 1}; + impCI.mipLevels = 1; + impCI.arrayLayers = 1; + impCI.samples = VK_SAMPLE_COUNT_1_BIT; + impCI.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + impCI.usage = importUsage(); + impCI.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + impCI.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + r = importDev.fn.CreateImage(importDev.device, &impCI, nullptr, &impImage); + if (r != VK_SUCCESS) { + fail(Step::ImportImageCreate, r, + "vkCreateImage on the IMPORT device rejected the exporter's " + "modifier + plane layouts"); + return; + } + + VkMemoryRequirements impReqs{}; + importDev.fn.GetImageMemoryRequirements(importDev.device, impImage, &impReqs); + res.importMemReqSize = impReqs.size; + res.importMemTypeBits = impReqs.memoryTypeBits; + + //-------------------------------------------------------------------- + // 8. Which memory types can hold THIS fd on THIS device. Spec-mandated + // for dma-buf (VUID-VkMemoryAllocateInfo-memoryTypeIndex-00648); the + // library warns that a type outside this mask can import + // "successfully" on NVIDIA and then read garbage + // (vulkan_video_encoder_ext.cpp:3328-3343). A test must not guess. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ImportFdMemoryTypes; + VkMemoryFdPropertiesKHR fdProps{VK_STRUCTURE_TYPE_MEMORY_FD_PROPERTIES_KHR}; + r = importDev.fn.GetMemoryFdPropertiesKHR( + importDev.device, VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT, + exportFd, &fdProps); + if (r != VK_SUCCESS) { + fail(Step::ImportFdMemoryTypes, r, + "vkGetMemoryFdPropertiesKHR failed on the import device - the " + "importable memory-type mask for this fd is unknown and the " + "test refuses to guess one"); + return; + } + res.fdMemTypeBits = fdProps.memoryTypeBits; + + res.lastStepStarted = Step::ImportMemoryTypeSelect; + const uint32_t candidates = impReqs.memoryTypeBits & fdProps.memoryTypeBits; + if (candidates == 0) { + std::ostringstream os; + os << "no memory type satisfies both the import image (0x" << std::hex + << impReqs.memoryTypeBits << ") and this dma-buf (0x" + << fdProps.memoryTypeBits << ")" << std::dec; + fail(Step::ImportMemoryTypeSelect, VK_ERROR_OUT_OF_DEVICE_MEMORY, os.str()); + return; + } + uint32_t impTypeIdx = chooseMemoryType(candidates, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT); + if (impTypeIdx == UINT32_MAX) { + impTypeIdx = chooseMemoryType(candidates, 0); + } + res.importMemTypeIndex = impTypeIdx; + + //-------------------------------------------------------------------- + // 9. Import. dup() first: the fd handed to vkAllocateMemory belongs to + // the driver from the moment that call SUCCEEDS and never before, so + // the master fd stays ours for the second arm and for cleanup. + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ImportMemoryAllocate; + importFd = ::dup(exportFd); + if (importFd < 0) { + fail(Step::ImportMemoryAllocate, VK_ERROR_OUT_OF_HOST_MEMORY, + "dup() of the exported dma-buf fd failed"); + return; + } + + VkMemoryDedicatedAllocateInfo impDedicated{ + VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO}; + impDedicated.image = impImage; + + VkImportMemoryFdInfoKHR importFdInfo{VK_STRUCTURE_TYPE_IMPORT_MEMORY_FD_INFO_KHR}; + importFdInfo.pNext = &impDedicated; + importFdInfo.handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + importFdInfo.fd = importFd; + + VkMemoryAllocateInfo impAlloc{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + impAlloc.pNext = &importFdInfo; + // The EXPORTER's size. Deriving it from the importer's + // vkGetImageMemoryRequirements is wrong for dma-buf on NVIDIA - see + // vulkan_video_encoder_ext.cpp:3485-3495. + impAlloc.allocationSize = res.exportAllocSize; + impAlloc.memoryTypeIndex = impTypeIdx; + + r = importDev.fn.AllocateMemory(importDev.device, &impAlloc, nullptr, &impMem); + if (r != VK_SUCCESS) { + std::ostringstream os; + os << "vkAllocateMemory(import) failed on device " << importDev.label + << ": size=" << res.exportAllocSize + << " (importer's own requirement was " << impReqs.size << ")" + << " memTypeIdx=" << impTypeIdx; + fail(Step::ImportMemoryAllocate, r, os.str()); + return; // cleanup closes importFd: the driver never took it + } + importFd = -1; // handed off; closing it now would be a double close + + res.lastStepStarted = Step::ImportBindMemory; + r = importDev.fn.BindImageMemory(importDev.device, impImage, impMem, 0); + if (r != VK_SUCCESS) { + fail(Step::ImportBindMemory, r, + "vkBindImageMemory failed on the import device"); + return; + } + res.importSucceeded = true; + + //-------------------------------------------------------------------- + // 10. Use the memory. This is the part the Windows sibling stops short + // of: without it, "import succeeded" is a claim about an API return + // code and not about the memory. + //-------------------------------------------------------------------- + if (!doContent) { + res.detail = m_config.contentCheck + ? (m_formatDesc.known + ? "content round-trip unavailable: this modifier's tiling " + "features lack TRANSFER_SRC/DST" + : "content round-trip unavailable: no plane description for " + "this format") + : "content round-trip disabled (--no-content-check)"; + res.lastStepStarted = Step::Complete; + return; + } + + res.lastStepStarted = Step::AcquireFromForeign; + + r = createStaging(importDev, totalStagingBytes, + VK_BUFFER_USAGE_TRANSFER_DST_BIT, download); + if (r != VK_SUCCESS) { + fail(Step::Readback, r, "host-visible readback buffer failed on the import device"); + return; + } + // Prefill with the complement of the pattern: if the copy never runs, + // the comparison cannot pass by accident. + { + uint8_t* dst = static_cast(download.mapped); + for (VkDeviceSize i = 0; i < totalStagingBytes; ++i) { + dst[i] = static_cast(~patternByte(i)); + } + if (!download.coherent) { + VkMappedMemoryRange range{VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE}; + range.memory = download.memory; + range.offset = 0; + range.size = VK_WHOLE_SIZE; + importDev.fn.FlushMappedMemoryRanges(importDev.device, 1, &range); + } + } + + VkCommandBufferAllocateInfo cbAI{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cbAI.commandPool = importDev.cmdPool; + cbAI.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; + cbAI.commandBufferCount = 1; + r = importDev.fn.AllocateCommandBuffers(importDev.device, &cbAI, &cmdB); + if (r != VK_SUCCESS) { + fail(Step::Readback, r, "vkAllocateCommandBuffers failed on the import device"); + return; + } + + VkCommandBufferBeginInfo begin{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + begin.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + r = importDev.fn.BeginCommandBuffer(cmdB, &begin); + if (r != VK_SUCCESS) { + fail(Step::Readback, r, "vkBeginCommandBuffer failed on the import device"); + return; + } + + // Acquire from FOREIGN. oldLayout/newLayout mirror the release on A. + VkImageMemoryBarrier acquire{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + acquire.srcAccessMask = 0; + acquire.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + acquire.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + acquire.newLayout = VK_IMAGE_LAYOUT_GENERAL; + acquire.srcQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT; + acquire.dstQueueFamilyIndex = importDev.queueFamily; + acquire.image = impImage; + acquire.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + importDev.fn.CmdPipelineBarrier(cmdB, + VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &acquire); + + res.lastStepStarted = Step::Readback; + importDev.fn.CmdCopyImageToBuffer(cmdB, impImage, VK_IMAGE_LAYOUT_GENERAL, + download.buffer, + (uint32_t)regions.size(), regions.data()); + + const char* failedCall = nullptr; + r = submitOneShot(importDev, cmdB, failedCall); + if (r != VK_SUCCESS) { + fail(Step::Readback, r, + std::string(failedCall ? failedCall : "submit") + + " failed on the import device while reading the shared image back"); + return; + } + + if (!download.coherent) { + VkMappedMemoryRange range{VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE}; + range.memory = download.memory; + range.offset = 0; + range.size = VK_WHOLE_SIZE; + importDev.fn.InvalidateMappedMemoryRanges(importDev.device, 1, &range); + } + + //-------------------------------------------------------------------- + // 11. Compare only the plane byte ranges (the inter-plane alignment + // padding is never written by either copy, so comparing it would + // manufacture a failure). + //-------------------------------------------------------------------- + res.lastStepStarted = Step::ContentCompare; + const uint8_t* got = static_cast(download.mapped); + for (const auto& pr : planeRanges) { + for (uint64_t i = 0; i < pr.second; ++i) { + const uint64_t off = pr.first + i; + const uint8_t expected = patternByte(off); + if (got[off] != expected) { + res.firstMismatchOffset = off; + res.expectedByte = expected; + res.actualByte = got[off]; + fail(Step::ContentCompare, VK_SUCCESS, + "the import device read back different bytes than the " + "export device wrote - the import bound memory that is " + "not the exported allocation, or the layouts disagree"); + return; + } + res.comparedBytes++; + } + } + res.contentVerified = true; + res.lastStepStarted = Step::Complete; + }; + + body(); + + //------------------------------------------------------------------------- + // Cleanup. Order matters: images before their memory, and the fds last. + // A failed/timed-out submit can leave work in flight; freeing command + // buffers or destroying images under it is undefined behaviour, so drain + // both devices before touching anything. + if (importDev.fn.DeviceWaitIdle != nullptr) { + importDev.fn.DeviceWaitIdle(importDev.device); + } + if (m_devA.fn.DeviceWaitIdle != nullptr) { + m_devA.fn.DeviceWaitIdle(m_devA.device); + } + + if (cmdB != VK_NULL_HANDLE) { + importDev.fn.FreeCommandBuffers(importDev.device, importDev.cmdPool, 1, &cmdB); + } + if (cmdA != VK_NULL_HANDLE) { + m_devA.fn.FreeCommandBuffers(m_devA.device, m_devA.cmdPool, 1, &cmdA); + } + destroyStaging(importDev, download); + destroyStaging(m_devA, upload); + + if (impImage != VK_NULL_HANDLE) { + importDev.fn.DestroyImage(importDev.device, impImage, nullptr); + } + if (impMem != VK_NULL_HANDLE) { + importDev.fn.FreeMemory(importDev.device, impMem, nullptr); + } + if (expImage != VK_NULL_HANDLE) { + m_devA.fn.DestroyImage(m_devA.device, expImage, nullptr); + } + if (expMem != VK_NULL_HANDLE) { + m_devA.fn.FreeMemory(m_devA.device, expMem, nullptr); + } + + // importFd is >= 0 only when vkAllocateMemory did NOT take it. + if (importFd >= 0) { + ::close(importFd); + } + if (exportFd >= 0) { + ::close(exportFd); + } + + return res; +} + +//============================================================================= +// Run every usable modifier through both arms +//============================================================================= + +std::vector LinuxDmaBufImportTest::run() { + std::vector results; + + std::cout << "\n=== Linux dma-buf second-device import ===\n" + << "Format: " << formatName(m_config.format) << "\n" + << "Size: " << m_config.width << "x" << m_config.height << "\n" + << "Handle: VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT\n" + << "Export: usage 0x" << std::hex << exportUsage() + << " flags 0x" << exportFlags() << std::dec + << (m_config.videoUsage ? " (VIDEO_ENCODE_SRC)" : " (no video usage)") + << "\n" + << "Import: usage 0x" << std::hex << importUsage() + << " flags 0x" << importFlags() << std::dec + << (importVideoUsage() ? " (VIDEO_ENCODE_SRC)" : " (no video usage)") + << "\n" + << "Content: " << (m_config.contentCheck + ? "write on A, read back on B, byte compare" + : "DISABLED (--no-content-check)") << "\n\n"; + + for (const auto& mod : m_modifiers) { + if (!mod.usable) { + std::cout << "[SKIP] modifier " << modifierToString(mod.modifier) + << ": " << (mod.note.empty() ? "not usable" : mod.note) << "\n"; + continue; + } + + std::cout << "[RUN ] modifier " << modifierToString(mod.modifier) + << " memPlanes=" << mod.memoryPlaneCount << "\n"; + + results.push_back(runCycle("second-device", m_devB, mod)); + if (m_config.sameDeviceControl) { + results.push_back(runCycle("same-device", m_devA, mod)); + } + } + + return results; +} + +//============================================================================= +// Reporting +//============================================================================= + +void LinuxDmaBufImportTest::printResults(const std::vector& results) const { + std::cout << "\n" << std::string(108, '=') << "\n" + << " LINUX DMA-BUF SECOND-DEVICE IMPORT RESULTS\n" + << std::string(108, '=') << "\n\n"; + + std::cout << "Modifier capability (physical-device scoped query):\n"; + for (const auto& c : m_modifiers) { + std::cout << " " << std::left << std::setw(22) << modifierToString(c.modifier) + << " planes=" << c.memoryPlaneCount + << " exportable=" << (c.exportable ? "yes" : "no ") + << " importable=" << (c.importable ? "yes" : "no ") + << " " << c.note << "\n"; + } + std::cout << "\n"; + + std::cout << std::left + << std::setw(16) << "Arm" + << std::setw(22) << "Modifier" + << std::setw(10) << "Import" + << std::setw(10) << "Content" + << std::setw(24) << "Stopped at" + << "Result\n" + << std::string(108, '-') << "\n"; + + for (const auto& r : results) { + const char* importStr = r.importSucceeded ? "OK" : "FAIL"; + const char* contentStr = r.contentVerified + ? "OK" + : (r.contentAttempted ? "FAIL" : "n/a"); + std::cout << std::left + << std::setw(16) << r.arm + << std::setw(22) << modifierToString(r.modifier) + << std::setw(10) << importStr + << std::setw(10) << contentStr + << std::setw(24) << (r.ok() ? "-" : stepName(r.failedAt)) + << (r.ok() ? std::string("PASS") + : (std::string("FAIL ") + vkResultName(r.vkResult))) + << "\n"; + } + std::cout << std::string(108, '-') << "\n\n"; + + for (const auto& r : results) { + if (r.ok() && r.detail.empty()) { + continue; + } + std::cout << (r.ok() ? "NOTE " : "FAIL ") + << r.arm << " / modifier " << modifierToString(r.modifier) << "\n"; + if (!r.ok()) { + std::cout << " stopped at : " << stepName(r.failedAt) + << " (" << vkResultName(r.vkResult) << ")\n"; + } + if (!r.detail.empty()) { + std::cout << " detail : " << r.detail << "\n"; + } + std::cout << " devices : export VkDevice 0x" << std::hex + << r.exportDeviceHandle << ", import VkDevice 0x" + << r.importDeviceHandle << std::dec << "\n"; + if (r.exportAllocSize != 0) { + std::cout << " export : size=" << r.exportAllocSize + << " memTypeIdx=" << r.exportMemTypeIndex + << " modifier=" << modifierToString(r.readbackModifier) << "\n"; + } + if (r.importMemReqSize != 0) { + std::cout << " import : reqSize=" << r.importMemReqSize + << " reqBits=0x" << std::hex << r.importMemTypeBits + << " fdBits=0x" << r.fdMemTypeBits << std::dec + << " chosen=" << (int64_t)(int32_t)r.importMemTypeIndex << "\n"; + } + if (r.firstMismatchOffset != UINT64_MAX) { + std::cout << " first bad : byte " << r.firstMismatchOffset + << " expected 0x" << std::hex << r.expectedByte + << " got 0x" << r.actualByte << std::dec + << " (after " << r.comparedBytes << " matching bytes)\n"; + } + std::cout << "\n"; + } +} + +Verdict LinuxDmaBufImportTest::verdict(const std::vector& results, + std::string& reasonOut) const +{ + if (results.empty()) { + reasonOut = "no arm ran: no usable modifier survived the capability query"; + return Verdict::CouldNotRun; + } + + // Pair the arms per modifier so a failure can be attributed. + for (const auto& r : results) { + if (r.arm != "second-device" || r.ok()) { + continue; + } + bool controlAlsoFailed = false; + bool controlRan = false; + for (const auto& c : results) { + if ((c.arm == "same-device") && (c.modifier == r.modifier)) { + controlRan = true; + controlAlsoFailed = !c.ok(); + } + } + reasonOut = std::string("second-device arm failed at ") + + stepName(r.failedAt) + " for modifier " + + modifierToString(r.modifier); + if (controlRan && controlAlsoFailed) { + reasonOut += " -- the same-device control failed too, so this is an " + "export/environment defect and says NOTHING about " + "device independence"; + } else if (controlRan) { + reasonOut += " -- the same-device control PASSED, so this is a " + "device-independence defect"; + } + return Verdict::ClaimFailed; + } + + for (const auto& r : results) { + if ((r.arm == "second-device") && r.importSucceeded && r.contentVerified) { + reasonOut = "an image exported by VkDevice A was imported by a " + "different VkDevice B on the same VkPhysicalDevice, and " + "B read back exactly the bytes A wrote"; + return Verdict::ClaimVerified; + } + } + + if (!m_config.contentCheck) { + reasonOut = "--no-content-check was passed: import reachability was " + "exercised, memory USABILITY was not. This is not a pass."; + return Verdict::CouldNotRun; + } + + reasonOut = "every second-device import returned VK_SUCCESS but no content " + "round-trip completed, so nothing proved the memory is usable"; + return Verdict::CouldNotRun; +} + +} // namespace linux_dmabuf_import_test + +#endif // __linux__ diff --git a/common/libs/tests/linux_dmabuf_import/src/main.cpp b/common/libs/tests/linux_dmabuf_import/src/main.cpp new file mode 100644 index 00000000..845ff4d2 --- /dev/null +++ b/common/libs/tests/linux_dmabuf_import/src/main.cpp @@ -0,0 +1,204 @@ +/* + * Copyright 2024-2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#if defined(__linux__) + +#include "LinuxDmaBufImportTest.h" + +#include // strcasecmp + +#include +#include +#include +#include +#include + +using namespace linux_dmabuf_import_test; + +// Exit codes are three-valued on purpose. "The test could not run" must not be +// spellable as success by anything reading $? - see the banner below. +static const int kExitClaimVerified = 0; +static const int kExitClaimFailed = 1; +static const int kExitCouldNotRun = 2; +static const int kExitBadUsage = 64; + +static void printHelp(const char* prog) { + std::cout + << "Linux dma-buf SECOND-DEVICE import test\n\n" + << "Exports an image from one VkDevice and imports it into a DIFFERENT\n" + << "VkDevice on the SAME VkPhysicalDevice, then proves the memory is\n" + << "usable by writing a pattern through device A and reading it back\n" + << "through device B.\n\n" + << "This is the Linux counterpart of win32_opaque_import.\n\n" + << "Usage: " << prog << " [options]\n\n" + << "Options:\n" + << " --help, -h Show this help\n" + << " --verbose, -v Per-operation logging\n" + << " --validation Enable Vulkan validation layers\n" + << " --width Image width (default 1920)\n" + << " --height Image height (default 1080)\n" + << " --format nv12 | nv16 | p010 | p012 | rgba8 | bgra8\n" + << " (default nv12)\n" + << " --no-video-usage Drop VIDEO_ENCODE_SRC and the video create flags\n" + << " on BOTH the export and the import image\n" + << " --import-no-video-usage\n" + << " Drop them on the IMPORT image only. This is the\n" + << " Chromium shape: a foreign (GBM) exporter, and an\n" + << " importer whose usage the VEA picked itself.\n" + << " --import-video-usage Force them ON for the import image only\n" + << " --import-usage Raw VkImageUsageFlags for the import image\n" + << " --import-flags Raw VkImageCreateFlags for the import image\n" + << " --linear-only Only test DRM_FORMAT_MOD_LINEAR\n" + << " --modifier Only test this DRM modifier (e.g. 0x300000000c000001)\n" + << " --no-content-check Stop after bind; do NOT prove the memory is\n" + << " usable. Reports COULD-NOT-RUN, never PASS.\n" + << " --no-control Skip the same-device control arm\n\n" + << "Exit codes:\n" + << " 0 claim verified (second-device import + content round-trip OK)\n" + << " 1 claim FAILED\n" + << " 2 could not run (no GPU / no extension / nothing proved)\n" + << " 64 bad usage\n"; +} + +static bool parseFormat(const char* s, VkFormat& out) { + if (!strcasecmp(s, "nv12")) { out = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; return true; } + if (!strcasecmp(s, "nv16")) { out = VK_FORMAT_G8_B8R8_2PLANE_422_UNORM; return true; } + if (!strcasecmp(s, "p010")) { out = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16; return true; } + if (!strcasecmp(s, "p012")) { out = VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16; return true; } + if (!strcasecmp(s, "rgba8")) { out = VK_FORMAT_R8G8B8A8_UNORM; return true; } + if (!strcasecmp(s, "bgra8")) { out = VK_FORMAT_B8G8R8A8_UNORM; return true; } + return false; +} + +static bool parseArgs(int argc, char* argv[], TestConfig& cfg, bool& helpShown) { + helpShown = false; + for (int i = 1; i < argc; ++i) { + const char* a = argv[i]; + if (!strcmp(a, "--help") || !strcmp(a, "-h")) { + printHelp(argv[0]); + helpShown = true; + return false; + } else if (!strcmp(a, "--verbose") || !strcmp(a, "-v")) { + cfg.verbose = true; + } else if (!strcmp(a, "--validation")) { + cfg.validation = true; + } else if (!strcmp(a, "--no-video-usage")) { + cfg.videoUsage = false; + } else if (!strcmp(a, "--import-no-video-usage")) { + cfg.importVideoUsage = 0; + } else if (!strcmp(a, "--import-video-usage")) { + cfg.importVideoUsage = 1; + } else if (!strcmp(a, "--import-usage") && (i + 1 < argc)) { + cfg.importUsageRaw = (uint32_t)strtoul(argv[++i], nullptr, 0); + } else if (!strcmp(a, "--import-flags") && (i + 1 < argc)) { + cfg.importFlagsRaw = (uint32_t)strtoul(argv[++i], nullptr, 0); + } else if (!strcmp(a, "--linear-only")) { + cfg.linearOnly = true; + } else if (!strcmp(a, "--no-content-check")) { + cfg.contentCheck = false; + } else if (!strcmp(a, "--no-control")) { + cfg.sameDeviceControl = false; + } else if (!strcmp(a, "--width") && (i + 1 < argc)) { + cfg.width = (uint32_t)strtoul(argv[++i], nullptr, 0); + } else if (!strcmp(a, "--height") && (i + 1 < argc)) { + cfg.height = (uint32_t)strtoul(argv[++i], nullptr, 0); + } else if (!strcmp(a, "--modifier") && (i + 1 < argc)) { + cfg.onlyModifier = strtoull(argv[++i], nullptr, 0); + } else if (!strcmp(a, "--format") && (i + 1 < argc)) { + if (!parseFormat(argv[++i], cfg.format)) { + std::cerr << "Unknown format: " << argv[i] << "\n"; + return false; + } + } else { + std::cerr << "Unknown option: " << a << "\n"; + printHelp(argv[0]); + return false; + } + } + return true; +} + +int main(int argc, char* argv[]) { + std::cout << "=================================================\n" + << " Linux dma-buf SECOND-DEVICE import test\n" + << "=================================================\n"; + + TestConfig config; + bool helpShown = false; + if (!parseArgs(argc, argv, config, helpShown)) { + return helpShown ? kExitClaimVerified : kExitBadUsage; + } + + LinuxDmaBufImportTest test; + VkResult initResult = test.init(config); + if (initResult != VK_SUCCESS) { + std::cerr << "\n" << std::string(72, '!') << "\n" + << "COULD NOT RUN - THIS IS NOT A PASS\n" + << std::string(72, '!') << "\n" + << " stopped at : " << stepName(test.initFailedAt()) << "\n" + << " VkResult : " << vkResultName(initResult) << "\n" + << " detail : " << test.initFailureDetail() << "\n\n" + << "The second-device dma-buf import claim is UNTESTED\n" + << "on this host. Do not read this run as evidence for or\n" + << "against it.\n"; + return kExitCouldNotRun; + } + + std::vector results = test.run(); + test.printResults(results); + + std::string reason; + const Verdict v = test.verdict(results, reason); + + switch (v) { + case Verdict::ClaimVerified: + std::cout << std::string(72, '=') << "\n" + << "VERDICT: CLAIM VERIFIED\n" + << std::string(72, '=') << "\n" + << " " << reason << "\n"; + return kExitClaimVerified; + + case Verdict::ClaimFailed: + std::cout << std::string(72, '!') << "\n" + << "VERDICT: CLAIM FAILED\n" + << std::string(72, '!') << "\n" + << " " << reason << "\n"; + return kExitClaimFailed; + + case Verdict::CouldNotRun: + default: + std::cout << std::string(72, '!') << "\n" + << "VERDICT: COULD NOT RUN - THIS IS NOT A PASS\n" + << std::string(72, '!') << "\n" + << " " << reason << "\n"; + return kExitCouldNotRun; + } +} + +#else // !__linux__ + +#include + +// dma-buf is a Linux kernel object; there is no cross-platform equivalent to +// test here. The Windows sibling is common/libs/tests/win32_opaque_import. +int main() { + std::cout << "Linux dma-buf second-device import test: NOT APPLICABLE " + "(dma-buf is Linux-only). See win32_opaque_import for the " + "Windows OPAQUE_WIN32 sibling.\n"; + return 0; +} + +#endif // __linux__ diff --git a/common/libs/tests/src/ColorConversion.cpp b/common/libs/tests/src/ColorConversion.cpp index 1dc375f4..d8532f3b 100644 --- a/common/libs/tests/src/ColorConversion.cpp +++ b/common/libs/tests/src/ColorConversion.cpp @@ -271,6 +271,38 @@ void generateRGBATestPattern(TestPatternType type, break; } + case TestPatternType::PurePrimaryQuadrants: { + // TL = red, TR = green, BL = blue, BR = white. Fully saturated: + // 255 and 0, no intermediate values anywhere, so every quadrant + // has an unambiguous (Cb, Cr) signature and a component or plane + // swap relocates it to a value nothing else in the frame produces. + // + // The quadrant split uses the SAME rounding for both axes as the + // 4:2:0 chroma subsample (halves at width/2, height/2), so no + // 2x2 chroma block ever straddles two quadrants on an + // even-dimensioned image and the expected chroma is exact rather + // than a boundary average. + const uint32_t halfW = width / 2; + const uint32_t halfH = height / 2; + for (uint32_t y = 0; y < height; y++) { + for (uint32_t x = 0; x < width; x++) { + const bool right = (x >= halfW); + const bool bottom = (y >= halfH); + uint8_t r, g, b; + if (!bottom && !right) { r = 255; g = 0; b = 0; } // TL red + else if (!bottom && right) { r = 0; g = 255; b = 0; } // TR green + else if (bottom && !right) { r = 0; g = 0; b = 255; } // BL blue + else { r = 255; g = 255; b = 255; } // BR white + const uint32_t offset = (y * width + x) * 4; + data[offset + 0] = r; + data[offset + 1] = g; + data[offset + 2] = b; + data[offset + 3] = 255; + } + } + break; + } + case TestPatternType::Random: { uint32_t seed = 12345; for (size_t i = 0; i < data.size(); i += 4) { diff --git a/common/libs/tests/src/FilterTestApp.cpp b/common/libs/tests/src/FilterTestApp.cpp index b2f49229..a61cc9f0 100644 --- a/common/libs/tests/src/FilterTestApp.cpp +++ b/common/libs/tests/src/FilterTestApp.cpp @@ -14,6 +14,9 @@ * limitations under the License. */ +#include +#include +#include #include "FilterTestApp.h" #include "TestCases.h" #include "ColorConversion.h" @@ -22,6 +25,7 @@ #include #include #include +#include #include "nvidia_utils/vulkan/ycbcrvkinfo.h" #include "VkCodecUtils/Helpers.h" // For vk::DeviceUuidUtils @@ -461,12 +465,72 @@ void FilterTestApp::registerTest(const TestCaseConfig& config) { m_testCases.push_back(config); } +namespace { +// ----------------------------------------------------------------------- +// Registry of cases whose output is genuinely read back, genuinely compared, +// and genuinely DISAGREES -- because of a defect in the FILTER, not in this +// harness. +// +// Entries are pinned here rather than silenced, because the two ways of +// making the suite green again are both worse: dropping the comparison +// rebuilds the exact check-that-cannot-fail this harness exists to remove, +// and letting the suite go red on arrival gets the whole job ignored. So a +// listed case that FAILS is reported XFAIL and does not sink the run -- and a +// listed case that PASSES is a hard FAILURE, so a fix cannot land without +// this list being updated. +// +// The list is EMPTY. It held two entries, both removed when the single filter +// defect behind them was fixed: +// +// TC043_YCbCrCopy_NV16 -- YCBCRCOPY to 4:2:2 wrote only the top half of +// the chroma plane (rows 540..1079 all zero, +// 1032750 of 2073600 chroma bytes wrong). +// TC044_YCbCrCopy_YUV444 -- YCBCRCOPY to 4:4:4 decimated chroma 2x2 (Cb +// read 72,72,106,106,... for an input stepping +// by 17, row 1 a duplicate of row 0; all 2073600 +// Cb bytes wrong). +// +// One cause, two faces: InitYCBCRCOPY() hardcoded the shader's luma block to +// 2x2 while the host dispatch was already sized from the OUTPUT format's +// chroma subsampling, so shader and dispatch disagreed for every non-4:2:0 +// output. See the comment at that call site in VulkanFilterYuvCompute.cpp. +// Both planes are now byte-exact against the CPU reference on an A4000 +// (maxdiff=0, psnr=100), and the XPASS arm below is what forced this list to +// be updated in the same change. +// +// Keep the mechanism: the next real filter defect belongs in here, not in a +// deleted assertion. +// ----------------------------------------------------------------------- +struct KnownFilterDefect { + const char* testName; + const char* reason; +}; + +// std::vector, not a C array: a zero-length array is not standard C++, and +// the whole point of this list is that it is allowed to be empty. +const std::vector kKnownFilterDefects = { +}; + +const KnownFilterDefect* FindKnownFilterDefect(const std::string& name) { + for (const auto& entry : kKnownFilterDefects) { + if (name == entry.testName) { + return &entry; + } + } + return nullptr; +} +} // namespace + TestResult FilterTestApp::runTest(const TestCaseConfig& config) { TestResult result; result.testName = config.name; auto startTime = std::chrono::high_resolution_clock::now(); - + + // VkImage handles are recycled once the previous test released its + // resources, so a stale entry here would alias an unrelated image. + m_hostUploadedOptimal.clear(); + std::cout << "[Test] Running: " << config.name << std::endl; // Validate configuration @@ -543,10 +607,8 @@ TestResult FilterTestApp::runTest(const TestCaseConfig& config) { std::vector> inputImages; std::vector> inputImageViews; std::vector> inputBuffers; - - // Bytes actually written into input slot 0, kept so the CPU reference model can be - // computed from the same data the shader reads. std::vector firstInputPattern; + std::vector firstInputReference; for (const auto& inputSlot : config.inputs) { VkSharedBaseObj image; @@ -564,11 +626,14 @@ TestResult FilterTestApp::runTest(const TestCaseConfig& config) { inputImageViews.push_back(imageView); inputBuffers.push_back(buffer); - // Generate test pattern + // Generate test pattern. Capture the first input twice: the staged bytes, + // which are uploaded verbatim, and the same picture in logical component + // order, which is what the CPU reference conversion is computed from. if (inputSlot.generateTestPattern) { + const bool isFirstInput = (&inputSlot == &config.inputs[0]); vkResult = generateTestPattern(inputSlot, image, buffer, - (&inputSlot == &config.inputs[0]) ? &firstInputPattern - : nullptr); + isFirstInput ? &firstInputPattern : nullptr, + isFirstInput ? &firstInputReference : nullptr); if (vkResult != VK_SUCCESS) { result.errorMessage = "Failed to generate test pattern"; result.passed = false; @@ -651,6 +716,60 @@ TestResult FilterTestApp::runTest(const TestCaseConfig& config) { VkCommandBufferBeginInfo beginInfo{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; beginInfo.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; m_vkDevCtx.BeginCommandBuffer(cmdBuffer, &beginInfo); + + // ---- Put every filter image into VK_IMAGE_LAYOUT_GENERAL. ---- + // + // This harness recorded NO layout transition at all, so every image + // reached vkCmdDispatch in the layout it was created in, while the + // filter's descriptors -- correctly -- declare GENERAL, the only layout a + // VK_DESCRIPTOR_TYPE_STORAGE_IMAGE descriptor admits + // (VUID-VkDescriptorImageInfo-imageLayout-00344). That is + // VUID-vkCmdDraw-None-09600, 501 per --all run, and it is a defect in the + // TEST, not in the filter: the filter is handed images by its caller and + // cannot know what layout they are in. The encoder's caller + // (VkVideoEncoder::StageInputFrame) records exactly these two barriers. + // + // oldLayout is read from the create info rather than assumed, so the + // PREINITIALIZED (host-written LINEAR) and UNDEFINED (device-local + // OPTIMAL) cases each name the layout the image is actually in. Naming + // UNDEFINED for a PREINITIALIZED image would validate, and would discard + // the test pattern -- the failure this exists to avoid, not to trade for. + { + std::vector toGeneral; + auto addBarrier = [&toGeneral, this](const VkSharedBaseObj& img) { + if (img == nullptr) { + return; + } + VkImageMemoryBarrier b{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + b.srcAccessMask = VK_ACCESS_HOST_WRITE_BIT; + b.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT; + b.oldLayout = img->GetImageCreateInfo().initialLayout; + b.newLayout = VK_IMAGE_LAYOUT_GENERAL; + b.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + b.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + b.image = img->GetImage(); + // COLOR_BIT covers every plane of a multi-planar image, which is + // what the per-plane storage views are built over. + b.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + toGeneral.push_back(b); + }; + for (const auto& img : inputImages) { + addBarrier(img); + } + for (const auto& img : outputImages) { + addBarrier(img); + } + if (!toGeneral.empty()) { + m_vkDevCtx.CmdPipelineBarrier(cmdBuffer, + VK_PIPELINE_STAGE_HOST_BIT, + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + 0, + 0, nullptr, + 0, nullptr, + (uint32_t)toGeneral.size(), + toGeneral.data()); + } + } // Record filter commands auto* yuvFilter = static_cast(filter.get()); @@ -852,9 +971,9 @@ TestResult FilterTestApp::runTest(const TestCaseConfig& config) { // get any bytes back", which a filter that writes garbage -- or writes only // part of the frame -- passes just as happily as a correct one. std::vector referenceData = - generateReferenceOutput(config, firstInputPattern); + generateReferenceOutput(config, firstInputReference); - if (referenceData.empty() && !firstInputPattern.empty()) { + if (referenceData.empty() && !firstInputReference.empty()) { // No CPU model for this conversion yet. Say so instead of reporting a // pass: an unvalidated case must not look like a validated one. result.passed = false; @@ -893,13 +1012,57 @@ TestResult FilterTestApp::runTest(const TestCaseConfig& config) { result.passed = true; } + // A case that is REQUIRED to disagree with the reference. See + // TestCaseConfig::expectReferenceMismatch. Applied before the + // known-defect registry because it is a property of the configuration, + // not a defect awaiting a fix. + if (config.expectReferenceMismatch) { + if (result.passed) { + std::cout << "[UNEXPECTED-MATCH] " << config.name + << ": this configuration cannot be colour-correct, yet it " + "matched the reference." << std::endl; + result.errorMessage = + "Expected a reference MISMATCH (the configuration under test " + "cannot reproduce the source colours) but the output matched. " + "Re-derive the premise before trusting this."; + result.passed = false; + } else { + std::cout << "[EXPECTED-MISMATCH] " << config.name + << ": disagreed with the reference, as required (observed: " + << result.errorMessage << ")" << std::endl; + result.errorMessage = + "Expected mismatch confirmed: " + result.errorMessage; + result.passed = true; + } + } + + // Known filter defects: XFAIL is tolerated, XPASS is not. See + // kKnownFilterDefects above for why this is a registry and not a mute. + const KnownFilterDefect* known = FindKnownFilterDefect(config.name); + if (known != nullptr) { + if (!result.passed) { + std::cout << "[XFAIL] " << config.name << ": known filter defect -- " + << known->reason << " (observed: " << result.errorMessage << ")" + << std::endl; + result.errorMessage = "XFAIL (known filter defect): " + std::string(known->reason); + result.passed = true; + } else { + std::cout << "[XPASS] " << config.name + << ": listed in kKnownFilterDefects but PASSED." << std::endl; + result.errorMessage = + "XPASS: this case is listed in kKnownFilterDefects but now passes. " + "If the filter was fixed, delete its entry from that list."; + result.passed = false; + } + } + auto endTime = std::chrono::high_resolution_clock::now(); result.executionTimeMs = std::chrono::duration(endTime - startTime).count(); std::cout << "[Test] " << config.name << ": " << (result.passed ? "PASSED" : (result.unvalidated ? "UNVALIDATED" : "FAILED")) << " (" << result.executionTimeMs << " ms)" << std::endl; - + return result; } @@ -946,9 +1109,9 @@ void FilterTestApp::printSummary(const std::vector& results) { << ", Failed: " << failed << ", Unvalidated: " << unvalidated << std::endl; if (passed == 0) { - // A run with nothing validated is not a green run, whatever the failure count - // says: a suite that checks no pixels at all reports the same totals as one that - // checks them and finds them right. Call it out. + // A run with nothing validated is not a green run, whatever the failure + // count says: a suite can report every case passing while checking no + // pixels at all. Call it out. std::cout << "WARNING: no case in this run validated its output pixels." << std::endl; } @@ -1068,8 +1231,17 @@ VkResult FilterTestApp::createTestInput(const TestIOSlot& slot, subresRange.baseArrayLayer = 0; subresRange.layerCount = 1; - result = VkImageResourceView::Create(&m_vkDevCtx, outImage, subresRange, - VK_IMAGE_USAGE_STORAGE_BIT, outImageView); + // The compute filter binds every image slot it is handed -- the + // single-plane combined view and the per-plane YCbCr views alike -- + // as a VK_DESCRIPTOR_TYPE_STORAGE_IMAGE, so storage is the only + // access form the views have to carry. The usage a view requests must + // be a subset of the usage its image was created with + // (VUID-VkImageViewCreateInfo-pNext-02662), which is what ties this + // to |kTestImageUsage|: the two are one declaration expressed twice + // and must be read together. + result = VkImageResourceView::Create(&m_vkDevCtx, outImage, subresRange, + VK_IMAGE_USAGE_STORAGE_BIT, + outImageView); if (result != VK_SUCCESS) { return result; } @@ -1113,22 +1285,78 @@ VkResult FilterTestApp::createStagingBuffer(size_t size, ); } +// Per-plane description of the packed host pattern buffer. Mirrors exactly +// the layout generateTestPattern() produces and calculateImageSize() +// measures: planes back to back, each tightly packed, no padding. +namespace { +struct PlaneCopyDesc { + VkImageAspectFlagBits aspect; + uint32_t width; // plane width, in plane texels + uint32_t height; // plane height, in plane texels + uint32_t texelBytes; // bytes per plane texel +}; + +uint32_t describePlanes(TestFormat format, uint32_t width, uint32_t height, + PlaneCopyDesc planes[3]) { + switch (format) { + case TestFormat::RGBA8: + case TestFormat::BGRA8: + case TestFormat::Y410: // packed 4:4:4, single plane + planes[0] = {VK_IMAGE_ASPECT_COLOR_BIT, width, height, 4}; + return 1; + case TestFormat::NV12: + planes[0] = {VK_IMAGE_ASPECT_PLANE_0_BIT, width, height, 1}; + planes[1] = {VK_IMAGE_ASPECT_PLANE_1_BIT, width / 2, height / 2, 2}; + return 2; + case TestFormat::P010: + case TestFormat::P012: + planes[0] = {VK_IMAGE_ASPECT_PLANE_0_BIT, width, height, 2}; + planes[1] = {VK_IMAGE_ASPECT_PLANE_1_BIT, width / 2, height / 2, 4}; + return 2; + case TestFormat::I420: + planes[0] = {VK_IMAGE_ASPECT_PLANE_0_BIT, width, height, 1}; + planes[1] = {VK_IMAGE_ASPECT_PLANE_1_BIT, width / 2, height / 2, 1}; + planes[2] = {VK_IMAGE_ASPECT_PLANE_2_BIT, width / 2, height / 2, 1}; + return 3; + case TestFormat::NV16: + planes[0] = {VK_IMAGE_ASPECT_PLANE_0_BIT, width, height, 1}; + planes[1] = {VK_IMAGE_ASPECT_PLANE_1_BIT, width / 2, height, 2}; + return 2; + case TestFormat::P210: + planes[0] = {VK_IMAGE_ASPECT_PLANE_0_BIT, width, height, 2}; + planes[1] = {VK_IMAGE_ASPECT_PLANE_1_BIT, width / 2, height, 4}; + return 2; + case TestFormat::YUV444: + planes[0] = {VK_IMAGE_ASPECT_PLANE_0_BIT, width, height, 1}; + planes[1] = {VK_IMAGE_ASPECT_PLANE_1_BIT, width, height, 1}; + planes[2] = {VK_IMAGE_ASPECT_PLANE_2_BIT, width, height, 1}; + return 3; + default: + return 0; + } +} +} // namespace + VkResult FilterTestApp::generateTestPattern(const TestIOSlot& slot, VkSharedBaseObj& image, VkSharedBaseObj& buffer, - std::vector* pOutPatternData) { + std::vector* pOutPatternData, + std::vector* pOutReferencePattern) { std::vector patternData; // Generate test pattern based on input format switch (slot.format) { case TestFormat::RGBA8: case TestFormat::BGRA8: { - // Generate RGBA color bars pattern - generateRGBATestPattern(TestPatternType::ColorBars, + // Generate the pattern in LOGICAL R,G,B,A byte order for BOTH + // formats. The BGRA staging swap happens below, after + // |outPatternData| has been handed the logical bytes -- see the + // comment there for why the two must differ. + generateRGBATestPattern(slot.pattern, slot.width, slot.height, patternData); break; } - + case TestFormat::NV12: case TestFormat::I420: { // Generate NV12 test pattern by converting from RGBA @@ -1196,6 +1424,34 @@ VkResult FilterTestApp::generateTestPattern(const TestIOSlot& slot, } } + // BGRA8: stage the SAME PICTURE, written the way that format spells it. + // + // This is the whole point of the BGRA cases, so it is worth being exact + // about what is being asserted. The pattern above is in logical R,G,B,A + // byte order. A VK_FORMAT_B8G8R8A8_UNORM image spells the identical colour + // with bytes 0 and 2 exchanged. So: + // + // * |patternData| (swapped below) is uploaded -- the bytes a real BGRA + // producer would hand us. + // * |outPatternData| keeps the LOGICAL RGBA bytes, and the CPU reference + // is computed from those. + // + // Consequence, and this is the test: a correct implementation must produce + // BYTE-IDENTICAL YCbCr for the RGBA8 and BGRA8 cases, because they are the + // same picture. An implementation that reads BGRA memory as if it were + // RGBA -- exactly what a `rgba8` storage qualifier on a BGRA view does -- + // produces the red/blue-swapped picture and fails against the shared + // reference. Feeding the swapped bytes to a swapped reference would cancel + // the two errors out and assert nothing at all. + const bool stageAsBgra = (slot.format == TestFormat::BGRA8) && slot.bgraStageSwap; + std::vector logicalRgbaPattern; + if (stageAsBgra && !patternData.empty()) { + logicalRgbaPattern = patternData; + for (size_t i = 0; (i + 3) < patternData.size(); i += 4) { + std::swap(patternData[i + 0], patternData[i + 2]); + } + } + // Upload pattern data to resource if (buffer && !patternData.empty()) { VkDeviceSize maxSize; @@ -1209,20 +1465,36 @@ VkResult FilterTestApp::generateTestPattern(const TestIOSlot& slot, // Optimal-tiled images cannot be written from the host at all. Their upload is a // staging buffer plus a vkCmdCopyBufferToImage, which runTest() sets up as the // filter's pre-transfer -- the staging buffer has to outlive this function, and the - // copy has to be in the same submission as the compute dispatch. Neither is possible - // from here, which is why this function only produces the bytes: a staging buffer - // filled and dropped on return feeds the shader uninitialised memory, and the case - // still reports success. + // copy has to be in the same submission as the compute dispatch. A staging + // buffer filled and then dropped on return, with the copy left undone, feeds + // the shader uninitialised memory while the case still reports success -- + // which is why the buffer's lifetime and the submission are arranged here + // rather than locally. // Note: linear images are NOT written through their host mapping here. See the // staging path in runTest() -- a direct memcpy assumes the planes are tightly packed, // which is not true on every driver. - // Hand back exactly what was written, so the reference model is computed from the - // same bytes the shader will read rather than from a regenerated pattern. + // Two answers, and they differ for exactly one slot format. Handing back a single + // vector for both jobs is wrong for BGRA8 in whichever direction it is resolved. + // + // |pOutPatternData| the bytes AS STAGED. The upload copies these into the + // image, so this is the picture the shader really reads, + // spelled the way the slot's VkFormat spells it. + // |pOutReferencePattern| the same picture in LOGICAL R,G,B,A order. The reference + // generators read pixel[0] as red, so giving them the + // exchanged bytes would describe the byte order instead of + // the picture -- and the exchange would cancel against + // itself, leaving the comparison asserting nothing. + // + // They coincide for every format that stages its bytes in logical order, which is + // every format except a BGRA8 slot with bgraStageSwap set. if (pOutPatternData != nullptr) { *pOutPatternData = patternData; } + if (pOutReferencePattern != nullptr) { + *pOutReferencePattern = stageAsBgra ? logicalRgbaPattern : patternData; + } return VK_SUCCESS; } @@ -1252,6 +1524,81 @@ TestResult FilterTestApp::validateOutput(const TestCaseConfig& config, // packed. Everything arrives through the readback staging buffer instead. (void)outputImage; + // --- Readback fingerprint ------------------------------------------- + // The reference-comparison arm below is only reached when a caller + // supplies reference bytes, and runTest() passes an empty vector, so a + // PASS from this function says "some data came back", never "the right + // data came back". That is not enough to tell a working descriptor arm + // from one that binds a never-written descriptor set, nor a BT.709 + // matrix from a BT.2020 one -- both still produce a full buffer. + // + // So emit a checksum of the raw readback unconditionally, and dump the + // bytes when VKFT_DUMP names a directory. An A/B that must change (a + // matrix or range fix) and an A/B that must NOT change (swapping the + // descriptor arm) are then both decidable from the same output. + if (!actualData.empty()) { + unsigned long long sum = 0; + unsigned int fnv = 2166136261u; + for (size_t i = 0; i < actualData.size(); i++) { + sum += actualData[i]; + fnv = (fnv ^ actualData[i]) * 16777619u; + } + std::cout << "[CHECKSUM] " << config.name + << " bytes=" << actualData.size() + << " sum=" << sum + << " fnv1a=" << fnv << std::endl; + + const char* dumpDir = getenv("VKFT_DUMP"); + if (dumpDir && dumpDir[0]) { + std::string path = std::string(dumpDir) + "/" + config.name + ".bin"; + FILE* f = fopen(path.c_str(), "wb"); + if (f) { + fwrite(actualData.data(), 1, actualData.size(), f); + fclose(f); + std::cout << "[DUMP] " << path << std::endl; + } else { + std::cerr << "[DUMP] FAILED to open " << path << std::endl; + } + } + } + + // Report the raw agreement between GPU output and CPU reference before + // any tolerance is applied, so a threshold argument can be had against + // numbers rather than against a bare PASS/FAIL. + if (!referenceData.empty() && !actualData.empty()) { + // Report in the format's OWN sample units. Reporting a 16-bit-per- + // sample format byte-wise made this line disagree with the verdict it + // is supposed to explain: TC002_RGBA_to_P010 printed "maxdiff=255 + // psnr=10.26" and then PASSED, because a 107/65535 disagreement lands + // almost entirely in the low byte and reads as a 255-unit byte error. + // A diagnostic that contradicts the gate is worse than none. + const bool is16Bit = formatIs16BitSamples(outputSlot.format); + const size_t bytes = std::min(actualData.size(), referenceData.size()); + const size_t n = is16Bit ? (bytes / 2) : bytes; + const uint8_t* ab = actualData.data(); + const uint8_t* rb = referenceData.data(); + uint32_t maxDiff = 0; + double sumSq = 0.0; + for (size_t i = 0; i < n; i++) { + const int av = is16Bit ? (ab[2 * i] | (ab[2 * i + 1] << 8)) : (int)ab[i]; + const int rv = is16Bit ? (rb[2 * i] | (rb[2 * i + 1] << 8)) : (int)rb[i]; + const int d = av - rv; + const uint32_t ad = (uint32_t)(d < 0 ? -d : d); + if (ad > maxDiff) { + maxDiff = ad; + } + sumSq += (double)d * (double)d; + } + const double peak = is16Bit ? 65535.0 : 255.0; + const double mse = (n != 0) ? (sumSq / (double)n) : 0.0; + const double psnr = (mse == 0.0) ? 100.0 : 10.0 * std::log10((peak * peak) / mse); + std::cout << "[REFCMP] " << config.name + << " unit=" << (is16Bit ? "u16" : "u8") + << " n=" << n + << " maxdiff=" << maxDiff + << " psnr=" << psnr << std::endl; + } + // If we have reference data, compare if (!referenceData.empty() && !actualData.empty()) { // Determine comparison method based on output format @@ -1529,9 +1876,27 @@ std::vector FilterTestApp::generateReferenceOutput(const TestCaseConfig // For clear, generate expected cleared values size_t size = calculateImageSize(output.format, output.width, output.height); referenceData.resize(size); - - // Initialize with 50% gray for Y/R=0.5, and neutral for CbCr=0.5 (128 for 8-bit) - std::fill(referenceData.begin(), referenceData.end(), 128); + + if (formatIs16BitSamples(output.format)) { + // The CLEAR shader does imageStore(..., vec4(0.5, ...)) into a + // plane view whose storage is 16 bits per sample, so the value + // that lands is round(0.5 * 65535) = 32767 -- measured 32767 + // on an A4000 for every luma AND chroma sample of TC051. + // Filling BYTES with 128 made the reference 0x8080 = 32896, + // which is not 50% of anything; it is the same units mistake + // convertRGBAtoP010() makes, in the other direction. The + // 16-bit comparison arm above tolerates the +-1 LSB that + // another driver's rounding could produce (a 1/65535 error is + // ~96 dB, far above the 30 dB gate). + const uint16_t mid = 32767; + for (size_t i = 0; i + 1 < size; i += 2) { + referenceData[i] = (uint8_t)(mid & 0xFF); + referenceData[i + 1] = (uint8_t)((mid >> 8) & 0xFF); + } + } else { + // 50% gray for Y=0.5 and neutral CbCr=0.5, i.e. 128 for 8-bit. + std::fill(referenceData.begin(), referenceData.end(), 128); + } break; } @@ -1542,10 +1907,177 @@ std::vector FilterTestApp::generateReferenceOutput(const TestCaseConfig return referenceData; } -VkResult FilterTestApp::copyImageToStagingBuffer(VkSharedBaseObj& image, +// --------------------------------------------------------------------------- +// Read an OPTIMAL-tiled image back into a host-visible staging buffer. +// +// This was a `return VK_SUCCESS;` stub with no callers at all, and runTest() +// took that as licence to short-circuit every optimal-tiled OUTPUT to +// passed = true WITHOUT READING A BYTE -- 46 of the 54 --all cases. The 8 +// linear-output cases were the only ones validating anything. A green that +// cannot go red is the same defect the input side had, pointing the other +// way. +// +// Two things the input-side fix paid for, applied here rather than +// rediscovered: +// +// 1. oldLayout is VK_IMAGE_LAYOUT_GENERAL, NOT the create-info +// VK_IMAGE_LAYOUT_UNDEFINED. runTest()'s barrier puts every filter image +// into GENERAL before the dispatch, and the filter's own trailing +// barrier (VulkanFilterYuvCompute::RecordCommandBuffer, the +// GENERAL->GENERAL / SHADER_WRITE->MEMORY_READ one) leaves the output +// there. Naming the create-info UNDEFINED would validate cleanly and +// would be free to DISCARD the very pixels this function exists to read +// -- the symmetric twin of the hazard that nearly ate the input pattern. +// +// 2. The DESTINATION packing is ours to pick, so pick tight: +// bufferRowLength / bufferImageHeight = 0 means "tightly packed to +// imageExtent". A copy-to-buffer does not inherit the image's row +// padding, so this sidesteps the trap the LINEAR readback had to solve +// with GetPlaneLayout() -- there, luma was padded 2073600 -> 2 MiB and a +// flat read put the whole chroma plane 23552 bytes out. Here the result +// is byte-for-byte what calculateImageSize() measures. +// --------------------------------------------------------------------------- +VkResult FilterTestApp::copyImageToStagingBuffer(const TestIOSlot& slot, + VkSharedBaseObj& image, VkSharedBaseObj& stagingBuffer) { - // TODO: Implement image-to-buffer copy for optimal tiled images - // For now, this is a stub that returns success since we're mainly using linear images + if (image == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + PlaneCopyDesc planes[3]; + const uint32_t numPlanes = describePlanes(slot.format, slot.width, slot.height, planes); + if (numPlanes == 0) { + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + // Each region's bufferOffset must be a multiple of the PLANE's texel + // size (VUID-vkCmdCopyImageToBuffer-srcImage-07976), and tight packing of + // the earlier planes does not guarantee that: TC101 is 1921x1081, so the + // NV12 luma plane is 2076601 bytes and plane 1's texel size is 2. So the + // copy is recorded at ALIGNED offsets and the planes are compacted down + // to tight packing on the host once the fence signals -- the caller still + // gets the tightly packed buffer it expects, and the copy stays legal. + VkDeviceSize alignedOffset[3] = {0, 0, 0}; + VkDeviceSize tightOffset[3] = {0, 0, 0}; + VkDeviceSize planeBytes[3] = {0, 0, 0}; + VkDeviceSize alignedCursor = 0; + VkDeviceSize tightCursor = 0; + for (uint32_t i = 0; i < numPlanes; i++) { + const VkDeviceSize align = (planes[i].texelBytes != 0) ? planes[i].texelBytes : 1; + alignedCursor = ((alignedCursor + align - 1) / align) * align; + alignedOffset[i] = alignedCursor; + tightOffset[i] = tightCursor; + planeBytes[i] = (VkDeviceSize)planes[i].width * planes[i].height * planes[i].texelBytes; + alignedCursor += planeBytes[i]; + tightCursor += planeBytes[i]; + } + if (tightCursor == 0) { + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + VkResult result = createStagingBuffer((size_t)alignedCursor, stagingBuffer); + if (result != VK_SUCCESS) { + return result; + } + + VkCommandBufferAllocateInfo allocInfo{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + allocInfo.commandPool = m_commandPool; + allocInfo.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; + allocInfo.commandBufferCount = 1; + + VkCommandBuffer cmdBuffer = VK_NULL_HANDLE; + result = m_vkDevCtx.AllocateCommandBuffers(m_vkDevCtx.getDevice(), &allocInfo, &cmdBuffer); + if (result != VK_SUCCESS) { + return result; + } + + VkCommandBufferBeginInfo beginInfo{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + beginInfo.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + m_vkDevCtx.BeginCommandBuffer(cmdBuffer, &beginInfo); + + // GENERAL, not the create-info UNDEFINED -- see (1) in the block above. + VkImageMemoryBarrier toTransferSrc{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + toTransferSrc.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT; + toTransferSrc.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + toTransferSrc.oldLayout = VK_IMAGE_LAYOUT_GENERAL; + toTransferSrc.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + toTransferSrc.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + toTransferSrc.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + toTransferSrc.image = image->GetImage(); + // COLOR_BIT covers every plane of a multi-planar image. + toTransferSrc.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + m_vkDevCtx.CmdPipelineBarrier(cmdBuffer, + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &toTransferSrc); + + VkBufferImageCopy regions[3]{}; + for (uint32_t i = 0; i < numPlanes; i++) { + regions[i].bufferOffset = alignedOffset[i]; + regions[i].bufferRowLength = 0; // tightly packed -- see (2) above + regions[i].bufferImageHeight = 0; + regions[i].imageSubresource.aspectMask = planes[i].aspect; + regions[i].imageSubresource.mipLevel = 0; + regions[i].imageSubresource.baseArrayLayer = 0; + regions[i].imageSubresource.layerCount = 1; + regions[i].imageOffset = {0, 0, 0}; + regions[i].imageExtent = {planes[i].width, planes[i].height, 1}; + } + + m_vkDevCtx.CmdCopyImageToBuffer(cmdBuffer, + image->GetImage(), + VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + stagingBuffer->GetBuffer(), + numPlanes, regions); + + // HOST_COHERENT memory needs no cache maintenance, but the transfer write + // still has to be made AVAILABLE to the host domain. A fence wait alone + // does not do that. + VkMemoryBarrier toHost{VK_STRUCTURE_TYPE_MEMORY_BARRIER}; + toHost.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + toHost.dstAccessMask = VK_ACCESS_HOST_READ_BIT; + m_vkDevCtx.CmdPipelineBarrier(cmdBuffer, + VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_PIPELINE_STAGE_HOST_BIT, + 0, 1, &toHost, 0, nullptr, 0, nullptr); + + m_vkDevCtx.EndCommandBuffer(cmdBuffer); + + VkFence fence = VK_NULL_HANDLE; + VkFenceCreateInfo fenceInfo{VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + m_vkDevCtx.CreateFence(m_vkDevCtx.getDevice(), &fenceInfo, nullptr, &fence); + + VkSubmitInfo submitInfo{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + submitInfo.commandBufferCount = 1; + submitInfo.pCommandBuffers = &cmdBuffer; + + result = m_vkDevCtx.QueueSubmit(m_vkDevCtx.GetComputeQueue(), 1, &submitInfo, fence); + if (result == VK_SUCCESS) { + result = m_vkDevCtx.WaitForFences(m_vkDevCtx.getDevice(), 1, &fence, VK_TRUE, UINT64_MAX); + } + + m_vkDevCtx.DestroyFence(m_vkDevCtx.getDevice(), fence, nullptr); + m_vkDevCtx.FreeCommandBuffers(m_vkDevCtx.getDevice(), m_commandPool, 1, &cmdBuffer); + + if (result != VK_SUCCESS) { + return result; + } + + // Compact the aligned plane layout down to tight packing. Only ever moves + // bytes DOWNWARD, and only when an alignment gap was actually inserted, + // so for every even-dimensioned case this is a no-op. + VkDeviceSize mappedSize = 0; + uint8_t* mapped = stagingBuffer->GetDataPtr(0, mappedSize); + if ((mapped == nullptr) || (mappedSize < alignedCursor)) { + return VK_ERROR_MEMORY_MAP_FAILED; + } + for (uint32_t i = 1; i < numPlanes; i++) { + if (alignedOffset[i] != tightOffset[i]) { + memmove(mapped + tightOffset[i], mapped + alignedOffset[i], + (size_t)planeBytes[i]); + } + } + return VK_SUCCESS; } diff --git a/common/libs/tests/src/TestCases.cpp b/common/libs/tests/src/TestCases.cpp index 1ab1c54f..171bf375 100644 --- a/common/libs/tests/src/TestCases.cpp +++ b/common/libs/tests/src/TestCases.cpp @@ -176,13 +176,13 @@ TestCaseConfig TC001_RGBA_to_NV12() { // case actually exercises the shader: input pattern in, filter runs, output compared // against convertRGBAtoNV12() to within the configured tolerance. // -// Keep at least one linear case in every suite. A suite with no validated case reports -// only that submission did not error, which is a green run over a filter that could be -// writing anything at all. -// Linear 4:2:2 control. This is the case that pins the shader's luma block ratio to the -// dispatch grid: a block size that disagrees with the grid leaves part of the 4:2:2 -// chroma plane unwritten, which a PSNR comparison against convertRGBAtoNV16() catches -// and a "did it submit" check does not. +// Keep at least one linear case in every suite. Without one, a suite of +// optimal-tiled cases reports only that submission did not error, which is +// indistinguishable from a suite that checks nothing. +// Linear 4:2:2 control. This is the case that verifies the block ratio and the +// dispatch: a hard-wired 2x2 block against a (w/2, h) dispatch leaves the +// bottom half of the 4:2:2 chroma plane unwritten, which a PSNR comparison +// against convertRGBAtoNV16() catches and a "did it submit" check does not. TestCaseConfig TC005L_RGBA_to_NV16_Linear() { TestCaseConfig config = createRGBA2YCbCr("TC005L_RGBA_to_NV16_Linear", TestFormat::NV16, @@ -276,7 +276,7 @@ TestCaseConfig TC006_RGBA_to_P210() { // P212 is the exact 12-bit counterpart of TC006/TC015's P210: same 2-plane 4:2:2 layout, // same 16-bit container, same R16/R16G16 plane views -- only the X4-vs-X6 padding differs. // It is here because requesting this format as a compute-filter output HANGS the GPU in TRV -// (task #19). Running it in this harness is safe: isFormatSupported() does a per-plane +// Running it in this harness is safe: isFormatSupported() does a per-plane // feature check AND an image-level vkGetPhysicalDeviceImageFormatProperties for the exact // image, and reports the case "unvalidated" rather than submitting work that wedges the GPU. TestCaseConfig TC006b_RGBA_to_P212() { @@ -844,6 +844,64 @@ TestCaseConfig TC080_RGBA_to_NV12_Linear() { return config; } +TestCaseConfig TC092_RGBA_to_NV12_Linear_LimitedRange() { + TestCaseConfig config = createRGBA2YCbCr("TC092_RGBA_to_NV12_Linear_LimitedRange", + TestFormat::NV12, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709, + VK_SAMPLER_YCBCR_RANGE_ITU_NARROW); + // Linear INPUT too: the optimal-tiled upload path is a documented TODO + // (FilterTestApp.cpp), so an optimal RGBA input arrives as all-zero black + // -- and black converts identically under every matrix, which is exactly + // the case a colour-matrix test must not be. + config.inputs[0].tiling = TilingMode::Linear; + config.outputs[0].tiling = TilingMode::Linear; + return config; +} + +TestCaseConfig TC096_RGBA_to_NV12_Linear_BT709_FullRange() { + // The matched baseline for TC092/TC093/TC094: same linear input, same + // linear output, BT.709 full range. Holding everything but one variable + // is what makes those three decidable -- TC080 cannot serve, because its + // optimal-tiled input arrives black. + TestCaseConfig config = createRGBA2YCbCr("TC096_RGBA_to_NV12_Linear_BT709_FullRange", + TestFormat::NV12, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709, + VK_SAMPLER_YCBCR_RANGE_ITU_FULL); + config.inputs[0].tiling = TilingMode::Linear; + config.outputs[0].tiling = TilingMode::Linear; + return config; +} + +TestCaseConfig TC093_RGBA_to_NV12_Linear_BT601() { + TestCaseConfig config = createRGBA2YCbCr("TC093_RGBA_to_NV12_Linear_BT601", + TestFormat::NV12, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_601, + VK_SAMPLER_YCBCR_RANGE_ITU_FULL); + config.inputs[0].tiling = TilingMode::Linear; + config.outputs[0].tiling = TilingMode::Linear; + return config; +} + +TestCaseConfig TC094_RGBA_to_NV12_Linear_BT2020() { + TestCaseConfig config = createRGBA2YCbCr("TC094_RGBA_to_NV12_Linear_BT2020", + TestFormat::NV12, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_2020, + VK_SAMPLER_YCBCR_RANGE_ITU_FULL); + config.inputs[0].tiling = TilingMode::Linear; + config.outputs[0].tiling = TilingMode::Linear; + return config; +} + +TestCaseConfig TC095_RGBA_to_P010_Linear_LimitedRange() { + TestCaseConfig config = createRGBA2YCbCr("TC095_RGBA_to_P010_Linear_LimitedRange", + TestFormat::P010, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709, + VK_SAMPLER_YCBCR_RANGE_ITU_NARROW); + config.inputs[0].tiling = TilingMode::Linear; + config.outputs[0].tiling = TilingMode::Linear; + return config; +} + TestCaseConfig TC081_RGBA_to_P010_Linear() { TestCaseConfig config = createRGBA2YCbCr("TC081_RGBA_to_P010_Linear", TestFormat::P010); config.outputs[0].tiling = TilingMode::Linear; @@ -1209,6 +1267,116 @@ TestCaseConfig TC104_Minimum_Resolution_2x2() { VK_SAMPLER_YCBCR_RANGE_ITU_FULL, 2, 2); } +// ODD extents, which no 4:2:0 case can carry: Vulkan requires each dimension to be a +// multiple of that axis's chroma subsampling, so an odd extent has to be carried by a +// format that does not subsample that axis. 4:4:4 subsamples neither, 4:2:2 subsamples +// width only, and between them the two cases below put an odd extent on both axes. +// +// An odd extent is the one shape in which the dispatch grid cannot be derived by +// halving: the last column and the last row are covered by a partial block and a +// partial workgroup, so a grid or a bounds check that rounds the wrong way drops them +// entirely rather than merely mis-sizing the work. +// +// SMALL ON PURPOSE. The verdict for these formats is a frame-average PSNR against a +// 30 dB threshold, and an average dilutes an edge: one dropped column out of 1921 is +// 40 dB and passes, while the same defect on 65 is 19 dB and fails. The extents are +// therefore the smallest that are still odd and still not a multiple of the workgroup +// or block size, so the edge is a large enough fraction of the frame for the gate to +// resolve it. Large unaligned extents are covered by the 4:2:0 case above. +TestCaseConfig TC105_Odd_Resolution_65x33_YUV444() { + return createRGBA2YCbCr("TC105_Odd_Resolution_65x33_YUV444", TestFormat::YUV444, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709, + VK_SAMPLER_YCBCR_RANGE_ITU_FULL, 65, 33); +} + +TestCaseConfig TC106_Odd_Height_66x33_NV16() { + return createRGBA2YCbCr("TC106_Odd_Height_66x33_NV16", TestFormat::NV16, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709, + VK_SAMPLER_YCBCR_RANGE_ITU_FULL, 66, 33); +} + +// ============================================================================= +// RGBA/BGRA component-order tests +// ============================================================================= +// +// WHAT THESE CASES ESTABLISH, and why colour bars cannot. +// +// The filter binds an RGBA input as one VK_DESCRIPTOR_TYPE_STORAGE_IMAGE and reads +// it with imageLoad(). GLSL offers exactly one 8-bit four-component storage format +// qualifier, `rgba8`, so a VK_FORMAT_B8G8R8A8_UNORM view is necessarily declared +// `rgba8`: the declaration cannot name the view's component order. The contract is +// that this does not matter -- component order is a property of the format, not of +// the access, so imageLoad() returns logical R,G,B,A for either format. BGRA8 is +// what a compositor most often produces, so that contract carries real traffic. +// +// All three cases stage THE SAME PICTURE -- four saturated quadrants: red, green, +// blue, white -- and are validated against ONE reference computed from the logical +// RGB values. A BGRA8 slot writes that picture the way its own format spells it, +// bytes 0 and 2 exchanged (TestIOSlot::bgraStageSwap): +// +// TC130 RGBA8, bytes in the format's order : must MATCH (the baseline) +// TC132 BGRA8, bytes in the format's order : must MATCH (the deliverable: +// component order +// resolved from the +// VkFormat) +// TC133 BGRA8, bytes NOT exchanged : must DIFFER (the control) +// +// WHY TC133 IS WHAT MAKES THE OTHER TWO MEAN ANYTHING. TC130 and TC132 both passing +// is also exactly what would be observed if the BGRA staging exchange never happened +// AND the VkFormat were ignored -- two errors cancelling. TC133 keeps the format and +// removes the exchange, so the picture in memory really is red/blue exchanged and the +// output MUST disagree with the reference. +// +// A red/blue exchange conserves the byte histogram exactly, so it is invisible to any +// checksum, size or histogram test of the image alone. Saturated primaries are what +// make it visible: red and blue sit at opposite ends of both chroma axes, so the error +// lands in Cb/Cr where the PSNR comparison reports it. + +static TestCaseConfig createRgbaComponentOrderCase(const char* name, + TestFormat inputFormat, + bool expectMismatch) { + // 256x256 rather than 1920x1080: the quadrant split lands on the 4:2:0 chroma + // grid, so no chroma sample straddles a colour boundary and the expected chroma + // is exact rather than a boundary average; and the frame is small enough to read + // back and compare without dominating the suite's run time. + TestCaseConfig config = createRGBA2YCbCr(name, TestFormat::NV12, + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709, + VK_SAMPLER_YCBCR_RANGE_ITU_FULL, + 256, 256); + config.inputs[0].format = inputFormat; + config.inputs[0].pattern = TestPatternType::PurePrimaryQuadrants; + config.expectReferenceMismatch = expectMismatch; + return config; +} + +TestCaseConfig TC130_RGBA_to_NV12_PurePrimaries_Storage() { + return createRgbaComponentOrderCase("TC130_RGBA_to_NV12_PurePrimaries_Storage", + TestFormat::RGBA8, + /*expectMismatch*/ false); +} + +TestCaseConfig TC132_BGRA_to_NV12_PurePrimaries_Storage() { + // The required result is the YCbCr TC130 produces: the two cases are the same + // picture, written the two ways the two formats spell it. + return createRgbaComponentOrderCase("TC132_BGRA_to_NV12_PurePrimaries_Storage", + TestFormat::BGRA8, + /*expectMismatch*/ false); +} + +TestCaseConfig TC133_BGRA_NoSwapControl_MustDiffer() { + // THE CONTROL. A BGRA8 image deliberately loaded with UNEXCHANGED (logical + // R,G,B,A) bytes. The picture in memory is therefore red/blue exchanged relative + // to the reference, and a pipeline that honours the VkFormat must produce output + // that DISAGREES with it. A pass here is what licenses reading TC130 and TC132 as + // results rather than as a coincidence. + TestCaseConfig config = + createRgbaComponentOrderCase("TC133_BGRA_NoSwapControl_MustDiffer", + TestFormat::BGRA8, + /*expectMismatch*/ true); + config.inputs[0].bgraStageSwap = false; + return config; +} + // ============================================================================= // Test Set Getters // ============================================================================= @@ -1228,7 +1396,21 @@ std::vector getAllStandardTests() { TC005_RGBA_to_NV16(), TC006_RGBA_to_P210(), TC007_RGBA_to_YUV444(), - // TC008_RGBA_to_Y410(), // Disabled: Y410 is packed format, needs special shader + // TC008_RGBA_to_Y410(), // Still disabled, but NO LONGER for the + // reason recorded here before, which was a real defect and is now + // fixed. The RGBA2YCBCR packed-output arm used to declare + // outputImageRGB an image2DArray while storing into it with an ivec2, + // so its GLSL did not compile ("imageStore: no matching overloaded + // function found"). ShaderGenerateImagePlaneDescriptors now declares + // the single-plane arm image2D unconditionally -- what the combined + // VK_IMAGE_VIEW_TYPE_2D view it binds required all along -- and the + // two index forms agree. + // + // Measured after that change: the case builds its pipeline, dispatches, + // and stops at UNVALIDATED -- "no CPU reference model for this format + // pair". That is the remaining work, and it is in + // generateReferenceOutput, not in the shader generator. Enabling the + // case before then would add a row that cannot judge its own pixels. // YCbCr to RGBA -- DISABLED. YCBCR2RGBA has several independent shader-generation // defects; dump the generated GLSL with VK_FILTER_DUMP_SHADERS=1 to see them: @@ -1250,7 +1432,12 @@ std::vector getAllStandardTests() { // TC014_NV16_to_RGBA(), // TC015_P210_to_RGBA(), // TC016_YUV444_to_RGBA(), - // TC017_Y410_to_RGBA(), + // TC017_Y410_to_RGBA(), // Disabled, and not stale. Two faults, in + // this order: the arm derives a bit depth through YcbcrVkFormatInfo, + // which answers NULL for A2B10G10R10_UNORM_PACK32 -- that dereference + // used to crash and is now guarded -- and only then emits + // inputImageY/inputImageCbCr, which a packed input never declares, so + // the shader does not compile. The arm is deprecated besides. // Color primaries (BT.601, BT.709, BT.2020) TC020_RGBA_to_NV12_BT601(), @@ -1309,6 +1496,11 @@ std::vector getAllStandardTests() { // Linear tiling TC080_RGBA_to_NV12_Linear(), TC081_RGBA_to_P010_Linear(), + TC092_RGBA_to_NV12_Linear_LimitedRange(), + TC093_RGBA_to_NV12_Linear_BT601(), + TC094_RGBA_to_NV12_Linear_BT2020(), + TC095_RGBA_to_P010_Linear_LimitedRange(), + TC096_RGBA_to_NV12_Linear_BT709_FullRange(), TC082_Linear_NV12_to_Optimal_NV12(), TC083_Optimal_NV12_to_Linear_NV12(), @@ -1317,12 +1509,19 @@ std::vector getAllStandardTests() { TC090_Dual_Output_Optimal_Linear(), TC091_Triple_Output_with_Subsampled(), + // Component order + TC130_RGBA_to_NV12_PurePrimaries_Storage(), + TC132_BGRA_to_NV12_PurePrimaries_Storage(), + TC133_BGRA_NoSwapControl_MustDiffer(), + // Edge cases TC100_Small_Resolution_64x64(), TC101_Unaligned_Resolution_1922x1082(), TC102_4K_Resolution_3840x2160(), TC103_8K_Resolution_7680x4320(), // May exceed GPU memory TC104_Minimum_Resolution_2x2(), + TC105_Odd_Resolution_65x33_YUV444(), + TC106_Odd_Height_66x33_NV16(), }; } diff --git a/common/libs/tests/src/main.cpp b/common/libs/tests/src/main.cpp index 40acdbfd..b17362e5 100644 --- a/common/libs/tests/src/main.cpp +++ b/common/libs/tests/src/main.cpp @@ -100,6 +100,32 @@ void listTests() { std::cout << std::endl; } + +// --------------------------------------------------------------------------- +// CTest skip semantics. +// +// A host with no usable Vulkan device has proved nothing about this test's +// subject, so it must report SKIPPED (CTest's conventional 77), not a pass and +// not a failure. Only device-AVAILABILITY results map to 77. Every other +// VkResult -- VK_ERROR_DEVICE_LOST, VK_ERROR_OUT_OF_*_MEMORY, anything else -- +// stays a hard exit 1, so a real regression inside init() cannot hide behind +// this arm and get itself reported as "skipped". +// --------------------------------------------------------------------------- +static const int kCTestSkipExitCode = 77; + +static bool IsNoUsableDeviceResult(VkResult result) { + switch (result) { + case VK_ERROR_INCOMPATIBLE_DRIVER: // -9: no ICD the loader can use + case VK_ERROR_INITIALIZATION_FAILED: // -3: loader/ICD refused to come up + case VK_ERROR_EXTENSION_NOT_PRESENT: // -7: no device offers what we need + case VK_ERROR_LAYER_NOT_PRESENT: // -6 + case VK_ERROR_FEATURE_NOT_PRESENT: // -8: device lacks a required feature + return true; + default: + return false; + } +} + int main(int argc, char* argv[]) { bool verbose = false; bool validate = false; @@ -158,6 +184,12 @@ int main(int argc, char* argv[]) { VkResult result = app.init(verbose, validate, deviceUuid.empty() ? nullptr : deviceUuid.c_str()); if (result != VK_SUCCESS) { std::cerr << "Failed to initialize test application: " << result << std::endl; + if (IsNoUsableDeviceResult(result)) { + std::cerr << "No usable Vulkan device on this host: reporting SKIPPED (" + << kCTestSkipExitCode << "). Nothing was proved either way." + << std::endl; + return kCTestSkipExitCode; + } return 1; } diff --git a/common/libs/tests/win32_opaque_import/CMakeLists.txt b/common/libs/tests/win32_opaque_import/CMakeLists.txt index f30d2a9e..e2824102 100644 --- a/common/libs/tests/win32_opaque_import/CMakeLists.txt +++ b/common/libs/tests/win32_opaque_import/CMakeLists.txt @@ -40,6 +40,7 @@ set(TEST_HEADERS set(VKCODECUTILS_DIR "${CMAKE_CURRENT_SOURCE_DIR}/../../VkCodecUtils") set(VKCODECUTILS_SOURCES ${VKCODECUTILS_DIR}/VulkanDeviceContext.cpp + ${VKCODECUTILS_DIR}/VkEncoderStdioLatch.cpp ${VKCODECUTILS_DIR}/VulkanDeviceMemoryImpl.cpp ${VKCODECUTILS_DIR}/VkImageResource.cpp ${VKCODECUTILS_DIR}/VkBufferResource.cpp @@ -54,6 +55,7 @@ set(VKCODECUTILS_SOURCES ${VKCODECUTILS_DIR}/VulkanFenceSet.cpp ${VKCODECUTILS_DIR}/Helpers.cpp ${VK_DISPATCH_TABLE_SOURCE} + ${VK_DISPATCH_TABLE_HEADER} ${VKCODECUTILS_DIR}/nvVkFormats.cpp ) @@ -105,6 +107,12 @@ install(TARGETS ${PROJECT_NAME} enable_testing() add_test(NAME Win32OpaqueImportSmokeTest COMMAND ${PROJECT_NAME}) add_test(NAME Win32OpaqueImportAllTests COMMAND ${PROJECT_NAME} --verbose) +# Labelled for the same CI split as the Linux tests. No SKIP_RETURN_CODE is +# claimed here because this arm is Windows-only and was NOT run or verified +# when the labels were added -- do not assume it skips cleanly without a GPU. +set_tests_properties(Win32OpaqueImportSmokeTest + Win32OpaqueImportAllTests PROPERTIES + LABELS "gpu") # Print status message(STATUS "win32_opaque_import_test: Configured for Windows") diff --git a/scripts/run_ctest_ci.sh b/scripts/run_ctest_ci.sh new file mode 100755 index 00000000..336d7ba1 --- /dev/null +++ b/scripts/run_ctest_ci.sh @@ -0,0 +1,107 @@ +#!/usr/bin/env bash +# +# Run the CTest suite the way CI runs it. +# +# A plain `ctest` is not enough on a GPU-less runner: the suite reports +# 2 passed / 4 skipped / 8 failed there, and a job that is red on arrival is a +# job people learn to ignore. +# +# So the suite is split by LABEL: +# +# device-free needs no Vulkan device; must be GREEN everywhere; gating. +# gpu needs a real GPU; exits 77 (or 2 for the dma-buf import test) +# and is reported SKIPPED when there is no device, so this set +# is also green on a GPU-less runner -- and really gates on a +# runner that has one. +# +# Both sets are gating. `--no-tests=error` means a label typo or an empty set +# fails the job instead of silently passing, and the audit below means a new +# add_test() that forgets its label fails the job too. Both are there because +# the failure mode this script exists to fix is a check that cannot fail. +# +# Usage: scripts/run_ctest_ci.sh [build-dir] (default: BUILD) + +set -u -o pipefail + +BUILD_DIR="${1:-BUILD}" + +if [ ! -f "${BUILD_DIR}/CTestTestfile.cmake" ]; then + echo "ERROR: no CTestTestfile.cmake in '${BUILD_DIR}'." >&2 + echo " Configure with CTest enabled first, e.g." >&2 + echo " cmake -B ${BUILD_DIR} -DCMAKE_BUILD_TYPE=Release" >&2 + exit 1 +fi + +count_tests() { + # $@ are extra ctest args. Read the count out of `ctest -N`, and read the + # exit status of ctest itself rather than of the pipeline tail. + local out status + out="$(ctest --test-dir "${BUILD_DIR}" -N "$@" 2>&1)" + status=$? + if [ "${status}" -ne 0 ]; then + echo "ERROR: 'ctest -N $*' failed:" >&2 + echo "${out}" >&2 + exit 1 + fi + echo "${out}" | sed -n 's/^Total Tests: \([0-9]*\)$/\1/p' +} + +echo "==============================================================" +echo " CTest label audit" +echo "==============================================================" +TOTAL="$(count_tests)" +LABELLED="$(count_tests -L 'device-free|gpu')" + +if [ -z "${TOTAL}" ] || [ -z "${LABELLED}" ]; then + echo "ERROR: could not parse the test counts out of 'ctest -N'." >&2 + exit 1 +fi + +echo "registered tests: ${TOTAL}" +echo "labelled tests : ${LABELLED}" + +if [ "${TOTAL}" -eq 0 ]; then + echo "ERROR: zero tests registered. Either the top-level enable_testing()" >&2 + echo " was dropped or BUILD_TESTS is OFF. A ctest step over an empty" >&2 + echo " suite is a CI step that can never fail -- refusing to pass." >&2 + exit 1 +fi + +if [ "${TOTAL}" -ne "${LABELLED}" ]; then + echo "ERROR: $(( TOTAL - LABELLED )) registered test(s) carry neither the" >&2 + echo " 'device-free' nor the 'gpu' label, so they are in no CI set" >&2 + echo " and gate nothing. Unlabelled:" >&2 + comm -23 \ + <(ctest --test-dir "${BUILD_DIR}" -N | sed -n 's/^ Test *#[0-9]*: //p' | sort) \ + <(ctest --test-dir "${BUILD_DIR}" -N -L 'device-free|gpu' | sed -n 's/^ Test *#[0-9]*: //p' | sort) \ + | sed 's/^/ /' >&2 + exit 1 +fi +echo "OK: every registered test is in exactly one CI set." +echo + +echo "==============================================================" +echo " device-free suite (gating; must be green on any host)" +echo "==============================================================" +ctest --test-dir "${BUILD_DIR}" --output-on-failure --no-tests=error \ + -L '^device-free$' +DEVICE_FREE_STATUS=$? +echo "device-free ctest exit status: ${DEVICE_FREE_STATUS}" +echo + +echo "==============================================================" +echo " gpu suite (gating; all-SKIP when the host has no GPU)" +echo "==============================================================" +ctest --test-dir "${BUILD_DIR}" --output-on-failure --no-tests=error \ + -L '^gpu$' +GPU_STATUS=$? +echo "gpu ctest exit status: ${GPU_STATUS}" +echo + +if [ "${DEVICE_FREE_STATUS}" -ne 0 ] || [ "${GPU_STATUS}" -ne 0 ]; then + echo "FAILED (device-free=${DEVICE_FREE_STATUS} gpu=${GPU_STATUS})" + exit 1 +fi + +echo "PASSED (device-free=0 gpu=0)" +exit 0 diff --git a/tests/decode_samples.json b/tests/decode_samples.json index aedc4196..e5355847 100644 --- a/tests/decode_samples.json +++ b/tests/decode_samples.json @@ -95,7 +95,7 @@ "source_url": "https://storage.googleapis.com/vulkan-video-samples/avc/clip-a.h264", "source_checksum": "b119d5d667eb7791613d0e39f375b61f0fa41baa5ee1f216a678e0483c69333c", "source_filepath": "video/avc/clip-a.h264", - "expected_output_y4m_md5": "472dc5dbe403441861340a5f64946155" + "expected_output_y4m_md5": "83c30cbef96f357bbf7556a3621857a4" }, { "name": "h264_clip_c", diff --git a/tests/libs/video_test_framework_base.py b/tests/libs/video_test_framework_base.py index f629a121..eac9e726 100644 --- a/tests/libs/video_test_framework_base.py +++ b/tests/libs/video_test_framework_base.py @@ -25,6 +25,7 @@ from pathlib import Path from typing import Dict, List, Optional +from tests.libs.video_test_output_check import check_declared_output from tests.libs.video_test_config_base import ( BaseTestConfig, ExpectedResult, @@ -559,8 +560,14 @@ def execute_test_command( config: BaseTestConfig, timeout: int = DEFAULT_TEST_TIMEOUT, cwd: Optional[Path] = None, + output_file: Optional[Path] = None, ) -> TestResult: - """Execute a test command and return result.""" + """Execute a test command and return result. + + |output_file| names the artifact this command was asked to produce, + when it was asked to produce one. A run that names an output and + exits successfully has to have written it; see _check_declared_output. + """ command_line = ' '.join(cmd) if self.verbose: @@ -586,7 +593,7 @@ def execute_test_command( result = subprocess.run(cmd, check=False, **subprocess_kwargs) self._detect_driver_from_output(result.stdout, result.stderr) - return TestResult( + test_result = TestResult( config=config, returncode=result.returncode, stdout=result.stdout, @@ -596,6 +603,9 @@ def execute_test_command( result.returncode, result.stderr), command_line=command_line ) + check_declared_output(test_result, config, output_file, + cmd) + return test_result except subprocess.TimeoutExpired: return create_error_result( @@ -620,11 +630,11 @@ def build_decoder_command( # pylint: disable=too-many-arguments ] # --enablePostProcessFilter takes a filter TYPE, not a boolean. The decoder's # default is -1, which disables the post-process pass; 0 is a legacy value that - # selects the first filter. Do not pass "0" here to mean "off": it routes every - # decode that did not ask for a filter through a compute shader, so a decode - # cell validates decode + filter and a filter defect reads as a decoder defect. - # Omit the option to get the default: the argument parser rejects "-1" because - # it starts with a dash. + # selects the first filter. Passing "0" to mean "off" silently routes every + # decode that did not ask for a filter through a compute shader -- decode cells + # then validate decode + filter, and a filter defect reads as a decoder defect. + # Omit the option to get the default: the argument parser + # rejects "-1" because it starts with a dash. if output_file: cmd.extend(["-o", str(output_file)]) diff --git a/tests/libs/video_test_framework_decode.py b/tests/libs/video_test_framework_decode.py index f37a8a7f..addc6725 100755 --- a/tests/libs/video_test_framework_decode.py +++ b/tests/libs/video_test_framework_decode.py @@ -249,7 +249,8 @@ def _run_decoder_test(self, config: DecodeTestSample) -> TestResult: # Use base class to execute (handles subprocess details) run_cwd = self._default_run_cwd() result = self.execute_test_command( - cmd, config, timeout=self.timeout, cwd=run_cwd + cmd, config, timeout=self.timeout, cwd=run_cwd, + output_file=output_file ) # Verify MD5 if enabled and test succeeded diff --git a/tests/libs/video_test_framework_encode.py b/tests/libs/video_test_framework_encode.py index cf5d7bd8..21b22a12 100755 --- a/tests/libs/video_test_framework_encode.py +++ b/tests/libs/video_test_framework_encode.py @@ -297,7 +297,8 @@ def _run_encoder_test(self, config: EncodeTestSample) -> TestResult: # Use base class to execute (handles subprocess details) run_cwd = self._default_run_cwd() result = self.execute_test_command( - cmd, config, timeout=self.timeout, cwd=run_cwd + cmd, config, timeout=self.timeout, cwd=run_cwd, + output_file=output_file ) # Analyze output diff --git a/tests/libs/video_test_output_check.py b/tests/libs/video_test_output_check.py new file mode 100644 index 00000000..e4b0a2c4 --- /dev/null +++ b/tests/libs/video_test_output_check.py @@ -0,0 +1,116 @@ +"""Did the run produce the output the cell declared? + +Split out of video_test_framework_base so that module stays under the +line ceiling, and because this is a separable question: everything here reads +a finished command line and a finished file, and touches no framework state. + +Copyright 2025 NVIDIA Corporation. +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +""" + +from tests.libs.video_test_config_base import ( + BaseTestConfig, + ExpectedResult, + TestResult, + VideoTestStatus, +) + + +# Flags under which an empty or absent declared output is the result +# that was asked for. They are read off the command that was actually +# issued, so a flag the framework adds counts the same as one a cell +# declared, and no codec and no filename appears here. +# +# --disableFileOutput -- suppresses every write to the bitstream file +# and wins over an -o that also names one, so the path is opened +# and left at zero bytes. It is the flag an in-memory capture uses. +# --numFrames 0 -- asks the encoder for no frames, so there is no +# bitstream for it to write. +_NO_OUTPUT_FLAGS = ("--disableFileOutput",) +_NO_OUTPUT_FLAG_VALUES = (("--numFrames", "0"),) + + +def command_declines_output(cmd) -> bool: + """True when the command itself says it will produce no output.""" + if not cmd: + return False + argv = [str(arg) for arg in cmd] + if any(flag in argv for flag in _NO_OUTPUT_FLAGS): + return True + for flag, value in _NO_OUTPUT_FLAG_VALUES: + for index in range(len(argv) - 1): + if (argv[index] == flag) and (argv[index + 1] == value): + return True + return False + + +def check_declared_output(result: TestResult, + config: BaseTestConfig, + output_file, + cmd=None) -> None: + """Downgrade a successful run that produced nothing. + + A return code says the process reached its own exit, not that it did + the work. An encode that loses the device, or that stops on its first + frame, can still open its output, write no bytes and exit 0 -- and a + verdict read from the exit code alone scores that as a pass over a + file of zero bytes. + + Deliberately narrow, because "produced no output" is a legitimate + result in four shapes this must not touch: + + * a command that was never asked for an output. The decoder runs + with no -o whenever there is no golden to compare against, and + the decode-side validation of an encode never names one at all; + |output_file| is None for both and there is nothing to check. + * a NEGATIVE cell, whose whole point is that the codec, profile or + bit depth is refused before a bitstream exists. Those declare + expected_result: unsupported and are scored on their exit code + and VkResult, not on their artifacts. + * a run that already failed, crashed, was skipped or reported the + feature unsupported. Its status is the diagnosis; replacing it + with this one would lose the more specific answer. + * a command that ASKED for no output. An -o is appended to every + encode command, so a cell that also carries a flag suppressing + the bitstream, or that asks for no frames, names a path it was + never going to fill. The command is what says so. + + Nothing here knows a codec or a filename. The path is the one the + caller put on the command line, whatever it named it. + """ + if output_file is None: + return + if command_declines_output(cmd): + return + if result.status != VideoTestStatus.SUCCESS: + return + expected = getattr(config, "expected_result", ExpectedResult.SUCCESS) + if expected != ExpectedResult.SUCCESS: + return + + try: + size = output_file.stat().st_size + except OSError: + size = None + + result.meta["output_size"] = size + if size is None: + result.status = VideoTestStatus.ERROR + result.error_message = ( + f"Exited successfully but wrote no output: " + f"{output_file} does not exist" + ) + elif size == 0: + result.status = VideoTestStatus.ERROR + result.error_message = ( + f"Exited successfully but produced an empty output: " + f"{output_file} is 0 bytes" + ) diff --git a/tests/unit_tests/mock_framework.py b/tests/unit_tests/mock_framework.py index 49df5bc7..384ef76f 100644 --- a/tests/unit_tests/mock_framework.py +++ b/tests/unit_tests/mock_framework.py @@ -21,7 +21,11 @@ limitations under the License. """ -from tests.libs.video_test_config_base import SkipFilter +from tests.libs.video_test_config_base import ( + SkipFilter, + TestResult, + VideoTestStatus, +) from tests.libs.video_test_framework_base import VulkanVideoTestFrameworkBase @@ -47,3 +51,24 @@ def create_test_suite(self): def run_single_test(self, _config): """Nothing is executed; the tests call the scoring methods directly.""" + + +def make_result(config, returncode=0, status=VideoTestStatus.SUCCESS, + stdout="", stderr=""): + """Build a TestResult the way execute_test_command would. + + Shared for the same reason MockFramework is: two test modules built an + identical one, which is a copy waiting to drift rather than two facts. + + The defaults describe an ORDINARY PASSING RUN, so a test that cares about + one field states that field and nothing else. Callers exercising a failure + pass the status, the return code, or both. + """ + return TestResult( + config=config, + returncode=returncode, + execution_time=0.0, + status=status, + stdout=stdout, + stderr=stderr, + ) diff --git a/tests/unit_tests/test_expected_rejection.py b/tests/unit_tests/test_expected_rejection.py index 07afd6f7..5e8081c0 100644 --- a/tests/unit_tests/test_expected_rejection.py +++ b/tests/unit_tests/test_expected_rejection.py @@ -40,10 +40,9 @@ BaseTestConfig, CodecType, ExpectedResult, - TestResult, VideoTestStatus, ) -from tests.unit_tests.mock_framework import MockFramework +from tests.unit_tests.mock_framework import MockFramework, make_result # What the apps actually print when the query rejects a profile, copied from a # real run on an RTX 5080. Both spellings of the result appear. @@ -69,18 +68,6 @@ def make_config(expected_result=ExpectedResult.UNSUPPORTED, ) -def make_result(config, returncode, status, stdout="", stderr=""): - """Build a TestResult as execute_test_command would.""" - return TestResult( - config=config, - returncode=returncode, - execution_time=0.0, - status=status, - stdout=stdout, - stderr=stderr, - ) - - class TestExpectedRejectionScoring: """Scoring of expected_result: unsupported cells.""" diff --git a/tests/unit_tests/test_status_determination.py b/tests/unit_tests/test_status_determination.py index 8c3dd860..f56638f8 100644 --- a/tests/unit_tests/test_status_determination.py +++ b/tests/unit_tests/test_status_determination.py @@ -1,7 +1,9 @@ """ Unit tests for test status determination. -Tests determine_test_status() method return code mapping. +Tests determine_test_status() method return code mapping, and the artifact +check that completes it: a run that exits successfully and produced nothing +is not a pass, and the exit code cannot say so on its own. Copyright 2025 Igalia S.L. @@ -18,12 +20,33 @@ limitations under the License. """ +# check_declared_output is the scoring step for a declared output. Reaching it +# through the public surface would mean running a real encoder against real +# content, which is what the integration suites do; these tests exist to pin +# the scoring down without hardware. +# pylint: disable=protected-access + from unittest.mock import patch import pytest -from tests.libs.video_test_config_base import VideoTestStatus -from tests.unit_tests.mock_framework import MockFramework +from tests.libs.video_test_config_base import ( + BaseTestConfig, + CodecType, + ExpectedResult, + VideoTestStatus, +) +from tests.libs.video_test_output_check import check_declared_output +from tests.unit_tests.mock_framework import MockFramework, make_result + + +def make_config(expected_result=ExpectedResult.SUCCESS): + """Build an ordinary encode cell, or by argument a negative one.""" + return BaseTestConfig( + name="h264_1080p", + codec=CodecType.H264, + expected_result=expected_result, + ) class TestDetermineVideoTestStatus: @@ -122,3 +145,143 @@ def test_status_comparison(self): """Test VideoTestStatus enum comparison""" assert VideoTestStatus.SUCCESS != VideoTestStatus.ERROR assert VideoTestStatus.CRASH != VideoTestStatus.NOT_SUPPORTED + + +class TestDeclaredOutputCheck: + """The artifact half of the verdict. + + A process that exits 0 has reached its own exit; it has not necessarily + done the work. An encode that loses the device on its first frame opens + its bitstream, writes nothing and exits 0, and until the artifact is + looked at that run is indistinguishable from a complete one. + """ + + @pytest.fixture + def framework(self): + """Create mock framework for tests""" + return MockFramework() + + def test_empty_output_fails_a_zero_exit(self, tmp_path): + """A 0-byte artifact turns a successful exit into an ERROR.""" + output = tmp_path / "out.264" + output.write_bytes(b"") + config = make_config() + result = make_result(config) + + check_declared_output(result, config, output) + + assert result.status == VideoTestStatus.ERROR + assert "0 bytes" in result.error_message + assert result.returncode == 0 + + def test_missing_output_fails_a_zero_exit(self, tmp_path): + """An artifact that was never written is the same failure.""" + output = tmp_path / "never_written.265" + config = make_config() + result = make_result(config) + + check_declared_output(result, config, output) + + assert result.status == VideoTestStatus.ERROR + assert "does not exist" in result.error_message + + def test_non_empty_output_stays_a_pass(self, tmp_path): + """The ordinary case: bytes were written, the verdict stands.""" + output = tmp_path / "out.ivf" + output.write_bytes(b"\x00" * 64) + config = make_config() + result = make_result(config) + + check_declared_output(result, config, output) + + assert result.status == VideoTestStatus.SUCCESS + assert result.meta["output_size"] == 64 + + def test_no_declared_output_is_not_judged(self): + """A run that was never asked for an output has nothing to check. + + The decoder runs with no -o whenever there is no golden to compare + against, and the decode-side validation of an encode never names one. + """ + config = make_config() + result = make_result(config) + + check_declared_output(result, config, None) + + assert result.status == VideoTestStatus.SUCCESS + assert "output_size" not in result.meta + + def test_negative_cell_is_not_judged(self, tmp_path): + """A cell whose pass condition is a refusal writes no bitstream.""" + output = tmp_path / "out.ivf" + config = make_config(expected_result=ExpectedResult.UNSUPPORTED) + result = make_result(config, returncode=69, + status=VideoTestStatus.SUCCESS) + + check_declared_output(result, config, output) + + assert result.status == VideoTestStatus.SUCCESS + + def test_an_existing_verdict_is_not_overwritten(self, tmp_path): + """A more specific status survives; this one only adds to a pass.""" + output = tmp_path / "out.264" + config = make_config() + for status in (VideoTestStatus.NOT_SUPPORTED, + VideoTestStatus.SKIPPED, + VideoTestStatus.CRASH): + result = make_result(config, status=status) + check_declared_output(result, config, output) + assert result.status == status + + def test_suppressed_output_is_not_judged(self, tmp_path): + """--disableFileOutput asks for no bitstream and wins over -o. + + The encode command always ends in -o, so a run that suppresses the + bitstream still names a path. It opens that path and leaves it at + zero bytes, which is the result it was asked for. + """ + output = tmp_path / "out.264" + output.write_bytes(b"") + config = make_config() + result = make_result(config) + cmd = ["vk-video-enc", "-i", "in.yuv", "--codec", "h264", + "--disableFileOutput", "-o", str(output)] + + check_declared_output(result, config, output, cmd) + + assert result.status == VideoTestStatus.SUCCESS + assert "output_size" not in result.meta + + def test_zero_frames_is_not_judged(self, tmp_path): + """A run asked for no frames has no bitstream to write.""" + output = tmp_path / "out.265" + output.write_bytes(b"") + config = make_config() + result = make_result(config) + cmd = ["vk-video-enc", "-i", "in.yuv", "--codec", "h265", + "--numFrames", "0", "--repeatInputFrames", "-o", str(output)] + + check_declared_output(result, config, output, cmd) + + assert result.status == VideoTestStatus.SUCCESS + + def test_an_ordinary_command_is_still_judged(self, tmp_path): + """The exemption is read off the command and nothing else. + + A command carrying neither flag is judged exactly as before, and a + frame count that is not zero is not a frame count of zero. + """ + output = tmp_path / "out.ivf" + output.write_bytes(b"") + config = make_config() + cmd = ["vk-video-enc", "-i", "in.yuv", "--codec", "av1", + "--numFrames", "16", "-o", str(output)] + + result = make_result(config) + check_declared_output(result, config, output, cmd) + assert result.status == VideoTestStatus.ERROR + + # And with no command at all, which is how the older callers reach it. + result = make_result(config) + check_declared_output(result, config, output) + assert result.status == VideoTestStatus.ERROR diff --git a/vk_video_decoder/demos/vk-video-dec/CMakeLists.txt b/vk_video_decoder/demos/vk-video-dec/CMakeLists.txt index 1c9d1bbf..5f2348a1 100644 --- a/vk_video_decoder/demos/vk-video-dec/CMakeLists.txt +++ b/vk_video_decoder/demos/vk-video-dec/CMakeLists.txt @@ -7,6 +7,7 @@ set(sources ${VK_DISPATCH_TABLE_SOURCE} ${VK_DISPATCH_TABLE_HEADER} ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkEncoderStdioLatch.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanShaderCompiler.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/${VK_SHADER_COMPILER_BACKEND_SOURCE} @@ -52,7 +53,11 @@ set(sources ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanCommandBufferPool.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkVideoFrameToFile.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkVideoCrc.cpp - ${VK_VIDEO_DECODER_LIBS_SOURCE_ROOT}/VkDecoderUtils/FFmpegDemuxer.cpp + # NO unconditional FFmpegDemuxer.cpp here. Listing it here as well as + # under the if(FFMPEG_AVAILABLE) below defeats the guard: the file then + # compiles even with FFMPEG_AVAILABLE OFF, and it fails on + # the moment the headers are absent. The + # conditional append below is the only place it belongs. ${VK_VIDEO_DECODER_LIBS_SOURCE_ROOT}/VkDecoderUtils/VideoStreamDemuxer.cpp ${VK_VIDEO_DECODER_LIBS_SOURCE_ROOT}/VkDecoderUtils/VideoStreamDemuxer.h ${VK_VIDEO_DECODER_LIBS_SOURCE_ROOT}/VkDecoderUtils/ElementaryStream.cpp @@ -98,11 +103,25 @@ link_directories( ${VULKAN_LIB_DIR} ) +# FFmpeg is a DEMUXER-ONLY dependency here, so the link flags must carry the +# same if(FFMPEG_AVAILABLE) guard the sources and the -DFFMPEG_DEMUXER_SUPPORT +# define already carry above. Without it the demuxer is compiled out but +# -lavformat still reaches the linker, and the target fails to link on any host +# whose libavformat is missing -- which is what happened here: FindFFmpeg found +# libavcodec and libavutil in /usr/lib but no libavformat, FFMPEG_AVAILABLE went +# OFF, and this target stopped linking while nothing in it referenced a single +# avformat_* symbol. Same guard, same reason, as +# vk_video_decoder/test/vulkan-video-dec/CMakeLists.txt. if(WIN32) - list(APPEND libraries PUBLIC ${AVCODEC_LIB} ${AVFORMAT_LIB} ${AVUTIL_LIB} ${VULKAN_VIDEO_PARSER_LIB} ${VK_SHADER_COMPILER_LIBS}) + if(FFMPEG_AVAILABLE) + list(APPEND libraries PUBLIC ${AVCODEC_LIB} ${AVFORMAT_LIB} ${AVUTIL_LIB}) + endif() + list(APPEND libraries PUBLIC ${VULKAN_VIDEO_PARSER_LIB} ${VK_SHADER_COMPILER_LIBS}) else() list(APPEND libraries PRIVATE -lX11) - list(APPEND libraries PRIVATE -lavcodec -lavutil -lavformat) + if(FFMPEG_AVAILABLE) + list(APPEND libraries PRIVATE -lavcodec -lavutil -lavformat) + endif() list(APPEND libraries PRIVATE ${VK_SHADER_COMPILER_LIBS}) list(APPEND libraries PRIVATE -L${CMAKE_INSTALL_LIBDIR} -l${VULKAN_VIDEO_PARSER_LIB}) list(APPEND libraries PRIVATE -L${LIBNVPARSER_BINARY_ROOT} -l${VULKAN_VIDEO_PARSER_LIB}) diff --git a/vk_video_decoder/include/vkvideo_parser/PictureBufferBase.h b/vk_video_decoder/include/vkvideo_parser/PictureBufferBase.h index 6967f9eb..fd053cf8 100644 --- a/vk_video_decoder/include/vkvideo_parser/PictureBufferBase.h +++ b/vk_video_decoder/include/vkvideo_parser/PictureBufferBase.h @@ -61,7 +61,22 @@ class vkPicBuffBase : public VkPicIf { } vkPicBuffBase() - : m_refCount(0) + // NAMING THE BASE IS LOAD-BEARING. VkPicIf carries decodeWidth, + // decodeHeight, decodeSuperResWidth and reserved[] as raw int32_t with + // no initializers. Leaving it out of this list DEFAULT-initializes the + // base subobject, so those fields hold indeterminate values rather + // than zero -- and nothing else zeroes them: the pool stores these in + // a std::vector, whose value-initialization degenerates to calling + // this user-provided constructor, and Reset() clears only m_refCount, + // so the fields also survive pool recycling untouched. + // + // With the AV1 parser now writing all three before they are read, this + // is belt-and-braces rather than the primary fix -- but it is what + // makes a MISSING write read as 0 instead of as a recycled + // allocation's contents, which is the difference between a bug that + // reproduces and one that only shows up under memory pressure. + : VkPicIf() + , m_refCount(0) , m_picIdx(-1) , m_displayOrder((uint32_t)-1) , m_decodeOrder(0) diff --git a/vk_video_decoder/libs/CMakeLists.txt b/vk_video_decoder/libs/CMakeLists.txt index af1e8576..b7b7071c 100644 --- a/vk_video_decoder/libs/CMakeLists.txt +++ b/vk_video_decoder/libs/CMakeLists.txt @@ -7,6 +7,7 @@ set(LIBVKVIDEODECODER_SRC ${VK_DISPATCH_TABLE_SOURCE} ${VK_DISPATCH_TABLE_HEADER} ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkEncoderStdioLatch.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanShaderCompiler.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/${VK_SHADER_COMPILER_BACKEND_SOURCE} diff --git a/vk_video_decoder/libs/NvVideoParser/src/VulkanAV1Decoder.cpp b/vk_video_decoder/libs/NvVideoParser/src/VulkanAV1Decoder.cpp index 1ac735fd..30fc4033 100644 --- a/vk_video_decoder/libs/NvVideoParser/src/VulkanAV1Decoder.cpp +++ b/vk_video_decoder/libs/NvVideoParser/src/VulkanAV1Decoder.cpp @@ -299,6 +299,51 @@ bool VulkanAV1Decoder::BeginPicture(VkParserPictureData* pnvpd) m_pClient->AllocPictureBuffer(&m_pCurrPic); } + // PUBLISH THIS PICTURE'S DIMENSIONS FOR LATER FRAMES THAT REFERENCE IT. + // + // AV1 frame_size_with_refs() has the current frame INHERIT its size from a + // reference: UpscaledWidth = RefUpscaledWidth[i], FrameHeight = + // RefFrameHeight[i]. SetupFrameSizeWithRefs() duly reads them back out of + // the referenced VkPicIf -- and until now NOTHING ANYWHERE WROTE THEM ON + // THE AV1 PATH. decodeSuperResWidth had 0 writers among its 2 occurrences + // in the tree, and `git log -S` finds no commit that ever added one: the + // field has been read-only since it was introduced. decodeWidth and + // decodeHeight had exactly one writer each and both are in the VP9 parser, + // so on a pure-AV1 stream all three were read while indeterminate -- + // vkPicBuffBase's constructor does not name its VkPicIf base, so they are + // not even zero. + // + // The wider exposure is not frame_size_with_refs at all: VulkanVideoParser + // builds the Vulkan reference VkVideoPictureResourceInfoKHR::codedExtent + // from decodeWidth/decodeHeight for EVERY AV1 reference slot on EVERY + // frame that has references, with no frame_size_override_flag gate. The + // narrow branch is where the garbage becomes visible; this is where it is + // consumed. + // + // WRITTEN HERE, AND DELIBERATELY NOT IN UpdateFramePointers(). That is the + // function that publishes the picture into the reference slots and looks + // like the natural home, but it is ALSO called on the show_existing_frame + // + reset_decoder_state path with a PREVIOUSLY DECODED picture, while the + // decoder's frame_width/frame_height/upscaled_width members still hold the + // last real frame's values. Writing there would overwrite a correct stored + // dimension with an unrelated one every time a show_existing_frame + // refreshes the slots -- turning a read-of-garbage bug into a + // corrupt-good-data bug. BeginPicture is reached from end_of_picture only + // after the whole frame header is parsed, so all three values are final. + // + // OUTSIDE the allocation guard above, not inside it. The VP9 writer this + // mirrors sits inside its own `if (m_pCurrPic == nullptr)`, so a retained + // picture keeps a stale size; that flaw is not worth copying. + // + // decodeWidth carries frame_width (the POST-superres, coded width) rather + // than upscaled_width, because that is what the reference codedExtent + // needs -- Vulkan wants the decoded extent, not the display extent. + if (m_pCurrPic != nullptr) { + m_pCurrPic->decodeSuperResWidth = upscaled_width; // RefUpscaledWidth + m_pCurrPic->decodeWidth = frame_width; // coded, post-superres + m_pCurrPic->decodeHeight = frame_height; // RefFrameHeight + } + pnvpd->PicWidthInMbs = nvsi.nCodedWidth >> 4; pnvpd->FrameHeightInMbs = nvsi.nCodedHeight >> 4; pnvpd->pCurrPic = m_pCurrPic; diff --git a/vk_video_decoder/libs/VulkanVideoFrameBuffer/VulkanVideoFrameBuffer.cpp b/vk_video_decoder/libs/VulkanVideoFrameBuffer/VulkanVideoFrameBuffer.cpp index a4087936..bf5b9faa 100644 --- a/vk_video_decoder/libs/VulkanVideoFrameBuffer/VulkanVideoFrameBuffer.cpp +++ b/vk_video_decoder/libs/VulkanVideoFrameBuffer/VulkanVideoFrameBuffer.cpp @@ -41,6 +41,15 @@ class NvPerFrameDecodeResources : public vkPicBuffBase { struct ImageViewState { VkImageLayout currentLayerLayout; + // The image itself, tracked separately from the view. A view is not + // creatable over every image this pool allocates: a transfer-only + // linear output image carries just VK_IMAGE_USAGE_TRANSFER_DST_BIT, + // which is not view-compatible (VUID-VkImageViewCreateInfo-image-04441), + // so VkImageResourceView::Create legitimately produces no view for it. + // Such an image is still perfectly usable -- vkCmdCopyImage and image + // barriers take the raw VkImage -- so existence must be keyed on the + // IMAGE, not on the view, or the resource can never be handed out. + VkSharedBaseObj imageResource; VkSharedBaseObj view; VkSharedBaseObj singleLevelView; uint32_t recreateImage : 1; @@ -106,7 +115,8 @@ class NvPerFrameDecodeResources : public vkPicBuffBase { return false; } - return (!!m_imageViewState[imageTypeIdx].view && (m_imageViewState[imageTypeIdx].view->GetImageView() != VK_NULL_HANDLE)); + return (!!m_imageViewState[imageTypeIdx].imageResource && + (m_imageViewState[imageTypeIdx].imageResource->GetImage() != VK_NULL_HANDLE)); } void InvalidateImageLayout(uint8_t imageTypeIdx) { @@ -129,10 +139,12 @@ class NvPerFrameDecodeResources : public vkPicBuffBase { } if (pPictureResourceInfo) { - pPictureResourceInfo->image = m_imageViewState[imageTypeIdx].view->GetImageResource()->GetImage(); - pPictureResourceInfo->imageFormat = m_imageViewState[imageTypeIdx].view->GetImageResource()->GetImageCreateInfo().format; + pPictureResourceInfo->image = m_imageViewState[imageTypeIdx].imageResource->GetImage(); + pPictureResourceInfo->imageFormat = m_imageViewState[imageTypeIdx].imageResource->GetImageCreateInfo().format; pPictureResourceInfo->currentImageLayout = m_imageViewState[imageTypeIdx].currentLayerLayout; - pPictureResourceInfo->baseArrayLayer = m_imageViewState[imageTypeIdx].view->GetImageSubresourceRange().baseArrayLayer; + // A viewless image is single-layer by construction; layer 0 is the only one. + pPictureResourceInfo->baseArrayLayer = m_imageViewState[imageTypeIdx].view ? + m_imageViewState[imageTypeIdx].view->GetImageSubresourceRange().baseArrayLayer : 0; } if (VK_IMAGE_LAYOUT_MAX_ENUM != newImageLayout) { @@ -140,7 +152,10 @@ class NvPerFrameDecodeResources : public vkPicBuffBase { } if (pPictureResource) { - pPictureResource->imageViewBinding = m_imageViewState[imageTypeIdx].view->GetImageView(); + // No view for a transfer-only image; the binding stays null and the + // resource is consumed through its raw VkImage. + pPictureResource->imageViewBinding = m_imageViewState[imageTypeIdx].view ? + m_imageViewState[imageTypeIdx].view->GetImageView() : VK_NULL_HANDLE; } return true; @@ -904,6 +919,11 @@ VkResult NvPerFrameDecodeResources::CreateImage( const VulkanDeviceContext* vkDe imageResource = imageArrayParent; } + // Record the image before any view is attempted. The view is optional + // (see ImageViewState::imageResource); the image is what makes the + // resource exist. + m_imageViewState[pImageSpec->imageTypeIdx].imageResource = imageResource; + if (!imageViewArrayParent) { uint32_t baseArrayLayer = imageArrayParent ? imageIndex : 0; @@ -994,6 +1014,7 @@ void NvPerFrameDecodeResources::Deinit(const VulkanDeviceContext* vkDevCtx) for (uint32_t imageTypeIdx = 0; imageTypeIdx < DecodeFrameBufferIf::MAX_PER_FRAME_IMAGE_TYPES; imageTypeIdx++) { m_imageViewState[imageTypeIdx].view = nullptr; + m_imageViewState[imageTypeIdx].imageResource = nullptr; m_imageViewState[imageTypeIdx].singleLevelView = nullptr; } diff --git a/vk_video_decoder/test/vulkan-video-dec/CMakeLists.txt b/vk_video_decoder/test/vulkan-video-dec/CMakeLists.txt index 6cdb2fba..97c517c3 100644 --- a/vk_video_decoder/test/vulkan-video-dec/CMakeLists.txt +++ b/vk_video_decoder/test/vulkan-video-dec/CMakeLists.txt @@ -7,6 +7,7 @@ set(VULKAN_VIDEO_DEC_SOURCES ${VK_DISPATCH_TABLE_SOURCE} ${VK_DISPATCH_TABLE_HEADER} ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkEncoderStdioLatch.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanShaderCompiler.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/${VK_SHADER_COMPILER_BACKEND_SOURCE} diff --git a/vk_video_decoder/test/vulkan-video-simple-dec/CMakeLists.txt b/vk_video_decoder/test/vulkan-video-simple-dec/CMakeLists.txt index 98c56898..18180512 100644 --- a/vk_video_decoder/test/vulkan-video-simple-dec/CMakeLists.txt +++ b/vk_video_decoder/test/vulkan-video-simple-dec/CMakeLists.txt @@ -1,6 +1,7 @@ set(VULKAN_VIDEO_SIMPLE_DEC_SOURCES Main.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkVideoFrameToFile.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkEncoderStdioLatch.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkVideoCrc.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/nvVkFormats.cpp ${VK_VIDEO_DECODER_LIBS_SOURCE_ROOT}/VkDecoderUtils/VideoStreamDemuxer.cpp diff --git a/vk_video_encoder/CMakeLists.txt b/vk_video_encoder/CMakeLists.txt index ba7dd463..85428409 100644 --- a/vk_video_encoder/CMakeLists.txt +++ b/vk_video_encoder/CMakeLists.txt @@ -115,6 +115,14 @@ else() MESSAGE(STATUS "VULKAN_VIDEO_ENCODER_INCLUDE path is not set. Setting the default path location to ${CMAKE_CURRENT_SOURCE_DIR}/include") set(VULKAN_VIDEO_ENCODER_INCLUDE "${CMAKE_CURRENT_SOURCE_DIR}/include" CACHE PATH "Path to Vulkan Video Encoder include directory" FORCE) endif() + +# The internal header set. NOT exported by any target and NOT installed: it is +# reachable from the encoder library and from the in-tree tests that include it, +# and from nothing a consumer links or installs. Keeping it out of +# VULKAN_VIDEO_ENCODER_INCLUDE is what makes "internal" a build fact rather than +# a filename convention. +set(VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE + "${CMAKE_CURRENT_SOURCE_DIR}/internal") ############ VULKAN_VIDEO_ENCODER_INCLUDE ###################################### ############ VULKAN_VIDEO_ENCODER_LIB_PATH ###################################### @@ -583,6 +591,235 @@ else() endif() add_subdirectory(test/vulkan-video-enc) +# Device-free coverage of the ext API's chained-descriptor walk. Links the +# static encoder archive for the internal header's test seams. +add_subdirectory(test/encoder-ext-sync) +add_subdirectory(test/encoder-ext-reconfigure) +# Real-device coverage of the per-frame RELEASE fence export. Skips (CTest +# SKIP, exit 77) on a host with no encode-capable GPU. +# LINUX-ONLY TESTS, guarded individually below. +# +# Each of these reads a POSIX facility as the SUBJECT of the test, not as a +# convenience: dlopen for loader adoption, /proc fd enumeration and dirent for +# dma-buf fd ownership, unistd for the fd lifetime assertions. There is no +# Windows equivalent to assert about, so they are Linux-only by construction +# rather than merely unported -- porting them would mean inventing a different +# test, not translating this one. +# +# The library itself builds on Windows; only these tests do not. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-release-fence) +endif() +# Real-device coverage of the per-frame ACQUIRE fence's fd OWNERSHIP -- that +# every refusal exit consumes the caller's sync_fd, and that the success path +# still consumes it exactly once. Needs a GPU for the same reason its sibling +# does: a real sync_fd needs a real queue signal to export from, and the +# success case needs a real import. Skips (CTest SKIP, exit 77) without one. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-acquire-fd) +endif() +# Device-free coverage of the input-format taxonomy (which rung of the +# adaptation ladder a format needs), of the preprocess-conversion decision the +# binder derives from it, and of the transfer-function declaration. All are +# pure functions of their arguments, so this needs no GPU; what the filter then +# DOES with the pixels needs one and lives in vk_filter_test. +add_subdirectory(test/encoder-ext-filter) +# Real-device REGRESSION test for a defect this guards against: +# DrainPendingFrames() permanently disabled async assembly, and the synchronous +# fallback publishes no completion record, so every frame submitted after that +# call was encoded and then never became acquirable. DrainAndRestartThreads() +# closed it. This is a plain gating test -- it exits 0 and carries NO +# WILL_FAIL; see the note in the test's own CMakeLists.txt for why leaving +# WILL_FAIL on would have reported the working fix as a failure. Skips +# (CTest SKIP, exit 77) without a GPU. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-drain-assembly) +endif() +# Device-free coverage of the CONTRACT the test above rests on: that +# VkVideoEncoder::AssembleBitstreamData -- the synchronous assembly path -- +# publishes NO completion record, in either output mode. Seven sites in the +# tree are built on that premise, including the guard in ProcessOrderedFrames +# and the WriteDataToFile contract comment. This entry is the ONLY thing that +# executes the function: the ext surface cannot reach it (asyncAssembly +# is pinned on, a subscriber is always registered, and ProcessOrderedFrames +# refuses the sync fallback whenever one exists) and the file-based CLI that +# can reach it exposes no VkVideoEncoder to observe. Subclasses the encoder +# with a null device context, so there is no GPU, driver or display dependence +# and nothing to skip. +add_subdirectory(test/encoder-sync-assembly) +# Real-device coverage of the DIRECT submit's FIXED 8-slot wait array. The +# direct-encode path assembles waits into VkSemaphoreSubmitInfoKHR[8] and its +# fill loop is bounded by the capacity, so surplus waits were DISCARDED with no +# log, no assert and a SUCCESS status -- and because the ext layer appends the +# imported acquire-fence semaphore LAST, that producer fence is the first +# casualty. Registers two entries: the DIRECT subject and a STAGED control that +# runs the identical case against the non-truncating vector assembly, which is +# what makes a green subject arm distinguishable from a test that cannot fail. +# Skips (CTest SKIP, exit 77) without an encode-capable GPU. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-direct-wait-capacity) +endif() +# Real-device coverage of the ADOPT-mode SESSION: the embedder owns the +# VkInstance and names the VkPhysicalDevice, and the LIBRARY creates its own +# VkDevice on them and encodes. That combination -- externalInstance + +# externalPhysicalDevice with externalDevice left NULL -- is the supported +# and this suite is the only thing in the tree that exercises +# externalInstance, externalPhysicalDevice or externalDevice at all: the only +# caller was the out-of-tree Chromium embedder. +# Four entries: the ADOPT subject, an OWN control that makes a red +# subject attributable to the adopted handles rather than to the host, the +# physical-device pin, and config.validate over a borrowed instance. +# Skips (CTest SKIP, exit 77) without an encode-capable GPU. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-adopt-device) +endif() +# Real-device coverage of INPUT RESIDENCY on an OS-HANDLE registration. Before +# this suite every handleType in vk_video_encoder/test was +# ..._VK_IMAGE -- eight registration sites, zero OPAQUE_FD -- so the whole +# OS-handle import path, and every rule the ext layer applies only to it, had +# no test of any kind. The residency read in SubmitRegisteredFrame is exactly +# such a rule, and a suite that only registers VK_IMAGE takes its other arm +# every time. +# +# The registration is a SELF-IMPORT: the test exports OPAQUE_FD from the +# LIBRARY's own VkDevice and hands the fd straight back, which is what +# Chromium's shipping CPU staging tier does and which no in-tree test did. +# +# Asserts on a library-side counter (VkVideoEncoderInputResidencyInfo), NOT on +# a validation-error count, and that is forced rather than preferred: the two +# barrier programs the residency decision selects between are both spec-clean +# and both leave the image in the same layout, so a layer-based assertion here +# would be green either way -- a test that cannot fail. Five entries: the +# OPAQUE_FD + explicit-LOCAL subject, a VK_IMAGE control over the IDENTICAL +# image that makes a red subject attributable to handleType alone, an explicit +# FOREIGN and an AUTO pin that together stop the fix degenerating into "never +# acquire", and a mirror of the Chromium tier-3 layout declaration which +# PASSES -- the design registered it WILL_FAIL and the hardware disagreed, so +# the registration was corrected rather than the measurement explained away; +# the suite's own CMakeLists records the reasoning. +# Skips (CTest SKIP, exit 77) without an encode-capable GPU. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-input-residency) +endif() +# Real-device INPUT FORMAT MATRIX. VkEncClassifyInputFormat names nine formats; +# whether one REGISTERS, and onto which input path, is a fact about a device, a +# session and a descriptor that the taxonomy table cannot see. This walks all +# nine plus two excluded controls against the library-owned device and asserts +# both halves per arm -- registration status AND the slot's resolved inputPath, +# read from VkEncProbeResource rather than inferred from the config flag, which +# is true of a registration that resolved to STAGED as well. Each arm also runs +# a NEGATIVE control with its own declaration withheld, so a fix that merely +# deleted a refusal fails here. Skips (CTest SKIP, exit 77) without an +# encode-capable GPU. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-format-matrix) +endif() +add_subdirectory(test/encoder-ext-input-format-query) +# The v3 interface, driven the way a host drives it: Result and Expected +# semantics, role discovery, the aliasing reference that keeps a session alive +# behind a role, and the two configuration rules the library owns -- which +# input path a format resolves to, and which codecs carry HDR metadata. The +# checks that need no device always run; the rest report SKIP without an +# encode-capable GPU rather than passing vacuously. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-interface) +endif() +add_subdirectory(test/encoder-drain) +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-release-fence) +endif() + +# The ENCODE half of the same matrix: known primaries pattern in, real frames +# submitted, bitstream written out for an INDEPENDENT decoder. A 3-plane row +# that does not resolve to FILTER is abandoned before any submit. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + add_subdirectory(test/encoder-ext-format-encode) +endif() + +# Real-device coverage of the dma-buf IMPORT-ORDINAL +# GUARD, which shipped with a coverage floor of exactly zero: no +# test, CMake entry or CI file referenced it, and +# VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF reached +# VkEncImportExternalImage from NO encoder-ext test at all -- every +# registration in the tree was VK_IMAGE or OPAQUE_FD, on both of +# which the guard returns at its second line. The suite was green +# with the workaround deleted. +# +# This exports a REAL dma-buf from a real NVIDIA VkDevice and imports +# it, with counting thunks in the device dispatch table, so what the +# guard did is read off the driver rather than off a log line. +# +# THE GUARD IS DISABLED BY DEFAULT (kVkEncImportOrdinalGuardCount = 0): +# it shifts the PHASE of the driver's import-damage pattern and does +# not reduce it, the damage rate being the same for every K, so the +# shipped build takes no sacrificial import at all. +# The default entry therefore asserts INERTNESS: one image for one +# caller import, nothing retained, nothing released, verdict DISABLED. +# The ARMED behaviour has not lost its coverage; the same binary takes +# --guard-count=N and asserts all of it against a library rebuilt with +# that N, which is the only way a build constant can be tested. Two +# ctest entries: the retired subject and the kill switch. Skips (exit +# 77) with no GPU, no dma-buf export, or a non-NVIDIA GPU, where the +# guard refuses to run by design. +add_subdirectory(test/encoder-ext-import-ordinal-guard) + +# The same guard's REPORTING CHANNEL, device-free. The suite above asks what +# the guard did on real hardware; this one asks whether anything OUTSIDE the +# library can find out, and pins the build count the library ships. The guard used to answer only on +# stderr, which the shipping Chromium configuration discards wholesale +# (silenceStdio), so both of its failure paths were silent in the one +# deployment it exists for. This gates the carrier that replaced that +# narration: VkVideoEncoderImportGuardInfo, chainable onto the registration +# status echo and onto the completion snapshot. It covers the CARRIER only -- +# COMPLETE, INCOMPLETE and DISABLED-on-a-real-import need a real dma-buf +# import on an NVIDIA device and belong to the suite above and to hardware. +# Note the carrier got harder to test, not easier, when the guard was retired: +# the build count is 0, so every field of an honest verdict is now 0 too, and +# these cases have to poison the caller's struct to prove the library wrote it +# at all. +add_subdirectory(test/encoder-ext-import-guard) + +# The dma-buf import CONTENT probe -- what the DRIVER actually put in the +# buffer, as opposed to what the (retired) ordinal guard did about it. The +# suite above gates a workaround's REPORT; this one gates a DETECTOR, and the +# distinction matters for what a green means. +# +# It covers the decision and the carrier, device-free: the predicate over +# every (Y,U,V) liveness quadrant, the scorer over synthetic NV12 buffers laid +# out with the plane offsets, padded row pitch and skipped rows a real +# HOST_VISIBLE LINEAR image has, the once-per-REGISTRATION latch, the +# oldest-damaged-first report ordering, the drain on retirement, and the +# VkVideoEncoderImportContentInfo chain on both of its two chain points. +# +# It does NOT cover whether the probe fires on a buffer the driver really +# poisoned. That needs a driver that exhibits the defect and a real GBM +# dma-buf import, and is +# proven on hardware or not at all. What this buys is that the DECISION -- +# the part that would otherwise be exercised only on the one host with the +# broken driver -- is regression-tested everywhere, including the GPU-less CI +# runner. The predicate is asserted quadrant by quadrant rather than +# spot-checked because the CHROMA-ONLY version of it has produced a wrong +# conclusion on this defect twice, and a fully legal black frame is asserted +# CLEAN because that is the false positive that would reroute working buffers. +add_subdirectory(test/encoder-ext-import-content) + +# H.264 Baseline must not emit CABAC. profile_idc 66 covers Baseline and +# Constrained Baseline alike and neither admits CABAC: entropy_coding_mode_flag +# must be 0 there. Device-free. +# +# The entropy coder is chosen by nobody: there is no command-line switch for it, +# it is taken from the device preferredStdEntropyCodingModeFlag, which is CABAC +# on this vendor hardware. So an explicit --profile baseline emitted profile_idc +# 66 with entropy_coding_mode_flag 1 -- a non-conformant bitstream, and +# reachable from the WebRTC default profile. +# +# Three parts: the rule over the whole profile x entropy matrix, the emitted +# SPS/PPS pair, and the composition with the profile auto-upgrade in +# InitProfileLevel(). The third is the one worth having. That upgrade does NOT +# cover this case -- it is guarded by profileIdc == INVALID and so never runs on +# an explicit request -- and the suite fails if someone later widens it to fire +# on requests and assumes it subsumes the clamp. +add_subdirectory(test/encoder-h264-baseline-entropy) if(BUILD_DEMOS AND NOT DEFINED DEQP_TARGET) add_subdirectory(demos) diff --git a/vk_video_encoder/demos/vk-video-enc/CMakeLists.txt b/vk_video_encoder/demos/vk-video-enc/CMakeLists.txt index 7d278107..d872f1fe 100644 --- a/vk_video_encoder/demos/vk-video-enc/CMakeLists.txt +++ b/vk_video_encoder/demos/vk-video-enc/CMakeLists.txt @@ -20,13 +20,29 @@ set(sources ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderAV1.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkEncoderConfig.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoder.cpp + # THIS LIST IS A DUPLICATE OF libs/CMakeLists.txt's, and that is the + # defect it just caused: the demo compiles the encoder sources itself + # instead of linking libvkvideo-encoder-static, so every file added there + # has to be added here too, with nothing to enforce it. HdrMetadata was + # added to libs/ and not here, and the demo has not linked since -- + # undefined reference to VkEncBuildH265HdrSeiNal and + # VkEncBuildAv1HdrMetadataObus. It is invisible in a -DBUILD_DEMOS=OFF + # configure and in ctest, which is why it survived. + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderHdrMetadata.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderHdrMetadata.h + # Same seam, same selection rule as vk_video_encoder/libs/CMakeLists.txt. + $<$:${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.cpp> + $<$:${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderOsAdapterWindows.cpp> + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderPsnr.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderContentProbe.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderPsnr.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/json/EncoderConfigJsonLoader.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoGopStructure.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoGopStructure.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoder.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/YCbCrConvUtilsCpu.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkEncoderStdioLatch.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/YCbCrConvUtilsCpu.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkShell/Shell.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkShell/ShellDirect.cpp @@ -131,11 +147,25 @@ link_directories( ${VULKAN_LIB_DIR} ) +# FFmpeg is a DEMUXER-ONLY dependency here, so the link flags must carry the +# same if(FFMPEG_AVAILABLE) guard the sources and the -DFFMPEG_DEMUXER_SUPPORT +# define already carry above. Without it the demuxer is compiled out but +# -lavformat still reaches the linker, and the target fails to link on any host +# whose libavformat is missing -- which is what happened here: FindFFmpeg found +# libavcodec and libavutil in /usr/lib but no libavformat, FFMPEG_AVAILABLE went +# OFF, and this target stopped linking while nothing in it referenced a single +# avformat_* symbol. Same guard, same reason, as +# vk_video_decoder/test/vulkan-video-dec/CMakeLists.txt. if(WIN32) - list(APPEND libraries PRIVATE ${AVCODEC_LIB} ${AVFORMAT_LIB} ${AVUTIL_LIB} ${VULKAN_VIDEO_PARSER_LIB} ${VK_SHADER_COMPILER_LIBS}) + if(FFMPEG_AVAILABLE) + list(APPEND libraries PRIVATE ${AVCODEC_LIB} ${AVFORMAT_LIB} ${AVUTIL_LIB}) + endif() + list(APPEND libraries PRIVATE ${VULKAN_VIDEO_PARSER_LIB} ${VK_SHADER_COMPILER_LIBS}) else() list(APPEND libraries PRIVATE -lX11) - list(APPEND libraries PRIVATE -lavcodec -lavutil -lavformat) + if(FFMPEG_AVAILABLE) + list(APPEND libraries PRIVATE -lavcodec -lavutil -lavformat) + endif() list(APPEND libraries PRIVATE ${VK_SHADER_COMPILER_LIBS}) list(APPEND libraries PRIVATE -L${CMAKE_INSTALL_LIBDIR} -l${VULKAN_VIDEO_PARSER_LIB}) list(APPEND libraries PRIVATE -L${LIBNVPARSER_BINARY_ROOT} -l${VULKAN_VIDEO_PARSER_LIB}) diff --git a/vk_video_encoder/demos/vk-video-enc/Main.cpp b/vk_video_encoder/demos/vk-video-enc/Main.cpp index 0b348830..54201793 100644 --- a/vk_video_encoder/demos/vk-video-enc/Main.cpp +++ b/vk_video_encoder/demos/vk-video-enc/Main.cpp @@ -134,7 +134,7 @@ int main(int argc, const char* argv[]) } VkQueueFlags requestVideoComputeQueueMask = 0; - if (encoderConfig->enablePreprocessComputeFilter == VK_TRUE) { + if (encoderConfig->IsPreprocessComputeFilterEnabled()) { requestVideoComputeQueueMask = VK_QUEUE_COMPUTE_BIT; } @@ -199,7 +199,7 @@ int main(int argc, const char* argv[]) true, // createGraphicsQueue true, // createDisplayQueue ((encoderConfig->selectVideoWithComputeQueue == 1) || // createComputeQueue - (encoderConfig->enablePreprocessComputeFilter == VK_TRUE)) + encoderConfig->IsPreprocessComputeFilterEnabled()) ); if (result != VK_SUCCESS) { if (IsVideoUnsupportedResult(result)) { @@ -260,7 +260,7 @@ int main(int argc, const char* argv[]) false, // createGraphicsQueue false, // createDisplayQueue ((encoderConfig->selectVideoWithComputeQueue == 1) || // createComputeQueue - (encoderConfig->enablePreprocessComputeFilter == VK_TRUE)) + encoderConfig->IsPreprocessComputeFilterEnabled()) ); if (result != VK_SUCCESS) { if (IsVideoUnsupportedResult(result)) { @@ -311,10 +311,31 @@ int main(int argc, const char* argv[]) } } - encoder->WaitForThreadsToComplete(); + // The drain carries the verdict from the encoder and assembly threads; + // |result| carries the frame loop's own. Both are collected before + // anything is reported, so the summary below describes the run that + // actually happened. + const bool completed = encoder->WaitForThreadsToComplete(); std::cout << "Done processing " << curFrameIndex << " input frames!" << std::endl << "Encoded file's location is at " << encoderConfig->outputFileHandler.GetFileName() << std::endl; + + // THE EXIT STATUS IS THE ONLY THING A HARNESS READS. An encode that + // stopped early leaves a bitstream behind either way -- a short one, or + // an empty one when it stopped on the first frame -- so the file cannot + // distinguish a finished run from an abandoned one and the status has to. + if (result != VK_SUCCESS) { + fprintf(stderr, "Encoding stopped at input frame %u: the frame could " + "not be read, staged, recorded or submitted (0x%x)\n", + curFrameIndex, result); + return EXIT_FAILURE; + } + if (!completed) { + fprintf(stderr, "Encoding did not complete: the encoder reported a " + "failure on one or more of the %u frames it was given\n", + curFrameIndex); + return EXIT_FAILURE; + } return 0; } diff --git a/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.html b/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.html new file mode 100644 index 00000000..ae2185d3 --- /dev/null +++ b/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.html @@ -0,0 +1,330 @@ + + + + + + + The dma-buf import content probe + + + + + + +
+ +
+
+
+ +
+

The dma-buf import content probe

+

Status: internal mechanism. It is not part of the public encoder interface, and this document does not propose making it one.

+ +
+
+ +
+
+ +
+ +

1. The failure it exists to catch

+ +

A dma-buf import can succeed at every level the API can report on — vkAllocateMemory, vkBindImageMemory, image view creation, all VK_SUCCESS — and still bind an image to memory the producer's writes never reach. The encoder then encodes that image faithfully and emits a structurally valid bitstream of a dead or green frame.

+ +

Nothing else in the library can see this. Every layer above works from handles and descriptors, and those are all correct; the damage is only visible in the pixels. The probe is the only place that reads imported pixels on the host, so it is the only place that can hold an opinion.

+ +

2. The shape of the answer

+ +

The verdict is latched per registration, not per frame.

+ +

The defect is a property of the import: it is permanent for the life of that buffer, a damaged import is not recoverable, re-importing at the next ordinal rescues it in 0 of 8 attempts, and 100% of the frames backed by a damaged buffer carry the damage. A second look therefore costs a readback and can learn nothing. A 5125-frame session with 5 registered buffers performs 5 probes, not 5125.

+ +

3. The sequence

+ +
+ Arm, capture, score, echo +
Arm, capture, score, echo
+
+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
#WhenCallWhat happens
1Encoder initialisationConfigure()Pool depth is the encode queue depth + 2: a node is held from the record site until the post-fence score, so the in-flight set can briefly exceed the queue depth
2The host registers an external bufferArmRegistration(resource, captureSite)State becomes ARMED, or NOT_APPLICABLE when the readback cannot ride this registration. Latched here rather than discovered per frame, so the answer travels back in the registration echo
3First frame that uses the registrationNeedsCapture() gates RecordCapture()Two vkCmdCopyImage — Y and UV — appended to the staging command buffer that already exists, one command after CopyLinearToOptimalImage has read the same image in the same layout. No extra submit, fence or queue; the image is already in TRANSFER_SRC_OPTIMAL
4After that command buffer's fence is waitedScoreCapture()Host-side strided plane means off a HOST_VISIBLE|HOST_COHERENT LINEAR pool image, from the same two post-fence sites the PSNR readback uses
5Any later callchained VkVideoEncoderImportContentInfoThe caller reads state, meanY/U/V and the session totals
—UnregistrationForgetRegistration()Per-registration state is erased. Session totals are deliberately not decremented
+ +

ScoreCapture() must run only after the fence has been waited. The capture is recorded into the caller's command buffer; the score reads host memory that is only valid once that submission has completed.

+ +

4. The predicate

+ +

A plane is dead when its mean is strictly below 2.0/255, expressed in Q8 as kDeadPlaneMeanQ8 = 512. The public constant VK_VIDEO_ENCODER_IMPORT_CONTENT_DEAD_PLANE_MEAN_Q8 must equal it, and a static_assert pins the two together.

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
StateMeaning
NOT_EVALUATEDNo verdict. The registration was never armed, or has been forgotten
NOT_APPLICABLEThe readback cannot ride this registration, so no verdict is possible
ARMEDA verdict was promised and has not arrived yet
CLEANScored, and no plane is dead
DAMAGED_CHROMAThe chroma planes are dead, luma is not
DAMAGED_ALLEvery plane is dead
+ +

Only VK_FORMAT_G8_B8R8_2PLANE_420_UNORM is probeable. The predicate is stated over Y, U and V plane means, so a format with no chroma planes to score has no verdict to give, and a 10- or 12-bit packed format stores its samples in the high bits of 16-bit words — a byte-wise mean would be reading the wrong half. Widening this means teaching the scorer the word layout, not adding a case to the predicate, and NOT_APPLICABLE is the honest answer until someone does.

+ +

Only every 8th row of each plane is read (kRowStride). That is a requirement rather than a micro-optimisation: a byte-wise walk of the same 3.1 MB of HOST_VISIBLE memory costs VkVideoEncoderPsnr the difference between ~200 fps and 3.6 fps. A strided mean is ample for "is this plane dead", which is the only question asked.

+ +

5. Read armedRegistrationCount before believing anything else

+ +

probedRegistrationCount == 0 && damagedRegistrationCount == 0 is ambiguous. It is what a session reports when every buffer was probed and every one was clean, and it is equally what a session reports when no capture ever ran.

+ +

armedRegistrationCount is what tells them apart: it counts registrations that were promised a verdict and have not reached one.

+ +
    +
  • Non-zero transiently is expected and correct — a registration is armed at import and scored
  • +
+ +

a frame or two later, so a mid-session poll legitimately catches buffers in flight.

+ +
    +
  • Non-zero at the end of a session, or in a long-running session where it never falls, means
  • +
+ +

the capture site is not being reached. The absence of damage reports is then an absence of measurement, not an absence of damage.

+ +

A consumer that reports "no damage" without checking this field cannot distinguish a healthy session from a probe that never ran.

+ +

probeGeneration is a non-zero build constant stamped on every path, including the paths that produce no verdict. It is the writer proof: a zeroed structure means nothing filled it in.

+ +

6. What it deliberately does not do

+ +

It does not write to stderr. A host that silences stdio, or that runs the encoder in a child process whose stderr it never reads, would lose every finding — and a diagnostic whose only signal can be discarded by the caller it exists for tells that caller nothing. Every answer leaves through the chained structure, which the caller must read.

+ +

It does not decrement session totals on unregistration. "Three of the buffers this session imported were damaged" stays true after those three are retired, and a consumer watching a falling total would conclude the damage went away.

+ +

It does not re-probe. See §2.

+ +

7. Reachability

+ +

The probe is internal, and the structures that carry its verdict — VkVideoEncoderImportContentInfo, VkVideoEncoderImportContentState and the companion VkVideoEncoderImportGuardInfo — are declared in the descriptor API's internal header, which is not on any consumer's include path.

+ +

Its consumers are the library's own white-box tests: encoder-ext-import-content, encoder-ext-import-guard and encoder-ext-format-encode --content-probe. Those tests name the internal directory explicitly in their build, which is the point — a test that reaches past the public surface should have to say so.

+ +

The public C++ encoder interface does not expose the probe and is not intended to. A host that needs import verification today gets it by running those tests against its own configuration, not by chaining a structure at runtime.

+ +
+
+ +
+
+

+
+
+ + diff --git a/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.md b/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.md new file mode 100644 index 00000000..4dd5dc0d --- /dev/null +++ b/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.md @@ -0,0 +1,121 @@ +# The dma-buf import content probe + +**Status:** internal mechanism. It is not part of the public encoder interface, and this +document does not propose making it one. + +--- + +## 1. The failure it exists to catch + +A dma-buf import can succeed at every level the API can report on — `vkAllocateMemory`, +`vkBindImageMemory`, image view creation, all `VK_SUCCESS` — and still bind an image to memory +the producer's writes never reach. The encoder then encodes that image faithfully and emits a +structurally valid bitstream of a dead or green frame. + +Nothing else in the library can see this. Every layer above works from handles and descriptors, +and those are all correct; the damage is only visible in the pixels. The probe is the only place +that reads imported pixels on the host, so it is the only place that can hold an opinion. + +## 2. The shape of the answer + +The verdict is latched **per registration**, not per frame. + +The defect is a property of the import: it is permanent for the life of that buffer, a damaged +import is not recoverable, re-importing at the next ordinal rescues it in 0 of 8 attempts, and +100% of the frames backed by a damaged buffer carry the damage. A second look therefore costs a +readback and can learn nothing. A 5125-frame session with 5 registered buffers performs 5 probes, +not 5125. + +## 3. The sequence + +![Arm, capture, score, echo](ENCODER_IMPORT_CONTENT_PROBE.svg) + +| # | When | Call | What happens | +|---|---|---|---| +| 1 | Encoder initialisation | `Configure()` | Pool depth is the encode queue depth **+ 2**: a node is held from the record site until the post-fence score, so the in-flight set can briefly exceed the queue depth | +| 2 | The host registers an external buffer | `ArmRegistration(resource, captureSite)` | State becomes `ARMED`, or `NOT_APPLICABLE` when the readback cannot ride this registration. Latched here rather than discovered per frame, so the answer travels back in the registration echo | +| 3 | First frame that uses the registration | `NeedsCapture()` gates `RecordCapture()` | Two `vkCmdCopyImage` — Y and UV — appended to the staging command buffer that already exists, one command after `CopyLinearToOptimalImage` has read the same image in the same layout. No extra submit, fence or queue; the image is already in `TRANSFER_SRC_OPTIMAL` | +| 4 | After that command buffer's fence is waited | `ScoreCapture()` | Host-side strided plane means off a `HOST_VISIBLE|HOST_COHERENT` LINEAR pool image, from the same two post-fence sites the PSNR readback uses | +| 5 | Any later call | chained `VkVideoEncoderImportContentInfo` | The caller reads `state`, `meanY/U/V` and the session totals | +| — | Unregistration | `ForgetRegistration()` | Per-registration state is erased. Session totals are deliberately **not** decremented | + +`ScoreCapture()` must run only after the fence has been waited. The capture is recorded into the +caller's command buffer; the score reads host memory that is only valid once that submission has +completed. + +## 4. The predicate + +A plane is dead when its mean is **strictly below 2.0/255**, expressed in Q8 as +`kDeadPlaneMeanQ8 = 512`. The public constant +`VK_VIDEO_ENCODER_IMPORT_CONTENT_DEAD_PLANE_MEAN_Q8` must equal it, and a `static_assert` pins +the two together. + +| State | Meaning | +|---|---| +| `NOT_EVALUATED` | No verdict. The registration was never armed, or has been forgotten | +| `NOT_APPLICABLE` | The readback cannot ride this registration, so no verdict is possible | +| `ARMED` | A verdict was promised and has not arrived yet | +| `CLEAN` | Scored, and no plane is dead | +| `DAMAGED_CHROMA` | The chroma planes are dead, luma is not | +| `DAMAGED_ALL` | Every plane is dead | + +Only `VK_FORMAT_G8_B8R8_2PLANE_420_UNORM` is probeable. The predicate is stated over Y, U and V +plane means, so a format with no chroma planes to score has no verdict to give, and a 10- or +12-bit packed format stores its samples in the high bits of 16-bit words — a byte-wise mean would +be reading the wrong half. Widening this means teaching the scorer the word layout, not adding a +case to the predicate, and `NOT_APPLICABLE` is the honest answer until someone does. + +Only every 8th row of each plane is read (`kRowStride`). That is a requirement rather than a +micro-optimisation: a byte-wise walk of the same 3.1 MB of `HOST_VISIBLE` memory costs +`VkVideoEncoderPsnr` the difference between ~200 fps and 3.6 fps. A strided mean is ample for +"is this plane dead", which is the only question asked. + +## 5. Read `armedRegistrationCount` before believing anything else + +`probedRegistrationCount == 0 && damagedRegistrationCount == 0` is **ambiguous**. It is what a +session reports when every buffer was probed and every one was clean, and it is equally what a +session reports when no capture ever ran. + +`armedRegistrationCount` is what tells them apart: it counts registrations that were promised a +verdict and have not reached one. + +* Non-zero **transiently** is expected and correct — a registration is armed at import and scored + a frame or two later, so a mid-session poll legitimately catches buffers in flight. +* Non-zero **at the end of a session**, or in a long-running session where it never falls, means + the capture site is not being reached. The absence of damage reports is then an absence of + *measurement*, not an absence of damage. + +A consumer that reports "no damage" without checking this field cannot distinguish a healthy +session from a probe that never ran. + +`probeGeneration` is a non-zero build constant stamped on every path, including the paths that +produce no verdict. It is the writer proof: a zeroed structure means nothing filled it in. + +## 6. What it deliberately does not do + +**It does not write to stderr.** A host that silences stdio, or that runs the encoder in a child +process whose stderr it never reads, would lose every finding — and a diagnostic whose only signal +can be discarded by the caller it exists for tells that caller nothing. Every answer leaves through +the chained structure, which the caller must read. + +**It does not decrement session totals on unregistration.** "Three of the buffers this session +imported were damaged" stays true after those three are retired, and a consumer watching a falling +total would conclude the damage went away. + +**It does not re-probe.** See §2. + +## 7. Reachability + +The probe is internal, and the structures that carry its verdict — +`VkVideoEncoderImportContentInfo`, `VkVideoEncoderImportContentState` and the companion +`VkVideoEncoderImportGuardInfo` — are declared in the descriptor API's internal header, which is +not on any consumer's include path. + +Its consumers are the library's own white-box tests: `encoder-ext-import-content`, +`encoder-ext-import-guard` and `encoder-ext-format-encode --content-probe`. Those tests name the +internal directory explicitly in their build, which is the point — a test that reaches past the +public surface should have to say so. + +The public C++ encoder interface does not expose the probe and is not intended to. A host that +needs import verification today gets it by running those tests against its own configuration, not +by chaining a structure at runtime. diff --git a/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.svg b/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.svg new file mode 100644 index 00000000..6c6c2588 --- /dev/null +++ b/vk_video_encoder/docs/ENCODER_IMPORT_CONTENT_PROBE.svg @@ -0,0 +1,136 @@ + + + + + + + + + dma-buf import content probe — arm once, capture once, score once + A verdict is latched per REGISTRATION, not per frame. The capture rides the staging command buffer that already exists. + + + + + + + + + Host + Descriptor API layer + Encoder core + Content probe + + + + register + first frame that uses it + after the staging fence + any later call + + + + + + + + register external image + + chained ContentInfo + read the echo + state + means + totals + + + + + + + + ArmRegistration + (resource, captureSite) + fill ContentInfo + from probe state + + + + + + + + StageInputFrame + image already TRANSFER_SRC + post-fence sites + same two the PSNR path uses + + + + + + + + + latch ARMED + or NOT_APPLICABLE + RecordCapture + 2x vkCmdCopyImage to LINEAR + ScoreCapture + strided plane means + + + + + + + + + + + + + + NeedsCapture() gates this — once per registration + + The verdict states + + + + + + + + + + NOT_EVALUATED + ARMED + NOT_APPLICABLE + CLEAN + DAMAGED_CHROMA + DAMAGED_ALL + + + + + + + + + + arm + format not probeable + score + + + + Read armedRegistrationCount FIRST + + probed=0 damaged=0 is ambiguous. It is what + a session reports when every buffer was clean, + and equally what one reports when nothing was + ever measured. + armed > 0 at session end = buffers were + promised a verdict and never got one. + + + Once per registration: the defect is a property of the IMPORT, permanent for that buffer, and carried by every frame it backs. + A 5125-frame session with 5 registered buffers performs 5 probes, not 5125. + Internal mechanism — not exposed through the public encoder interface. + diff --git a/vk_video_encoder/include/vulkan_video_encoder.h b/vk_video_encoder/include/vulkan_video_encoder.h index f170fd4a..7ee4faa8 100644 --- a/vk_video_encoder/include/vulkan_video_encoder.h +++ b/vk_video_encoder/include/vulkan_video_encoder.h @@ -1,5 +1,5 @@ /* - * Copyright 2024 NVIDIA Corporation. + * Copyright 2026 NVIDIA Corporation. * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. @@ -17,42 +17,1189 @@ #ifndef _VULKAN_VIDEO_ENCODER_H_ #define _VULKAN_VIDEO_ENCODER_H_ -// VK_VIDEO_ENCODER_EXPORT tags symbol that will be exposed by the shared library. +// The encoder interface: abstract, reference-counted C++ roles. +// +// THE LANGUAGE FLOOR IS C++17, and it is a hard constraint on this file rather +// than a preference. Consumers include this header; they do not compile the +// implementation, which targets C++20. Nothing here may use std::span, +// std::expected, designated initialisers, concepts or coroutines. A +// syntax-only compile at -std=c++17 is part of the test suite. +// +// THE CALLER NEVER LAYS OUT AN OBJECT THIS LIBRARY OWNS. Every interface below +// is allocated by the library and reached through a Ref. That is what allows a +// later version to add behaviour: a new interface derives from an old one, and +// callers built against the old one neither recompile nor notice. Types the +// CALLER constructs -- PlatformCreateInfo, RateControl, FrameSubmit -- are +// plain aggregates and may never gain a field, because their layout is pinned +// by every caller that compiles against them. Extension happens on interfaces. +// +// A FEATURE THE IMPLEMENTATION DOES NOT PROVIDE IS A NULL Ref, not an error +// returned from a call. Query() is both the capability test and the way to +// reach the capability, so a caller checks once at setup instead of at every +// call site. + +#include +#include +#include +#include +#include + +#include "vulkan_interfaces.h" + #if defined(VK_VIDEO_ENCODER_SHAREDLIB) -#if defined(_WIN32) -#if defined(VK_VIDEO_ENCODER_IMPLEMENTATION) -#define VK_VIDEO_ENCODER_EXPORT __declspec(dllexport) +# if defined(_WIN32) +# if defined(VK_VIDEO_ENCODER_IMPLEMENTATION) +# define VK_ENC_EXPORT __declspec(dllexport) +# else +# define VK_ENC_EXPORT __declspec(dllimport) +# endif +# else +# if defined(VK_VIDEO_ENCODER_IMPLEMENTATION) +# define VK_ENC_EXPORT __attribute__((visibility("default"))) +# else +# define VK_ENC_EXPORT +# endif +# endif #else -#define VK_VIDEO_ENCODER_EXPORT __declspec(dllimport) -#endif -#else -#if defined(VK_VIDEO_ENCODER_IMPLEMENTATION) -#define VK_VIDEO_ENCODER_EXPORT __attribute__((visibility("default"))) -#else -#define VK_VIDEO_ENCODER_EXPORT -#endif -#endif -#else -#define VK_VIDEO_ENCODER_EXPORT +# define VK_ENC_EXPORT #endif -#include "vulkan_interfaces.h" -#include "VkCodecUtils/VkVideoRefCountBase.h" +namespace vk { +namespace video { +namespace enc { + +//============================================================================= +// STATUS +//============================================================================= + +// NOTHING IN THIS HEADER MAY BE NAMED Success, None, Status, Bool, Always, +// Complex, Convex, Above, Below, KeyPress OR CursorShape. Xlib defines every +// one of them as an object-like macro, so a consumer building with +// VK_USE_PLATFORM_XLIB_KHR -- which is every browser and every compositor on +// Linux -- would have them substituted before this header is even parsed. A +// namespace does not help against the preprocessor. +// +// Undefining them here is not an option: it would silently break the +// consumer's own use of Xlib. So the names are simply not used, which is why +// the status type is Result and its code type is ResultCode. +enum class ResultCode : int32_t { + Ok = 0, + + // The caller can fix these. + InvalidArgument, + NotConfigured, + AlreadyConfigured, + UnsupportedCodec, + UnsupportedProfile, + UnsupportedFormat, + UnsupportedFeature, + OutOfRange, + + // Try again, or wait. + NotReady, + Timeout, + WouldBlock, + OutOfResources, + + // The session is finished. + DeviceLost, + InternalError, +}; + +// A code and a pointer to a string literal the library owns: 16 bytes, +// trivially copyable, no allocation. SubmitFrame returns one of these per +// frame, so it must not touch the heap. +// +// detail() is a literal, never a formatted message. Anything needing runtime +// context -- which frame, which format -- belongs on IDiagnostics. +class Result { +public: + Result() : m_code(ResultCode::Ok), m_detail("") { } + Result(ResultCode code, const char* detail = "") + : m_code(code), m_detail(detail ? detail : "") { } + + bool ok() const { return m_code == ResultCode::Ok; } + explicit operator bool() const { return ok(); } + + ResultCode code() const { return m_code; } + const char* detail() const { return m_detail; } + + bool operator==(ResultCode c) const { return m_code == c; } + bool operator!=(ResultCode c) const { return m_code != c; } + +private: + ResultCode m_code; + const char* m_detail; +}; + +// A T, or the Result explaining why there is no T. std::expected is C++23, so +// this is permanent rather than a placeholder. +// +// Reading value() when the Result is an error is a caller bug; the value is +// default-constructed in that case rather than undefined, so a missed check +// misbehaves predictably instead of corrupting memory. +template +class Expected { +public: + Expected(const T& value) : m_value(value) { } + Expected(T&& value) : m_value(std::move(value)) { } + Expected(const Result& status) : m_value(), m_status(status) { } + + bool ok() const { return m_status.ok(); } + explicit operator bool() const { return ok(); } + + const T& value() const { return m_value; } + T& value() { return m_value; } + const T& operator*() const { return m_value; } + T& operator*() { return m_value; } + const T* operator->() const { return &m_value; } + T* operator->() { return &m_value; } + + const Result& status() const { return m_status; } + +private: + T m_value; + Result m_status; +}; + +//============================================================================= +// ARRAY VIEW +//============================================================================= + +// A borrowed, contiguous, read-only view: std::span without the C++20. The +// storage belongs to the library and stays valid until the object that +// returned it is released, which is what lets the enumerators return a view +// instead of making every caller run the count-then-fill dance twice. +template +class ArrayView { +public: + ArrayView() : m_data(nullptr), m_size(0) { } + ArrayView(const T* data, size_t size) : m_data(data), m_size(size) { } + ArrayView(const std::vector& v) : m_data(v.data()), m_size(v.size()) { } + + const T* data() const { return m_data; } + size_t size() const { return m_size; } + bool empty() const { return m_size == 0; } + + const T* begin() const { return m_data; } + const T* end() const { return m_data + m_size; } + const T& operator[](size_t i) const { return m_data[i]; } + +private: + const T* m_data; + size_t m_size; +}; + +//============================================================================= +// OBJECT MODEL +//============================================================================= + +template using Ref = std::shared_ptr; + +// The base of every library-owned object. It has exactly one virtual besides +// the destructor, and that one never grows: all extension happens by deriving +// new interfaces and answering their id here. +class IObject { +public: + virtual ~IObject() = default; + + // Returns a pointer to the named interface, or nullptr if this object does + // not implement it. Callers use Query() rather than calling this. + virtual void* QueryInterface(std::string_view id) = 0; + +protected: + IObject() = default; +}; + +// Reach a role on an object. Returns null when the implementation does not +// provide it, which is the capability query. +// +// The returned Ref shares the owner's reference count (the aliasing +// constructor), so holding a role keeps the object that vends it alive. A +// caller may drop the session Ref and keep only the roles it uses. +template +Ref Query(const Ref& obj) +{ + if (!obj) { + return Ref(); + } + void* iface = obj->QueryInterface(I::kId); + if (iface == nullptr) { + return Ref(); + } + return Ref(obj, static_cast(iface)); +} + +//============================================================================= +// VOCABULARY +//============================================================================= + +enum class Codec : uint32_t { + H264 = 0, + H265, + AV1, +}; + +// The named profiles the library maps onto codec-specific profile_idc values. +// Default lets the library choose from the codec and the input geometry. +enum class Profile : uint32_t { + Default = 0, + H264Baseline, H264Main, H264High, H264High10, + H265Main, H265Main10, H265MainStillPicture, H265Rext, + AV1Main, AV1High, AV1Professional, +}; + +// How to read the samples in the caller's input image. FromFormat asks the +// library to decide from the VkFormat alone, which is right whenever the +// format is unambiguous. +enum class ColorModel : uint32_t { + FromFormat = 0, + Rgb, + YCbCr, +}; + +enum class PictureType : uint32_t { + Intra = 0, // includes IDR and intra-refresh + Predicted, + Bidirectional, +}; + +enum class FrameState : uint32_t { + Unknown = 0, // never submitted, or already released + Pending, // submitted, not yet encoded + Ready, // an Acquire will deliver it + Acquired, // delivered, awaiting Release +}; + +enum class RateControlMode : uint32_t { + Default = 0, + Disabled, // constant QP + Cbr, + Vbr, +}; + +enum class TuningMode : uint32_t { + Default = 0, + HighQuality, + LowLatency, + UltraLowLatency, + Lossless, +}; + +// How the library obtains an input image the caller already owns. +enum class ExternalHandleType : uint32_t { + NoHandle = 0, + OpaqueFd, + DmaBuf, + OpaqueWin32, + D3D11Texture, + VkImageHandle, // an image already resident on the shared VkDevice +}; + +// What InputPath the configuration resolves to. The library decides this from +// the input format, the colour model and the image layout; the caller asks +// rather than deriving it, so the rule has one implementation. +enum class InputPath : uint32_t { + // The encoder reads the caller's image directly. No intermediate. + Direct = 0, + // A transfer-queue copy into an encodable image. The preferred path + // whenever the samples are already in an encodable format and layout. + Copy, + // A compute shader converts the samples. Used only when the format, the + // colour model or the plane layout cannot be resolved by a copy. + ComputeFilter, +}; + +// What the compute filter needs of the caller's own image, when a format +// reaches the encoder through it. A host allocating the buffer has to know +// this BEFORE it allocates, and the answer is a property of the format rather +// than of any device or session. +enum class FilterAccess : uint32_t { + // No filter runs; nothing extra is required of the image. + NoFilter = 0, + // The source is planar and the filter addresses its planes individually, + // so the image needs per-plane storage views -- which in turn needs + // MUTABLE_FORMAT and a format list naming the plane formats. + PlaneStorage, + // The source is a single plane, packed or RGB, and the filter reads it as + // a storage image, so the image needs STORAGE usage. + StorageRead, +}; + +// How one input format reaches the encoder. Answered without a device, +// because a host chooses its buffer format before it has one. +struct FormatRouting { + bool supported = false; + + InputPath path = InputPath::Copy; + FilterAccess filterAccess = FilterAccess::NoFilter; + + // What the session runs at. Equal to the input format on the direct path; + // the conversion target when a filter runs. VK_FORMAT_UNDEFINED when the + // answer needs a device, which is the RGB case: an RGB session takes the + // device's first advertised encode-source format. + VkFormat sessionFormat = VK_FORMAT_UNDEFINED; +}; + +// Optional behaviour a caller may need to know about before committing to a +// configuration. Supports() answers for the device the platform is bound to. +enum class Feature : uint32_t { + DmaBufImport = 0, + ExternalSemaphores, + IntraRefresh, + HdrMetadata, + Reconfigure, + RegisteredResources, + BFrames, +}; + +//============================================================================= +// VALUE TYPES +// +// Constructed by the caller, so their layout is pinned by every caller that +// compiles. They may never gain a field. When one needs to grow, a new +// interface version takes a new type. +//============================================================================= + +// Adopt the caller's Vulkan objects. A null device asks the library to create +// its own, which is what the standalone demos do; a browser or a compositor +// always supplies one, because the encoder must run on the device that already +// owns the images. +struct PlatformCreateInfo { + VkInstance instance = VK_NULL_HANDLE; + VkPhysicalDevice physicalDevice = VK_NULL_HANDLE; + VkDevice device = VK_NULL_HANDLE; + uint32_t encodeQueueFamilyIndex = UINT32_MAX; + uint32_t computeQueueFamilyIndex = UINT32_MAX; + + // Used only when device is null: pick a device by index, or by UUID when + // gpuUuidValid is set. A UUID is the stable choice across reboots. + int32_t deviceIndex = -1; + uint8_t gpuUuid[VK_UUID_SIZE] = {}; + bool gpuUuidValid = false; + + // Enable the Vulkan validation layers on a library-created instance. No + // effect when the caller supplies its own instance. + bool enableValidation = false; + + // Keep the library off stdout and stderr. A sandboxed host has no usable + // stdio, and a library that writes to it anyway is a crash or a leak of + // whatever the host had bound to those descriptors. + // + // THE EFFECT IS PROCESS-WIDE AND ONE-WAY. Silence holds while any live + // platform, session or internal query is requesting it, and ends only when + // the last of them is gone. Two callers cannot each choose: leaving this + // false does not make the library audible while another owner needs it + // quiet. Setting it true silences output the other owner would have seen. + // + // It covers the library's gated diagnostics. It is not a redirection of + // the process's file descriptors and it does not relax a sandbox. + bool silenceStdio = false; +}; + +struct RateControl { + RateControlMode mode = RateControlMode::Default; + + uint32_t averageBitrate = 0; // bits/sec + uint32_t maxBitrate = 0; // bits/sec, VBR only + uint32_t vbvBufferSize = 0; // bits; 0 asks the library to derive one + + // Used when mode is Disabled. Ignored otherwise. + int32_t constQpIntra = 0; + int32_t constQpPredicted = 0; + int32_t constQpBidirectional = 0; + + // Clamps applied in every mode. Left at zero, the library uses the range + // the device advertises. + int32_t minQp = 0; + int32_t maxQp = 0; + + uint32_t qualityLevel = 0; // 0 asks for the driver default + TuningMode tuning = TuningMode::Default; +}; + +struct GopStructure { + uint32_t gopLength = 0; // frames; 0 means infinite + uint32_t idrPeriod = 0; // frames; 0 means IDR only at the start + uint32_t consecutiveBFrames = 0; + bool closedGop = false; +}; + +// How the caller's own samples are coded, in the code points of the video +// standards (ITU-T H.273). The library writes these into the bitstream and +// uses them to decide whether a conversion is needed. +struct ColourInfo { + uint8_t colourPrimaries = 2; // 2 = unspecified + uint8_t transferCharacteristics = 2; + uint8_t matrixCoefficients = 2; + bool fullRange = false; + + // The transfer function the caller's input samples already carry. + // + // 0 is the only value that asserts nothing, and it is the default: the + // input is taken to be in transferCharacteristics already. Every other + // value is a positive claim, checked against transferCharacteristics and + // refused when the two disagree -- this library converts the colour MODEL + // and implements no transfer function, so it cannot reconcile them. Note + // that 2 ("unspecified") is such a claim, not an absence: it is a value + // the bitstream fields above may legitimately carry, but here it would + // conflict with any bitstream that declares a real transfer function. + uint8_t inputTransferCharacteristics = 0; +}; + +// SMPTE ST 2086 mastering display and content light level, in the units the +// standard codes and a producer already holds: chromaticity scaled by 50000, +// luminance by 10000. AV1 codes the same quantities in a different fixed-point +// format and primary order; the library converts, and the caller supplies one +// spelling only. +// +// All zeros is a claim, not an absence -- it says the mastering display is +// black -- so each payload has its own presence flag and they are written +// independently. +struct HdrMetadata { + bool masteringDisplayPresent = false; + uint16_t displayPrimaryX[3] = {}; // ST 2086 order: 0 green, 1 blue, 2 red + uint16_t displayPrimaryY[3] = {}; + uint16_t whitePointX = 0; + uint16_t whitePointY = 0; + uint32_t maxLuminance = 0; + uint32_t minLuminance = 0; + + bool contentLightLevelPresent = false; + uint16_t maxContentLightLevel = 0; + uint16_t maxFrameAverageLightLevel = 0; +}; + +struct PlaneLayout { + uint64_t offset = 0; + uint64_t rowPitch = 0; + uint64_t size = 0; +}; + +enum { kMaxPlanes = 4 }; + +// A registered image or semaphore, named by an opaque value the library +// chooses. Zero is never a live registration, so a caller may use it as +// "none" without a companion flag. +// +// Opaque rather than an index into anything: a caller that could compute one +// could also collide with one, and the library is free to change what it +// stores behind it. +typedef uint64_t ResourceId; +const ResourceId kNoResource = 0; + +// What happens to an OS handle the caller passes in. +// +// This is not a detail. Under Transfer the library closes the handle on +// EVERY exit path, including a failed registration, so a caller that also +// closes it double-closes -- and a double close is not a leak, it is a +// close of whatever unrelated descriptor has since taken the number. +enum class HandleOwnership : uint32_t { + // The library takes the handle. The caller must not close it, on any + // outcome, including a failed registration. + // + // Who closes it in the end depends on how far the import got, and a + // caller can neither observe nor rely on which: once the handle has been + // handed to the driver, the driver owns it -- even if the allocation then + // fails -- and the library will not close it, because that would be a + // second close of a number this process may already have recycled. Short + // of that handoff the library closes it. Either way it is not yours. + // + // The default, because it is what the layer below does and what a + // producer handing over a dma-buf almost always wants: one owner. + Transfer = 0, + // The caller keeps the handle and closes it once registration returns. + // The library duplicates whatever it needs. + Borrow, +}; + +// Where an image lives relative to the encoder's device, which decides +// whether a queue-family ownership transfer is needed on acquire. +enum class Residency : uint32_t { + // Inferred from the handle type. Right for a dma-buf or an OS handle, + // where the library can tell. + Auto = 0, + // Allocated on the encoder's own device. No ownership transfer. + Local, + // Owned by VK_QUEUE_FAMILY_FOREIGN_EXT; acquired before use. + Foreign, +}; + +// An image the caller already owns, described well enough for the library to +// import it without guessing. Only the fields the handle type needs are read. +struct ExternalImage { + ExternalHandleType handleType = ExternalHandleType::NoHandle; + + // The OS handle, or the Vulkan image when handleType is VkImageHandle. + // + // WHO OWNS fd IS |ownership|, AND IT DEFAULTS TO Transfer: once passed, + // it is not yours to close on any outcome. Set Borrow to keep it -- the + // library then works from a private duplicate taken before anything that + // can fail. A Win32 handle is never closed by the library under either + // rule, and a VkImage is not a handle this rule applies to. + int fd = -1; + void* win32Handle = nullptr; + VkImage existingImage = VK_NULL_HANDLE; + + HandleOwnership ownership = HandleOwnership::Transfer; + + VkFormat format = VK_FORMAT_UNDEFINED; + uint32_t width = 0; + uint32_t height = 0; + VkImageTiling tiling = VK_IMAGE_TILING_OPTIMAL; + VkImageLayout layout = VK_IMAGE_LAYOUT_UNDEFINED; + ColorModel colorModel = ColorModel::FromFormat; + + // The usage and flags the image was created with. The library cannot + // discover these from a VkImage handle, and it needs them to decide + // whether the encoder can read the image directly or must stage it, so + // leaving them zero makes an otherwise valid registration fail. + VkImageUsageFlags usage = 0; + VkImageCreateFlags createFlags = 0; + + Residency residency = Residency::Auto; + + bool hasDrmFormatModifier = false; + uint64_t drmFormatModifier = 0; + + uint32_t planeCount = 0; + PlaneLayout planeLayouts[kMaxPlanes] = {}; + + uint64_t allocationSize = 0; + uint32_t memoryTypeBits = 0; // 0 when the caller does not know + uint32_t memoryTypeIndex = UINT32_MAX; + + // WHICH DEVICE ALLOCATED THIS, so the library can refuse an image that + // belongs to another one. On a multi-GPU host an import that silently + // succeeds against the wrong device is worse than a refusal: it fails + // later, in the driver, with nothing naming the cause. + // + // ALL-ZERO MEANS "NOT STATED", and is what a caller that genuinely does + // not know passes. The library then skips the check, so leaving these + // empty is not the safe default -- it is the unchecked one. + uint8_t deviceUUID[VK_UUID_SIZE] = {}; + uint8_t driverUUID[VK_UUID_SIZE] = {}; + uint8_t deviceLUID[VK_LUID_SIZE] = {}; + bool deviceLuidValid = false; +}; + +// A semaphore the caller owns elsewhere, imported by OS handle. Timeline +// semaphores only: a binary semaphore cannot express "frame N is done" to a +// consumer that was not waiting at the moment it was signalled. +// +// Who closes the handle is |ownership| below, and it defaults to Transfer. +struct ExternalSemaphore { + ExternalHandleType handleType = ExternalHandleType::NoHandle; + int fd = -1; + void* win32Handle = nullptr; + + // As for an image: Transfer means the library closes fd, including on + // failure. + HandleOwnership ownership = HandleOwnership::Transfer; +}; + +// A wait/signal pair the library honours around the encode submission. A +// timeline value of zero means the semaphore is binary. +struct SemaphoreWait { + VkSemaphore semaphore = VK_NULL_HANDLE; + uint64_t value = 0; +}; + +// One frame handed to the encoder. frameId is the caller's, echoed back on +// every result and status query, and is how a caller correlates without +// holding library state. +struct FrameSubmit { + uint64_t frameId = 0; + uint64_t pts = 0; + + // Exactly one of these describes the samples: an image registered earlier + // with IResourceRegistry, or an image described inline. A registered image + // costs no import here, which is why a compositor recycling buffers + // registers them once. + ResourceId registeredImage = kNoResource; + ExternalImage image; -// High-level interface of the video encoder -class VulkanVideoEncoder : public virtual VkVideoRefCountBase { + bool forceIdr = false; + bool isLastFrame = false; + bool hasQpOverride = false; + int32_t qpOverride = 0; + + ArrayView waitSemaphores; + ArrayView signalSemaphores; + + // ---- fences, for a producer this encoder does not share a queue with ---- + + // Wait for this fence before reading the image. A compositor handing over + // a buffer it has just rendered into passes the fence that render + // signalled; without it the encode may read the frame half-written. + // + // THE LIBRARY CLOSES THIS FD, on every exit path including a refusal, and + // does NOT write -1 back. A caller that re-submits the same frame must + // therefore pass a fresh descriptor each time -- in practice a dup, so + // that the caller keeps the original and every attempt waits properly. + // -1 means no ordering is required. + int acquireFenceFd = -1; + + // Receives a fence signalled when the encoder has finished reading the + // image, so the caller knows when it may write the buffer again. + // + // The library writes -1 through this pointer before anything in the + // submission can refuse, so the storage never keeps a stale fd from a + // previous attempt. The caller owns whatever it receives and closes it. + // Null asks for no release fence. + int* releaseFenceFd = nullptr; +}; + +// An encoded frame. The bitstream points into library storage and stays valid +// until Release(frameId); a caller that needs it longer copies it. +struct EncodedFrame { + uint64_t frameId = 0; + uint64_t pts = 0; + uint64_t dts = 0; + + ArrayView bitstream; + + PictureType pictureType = PictureType::Intra; + bool isIdr = false; + uint32_t temporalLayerId = 0; + + // WHAT HAPPENED TO THIS FRAME. Ok means |bitstream| is its encode. + // + // Timeout means the frame was dropped against its deadline: it is still + // delivered, still must be Released, and carries no bitstream. A caller + // that treats every delivered frame as an encode emits an empty one and + // calls it output. + // + // This is a property of the frame, not of the Acquire call, which is why + // it rides here rather than in the Result of the acquisition: acquiring a + // dropped frame succeeded. + ResultCode outcome = ResultCode::Ok; +}; + +struct InputFormat { + VkFormat format = VK_FORMAT_UNDEFINED; + ColorModel colorModel = ColorModel::FromFormat; + + // What accepting this format costs. A format that resolves to + // InputPath::ComputeFilter is supported but not free. + InputPath path = InputPath::Direct; + bool isOptimal = true; +}; + +struct QpRange { + int32_t minQp = 0; + int32_t maxQp = 0; + bool known = false; +}; + +struct RateControlCaps { + bool supportsCbr = false; + bool supportsVbr = false; + bool supportsConstantQp = false; + uint32_t maxQualityLevels = 0; + + // The highest bitrate the device accepts for this profile, in bits/sec. + // Zero means the device states no ceiling. A host that clamps its own + // request needs this before it builds a configuration, which is why it is + // a capability rather than a refusal at session creation. + uint32_t maxBitrate = 0; +}; + +// What the session has done and what it is holding. A caller watching for a +// stall reads this: a completion counter that does not move while frames are +// outstanding and output buffers are free is the signature of one. +// +// The descriptor API answers this as two chained structures; here it is one +// value, because a caller that wants the counters always wants the +// diagnostics alongside them. +struct CompletionStats { + uint64_t completionCounter = 0; // monotonic; the drain target + uint64_t framesTimedOut = 0; // deadline drops delivered + uint64_t lateCaptures = 0; // captures discarded after a timeout drop + uint64_t framesCancelled = 0; + + uint32_t framesPending = 0; // submitted, not yet encoded + uint32_t framesReady = 0; // retrievable now + uint32_t framesAcquired = 0; // delivered, awaiting Release + + // Misuses the library has recorded, and the most recent one. A non-zero + // count is a caller bug the library chose to survive rather than refuse. + uint64_t diagnosticCount = 0; + char lastDiagnostic[256] = {}; +}; + +struct RuntimeInfo { + char implementationName[64] = {}; + + bool isHardwareAccelerated = false; + // The encoder can take a frame by handle rather than by copy. A host + // decides whether to allocate importable buffers on this. + bool supportsNativeHandle = false; + // The rate controller is the hardware's, so a host must not second-guess + // it by re-driving the bitrate per frame. + bool trustedRateController = false; + bool supportsSimulcast = false; + // Resolution can change without rebuilding the session. + bool supportsFrameSizeChange = false; + bool reportsAverageQp = false; + + // What the encoder wants the coded extent rounded to. One means no + // constraint; a host that ignores these gets a refusal at session + // creation rather than a silent crop. + uint32_t resolutionAlignmentWidth = 1; + uint32_t resolutionAlignmentHeight = 1; + // Whether that alignment applies to every simulcast layer or only the + // base one. + bool applyAlignmentToAllSimulcastLayers = false; +}; + +//============================================================================= +// PLATFORM-SCOPE INTERFACES +//============================================================================= + +// What the device can do, answerable before any session exists. Every view +// returned here is owned by the platform and valid for its lifetime. +class IEncoderCaps : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IEncoderCaps/1"; + + // Empty when no device reachable from this platform can encode. A + // platform is created from an instance, which succeeds before any device + // is examined, so this is the test for "there is an encoder here at all". + virtual ArrayView Codecs() const = 0; + virtual ArrayView Profiles(Codec codec) const = 0; + + // The input formats this profile accepts, each tagged with the path it + // resolves to. A caller picks a format whose path it is willing to pay for + // instead of discovering the cost at session creation. + virtual ArrayView InputFormats(Profile profile) const = 0; + virtual ArrayView DrmModifiers(Profile profile, VkFormat format) const = 0; + + virtual RateControlCaps RateControl(Profile profile) const = 0; + + virtual VkExtent2D MinCodedExtent(Profile profile) const = 0; + virtual VkExtent2D MaxCodedExtent(Profile profile) const = 0; + + virtual bool Supports(Feature feature) const = 0; +}; + +// The Vulkan objects the encoder runs on, whether the caller supplied them or +// the library created them. +class IDeviceBinding : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IDeviceBinding/1"; + + virtual VkInstance Instance() const = 0; + virtual VkPhysicalDevice PhysicalDevice() const = 0; + virtual VkDevice Device() const = 0; + virtual uint32_t EncodeQueueFamilyIndex() const = 0; + virtual uint32_t ComputeQueueFamilyIndex() const = 0; + virtual PFN_vkGetInstanceProcAddr GetInstanceProcAddr() const = 0; + + // WHICH device this is, by the identity the Vulkan driver reports rather + // than by the handle. A host that adopted a device needs to confirm the + // encoder bound the one it meant: on a multi-GPU machine the wrong answer + // is not an error here, it is every subsequent import failing for a reason + // that names a buffer instead of a device. + // + // Handles cannot answer this. A VkPhysicalDevice from one instance and one + // from another are different values for the same hardware, and the encoder + // may hold its own instance. + // + // False when the device cannot be identified, in which case |outUuid| is + // untouched. + virtual bool DeviceUuid(uint8_t outUuid[VK_UUID_SIZE]) const = 0; +}; + +// HDR10 static metadata, reachable through Query on a configuration whose +// codec can carry it. H.264 has no mastering-display or content-light SEI, so +// Query returns null for an H.264 configuration and the caller learns that +// without creating a session. +class IHdrMetadataConfig : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IHdrMetadataConfig/1"; + + virtual void SetHdrMetadata(const HdrMetadata& metadata) = 0; +}; + +// A configuration under construction. The library owns it, so it can answer +// questions about itself rather than being an inert record the caller fills +// and hopes about. +class IEncoderConfig : public IObject { public: - virtual VkResult Initialize(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, - int argc, const char** argv) = 0; - virtual int64_t GetNumberOfFrames() = 0; - virtual VkResult EncodeNextFrame(int64_t& frameNumEncoded) = 0; - virtual VkResult GetBitstream() = 0; + static constexpr std::string_view kId = "vk.video.enc.IEncoderConfig/1"; + + virtual IEncoderConfig& SetCodedExtent(uint32_t width, uint32_t height) = 0; + virtual IEncoderConfig& SetInputExtent(uint32_t width, uint32_t height) = 0; + virtual IEncoderConfig& SetFrameRate(uint32_t numerator, uint32_t denominator) = 0; + virtual IEncoderConfig& SetInputFormat(VkFormat format, ColorModel colorModel) = 0; + virtual IEncoderConfig& SetRateControl(const RateControl& rateControl) = 0; + virtual IEncoderConfig& SetGop(const GopStructure& gop) = 0; + virtual IEncoderConfig& SetColourInfo(const ColourInfo& colour) = 0; + + // Refuse a session whose device lacks the extensions the import paths + // need, naming the missing one. + // + // Off, a device missing them is accepted here and fails at the first + // import, where the message is about a handle rather than about the + // device -- so a host that intends to import anything asks for this and + // learns at session creation instead. + virtual IEncoderConfig& RequireImportExtensions(bool require) = 0; + + // Write the bitstream to this path. Absent, the encoder keeps every frame + // in memory for IBitstreamSource, which is what an embedding host wants: + // it has nowhere to write and no reason to. + virtual IEncoderConfig& SetOutputPath(const char* path) = 0; + + virtual Codec GetCodec() const = 0; + virtual Profile GetProfile() const = 0; + + // Which path this configuration resolves to, and the format the session + // will actually run at. Both are decided here so the rule has a single + // implementation and a caller can log the answer instead of predicting it. + virtual InputPath ResolveInputPath() const = 0; + virtual VkFormat ResolveSessionFormat() const = 0; + + // Everything checkable without a session, checked in one place. + virtual Result Validate() const = 0; }; +//============================================================================= +// SESSION-SCOPE INTERFACES +//============================================================================= + +// Hand frames to the encoder. +class IFrameSubmitter : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IFrameSubmitter/1"; + + // Submit one frame. Returns once the work is queued, not once it is + // encoded; the frame is complete when it reaches FrameState::Ready. + virtual Result SubmitFrame(const FrameSubmit& frame) = 0; + + // Drop a submitted frame that has not been encoded yet. A frame already + // Ready is not cancellable and must be acquired and released. + virtual Result CancelFrame(uint64_t frameId) = 0; + + virtual FrameState GetFrameState(uint64_t frameId) const = 0; +}; + +// Take encoded frames out. +// +// Every successful Acquire is paired with a Release. Until Release, the +// bitstream storage for that frame is pinned, and an encoder that runs out of +// storage stops accepting submissions. +class IBitstreamSource : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IBitstreamSource/1"; + + // The next frame in encode order, or ResultCode::NotReady when none is ready. + virtual Expected AcquireNext() = 0; + + // A specific frame by the id its submission carried. ResultCode::NotReady + // while it is still encoding; ResultCode::InvalidArgument if it was never + // submitted or has already been released. + virtual Expected Acquire(uint64_t frameId) = 0; + + virtual void Release(uint64_t frameId) = 0; +}; + +// Learn that frames have completed without polling. +// +// A caller uses exactly one of these styles. The callback runs on a library +// thread and must not call back into the session; it exists to wake the +// caller's own loop. +class ICompletionSignal : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.ICompletionSignal/1"; + + typedef void (*CompletionCallback)(uint64_t frameId, void* userData); + + // Invoked exactly once when the encoder is done with a |userData| it was + // given -- on replacement, on detach, or at teardown -- and only after any + // invocation still running has returned. + typedef void (*UserDataRelease)(void* userData); + + // Install, replace, or (with a null callback) clear the completion + // callback. + // + // |release|, when given, hands ownership of |userData| to the encoder. + // Prefer it to keeping the cookie alive by arrangement with teardown + // order: the encoder knows when the last invocation has returned and the + // caller does not. + // + // Clearing is a quiesce point -- once it returns, no invocation is in + // flight -- so a caller that manages the cookie itself may destroy it + // immediately afterwards. + virtual Result SetCallback(CompletionCallback callback, void* userData, + UserDataRelease release = nullptr) = 0; + + // Monotonic count of frames that have become Ready. A caller that samples + // it needs no callback and no handle. + virtual uint64_t CompletedCount() const = 0; + + // A semaphore the library signals as frames complete, for a caller that + // already waits on Vulkan. Null when the implementation has none. + virtual VkSemaphore CompletionSemaphore() const = 0; + + // An OS handle to wait on from a non-Vulkan loop. + // + // Each call returns a NEW handle and the caller closes that one. The + // library keeps its own and closes it at teardown, so closing an exported + // handle never disturbs the encoder, and the encoder's teardown never + // invalidates a handle a caller still holds. + // + // Exports share the underlying completion object. They are independent + // close obligations, not independent queues: a completion drained through + // one export is not redelivered to another. Waking on any of them means + // "drain what is available and reconcile against CompletedCount()", which + // is the same discipline a single handle needs, because the platforms + // differ in what an unread signal accumulates to. + // + // The handle is close-on-exec where the platform expresses that. + virtual Expected ExportCompletionHandle() = 0; +}; + +// What a terminal Finish() actually proved. +// +// Finish() returning an error and the GPU still holding the caller's input are +// different questions, and a single Result cannot answer both. A producer that +// lent the encoder an image asks this one before it takes the image back. +enum class ShutdownDisposition : int32_t { + // Every library worker joined, the whole-device wait succeeded and no + // device loss was seen. The input may be released -- even when Finish() + // reported an earlier encode or file-output failure. Bitstreams already + // copied out stay valid until Release(). + Idle = 0, + + // The workers joined and the device was lost. Pending Vulkan use is + // retired, so nothing is still reading -- but shared external contents and + // peer image layouts are not valid, which is not the same permission. + LostDeviceRetired, + + // Any other terminal wait failure, or a shutdown still underway. Nothing + // established that the device stopped using the images, imported + // semaphores and registrations involved: keep every one of them, and every + // producer-side dependency, alive. Cancelling or abandoning frames does + // not change this. + Unproven, +}; + +// Thread-safe to read at any time; meaningful after Finish() returns. +struct ShutdownSnapshot { + ShutdownDisposition disposition = ShutdownDisposition::Unproven; + + // The FIRST encode or shutdown failure, kept. A clean later step does not + // clear it. + ResultCode firstError = ResultCode::Ok; + + bool workersJoined = false; + bool callbackDetached = false; + + // Sticky, and separate from the disposition on purpose: a lost device + // followed by a wait that returns success is still a lost device. + bool deviceLostObserved = false; + + // False while a shutdown is still underway or was left Unproven. + bool complete = false; +}; + +// Why Finish()'s Result is not enough, and what to do about it. +class IShutdownDiagnostics : public IObject { +public: + static constexpr std::string_view kId = + "vk.video.enc.IShutdownDiagnostics/1"; + + virtual ShutdownSnapshot Shutdown() const = 0; +}; + +// Register images and semaphores once and refer to them by index afterwards. +// +// This exists because importing a dma-buf costs a Vulkan image creation, and a +// compositor recycles the same handful of buffers for the life of a stream. +// Registration moves that cost out of the per-frame path. +class IResourceRegistry : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IResourceRegistry/1"; + + // Returns the id to put in FrameSubmit::registeredImage. + // + // |outHandleConsumed|, when given, is WRITTEN ON EVERY RETURN, including + // every refusal -- which is the only reason it is an out-parameter rather + // than part of the returned value. Under HandleOwnership::Transfer a + // refusal still consumes the handle, so a caller that learns this only on + // success learns it exactly when it does not matter. + // + // DIAGNOSTIC ONLY. A false under Transfer is a library defect, and still + // not an instruction to close: the caller cannot know how far the import + // got, so closing on it risks the double close the ownership rule exists + // to prevent. Assert on it; do not act on it. + virtual Expected RegisterImage( + const ExternalImage& image, + bool* outHandleConsumed = nullptr) = 0; + virtual Result UnregisterImage(ResourceId id) = 0; + + // Whether this image could be registered, and what it would cost, without + // registering it. A caller uses this to choose a buffer format before it + // allocates. + virtual Expected QueryImageSupport(const ExternalImage& image) = 0; + + virtual Expected RegisterSemaphore(const ExternalSemaphore& semaphore) = 0; + virtual Result UnregisterSemaphore(ResourceId id) = 0; +}; + +// What the implementation is and what it did. Nothing here changes encoder +// behaviour; it is for logs and for the runtime-context detail that Result +// deliberately does not carry. +class IDiagnostics : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IDiagnostics/1"; + + virtual RuntimeInfo GetRuntimeInfo() const = 0; + + // The last error in full, with the context Result omits. Valid until the + // next call that returns a failing Result. + virtual const char* LastErrorDetail() const = 0; + + virtual uint64_t FramesSubmitted() const = 0; + virtual uint64_t FramesEncoded() const = 0; + + // The session's own accounting, which is authoritative where the two + // counters above are only this interface's view of it. + virtual Expected GetCompletionStats() const = 0; +}; + +// An encoding session: a configured encoder with frames in flight. +// +// The session is lifecycle only. Everything a caller does with frames is on a +// role reached through Query, so a caller depends on the four methods it uses +// rather than on every method the encoder has. +class IEncoderSession : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IEncoderSession/1"; + + // Apply what can change mid-stream and refuse what cannot, naming it. + // + // Rate control and frame rate move, and take effect at the next encoded + // frame rather than at an IDR boundary. Everything else -- resolution, + // profile, input format, colour, GOP structure, quality level -- is + // settled in the sequence header written once, or in the input routing + // the session was built around, so changing one needs a new session. Such + // a change is refused with ResultCode::UnsupportedFeature, the session + // keeps running exactly as it was, and IDiagnostics::LastErrorDetail + // names the offending field. + // + // A FIELD LEFT UNSET MEANS "LEAVE IT ALONE", not "set it to zero". The + // session holds the configuration it was created with and overlays only + // what this one states, so a caller may pass either the configuration it + // built the session from with the rate changed, or a fresh one carrying + // nothing but the new rate. Without that rule, moving the frame rate + // would mean restating the resolution, the format and the bitrate purely + // to avoid being refused for changing them. + // + // The rate control MODE is not among what moves: a session created + // without rate control cannot be moved into CBR, only re-created. + virtual Result Reconfigure(const Ref& config) = 0; + + // Encode everything submitted and wait for it, WITHOUT ending the + // stream. The session stays fully usable: further submissions are + // accepted and behave exactly as they would have without the drain, and + // a caller may drain as often as it likes. + // + // This is the one to reach for. It is what "encode what I have given you + // so far" means. + virtual Result Drain() = 0; + + // END the stream: encode everything pending, write the trailing bitstream + // a reordered GOP still owes, and release the encoder. + // + // TERMINAL, and named so rather than "flush" because that name invites a + // caller to treat it as a checkpoint. After this, a submission is refused + // with NotConfigured and CompletionSemaphore() answers VK_NULL_HANDLE. + // Frames already encoded stay acquirable and their bitstream pointers + // stay valid, so a caller finishes collecting after finishing the stream. + virtual Result Finish() = 0; + + // Discard everything in flight. Returns how many frames were dropped. + // Frames already Ready are not dropped. + virtual Expected AbandonAll() = 0; +}; + +//============================================================================= +// PLATFORM +//============================================================================= + +// Releasing the process-wide Vulkan instance the encoder stands up. +// +// An OWN-mode platform caches its VkInstance for the process lifetime so that +// repeated create/destroy cycles reuse one instance rather than issuing a +// second vkCreateInstance -- which a sandboxed process may no longer be +// permitted to do. Dropping every Ref therefore does NOT destroy the instance, +// and a process that exits without calling Retire() leaves it standing. Under +// a driver that audits allocations at exit, that is reported as a leak. +// +// Retire() is how a caller that owns the process lifetime releases it at a +// moment of its choosing: late enough that no encoding remains, early enough +// that calling the driver is still allowed. It is permanent -- after it, every +// CreateSession and every VkEncCreatePlatform for this device fails rather +// than standing a second instance up -- and idempotent. +// +// A caller that simply runs to process exit need not call it at all. +class IPlatformLifetime : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IPlatformLifetime/1"; + + // Returns the number of cached device contexts released; zero if the + // instance was already retired. + virtual uint32_t Retire() = 0; +}; + +// The root object: a device the encoder can run on. Everything else is +// created from here. +class IEncoderPlatform : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.IEncoderPlatform/1"; + + virtual Ref Caps() const = 0; + virtual Ref DeviceBinding() const = 0; + + virtual Expected> CreateConfig(Codec codec, Profile profile) = 0; + virtual Expected> CreateSession(const Ref& config) = 0; +}; + +} // namespace enc +} // namespace video +} // namespace vk + +//============================================================================= +// THE EXPORTED SYMBOLS +// +// Two, and together they are the whole ABI: VkEncCreatePlatform below, and +// VkEncClassifyFormat after it. +// +// A host that treats the encoder as an optional dependency -- delay-loaded on +// Windows, dlopen'd elsewhere -- resolves them by name. The absence of +// VkEncCreatePlatform is how such a host decides to run without an encoder, +// so both are extern "C" and their signatures are fixed. +//============================================================================= + +extern "C" VK_ENC_EXPORT +VkResult VkEncCreatePlatform(const vk::video::enc::PlatformCreateInfo& createInfo, + vk::video::enc::Ref& outPlatform); -extern "C" VK_VIDEO_ENCODER_EXPORT -VkResult CreateVulkanVideoEncoder(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, - int argc, const char** argv, - VkSharedBaseObj& vulkanVideoEncoder); +// How a format reaches the encoder, answered without a platform. +// +// The second of the two, and it is separate from everything above for a +// reason a caller feels: this is a property of the format, so +// requiring a platform to ask would mean creating a device to answer a +// question that does not depend on one. A host picks its buffer format while +// negotiating with a producer, long before the encoder exists. +// +// Returns VK_ERROR_FORMAT_NOT_SUPPORTED, with outRouting->supported false, for +// a format this library cannot take on any route. +extern "C" VK_ENC_EXPORT +VkResult VkEncClassifyFormat(VkFormat format, + vk::video::enc::ColorModel colorModel, + vk::video::enc::FormatRouting* outRouting); #endif /* _VULKAN_VIDEO_ENCODER_H_ */ diff --git a/vk_video_encoder/include/vulkan_video_encoder_argv.h b/vk_video_encoder/include/vulkan_video_encoder_argv.h new file mode 100644 index 00000000..773594da --- /dev/null +++ b/vk_video_encoder/include/vulkan_video_encoder_argv.h @@ -0,0 +1,58 @@ +/* + * Copyright 2024 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef _VULKAN_VIDEO_ENCODER_ARGV_H_ +#define _VULKAN_VIDEO_ENCODER_ARGV_H_ + +// VK_VIDEO_ENCODER_EXPORT tags symbol that will be exposed by the shared library. +#if defined(VK_VIDEO_ENCODER_SHAREDLIB) +#if defined(_WIN32) +#if defined(VK_VIDEO_ENCODER_IMPLEMENTATION) +#define VK_VIDEO_ENCODER_EXPORT __declspec(dllexport) +#else +#define VK_VIDEO_ENCODER_EXPORT __declspec(dllimport) +#endif +#else +#if defined(VK_VIDEO_ENCODER_IMPLEMENTATION) +#define VK_VIDEO_ENCODER_EXPORT __attribute__((visibility("default"))) +#else +#define VK_VIDEO_ENCODER_EXPORT +#endif +#endif +#else +#define VK_VIDEO_ENCODER_EXPORT +#endif + +#include "vulkan_interfaces.h" +#include "VkCodecUtils/VkVideoRefCountBase.h" + +// High-level interface of the video encoder +class VulkanVideoEncoder : public virtual VkVideoRefCountBase { +public: + virtual VkResult Initialize(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, + int argc, const char** argv) = 0; + virtual int64_t GetNumberOfFrames() = 0; + virtual VkResult EncodeNextFrame(int64_t& frameNumEncoded) = 0; + virtual VkResult GetBitstream() = 0; +}; + + +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoder(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, + int argc, const char** argv, + VkSharedBaseObj& vulkanVideoEncoder); + +#endif /* _VULKAN_VIDEO_ENCODER_ARGV_H_ */ diff --git a/vk_video_encoder/include/vulkan_video_encoder_ext.h b/vk_video_encoder/include/vulkan_video_encoder_ext.h deleted file mode 100644 index 7b7ffc16..00000000 --- a/vk_video_encoder/include/vulkan_video_encoder_ext.h +++ /dev/null @@ -1,307 +0,0 @@ -/* - * Copyright 2024-2025 NVIDIA Corporation. - * - * Licensed under the Apache License, Version 2.0 (the "License"); - * you may not use this file except in compliance with the License. - * You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -#ifndef _VULKAN_VIDEO_ENCODER_EXT_H_ -#define _VULKAN_VIDEO_ENCODER_EXT_H_ - -#include "vulkan_video_encoder.h" -#include - -//============================================================================= -// External Frame Input with Synchronization -// -// Extends VulkanVideoEncoder for frame-at-a-time operation with -// externally-provided VkImages and timeline semaphore synchronization. -// This is the interface for cross-process encoder services. -// -// Usage flow: -// 1. CreateVulkanVideoEncoderExt() to create the encoder -// 2. InitializeExt() with structured config (not argc/argv) -// 3. For each frame: -// a. SubmitExternalFrame() with imported VkImage + sync info -// b. Poll GetEncodedFrame() for completed bitstream -// 4. Flush() to drain pending frames -//============================================================================= - -//============================================================================= -// Encoder Configuration (structured, not argc/argv) -//============================================================================= -struct VkVideoEncoderConfig { - // Codec - VkVideoCodecOperationFlagBitsKHR codec; - - // Encode output resolution - uint32_t encodeWidth; - uint32_t encodeHeight; - - // Input format (what the external frames will be) - VkFormat inputFormat; - uint32_t inputWidth; - uint32_t inputHeight; - - // Rate control - // 0 = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR - // 1 = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_KHR (constant QP) - // 2 = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_KHR - // 3 = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_KHR - uint32_t rateControlMode; - uint32_t averageBitrate; // bits/sec - uint32_t maxBitrate; // bits/sec (VBR) - uint32_t vbvBufferSize; // bits (0 = default) - - // Constant QP (when rateControlMode == DISABLED) - int32_t constQpI; - int32_t constQpP; - int32_t constQpB; - int32_t minQp; - int32_t maxQp; - - // GOP structure - uint32_t gopLength; // Frames per GOP - uint32_t consecutiveBFrames;// B-frames between I/P (0 = no B-frames) - uint32_t idrPeriod; // 0 = every GOP starts with IDR - VkBool32 closedGop; - - // Frame rate - uint32_t frameRateNum; - uint32_t frameRateDen; - - // Quality - uint32_t qualityLevel; // 0 = default - - // Tuning mode (VkVideoEncodeTuningModeKHR): - // 0 = DEFAULT, 1 = HIGH_QUALITY, 2 = LOW_LATENCY, - // 3 = ULTRA_LOW_LATENCY, 4 = LOSSLESS - // LOSSLESS engages transquant-bypass + QP0 in the codec config, producing - // bit-exact output (per the Vulkan spec). Requires rateControlMode DISABLED. - uint32_t tuningMode; - - // Color info (VUI) - uint8_t colourPrimaries; - uint8_t transferCharacteristics; - uint8_t matrixCoefficients; - VkBool32 videoFullRange; - - // Enable the built-in compute filter for input preprocessing - // Set to VK_TRUE if input format may not be directly encodable - // (e.g. RGBA input that needs RGBA->NV12 conversion) - VkBool32 enablePreprocessFilter; - - // Device selection (-1 = auto, matches first discrete GPU) - int32_t deviceId; - uint8_t gpuUUID[VK_UUID_SIZE]; // Preferred GPU UUID (all zeros = auto) - - // Bitstream output file path (null or empty = encoder library default, e.g. out.264/out.265/out.ivf) - const char* outputPath; - - // Debug - VkBool32 verbose; - VkBool32 validate; // Vulkan validation layers - - // External VkInstance (optional) - // When non-null, the encoder creates its VkDevice on this instance - // instead of creating its own. Required for cross-process import - // on Windows where opaque Win32 handles are scoped per-instance. - VkInstance externalInstance; -}; - -//============================================================================= -// External Frame Descriptor -// -// Describes a frame to encode that was allocated externally -// (e.g. imported from DMA-BUF in a cross-process encoder service). -//============================================================================= -struct VkVideoEncodeInputFrame { - // The VkImage to encode (must be on the same device as the encoder) - VkImage image; - VkImageView imageView; // Can be VK_NULL_HANDLE if not needed - - // Image properties (must match the actual image) - VkFormat format; - uint32_t width; - uint32_t height; - VkImageTiling imageTiling = VK_IMAGE_TILING_OPTIMAL; // Must match actual image for path selection - VkImageLayout currentLayout; // Current layout of the image - - // Frame identification - uint64_t frameId; // Unique frame identifier - uint64_t pts; // Presentation timestamp (90kHz or custom) - - // Frame type overrides (0 = let encoder decide via GOP structure) - VkBool32 forceIDR; - VkBool32 forceIntra; - - // Set to VK_TRUE for the last frame to properly close the GOP - // and write end-of-stream markers. Without this, decoders may - // not be able to decode the trailing frames. - VkBool32 isLastFrame = VK_FALSE; - - // Per-frame QP override (-1 = use session default) - int32_t qpOverride; - - // Unique image index for query pool slot mapping and in-flight tracking. - // Same concept as the internal image pool index and debug tracking. - // Set to -1 to use the internal image pool index (legacy default). - int32_t uniqueImageIndex = -1; - - // Synchronization: wait semaphores - // The encoder will wait on these before accessing the image. - // Typically this is the producer's graph timeline semaphore. - uint32_t waitSemaphoreCount; - VkSemaphore* pWaitSemaphores; // Array of semaphores to wait on - uint64_t* pWaitSemaphoreValues; // Timeline values (0 for binary semaphores) - - // Synchronization: signal semaphores - // The encoder will signal these after the image is no longer needed. - // Typically this is the consumer's release timeline semaphore. - uint32_t signalSemaphoreCount; - VkSemaphore* pSignalSemaphores; // Array of semaphores to signal - uint64_t* pSignalSemaphoreValues; // Timeline values (0 for binary) -}; - -//============================================================================= -// Encoded Frame Result -// -// Returned by GetEncodedFrame() after encoding completes. -//============================================================================= -struct VkVideoEncodeResult { - uint64_t frameId; // Matches VkVideoEncodeInputFrame::frameId - uint64_t pts; // Pass-through from input - uint64_t dts; // Decode timestamp (encoder-assigned) - - // Bitstream - const uint8_t* pBitstreamData; // Pointer to encoded data (valid until next GetEncodedFrame) - uint32_t bitstreamSize; // Size in bytes - - // Frame info - uint32_t pictureType; // 0=I, 1=P, 2=B - VkBool32 isIDR; - uint32_t temporalLayerId; - - // Encode status - VkResult status; // VK_SUCCESS or error -}; - -//============================================================================= -// Extended Encoder Interface -// -// Extends VulkanVideoEncoder with external frame input and sync support. -// The base VulkanVideoEncoder methods (Initialize, EncodeNextFrame, etc.) -// remain for backward compatibility with file-based encoding. -//============================================================================= -class VulkanVideoEncoderExt : public VulkanVideoEncoder { -public: - // Initialize with structured config (alternative to argc/argv) - virtual VkResult InitializeExt(const VkVideoEncoderConfig& config) = 0; - - // Submit an externally-provided frame for encoding. - // The encoder will: - // 1. Wait on the input frame's wait semaphores - // 2. If format conversion is needed, run the compute filter - // 3. Copy the external image to an internal pool image (staging) - // 4. Signal the input frame's signal semaphores (staging complete) - // 5. Encode from the internal pool image - // - // This is non-blocking: the frame is queued for encoding. - // Call GetEncodedFrame() to retrieve the bitstream. - // - // pStagingCompleteSemaphore [out, optional]: if non-null, receives the - // binary semaphore that is signaled when the staging copy completes. - // This is useful when the caller needs to chain additional GPU work - // (e.g. display blit) that reads the same external image and needs - // to know when the encoder is done reading it. The caller can then - // signal their own release semaphore after both operations complete. - // - // If the caller passes signal semaphores in the frame, those are - // signaled at staging completion time (same point as this semaphore). - // If the caller needs the release to happen AFTER additional work - // (e.g. display blit), do NOT pass signal semaphores in the frame; - // instead use pStagingCompleteSemaphore to chain the work, then - // signal the release semaphore from the final submission. - // - // Returns VK_SUCCESS if the frame was accepted for encoding. - // Returns VK_NOT_READY if the encoder's internal queue is full (try again later). - virtual VkResult SubmitExternalFrame(const VkVideoEncodeInputFrame& frame, - VkSemaphore* pStagingCompleteSemaphore = nullptr) = 0; - - // === Asynchronous Bitstream Retrieval === - // - // After SubmitExternalFrame(), the encode happens asynchronously. - // Use these methods to retrieve the encoded bitstream without blocking - // the encode pipeline. - - // Poll: check if a specific frame's encode has completed. - // Returns VK_SUCCESS if the bitstream is ready to read. - // Returns VK_NOT_READY if still encoding. - virtual VkResult PollEncodeComplete(uint64_t frameId) = 0; - - // Get the next completed encoded frame (FIFO order). - // Returns VK_SUCCESS and fills 'result' if a frame is ready. - // Returns VK_NOT_READY if no frames are ready yet. - // - // The pBitstreamData pointer in result is valid until ReleaseEncodedFrame() - // is called for this frameId. This allows the caller to read the bitstream - // at their own pace (write to file, send via IPC, etc.) while encoding - // continues on subsequent frames. - virtual VkResult GetEncodedFrame(VkVideoEncodeResult& result) = 0; - - // Release an encoded frame's bitstream buffer back to the pool. - // Must be called after the caller is done reading pBitstreamData. - // The bitstream buffer is returned to the pool for reuse. - virtual void ReleaseEncodedFrame(uint64_t frameId) = 0; - - // Get the fence associated with a frame's encode completion. - // The caller can wait on this fence externally (e.g. in a thread pool) - // instead of polling PollEncodeComplete(). - // Returns VK_NULL_HANDLE if the frame hasn't been submitted yet. - virtual VkFence GetEncodeFence(uint64_t frameId) = 0; - - // === Flush and Drain === - - // Flush: encode all pending frames and make their bitstreams available. - // Blocks until all pending frames are encoded. - virtual VkResult Flush() = 0; - - // === Dynamic Reconfiguration === - - // Change rate control parameters mid-stream without session reset. - // Takes effect at the next IDR frame (or immediately if forceIDR). - virtual VkResult Reconfigure(const VkVideoEncoderConfig& config) = 0; - - // === Capability Query === - - // Query encoder capabilities for the configured codec. - // Can be called before InitializeExt() to check support. - virtual VkBool32 SupportsFormat(VkFormat inputFormat) const = 0; - virtual uint32_t GetMaxWidth() const = 0; - virtual uint32_t GetMaxHeight() const = 0; - - // === Device Access === - - // Get the encoder's Vulkan device handles. - // Use these for DMA-BUF import, semaphore creation, etc. - // The encoder owns these handles — caller must NOT destroy them. - virtual VkDevice GetVkDevice() const = 0; - virtual VkPhysicalDevice GetVkPhysicalDevice() const = 0; - virtual VkInstance GetVkInstance() const = 0; -}; - -// Factory function for the extended encoder interface -extern "C" VK_VIDEO_ENCODER_EXPORT -VkResult CreateVulkanVideoEncoderExt( - VkSharedBaseObj& vulkanVideoEncoder); - -#endif /* _VULKAN_VIDEO_ENCODER_EXT_H_ */ diff --git a/vk_video_encoder/internal/vulkan_video_encoder_ext.h b/vk_video_encoder/internal/vulkan_video_encoder_ext.h new file mode 100644 index 00000000..7f55a7ff --- /dev/null +++ b/vk_video_encoder/internal/vulkan_video_encoder_ext.h @@ -0,0 +1,3325 @@ +/* + * Copyright 2024-2025 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef _VULKAN_VIDEO_ENCODER_EXT_H_ +#define _VULKAN_VIDEO_ENCODER_EXT_H_ + +#include "vulkan_video_encoder_argv.h" +#include + +//============================================================================= +// API versioning and structure typing +// +// VK_VIDEO_ENCODER_EXT_API_VERSION identifies this release of the interface. +// It is NOT a compatibility counter: nothing raises it on a breaking change, +// and nothing reads it at runtime. Do not branch on it. +// +// Layout compatibility is enforced at BUILD TIME: a public struct that +// changes shape stops the build rather than passing silently. +// +// Skew across the boundary is closed by VENDORING. Take this header and the +// implementation together and build them as ONE unit; rolling either alone +// reintroduces the skew these rules exist to remove. Vendoring is also what +// covers the C++ interface below, where adding, removing or reordering a +// virtual method breaks the ABI exactly as a struct change does. +// +// Every public struct that rides a pNext chain opens with sType/pNext. Two +// structures carry no such prefix and chain nowhere: VkVideoEncoderPlaneLayout, +// an element of a fixed array inside another structure, and +// VkVideoEncoderInputFormatProperties, an element of an array the library +// writes. The rules below govern the chainable ones: +// * The ONLY legal way to extend a struct is a new pNext-chained struct +// with a new VkVideoEncoderStructureType value. Appending a field to an +// existing struct is NOT an option: the sType is unchanged, so a consumer +// built against the older layout still passes the gate and is then read +// with a shifted layout -- silently. A version bump does not save it, +// because nothing checks the version at runtime. +// * sType values are never reused or renumbered. +// * The library REJECTS a struct whose sType it does not recognize +// rather than guessing -- VK_ERROR_INITIALIZATION_FAILED from the +// entry points that return a VkResult, +// VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN from those +// that return a typed status. +// * That rejection applies at EVERY public pNext position, entry and +// result structs alike -- not only to each call's primary struct. An +// unknown sType anywhere in a chain is a typed error, never a skip, +// and a struct for which no extension struct is defined refuses ANY +// chain. Where a call DOES define more than one optional link, the +// walk is flat and order-independent in Vulkan's own manner: each +// link is classified by its sType and not by its position, so a +// known link is accepted wherever in that chain it sits. A structure +// type must appear at most ONCE in a chain. On the chains the library +// WRITES its answer into -- GetCompletionInfo's, and the pStatus and +// support chains of RegisterImageResource and QueryImageSupport -- a +// repeat is refused exactly as an unknown type is; that diagnosis is +// not made on every chain, so naming a type twice is a caller error +// rather than something to rely on being reported. +// * There is deliberately NO size field, and none will be added: a size +// field invites reading fewer bytes than the caller wrote -- silent +// field loss, the exact defect class structure typing exists to +// prevent. Extend by chaining. +// +// Structs self-stamp via default member initializers, so `T x = {};` is +// correctly typed without a constructor call (IPC-deserialization safe). +//============================================================================= + +// The release identifier for this interface, per the note above. There is no +// runtime version query and no version negotiation. +#define VK_VIDEO_ENCODER_EXT_API_VERSION 1 + +// consecutiveBFrames: ask the driver for its preferred count instead of +// naming one. Distinct from 0, which means no B-frames. +#define VK_VIDEO_ENCODER_B_FRAMES_DRIVER_PREFERRED 0xFFFFFFFFu + +// Maximum planes in an imported image descriptor. Four covers every format +// this encoder accepts. +#define VK_VIDEO_ENCODER_MAX_PLANES 4 + + +typedef enum VkVideoEncoderStructureType { + VK_VIDEO_ENCODER_STRUCTURE_TYPE_UNDEFINED = 0, + // Private range base 'VE' << 16 (0x56450000); values are permanent. + VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG = 0x56450001, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME = 0x56450002, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_ENCODE_RESULT = 0x56450003, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_RUNTIME_INFO = 0x56450004, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES = 0x56450005, + // Handle exchange. 0x5645000D .. 0x56450011 remain + // reserved for the rest of the registry surface; never reuse. + VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR = 0x56450006, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_SYNC_DESCRIPTOR = 0x56450007, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS = 0x56450008, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT = 0x56450009, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_SEMAPHORE_DESCRIPTOR = 0x5645000A, + // Chained onto VkVideoEncoderImageSupport::pNext: the renegotiation + // modifier list (see the struct). + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT_DETAILS = 0x5645000B, + // The registration status echo. + VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS = 0x5645000C, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO = 0x56450012, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_DEADLINE_INFO = 0x56450013, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_VALIDATION_INFO = 0x56450014, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_DIAGNOSTIC_INFO = 0x56450015, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR = 0x56450016, + // The context surface. + VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONTEXT_CREATE_INFO = 0x56450017, + VK_VIDEO_ENCODER_STRUCTURE_TYPE_DEVICE_IDENTITY = 0x56450018, + // 0x56450019 .. 0x5645001D are assigned and never reused. + // HDR10 static metadata; see VkVideoEncoderHdrMetadataInfo. Chained onto + // VkVideoEncoderConfig::pNext. + VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO = 0x5645001E, + // How the caller's OWN samples are coded; see + // VkVideoEncoderInputColourInfo. Chained onto VkVideoEncoderConfig::pNext. + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO = 0x5645001F, +} VkVideoEncoderStructureType; + +//============================================================================= +// HDR10 STATIC METADATA -- SMPTE ST 2086 mastering display volume and +// MaxCLL/MaxFALL content light level. +// +// Chain onto VkVideoEncoderConfig::pNext. Absent, the encoder emits no SEI +// and no metadata OBU. +// +// WHERE IT LANDS. +// H.265 a PREFIX SEI NAL carrying mastering_display_colour_volume +// (payloadType 137) and/or content_light_level_info (144), appended +// to the VPS/SPS/PPS the driver writes, so it repeats at every IDR. +// AV1 METADATA_TYPE_HDR_MDCV (2) and/or METADATA_TYPE_HDR_CLL (1) +// metadata OBUs, appended to the sequence header OBU. +// H.264 NOTHING, and that is a refusal rather than an omission: H.264 has +// no standard mastering-display or content-light SEI. An H.264 +// session that chains this struct is REJECTED at InitializeExt. +// +// UNITS ARE SMPTE ST 2086's, i.e. exactly what the H.265 SEI codes and what a +// producer already holds (Chromium's gfx::HdrMetadataSmpteSt2086, ffmpeg's +// AVMasteringDisplayMetadata): chromaticity in increments of 0.00002 +// (coordinate x 50000), luminance in increments of 0.0001 cd/m^2 (nits x +// 10000). AV1 codes the same quantities in DIFFERENT fixed-point formats and +// in a DIFFERENT primary order; the library converts. The caller supplies one +// spelling and never sees the other. +// +// A DISPLAY OF ALL ZEROS IS A CLAIM, not an absence -- it says the mastering +// display is black -- so the two payloads have explicit presence flags and +// are emitted independently. A caller that knows MaxCLL and nothing about the +// mastering display sets contentLightLevelPresent alone. +// +// DO NOT LEAVE THIS CHAINED ON A CONFIG YOU PASS TO Reconfigure(), which +// refuses ANY pNext. This metadata is written into the parameter sets at the +// first IDR, alongside the colour fields Reconfigure holds immutable, so +// changing the colour volume mid-stream needs a session re-init. A caller +// that builds one VkVideoEncoderConfig and hands it to both entry points must +// clear pNext for the Reconfigure call. +struct VkVideoEncoderHdrMetadataInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO; + const void* pNext = nullptr; + + // ---- mastering_display_colour_volume / metadata_hdr_mdcv ---- + VkBool32 masteringDisplayPresent = VK_FALSE; + // ST 2086 order: index 0 GREEN, 1 BLUE, 2 RED. Coordinate x 50000. + uint16_t displayPrimaryX[3] = {0, 0, 0}; + uint16_t displayPrimaryY[3] = {0, 0, 0}; + uint16_t whitePointX = 0; + uint16_t whitePointY = 0; + // cd/m^2 x 10000. E.g. 1000 nits is 10000000; 0.0001 nits is 1. + uint32_t maxDisplayMasteringLuminance = 0; + uint32_t minDisplayMasteringLuminance = 0; + + // ---- content_light_level_info / metadata_hdr_cll ---- + VkBool32 contentLightLevelPresent = VK_FALSE; + uint16_t maxContentLightLevel = 0; // MaxCLL, cd/m^2 + uint16_t maxFrameAverageLightLevel = 0; // MaxFALL, cd/m^2 +}; + +//============================================================================= +// THE INPUT'S OWN COLOUR CODING. + +// The input's quantisation range. UNDECLARED IS NOT LIMITED: see the per-lane, +// per-codec policy table below, and note that on the DIRECT lane the library +// applies nothing, so this declaration is a FACT about the caller's samples +// and not a request. +typedef enum VkVideoEncoderRangeDeclaration { + VK_VIDEO_ENCODER_RANGE_UNDECLARED = 0, + VK_VIDEO_ENCODER_RANGE_LIMITED = 1, // studio swing / narrow + VK_VIDEO_ENCODER_RANGE_FULL = 2, +} VkVideoEncoderRangeDeclaration; + +// How the caller's own samples are coded. Chain onto VkVideoEncoderConfig::pNext. +// +// THE PRINCIPLE: the caller declares what its buffer IS; the library declares +// what it DID. The colour fields on VkVideoEncoderConfig state what the +// BITSTREAM should advertise; these state what the INPUT carries. They are +// different questions and, apart from the transfer axis, they were one field. +// +// AN ABSENT CHAIN IS "UNDECLARED", ON EVERY AXIS AT ONCE, and produces exactly +// the previous behaviour bit for bit. There is no sentinel to discover and no +// value pattern to reverse-engineer from a binder gate. A caller that says +// nothing is no worse off than it was, and it becomes able to say so on +// purpose. +// +// 0 IS UNDECLARED ON EACH AXIS, matching VkVideoEncoderConfig's own colour +// fields, so a partly-filled chain is truthful field by field. +// +// WHAT THE LIBRARY DOES WITH IT. The RGBA->Y'CbCr preprocess filter derives its +// conversion matrix from inputColourPrimaries when this chain declares them. +// When it does not, the derivation falls back to +// VkVideoEncoderConfig::colourPrimaries -- an OUTPUT field -- and that fallback +// is sound ONLY because this library implements NO primaries conversion, so an +// input's primaries and its bitstream's primaries are necessarily the same. It +// is stated here rather than left implicit precisely because it stops being +// sound the moment a primaries conversion exists. +// +// WHEN BOTH ARE DECLARED AND THEY DISAGREE on an axis this library cannot +// convert, initialization is REFUSED with the reason -- the same rule +// VkVideoEncoderConfig::inputTransferCharacteristics already applies to the +// transfer axis. One mechanism, three more axes. +// +// THE RANGE AXIS HAS THE OTHER RULE, AND IT IS THE ONE WORTH READING. The +// three axes above are COMPARED on both lanes and applied on neither. The +// range is APPLIED on the Y'CbCr lane and COMPARED on the RGB lane, because +// that is where the library's own behaviour divides: +// +// * Y'CbCr in. Nothing scales the samples on the way to the encoder, so the +// range they carry IS the range the bitstream codes. A declared +// inputRange therefore WRITES video_full_range_flag and raises +// video_signal_type_present_flag with it -- see the policy table below +// for why an unsignalled range is not a safe silence. It does not have to +// agree with videoFullRange; it decides, and it is refused only against +// videoFullRange == VK_TRUE, which is the one direction that is a real +// contradiction (VK_FALSE is a VkBool32's zero and cannot be told from +// silence). +// +// * RGB in. The RGBA->Y'CbCr filter PRODUCES the output range, from +// videoFullRange. The caller's declaration is about its RGB buffer, which +// is a different quantity, so it does not retarget the output. The filter +// samples its RGB over the full range and performs no input expansion, so +// a declared LIMITED describes a conversion that does not happen and is +// REFUSED with the reason, exactly as a mismatched matrix is; a declared +// FULL agrees with what the filter reads and constrains nothing else. +// +// So a declared input range is recorded AND USED, and this sentence is what +// the taxonomy test's range case asserts rather than what it assumes. +// +// THE PER-LANE, PER-CODEC POLICY FOR "UNDECLARED". This is what no document +// said, and the AV1 row is the one that must not be folded into the others. +// +// H.264 / H.265, DIRECT lane (Y'CbCr in, Y'CbCr out), primaries / transfer / +// matrix undeclared: the library writes NOTHING and leaves +// video_signal_type_present_flag at 0. Nothing was applied, so it has +// nothing to say, and a decoder infers Unspecified on all three. +// +// H.264 / H.265, DIRECT lane, RANGE undeclared: the library writes nothing -- +// AND BOTH HALVES OF WHAT THAT MEANS NEED SAYING. Normatively, +// video_full_range_flag is nested inside video_signal_type_present_flag +// and, absent, "shall be inferred to be equal to 0" (H.264 E.2.1, H.265 +// E.3.1) -- studio swing. OBSERVABLY, decoders do not apply that inference +// at their API boundary: ffprobe reports color_range=unknown for an absent +// description and tv for an explicit 0, and hands UNSPECIFIED downstream +// for a scaler or a compositor to guess. A DECLARE-NOTHING CALLER IS +// THEREFORE IMPLEMENTATION-DEPENDENT, NOT SAFELY LIMITED. A caller that +// cares must declare. +// +// H.264 / H.265, DIRECT lane, RANGE DECLARED: it is written, and it is +// SIGNALLED. video_signal_type_present_flag goes up with video_format 5 +// (Unspecified) beside it, so a declaration survives as something a +// decoder reads back rather than as something a caller has to hope was +// inferred. This is the one axis on which a declaration about the INPUT +// changes the bitstream, and it does so because on this lane the input's +// range is the bitstream's range -- there is no conversion between them +// to make the two questions different. +// +// H.264 / H.265, FILTER lane (RGB in), matrix and range undeclared: BT.709 +// and limited are APPLIED and NOT SIGNALLED. The conversion did happen and +// the library knows what it did; whether it should also say so is a +// separate decision and is not taken here. videoFullRange is what moves +// the applied range on this lane, and it moves the signalled one with it, +// because they are one variable. +// +// H.264 / H.265, FILTER lane, RANGE DECLARED: a declaration of FULL is +// accepted and changes nothing -- it agrees with the full-range RGB the +// filter reads -- and a declaration of LIMITED is REFUSED, because no +// input expansion exists to honour it. What the STREAM carries stays +// videoFullRange's to say on this lane. +// +// H.264 / H.265, FILTER lane, primaries and transfer undeclared: Unspecified +// (2). Nothing converted them, and 2 is the honest code point. +// +// AV1, ANY lane, RANGE: THERE IS NO ABSENT STATE. color_range is +// unconditional AV1 syntax (AV1 5.5.2 color_config()): a mandatory f(1) on +// the mono-chrome and general paths, and assigned 1 on the sRGB/identity +// path. So SOME value is in every AV1 bitstream whether this library writes +// it or a driver does, and the library COMMITS TO ONE ON EVERY AV1 STREAM: +// 0 (studio) when nothing was declared. That is a per-codec asymmetry, not +// a per-lane one, and it is stated here so no caller has to rediscover it. +// A declaration is what makes the committed value the caller's rather than +// the default; there is no "signalled or not" half of the question to ask +// on this codec, only which bit. +// +// AV1, primaries / transfer / matrix undeclared: color_description_present_flag +// is 0 and the OBU omits all three. Symmetrical with H.26x. +// +// WHAT A CONSUMER DOES WITH THIS. A producer that knows its buffer -- a +// compositor handing over BT.2020 RGBA, a camera pipeline handing over full- +// range BT.601 Y'CbCr -- states it here and states the bitstream it wants on +// the config, and the library either honours the pair or refuses it. A producer +// that does not know says nothing and gets the behaviour it had. Neither has to +// model the library's routing to decide which fields to set, which is the +// property that lets a consumer delete its copy of the route taxonomy. +// +// DO NOT LEAVE THIS CHAINED ON A CONFIG YOU PASS TO Reconfigure(), which +// refuses ANY pNext. Every colour axis is already immutable for the life of the +// session, so this chain inherits the right lifetime; a caller that hands one +// VkVideoEncoderConfig to both entry points must clear pNext for the +// Reconfigure call. +struct VkVideoEncoderInputColourInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO; + const void* pNext = nullptr; + + // ISO/IEC 23091-2 / 23091-4 code points, as the config's output-side + // fields use. 0 on any axis is UNDECLARED and asserts nothing. + uint8_t inputColourPrimaries = 0; + // Mirrors VkVideoEncoderConfig::inputTransferCharacteristics and is + // subject to the identical rule: this library applies no transfer + // function, so a non-zero value that differs from + // transferCharacteristics is refused. Supplying both is allowed and they + // must agree. + uint8_t inputTransferCharacteristics = 0; + uint8_t inputMatrixCoefficients = 0; + uint8_t reserved = 0; // must be 0 + // The quantisation range the caller's own samples carry. On a Y'CbCr + // input this DECIDES the bitstream's range, because the library scales + // nothing between them; on an RGB input it is checked against what the + // filter can read and the bitstream's range stays + // VkVideoEncoderConfig::videoFullRange's to state. See the range + // paragraph above for both halves and for the one contradiction that is + // refused. + VkVideoEncoderRangeDeclaration inputRange = VK_VIDEO_ENCODER_RANGE_UNDECLARED; +}; + +// Optional extension of VkVideoEncoderConfig: chain this onto the config's +// pNext to override the per-frame completion deadline. Absent, the default +// (8 s) applies. +struct VkVideoEncoderFrameDeadlineInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_DEADLINE_INFO; + const void* pNext = nullptr; + + // Nanoseconds; 0 selects the default (8 s). A non-zero value below 6 s is + // raised to 6 s: any lower and a slow-but-successful frame would be + // declared timed out and its real capture discarded as late, converting a + // slow frame into a lost one. + // A frame past its deadline is delivered by the Acquire methods as a + // 0-byte drop frame with status VK_TIMEOUT (it still requires + // ReleaseEncodedFrame); the session continues. + uint64_t frameCompletionTimeoutNs = 0; +}; + +// Opt-in validations performed once, at InitializeExt (see +// VkVideoEncoderValidationInfo below). +typedef enum VkVideoEncoderValidationFlagBits { + // Hard-fail InitializeExt (VK_ERROR_EXTENSION_NOT_PRESENT) when the + // session's device lacks an extension the external-input import surface + // uses, NAMING each missing one in the log, instead of passing creation + // and failing later at import. + // + // SCOPE DIFFERS BY DEVICE OWNERSHIP. On the library-created device this + // validates ENABLEMENT. On an IMPORTED device Vulkan offers no way to read + // the enabled-extension set, so it validates PHYSICAL-DEVICE SUPPORT only + // and enablement remains the device creator's to audit. + VK_VIDEO_ENCODER_VALIDATE_EXTENSIONS_BIT = 0x00000001, +} VkVideoEncoderValidationFlagBits; +typedef uint32_t VkVideoEncoderValidationFlags; + +// Optional extension of VkVideoEncoderConfig: chain onto the config's +// pNext. Unknown flag bits are rejected at InitializeExt +// (VK_ERROR_INITIALIZATION_FAILED): a caller asking for a validation this +// build does not know is not getting it, and must hear so. +struct VkVideoEncoderValidationInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_VALIDATION_INFO; + const void* pNext = nullptr; + + VkVideoEncoderValidationFlags flags = 0; +}; + +//============================================================================= +// Explicit input-image residency (queue-family ownership). +// +// FOREIGN names an image owned by VK_QUEUE_FAMILY_FOREIGN_EXT -- a dma-buf or +// pixmap import -- which the encoder must acquire from that family before it +// may read the pixels. LOCAL names one allocated on the encoder's own device, +// where such an acquire is unnecessary and, on host-written content, illegal +// (VUID-VkImageMemoryBarrier2-srcStageMask-03854). Getting it wrong is not a +// performance nicety in either direction; see +// VkVideoEncoderExternalImageDescriptor::residency. +// +// THE DISCRIMINATOR IS THE ALLOCATOR, NOT THE WRITER. "Host-written" does +// not imply LOCAL: the Wayland zero-copy lane host-writes its pixels through +// a mapping and is correctly FOREIGN, because GBM/DRM allocated the buffer. +// Conversely an OS-handle import can be correctly LOCAL -- a self-import, +// where the image was allocated on the encoder's own device, exported and +// re-imported into it. Ask who allocated the memory and on whose queue family +// it sits, never who last wrote its contents. +// +// THE LAYOUT QUALIFIES THE DECLARATION. A frame presented in +// VK_IMAGE_LAYOUT_PREINITIALIZED is treated as local whatever residency it +// declared, because that layout names host-written content, which cannot be +// combined with a queue-family transfer. +// +// AUTO INFERS FROM THE LAYOUT: FOREIGN iff currentLayout != +// VK_IMAGE_LAYOUT_PREINITIALIZED. That inference is correct only for +// FIRST-USE local staging images. A REUSED CPU-staging image is left in +// TRANSFER_SRC_OPTIMAL, which AUTO misclassifies as a foreign import. A +// CALLER POOLING OR REUSING INPUT IMAGES MUST STATE THE RESIDENCY +// EXPLICITLY, and that holds on every handle type. On an OS handle AUTO never +// reaches the layout inference at all: it is derived as FOREIGN. +//============================================================================= +enum VkVideoEncoderInputResidency { + VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO = 0, // inferred; see above + VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL = 1, // same-device allocation, no QFOT + VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN = 2, // FOREIGN_EXT-owned import, QFOT acquire +}; + +//============================================================================= +// Handle exchange +// +// The library imports external memory itself. A consumer describes what it +// exported and hands over an OS handle; it does not create a VkImage, and it +// does not need to know what the encoder's device is. +// +// HANDLE OWNERSHIP -- one unconditional rule: +// +// * The library CONSUMES a POSIX fd it is given. Always, on every exit +// path, success or failure. After any call that takes an fd, the caller +// must not close it, use it, or look at it; it is closed exactly once, +// and not by the caller. +// +// * The library NEVER closes a Win32 handle. Not on success, not on +// failure. The caller retains it and must CloseHandle it once the +// resource is unregistered. +// +// The asymmetry is Vulkan's: OPAQUE_FD and DMA_BUF transfer ownership to the +// implementation on import, NT handles do not. +// +// That rule is the TRANSFER mode -- the default, and what {} gives you. The +// mode is chosen per REGISTRATION, never per call outcome; see +// VkVideoEncoderHandleOwnership below. Both modes answer the same on every +// exit path, and the library REPORTS which rule it applied through +// VkVideoEncoderStatus::handlesConsumed (defined below), so a caller can +// assert rather than guess. +//============================================================================= + +// Ownership mode, set per REGISTRATION -- not per frame, and never per +// outcome: +// +// TRANSFER: the rule above, verbatim. After the call the handle is not +// the caller's, on every exit path, success or failure. This is the +// default, and what {} gives you. +// +// BORROW: the handle stays the caller's, on every exit path, success or +// failure. The library duplicates it before anything that can fail and +// applies the TRANSFER rule to its own copy. The caller closes its +// original whenever it likes. +// +// BORROW exists for a caller that must import one received handle twice -- a +// single frame fd into both a display device and the encoder, say. It is +// meaningless across an IPC boundary, where the transport already duplicated +// the handle. On Win32 both modes are identical (the library never closes an +// NT handle) and BORROW is accepted as a no-op. +typedef enum VkVideoEncoderHandleOwnership { + VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_TRANSFER = 0, + VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_BORROW = 1, +} VkVideoEncoderHandleOwnership; + +typedef enum VkVideoEncoderExternalHandleType { + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_NONE = 0, + // POSIX: ownership transfers to the library (see above). + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD = 1, + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF = 2, + // Win32: caller retains. + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_WIN32 = 3, + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_D3D11_TEXTURE = 4, + // An already-imported image on the ENCODER's own device. The + // same-device path; no import is performed and no handle is consumed. + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE = 5, +} VkVideoEncoderExternalHandleType; + +// Typed status: a refusal a caller can act on. A bare VkResult cannot tell +// "this modifier is not encodable" from "wrong GPU" from "that format is +// unsupported", and a client has to know whether to renegotiate the +// allocation or fail the stream. +typedef enum VkVideoEncoderStatusCode { + VK_VIDEO_ENCODER_STATUS_SUCCESS = 0, + VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + // Reserved; nothing returns it. Not removed, because the values are + // positional and deleting one renumbers every code below it. + VK_VIDEO_ENCODER_STATUS_ERROR_API_VERSION_UNSUPPORTED, + VK_VIDEO_ENCODER_STATUS_ERROR_DEVICE_MISMATCH, + VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED, + VK_VIDEO_ENCODER_STATUS_ERROR_EXTENSION_MISSING, + VK_VIDEO_ENCODER_STATUS_ERROR_FORMAT_UNSUPPORTED, + VK_VIDEO_ENCODER_STATUS_ERROR_MODIFIER_UNSUPPORTED, + VK_VIDEO_ENCODER_STATUS_ERROR_USAGE_INSUFFICIENT, + VK_VIDEO_ENCODER_STATUS_ERROR_PLANE_LAYOUT_INVALID, + VK_VIDEO_ENCODER_STATUS_ERROR_ALLOCATION_SIZE_INVALID, + VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED, + VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED, + VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED, + VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT, + VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN, + VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED, + // New codes are APPENDED here, never inserted: every value above and + // below is ABI for a consumer built against an older header. + VK_VIDEO_ENCODER_STATUS_ERROR_SHARING_MODE_UNSUPPORTED, + VK_VIDEO_ENCODER_STATUS_ERROR_EXTENT_INVALID, + // Flow control, not an error -- VkResult's VK_NOT_READY as a status: + // the submit path is at capacity, so drain completions and retry the + // same call. Distinct from ERROR_RESOURCE_LIMIT above, which retrying + // alone never clears. + // + // ERROR_RESOURCE_LIMIT has TWO sources. (1) A registry's 4096-slot image + // or semaphore table is full; an Unregister clears it. (2) A DIRECT + // submit was handed more waits than its eight-entry array can carry; + // presenting fewer waits clears that, and unregistering does nothing for + // it. Read the code as "this cannot succeed until the request or the + // registry changes" -- it does not by itself say which, and an unchanged + // retry succeeds in neither case. + VK_VIDEO_ENCODER_STATUS_NOT_READY, + // The colorModel DECLARATION cannot be read against the format it was + // declared over: RGB named over a Y'CbCr format, Y'CbCr named over an + // RGBA layout that carries no packed 4:4:4 reading, or a value the + // enumeration does not define. + // + // The offending field is colorModel, not format. NV12 declared RGB is + // refused under this code while NV12 itself is directly encodable, so a + // caller acting on the code alone corrects the declaration -- which is + // the only thing that can be corrected. Changing the format instead + // would answer a question that was never asked. + VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED, +} VkVideoEncoderStatusCode; + +// Library-minted resource id, safe to carry across an IPC boundary. A stale +// id -- one minted before an unregister -- is REJECTED rather than silently +// matching a recycled slot. +typedef uint64_t VkVideoEncoderResource; +#define VK_VIDEO_ENCODER_RESOURCE_NULL ((VkVideoEncoderResource)0) + +typedef struct VkVideoEncoderPlaneLayout { + uint64_t offset; + // NOT READ. The library writes 0 into the Vulkan structure itself, which + // is what VUID-VkImageDrmFormatModifierExplicitCreateInfoEXT-size-02267 + // requires on the explicit path. The other four members of this structure + // are taken as declared. + uint64_t size; + uint64_t rowPitch; + uint64_t arrayPitch; // 0 when arrayLayers == 1 + uint64_t depthPitch; // 0 for 2D +} VkVideoEncoderPlaneLayout; + +// The colour model a caller's samples are in. +// +// A Vulkan format names a COMPONENT LAYOUT, and for one family of inputs the +// layout does not determine the colour model. The packed 4:4:4 Y'CbCr layouts +// have no Vulkan enumerant of their own and ride the matching RGBA ones -- +// AYUV on VK_FORMAT_R8G8B8A8_UNORM, Y410 on VK_FORMAT_A2B10G10R10_UNORM_PACK32 +// -- which are also the enumerants an ordinary RGBA producer declares. The +// format alone therefore cannot say whether a texel holds red, green and blue +// or luma and two chroma, and the difference decides whether a colour +// conversion runs. +// +// Declaring the model is how a caller settles that. It is a fact about the +// caller's own pixels, and nothing the library can measure recovers it. +typedef enum VkVideoEncoderColorModel { + // Read the colour model off the format. Every Vulkan format except the + // packed 4:4:4 Y'CbCr layouts above names exactly one, so this is the + // right answer for all of them -- and it is what a zero-initialised + // structure says, so a caller that has no packed input never sets this + // field. + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT = 0, + // Red, green and blue, with the transfer function already applied. The + // library converts to the encoder's Y'CbCr input using the matrix + // VkVideoEncoderConfig::matrixCoefficients names. + VK_VIDEO_ENCODER_COLOR_MODEL_RGB = 1, + // Luma and two chroma. Over a packed 4:4:4 layout this is the + // declaration that makes AYUV and Y410 nameable; over any other Y'CbCr + // format it states what the format already says and changes nothing. + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR = 2, +} VkVideoEncoderColorModel; + +// THE IPC PAYLOAD. A producer in another process fills this by field copy and +// the OS handle rides out of band (SCM_RIGHTS on POSIX, a duplicated handle on +// Windows), so the member offsets below are the wire format rather than a +// property of one build. It is a superset of what a producer already holds +// about an image it exported. +// +// Two members are not part of that payload, and are the only two that are not +// plain data: pNext, which must be NULL and is refused otherwise, and +// existingImage, which is read only for handleType == VK_IMAGE -- a +// same-process arm, never an IPC one. Everything else is scalar, enum, flags, +// or a fixed array of those. +typedef struct VkVideoEncoderExternalImageDescriptor { + VkVideoEncoderStructureType sType; + const void* pNext; // MUST be NULL; a chain is refused + + VkVideoEncoderExternalHandleType handleType; + + // ---- The exporter's ACTUAL VkImageCreateInfo. Never guessed. ---- + VkFormat format; + uint32_t width; + uint32_t height; + VkImageType imageType; // 0 => VK_IMAGE_TYPE_2D + uint32_t mipLevels; // 0 => 1 + uint32_t arrayLayers; // 0 => 1 + VkSampleCountFlagBits samples; // 0 => VK_SAMPLE_COUNT_1_BIT + VkImageTiling tiling; // OPTIMAL | LINEAR | DRM_FORMAT_MODIFIER + // 0 is INVALID on the OS-handle arms, not a default: the library never + // invents a usage the exporter did not grant. The VK_IMAGE arm alone may + // leave it 0, and is then accepted as transfer-source only -- see + // RegisterImageResource. + VkImageUsageFlags imageUsage; + VkImageCreateFlags imageFlags; + VkSharingMode sharingMode; + // NOT mirrored: the exporter's create-time VkVideoProfileListInfoKHR. A + // profile-DEPENDENT exporter therefore cannot be registered faithfully. + // + // WHICH MAKES VIDEO-PROFILE COMPATIBILITY A CALLER OBLIGATION ON THE + // DIRECTLY ENCODABLE ARM, and one this library cannot discharge for it. + // A registration whose tiling is not LINEAR, whose format the library + // encodes without conversion, and which grants + // VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR is encoded FROM THE CALLER'S + // OWN IMAGE, with no staging copy in between. It must therefore ALREADY + // be compatible with the video profile the session negotiates + // (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-08206) -- created either with a + // VkVideoProfileListInfoKHR naming that profile in its VkImageCreateInfo + // pNext chain, or with VK_IMAGE_CREATE_VIDEO_PROFILE_INDEPENDENT_BIT_KHR. + // Registration is handed an image that already exists, so nothing done + // here can repair one created with neither. + // + // PROFILE INDEPENDENCE IS NOT A BLANKET EXEMPTION. A profile-independent + // image is compatible with a profile only while EVERY usage it declares + // is one vkGetPhysicalDeviceVideoFormatPropertiesKHR reports for that + // profile and format. VK_IMAGE_USAGE_STORAGE_BIT is commonly absent from + // that set for encode-source formats, so an image carrying STORAGE may be + // registered for the compute filter -- which reads it through a storage + // descriptor, and for which STORAGE is required -- but must not ALSO + // declare VIDEO_ENCODE_SRC and ask to be encoded directly. A producer + // that wants both should present the image without VIDEO_ENCODE_SRC and + // let the registration route through the staging copy. + + // ---- DRM format modifier ---- + // Separate presence flag because modifier 0 is REAL -- it is + // DRM_FORMAT_MOD_LINEAR. A single uint64 cannot distinguish "no modifier, + // import as OPTIMAL" from "the modifier is literally zero". + VkBool32 hasDrmFormatModifier; + uint64_t drmFormatModifier; + + uint32_t planeCount; + // All planes must live in the SINGLE memory object named by the handle + // passed to RegisterImageResource. A disjoint (multi-buffer-object) image + // cannot be expressed here. + // + // THIS IS A CALLER OBLIGATION, and it cannot be delegated. Registration + // is handed ONE handle, so the library cannot see where the other planes + // came from; nothing in this descriptor carries their provenance. + // RegisterImageResource rejects the shape that is provably impossible in + // one allocation -- two planes at the same offset, which is what a + // disjoint producer writes when every plane starts at the base of its own + // buffer -- but a disjoint producer whose chroma plane sits at a non-zero + // offset inside its OWN buffer passes that check and yields silently + // wrong chroma. Do not treat the offset rule as a disjointness test. + // + // A caller that holds one handle per plane (gfx::NativePixmapHandle, + // gbm_bo_get_fd_for_plane, ...) MUST compare their identity itself -- + // fstat st_dev/st_ino on POSIX -- and only then present the buffer here. + // On NVIDIA/gbm the per-plane exports of one BO return the same cached + // dma_buf, so equal (st_dev, st_ino) is positive proof of jointness, and + // distinct BOs never collide. A caller that cannot run that test should + // stage a copy instead of registering. + VkVideoEncoderPlaneLayout planeLayouts[VK_VIDEO_ENCODER_MAX_PLANES]; + + // ---- Memory, as reported by the EXPORTER ---- + // Deriving allocationSize from vkGetImageMemoryRequirements is wrong for + // dma-buf on NVIDIA. 0 means unknown and is accepted with a log. + uint64_t allocationSize; + uint32_t memoryTypeBits; // 0 = unknown (a real mask always has a + // bit set: the exporter's own allocation + // occupies a type) + uint32_t memoryTypeIndex; // UINT32_MAX = unknown + // memoryTypeIndex is a PREFERENCE on the dma-buf arm and a REQUIREMENT + // on the opaque arms: + // - DMA_BUF: the library asks vkGetMemoryFdPropertiesKHR which types + // can import THIS fd and honours memoryTypeIndex only inside that + // mask intersected with the image's requirements + // (VUID-VkMemoryAllocateInfo-memoryTypeIndex-00648). If the query + // is unavailable, the legacy preference order runs as a logged + // fallback, never silently. + // - OPAQUE_FD / OPAQUE_WIN32: querying is forbidden and the import + // must reuse the exporter's allocation parameters + // (VUID-VkMemoryAllocateInfo-allocationSize-01742 / -01743), so + // supply memoryTypeIndex; on OPAQUE_FD, memoryTypeBits additionally + // lets the library choose safely when the exact index is not + // available. An OPAQUE_WIN32 import without it is refused: on WDDM + // a guessed type is a hard error, not a fallback. + + // ---- Provenance: multi-GPU safety ---- + // A supplied deviceUUID or deviceLUID that does not match the encode + // device fails DEVICE_MISMATCH at registration, rather than failing the + // import later with an unhelpful driver error. driverUUID binds on the + // OPAQUE handle types, where the external-memory compatibility table + // requires driver identity; a dma-buf is a kernel object and a + // cross-driver import of one is legal, so a driverUUID difference there + // is reported rather than refused. + uint8_t deviceUUID[VK_UUID_SIZE]; + uint8_t driverUUID[VK_UUID_SIZE]; + uint8_t deviceLUID[VK_LUID_SIZE]; + VkBool32 deviceLUIDValid; + + // THE LAYOUT THE LIBRARY WILL FIND THIS IMAGE IN ON THE FIRST FRAME. + // + // A STATEMENT OF FACT about the image, not a preference: it is named as + // the oldLayout of the FIRST acquire barrier, so a false declaration + // violates VUID-VkImageMemoryBarrier2-oldLayout-01197. + // + // ONE VALUE IS NOT NAMED BACK. VK_IMAGE_LAYOUT_UNDEFINED declared here + // is read as "the producer stated nothing", and the library substitutes + // a layout of its own rather than name UNDEFINED in a barrier -- an + // UNDEFINED oldLayout permits the implementation to discard the very + // pixels the producer just wrote. WHICH layout it substitutes is NOT part + // of this contract and differs by path. A caller that needs a known + // layout must declare the layout the image is really in rather than rely + // on the substitution. Every other layout is carried through unchanged. + // + // WHERE THE LIBRARY LEAVES THE IMAGE BETWEEN FRAMES is likewise NOT part + // of this contract. It differs by path, and is constrained by what a + // barrier may name as a destination + // (VUID-VkImageMemoryBarrier2-newLayout-01198), which excludes + // VK_IMAGE_LAYOUT_UNDEFINED and VK_IMAGE_LAYOUT_PREINITIALIZED. A caller + // that needs the image in a particular layout between frames must put it + // there itself and state it per frame, on + // VkVideoEncoderFrameSubmitInfo::currentLayout, rather than infer it from + // what was declared here. + // + // TO SEE WHAT THE LIBRARY ACTUALLY NAMED, set VKENC_DEBUG_LAYOUT in the + // environment and the library traces the layouts it puts in its + // barriers. DIAGNOSTIC ONLY, and deliberately so: it is the only way to + // observe the two answers this contract has just declined to specify -- + // which layout is substituted for UNDEFINED, and where the image is + // left between frames -- and observing them does not turn either into a + // promise. Do not branch on what it reports. It writes to stderr + // directly, so unlike the library's other diagnostics it is NOT + // suppressed by silenceStdio. + // + // WHEN THIS FIELD STOPS BEING READ. On the STAGED path the library moved + // the image itself, so from the second frame on it names its own record + // instead: a declaration that was true once does not have to be kept true + // across frames the caller never touched the image on. TO OVERRIDE THAT + // RECORD ON A GIVEN FRAME -- because you moved the image yourself between + // submits -- set VkVideoEncoderFrameSubmitInfo::currentLayout for that + // frame. A non-UNDEFINED value there is an explicit per-frame statement + // of fact and beats the record; UNDEFINED there resolves to the record + // where one was kept, and to this field where it was not. + // + // THREE CASES WHERE THIS FIELD IS RE-READ EVERY FRAME, because no record + // exists or it is not authoritative: + // * a DIRECTLY ENCODABLE registration, which is never staged; + // * an image the library RELEASED to VK_QUEUE_FAMILY_FOREIGN_EXT, since a + // foreign agent held it between frames and may have transitioned it, + // and the release CLEARS the record rather than leave a stale one; + // * the SubmitExternalFrame lane, which has no registration at all and + // takes its layout from the frame. + // + // KNOWN GAP, FIRST FRAME. The library does not transition the image at + // registration, so this declaration must ALREADY be true when the first + // frame is submitted, and only the caller can make it so. An image with + // external memory can be created only UNDEFINED + // (VUID-VkImageCreateInfo-pNext-01443), so on the OS-handle tiers the + // image the library imports genuinely IS in UNDEFINED on frame 1, + // whatever is declared here -- and declaring UNDEFINED does not carry + // that fact through, because UNDEFINED is the one value not named back. + // No declaration therefore makes frame 1 truthful on those tiers; the + // caller's remaining lever is to transition the image into the declared + // layout itself before the first submit. + VkImageLayout defaultLayout; + + + // HONOURED ON EVERY HANDLE TYPE. An explicit LOCAL or FOREIGN is a + // statement of fact by the only party that can make it, and the library + // takes it as given wherever it is made -- with one layout-driven + // qualification, and only on the STAGED path. There, a frame presented in + // VK_IMAGE_LAYOUT_PREINITIALIZED is treated as local even under a FOREIGN + // declaration, because PREINITIALIZED names host-written content, which + // is not a legal pairing with a queue-family ownership transfer + // (VUID-VkImageMemoryBarrier2-srcStageMask-03854). The DIRECT path reads + // this field alone, with no layout term. An explicit LOCAL carries no + // qualification on either path. + // + // IT IS NOT DERIVED FROM THE HANDLE TYPE. That the library performed the + // import does not make the memory foreign to the encode device: a + // SELF-IMPORT -- an image allocated on THIS device, exported, and + // re-imported into it, which is what an embedder does when it wants the + // library to own the staging allocation -- has no second device and no + // second queue family. The discriminator is OWNERSHIP BY AN EXTERNAL + // ALLOCATOR OR QUEUE FAMILY, never who wrote the pixels; see the enum + // above for the two cases that make the difference. + // + // AUTO (== 0, so also what a zero-initialised descriptor says) IS STILL + // DERIVED AS FOREIGN ON AN OS HANDLE, and only there: a caller that + // propagates no residency across a process boundary has told the library + // nothing, and a genuine import is overwhelmingly the likelier reading. + // On VK_IMAGE, AUTO continues to mean the layout-derived inference. + // + // GETTING IT WRONG IS NOT A PERFORMANCE NICETY, in either direction. + // Declaring FOREIGN for a local image takes a two-sided queue-family + // ownership transfer it does not need: an acquire FROM + // VK_QUEUE_FAMILY_FOREIGN_EXT and, at the last use of the image, a + // matching release BACK to it. The release is the expensive half to get + // wrong -- it gives a local image away to an owner that does not exist + // and discards the library's residual-layout record, so the next frame + // acquires from a layout the image is no longer in. Declaring LOCAL for + // a genuinely foreign one skips an acquire the transfer needs, and the + // copy then reads undefined content. + VkVideoEncoderInputResidency residency; + + // Only for handleType == VK_IMAGE: the caller's already-imported image. + // Ignored for every OS-handle type, where the library performs the import + // and therefore knows the answer without being told. + VkImage existingImage; + + // Ownership mode for the handle passed to RegisterImageResource (see + // VkVideoEncoderHandleOwnership). Zero-init == TRANSFER, deliberately: + // the safe thing is what {} gives you. + VkVideoEncoderHandleOwnership ownership; + + // The colour model the samples in this image are in (see + // VkVideoEncoderColorModel). Zero-init reads it off |format|, which is + // correct for every format but the packed 4:4:4 Y'CbCr layouts. + // + // A declaration the format cannot carry is REFUSED at registration, and + // at QueryImageSupport, with + // VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED -- which names + // this field and not |format|. It is never reconciled to one of the two + // readings: refusing at the negotiation point is what lets the producer + // correct the declaration while the allocation can still change. + VkVideoEncoderColorModel colorModel; +} VkVideoEncoderExternalImageDescriptor; + +// Per-frame submission against a registration. +// +// Everything describing the IMAGE lives in the registration; this carries +// only what genuinely varies per frame. +typedef struct VkVideoEncoderFrameSubmitInfo { + VkVideoEncoderStructureType sType; // ..._FRAME_PARAMS + const void* pNext; + + VkVideoEncoderResource resource; // from RegisterImageResource + + uint64_t frameId; + uint64_t pts; + VkBool32 forceIDR; + VkBool32 isLastFrame; + // -1 = no override. Honoured only in DISABLED (caller-managed) rate + // control; in CBR/VBR the encoder owns QP. + int32_t qpOverride; + // The layout the producer left the image in. VK_IMAGE_LAYOUT_UNDEFINED + // means "as declared at registration". + VkImageLayout currentLayout; + + uint32_t waitSemaphoreCount; + const VkSemaphore* pWaitSemaphores; + const uint64_t* pWaitSemaphoreValues; + uint32_t signalSemaphoreCount; + const VkSemaphore* pSignalSemaphores; + const uint64_t* pSignalSemaphoreValues; +} VkVideoEncoderFrameSubmitInfo; + +// Picture type of an encoded frame (VkVideoEncodeResult::pictureType). +typedef enum VkVideoEncoderPictureType { + VK_VIDEO_ENCODER_PICTURE_TYPE_I = 0, // intra (incl. IDR / intra-refresh) + VK_VIDEO_ENCODER_PICTURE_TYPE_P = 1, + VK_VIDEO_ENCODER_PICTURE_TYPE_B = 2, +} VkVideoEncoderPictureType; + +// State of a submitted frame (GetFrameStatus). The query CANNOT fail. +// "Never submitted" and "already released" share UNKNOWN deliberately: both +// are caller bugs, and both call for the same response. +typedef enum VkVideoEncoderFrameState { + VK_VIDEO_ENCODER_FRAME_STATE_UNKNOWN = 0, // never submitted, or already released + VK_VIDEO_ENCODER_FRAME_STATE_PENDING = 1, // submitted, no capture yet + VK_VIDEO_ENCODER_FRAME_STATE_READY = 2, // an Acquire call will deliver it + VK_VIDEO_ENCODER_FRAME_STATE_ACQUIRED = 3, // delivered, awaiting ReleaseEncodedFrame +} VkVideoEncoderFrameState; + +// Completion callback: invoked the moment a frame's capture becomes +// retrievable (the ONE producer-raised readiness edge). Contract: +// * MUST NOT THROW. Unwinding out of the callback leaves the encoder in a +// state with no route back; where the toolchain makes noexcept part of +// the function type (C++17 and later) the compiler enforces this, +// elsewhere the library catches and aborts, naming the frame. +// * MUST NOT BLOCK -- post to your own executor and return (a consumer +// that does work here stalls the encode pipeline). +// * MUST NOT call session-serial (class a) methods; the library rejects +// such re-entry with VK_ERROR_NOT_PERMITTED_KHR. The thread-safe +// (class c) retrieval methods ARE legal here. +// * MUST NOT destroy the encoder from inside the callback. Detach first +// (SetCompletionCallback(nullptr, nullptr), which is itself a quiesce +// point), then destroy; teardown joins the thread you are running on. +// * Notifications COALESCE: one callback may cover several ready frames. +// Drain in a loop and reconcile against GetCompletionCounter() -- +// assuming one-callback-one-frame silently drops frames under load. +// * Deadline-synthesized VK_TIMEOUT drops do NOT raise the callback +// (there is no capture); they become retrievable by deadline passage. +#if defined(__cplusplus) && (__cplusplus >= 201703L) +#define VK_VIDEO_ENCODER_CB_NOEXCEPT noexcept +#else +#define VK_VIDEO_ENCODER_CB_NOEXCEPT +#endif + +typedef void (*PFN_vkVideoEncoderCompletionCallback)(uint64_t frameId, + void* pUserData) + VK_VIDEO_ENCODER_CB_NOEXCEPT; + +// Invoked EXACTLY ONCE when the encoder stops using a pUserData previously +// given to SetCompletionCallback -- on replacement, on detach, or at +// destruction -- and only after any in-flight completion invocation has +// returned. +// +// Prefer this to keeping the cookie alive by arrangement with the encoder's +// teardown order. A weak reference INSIDE the cookie protects the work the +// callback does; it does nothing for the cookie itself, which the encoder +// must dereference in order to reach it. +// +// Optional: pass nullptr to keep owning the cookie yourself. +typedef void (*PFN_vkVideoEncoderUserDataRelease)(void* pUserData) + VK_VIDEO_ENCODER_CB_NOEXCEPT; + +// Completion observability snapshot (GetCompletionInfo). +// +// Chainable onto pNext: VkVideoEncoderDiagnosticInfo. Unknown or repeated +// sTypes are refused, not ignored. +struct VkVideoEncoderCompletionInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + const void* pNext = nullptr; + + uint64_t completionCounter; // monotonic frame-completion count (drain target) + uint64_t framesTimedOut; // deadline drops delivered + uint64_t lateCaptures; // captures discarded after a timeout drop + uint64_t framesCancelled; // CancelFrame + uint32_t framesPending; // submitted, no capture yet + uint32_t framesReady; // retrievable now + uint32_t framesAcquired; // delivered, awaiting ReleaseEncodedFrame +}; + +// Diagnostic side-channel: chain to VkVideoEncoderCompletionInfo::pNext +// on a GetCompletionInfo() call. +// +// The library's runtime misuse messages otherwise go to stderr, which +// silenceStdio discards -- so under silenceStdio a real API misuse (for +// example ReleaseEncodedFrame on an unknown frame id: a double release, or a +// typo'd id) is a SILENT no-op. Chaining this struct returns the running +// misuse count and the most recent message through the snapshot call the +// consumer already makes, so the text can land in the consumer's OWN logging. +// +// lastDiagnostic is NUL-terminated and empty until the first recorded +// misuse; the count never resets for the life of the encoder object. +#define VK_VIDEO_ENCODER_MAX_DIAGNOSTIC_CHARS 192 +struct VkVideoEncoderDiagnosticInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_DIAGNOSTIC_INFO; + const void* pNext = nullptr; + + uint64_t diagnosticCount; // misuses recorded since creation + char lastDiagnostic[VK_VIDEO_ENCODER_MAX_DIAGNOSTIC_CHARS]; +}; + +//============================================================================= +// External Frame Input with Synchronization +// +// Extends VulkanVideoEncoder for frame-at-a-time operation with +// externally-provided VkImages and timeline semaphore synchronization. +// This is the interface for cross-process encoder services. +// +// Usage flow: +// 1. CreateVulkanVideoEncoderExt() to create the encoder +// 2. InitializeExt() with structured config (not argc/argv) +// 3. For each frame: +// a. RegisterImageResource() once per buffer, then +// SubmitRegisteredFrame() naming that registration per frame. +// SubmitExternalFrame() with a VkImage still works; registration is +// what a cross-process or pooled producer needs. +// b. Wait for the completion callback (SetCompletionCallback), then +// drain with AcquireNextEncodedFrame in a LOOP -- notifications +// coalesce. Polling still works but is not the intended shape. +// 4. Flush() to drain pending frames +//============================================================================= + +//============================================================================= +// Encode profile +// +// A profile is named by THE CODEC STANDARD'S OWN NUMBER: H.264 profile_idc +// (ITU-T H.264 Annex A, Table A-1), H.265 general_profile_idc (ITU-T H.265 +// Annex A), AV1 seq_profile (AV1 6.4.1). VkVideoEncoderConfig::profile is a +// plain uint32_t read against VkVideoEncoderConfig::codec, so the value space +// belongs to the standard and not to this header: a profile the constants +// below do not name is still expressible, and whether this library and this +// device can encode it is answered by +// EnumerateVulkanVideoEncoderProfileCapabilities* and, failing that, by a +// refusal at InitializeExt naming the value. A profile request is never +// accepted and ignored. +// +// The constants below carry the same values as StdVideoH264ProfileIdc / +// StdVideoH265ProfileIdc / StdVideoAV1Profile, spelled out so this header +// puts no dependency onto a consumer. THEY ENUMERATE EVERY +// PROFILE THIS LIBRARY BINDS, and that is a checked property rather than a +// promise: the taxonomy test sweeps the whole value space of each codec's +// profile syntax element through the binder and fails when a number binds +// that no constant here names, and when a constant here names a number the +// binder refuses. A library change that widens the bind set therefore fails +// until this block is widened with it. +// +// NAMING THEM IS WHAT KEEPS THE HEADER SELF-SUFFICIENT. Without a name a +// caller that wants High 4:4:4 Predictive -- which is what the DEFAULT +// derivation itself selects from 4:4:4 input -- must either write the bare +// integer 244 or include , which is +// exactly the dependency this block exists to spare it. +// +// THE VALUES ARE THE STANDARD'S, NOT A LIBRARY NUMBERING, so a profile added +// to a standard is expressible here on the day it is assigned, and the only +// thing a library change adds is the ability to BIND it. +// +// VK_VIDEO_ENCODER_PROFILE_DEFAULT (0) asks the library to derive the profile +// from the input's bit depth and chroma subsampling. It is 0 so that a +// zeroed config selects the derivation on every codec instead of a real +// profile. +// +// ONE OVERLAP, STATED RATHER THAN DISCOVERED. AV1 seq_profile 0 IS Main, and +// is therefore the same value as DEFAULT. On an AV1 session 0 is read as +// "derive", and the derivation reads both halves of the input: 8/10-bit +// 4:2:0 picks Main, which is what seq_profile 0 names, so the two readings +// agree there; 4:4:4 picks High and 12-bit or 4:2:2 picks Professional, +// none of which seq_profile 0 can carry. 0 is not a legal H.264 profile_idc +// and is not an assigned H.265 general_profile_idc, so no other codec +// carries the overlap. +#define VK_VIDEO_ENCODER_PROFILE_DEFAULT 0u + +// H.264 profile_idc. WHAT EACH ONE ADMITS IS THE STANDARD'S RULE, on both +// axes, and a request against input the profile cannot carry is refused at +// InitializeExt rather than emitted out-of-spec. Per ITU-T H.264 Annex A, +// Table A-1: Baseline, Main and High are 8-bit 4:2:0; High 10 reaches ten +// bits at 4:2:0; High 4:2:2 reaches ten bits and 4:2:2; High 4:4:4 +// Predictive reaches fourteen bits and 4:4:4. +enum VkVideoEncoderProfileH264 { + VK_VIDEO_ENCODER_PROFILE_H264_BASELINE = 66, + VK_VIDEO_ENCODER_PROFILE_H264_MAIN = 77, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH = 100, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH_10 = 110, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH_422 = 122, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH_444_PREDICTIVE = 244, +}; + +// H.265 general_profile_idc. Main is 8-bit 4:2:0 (H.265 A.3.2), Main Still +// Picture is Main's single-picture form (A.3.4) and admits the same input, +// Main 10 reaches ten bits at 4:2:0 (A.3.3), and Range Extensions and Screen +// Content Coding Extensions reach 4:2:2, 4:4:4 and sixteen bits (A.3.5, +// A.3.7). Input a named profile cannot carry is refused; DEFAULT derives a +// profile that admits the input on both axes. +enum VkVideoEncoderProfileH265 { + VK_VIDEO_ENCODER_PROFILE_H265_MAIN = 1, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN10 = 2, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN_STILL_PICTURE = 3, + VK_VIDEO_ENCODER_PROFILE_H265_FORMAT_RANGE_EXTENSIONS = 4, + VK_VIDEO_ENCODER_PROFILE_H265_SCC_EXTENSIONS = 9, +}; + +// AV1 seq_profile. Main is 8/10-bit 4:2:0, High is 8/10-bit 4:4:4, and +// Professional is the one that reaches 4:2:2 and twelve bits (AV1 6.4.1, +// A.2). Note the overlap above: Main's value is also +// VK_VIDEO_ENCODER_PROFILE_DEFAULT, so it is the one profile a caller cannot +// name distinctly from the derivation -- which costs nothing, because on +// 4:2:0 input at 8 or 10 bits the derivation selects it. +enum VkVideoEncoderProfileAV1 { + VK_VIDEO_ENCODER_PROFILE_AV1_MAIN = 0, + VK_VIDEO_ENCODER_PROFILE_AV1_HIGH = 1, + VK_VIDEO_ENCODER_PROFILE_AV1_PROFESSIONAL = 2, +}; + +// The capability enumeration answers per (codec, profile) pair, so a profile +// this library binds but this device cannot encode is reported there rather +// than discovered at session creation. +//============================================================================= + +//============================================================================= +// Encoder Configuration (structured, not argc/argv) +//============================================================================= +struct VkVideoEncoderConfig { + VkVideoEncoderStructureType sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + const void* pNext = nullptr; + + // Codec + VkVideoCodecOperationFlagBitsKHR codec; + + // Encode profile within the selected codec: the codec standard's own + // profile number, read against |codec| above. The profile constants above + // name every number this library binds -- checked, not asserted -- and + // state what input each one admits. + // VK_VIDEO_ENCODER_PROFILE_DEFAULT (0) derives the profile from the input. + uint32_t profile = VK_VIDEO_ENCODER_PROFILE_DEFAULT; + + // Encode output resolution + uint32_t encodeWidth; + uint32_t encodeHeight; + + // Input format -- what the external frames will be. + // + // THE LIBRARY DECIDES WHETHER A CONVERSION RUNS, from this field and + // |inputColorModel| below. The caller does not ask for one, and there is + // no flag to set. + // + // WHETHER A FORMAT IS ACCEPTED is decided by the declared pair -- this + // field and |inputColorModel| -- TOGETHER WITH the codec, the profile and + // the device, and InitializeExt settles both halves before it creates a + // session. + // + // * The LIBRARY half is settled with no device involved: a pair this + // library does not route, or one the named profile cannot carry, is + // refused with the reason and is not renegotiable by trying another + // GPU. + // * The DEVICE half is settled against the profile the configuration + // derives: a pair this device will not encode is refused NAMING the + // format, its chroma subsampling and its bit depth, rather than + // surviving to the driver's own refusal at video-session creation -- + // which arrives later and names neither. + // + // VkEncEnumerateInputFormats IS THAT ANSWER, ASKED BEFORE THE CALL. It + // lists the formats this library can route to an encoder input that a + // given device accepts for a given (codec, profile), each flagged OPTIMAL + // or SUBOPTIMAL -- and it is computed by the same function InitializeExt + // gates on. So for a (codec, profile) the advertised set IS the accepted + // set: a format on the list initialises, and a format the list omits is + // refused. There is one surface to consult, not two. + // + // THE ONE AXIS THE LIST CANNOT CARRY IS THE COLOUR MODEL, and it is why + // VkEncQueryInputFormatSupport takes one and the list does not. The packed + // 4:4:4 Y'CbCr layouts AYUV and Y410 have no Vulkan enumerant of their own + // and ride the RGBA ones, so they are accepted only where + // |inputColorModel| declares them; undeclared, the same enumerant is an + // ordinary RGBA image. A list keyed on formats alone can show neither of + // them as itself -- AYUV's enumerant appears there under its RGB reading, + // and Y410's enumerant does not appear at all -- so on THAT axis, and only + // that one, absence from the list is not a refusal. The point query is + // where a caller holding AYUV or Y410 frames gets the answer, and it gives + // the same verdict this call will. + // + // TWO CONSEQUENCES OF THE SAME RULE, spelled out because they decide what + // a caller allocates: + // + // * Y416 (VK_FORMAT_R16G16B16A16_UNORM) is not accepted under any + // declaration: 16 bits per component is not an encode component bit + // depth. It is on no list and is refused under every colour model, + // so the two agree about it as they do about every other format. + // * On a Y'CbCr input the chroma subsampling OF THE FORMAT is what the + // encode profile is derived from, so a 4:4:4 Y'CbCr input encodes as + // 4:4:4 without anything further being asked for. An RGBA input + // carries no subsampling to read and selects no profile this way: + // what its bitstream is coded at is the encodeFormat of its + // advertised entry, per VkVideoEncoderInputFormatProperties. + // + // A format this library does not accept -- under either reading, where + // the enumerant carries two -- is refused at InitializeExt with the + // reason. + // + // A conversion this build or this device cannot perform is an init + // failure with a reason, never a quiet acceptance. THE TWO WAYS IT CAN + // BE UNAVAILABLE are both settled by the caller BEFORE this call, and + // neither is renegotiable after it: + // + // * THE FILTER IS NOT COMPILED INTO THE BUILD. It is a CMake option + // of the unit you vendor -- BUILD_ENCODER_COMPUTE_FILTER -- so the + // consumer that builds this header together with its + // implementation is the party that decides it. With it OFF, every + // format that is encodable only through the filter is refused. + // * THE SESSION'S DEVICE EXPOSES NO COMPUTE QUEUE FAMILY for the + // filter to run on. Under a caller-supplied VkDevice that is the + // caller's own device-creation choice, so InitializeExt verifies it + // rather than assuming it; the only remedy is to create the device + // again with a compute queue, which is a pre-InitializeExt action. + // + // NEITHER IS DISCOVERABLE AT RUNTIME BEFORE INIT. Both refuse with the + // same VK_ERROR_INITIALIZATION_FAILED, and the text that tells them + // apart goes to the library's stderr diagnostics, which silenceStdio + // suppresses. The queries named below are session-scoped, so they + // cannot answer either question until a session exists to ask them of. + // + // HOW A CALLER FINDS OUT, BEFORE IT ALLOCATES A FRAME POOL: + // QueryImageSupport() answers for a whole image descriptor and returns + // CONVERSION_REQUIRED naming what is missing. That query is the + // surface the decision is visible through. + VkFormat inputFormat; + + // ALPHA IS NOT ENCODED, AND THAT IS A REFUSAL RATHER THAN AN OMISSION. + // The RGBA spellings above are accepted as inputs and their A channel is + // DROPPED: the preprocess filter reads R, G and B, and what the encoder + // is handed is the Y'CbCr the matrix produced. There is nowhere for the + // alpha to go. Vulkan Video's encode extensions define no auxiliary + // picture layer, no alpha component and no second coded picture, and + // VkVideoProfileInfoKHR carries exactly three colour dimensions -- chroma + // subsampling, luma depth, chroma depth -- with no fourth. So this is an + // API-level absence, not a device one, and no driver changes it. + // + // THE OUT-OF-BAND ROUTE, for a caller that needs transparency: encode the + // colour here as usual and encode the alpha plane as the LUMA of a SECOND + // stream from a second encoder instance, with neutral chroma, then carry + // the two elementary streams together in the container. Several instances + // per process are supported, so the composition belongs above this + // library. This header names the route so that "the encoder does not do + // alpha" is not mistaken for "alpha cannot be carried". + + // The colour model |inputFormat|'s samples are in (see + // VkVideoEncoderColorModel). Zero-init reads it off |inputFormat|. + // + // The session is declared in this PAIR: |inputFormat| alone does not name + // an input, because the packed 4:4:4 Y'CbCr layouts share their enumerants + // with RGBA. Both halves are properties of the frames the session will be + // given, so both are stated once here and neither is sent per frame. + // + // IMMUTABLE ACROSS Reconfigure(), as |inputFormat| is. That call compares + // what the declaration RESOLVES to: naming the model the format already + // carries, or leaving the field FROM_FORMAT, resolves to what is in force + // and is accepted, while a declaration that resolves to the other model + // -- which is what the packed 4:4:4 readings of the RGBA enumerants are + // -- is refused rather than accepted and ignored. + VkVideoEncoderColorModel inputColorModel; + + uint32_t inputWidth; + uint32_t inputHeight; + + // Rate-control mode: the Vulkan enum directly. NOTE the values are bit + // flags -- DEFAULT = 0, DISABLED_BIT = 1, CBR_BIT = 2, VBR_BIT = 4 + // (VBR is 4, not 3). + VkVideoEncodeRateControlModeFlagBitsKHR rateControlMode = + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR; + uint32_t averageBitrate; // bits/sec + uint32_t maxBitrate; // bits/sec (VBR) + uint32_t vbvBufferSize; // bits (0 = default) + + // Constant quantizer for I, P and B pictures, applied when + // rateControlMode is DISABLED. -1 means "this config does not name that + // quantizer" and leaves it to the library, which resolves P from I and B + // from P. + // + // UNITS ARE THE CODEC'S OWN, AND THE RANGE FOLLOWS THE UNIT: + // + // H.264 / H.265 a QP, 0 .. 51 + // AV1 a quantizer index, 0 .. 255 + // + // A value ABOVE the codec's maximum is REFUSED, with + // VK_ERROR_INITIALIZATION_FAILED and a message on the error stream + // naming the field, the value and the range. It is not clamped, not + // reinterpreted and not truncated. Both entry points enforce this: + // InitializeExt() and Reconfigure(). 52 is therefore a legal AV1 + // quantizer index and a refused H.26x QP -- one range does not serve + // both codecs, and a caller that computes these values has to know which + // codec it is configuring. + // + // 0 is legal on every codec and is carried as given. + int32_t constQpI; + int32_t constQpP; + int32_t constQpB; + // Rate-control QP clamps for CBR/VBR sessions, H.26x units (QP 0..51). + // 0 = unset: no clamp reaches the driver. An explicit minQp of 0 + // collapses onto unset BY DESIGN -- QP 0 is the codec floor, so + // "clamp at 0" and "no lower clamp" admit the same QP range. A maxQp + // of 0 (force every frame to QP 0) is NOT expressible through these + // fields. AV1 rate + // control is quantizer-index based (0..255, a different unit): a + // non-zero value on an AV1 session is REJECTED at InitializeExt + // rather than reinterpreted or ignored. Values are validated against + // the codec range at InitializeExt and against the device's supported + // QP window at device init. + int32_t minQp; + int32_t maxQp; + + // GOP structure. gopLength 0 leaves the GOP length unstated, and the + // device's preferred GOP length applies; see idrPeriod below for the + // matching rule on IDRs. + uint32_t gopLength; // Frames per GOP + // B-frames between I/P pictures. + // + // 0 -> no B-frames (IPPP). + // 1 .. 254 -> that many B-frames. + // VK_VIDEO_ENCODER_B_FRAMES_DRIVER_PREFERRED -> let the driver choose. + // + // Driver-preferred has its own value rather than overloading 0, so that + // a caller asking for no reordering cannot silently get mini-GOPs. + // Anything else above 254 is rejected at init; 255 is reserved. + uint32_t consecutiveBFrames; + // Frames between IDRs. 0 does NOT mean "no IDRs" and does not mean "IDR + // on every frame": it leaves the period UNSTATED, and the device's own + // preferred IDR period -- reported per quality level and adopted when + // the session reads its device capabilities -- is what applies. It is + // read independently of gopLength. On a device that reports no + // preference the period stays unstated and only the first frame of the + // stream is an IDR, so a caller that needs random access at a known + // cadence states the period here. + uint32_t idrPeriod; + VkBool32 closedGop; + + // Frame rate + uint32_t frameRateNum; + uint32_t frameRateDen; + + // (The per-frame completion deadline lives in + // VkVideoEncoderFrameDeadlineInfo, chained onto this struct's pNext -- + // see the ABI rules at the top of this header.) + + // Quality + uint32_t qualityLevel; // 0 = default + + // Tuning mode. LOSSLESS engages transquant-bypass + QP0 in the codec + // config, producing bit-exact output (per the Vulkan spec); it requires + // rateControlMode DISABLED. + VkVideoEncodeTuningModeKHR tuningMode = + VK_VIDEO_ENCODE_TUNING_MODE_DEFAULT_KHR; + + // Colour info (VUI). ISO/IEC 23091-4 code points, shared verbatim by + // H.264, H.265 and AV1. + // + // 0 MEANS "NOT SUPPLIED" on this surface, and each of the three is read + // INDEPENDENTLY. Supplying only transferCharacteristics = 16 (PQ) + // declares PQ and leaves the other two Unspecified (code point 2). It + // does NOT also declare Reserved primaries and an Identity/GBR matrix: + // the three are read one at a time, never copied as a group. + // + // The cost of that sentinel: code point 0 itself (Reserved + // primaries/transfer, Identity/GBR matrix) cannot be REQUESTED through + // these fields. This encoder converts to YCbCr and has no Identity path + // to offer, so nothing is lost. + // + // If NONE of the three is supplied the bitstream carries no colour + // description at all, which a decoder reads as Unspecified -- not as a + // declaration of BT.709, and not as RGB. + // + // MATRIX, AND WHAT IT SELECTS: on an RGBA input, + // matrixCoefficients selects the matrix that is APPLIED, so the + // label the bitstream carries describes the pixels it carries. 1, 5, 6 + // and 9 are taken as named. 2 (Unspecified) is accepted and the matrix + // is DERIVED FROM colourPrimaries -- 9 from primaries 9, 6 from + // primaries 5, 6 or 7, and 1 otherwise -- then applied AND signalled + // back in place of the 2, so a caller that declares BT.2020 primaries + // and names no matrix gets BT.2020 chroma under a BT.2020 label. 10 + // (BT.2020 constant luminance) is the one accepted-and-approximated + // code point: it is signalled as asked and converted with the + // non-constant-luminance matrix, the only BT.2020 derivation available + // here. A code point that names a matrix the filter cannot produce -- + // 7 (SMPTE 240M), and anything outside BT.709/BT.601/BT.2020 -- is + // REFUSED at InitializeExt with a reason, not converted as BT.709 under + // the caller's label. 0 never reaches that gate: on this surface it is + // the "not supplied" sentinel described above, and is read as + // Unspecified. + // + // THE REFUSAL IS THE RGBA ARM'S ALONE, and the scope is stated because + // the paragraph above could be read as unconditional. On the DIRECT lane + // -- a Y'CbCr input the encoder reads as it lies -- this library applies + // NO matrix, so matrixCoefficients is a LABEL FOR THE CALLER'S OWN + // SAMPLES and there is nothing to refuse. Any code point the codec can + // express is carried into the VUI unexamined, including the ones the + // filter could not have produced. Refusing there would reject a truthful + // description of a picture this library never touched. + uint8_t colourPrimaries; + uint8_t transferCharacteristics; + uint8_t matrixCoefficients; + VkBool32 videoFullRange; + + // The transfer function the SUBMITTED FRAMES carry, as an ISO/IEC 23091-4 + // code point. transferCharacteristics above declares the transfer + // function the ENCODED BITSTREAM advertises, which is the one the encoder + // works in; this one declares what the input arrives in. Declaring both + // is how a caller states an OTF requirement rather than implying one. + // + // 0 means "not declared" and asserts nothing: the input is taken to be in + // transferCharacteristics already. + // + // THE CONTRACT. This library converts the COLOUR MODEL only -- RGB to + // YCbCr, and between YCbCr plane layouts and bit depths. It implements NO + // transfer function and applies none: the code values it writes are the + // code values it read. So the input and the bitstream must name the same + // transfer function, and a non-zero inputTransferCharacteristics that + // differs from transferCharacteristics is REFUSED at InitializeExt + // (VK_ERROR_INITIALIZATION_FAILED) with the reason. Refusing is the + // point: an unapplied transfer function produces pixels that are close + // enough to look plausible and wrong everywhere. + // + // WHAT THIS DECIDES ABOUT THE PREPROCESS CONVERSION. A colour-MODEL + // difference is what the preprocess filter exists to close, and the + // library engages it on that difference alone -- see inputFormat above. + // A transfer-function difference is not something the filter can close, + // so it is refused here instead of being filtered. + // + // *_SRGB input formats are refused outright, on a separate rule enforced + // by inputFormat. What this field decides about the _UNORM formats that + // ARE accepted is that their code values must already be in + // transferCharacteristics: the matrix is defined on gamma-encoded + // R'G'B', and nothing on the path re-encodes them. + uint8_t inputTransferCharacteristics; + + // Device selection. -1 is the library's default: the first enumerated + // device carrying the required device extensions and queue families. + // + // 0 IS NOT "UNSET". It is read as a PCI device ID, so a config that + // reaches this field holding 0 -- a memset, or any initialization that + // bypasses the default member initializer -- selects nothing and fails + // VK_ERROR_FEATURE_NOT_PRESENT. + int32_t deviceId = -1; + uint8_t gpuUUID[VK_UUID_SIZE]; // Preferred GPU UUID (all zeros = auto) + + // Bitstream output file path (null or empty = encoder library default, e.g. out.264/out.265/out.ivf) + const char* outputPath; + + // Debug + VkBool32 verbose; + VkBool32 validate; // Vulkan validation layers + + // When VK_TRUE, the encoder does not write to outputPath + // and instead captures the encoded bitstream in memory; the data is + // returned via VkVideoEncodeResult::pBitstreamData / bitstreamSize. + // The caller must copy out before invoking ReleaseEncodedFrame. + // Default VK_FALSE preserves the original file-output behavior. + // + // The COMPLETION surface does not depend on this flag. In both modes + // every submitted frame raises the completion edge (callback / event + // handle / counter) and becomes acquirable exactly once. In file-output + // mode the acquired result is a metadata record: bitstreamSize == 0, + // pBitstreamData == nullptr, status VK_SUCCESS when the bytes reached + // the file (the file-write error code otherwise), with isIDR and + // pictureType describing the frame. Distinguish it from a deadline drop + // by status (VK_SUCCESS vs VK_TIMEOUT). Release rules are identical: + // every delivered frame requires ReleaseEncodedFrame, and releasing a + // still-PENDING frame remains legal (its completion record is then + // discarded when it arrives). + VkBool32 disableFileOutput = VK_FALSE; + + // When VK_TRUE, the encoder library sends its own diagnostic output to a + // null stream instead of the console. An embedder that runs the encoder + // inside a sandboxed process, where writing to stdout/stderr is at best + // lost and at worst trips sandbox diagnostics, sets this VK_TRUE. + // Latched process-wide at InitializeExt() time; default VK_FALSE. + // + // IT DOES NOT SILENCE THE LIBRARY. Only the library's own gated output is + // covered. Some paths write to stdout and stderr directly and are not + // intercepted, and those are not all chatter -- the encoder's + // bitstream-readback failure reports are among them, as is the usage and + // per-option text written by the argv bridge in vulkan_video_encoder.h. + // + // So treat VK_TRUE as "the library stops volunteering status", not as a + // guarantee that nothing reaches stdout or stderr. An embedder that needs + // the guarantee has to redirect the descriptors itself. + VkBool32 silenceStdio = VK_FALSE; + + // ====================================================================== + // DEVICE OWNERSHIP. Three configurations, and which one you get is + // decided entirely by which of the three handles below are non-null. + // + // OWN all three VK_NULL_HANDLE. The library creates the instance, + // selects a physical device and creates the logical device. + // The only configuration available to a process that has no + // Vulkan implementation of its own. + // + // ADOPT externalInstance + externalPhysicalDevice, externalDevice + // LEFT NULL. The library BORROWS the instance, is PINNED to the + // supplied physical device, and creates its OWN VkDevice on it, + // probing its own queue families. Nothing of the caller's is + // destroyed at teardown. This is the supported way for an + // embedder that already has Vulkan up to keep the encoder on the + // same GPU it composites on without lending it a logical device. + // + // IMPORT all three supplied. The library encodes on the CALLER's + // VkDevice and binds the caller's queue families. For an + // embedder that must put the encoder on a logical device it + // already owns; the context path + // (CreateVulkanVideoEncoderExtOnContext) does not take it. + // + // PREFER ADOPT TO IMPORT. Under ADOPT the library creates a device it + // fully specifies while the physical device still comes from the + // embedder, so landing on the wrong GPU is structurally impossible. + // Under IMPORT the encoder's queues, enabled extensions and device + // lifetime are whatever the embedder created for its own purposes. + // + // THE PIN IS ENFORCED, NOT ADVISORY. If deviceId or gpuUUID is also set + // and names something other than the supplied physical device, the + // library REFUSES (VK_ERROR_FEATURE_NOT_PRESENT out of InitializeExt). + // It does not fall back to enumerating and picking something else -- a + // pin that silently re-selects is worse than no pin, because it looks + // like one. + // ====================================================================== + + // External VkInstance (optional) + // When non-null, the encoder creates its VkDevice on this instance + // instead of creating its own instance. Required for cross-process + // import on Windows where opaque Win32 handles are scoped per-instance, + // and required for ADOPT. + // + // VALIDATION OVER A BORROWED INSTANCE: setting `validate` alongside this + // does NOT install a debug callback. A debug callback is an + // instance-level object, and only the party that called vkCreateInstance + // knows whether VK_EXT_debug_utils / VK_EXT_debug_report was enabled on + // it. The embedder's own layer and callback report as usual. + VkInstance externalInstance = VK_NULL_HANDLE; + + // Caller-supplied VkPhysicalDevice / VkDevice / queue families. + // When externalDevice is set, the encoder shares the caller-provided + // VkDevice instead of creating its own. Nothing the caller supplied is + // destroyed at teardown, and the instance is tracked separately from the + // device, which is what makes the ADOPT combination -- borrowed + // instance, library-owned device -- tear down correctly. + // + // Contract: + // * externalInstance non-null is allowed standalone (VkInstance can + // be shared without sharing the device). + // * externalPhysicalDevice non-null with externalDevice VK_NULL_HANDLE + // is ADOPT, and is fully supported on this factory path: the library + // pins to that physical device and creates its own logical device on + // it. A session created on a context already takes both handles from + // the context, so naming either one in the config is refused there + // instead -- see CreateVulkanVideoEncoderExtOnContext. + // * externalDevice non-null REQUIRES externalPhysicalDevice non-null + // (Vulkan provides no API to recover the physical device from a + // logical device handle). InitializeExt() rejects the mismatched + // combination with VK_ERROR_INITIALIZATION_FAILED. On this factory + // path this is the only combination of the two that is rejected -- + // the reverse, physical-device-without-device, is ADOPT. + // * Default-initialized VK_NULL_HANDLE / UINT32_MAX preserves the + // original behavior (library creates everything / probes families). + // + // externalEncodeQueueFamilyIndex / externalComputeQueueFamilyIndex: + // when externalDevice is set, the caller created the device's queues, so + // the library MUST bind the families the caller actually created queues + // for: vkGetDeviceQueue on a family the device was not created with is + // undefined behavior, and the library's own probe may legally pick a + // different one. A valid family index here is used verbatim for the + // encode / compute queue, after validating that the family exists and + // carries the required queue flags and codec ops. UINT32_MAX (the + // default; == VK_QUEUE_FAMILY_IGNORED) keeps the library's probed family, + // and is the only correct value when externalDevice is VK_NULL_HANDLE, + // where these fields are ignored. + VkPhysicalDevice externalPhysicalDevice = VK_NULL_HANDLE; + VkDevice externalDevice = VK_NULL_HANDLE; + uint32_t externalEncodeQueueFamilyIndex = UINT32_MAX; + uint32_t externalComputeQueueFamilyIndex = UINT32_MAX; +}; + + +//============================================================================= +// External Frame Descriptor +// +// Describes a frame to encode that was allocated externally +// (e.g. imported from DMA-BUF in a cross-process encoder service). +//============================================================================= +struct VkVideoEncodeInputFrame { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME; + const void* pNext = nullptr; + + // The VkImage to encode (must be on the same device as the encoder) + VkImage image; + + // Image properties (must match the actual image) + VkFormat format; + uint32_t width; + uint32_t height; + VkImageTiling imageTiling = VK_IMAGE_TILING_OPTIMAL; // Must match actual image for path selection + VkImageLayout currentLayout; // Current layout of the image + + // Frame identification + uint64_t frameId; // Unique frame identifier + uint64_t pts; // Presentation timestamp (90kHz or custom) + + // Force an IDR at this frame. VK_FALSE lets the GOP structure decide. + VkBool32 forceIDR; + + // Set to VK_TRUE for the last frame to properly close the GOP + // and write end-of-stream markers. Without this, decoders may + // not be able to decode the trailing frames. + VkBool32 isLastFrame = VK_FALSE; + + // Per-frame QP override (-1 = use session default) + // Per-frame quantizer, -1 for "use the session's configured constQp". + // + // Units are the codec's own QP units -- the same ones the config's + // constQp uses: 0..51 for H.264/H.265, qindex 0..255 for AV1. There is no + // second convention to learn. + // + // Honoured ONLY when the session was initialized with rate control + // DISABLED. In CBR/VBR the encoder owns QP and an override would fight its + // rate controller, so it is refused there and logged once rather than + // half-applied. + int32_t qpOverride; + + // A per-frame tag the CALLER owns, for the caller's own diagnostics. The + // library accepts it and interprets nothing: it does not compare it, + // index by it, or echo it back. It is the producer's identifier for the + // IMAGE this frame was submitted from -- typically a slot in its frame + // pool -- which is a different thing from frameId. + // + // -1 is "no tag" and is the default. Every other value is the caller's to + // define. There is no counterpart on VkVideoEncoderFrameSubmitInfo, where + // the registration already names the image. + int32_t uniqueImageIndex = -1; + + // Queue-family ownership of |image| (see the enum above). + // Leave AUTO for first-use images, whose residency the declared layout + // infers; set LOCAL for reusable host-written staging images and + // FOREIGN for dma-buf/pixmap imports. + // + // HONOURED AS DECLARED, ON EVERY HANDLE TYPE, WITH ONE LAYOUT + // QUALIFICATION -- the rule stated in full on + // VkVideoEncoderExternalImageDescriptor::residency. An explicit LOCAL is + // taken as given wherever it is made and never derives a foreign acquire. + // An explicit FOREIGN still defers to a PREINITIALIZED layout. AUTO is + // derived by that same layout rule. Handle type is declared on the + // descriptor at registration and plays no part in reading this field. + // + // SO THE LAYOUT DECLARATION IS NOT FREE. Changing currentLayout changes + // the routing, and can turn a local host-written staging image into a + // queue-family ownership acquire. + // + // A REUSED registration does not have to declare a RESTORABLE layout: + // the library records the layout its own handback left the image in and + // names that on the next acquire, so a declaration only has to be true + // on the FIRST frame. See + // VkVideoEncoderExternalImageDescriptor::defaultLayout. + VkVideoEncoderInputResidency inputResidency = + VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO; + + // Synchronization: wait semaphores + // The encoder will wait on these before accessing the image. + // Typically this is the producer's graph timeline semaphore. + // + // BOUNDED at eight on the zero-copy (DIRECT) submit path; the staged path + // carries no such bound. A frame this entry point routes DIRECT whose + // wait list exceeds eight entries is REFUSED by the call itself with + // VK_ERROR_TOO_MANY_OBJECTS, rather than having a wait silently dropped. + // + // Eight is the ceiling on everything the frame carries, not on this field + // alone: a session encoding with a QP map spends a slot, and so does the + // hardware load-balancing timeline, so a frame carrying those can be + // refused below eight declared waits. The registration path publishes the + // same limit as ERROR_RESOURCE_LIMIT out of SubmitRegisteredFrame. + uint32_t waitSemaphoreCount; + // Const because the library only ever READS these arrays: it copies each + // element into the submit info it builds and never writes back through + // them. Declared non-const they would force every caller holding a const + // array -- which the public submit info hands out -- to cast the + // qualifier away at the assignment, and a cast that strips const is + // indistinguishable at the call site from one that intends to write. + const VkSemaphore* pWaitSemaphores; // Array of semaphores to wait on + const uint64_t* pWaitSemaphoreValues; // Timeline values (0 for binary semaphores) + + // Synchronization: signal semaphores + // The encoder will signal these after the image is no longer needed. + // Typically this is the consumer's release timeline semaphore. + uint32_t signalSemaphoreCount; + const VkSemaphore* pSignalSemaphores; // Array of semaphores to signal + const uint64_t* pSignalSemaphoreValues; // Timeline values (0 for binary) +}; + +//============================================================================= +// Encoded Frame Result +// +// Returned by GetEncodedFrame() after encoding completes. +//============================================================================= +struct VkVideoEncodeResult { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_ENCODE_RESULT; + const void* pNext = nullptr; + + uint64_t frameId; // Matches VkVideoEncodeInputFrame::frameId + uint64_t pts; // Pass-through from input + uint64_t dts; // Decode timestamp (encoder-assigned) + + // Bitstream + // Valid until ReleaseEncodedFrame(frameId). NOT invalidated by the + // next retrieval call -- several frames may be held acquired at + // once. ONE carve-out: encoder teardown (the final release of the + // encoder object) frees this storage regardless of acquisition + // state, so a consumer must not hold delivered pointers across the + // encoder's destruction. + const uint8_t* pBitstreamData; + uint32_t bitstreamSize; // Size in bytes + + // Frame info + VkVideoEncoderPictureType pictureType; // I=0 (incl. IDR), P=1, B=2 + VkBool32 isIDR; + uint32_t temporalLayerId; + + // Encode status: VK_SUCCESS, or the per-frame assembly/readback + // failure code -- e.g. VK_INCOMPLETE when the encode query status was + // not COMPLETE (such as INSUFFICIENT_BITSTREAM_BUFFER_RANGE). On + // failure pBitstreamData is null and bitstreamSize is 0; the frame is + // still delivered (and must still be ReleaseEncodedFrame()d) so the + // caller can raise an actionable per-frame error instead of stalling. + // + // SIZING A CONSUMER BUFFER NEEDS NO SEPARATE QUERY: the in-memory capture + // path never truncates, so bitstreamSize already IS the size required for + // the caller's output buffer. Vulkan video exposes no size query, so a + // frame that fails this way reports no required size and cannot be + // re-encoded larger; the per-frame status above is what makes the + // condition observable and actionable. + VkResult status; +}; + +//============================================================================= +// Runtime info descriptor +// +// Caller-queryable snapshot of the encoder's runtime characteristics, for a +// consumer that has to describe the encoder to its own clients. +// +// Carries the fields that decide how a consumer advertises the encoder: +// trusted rate controller, resolution alignment, native-handle and +// hardware-acceleration flags. Per-temporal-layer SVC arrays are not +// reported. +//============================================================================= +struct VkVideoEncoderRuntimeInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_RUNTIME_INFO; + const void* pNext = nullptr; + + char implementationName[64]; + VkBool32 isHardwareAccelerated; + VkBool32 supportsNativeHandle; + VkBool32 trustedRateController; + VkBool32 supportsSimulcast; + VkBool32 supportsFrameSizeChange; + VkBool32 reportsAverageQp; + uint32_t requestedResolutionAlignmentWidth; + uint32_t requestedResolutionAlignmentHeight; + VkBool32 applyAlignmentToAllSimulcastLayers; +}; + +//============================================================================= +// An imported synchronization primitive, registered once. +// +// The per-frame path takes VkSemaphore handles, which are process-local: a +// producer in another process has nothing meaningful to put there. Import +// once at setup, name by id per frame. +//============================================================================= +typedef struct VkVideoEncoderSemaphoreDescriptor { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_SEMAPHORE_DESCRIPTOR; + const void* pNext = nullptr; + + // OPAQUE_FD on Linux, OPAQUE_WIN32 on Windows. DMA_BUF and VK_IMAGE are + // image handle types and are rejected here. + VkVideoEncoderExternalHandleType handleType = + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_NONE; + + // TIMELINE only. A binary semaphore cannot express "wait for frame N" and + // is single-use, so it cannot be registered once and named repeatedly -- + // the whole point of registering it. + VkSemaphoreType semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + + // Ownership mode for the handle passed to RegisterSemaphore (see + // VkVideoEncoderHandleOwnership). Defaults to TRANSFER. + VkVideoEncoderHandleOwnership ownership = + VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_TRANSFER; +} VkVideoEncoderSemaphoreDescriptor; + +//============================================================================= +// Per-frame synchronization, by registered id and value. +// +// Chain onto VkVideoEncoderFrameSubmitInfo::pNext. Present so the per-frame +// path carries INTEGERS only: the handles crossed the process boundary once, +// at registration. +// +// The submit info's raw VkSemaphore arrays remain for in-process callers. +// Where this descriptor NAMES a direction, the registered ids replace that +// direction's raw array outright. +// +// The override is PER DIRECTION. A chain that names only waits leaves the +// caller's signal array exactly as it found it, and the mirror image holds +// too. +//============================================================================= +typedef struct VkVideoEncoderFrameSyncDescriptor { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_SYNC_DESCRIPTOR; + const void* pNext = nullptr; + + uint32_t waitCount = 0; + const VkVideoEncoderResource* pWaitSemaphores = nullptr; + const uint64_t* pWaitValues = nullptr; + + uint32_t signalCount = 0; + const VkVideoEncoderResource* pSignalSemaphores = nullptr; + const uint64_t* pSignalValues = nullptr; +} VkVideoEncoderFrameSyncDescriptor; + +//============================================================================= +// Per-frame fences (Linux sync_fd). Chain onto +// VkVideoEncoderFrameSubmitInfo::pNext -- the SAME chain the sync descriptor +// above hangs off, never a sub-chain of it. +// +// The walk is a FLAT, ORDER-INDEPENDENT pNext list, per the chaining rules at +// the head of this header. All three of these are honoured identically, and +// an unknown sType anywhere in the chain is refused wherever it sits: +// +// info.pNext = &fence; // fence alone +// info.pNext = &fence; fence.pNext = &sync; // fence leading +// info.pNext = &sync; sync.pNext = &fence; // fence trailing +// +// Separate from RegisterSemaphore, which is timeline-only: a binary semaphore +// cannot express "wait for frame N" and cannot be registered once then named +// repeatedly, which is the entire point of registering it. But a cross-API +// compositor's sync currency IS binary -- a fence handle on Linux is a binary +// sync_fd -- so a binary fence travels per frame. +// +// A SYNC_FD payload is consumed by the FIRST wait, which is exactly what a +// per-frame fence wants and why a registered semaphore is not imported the +// same way. +//============================================================================= +typedef struct VkVideoEncoderFrameFenceDescriptor { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR; + const void* pNext = nullptr; + + // A fence the encoder waits on before reading the input image. -1 = none. + // The library takes ownership and closes it on every exit path. On Win32 + // this field is not used: the value there is not a file descriptor, the + // library imports nothing from it, and it closes nothing -- a Win32 caller + // must leave this at -1. + // + // CALLER OBLIGATION, AND THE LIBRARY CANNOT DISCHARGE IT FOR YOU. + // After ANY return from SubmitRegisteredFrame -- SUCCESS or any refusal -- + // the number still sitting in this field names a descriptor that is + // already gone. The library does NOT write -1 back here -- unlike + // pReleaseFenceFd below, which it writes through a pointer into the + // caller's own storage. + // + // So a caller that KEEPS this descriptor and submits it again -- parking a + // frame that was refused and retrying it later is the obvious shape, and + // NOT_READY makes it the expected one -- MUST re-arm this field with a + // freshly exported fd, or set it to -1, before the second call. + // Re-submitting the identical struct hands the library a stale number: it + // is closed a second time, and by then the process may have opened + // something unrelated at that number, so the victim is not the fence. + // + // ADDITIVE TO THE RAW WAIT ARRAY, not a replacement for it -- which is + // where this differs from the sync descriptor above, whose named + // direction replaces that direction outright. A sync_fd is a SINGLE fence + // object, so a caller that exported one of its producers as an fd and + // left the rest in pWaitSemaphores has said something the fence alone + // cannot say. The library therefore waits on BOTH: supplying an + // acquireFenceFd and a pWaitSemaphores array together is legal and loses + // neither. + // + // BOUNDED ON THE DIRECT PATH, which qualifies "loses neither" above: + // the direct submit assembles its waits into a fixed eight-entry array. + // A registration that routes DIRECT must satisfy + // waitSemaphoreCount + (acquireFenceFd >= 0 ? 1 : 0) <= 8 + // -- this fence is appended to the wait array BEFORE the count is + // checked, so it is counted. Above eight, SubmitRegisteredFrame REFUSES + // the frame with VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT instead + // of dropping the surplus. Seven caller waits plus this fence is eight + // and fits. A registration that is not zero-copy carries no such bound, + // so this is a property of the direct arm alone. + // + // The refusal does NOT hand this fd back: it is imported before the + // count is checked, so it is consumed on that exit exactly as on every + // other. The ownership rule above is not suspended by the refusal. + int acquireFenceFd = -1; + + // [out, optional] A binary SYNC_FD fence signalled when the encoder has + // finished READING the input image. That is the submission which CONSUMES + // the input, and NOT the encode's completion: the bitstream becoming + // retrievable is a different event on a different resource, and the two + // must not be conflated. + // + // Feed it into the access seam that owns the backing, as a READ fence. + // The library never discharges that obligation for you; it only hands you + // the input to it. + // + // OWNERSHIP: the exported fd is the CALLER's. The library never closes + // it. close(2) it, or hand it to exactly one import, which consumes it. + // This is the REVERSE of acquireFenceFd above, and of every other fd rule + // in this header -- all of which govern handles the library is GIVEN. + // + // WHEN IT IS WRITTEN: before anything in this call can refuse. A caller + // may declare `int fd;` and rely on it being defined on EVERY return -- + // including ERROR_IMPORT_FAILED, ERROR_RESOURCE_UNKNOWN and + // ERROR_STRUCTURE_TYPE_UNKNOWN, and including the chains that put the + // refusing node AHEAD of this descriptor. A successful export overwrites + // it after the submit. + // + // -1 means "no fence", NOT failure, and is a legal answer every caller + // must handle. It is returned when: the platform or device cannot export + // a SYNC_FD binary semaphore; the frame already carries four or more + // signal semaphores, which leaves no room for this fence in the direct + // submit's signal array (refusing the fence is preferred to handing out + // an fd nothing will signal); the input-consuming submission had not been + // issued by the time this call returned (a frame deferred into a B-frame + // reorder batch is submitted on a LATER call); or the fence was already + // signalled, which vkGetSemaphoreFdKHR itself reports as -1. On -1 the + // caller must fall back to its own ordering -- it must NOT read -1 as + // "the input is already released". + int* pReleaseFenceFd = nullptr; +} VkVideoEncoderFrameFenceDescriptor; + +//============================================================================= +// Answer to QueryImageSupport. Chain a VkVideoEncoderImageSupportDetails +// (below) onto pNext for the renegotiation modifiers; unknown chained sTypes +// are rejected. +//============================================================================= +typedef struct VkVideoEncoderImageSupport { + VkVideoEncoderStructureType sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT; + const void* pNext = nullptr; + + // Would RegisterImageResource accept this descriptor? + VkBool32 supported = VK_FALSE; + + // Why not, when supported is VK_FALSE. VK_VIDEO_ENCODER_STATUS_SUCCESS + // when it is VK_TRUE. + VkVideoEncoderStatusCode status = VK_VIDEO_ENCODER_STATUS_SUCCESS; +} VkVideoEncoderImageSupport; + +// Capacity of the renegotiation list below. Truncated silently past this: +// the list exists so a producer can pick SOME workable allocation, not to +// enumerate the device. +#define VK_VIDEO_ENCODER_MAX_DIRECT_MODIFIERS 32 + +//============================================================================= +// Optional extension of VkVideoEncoderImageSupport: chain onto its pNext to +// receive, alongside the verdict: +// +// * directModifiers -- when the descriptor names an OS-handle import, the +// DRM format modifiers with which this exact descriptor (same format, +// usage, flags, extent; modifier swapped) WOULD register. This is the +// renegotiation list MODIFIER_UNSUPPORTED refers to: a producer +// re-allocates with one of these and resubmits -- one round trip. +// Count 0 when not initialized, when the descriptor is not an OS-handle +// import, when the descriptor declares no usage (imageUsage 0 is +// refused as USAGE_INSUFFICIENT under ANY modifier, so there is +// nothing to renegotiate), or when the device supports no workable +// modifier. +//============================================================================= +typedef struct VkVideoEncoderImageSupportDetails { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT_DETAILS; + const void* pNext = nullptr; // MUST be NULL + + uint32_t directModifierCount = 0; // OUT + uint64_t directModifiers[VK_VIDEO_ENCODER_MAX_DIRECT_MODIFIERS] = {}; // OUT +} VkVideoEncoderImageSupportDetails; + +//============================================================================= +// Per-call status echo, and never a control input: the library REPORTS which +// ownership rule it applied, so the caller can assert rather than guess. +// handlesConsumed is VK_TRUE iff the CALLER's handle was consumed by the call +// it was passed to: always true for a POSIX fd type under TRANSFER (success +// or failure alike), always false under BORROW (the library only ever +// consumes its private duplicate), always false for Win32 handles and for +// VK_IMAGE (no handle is taken). A constant of (platform, handle type, mode), +// and deliberately not of the call's outcome. +// +// Pass NULL for pNext. An unknown link, or a second link of a type already +// seen, is refused (STRUCTURE_TYPE_UNKNOWN) rather than ignored, with the +// ownership rule still applied to the handle on that exit like every other -- +// which under TRANSFER means the fd is consumed by the refusal. +//============================================================================= +struct VkVideoEncoderStatus { + VkVideoEncoderStructureType sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; + const void* pNext = nullptr; + + VkBool32 handlesConsumed = VK_FALSE; // OUT +}; + +//============================================================================= +// Extended Encoder Interface +// +// Extends VulkanVideoEncoder with external frame input and sync support. +// The base VulkanVideoEncoder methods (Initialize, EncodeNextFrame, etc.) are +// the file-based encoding interface -- declared in vulkan_video_encoder.h, +// which ships alongside this header and carries its own factory. This class +// adds to that interface; it does not replace it. +//============================================================================= +// What a terminal shutdown actually PROVED -- which is more than a bool can +// carry. Flush() joins the host workers either way. The question a caller has +// to answer before it takes back an image it lent the encoder is whether the +// GPU is still reading it, and a joined host thread does not answer that: +// compute/transfer staging whose dependent encode never submitted is queued +// work no host thread is waiting on. +enum VkVideoEncoderShutdownDisposition { + // Every host worker joined, the whole-device wait returned VK_SUCCESS and + // no device loss was observed. Producer input may be released -- even when + // Flush() reports an earlier encode or file-output failure. Already-copied + // host bitstreams stay valid until released. + VK_VIDEO_ENCODER_SHUTDOWN_IDLE = 0, + + // Every host worker joined and device loss is latched. Vulkan pending use + // is retired, but shared external contents and peer image layouts are not + // valid. This retires the wait; it is not permission to reuse the producer + // surface as if it were healthy. + VK_VIDEO_ENCODER_SHUTDOWN_LOST_DEVICE_RETIRED = 1, + + // Any other terminal wait error, or a shutdown still underway. Every GPU + // object, imported sync object, registration and producer dependency must + // be preserved: nothing has established that the device stopped using + // them. + VK_VIDEO_ENCODER_SHUTDOWN_UNPROVEN = 2, +}; + +// A snapshot, not a chained structure: it carries no sType and is never linked +// into a pNext walk. |pInfo| is fully overwritten. +struct VkVideoEncoderShutdownInfo { + VkVideoEncoderShutdownDisposition disposition; + + // The FIRST encode or shutdown error, kept. A later success does not clear + // it -- a shutdown that ends cleanly after a frame failed is still a + // shutdown that lost a frame. + VkResult firstError; + + // The last whole-device wait result, or VK_NOT_READY before one has been + // attempted. + VkResult lastDeviceWait; + + VkBool32 workersJoined; + VkBool32 callbackDetached; + + // Sticky, and deliberately separate from lastDeviceWait: a lost device + // followed by a wait that returns VK_SUCCESS is still a lost device, and + // the shared contents are still gone. + VkBool32 deviceLostObserved; + + VkBool32 shutdownComplete; +}; + +class VulkanVideoEncoderExt : public VulkanVideoEncoder { +public: + // Initialize with structured config (alternative to argc/argv) + virtual VkResult InitializeExt(const VkVideoEncoderConfig& config) = 0; + + // === Direct image submit: a VkImage per frame, without a registration === + // + // Hand over a VkImage per frame rather than naming a registration. + // Prefer RegisterImageResource + SubmitRegisteredFrame for new code: + // - the import, allocation and validation happen once per producer pool + // slot instead of once per frame; + // - an unsupported format or DRM modifier fails ONCE, at a negotiation + // point, with a code the caller can act on -- rather than on every + // frame forever, mid-stream, as a driver-level error; + // - a uint64 registration id crosses an IPC boundary; a VkImage handle + // does not, so a cross-process caller has no alternative. + // + // WHEN THE FRAME'S SEMAPHORES FIRE. The encoder waits on the frame's + // wait semaphores before it reads the input image, and signals the + // frame's signal semaphores once it has FINISHED READING that image. + // That is the RELEASE POINT. It is not the same event as the bitstream + // becoming retrievable, which is collected separately; on the DIRECT + // path below both are produced by the one encode submission, so a + // release carries no implication that the bitstream is not yet ready. + // + // Two input paths, chosen by the library from the frame's format and + // tiling and not something the caller asks for. They differ in one thing + // the caller can observe, which is WHERE that release point falls: + // DIRECT -- the image is encoded as it stands, so the release point is + // the encode submission itself. + // STAGED -- everything else. The input is taken into an image of the + // library's own first, and the release point is the completion of + // that staging copy, which is earlier than the encode. + // + // THIS CALL IS NOT ASYNCHRONOUS: the whole CPU-side encode pipeline runs + // INLINE on the calling thread before it returns, so budget for a full + // CPU-side encode issue rather than an enqueue. It does not block on + // capacity -- a full queue is refused up front as VK_NOT_READY, before + // any state changes. What IS asynchronous is completion: the bitstream is + // retrieved with AcquireNextEncodedFrame (or GetEncodedFrame). + // Threading: class (b), below. + // + // pStagingCompleteSemaphore [out, optional]: if non-null, receives the + // binary semaphore signaled when the staging copy completes. Useful + // when the caller needs to chain further GPU work (e.g. a display blit) + // that reads the same external image and must know when the encoder is + // done reading it. + // + // On the DIRECT path there is no staging copy, so VK_NULL_HANDLE is + // written. The frame's own signal semaphores, signaled by the encode + // submission, are the release point there. + // + // Signal semaphores passed in the frame are signaled at that same + // point. If the release must happen AFTER additional work, do not pass + // signal semaphores in the frame; chain on pStagingCompleteSemaphore + // and signal the release semaphore from the final submission. + // + // Returns VK_SUCCESS if the frame was accepted for encoding. + // Returns VK_NOT_READY if the encoder's queue is full (retry later) -- + // flow control, not an error. + virtual VkResult SubmitExternalFrame( + const VkVideoEncodeInputFrame& frame, + VkSemaphore* pStagingCompleteSemaphore = nullptr) = 0; + + // === Asynchronous Bitstream Retrieval === + // + // After a submit RETURNS, the GPU execution and the bitstream capture + // complete asynchronously; the submit call itself is not + // asynchronous (see the class (b) warning below). Use these methods + // to retrieve the encoded bitstream without blocking the encode + // pipeline. + + // THREADING CONTRACT (whole interface): + // (a) session-serial -- one thread, never concurrent with each + // other: InitializeExt, Flush, DrainPendingFrames, + // Reconfigure, SetCompletionCallback. ENFORCED, not merely + // documented: each returns VK_ERROR_NOT_PERMITTED_KHR when + // called from inside the completion callback. Teardown is in + // the same class and has no way to return an error, so + // dropping the LAST encoder reference from inside the + // callback is a diagnosed abort rather than a silent + // self-join or use-after-free. + // (b) submit-thread-affine -- the same thread for the whole session: + // SubmitExternalFrame, RegisterImageResource, + // SubmitRegisteredFrame, UnregisterImageResource, + // RegisterSemaphore, UnregisterSemaphore. A submit is + // NOT asynchronous -- the whole CPU-side encode pipeline + // (GOP/DPB bookkeeping, command recording and the queue + // submission) runs inline on the calling thread before it + // returns. It does not BLOCK on capacity (that is an up-front + // not-ready refusal -- VK_NOT_READY, or + // VK_VIDEO_ENCODER_STATUS_NOT_READY from the typed-status + // surface -- never a condition-variable wait); what is + // asynchronous is completion, which the class (c) surface + // below delivers. + // (c) fully thread-safe -- any thread: the retrieval methods below + // (AcquireNextEncodedFrame, AcquireEncodedFrame, GetEncodedFrame, + // ReleaseEncodedFrame), GetFrameStatus, CancelFrame, + // AbandonAllFrames, GetCompletionCounter, GetCompletionInfo, + // GetCompletionEventHandle, GetCompletionSemaphore, + // ExportCompletionSemaphoreHandle, GetRuntimeInfo. + + // Get the next completed encoded frame in COMPLETION order -- the order + // outcomes became deliverable, not the order frames were submitted, so a + // frame that never completes cannot strand the frames behind it. With no + // GOP reorder the two coincide, which is why delivery stays monotonic for + // muxers. Returns VK_SUCCESS and fills |result|; the frame is + // then ACQUIRED and pBitstreamData stays valid until + // ReleaseEncodedFrame() for that frameId -- the ONE lifetime rule. + // Its single carve-out is encoder teardown: the final release of + // the encoder object invalidates every outstanding pBitstreamData, + // so release delivered frames before dropping the last encoder + // reference. + // Flush() does NOT invalidate them; nothing else does either. + // Returns VK_NOT_READY while no undelivered frame has either a capture + // or an expired deadline. + // A frame past frameCompletionTimeoutNs is delivered as a 0-byte drop + // with status VK_TIMEOUT: data loss, not session loss -- head-of-line + // blocking is bounded by the deadline instead of wedging forever. + // In file-output mode completed frames deliver as 0-byte VK_SUCCESS + // records (see disableFileOutput). + // Threading: class (c). + virtual VkResult AcquireNextEncodedFrame(VkVideoEncodeResult& result) = 0; + + // Keyed retrieval: deliver a SPECIFIC frame regardless of submit order + // (the caller opts into out-of-order delivery). Returns VK_NOT_READY + // while pending; VK_ERROR_NOT_PERMITTED_KHR when already acquired; + // VK_ERROR_UNKNOWN when never submitted or already released. + // Threading: class (c). + virtual VkResult AcquireEncodedFrame(uint64_t frameId, + VkVideoEncodeResult& result) = 0; + + // Frame-state query; cannot fail. Threading: class (c). + virtual VkVideoEncoderFrameState GetFrameStatus(uint64_t frameId) = 0; + + // Alias of AcquireNextEncodedFrame(), for a drain loop written in these + // terms. Identical behaviour, and neither spelling is preferred over the + // other. Threading: class (c). + virtual VkResult GetEncodedFrame(VkVideoEncodeResult& result) = 0; + + // Release a delivered frame: frees the bitstream storage and returns + // the pool node. Required for EVERY delivered frame, including + // VK_TIMEOUT drops and per-frame failures. Releasing a still-PENDING + // frame is also legal (see disableFileOutput): the frame's completion + // record is discarded silently when it arrives, and does not count + // toward lateCaptures. Unknown ids are a logged + // no-op. Threading: class (c). + virtual void ReleaseEncodedFrame(uint64_t frameId) = 0; + + // Register (or clear, with nullptr) the completion callback -- see the + // PFN typedef's contract. The callback is invoked on a library thread; + // invocations are serialized. This call is a QUIESCE POINT: it returns + // only after any in-flight invocation of the previous + // callback has returned, so once a nullptr detach returns the caller + // may safely destroy whatever the previous pUserData referenced. Not + // callable from inside the completion callback itself + // (VK_ERROR_NOT_PERMITTED_KHR). Threading: class (a). + // |releaseUserData|, when non-null, transfers ownership of |pUserData| + // to the encoder: it is invoked exactly once when the encoder is done + // with it, after any in-flight invocation has returned. Prefer it to + // keeping the cookie alive by arrangement with teardown order. + virtual VkResult SetCompletionCallback( + PFN_vkVideoEncoderCompletionCallback callback, void* pUserData, + PFN_vkVideoEncoderUserDataRelease releaseUserData = nullptr) = 0; + + // Monotonic count of frame completions made retrievable (in + // file-output mode these are metadata-only records). Because callback + // notifications coalesce, consumers drain until their own retrieval + // count matches this counter. Threading: class (c). + virtual uint64_t GetCompletionCounter() = 0; + + // Observability snapshot; |pInfo| must be value-initialized + // (self-stamped). Chain a VkVideoEncoderDiagnosticInfo to + // |pInfo|->pNext to read the misuse channel in the same snapshot; a + // chained struct the library does not recognize is refused + // (VK_ERROR_INITIALIZATION_FAILED), not ignored -- the walk rule + // every other entry point applies. Threading: class (c). + virtual VkResult GetCompletionInfo(VkVideoEncoderCompletionInfo* pInfo) = 0; + + // What the terminal shutdown proved -- see VkVideoEncoderShutdownInfo. + // Readable before, during and after Flush(); a caller deciding whether it + // may release the input it lent this session reads it AFTER Flush() + // returns. Threading: class (c). + virtual void GetShutdownInfo(VkVideoEncoderShutdownInfo* pInfo) const = 0; + + // OS-handle completion currency, for a consumer that cannot be handed a + // C function pointer -- which is every out-of-process consumer. + // + // Returns a handle the caller waits on. The handle is the platform's + // own: on Linux, an eventfd. No build ships a handle nothing signals, so + // a caller never has to branch on a handle that exists but is never + // raised. + // + // VK_VIDEO_ENCODER_STATUS_SUCCESS with |*outHandle| set on success; + // ERROR_HANDLE_TYPE_UNSUPPORTED when the adapter could not create one + // (|*outHandle| is then 0, which is a legal handle value elsewhere and + // must not be waited on); ERROR_STRUCTURE_TYPE_UNKNOWN when |outHandle| + // is null. + // + // The handle is created on first request, owned by the encoder, and + // valid for the LIFETIME OF THE ENCODER OBJECT: it survives session + // teardown and is closed by the destructor, last, after the worker + // join (teardown is the refcounted release of the encoder; there is + // no public Deinitialize). The caller MUST NOT close it. Requesting + // it twice returns the same handle rather than a second one. + // + // Semantics match the callback exactly, and for the same reason: + // notifications COALESCE. The handle says "at least one frame became + // ready", never "exactly one". A waiter drains in a loop and reconciles + // against GetCompletionCounter(), and a waiter that assumes one wake + // equals one frame will silently strand frames -- the symptom of which + // looks exactly like a library stall. + // + // Registering no callback and using only this handle is a supported + // configuration; so is using neither. + virtual VkVideoEncoderStatusCode GetCompletionEventHandle(uint64_t* outHandle) = 0; + + // Completion currency 3: a Vulkan timeline, for GPU ordering ONLY -- + // never a wakeup. + // + // A library-owned timeline semaphore, signaled ON THE GPU by the encode + // queue. Its counter reaching (frameId + 1) means the encode GPU work + // of every external frame submitted with an id <= frameId has retired + // on the device: the bitstream and reconstruction writes are visible to + // device work that waits on it. Signals are COALESCED with the + // running maximum, because encode order is not input order under + // B-frames and a timeline may not signal non-monotonically. The +1 + // exists because a timeline's initial value is 0 and frame ids may + // legally start at 0. Consumers of this currency MUST submit + // monotonically increasing frameIds. + // + // What it is NOT: a readiness or wakeup signal. It is not pollable (an + // OPAQUE_FD export's legal operations are dup/dup2/close -- not poll), + // and it says nothing about RETRIEVABILITY: the host-side capture + // (fence wait, query readback, byte copy) happens after this signal + // fires, so a frame whose encode has retired here may not yet be + // acquirable. Readiness is the completion callback and the OS event + // handle, exclusively. + // + // A frame whose submit FAILS never advances the counter; do not enqueue + // waits for it. A deadline-synthesized VK_TIMEOUT drop is a host-side + // delivery event and is independent of this counter, in both + // directions: the counter advances iff the GPU work retires. + // + // The encoder owns the semaphore: valid from InitializeExt success + // until Flush() -- terminal for the encode session, it releases the + // underlying encoder and this semaphore with it -- or encoder + // teardown, whichever comes first. VK_NULL_HANDLE outside that + // window (including the rare session whose creation failed -- the + // currency degrades, the session does not), never destroyed by the + // caller. Do not leave GPU waits enqueued on it past that window. + // Threading: class (c). + virtual VkSemaphore GetCompletionSemaphore() const = 0; + + // Export the completion timeline for cross-process GPU ordering. + // + // OPAQUE_FD only; the Win32 arm is reserved and refused by type until + // its milestone, like every other Win32 arm here. Each call mints a NEW + // fd via vkGetSemaphoreFdKHR, and -- the REVERSE of the registration + // rule, which governs handles GIVEN TO the library -- the CALLER owns + // this fd: close it, or hand it to exactly one vkImportSemaphoreFdKHR, + // which consumes it. + // + // The export exists so another process can enqueue GPU waits against + // encode completion. It is not pollable and MUST NOT be a peer's + // liveness or wakeup mechanism; peer death belongs to the transport: + // an OPAQUE_FD timeline is never force-signalled when its owner dies, + // so an obligation that must survive the peer belongs in the + // transport's own currency, not on this semaphore. + // This interface carries no sync-file handle type; its semaphore arms + // import and export OPAQUE_FD timelines only. + // + // Returns ERROR_HANDLE_TYPE_UNSUPPORTED when the physical device + // cannot export an OPAQUE_FD timeline (the semaphore still works + // in-process) and for the reserved Win32 arm -- one answer, one clean + // branch, same as a reserved arm anywhere else here. NOT_INITIALIZED + // before InitializeExt. Threading: class (c). + virtual VkVideoEncoderStatusCode ExportCompletionSemaphoreHandle( + VkVideoEncoderExternalHandleType handleType, + uint64_t* outHandle) = 0; + + // Cancel DELIVERY of a frame: an unacquired frame (pending or ready) + // is converted to a 0-byte drop with status VK_INCOMPLETE; a late + // capture is discarded. The GPU work itself is not recalled. Already- + // acquired frames return VK_ERROR_NOT_PERMITTED_KHR; unknown ids + // VK_ERROR_UNKNOWN. Cancelled frames still require + // ReleaseEncodedFrame. Threading: class (c). + virtual VkResult CancelFrame(uint64_t frameId) = 0; + + // Peer loss: abandon every unacquired frame WITHOUT requiring anyone to + // retrieve and release it, and drop the claims those frames hold on + // their registrations. + // + // Cancel is for a live consumer changing its mind, and keeps the + // requirement to collect what it cancelled. This is for a consumer that + // is gone -- an out-of-process peer whose channel dropped -- where + // nothing will ever call ReleaseEncodedFrame, so that same requirement + // would pin every registration those frames name for the encoder's + // remaining lifetime and leave a deferred UnregisterImageResource + // permanently deferred. + // + // ALREADY-ACQUIRED frames are left alone. Their bitstream pointers are + // out in the caller's hands; freeing underneath a holder is worse than + // holding memory a while longer, and on a peer-loss path the holder is + // usually the thing being torn down anyway. + // + // |pAbandonedCount| may be null. Threading: class (c). + virtual VkResult AbandonAllFrames(uint32_t* pAbandonedCount) = 0; + + // === Flush and Drain === + + // Flush: encode all pending frames and make their bitstreams available. + // Blocks until all pending frames are encoded. + // + // TERMINAL for the encode session: the underlying encoder is released + // on the way out, so subsequent submits are refused + // (VK_ERROR_NOT_PERMITTED_KHR / ERROR_NOT_INITIALIZED) and + // GetCompletionSemaphore() answers VK_NULL_HANDLE from then on. + // Retrieval and ReleaseEncodedFrame of already-submitted frames stay + // valid -- delivered bitstream pointers are NOT invalidated by Flush. + // Use DrainPendingFrames() below for the non-terminal drain. + // Threading: class (a). + virtual VkResult Flush() = 0; + + // Non-terminal: flush the deferred GOP tail and wait for all in-flight + // encodes to complete, WITHOUT releasing the encoder. After this, + // GetEncodedFrame() can retrieve every frame submitted so far. + // + // NON-TERMINAL MEANS THE COMPLETION SURFACE TOO, not just the encoder + // object. Unlike Flush(), the session stays fully usable: further + // SubmitExternalFrame/SubmitRegisteredFrame calls are accepted AND each + // of those frames raises the completion edge and becomes acquirable + // exactly once, in both output modes, exactly as it would have without + // the drain. A drain may be called any number of times. + // + // Returns VK_ERROR_NOT_PERMITTED_KHR when there is no session or when + // called from inside a completion callback. A failure return other than + // that one means the drain completed -- everything already submitted is + // encoded and retrievable -- but the session could not be brought back + // up, so it can no longer report completions and further submits will + // fail rather than silently stall. + // Threading: class (a). + virtual VkResult DrainPendingFrames() = 0; + + // === Dynamic Reconfiguration === + + // Change rate control mid-stream without a session reset. Takes effect at + // the NEXT ENCODED FRAME, not at an IDR boundary. + // + // WHAT IT CHANGES: averageBitrate, maxBitrate, frameRateNum, + // frameRateDen, the constant-QP defaults constQpI/constQpP/constQpB, + // and the quantizer clamps minQp/maxQp. pNext must be NULL -- a chain + // is refused (VK_ERROR_INITIALIZATION_FAILED) -- so a caller that + // shares one VkVideoEncoderConfig with InitializeExt must clear pNext + // first. + // + // ZERO IS NOT "LEAVE THIS ALONE" ON THREE OF THOSE FOUR: + // * averageBitrate of 0 FAILS THE WHOLE CALL with + // VK_ERROR_NOT_PERMITTED_KHR, before any of the four is applied, so + // a caller moving only the frame rate must still restate the + // bitrate already in force. + // * maxBitrate of 0 is COERCED TO averageBitrate rather than meaning + // "no cap": a session reconfigured with a zero here comes back + // capped at its own average. + // * frameRateDen of 0 is coerced to 1 when frameRateNum is non-zero. + // * frameRateNum of 0 is the one that does mean "unchanged": the + // frame rate is left as it was and the call still succeeds. + // + // WHAT IS IMMUTABLE FOR THE LIFE OF THE SESSION, and is REFUSED rather + // than ignored when it differs from what InitializeExt was given: codec, + // profile, encodeWidth, encodeHeight, inputFormat, inputColorModel, + // rateControlMode, colourPrimaries, transferCharacteristics, + // inputTransferCharacteristics, matrixCoefficients and videoFullRange. + // Each is settled either in the sequence header written once at + // InitializeExt or in the input routing the session was built around, so + // changing one needs a session re-init. inputColorModel is compared AS IT + // RESOLVES against inputFormat: re-spelling the model a format already + // carries is accepted, while a declaration resolving to the other model + // is not. + // + // WHAT IS ALSO REFUSED ON A CHANGE -- the encoding parameters this call + // cannot carry. Each one is settled at InitializeExt and can reach the + // bitstream, so answering VK_SUCCESS to a change would leave the + // session encoding one way while the caller believed another: + // + // inputWidth and inputHeight -- what the session input conversion was + // built around, the other half of a declaration whose format and + // colour model are refused above. + // + // vbvBufferSize -- what a command carries is a VBV duration in + // milliseconds computed from this AND from an initial delay derived + // from the buffer size at init, both against the bitrate at init. + // Moving one of the three alone describes a buffer nobody has. + // + // gopLength, consecutiveBFrames, idrPeriod, closedGop -- these drive + // the structure that sequences frame types and DPB references as well + // as rate control, so a half-applied change would tell the driver one + // GOP while the encoder sequenced another. + // + // qualityLevel and tuningMode -- baked into the video session + // parameters at creation and into the profile respectively. + // + // MINQP AND MAXQP ARE APPLIED, with three exceptions that are REFUSED + // rather than ignored, because on each of them the value would reach + // nothing: on an AV1 session (AV1 rate control is quantizer-index based + // and consumes no QP-unit clamp), on a constant-QP session (that mode + // takes its quantizer from constQpI/P/B, which this call does carry), + // and for a value outside the H.26x range 0..51, an inverted window, or + // a value outside the QP window the device reports. A clamp of 0 means + // "no clamp", the same reading InitializeExt gives it, so a clamp set + // through this call can also be cleared through it. + // + // WHAT IS NEITHER APPLIED NOR REFUSED. The init-time plumbing and the + // diagnostics, which are READ BY NOTHING HERE and answered VK_SUCCESS + // whatever they hold: + // + // deviceId, gpuUUID, outputPath, verbose, validate, disableFileOutput, + // silenceStdio, externalInstance, externalPhysicalDevice, + // externalDevice, externalEncodeQueueFamilyIndex, + // externalComputeQueueFamilyIndex. + // + // Not one of them can change an encoded bit, so not one can misdescribe + // the stream. Refusing them would buy no correctness and would break a + // caller that builds a fresh minimal config for the reconfigure rather + // than copying its stored one. + // + // A CONSTANT-QP SESSION HAS EXACTLY ONE SESSION-LEVEL LEVER HERE, and it + // is the constant-QP triple. rateControlMode is immutable, so a session + // initialized VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR stays + // constant-QP, and the four rate fields land in per-layer state such a + // session does not carry -- that mode commands layerCount 0. constQpI, + // constQpP and constQpB are applied instead: a NEGATIVE member means the + // config names no quantizer and that one is left alone, while 0 is a + // valid (lossless) QP and is applied as one. Per-frame, + // VkVideoEncoderFrameSubmitInfo::qpOverride still moves the rate as well. + virtual VkResult Reconfigure(const VkVideoEncoderConfig& config) = 0; + + // === Device Access === + + // Get the encoder's Vulkan device handles. + // Use these for DMA-BUF import, semaphore creation, etc. + // The encoder owns these handles — caller must NOT destroy them. + virtual VkDevice GetVkDevice() const = 0; + virtual VkPhysicalDevice GetVkPhysicalDevice() const = 0; + virtual VkInstance GetVkInstance() const = 0; + + // The loader entry point these handles were resolved through. + // + // WHY THIS EXISTS, and why the three getters above are not enough. An + // embedder that owns no Vulkan device of its own still has to resolve + // device-level functions to touch the handles above -- to fill a staging + // image, to import a semaphore. Its own function-pointer table cannot + // serve: on a host where the embedder never brought Vulkan up there is + // nothing in it, and -- far worse -- on a host where the embedder + // ATTEMPTED Vulkan and failed, the table can be left holding a + // vkGetInstanceProcAddr that is non-null and DANGLING, pointing into a + // loader mapping that was torn down when the attempt failed. A null + // check passes and the call jumps into a non-executable page. + // + // So the rule is not "use ours when yours is missing", it is "on a + // library-owned device, ours is the only entry point with a defined + // lifetime": it stays valid exactly as long as the encoder does. + // + // Returns nullptr before the device context is brought up. The library + // owns this; the caller must not unload the loader behind it. + virtual PFN_vkGetInstanceProcAddr GetVkGetInstanceProcAddr() const = 0; + + // === Runtime Info === + + // Populate *outInfo with the encoder's current runtime characteristics. + // Valid only after InitializeExt() succeeds (returns VK_NOT_READY + // otherwise; VK_ERROR_INITIALIZATION_FAILED if outInfo is null, if its + // sType is not VK_VIDEO_ENCODER_STRUCTURE_TYPE_RUNTIME_INFO, or if + // anything is chained onto its pNext). The rate-control MODE is fixed + // for the session -- Reconfigure() refuses a change to it -- so + // trustedRateController does not change once the session is up. + virtual VkResult GetRuntimeInfo(VkVideoEncoderRuntimeInfo* outInfo) const = 0; + + // === Handle exchange === + + // Import |descriptor| once and return an id naming the result. The import + // and every allocation it needs happen here, not on the submit path. + // + // The caller owns the key -> id map. The library deliberately does NOT + // register-if-absent, because that puts allocation back on submit. + // + // VK_IMAGE registrations are the same-device OPT-IN arm, with a hard + // precondition: the image must remain valid, and its handle value + // stable, for the LIFETIME of the registration, and the + // caller must Unregister BEFORE destroying it -- a driver may recycle + // the handle value, and a surviving registration would then name freed + // memory. Per-frame ephemeral images must use an OS-handle registration + // (DMA_BUF et al) instead. Supplying |imageUsage| for a VK_IMAGE + // registration lets the library build spec-clean views and, when it + // includes VIDEO_ENCODE_SRC, register the image as one the encoder can + // read as it stands. Leaving it 0 is accepted, but the slot is then + // presumed transfer-source only: access the caller never declared is + // never granted, so nothing here can make the encode read an image + // without encode usage. + // + // |osHandle| is an int fd on POSIX and a HANDLE on Windows, passed as + // uint64. Ownership follows |descriptor.ownership| (TRANSFER unless the + // caller says otherwise): under TRANSFER an fd is consumed on EVERY + // exit path including failure; under BORROW the caller keeps its fd on + // every exit path (the library works on an immediate private + // duplicate). A Win32 handle is never closed by the library in either + // mode. + // + // |pStatus|, optional and self-stamped, receives the ownership echo + // (handlesConsumed) on every return -- assert against it rather than + // guessing. A non-null |pStatus| with the wrong sType is version skew + // and is refused (STRUCTURE_TYPE_UNKNOWN), with the ownership rule + // still applied to the handle on that exit like every other. + // + // Returns a typed status. VK_VIDEO_ENCODER_STATUS_ERROR_MODIFIER_UNSUPPORTED + // is renegotiable; DEVICE_MISMATCH and USAGE_INSUFFICIENT are not. + virtual VkVideoEncoderStatusCode RegisterImageResource( + const VkVideoEncoderExternalImageDescriptor& descriptor, + uint64_t osHandle, + VkVideoEncoderResource* outResource, + VkVideoEncoderStatus* pStatus = nullptr) = 0; + + // Retire a registration. Deferred and REFCOUNTED: the underlying image is + // freed only once no submitted frame can still read it, and the call does + // not stall on a device wait to establish that. + // + // The id is invalid immediately on return; any later use is rejected + // rather than dereferenced. + virtual VkVideoEncoderStatusCode UnregisterImageResource( + VkVideoEncoderResource resource) = 0; + + // Answer whether a descriptor WOULD register, without importing + // anything and without a handle -- so a producer can negotiate before it + // allocates, rather than discovering the answer per frame, mid-stream, + // as a driver error. + // + // It runs the SAME predicate RegisterImageResource runs, so the query and + // the registration cannot disagree. + // + // |supported| VK_FALSE is never the end of the story: |status| names the + // reason, and the renegotiable ones (MODIFIER_UNSUPPORTED, + // CONVERSION_REQUIRED) are what let a producer pick different allocation + // parameters instead of giving up. + // + // Handle types whose arms are reserved but not yet implemented answer + // VK_FALSE with ERROR_HANDLE_TYPE_UNSUPPORTED, so a caller branches + // cleanly rather than building a pipeline on a promise. + virtual VkVideoEncoderStatusCode QueryImageSupport( + const VkVideoEncoderExternalImageDescriptor& descriptor, + VkVideoEncoderImageSupport* outSupport) = 0; + + // Import a synchronization primitive once and return an id naming it. + // + // |osHandle| follows the same ownership contract as image registration: + // |descriptor.ownership| selects the mode (TRANSFER by default: an fd + // is consumed on every exit path including failure; BORROW: the caller + // keeps it on every exit path); a Win32 handle is never closed by the + // library in either mode. |pStatus| is the same optional echo, with one + // difference: it carries no chain here. See the note on + // VkVideoEncoderStatus for why, and for what the refusal costs an fd + // presented under TRANSFER. + // + // CROSS-PROCESS ON WINDOWS: |osHandle| must already be valid in the + // encoder's process. A Win32 handle is process-local, and this interface + // carries no source PID, so the caller performs the + // OpenProcess(PROCESS_DUP_HANDLE) + DuplicateHandle itself and passes the + // duplicate. It also closes that duplicate once registration returns -- + // the library never closes a handle it did not create. + // + // Duplicate on the IMPORTING side, not the exporting side. Pre-injecting + // a duplicate from the exporter collides with the importer's handle + // table. + // + // Timeline semaphores only -- see the descriptor for why. + virtual VkVideoEncoderStatusCode RegisterSemaphore( + const VkVideoEncoderSemaphoreDescriptor& descriptor, + uint64_t osHandle, + VkVideoEncoderResource* outResource, + VkVideoEncoderStatus* pStatus = nullptr) = 0; + + // Retire a semaphore registration. Unlike an image, this does NOT + // defer: the imported VkSemaphore is destroyed before the call returns. + // The caller must therefore ensure that every submitted batch which + // waits on or signals this registration has completed execution + // (VUID-vkDestroySemaphore-semaphore-01137) before unregistering, and + // must not name the id afterwards. Registrations never retired are + // destroyed at session teardown, after the encoder's threads are + // joined and its queue is idle. + virtual VkVideoEncoderStatusCode UnregisterSemaphore( + VkVideoEncoderResource resource) = 0; + + // Submit a frame against a registration. This call performs no image + // import and allocates nothing for the input: the image, its memory, its + // view and its wrapper were created once at registration, and registered + // semaphores named by id are resolved rather than imported. + // + // A chained per-frame fence descriptor is the exception: an armed + // acquireFenceFd and a requested release fd are per-frame Vulkan objects, + // and both retire with the frame. + // + // The registration is reference-counted for the lifetime of the submitted + // frame, so an Unregister racing an in-flight frame defers rather than + // freeing memory the GPU is still reading. + // + // Returns VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN for a stale or + // never-registered id -- including one retired since it was minted, which + // the generation counter catches instead of silently matching a recycled + // slot. + // + // Threading: class (b), and the warning there applies: the CPU-side + // pipeline runs inline on this call; capacity is an up-front + // VK_VIDEO_ENCODER_STATUS_NOT_READY (drain completions and retry the + // same call), never a blocking wait; completion is what happens + // asynchronously. + // + // WAIT-COUNT BOUND, direct path only. A registration that routes DIRECT + // assembles its waits into a fixed eight-entry array, so the frame's + // wait count plus its acquire fence must be <= 8. Beyond that this call + // returns VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT and encodes + // nothing, rather than discarding the surplus waits. + // Retrying unchanged cannot succeed: present fewer waits, or use a + // registration that is not zero-copy, whose wait list is not bounded + // here. The refusal consumes an acquireFenceFd like + // every other exit. + virtual VkVideoEncoderStatusCode SubmitRegisteredFrame( + const VkVideoEncoderFrameSubmitInfo& info, + VkSemaphore* pStagingCompleteSemaphore) = 0; +}; + +// Factory function for the extended encoder interface +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoderExt( + VkSharedBaseObj& vulkanVideoEncoder); + + +//============================================================================= +// Encoder capability enumeration BEFORE InitializeExt(). +// +// A consumer typically enumerates the profiles it can offer at process +// startup, before any encoder session exists, so that it can advertise them +// to a capability query or a codec negotiation. These are therefore free +// functions, and none of them creates a VkDevice or an encode session. +// +// The scalars -- coded extent, rate-control modes and the rest -- come back +// in VkVideoEncoderCapabilities below. The Std syntax flags and the accepted +// input formats are LISTS, and each is read from its own two-call entry point +// on a context -- VkEncEnumerateStdFlags and VkEncEnumerateInputFormats -- so +// that the capacity is the caller's rather than a constant this header has to +// pick. A caller holding no Vulkan handles builds an OWN-mode context with a +// zero gpuUUID. +//============================================================================= +// Whether an advertised input format is the one the encoder wants. +// +// STATED, and not derivable from the entry's two formats: a format the +// encoder does not read can still name ITSELF as its encode format, so +// format == encodeFormat holds for some SUBOPTIMAL entries exactly as it does +// for every OPTIMAL one. +// +// A property of the FORMAT, under this codec, profile and device. Whether one +// particular IMAGE is taken as it lies is a property of that allocation -- +// its tiling and its modifier -- and QueryImageSupport is what answers it, +// per allocation. +typedef enum VkVideoEncoderInputFormatOptimality { + // The encoder's own input format here: a frame supplied in it is what the + // encoder reads. + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL = 0, + // Not the format the encoder reads. Accepted, and coded correctly, + // because the library converts it into encodeFormat first. That + // conversion is what the entry costs over an OPTIMAL one, so prefer an + // OPTIMAL entry wherever the producer can supply one. + // + // There is no knob here: which formats can be converted is what this list + // reports. A build that can perform no conversion advertises no + // SUBOPTIMAL entry at all, rather than advertising one and refusing the + // session that declares it; its OPTIMAL entries are unaffected. + VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL = 1, +} VkVideoEncoderInputFormatOptimality; + +// One input format this encoder accepts: the format itself, the format the +// bitstream is coded from, and whether the encoder wants it. +// +// encodeFormat is what the encoder is given, so encodeFormat -- not format -- +// is what decides the chroma subsampling and bit depth of the bitstream. A +// caller that hands over an RGBA frame reads from encodeFormat that the +// bitstream is coded from a Y'CbCr one. +// +// Read optimality for which entry to prefer and encodeFormat for what the +// bitstream carries; neither answers the other's question. +// +// Deliberately without sType/pNext. This is an array element the library +// WRITES, not a parameter structure a caller fills, and a pointer inside an +// array element is what stops that array being marshalled as one block. +struct VkVideoEncoderInputFormatProperties { + VkFormat format; + VkFormat encodeFormat; + VkVideoEncoderInputFormatOptimality optimality; +}; + +// A Std syntax-flag bitmask, as reported by the driver for one +// (codec, profile) pair. +// +// Interpret an entry against the codec the query named: +// VkVideoEncodeH264StdFlagsKHR, VkVideoEncodeH265StdFlagsKHR or +// VkVideoEncodeAV1StdFlagsKHR. All three are VkFlags and a query names +// exactly one codec, so one list carries whichever applies rather than three +// lists of which two are always empty. +typedef VkFlags VkVideoEncoderStdFlags; + +struct VkVideoEncoderCapabilities { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES; + const void* pNext = nullptr; + + VkVideoCodecOperationFlagBitsKHR codec; + + // Level range. maxLevelIdc is the numeric value of the codec-specific + // Std level the driver reports -- StdVideoH264LevelIdc, + // StdVideoH265LevelIdc or StdVideoAV1Level; minLevelIdc is 0 when the + // driver does not expose a floor (Vulkan has no min-level cap today). + uint32_t minLevelIdc; + uint32_t maxLevelIdc; + + // DPB / reference limits (VkVideoCapabilitiesKHR). + uint32_t maxDpbSlots; + uint32_t maxActiveReferencePictures; + + // Encode caps (VkVideoEncodeCapabilitiesKHR). + uint32_t maxQualityLevels; + uint64_t maxBitrate; + + // Picture access granularity (VkVideoCapabilitiesKHR::pictureAccessGranularity). + uint32_t pictureAccessGranularityWidth; + uint32_t pictureAccessGranularityHeight; + + VkVideoEncodeRateControlModeFlagsKHR supportedRateControlModes; + VkVideoEncodeCapabilityFlagsKHR flags; + + // Coded-extent range (VkVideoCapabilitiesKHR). + VkExtent2D minCodedExtent; + VkExtent2D maxCodedExtent; + + // The driver's Std syntax flags and the input formats this encoder + // accepts are NOT here: each is a LIST, answered by its own two-call + // entry point (VkEncEnumerateStdFlags, VkEncEnumerateInputFormats), where + // its membership rules are stated. What is left here is scalars, which + // keeps this structure trivially copyable and 1:1 IPC-shaped. pNext must + // be NULL; every entry point that fills this structure refuses a chain. + + // Optional-feature availability. supportsMaintenance1 and + // supportsQuantizationMap are device-extension presence; + // supportsIntraRefresh additionally requires the device to report at + // least one intra-refresh mode. supportsResizeWithoutIdr is false: no + // Vulkan capability expresses it, so a resize forces an IDR. + bool supportsQuantizationMap; + bool supportsIntraRefresh; + bool supportsMaintenance1; + bool supportsResizeWithoutIdr; // for caller-choice IDR-on-resize +}; + +//============================================================================= +// Capability probing, per codec AND profile. +// +// vkGetPhysicalDeviceVideoCapabilitiesKHR answers per VkVideoProfileInfoKHR, +// so a caller advertising H.264 Baseline/Main or H.265 Main-10 must probe +// those exact (profile, bit-depth) combinations rather than copy a sibling +// profile's result. VK_VIDEO_ENCODER_PROFILE_DEFAULT probes the codec's +// representative profile -- H.264 High, H.265 Main, AV1 Main, all 8-bit -- +// which is the whole of what a caller with no particular profile in mind +// needs to ask. +// +// Both return VK_SUCCESS and fill *outCaps on success, and +// VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR when `profile` is not +// one this library probes for `codec` -- including a profile number that +// belongs to a different codec. They differ in how they say "not this +// device". The caller-handles variant is GIVEN the device, so a codec op it +// does not probe is VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR and a +// device that does not support encode for `codec` is +// VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR. The ephemeral variant +// SELECTS the device first, so everything that leaves it with no candidate +// -- an unmatched deviceId, a device lacking the video extensions, or no +// device offering encode for `codec` -- is VK_ERROR_FEATURE_NOT_PRESENT. +// A caller treats any non-VK_SUCCESS as "do not advertise +// this profile" rather than as a hard error. +//============================================================================= + +// The caller supplies the Vulkan handles. Use this when the caller has +// already initialized its VkInstance and VkPhysicalDevice. The library wraps +// them in an internal device context, probes WITHOUT creating a VkDevice or +// an encode session, and does not destroy the caller's handles. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult EnumerateVulkanVideoEncoderProfileCapabilities( + VkInstance instance, // caller-supplied + VkPhysicalDevice physicalDevice, // caller-supplied + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkVideoEncoderCapabilities* outCaps); + +// No caller-supplied handles. Use this from a caller that has not +// initialized Vulkan yet, and pass -1 for the library's default (first +// capable) device. +// +// "Ephemeral" names the API contract -- the caller supplies no handles and +// gets none back. It does NOT mean Vulkan is torn down between calls: an +// instance created here is held for the process lifetime, because re-creating +// one after a sandbox has locked down is a device loss. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult EnumerateVulkanVideoEncoderProfileCapabilitiesEphemeral( + int32_t deviceId, // -1 = first capable + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkVideoEncoderCapabilities* outCaps); + +//============================================================================= +// The encoder context +// +// A refcounted object holding a live VkInstance and the enumerated physical +// devices, so capabilities -- codecs, profiles, input formats, DRM modifiers, +// rate-control modes -- can be answered WITHOUT creating and destroying a +// loader, an instance or a device per query. A context creates no VkDevice; +// a session created on one creates its own. +// +// IMMUTABLE AFTER CONSTRUCTION. The device list, the identities and the +// per-(codec, profile) capability snapshot are all filled during construction +// and are read-only afterwards. That is what lets unrelated sequences SHARE +// ONE CONTEXT WITH NO SYNCHRONISATION OF THEIR OWN, and it is why there is no +// refresh entry point: a new generation of capability truth is a new context. +// TWO ACCESSORS READ THE DRIVER RATHER THAN THE SNAPSHOT, and both do so +// because their answer depends on something not known at construction. The +// DRM-modifier query is keyed on (format, usage). VkEncEnumerateInputFormats +// is keyed on (codec, profile) and resolves every candidate live through the +// same resolver the point query and InitializeExt use -- which is what makes +// the advertised set and the accepted set one set rather than two that can +// drift, and is therefore the reason it cannot be served from a snapshot +// probed at a fixed envelope. NEITHER TOUCHES CONTEXT STATE, so the +// immutability above is exactly what it says: the context is not written +// after construction, by these or by anything else. +// +// Lifetime rules, each from a specific hazard: +// +// 1. THE LOADER HANDLE IS NEVER UNLOADED. An embedder resolves its own +// Vulkan entry points out of the same libvulkan.so.1 / vulkan-1.dll, and +// unloading it when a context goes away would invalidate pointers the +// embedder still holds. The context retains it for the process lifetime, +// in BOTH modes, including for a context created and released inside a +// single call. +// +// 2. A SUCCESSFULLY BUILT OWN-MODE CONTEXT IS NEVER DESTROYED. Re-issuing +// vkCreateInstance after a sandbox has locked down is a device loss +// rather than a slow path, so the library holds a floor reference to +// every OWN-mode context it builds, keyed by the create-info's gpuUUID: +// a second create returns the SAME context instead of standing up a +// second instance. A create whose bring-up FAILS is not registered, so a +// later create with the same key builds again. Callers layer their own +// references on top and may drop them freely. +// +// 3. ADOPT MODE NEVER DESTROYS. The VkInstance and VkPhysicalDevice belong to +// the embedder; release is a no-op on both. ADOPT contexts are +// deliberately NOT floor-referenced: a cached borrowed handle would +// outlive the embedder's ownership of it, and Vulkan handle values are +// recycled, so a stale entry could match a different object. +// +// 4. Context identity is the (VkInstance, VkPhysicalDevice, VkDevice) triple. +// A context creates no VkDevice, so the third element is VK_NULL_HANDLE +// for every context this API builds. +// +// A host that already has a VkInstance may use either mode. A host with no +// Vulkan implementation of its own must use OWN. +//============================================================================= + +// Opaque, refcounted, immutable after construction. Never defined here: the +// only operations on it are the free functions below. +class VulkanVideoEncoderContext; + +typedef enum VkVideoEncoderContextMode { + // Create and own a VkInstance, then enumerate physical devices. + VK_VIDEO_ENCODER_CONTEXT_MODE_OWN = 0, + // Borrow an instance and physical device the embedder already has. + // Nothing is created and nothing is destroyed on release. + VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT = 1, +} VkVideoEncoderContextMode; + +struct VkVideoEncoderContextCreateInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONTEXT_CREATE_INFO; + const void* pNext = nullptr; + + // Deliberately WITHOUT a default member initializer. OWN is 0, so a + // zero-filled create-info means OWN; giving this field a different + // default would make `T x = {};` and `memset` disagree about what the + // caller asked for. Set it explicitly. + VkVideoEncoderContextMode mode; + + // ADOPT only, and both are required TOGETHER -- either one alone is + // VK_ERROR_INITIALIZATION_FAILED. Supplying either in OWN mode is also an + // error: handles handed to a mode that would not borrow them mean the + // caller asked for one thing and meant another. + VkInstance adoptInstance = VK_NULL_HANDLE; + VkPhysicalDevice adoptPhysicalDevice = VK_NULL_HANDLE; + + // OWN only. An all-zero UUID means "enumerate everything"; a non-zero one + // pins the context to that single physical device. deviceID cannot do + // this job -- it is a PCI device id, so it cannot disambiguate two + // identical GPUs. A non-zero UUID in ADOPT mode is an error: the device + // is already chosen. + uint8_t gpuUUID[VK_UUID_SIZE]; + + // Latches the library's process-wide stdio-silence gate when VK_TRUE. + // VK_FALSE does NOT un-silence: a capability probe must not undo a + // session's choice, which is why this is not the unconditional set that + // InitializeExt does with the same-named config field. + VkBool32 silenceStdio = VK_FALSE; +}; + +// Stable identity across processes and runs. +struct VkVideoEncoderDeviceIdentity { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_DEVICE_IDENTITY; + const void* pNext = nullptr; + + uint8_t deviceUUID[VK_UUID_SIZE]; + uint8_t driverUUID[VK_UUID_SIZE]; + char deviceName[VK_MAX_PHYSICAL_DEVICE_NAME_SIZE]; + uint32_t vendorID; + // VkPhysicalDeviceProperties::deviceID. Carried because the library's own + // ephemeral capability entry points select on it, NOT because it is an + // identity: two identical GPUs share one deviceID. Use deviceUUID to + // identify a device and this only to reproduce a deviceID-based selection. + uint32_t deviceID; +}; + +// Build a context. See the lifetime rules above: in OWN mode this may return +// an existing context rather than a new one, and the returned context outlives +// every reference the caller drops -- only VkEncRetireOwnContexts() releases +// it. In ADOPT mode a fresh context is built every time and destroying it +// destroys nothing of the caller's. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoderContext( + const VkVideoEncoderContextCreateInfo* pCreateInfo, + VkSharedBaseObj& outContext); + +// Release every cached OWN-mode context, destroying the VkInstance each holds, +// and returns how many were released. Reached publicly as IPlatformLifetime. +// +// Retirement is permanent: an OWN-mode create after this call is refused +// rather than allowed to issue a second vkCreateInstance. Call it while the +// process can still reach the driver -- not from a static destructor. +uint32_t VkEncRetireOwnContexts(); + +// Create an encode session ON a context. This is the session-to-context +// link, and it is how a session is meant to obtain a borrowed instance and +// physical device. +// +// The session creates its own VkDevice on the context's |deviceIndex| +// physical device -- a context creates none, in either mode -- so two +// sessions on one context are two devices on one instance, not a shared +// device. +// +// The session holds a reference to |context| for its whole life. That keeps +// the CONTEXT OBJECT -- and the capability snapshot the caller selected from -- +// alive and addressable while the session runs. +// +// IT DOES NOT KEEP THE BORROWED VkInstance ALIVE, and no reference here could. +// In ADOPT the instance belongs to the embedder and the context destroys +// nothing on release (lifetime rule 3 above); in OWN the library's floor +// registry already holds the context above zero for the process lifetime +// (rule 2). AN EMBEDDER THAT DESTROYS ITS VkInstance WHILE A SESSION IS LIVE +// HAS A USE-AFTER-FREE, on this path exactly as on the config path. That +// hazard is the embedder's to avoid. +// +// THE SESSION DOES NOT CONSUME THE CONTEXT'S CAPABILITY SNAPSHOT, which is +// the assumption to avoid making. It reads the instance and the physical +// device out of the context and nothing else, then re-probes queue families, +// device extensions and encode capabilities for itself, so N sessions on one +// context run N+1 capability sweeps rather than one. +// +// A CONFIG THAT ALSO SELECTS A DEVICE IS REFUSED with +// VK_ERROR_INITIALIZATION_FAILED, not resolved: externalInstance, +// externalPhysicalDevice, a deviceId other than -1, and a non-zero gpuUUID all +// name a device the context has already chosen. (The config ADOPT path +// answers VK_ERROR_FEATURE_NOT_PRESENT for the same user error; two codes for +// "you named the device twice", and this path's is the typed one.) A stale +// config field outranking the context would send the session to a different +// GPU than the one whose capabilities the caller just read, which on a +// single-GPU host is entirely invisible. +// +// config.externalDevice is refused for a different reason: a context owns +// the instance and the physical device and creates no VkDevice (context +// rule 4), and a session built on one creates its own -- so a caller- +// supplied logical device has nowhere to land on this path. +// +// |deviceIndex| indexes the context's snapshot -- 0..VkEncGetPhysicalDeviceCount-1, +// and is always 0 in ADOPT mode. Out of range is VK_ERROR_INITIALIZATION_FAILED. +// +// ON FAILURE |vulkanVideoEncoder| IS LEFT UNTOUCHED, matching +// CreateVulkanVideoEncoderExt. It is not nulled: a failing create must not +// destroy a session the caller already had in that variable. +// +// InitializeExt() is still what starts the session. This decides only where +// its instance and physical device come from. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoderExtOnContext( + const VkSharedBaseObj& context, + uint32_t deviceIndex, + VkSharedBaseObj& vulkanVideoEncoder); + +// Number of physical devices in the context's snapshot. Always 1 in ADOPT +// mode. 0 for a null context -- the count query has no error channel, and a +// caller that then indexes 0..count-1 does nothing at all, which is the +// correct behaviour for a context that was never built. +extern "C" VK_VIDEO_ENCODER_EXPORT +uint32_t VkEncGetPhysicalDeviceCount(VulkanVideoEncoderContext* ctx); + +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult VkEncGetPhysicalDeviceIdentity( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoEncoderDeviceIdentity* pOut); + +// Read one (codec, profile) entry out of the snapshot. No driver call is +// made: the answer was computed in the constructor. Returns exactly what the +// probe returned at construction time, so a caller maps any non-VK_SUCCESS to +// "do not advertise this profile" the same way it does for the free-function +// enumerators. *pOut is written only when the entry was actually probed. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult VkEncGetEncodeCapabilities( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkVideoEncoderCapabilities* pOut); + +// The Std syntax flags the driver reports for (|codec|, |profile|) on +// |deviceIndex|. +// +// Two-call idiom: pFlags == nullptr writes the number of entries to *pCount; +// otherwise at most *pCount entries are written, *pCount is set to the number +// written, and VK_INCOMPLETE is returned if any entry was dropped. +// +// *pCount IS ALWAYS WRITTEN, and this rule is shared by all three list +// enumerators -- this one, VkEncEnumerateInputFormats and +// VkEncEnumerateDrmModifiers. A null pCount is the single exception, being +// the one failure that leaves nowhere to write to; on every other return, +// success or failure, the caller is left a defined count: the number of +// entries written on VK_SUCCESS and VK_INCOMPLETE, and 0 on every error. +// One wrapper can therefore cover all three and read *pCount without +// first having to ask which of them it called. +// +// The count is a property of the context's snapshot, which is immutable after +// construction, so the counting call and the fetching call cannot disagree. +// VK_INCOMPLETE here means the caller passed a smaller *pCount than it was +// told, and never that the answer moved underneath it. +// +// A (codec, profile) pair this library does not probe, or one the driver +// refused, propagates the same result VkEncGetEncodeCapabilities gives for +// that pair and writes a count of 0. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult VkEncEnumerateStdFlags( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t* pCount, + VkVideoEncoderStdFlags* pFlags); + +// The input formats this encoder accepts for (|codec|, |profile|) on +// |deviceIndex|, what each one is encoded as, and how it gets there. +// +// MEMBERSHIP. An entry is present when this library can encode that format on +// THIS device for THIS (codec, profile). Two rules, and they are the whole +// contract: +// +// * a format the encoder itself reads is advertised as +// VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, with encodeFormat == format; +// * a format the encoder does not read, but that the preprocess compute +// filter converts into one it does, is advertised as +// VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL, with encodeFormat naming the +// conversion's output and so what its bitstream is coded from -- and only +// when the device accepts that output as an encode input. A conversion +// whose result this device would not take is not a route, so it is not +// advertised. +// +// WHAT THE LIST IS FOR. It is the set of formats a caller may DECLARE as a +// session's input format and then hand in -- the WHOLE set, and not a +// preference inside a wider one. InitializeExt gates acceptance on this same +// answer, so for this (codec, profile) a format on the list initialises and a +// format the list omits is refused, with the format and its subsampling named. +// The one axis on which absence is not a refusal is the colour model, which +// this list cannot carry: see the paragraph on the packed 4:4:4 layouts below, +// and VkEncQueryInputFormatSupport, which takes one. +// +// A session is narrower than the profile: an OPTIMAL entry is taken by any +// session on this profile, but a SUBOPTIMAL entry is taken only by a session +// that DECLARED that format, because the conversion is built for the one input +// format the session was given. Sizing a producer pool from this list is safe +// precisely because the declaration comes first. +// +// AND A SUBOPTIMAL ENTRY IS TAKEN ONLY ON THE REGISTERED LANE. The conversion +// runs against a REGISTERED resource, so a SUBOPTIMAL format must be handed in +// through RegisterImageResource() followed by SubmitRegisteredFrame(). +// SubmitExternalFrame() refuses it with VK_ERROR_FORMAT_NOT_SUPPORTED: that +// path has no registration to attach a conversion to. An OPTIMAL entry is +// accepted on either lane. This is a qualification of the sentence above, not +// an exception to it -- the format is still one the session may declare; what +// is restricted is which submit carries it. +// +// The list carries no tiling and no modifier: tiling is a property of an +// image rather than of a format, and QueryImageSupport is what answers it for +// a specific allocation. It carries no colour model either -- the packed +// 4:4:4 Y'CbCr layouts have no Vulkan enumerant and ride the RGBA ones -- so +// an aliased enumerant that IS advertised appears here under its RGB reading, +// and its Y'CbCr reading is reached only by declaring it. An enumerant whose +// only accepted reading is the Y'CbCr one does not appear in this list at +// all, so ON THE COLOUR-MODEL AXIS absence from it is not a refusal. That is +// the single exception to "the list is what InitializeExt takes", it exists +// because a list of formats cannot express a declaration about samples, and +// VkEncQueryInputFormatSupport is the surface that closes it. +// +// ORDER AND MULTIPLICITY. Each format appears once, however many tilings the +// device reports it at; the OPTIMAL entries come first and the SUBOPTIMAL ones +// follow, each group in this library's own routable order. +// +// NOT IN THE DEVICE'S ORDER, and there is no longer such an order to give. +// Every candidate is resolved at the profile ITS OWN binding derives -- a +// 4:4:4 input at a 4:4:4 profile, a 10-bit one at a 10-bit profile -- so the +// entries of one list are answered against as many device queries as there are +// distinct derivations, and no single device list orders them. Do not read +// position as preference beyond the OPTIMAL/SUBOPTIMAL split, which is stated +// per entry and is the part that means something. +// +// Two-call idiom and the *pCount rule exactly as VkEncEnumerateStdFlags +// states them: pFormats == nullptr writes the number of entries to *pCount; +// otherwise at most *pCount entries are written, *pCount is set to the number +// written, and VK_INCOMPLETE is returned if any entry was dropped. A pair +// this library does not probe, or one the driver refused, propagates the +// result VkEncGetEncodeCapabilities gives for it and writes a count of 0. +// +// THE STABILITY RULE IS THE SAME AND ITS REASON IS NOT. The counting call and +// the fetching call cannot disagree, so VK_INCOMPLETE here means the caller +// passed a smaller *pCount than it was told and never that the answer moved +// underneath it -- but this list is NOT read out of the context's snapshot. +// It is recomputed live on every call, per candidate, against the device. +// What makes the two calls agree is that the resolve is a deterministic +// function of (context, device, codec, profile) and a context's devices do +// not change for its lifetime, which is the same premise the snapshot itself +// rests on rather than the snapshot itself. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult VkEncEnumerateInputFormats( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t* pCount, + VkVideoEncoderInputFormatProperties* pFormats); + +// DRM modifiers for |format| that carry the format features |usage| implies. +// Answerable WITHOUT a session -- required, because the producer picks a +// modifier at allocation time, long before any encoder exists. +// +// Two-call idiom: pModifiers == nullptr writes the matching count to *pCount; +// otherwise at most *pCount entries are written, *pCount is set to the number +// written, and VK_INCOMPLETE is returned if any match was dropped. *pCount is +// written on every return but a null pCount, as VkEncEnumerateStdFlags +// states for all three enumerators -- so both error returns described below +// leave a count of 0, and so does an unrecognised deviceIndex. +// +// usage == 0 means "no feature filter" and returns every modifier the device +// reports for the format. A usage bit this function has no format-feature +// mapping for is VK_ERROR_INITIALIZATION_FAILED, never a silently dropped +// term: the whole point of the filter is that the caller can trust it. +// +// THIS ANSWER IS PROFILE-BLIND, AND THAT IS A LIMIT ON WHAT IT MEANS. It comes +// from vkGetPhysicalDeviceFormatProperties2, whose query has no video-profile +// term at all -- unlike VkEncEnumerateInputFormats and +// VkEncQueryInputFormatSupport, which are keyed on (codec, profile). So a +// modifier reported here as carrying VIDEO_ENCODE_INPUT is NOT thereby usable +// for the profile the session will negotiate: it says the FORMAT supports the +// feature under that modifier, not that this codec and profile accept that +// format. Ask the input-format query about the (codec, profile, format) triple +// and this one about the modifier, and treat the two answers as conjoined. +// +// Returns VK_ERROR_EXTENSION_NOT_PRESENT when the device cannot answer in +// 64-bit format features (VK_KHR_format_feature_flags2 / Vulkan 1.3). The +// 32-bit modifier list cannot express VK_FORMAT_FEATURE_2_VIDEO_ENCODE_INPUT, +// which is the one feature this entry point exists to filter on, so an +// answer from it would be wrong rather than partial. A device with no +// VK_EXT_image_drm_format_modifier at all reports a count of 0 and +// VK_SUCCESS: no modifiers is a true answer, not a failure. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult VkEncEnumerateDrmModifiers( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkFormat format, + VkImageUsageFlags usage, + uint32_t* pCount, + uint64_t* pModifiers); + + +// Whether this library will take |format|, declared in |colorModel|, as the +// input of a session on (|codec|, |profile|) on |deviceIndex| -- and what the +// bitstream would then be coded from. +// +// A POINT QUESTION, ASKED BEFORE A SESSION EXISTS, which is what separates it +// from everything else here. QueryImageSupport answers for a whole image +// descriptor, but it is a session method: it cannot be reached until a session +// has already been built in the very pair being asked about. This is the +// question a producer has EARLIER than that -- before it allocates a frame +// pool, while changing its mind is still free. +// +// NOT AN ENUMERATOR. No pCount, no two-call idiom, no VK_INCOMPLETE. It +// answers the one pair the caller names and enumerates nothing, and that is +// precisely why it can carry a colour model at all: a declaration is the +// caller's statement about its own samples, so a LIST carrying one would be a +// cross-product of formats and declarations rather than a list of formats. +// +// WHY THE COLOUR MODEL IS A PARAMETER HERE AND AN OMISSION THERE. +// VkEncEnumerateInputFormats says of itself that an enumerant whose only +// accepted reading is the Y'CbCr one "does not appear in this list at all, so +// absence from it is not a refusal". The packed 4:4:4 layouts AYUV and Y410 +// have no Vulkan enumerant of their own and ride the RGBA ones, so that list +// can show neither of them as itself. This is where a caller holding AYUV or +// Y410 frames finds out. +// +// IT ANSWERS WHAT InitializeExt WILL ANSWER, AND BY THE SAME ROUTE, IN BOTH +// HALVES. The library half of the verdict is produced by running the binder +// InitializeExt runs, not by restating its rules -- so a pair reported here as +// supported is a pair that binds, and the profile rules (which profile admits +// which bit depth and which chroma subsampling) are applied in one place only. +// The device half is a live format query against |deviceIndex|, at the chroma +// subsampling and bit depth the input itself implies, and it is the SAME call +// InitializeExt makes before it creates a session. A pair refused here is +// refused there, with the format and its subsampling named. +// +// ONE RESOLVER, THREE SURFACES. VkEncEnumerateInputFormats answers the same +// question for the whole routable set at once, by running this function over +// every candidate; InitializeExt asks it about the one pair a session +// declared. So no two of them can disagree: a format on the advertised list is +// a format this entry point accepts and a format a session initialises with, +// naming the same encodeFormat and the same optimality. Use the list to +// discover, this to ask about one pair -- including one the list cannot carry, +// because the list has no colour-model argument and this does -- and expect +// initialisation to say exactly what they said. +// +// |profile| is the codec standard's own number, and +// VK_VIDEO_ENCODER_PROFILE_DEFAULT means "derive it from the input" exactly as +// VkVideoEncoderConfig::profile does. DEFAULT is the interesting value for a +// 4:4:4 input, because the derivation is what reaches a 4:4:4 profile: a +// profile number this library does not bind is refused here as it is at +// InitializeExt, so naming one is not a way round that. +// +// RETURNS +// VK_SUCCESS +// accepted, and this device encodes it. *pProperties is written. +// VK_ERROR_FORMAT_NOT_SUPPORTED +// the library will not take the pair on this (codec, profile), or this +// device does not encode what the pair would be coded from. +// VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR +// |codec| is not an encode codec this library carries. +// VK_ERROR_INITIALIZATION_FAILED +// |ctx| is null, or |deviceIndex| names no device. +// +// |pProperties| may be NULL when only the verdict is wanted. It is written +// ONLY on VK_SUCCESS, and never partially. +// +// A REFUSAL EXPLAINS ITSELF ON THE LIBRARY'S STDERR, as InitializeExt's does +// and for the same reason -- it is the same binder. A caller sweeping formats +// to size a pool should expect that output; silenceStdio is a session-level +// control and does not reach this call. +extern "C" VK_VIDEO_ENCODER_EXPORT +VkResult VkEncQueryInputFormatSupport( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkFormat format, + VkVideoEncoderColorModel colorModel, + VkVideoEncoderInputFormatProperties* pProperties); + + +#endif /* _VULKAN_VIDEO_ENCODER_EXT_H_ */ diff --git a/vk_video_encoder/internal/vulkan_video_encoder_ext_internal.h b/vk_video_encoder/internal/vulkan_video_encoder_ext_internal.h new file mode 100644 index 00000000..fa7f2c3b --- /dev/null +++ b/vk_video_encoder/internal/vulkan_video_encoder_ext_internal.h @@ -0,0 +1,1697 @@ +/* + * Internal to the encoder library and its tests. NOT part of the public ABI: + * nothing here may be relied on by a consumer. The observation structs below + * carry a structure type because they ride the public pNext chains; that + * type is an internal detail like the rest of this header. + * + * THAT IS A BUILD FACT, NOT A REQUEST. This file lives outside the include + * directory the encoder library exports, and it is not installed. A consumer + * that links the library, or that builds against an install prefix, does not + * have this header on its include path and cannot include it by name. The + * library and the in-tree tests that need it name this directory explicitly; + * that naming is what distinguishes an internal consumer from a client. + * + * Two layers of machinery, described below, that make + * "accepted and ignored" structurally detectable rather than a thing anyone + * has to remember. + * + * Layer 1 is the field table below. Every field of VkVideoEncoderConfig + * appears exactly once with a disposition saying what happens to it. Each + * entry carries an offsetof assertion, so renaming or removing a field fails + * the build here, and each carries the SIZE and the ALIGNMENT of the field it + * names so that the rows can be checked against the struct as a whole and not + * only against one another. + * + * A FIELD WITH NO ROW is the case the offsetof assertions cannot see: those + * assertions are taken FROM rows, so a field nobody wrote a row for is never + * named by one. The sizeof assertion on the struct, and the closing-member + * pins beside it, fail when an addition changes the size or slides a run -- + * but both are PROMPTS rather than proofs. They fail where the pin is, not at + * this table, and a field whose pin was updated and whose row was not written + * still compiles. That is how inputColorModel reached this struct with no row + * here, and nothing failed. + * + * WHAT CLOSES IT is the tiling check, which lives in this tree rather than in + * a consumer: vk_video_encoder/test/encoder-ext-filter, run by ctest as + * EncoderExtInputFormatTaxonomy. Sorted by offset, the rows must lie end to + * end across VkVideoEncoderConfig -- no row starting inside its predecessor, a + * gap before a row legal only while it is STRICTLY narrower than that row's + * alignment, the last row closing the struct, and the gaps totalling + * kVkEncCfgPaddingBytes. Between them those fail for the removal of ANY row in + * this table. + * + * THE ONE CASE NONE OF THEM SEES is a field added into padding that already + * exists: it moves neither the struct's size nor any offset, so there is + * nothing for an arithmetic check to count. Classifying a new field remains a + * decision an author makes; these checks are what put the question in front of + * them. + * + * Layer 3 is the binder conformance suite, which drives + * VkEncBuildAndProbeConfig per codec arm and asserts effect or explicit + * rejection for every BOUND field except outputPath, which has no probe + * projection: its binding is a conditional file-open, and a consumer that + * captures in memory nulls the field. Those assertions are written by hand; + * no test walks this table pairing dispositions with effects, so a new + * BOUND field is covered only once its author adds the assertion. The tests + * that DO iterate the table are the field-table cases of the taxonomy suite + * named above; they check its SHAPE -- exactly-once classification, in-struct + * offsets, and the tiling -- not field effect. A consumer of this header may + * carry a test of that shape too, but a guarantee this header states has to + * be one this library can run, so the in-tree cases are what it cites. + * The binder is exposed as a free function because it touches + * no member state, so the suite drives it with no Vulkan device at all. + */ + +#ifndef VULKAN_VIDEO_ENCODER_EXT_INTERNAL_H_ +#define VULKAN_VIDEO_ENCODER_EXT_INTERNAL_H_ + +#include +#include + +// The public header, which every target allowed to include THIS header +// already has on its include path: the public directory is what the library +// target exports, and this directory is named only by the library and by the +// in-tree tests that reach in here. Deliberately does NOT reach into the +// library's private headers -- see VkEncBoundConfigProbe. +#include "vulkan_video_encoder_ext.h" + +// --------------------------------------------------------------------------- +// INTERNAL OBSERVATION STRUCTS +// +// What the library did with a frame, and what a dma-buf import produced. +// None of it is part of the client contract: a caller encodes without naming +// any of these types, and the routing decisions and driver mitigations they +// report are implementation choices the library is free to change. +// +// They ride the public pNext chains -- VkVideoEncoderCompletionInfo::pNext on +// GetCompletionInfo, and VkVideoEncoderStatus::pNext on +// RegisterImageResource -- so each carries a structure type, taken from a +// band of the public numbering that is assigned to these and never reused. +// The public chain rules apply unchanged: value-initialize the struct so it +// self-stamps, chain at most one link of each type, and expect an unknown or +// repeated link to be refused rather than ignored. +// --------------------------------------------------------------------------- + +constexpr VkVideoEncoderStructureType + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FILTER_INFO = + (VkVideoEncoderStructureType)0x56450019; +constexpr VkVideoEncoderStructureType + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_RESIDENCY_INFO = + (VkVideoEncoderStructureType)0x5645001A; +constexpr VkVideoEncoderStructureType + VK_VIDEO_ENCODER_STRUCTURE_TYPE_STAGED_SUBMIT_INFO = + (VkVideoEncoderStructureType)0x5645001B; +constexpr VkVideoEncoderStructureType + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO = + (VkVideoEncoderStructureType)0x5645001C; +constexpr VkVideoEncoderStructureType + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO = + (VkVideoEncoderStructureType)0x5645001D; + +// Which preprocess conversion the library built for a session. +enum VkVideoEncoderFilterType { + VK_VIDEO_ENCODER_FILTER_TYPE_NONE = 0, + // Any YCbCr -> YCbCr conversion, including 3-plane I420 -> 2-plane NV12 + // (plane-count and bit-depth conversion) and the identity copy. + VK_VIDEO_ENCODER_FILTER_TYPE_YCBCR_COPY = 1, + VK_VIDEO_ENCODER_FILTER_TYPE_RGBA_TO_YCBCR = 2, + VK_VIDEO_ENCODER_FILTER_TYPE_YCBCR_TO_RGBA = 3, +}; + +// Filter dispatch: chain onto VkVideoEncoderCompletionInfo::pNext. +// +// Whether a preprocess conversion ran, and how much of the session took it. +// The session's input format decides whether a filter is BUILT; the route is +// chosen per frame, so a session that has one can still send some or all +// frames down the staging copy. +// +// filterDispatchCount and stagedCopyCount are the two arms of one per-frame +// decision and never both count the same frame, so they are read directly +// rather than by subtracting from a total: +// +// filterCreated == VK_FALSE no filter on this session +// created, dispatch == 0, copy > 0 configured; every frame copied +// created, dispatch > 0, copy == 0 the filter is the path +// dispatch == 0 && copy == 0 nothing reached staging -- a +// zero-copy route bypasses it -- or +// there is no snapshot to take +// +// All four fields are filled only while the session still holds an encoder. +// Read them before Flush() and before the last reference to the encoder goes; +// after either, the snapshot is filterCreated = VK_FALSE with both counts 0, +// which is the last row of the table and is honest but indistinguishable from +// a session that has no filter and never staged a frame. +// +// Counts are cumulative for the life of the encoder and never reset. +struct VkVideoEncoderFilterInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FILTER_INFO; + const void* pNext = nullptr; + + VkBool32 filterCreated; // a filter OBJECT exists + VkVideoEncoderFilterType filterType; // which conversion was built + uint64_t filterDispatchCount; // filter command buffers RECORDED + uint64_t stagedCopyCount; // frames that took the staging copy arm +}; + +// Staging acquire program: chain onto VkVideoEncoderCompletionInfo::pNext. +// +// Which of the two staging-acquire programs a registered external frame took +// -- the VK_QUEUE_FAMILY_FOREIGN_EXT ownership acquire, or the local +// HOST|TRANSFER availability barrier. The library picks one from the declared +// residency and layout. Both are self-consistent, both leave the image in the +// same layout, and neither violates a core VUID, so the counts are what tells +// them apart. +// +// The two counts are the two arms of one per-frame decision and never both +// count the same frame: +// +// foreign > 0, local == 0 every staged frame took the FOREIGN acquire +// foreign == 0, local > 0 every staged frame took the local restore +// both 0 nothing reached the staging tier -- a direct +// zero-copy registration bypasses it -- or +// there is no snapshot to take +// +// COUNTED AT THE ROUTING DECISION, NOT AT THE BARRIER: once for each frame +// whose staging barrier program was chosen and recorded, on either arm. A +// count taken inside the two arms' own release and handback pairs would +// undercount a session that mixes filtered and copied frames, and a superset +// counter can make a live tier read as dead. +// +// EXTERNAL INPUT ONLY. The library's file-input lane declares no residency +// and is not counted here; counting it would make localAcquireCount equal the +// frame count on a session that registered nothing. +// +// Filled only while the session still holds an encoder, so read before +// Flush() and before the last encoder reference goes. DrainPendingFrames() +// does not clear it and is the right place to read after: it joins the +// encoder threads, so every submitted frame has been routed by the time it +// returns. +// +// Counts are cumulative for the life of the encoder and never reset. +struct VkVideoEncoderInputResidencyInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_RESIDENCY_INFO; + const void* pNext = nullptr; + + // Frames whose staging acquire named VK_QUEUE_FAMILY_FOREIGN_EXT as its + // source family (an external allocator or queue family owns the memory). + uint64_t foreignAcquireCount; + // Frames whose staging acquire was a same-family availability barrier + // and whose handback RESTORED the layout instead of releasing ownership. + // Its scopes follow the arm that ran -- transfer for the staging copy, + // compute for the filter. + uint64_t localAcquireCount; +}; + +// Staged-input submit engine: chain onto VkVideoEncoderCompletionInfo::pNext. +// +// Which queue family the staged input work is recorded and submitted on -- +// the acquire, the copy or filter dispatch, the release and the submit alike. +// Both arms of the staging path take their command buffer from one pool, the +// pool is created on the compute family whenever the session has a preprocess +// filter, and a command buffer may only be submitted to a queue of its pool's +// family (VUID-vkQueueSubmit2-commandBuffer-03874). So a frame that +// dispatches no filter still stages on the compute family the moment the +// session has one, without that frame's own format having changed. +// +// The family is more than scheduling. A driver may lose the device executing +// a queue-family RELEASE to VK_QUEUE_FAMILY_FOREIGN_EXT of an image created +// with VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR or ..._DPB_BIT_KHR when the +// barrier is recorded off the graphics or optical-flow families; the usage +// bit is the gate there, not the family alone. The staged copy arm releases +// the CALLER's imported image, whose usage the caller declared, so safety is +// a joint property of this family and that usage and both have to be read. +// Neither family is a validation error, so reporting is the only way to see +// which one ran. +// +// SESSION-CONSTANT, and filled from the same accessors the staging and submit +// sites read rather than re-derived, so a barrier site and a submit site that +// name different families are observable here. Valid before any frame stages. +// +// Filled only while the session still holds an encoder; afterwards it reports +// submitTypeQueueFlags 0 and VK_QUEUE_FAMILY_IGNORED. +struct VkVideoEncoderStagedSubmitInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_STAGED_SUBMIT_INFO; + const void* pNext = nullptr; + + // The raw VK_QUEUE_* bit the staged-input batch is submitted with + // (VK_QUEUE_COMPUTE_BIT 0x2, VK_QUEUE_TRANSFER_BIT 0x4, + // VK_QUEUE_VIDEO_ENCODE_BIT_KHR 0x40). Reported as the flag rather than as + // a library enum so no translation table can drift from the submitted + // value. 0 means there is no session. + uint32_t submitTypeQueueFlags; + // The queue-family index that flag resolves to on this device, i.e. the + // family named as the DESTINATION of the staged FOREIGN acquire and as the + // SOURCE of the staged FOREIGN release. VK_QUEUE_FAMILY_IGNORED means + // there is no session. + uint32_t queueFamilyIndex; +}; + +// --------------------------------------------------------------------------- +// The dma-buf import-ordinal guard, and how to read its report. +// +// The guard mitigates a class of dma-buf import defect in which the imported +// image is bound to memory the exported buffer's contents never reach. It +// holds a fixed number of sacrificial imports on the device ahead of any +// caller-visible one, so caller imports land at later live-positions. That is +// a change of POSITION, not a repair: an import that lands damaged is still +// damaged, and what the buffer actually holds is the question +// VkVideoEncoderImportContentInfo below answers. +// +// THE GUARD IS DISABLED BY DEFAULT: a stock build retains nothing and +// reports DISABLED for every dma-buf import inside the guard's scope. Outside +// that scope the out-of-scope verdict is reported instead, so read |state| +// rather than assuming DISABLED. VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD in +// the environment disables the guard in a build that enables it. +// +// WHERE TO CHAIN VkVideoEncoderImportGuardInfo. +// +// * VkVideoEncoderStatus::pNext on RegisterImageResource -- the verdict for +// THAT registration, delivered by the call that ran the guard. This is +// the one to assert on. +// * VkVideoEncoderCompletionInfo::pNext on GetCompletionInfo -- the most +// recent verdict from a registration the guard evaluated, readable at any +// time. A registration outside the guard's scope leaves this snapshot +// unchanged, so it cannot erase the answer a dma-buf registration +// established. It reads NOT_EVALUATED before the first evaluated one. +// +// THE STATES. |state| is the verdict; requestedCount and retainedCount are +// the arithmetic behind it. requestedCount is a BUILD CONSTANT, filled on +// every path, so a reader compares against it rather than hard-coding it. +// +// STATE MEANING WHAT TO DO +// --------------- ----------------------------- --------------------- +// NOT_EVALUATED The guard did not run: a Nothing to read. Also +// VK_IMAGE registration, or the zero value. +// one refused before the +// import. +// NOT_APPLICABLE Out of the guard's scope: Nothing to read. +// not a DMA_BUF handle, or no +// device to hold a position on. +// NOT_NVIDIA Not a device the guard Nothing to read. +// applies to. +// DISABLED Off deliberately: this build Nothing to read. The +// retains 0, or the kill stock build's answer. +// switch is set. +// COMPLETE retainedCount == Caller imports land at +// requestedCount, on a build live-position +// that asked for a non-zero requestedCount + 1 or +// count. later. +// INCOMPLETE retainedCount < Read failureStatus and +// requestedCount: the shift failureErrno; also in +// asked for did not happen. the diagnostic channel. +// --------------------------------------------------------------------------- +typedef enum VkVideoEncoderImportGuardState { + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED = 0, + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_APPLICABLE = 1, + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_NVIDIA = 2, + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_DISABLED = 3, + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_COMPLETE = 4, + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE = 5, +} VkVideoEncoderImportGuardState; + +typedef struct VkVideoEncoderImportGuardInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO; + const void* pNext = nullptr; // MUST be NULL + + VkVideoEncoderImportGuardState state = + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED; // OUT + // Sacrificial imports this build retains where the guard applies. + // A build constant, filled on every path. 0 in a stock build: the + // guard is disabled by default. + uint32_t requestedCount = 0; // OUT + // Sacrificial imports actually live on the device right now. + uint32_t retainedCount = 0; // OUT + // INCOMPLETE only: what refused the sacrificial import. + // ERROR_IMPORT_FAILED with a non-zero failureErrno means dup(2) + // failed; any other value is the import's own status. + VkVideoEncoderStatusCode failureStatus = + VK_VIDEO_ENCODER_STATUS_SUCCESS; // OUT + int32_t failureErrno = 0; // OUT +} VkVideoEncoderImportGuardInfo; + +// --------------------------------------------------------------------------- +// The imported-buffer content probe, and how to read its verdict. +// +// A dma-buf import can come back bound to memory the producer's writes never +// reach. VkVideoEncoderImportGuardInfo above reports what a mitigation did; +// this reports what the imported buffer actually CONTAINS. +// +// It is a content observation of the FIRST frame each registration serves, +// taken at the staged copy. A registration that would otherwise encode +// directly sends that one frame through the staged path and every later +// frame direct, so the cost is one detoured frame per registration. It is +// not a repair and not a prediction, and it cannot see a buffer that has not +// yet carried a frame. The reaction to a damaged verdict is to stop using +// that registration. +// +// THE PREDICATE. Let meanY, meanU and meanV be the plane means of the +// imported frame, in 0..255: +// +// DAMAGED_ALL <= (meanY < 2) && (meanU < 2) && (meanV < 2) +// DAMAGED_CHROMA <= (meanY >= 2) && ((meanU < 2) || (meanV < 2)) +// CLEAN <= neither +// +// Luma and chroma are scored together rather than chroma alone, so an +// all-zero buffer and a chroma-zeroed one stay distinguishable. +// +// ZEROED IS NOT BLACK, which is what makes the test sound: legal black in +// NV12 is Y=16 (0 in full range) with U=V=128, so a zero chroma plane is a +// value no correct encoder input carries. The false-positive budget is a +// frame that is deliberately all-zero in every plane, which from inside the +// library is indistinguishable from the defect; the cost is one frame per +// registration, and the reaction to it is not destructive. +// +// WHEN THE VERDICT EXISTS. Not at registration -- the producer has written +// nothing yet, so there is nothing to score. The registration echo reports +// ARMED or NOT_APPLICABLE; the verdict arrives on a later GetCompletionInfo +// snapshot, once the first frame of that registration has been submitted and +// its fence waited. Frames submitted in the meantime encode against the +// buffer and cannot be recalled, because the library does not recall +// submitted GPU work (see CancelFrame): roughly one pipeline depth of frames +// is the price of scoring content that only exists once it is written. +// +// WHERE TO CHAIN IT. Exactly where VkVideoEncoderImportGuardInfo chains, and +// the two are independent -- either, both, or neither. +// +// * VkVideoEncoderStatus::pNext on RegisterImageResource. CHAINING IT HERE +// IS THE OPT-IN: a registration whose status carries this struct arms the +// probe; one that does not is never probed and pays nothing. There is no +// environment variable and no build flag. The value read back on a +// successful registration is ARMED or NOT_APPLICABLE, and +// NOT_EVALUATED on a registration this call refused -- never a verdict. +// * VkVideoEncoderCompletionInfo::pNext on GetCompletionInfo -- the verdict +// channel, readable at any time from any thread. It reports the OLDEST +// still-registered DAMAGED_* registration, so retiring that one exposes +// the next on the following poll and not reacting re-reports the same +// one: idempotent either way, and no verdict is lost between polls. With +// none damaged it reports the most recent CLEAN verdict, or NOT_EVALUATED +// before the first frame is scored. +// +// THE STATES. +// +// STATE MEANING WHAT TO DO +// --------------- ----------------------------- --------------------- +// NOT_EVALUATED No verdict yet: nothing Poll again later. Also +// armed, or nothing armed has the zero value. +// completed a frame. +// NOT_APPLICABLE This registration cannot be No verdict will come. +// scored. +// ARMED Set up, waiting for a frame. Expect a verdict on a +// A registration echo reports later snapshot. +// CLEAN Scored; the predicate did Keep using the +// not fire. registration. +// DAMAGED_CHROMA Scored: chroma dead, luma Stop using the +// alive. registration. +// DAMAGED_ALL Scored: every plane dead. Stop using it. +// +// NOT_APPLICABLE is usually decided at registration and arrives in the echo: +// the registration is FILTER-routed, which is a storage read and never a +// copy; the import carries no TRANSFER_SRC, so no copy may legally be +// recorded out of it; or the format is not 8-bit 2-plane 420, which the +// predicate needs in order to have a Y, a U and a V to score. It can also +// be latched on the first capture attempt -- the extent is degenerate, or +// the capture pool cannot serve that format and extent. An ARMED echo is +// therefore a verdict PENDING and not a verdict promised; +// armedRegistrationCount tells one still in flight from one that will never +// arrive. +// +// probeGeneration is a non-zero BUILD CONSTANT +// (VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION) stamped on every path, +// so probeGeneration != 0 is proof the library wrote the struct. +// +// meanY / meanU / meanV are the plane means in Q8 FIXED POINT -- the 0..255 +// mean times 256, so 128.0 reads as 32768 and the predicate's "< 2" is +// "< 512". Integers rather than floats because this struct crosses a process +// boundary. They are filled on CLEAN and DAMAGED_* alike, so a reader can +// log why. +// --------------------------------------------------------------------------- + +// Non-zero by contract: a non-zero probeGeneration is what proves the +// library wrote the struct at all. Bump it if the predicate or the sampling +// changes in a way that makes old and new verdicts non-comparable. +#define VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION 1u + +// The predicate's threshold, in the Q8 units meanY/meanU/meanV carry: a plane +// mean strictly below 2.0/255. Named rather than open-coded because the +// library's scorer and every assertion on it have to agree. +#define VK_VIDEO_ENCODER_IMPORT_CONTENT_DEAD_PLANE_MEAN_Q8 512u + +typedef enum VkVideoEncoderImportContentState { + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED = 0, + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_APPLICABLE = 1, + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED = 2, + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_CLEAN = 3, + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_CHROMA = 4, + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_ALL = 5, +} VkVideoEncoderImportContentState; + +typedef struct VkVideoEncoderImportContentInfo { + VkVideoEncoderStructureType sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO; + const void* pNext = nullptr; // MUST be NULL + + // Which registration this verdict belongs to. + // VK_VIDEO_ENCODER_RESOURCE_NULL when there is no verdict + // (NOT_EVALUATED), and on the registration echo, where the resource id is + // the call's own return value. + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; // OUT + VkVideoEncoderImportContentState state = + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED; // OUT + // Non-zero build constant, stamped on every path. The writer proof. + uint32_t probeGeneration = 0; // OUT + // Q8 fixed point: the 0..255 plane mean times 256. + uint32_t meanY = 0; // OUT + uint32_t meanU = 0; // OUT + uint32_t meanV = 0; // OUT + // Session totals: registrations that reached a CLEAN or DAMAGED_* + // verdict, and how many of those were DAMAGED_*. + uint32_t probedRegistrationCount = 0; // OUT + uint32_t damagedRegistrationCount = 0; // OUT + // REGISTRATIONS STILL WAITING FOR A VERDICT -- armed, not yet scored. + // + // READ THIS BEFORE BELIEVING damagedRegistrationCount == 0. The two + // counts above cannot distinguish "every buffer was probed and every one + // was clean" from "nothing was ever probed", because both report + // probed=0 damaged=0 when no capture ever ran. This field is what tells + // them apart: non-zero at the end of a session means that many buffers + // were promised a verdict and never got one, so the absence of damage + // reports is an absence of MEASUREMENT, not an absence of damage. + // + // Expected to be non-zero TRANSIENTLY -- a registration is armed at + // import and scored a frame or two later, so a mid-session poll legit- + // imately catches buffers in flight. It is a session that ENDS with this + // non-zero, or a long-running session where it never falls, that means + // the capture site is not being reached. + uint32_t armedRegistrationCount = 0; // OUT +} VkVideoEncoderImportContentInfo; + +// Disposition of a public config field. +// +// BOUND -- forwarded into EncoderConfig by the binder. The binder suite +// asserts the landing for every BOUND field except outputPath +// (no probe projection; see the file comment above). +// SESSION -- consumed by the ext layer itself (device/instance selection, +// stdio silencing, codec dispatch). Never reaches EncoderConfig, +// and must not: it shapes the session, not the encode. +// ABI_GATE -- consumed by the versioning gate before anything else runs. +// VALIDATED-- read as a DECLARATION and checked against the rest of the +// config; refused at init when it cannot be honoured. Reaches +// EncoderConfig nowhere, because there is nothing to forward: +// the field states a requirement, it does not set a knob. +enum VkVideoEncoderConfigFieldDisposition { + VK_ENC_FIELD_BOUND = 0, + VK_ENC_FIELD_SESSION, + VK_ENC_FIELD_ABI_GATE, + // NO "REJECTED" DISPOSITION, AND ITS ABSENCE IS THE DECISION. There was + // one, with a single definition and no row anywhere in the table below: + // it advertised a class -- "a non-default value is refused at init because + // honouring it is impossible in THIS BUILD" -- that this config surface + // does not contain. Every refusal the surface makes is either a VALIDATED + // declaration that cannot be honoured or an ABI_GATE, and both are + // properties of the config rather than of the build. Deleted rather than + // given a row: inventing a member to justify an enumerant is how a table + // stops describing the thing it tables. + VK_ENC_FIELD_VALIDATED, +}; + +// X(field, disposition, note) +#define VK_VIDEO_ENCODER_CONFIG_FIELDS(X) \ + X(sType, ABI_GATE, "structure-type gate") \ + X(pNext, ABI_GATE, "extension chain, walked at init") \ + X(codec, SESSION, "selects the codec config subclass") \ + X(profile, BOUND, "the codec standard's own profile " \ + "number; a number this library " \ + "cannot bind, or one the input " \ + "depth does not admit, is refused") \ + X(encodeWidth, BOUND, "cfg->encodeWidth") \ + X(encodeHeight, BOUND, "cfg->encodeHeight") \ + X(inputFormat, BOUND, "validated, then cfg->input.bpp, " \ + "cfg->input.chromaSubsampling " \ + "and cfg->input.numPlanes -- the " \ + "subsampling is what makes a " \ + "4:4:4 or 4:2:2 encode profile " \ + "reachable at all") \ + X(inputColorModel, BOUND, "with inputFormat, the pair a " \ + "session is declared in; " \ + "resolved to " \ + "cfg->input.colorSpace, which " \ + "selects the input plane count " \ + "and subsampling. A " \ + "declaration the format cannot " \ + "carry is refused at init") \ + X(inputWidth, BOUND, "cfg->input.width") \ + X(inputHeight, BOUND, "cfg->input.height") \ + X(rateControlMode, BOUND, "cfg->rateControlMode") \ + X(averageBitrate, BOUND, "cfg->averageBitrate") \ + X(maxBitrate, BOUND, "cfg->maxBitrate") \ + X(vbvBufferSize, BOUND, "cfg->vbvBufferSize") \ + X(constQpI, BOUND, "cfg->constQp.qpIntra " \ + "+ constQpSet") \ + X(constQpP, BOUND, "cfg->constQp.qpInterP " \ + "+ constQpSet") \ + X(constQpB, BOUND, "cfg->constQp.qpInterB " \ + "+ constQpSet") \ + X(minQp, BOUND, "cfg->minQp + minQpSet (H.26x); " \ + "rejected on AV1 (qIndex units)") \ + X(maxQp, BOUND, "cfg->maxQp + maxQpSet (H.26x); " \ + "rejected on AV1 (qIndex units)") \ + X(gopLength, BOUND, "gopStructure.SetGopFrameCount") \ + X(consecutiveBFrames, BOUND, "SetConsecutiveBFrameCount") \ + X(idrPeriod, BOUND, "gopStructure.SetIdrPeriod") \ + X(closedGop, BOUND, "gopStructure.SetClosedGop") \ + X(frameRateNum, BOUND, "cfg->frameRateNumerator") \ + X(frameRateDen, BOUND, "cfg->frameRateDenominator") \ + X(qualityLevel, BOUND, "cfg->qualityLevel") \ + X(tuningMode, BOUND, "cfg->tuningMode, validated") \ + X(colourPrimaries, BOUND, "cfg->colour_primaries (VUI)") \ + X(transferCharacteristics, BOUND, "cfg->transfer_characteristics") \ + X(matrixCoefficients, BOUND, "cfg->matrix_coefficients") \ + X(videoFullRange, BOUND, "cfg->video_full_range_flag; also " \ + "VALIDATED against a chained " \ + "VkVideoEncoderInputColourInfo::inputRange -- " \ + "VK_TRUE over a LIMITED Y'CbCr input is a " \ + "range conversion this library does not " \ + "perform and is refused at init") \ + X(inputTransferCharacteristics, VALIDATED, \ + "checked against transferCharacteristics; a " \ + "declared mismatch is refused at init. " \ + "Reaches EncoderConfig nowhere: this library " \ + "applies no transfer function, so there is " \ + "nothing to forward") \ + X(deviceId, SESSION, "physical-device selection") \ + X(gpuUUID, SESSION, "physical-device selection") \ + X(outputPath, BOUND, "cfg->outputFileHandler") \ + X(verbose, BOUND, "cfg->verbose") \ + X(validate, BOUND, "cfg->validate") \ + X(disableFileOutput, BOUND, "cfg->disableFileOutput") \ + X(silenceStdio, SESSION, "VkEncoderStdioSilenceScope") \ + X(externalInstance, SESSION, "caller-supplied instance") \ + X(externalPhysicalDevice, SESSION, "caller-supplied physical device") \ + X(externalDevice, SESSION, "caller-supplied logical device") \ + X(externalEncodeQueueFamilyIndex, SESSION, \ + "honoured only with externalDevice") \ + X(externalComputeQueueFamilyIndex, SESSION, \ + "honoured only with externalDevice") + +// Stable index per field. No test iterates by these indices today; the one +// live consumer is kVkEncCfgFieldCount, which pins the table length (the +// static_assert below, re-checked at run time by the field-table cases in +// vk_video_encoder/test/encoder-ext-filter). +#define VK_ENC_FIELD_ENUM(field, disposition, note) kVkEncCfgField_##field, +enum VkVideoEncoderConfigFieldId { + VK_VIDEO_ENCODER_CONFIG_FIELDS(VK_ENC_FIELD_ENUM) + kVkEncCfgFieldCount +}; +#undef VK_ENC_FIELD_ENUM + +struct VkVideoEncoderConfigFieldInfo { + const char* name; + VkVideoEncoderConfigFieldDisposition disposition; + const char* note; + size_t offset; + // The EXTENT of the field this row names. An offset alone says where a row + // starts and nothing about what it covers, so offsets alone can be checked + // only against each other; with the extent, the rows can be laid end to + // end and compared against the struct they claim to describe. + // + // DERIVED from the member, never written down beside it: a hand-copied + // width is one more thing that can go stale, and a stale one would make + // the tiling check agree with a layout the compiler does not have. + size_t size; + size_t align; +}; + +// Every field's existence is asserted by taking its offset: a rename or a +// removal stops compiling here, which is the point. +#define VK_ENC_FIELD_INFO(field, disposition, note) \ + {#field, VK_ENC_FIELD_##disposition, note, \ + offsetof(VkVideoEncoderConfig, field), \ + sizeof(VkVideoEncoderConfig::field), \ + alignof(decltype(VkVideoEncoderConfig::field))}, +static const VkVideoEncoderConfigFieldInfo kVkVideoEncoderConfigFields[] = { + VK_VIDEO_ENCODER_CONFIG_FIELDS(VK_ENC_FIELD_INFO) +}; +#undef VK_ENC_FIELD_INFO + +static_assert(sizeof(kVkVideoEncoderConfigFields) / + sizeof(kVkVideoEncoderConfigFields[0]) == + (size_t)kVkEncCfgFieldCount, + "field table and field enum disagree"); + +// The bytes of VkVideoEncoderConfig that no field occupies: alignment padding +// between the rows above, once they are laid out in offset order. Four gaps +// carry all of it today -- 4 before pNext, 1 before videoFullRange, 3 before +// deviceId, 4 before outputPath. +// +// WHY A WRITTEN-DOWN NUMBER, as the struct's size is. The per-gap rule on its +// own -- a gap is legal while it is narrower than the alignment of the member +// that follows it -- still passes when the removed row sat in front of a +// WIDELY aligned successor, because the bytes it freed fit inside slack that +// successor already had. Two rows in this table are of exactly that shape: +// silenceStdio, four bytes ahead of an eight-aligned pointer, and +// matrixCoefficients, one byte ahead of a four-aligned VkBool32. Pinning the +// total is what catches those two -- the freed bytes have to surface +// somewhere, and once the total is pinned, here is where they surface. +// +// Move it only alongside the layout change that made it true. +constexpr size_t kVkEncCfgPaddingBytes = 12; + +// A flat projection of everything the binder is supposed to have written. +// The binder suite asserts on this rather than on EncoderConfig, so the +// tests depend on the binding CONTRACT and not on the library's internal +// layout -- and the library keeps its private headers private. +struct VkEncBoundConfigProbe { + uint32_t encodeWidth; + uint32_t encodeHeight; + uint32_t inputWidth; + uint32_t inputHeight; + uint32_t inputBpp; + uint32_t rateControlMode; + uint32_t averageBitrate; + uint32_t maxBitrate; + uint32_t vbvBufferSize; + uint32_t constQpIntra; + uint32_t constQpInterP; + uint32_t constQpInterB; + int32_t minQp; + int32_t maxQp; + uint32_t minQpSet; + uint32_t maxQpSet; + uint32_t constQpSet; + uint32_t gopFrameCount; + uint32_t idrPeriod; + uint32_t consecutiveBFrames; + uint32_t closedGop; + uint32_t frameRateNumerator; + uint32_t frameRateDenominator; + uint32_t qualityLevel; + uint32_t tuningMode; + uint32_t colourPrimaries; + uint32_t transferCharacteristics; + uint32_t matrixCoefficients; + uint32_t videoFullRangeFlag; + uint32_t colorDescriptionPresent; + uint32_t videoSignalTypePresent; + // Chroma siting, as the H.26x VUI will carry it. Projected because it is + // the ONLY observable of the preprocess filter's 2x2 box average outside + // a decoded picture. "Present" and "type" have to be readable + // separately, or a raised flag advertising type 0 looks identical to no + // signal at all. + uint32_t chromaLocInfoPresent; + uint32_t chromaSampleLocType; + // The INPUT side, as the chained VkVideoEncoderInputColourInfo landed in + // EncoderConfig. It reaches the config through a pNext walk rather than a + // config field, so nothing in the flat field table above can see whether + // it bound. inputColourChainPresent is separate from the value fields for + // the same reason av1ColorConfigPresent is separate from + // av1ColorDescriptionPresent: 0 is UNDECLARED on every axis, so no value + // field can tell "absent" from "present and zero". + uint32_t inputColourChainPresent; + uint32_t inputColourPrimaries; + uint32_t inputTransferCharacteristics; + uint32_t inputMatrixCoefficients; + uint32_t inputRange; + uint32_t verbose; + uint32_t validate; + uint32_t disableFileOutput; + // EncoderConfig::enablePreprocessComputeFilter, not a public config + // field: it is where the library's own preprocess-conversion decision + // lands. Projected so that decision is assertable in BOTH directions -- + // a directly encodable input must write 0 here rather than inheriting + // EncoderConfig's default of true (nothing else in the probe would + // notice a filter object created behind the caller's back), and an input + // that is encodable only after a conversion must write 1. + uint32_t preprocessComputeFilter; + // EncoderConfig::input.numPlanes. EncoderConfig does not store the input + // format: it RECONSTRUCTS input.vkFormat from subsampling, bit depth and + // this count. Left unwritten the count inherits EncoderConfig's default of + // 3, which would describe every session's input as 3-plane I420. + // + // THIS IS NOT THE ONLY ROUTE TO "inputFormat WAS BOUND": inputVkFormat below + // projects the reconstruction itself. The + // count is still projected, and separately, because the two answer + // different questions: this one is an INPUT to the reverse derivation and + // that one is its OUTPUT, and a test that reads only the output cannot say + // which of the three terms was wrong when it disagrees. + uint32_t inputNumPlanes; + // EncoderConfig::input.chromaSubsampling, the other half of what + // inputFormat is read for. Projected for the same reason as the plane + // count: the format is not stored, so the derivation is only assertable + // through what it wrote. It is the field a 4:4:4 or 4:2:2 input has to + // change -- left at its 4:2:0 default, a 4:4:4 request encodes as 4:2:0 + // and reports success -- and it is what the codec arm derives the encode + // profile from. Carries the VkVideoChromaSubsamplingFlagBitsKHR value. + uint32_t inputChromaSubsampling; + // EncoderConfig::input.vkFormat AS IT STANDS AFTER InitializeParameters, + // which is the OUTPUT of the reverse derivation and the one quantity that + // says whether the library's two derivations of the input's identity + // agree. + // + // THERE ARE TWO OF THEM, IN OPPOSITE DIRECTIONS. The binder derives + // (chroma subsampling, bit depth, plane count) from the caller's + // VkFormat; EncoderInputImageParameters::VerifyInputs() reconstructs a + // VkFormat from those same three. The reverse one is LOAD-BEARING and + // cannot be deleted: the packed-alias arm deliberately leaves vkFormat + // unwritten so the reconstruction supplies it, which is the only route by + // which AYUV and Y410 are nameable at all. So the two have to agree, and + // this is what a test reads to say that they did. + // + // ON THE RGBA LANE THE REVERSE DERIVATION DOES NOT RUN -- VerifyInputs + // carries the caller's format through, because CodecGetVkFormat spells no + // RGB layout -- so this field projects the carry-through there. It is the + // same proposition either way: the config's idea of the input format is + // the caller's. + uint32_t inputVkFormat; + // THE OTHER SIDE OF THE SAME BOUNDARY: EncoderConfig's encode-side + // geometry, which describes the BITSTREAM where the input fields above + // describe the caller's buffer. + // + // WHY BOTH SIDES ARE PROJECTED WHEN ONE WRITER MAKES THEM EQUAL. They are + // separate fields precisely so that the encode value can differ from the + // input value -- a chroma resampler or a device-driven depth downgrade is + // what would make them -- and every codec arm's profile derivation and + // every syntax element that states the coded format is a function of THIS + // side. A test that could see only one side could not say which side an + // arm had read, which is how three arms came to read different ones. + uint32_t encodeChromaSubsampling; + uint32_t encodeBitDepthLuma; + uint32_t encodeBitDepthChroma; + + // Per-codec-arm effect projections, run PER CODEC ARM. A projection + // that stops at the shared EncoderConfig members cannot see + // codec-conditional consumption: a field can reach the base config and + // still have no effect on the arm that encodes. These project what each + // arm actually hands the driver. + // Zero when the arm was not exercised. + // + // H.26x: the rate-control layer info the codec arm builds. useMinQp / + // useMaxQp are what make the clamp values legally visible to the driver. + // What the arm's InitVuiParameters() ACTUALLY produced, as opposed to + // what the shared EncoderConfig holds. An arm may write a VUI field from + // the config and then overwrite it before the parameter set is built, + // which a config-level projection cannot see and no encode row that + // signals no siting would catch. + uint32_t vuiChromaLocInfoPresent; + uint32_t vuiChromaSampleLocTypeTop; + uint32_t vuiChromaSampleLocTypeBottom; + uint32_t rcUseMinQp; + uint32_t rcUseMaxQp; + int32_t rcMinQpI; + int32_t rcMaxQpI; + // H.265 only: EncoderConfigH265::GetCpbVclFactor()'s result, the quantity + // ITU-T H.265 Table A.8 states. Projected because it is what a chroma-flag + // / chroma_format_idc confusion silently gets wrong, and because every + // downstream observable of it is ALSO a function of the level or tier, so + // none of them reads the factor on its own. + // + // READ IN THE BINDER'S STATE, which is the state InitProfileLevel reads -- + // after InitializeParameters and before InitVideoProfile. The reading point + // matters: the depth term reads encodeBitDepthLuma / encodeBitDepthChroma, and + // deriving those in InitVideoProfile puts the derivation at session creation, + // AFTER level selection -- so the level-selection call site would read zero + // while InitRateControl read the real depth, from one function inside one + // configuration. The derivation belongs in InitializeParameters, beside + // encodeChromaSubsampling, so both call sites read the same value and this + // projection reads it too. Zero on the other arms. + uint32_t h265CpbVclFactor; + // H.265 only: EncoderConfigH265::levelIdc as InitProfileLevel() selected + // it, which is 30 x the level number. The factor's one DEVICE-FREE + // downstream observable -- the default vbvBufferSize is not, because the + // probe reads the config field and InitRateControl, which computes the + // default from the factor, runs later and needs a session. A too-high + // level is a legal level, which is why nothing caught the factor being + // wrong; pinning it is what makes the correction visible downstream of the + // arithmetic rather than only inside it. Zero on the other arms. + uint32_t h265LevelIdc; + // H.265 only: EncoderConfigH265::general_tier_flag, the OTHER half of what + // DetermineLevelTier() picked. It is the term that actually moves with the + // factor at 1080p: when main tier's bitrate ceiling (maxBitRateMainTier x + // cpbVclFactor) is exceeded the selection does not climb to the next + // level, it takes HIGH TIER at the same one -- so a level-only projection + // reads the same number on a right and a wrong factor. Zero on the other + // arms, which is also main tier, so this field is read together with + // h265LevelIdc and not alone. + uint32_t h265GeneralTierFlag; + // AV1: the sequence-header colour config the arm attaches (AV1's + // counterpart of the H.26x VUI colour description). + uint32_t av1ColorDescriptionPresent; + uint32_t av1ColorPrimaries; + uint32_t av1TransferCharacteristics; + uint32_t av1MatrixCoefficients; + uint32_t av1ColorRange; + uint32_t av1BitDepth; + // Whether the arm attached a colour config AT ALL (pColorConfig != + // nullptr). Distinct from av1ColorDescriptionPresent, and the distinction + // is the defect: AV1's color_config carries color_range, BitDepth and + // subsampling as well as the colour description, so a caller that + // declared only full range needs the STRUCT even though the description + // flag stays 0. Gating the whole struct on that flag would drop the + // range on AV1 alone, and no field below can tell "absent" from + // "present and zero". + uint32_t av1ColorConfigPresent; + uint32_t av1ChromaSamplePosition; + // The sequence header's STRUCTURAL subsampling, as the arm wrote it. + // Projected because its correctness is decided by a DIFFERENT function + // (InitProfileLevel, which picks seq_profile from the same input) than the + // one that writes it, and no device this project can obtain reaches the arm + // where the two disagree: AV1 High and Professional are absent from every + // driver available here, so a 4:4:4 or 4:2:2 AV1 session dies at the + // capability query before a sequence header exists. Device-free is the only + // place this fact is assertable at all. + uint32_t av1SubsamplingX; + uint32_t av1SubsamplingY; + // HDR10 static metadata, as the chained VkVideoEncoderHdrMetadataInfo + // landed in EncoderConfig. It reaches the config through a pNext walk + // rather than a config field, so nothing in the flat field table above + // can see whether it bound. + uint32_t hdrMasteringPresent; + uint32_t hdrContentLightPresent; + uint32_t hdrMaxDisplayMasteringLuminance; + uint32_t hdrMinDisplayMasteringLuminance; + uint32_t hdrMaxContentLightLevel; + uint32_t hdrMaxFrameAverageLightLevel; + // displayPrimaryX[0] / displayPrimaryY[0], i.e. ST 2086's GREEN. One + // pair is enough to catch a permuted or dropped array and keeps the + // projection from turning into a second copy of the struct. + uint32_t hdrGreenPrimaryX; + uint32_t hdrGreenPrimaryY; + // All arms: the codec-typed config's profile, read through the same + // virtual accessor session creation consumes (GetCodecProfile, read by + // EncoderConfig::InitVideoProfile), so this projects the profile the + // arm actually encodes with -- H.264 SPS profile_idc, H.265 PTL + // general_profile_idc, AV1 seq_profile, in the codec's own Std enum + // values. An explicit caller profile must be visible here or init must + // have failed. NOTE: STD_VIDEO_AV1_PROFILE_MAIN == 0, so on the AV1 + // arm 0 is a real value, not "arm not exercised". + uint32_t codecProfile; +}; + +// Run the binder and project the result. Reads only its arguments: no device, +// no instance, no encoder -- which is what lets the binder suite assert every +// BOUND field but outputPath from a plain gtest process. +// +// |requestedEncodeBitDepth| IS WHAT MAKES THE TWO GEOMETRIES SEPARABLE, and +// it exists for one reason. EncoderConfig::InitializeParameters derives the +// encode side from the input side under a zero-means-unset guard, and states +// beside that guard that "an explicit encode depth, if one is ever set before +// this runs, is a request and not a default". Until this parameter there was +// no way to set one, so the guard had a rationale and no mechanism -- and, +// more to the point, the encode and input sides were EQUAL ON EVERY REACHABLE +// STATE. A test written against equal values cannot say which of them a codec +// arm read: every assertion it makes is satisfied identically either way, so +// it is a guard against a wrong DERIVATION and no guard at all against a +// wrong SIDE. That is exactly the shape three arms regressed into once. +// +// Non-zero, it writes encodeBitDepthLuma before InitializeParameters runs, so +// the guard leaves it alone and the encode side differs from the input side +// on the depth axis. The profile a codec arm then derives says which side it +// read. Zero -- the default -- is the ordinary path and changes nothing, so +// every existing caller of the three-argument form is unaffected. +// +// IT IS NOT A BACK DOOR ONTO THE PUBLIC SURFACE. VkVideoEncoderConfig has no +// encode-depth field and this parameter reaches no public entry point; it is +// this header's, and this header is the internal one. +VkResult VkEncBuildAndProbeConfig(const VkVideoEncoderConfig& extConfig, + VkVideoCodecOperationFlagBitsKHR codecOp, + VkEncBoundConfigProbe* outProbe, + uint32_t requestedEncodeBitDepth = 0); + +// Byte-exact projection of the HDR10 payload builders. +// +// The two builders live below the ext layer (VkVideoEncoderHdrMetadata.h, +// which this header deliberately does not name -- the same rule that keeps +// EncoderConfig out of it), and the only production caller is inside a codec +// arm that needs a device and a driver-written parameter-set buffer. This +// wrapper takes the PUBLIC struct and hands back the bytes, so the payload +// can be pinned from a device-free test. +// +// That matters most for AV1, whose end-to-end path needs a device with AV1 +// encode; the OBU bytes can be pinned without one. +// +// |codecOp| selects H.265 (a prefix SEI NAL, start code included) or AV1 +// (metadata OBUs). Returns the byte count, or 0 if there was nothing to +// build or it did not fit. +uint32_t VkEncBuildHdrMetadataPayload(const VkVideoEncoderHdrMetadataInfo* info, + VkVideoCodecOperationFlagBitsKHR codecOp, + uint8_t* out, uint32_t capacity); + +// --------------------------------------------------------------------------- +// Input taxonomy. +// +// Adapting a caller's input to what the device encodes has three rungs -- +// hardware conversion, the compute filter, a transfer copy -- and the input's +// DECLARATION decides which of them are candidates: +// +// ENCODABLE_DIRECT the device takes this input as an encode source as +// it stands. Rung 1. +// ENCODABLE_VIA_FILTER rung 2, and rung 2 ONLY -- a transfer copy is not a +// substitute. The 3-plane family differs from the +// semi-planar encode format by PLANE COUNT, and a copy +// cannot drop or merge a plane; the packed 4:4:4 +// Y'CbCr layouts declared as such (AYUV, Y410) differ +// by plane count the other way, one interleaved plane +// against two; the 8-bit RGBA family +// (R8G8B8A8_UNORM, B8G8R8A8_UNORM, +// A8B8G8R8_UNORM_PACK32) differs by COLOUR MODEL, and +// a copy cannot convert colour at all. Both the RGBA +// family and the packed layouts are single-plane, so +// plane count does not follow from the class. +// UNSUPPORTED neither, in this build. +// +// THE DECLARATION, NOT THE FORMAT. A VkFormat names a component layout, and +// the packed 4:4:4 Y'CbCr layouts share their enumerants with RGBA, so the +// format alone cannot place them. VkVideoEncoderColorModel is what the caller +// states and what this function reads; VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT +// asks for the format's own answer and is what every caller that has no packed +// input passes. A declaration the format cannot carry is UNSUPPORTED, never +// silently reconciled. +// +// A property of the DECLARATION alone, so this is a free function and a test +// can drive every arm with no device. Whether a given SESSION can take an +// ENCODABLE_VIA_FILTER input is a second question -- the compute filter has +// to be compiled in, enabled, and able to read the specific image -- and it +// is answered by VulkanVideoEncoderExtImpl::SupportsFormat / +// ValidateImageDescriptor, which hold the session state this function +// deliberately does not. +enum VkEncInputFormatClass { + VK_ENC_INPUT_FORMAT_UNSUPPORTED = 0, + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, +}; +VkEncInputFormatClass VkEncClassifyInput(VkFormat inputFormat, + VkVideoEncoderColorModel colorModel); + +// "Could this library take this input on SOME path?" -- i.e. classified as +// anything but UNSUPPORTED. +// +// Not used inside the library: every gate here needs the finer answer, which +// is the class itself. It remains for an out-of-tree consumer that has the +// coarse question to ask. +VkBool32 VkEncSupportsInput(VkFormat inputFormat, + VkVideoEncoderColorModel colorModel); + +// The number of planes |inputFormat| is laid out in, which EncoderConfig +// needs in order to reconstruct input.vkFormat at all: it does not store the +// input format, it DERIVES it from chroma subsampling, bit depth and +// numPlanes. 0 for a format this library does not classify. +// +// Takes no colour model, and that is a property of the question rather than +// an omission: plane count is a fact about the LAYOUT, and the packed 4:4:4 +// layouts are one plane whichever model is declared over them. +uint32_t VkEncInputFormatPlaneCount(VkFormat inputFormat); + +// Every input format this library routes, in a fixed order. The +// advertisement walks this list; the classifier answers what each one is. +// +// DERIVED, NOT LISTED. The set is computed from the multi-planar Y'CbCr +// format table and the packed 4:4:4 table -- the same two tables the compute +// filter's shader generator reads -- so the list and the classifier cannot +// disagree: both are the same predicate over the same rows. |outCount| is +// therefore the derived count and is not a constant of this header; +// VK_ENC_MAX_ROUTABLE_INPUT_FORMATS bounds it. +// +// Built once, on first call, and immutable afterwards. +const VkFormat* VkEncRoutableInputFormats(uint32_t& outCount); + +// The encoder-input format |inputFormat| is converted INTO, given the +// formats the device accepts for the profile in question. +// +// For a Y'CbCr input the answer preserves the input's own chroma subsampling +// and bit depth and names the two-plane semi-planar form: the compute filter +// converts plane layout and packing, and resamples neither chroma nor depth, +// so a target that changed either would describe a conversion that does not +// happen. +// +// For an RGB input the answer is the device's FIRST advertised +// encode-source format, which is the same one the session takes: an RGB +// session states no encode-source request, precisely so that a packed 4:4:4 +// alias cannot be matched by enum against a genuine RGBA input. +// +// VK_FORMAT_UNDEFINED for an input this library does not convert, and for an +// empty device list. +VkFormat VkEncConversionTargetFormat(VkFormat inputFormat, + const VkFormat* deviceFormats, + uint32_t deviceFormatCount); + +// Decides whether ONE candidate input format is advertisable and, if so, what +// it is encoded as and by which route. Writes |outEntry| only when it returns +// true; |outEntry->format| is pre-set to the candidate, so an admission that +// only fills in encodeFormat and optimality is complete. +// +// THE PARAMETER IS THE WHOLE POINT. The production caller supplies the LIVE +// per-candidate resolver -- the same one the point query answers from, so the +// two surfaces cannot drift -- and the device-free tests supply a synthetic +// one built from a static device list. What is being tested through the +// synthetic one is everything BUT the admission rule: the ordering, the +// de-duplication, the capacity stop, and the build gate. +typedef bool (*VkEncInputFormatAdmitFn)( + void* userData, VkFormat candidate, + VkVideoEncoderInputFormatProperties* outEntry); + +// The advertised input-format list for one profile: every format this library +// can route to an encoder input the device accepts, each naming what it is +// encoded as. Writes at most |outCapacity| entries and returns how many were +// written. +// +// EVERY CANDIDATE COMES FROM THE ROUTABLE LIST and is offered to |admit| +// exactly once. The admission decides membership and optimality; this function +// decides order, uniqueness and capacity. +// +// OPTIMAL entries come first, then SUBOPTIMAL ones, each group in the ROUTABLE +// list's order. It is not the device's order any more, and it cannot be: each +// candidate is now resolved at the profile its own binding derives, so there is +// no single device list to order by. Each format is written once however many +// tilings a device reports it at, because tiling is a property of an image and +// not of a format. +// +// NOT ADVERTISED AT ALL WHEN THE FILTER IS NOT COMPILED IN: every SUBOPTIMAL +// entry is ENCODABLE_VIA_FILTER and InitializeExt refuses exactly that class in +// such a build, so advertising one would name a format the library then +// refuses. The gate is the build's, not the admission's, and it is applied here +// so that no admission can bypass it. +// +// |outCapacity| is the caller's array length; the advertised list can never +// be longer than the routable list, so an array sized from that list holds +// every answer this function can give. +// +// A pure function of |admit|, so a test drives it with no device. +uint32_t VkEncAdvertiseInputFormats( + VkEncInputFormatAdmitFn admit, void* userData, + VkVideoEncoderInputFormatProperties* outEntries, uint32_t outCapacity); + +// The buffer the library hands the DRIVER when it asks for a profile's +// VIDEO_ENCODE_SRC formats. A device list, not an advertised one. +enum { VK_ENC_MAX_DEVICE_INPUT_FORMATS = 16 }; + +// An upper bound on how many input formats this library can route, and so on +// how long an advertised list can be: every advertised entry names a distinct +// routable format. +// +// A BOUND, NOT A COUNT. The routable set is derived from the multi-planar +// Y'CbCr format table plus the three RGB spellings, so its size is a property +// of that table and moves when the table does. What this constant has to be +// is large enough to hold the derivation's answer, which the .cpp +// static_asserts against the table's own length rather than against a number +// written twice. +enum { VK_ENC_MAX_ROUTABLE_INPUT_FORMATS = 41 }; + +// Everything one (codec, profile) probe answers about a physical device: the +// scalar capabilities the public structure carries, and the std-syntax flag +// list, which is answered through its own two-call entry point. +// +// The list is held beside the scalars rather than inside them because a list in +// a public structure has to pick a capacity, and the capacity belongs to the +// caller. Inside the library the capacity is a private constant, sized from +// what the probe can actually produce. +// +// NO INPUT-FORMAT LIST. The probe issues one device format query per (codec, +// profile) at a fixed 4:2:0 envelope, which is the right envelope for the +// scalars and the wrong one for a format list -- a caller asking which formats +// it may feed the encoder is asking about the profile its own input derives, +// not about the one the probe happened to key on. The enumerator therefore +// resolves each candidate live, through the same function the point query +// answers from, and holds no list here to go stale against it. +enum { VK_ENC_MAX_STD_FLAG_ENTRIES = 4 }; + +struct VkEncProfileCapabilitySnapshot { + VkVideoEncoderCapabilities caps; + uint32_t stdFlagCount; + VkVideoEncoderStdFlags stdFlags[VK_ENC_MAX_STD_FLAG_ENTRIES]; +}; + +// The colour model these samples are ACTUALLY in: the caller's declaration +// when one was made, and what the format says otherwise. +// +// The single point at which FROM_FORMAT is resolved, so no other site has to +// know which formats name their own model. Returns +// VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT -- never a guess -- for a format +// this library does not place at all, and for a declaration the format cannot +// carry; both are UNSUPPORTED to VkEncClassifyInput. +VkVideoEncoderColorModel VkEncResolveColorModel( + VkFormat inputFormat, VkVideoEncoderColorModel declared); + +// Import memory-type selection. +// +// The candidate SET is spec-constrained; the preference ORDER within it is +// this library's policy (exporter's index if representable, else DEVICE_LOCAL, +// else the first importable type): +// - dma-buf / D3D11: |authoritativeMask| is what +// vkGetMemoryFd/Win32HandlePropertiesKHR reported for THIS handle +// (VUID-VkMemoryAllocateInfo-memoryTypeIndex-00648 / -00645). 0 means +// the query was unavailable, which degrades to the legacy heuristic -- +// the caller logs and counts that, never this function. +// - opaque handles (|exporterIndexExact|): the exporter's own parameters +// are the only legal ones (VUID-VkMemoryAllocateInfo-allocationSize- +// 01742/-01743); a known-but-unrepresentable index has no valid +// substitute and selects nothing. +// +// Free function with no device dependency, so the policy is testable in a +// device-free unit test exactly like the config binder above. +enum VkEncImportMemoryTypeOutcome { + VK_ENC_IMPORT_MEMTYPE_NONE = 0, // nothing importable: refuse + VK_ENC_IMPORT_MEMTYPE_EXPORTER_INDEX, // exporter's index, inside the mask + VK_ENC_IMPORT_MEMTYPE_MASK_DEVICE_LOCAL, + VK_ENC_IMPORT_MEMTYPE_MASK_FIRST, // importable but not DEVICE_LOCAL +}; + +struct VkEncImportMemoryTypeRequest { + uint32_t requirementsMask = 0; // vkGetImageMemoryRequirements + uint32_t authoritativeMask = 0; // handle-properties query; 0 = none + uint32_t exporterMask = 0; // descriptor.memoryTypeBits; 0 = unknown + uint32_t exporterIndex = UINT32_MAX; // descriptor.memoryTypeIndex + VkBool32 exporterIndexExact = VK_FALSE; // opaque handles: as-given or nothing +}; + +struct VkEncImportMemoryTypeChoice { + uint32_t index = UINT32_MAX; // valid iff outcome != NONE + VkEncImportMemoryTypeOutcome outcome = VK_ENC_IMPORT_MEMTYPE_NONE; + // The exporter named an index and the mask excluded it. SUCCESS-class + // (a different type was selected) -- the caller logs and counts it. + VkBool32 exporterIndexOverridden = VK_FALSE; +}; + +void VkEncSelectImportMemoryType(const VkEncImportMemoryTypeRequest& request, + const VkPhysicalDeviceMemoryProperties& memProps, + VkEncImportMemoryTypeChoice* outChoice); + +// The acceptance behind ModifierWouldRegister's modifier pre-check: a +// modifier the device cannot service must surface as the renegotiable +// MODIFIER_UNSUPPORTED, not fall through to vkCreateImage as +// IMPORT_FAILED. vkCreateImage's validity is defined +// against the vkGetPhysicalDeviceImageFormatProperties2 query for the same +// inputs THROUGH THE LIMITS THE QUERY RETURNS: extent +// (VUID-VkImageCreateInfo-extent-02252/-02253), mip levels +// (-mipLevels-02255), array layers (-arrayLayers-02256) and sample count +// (-samples-02258), with the descriptor's 0-means-default rules applied +// exactly as ImportImageLocked applies them when it builds +// VkImageCreateInfo. Consuming fewer of them admits a descriptor the +// import must then refuse. +// +// Free function with no device dependency, so the acceptance is provable in +// a device-free unit test exactly like the memory-type policy above; only +// the query itself needs hardware. +VkBool32 VkEncDescriptorWithinCreationLimits( + const VkVideoEncoderExternalImageDescriptor& desc, + const VkImageFormatProperties& limits); + +// --------------------------------------------------------------------------- +// The external-image import behind registration. +// +// Exposed here for the same reason as the memory-type policy above: the +// fd-ownership split below is decided against DRIVER-call failures that no +// test on working hardware can produce on demand, so the import takes the +// dispatch context by reference and a test drives its real code with a +// stubbed dispatch table and real pipe(2) fds. Only the happy path against +// a real driver needs hardware. +class VulkanDeviceContext; + +// Which side of the vkAllocateMemory handoff the import stopped on. The +// split has to be carried because once vkAllocateMemory has been CALLED +// with a VkImportMemoryFdInfoKHR chained, the fd is not the library's to +// close -- NVIDIA consumes it even when the allocation FAILS -- and on +// success the VkDeviceMemory owns it until vkFreeMemory. A close() on +// either post-handoff path is a double close, and in a multithreaded +// process the number is immediately recyclable, so the second close tears +// down whatever unrelated descriptor now holds it. +enum VkEncExternalImageImportResult { + VK_ENC_IMPORT_FAILED_BEFORE_ALLOCATE = 0, // fd never reached the driver + VK_ENC_IMPORT_FAILED_AFTER_ALLOCATE = 1, // the driver consumed the fd + VK_ENC_IMPORT_SUCCESS = 2, +}; + +struct VkEncImportedImage { + VkImage image = VK_NULL_HANDLE; // valid iff SUCCESS + VkDeviceMemory memory = VK_NULL_HANDLE; // valid iff SUCCESS + VkDeviceSize allocationSize = 0; // what was actually allocated + VkEncExternalImageImportResult result = + VK_ENC_IMPORT_FAILED_BEFORE_ALLOCATE; + // SUCCESS-class memory-type telemetry (heuristic selections, + // exporter-index overrides); the member wrapper folds these into the + // session counters. + uint32_t heuristicSelections = 0; + uint32_t exporterOverrides = 0; +}; + +// vkCreateImage + memory-type selection + vkAllocateMemory(import chain) + +// vkBindImageMemory, from |desc|'s TRUE properties. |desc| has already +// passed ValidateImageDescriptor, and under BORROW |osHandle| is already +// the library's private duplicate. +// +// fd ownership on exit -- the rule that the library closes anything it did +// NOT hand to the driver, on every exit path, applied literally: +// * FAILED_BEFORE_ALLOCATE: the fd never reached the driver; THIS +// function closes it before returning. +// * FAILED_AFTER_ALLOCATE: the failed vkAllocateMemory consumed it, or +// the allocated-then-unbindable VkDeviceMemory released it via the +// cleanup vkFreeMemory. Nobody closes it again -- not this function, +// not any caller. +// * SUCCESS: the VkDeviceMemory owns it; vkFreeMemory is the release. +// On every exit the fd is consumed from the API caller's point of view +// (handlesConsumed stays VK_TRUE); the split only names which owner +// performs the close, and that owner is exactly one. Win32 handles are +// never closed here on any path, per the public header's rule. +VkVideoEncoderStatusCode VkEncImportExternalImage( + const VulkanDeviceContext& vkDevCtx, + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t osHandle, + VkEncImportedImage* outImport); + +// --------------------------------------------------------------------------- +// The dma-buf import-ordinal guard's verdict, carried from the guard (which +// runs deep inside the import) out to RegisterImageResource, which is the +// only place that can hand it to a caller. The carrier a caller chains is +// VkVideoEncoderImportGuardInfo, declared earlier in this header; this is the +// internal hop behind it, which has no sType and is not ABI. +// +// THREAD-LOCAL, deliberately. The guard writes it on the thread running the +// import and RegisterImageResource reads it back on that same thread (class +// (b), submit-thread-affine), so two sessions registering concurrently cannot +// overwrite each other's verdict -- which a process global would let them do, +// silently, exactly when a consumer is trying to diagnose one of them. +struct VkEncImportOrdinalGuardReport { + VkVideoEncoderImportGuardState state = + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED; + // What this BUILD retains where the guard applies -- a constant, not an + // outcome, so it is filled even on the paths the guard never evaluates. + // 0 in a stock build, because the guard is disabled by default. + uint32_t requestedCount = 0; + uint32_t retainedCount = 0; + VkVideoEncoderStatusCode failureStatus = VK_VIDEO_ENCODER_STATUS_SUCCESS; + int32_t failureErrno = 0; +}; + +// Clear the calling thread's record and re-stamp requestedCount. Called at +// the top of every RegisterImageResource, so a registration that never +// reaches the import reports NOT_EVALUATED rather than the previous +// registration's answer. +void VkEncResetImportOrdinalGuardReport(); + +void VkEncGetImportOrdinalGuardReport(VkEncImportOrdinalGuardReport* outReport); + +// --------------------------------------------------------------------------- +// Release of the dma-buf import-ordinal guard. +// +// The guard retains a small number of sacrificial dma-buf image imports for +// the LIFE OF THE DEVICE, so that every caller-visible import lands that many +// live-positions later. It is DISABLED BY DEFAULT -- the build count is 0, +// nothing is retained, this function finds nothing and returns 0. Everything +// below is about a build that enables it. "For the life of the device" cuts +// both ways, and this function is the second half of it: +// +// * RELEASED TOO EARLY, the shift the count was chosen for is undone, and +// undone SILENTLY: the freed position is handed straight to the next +// caller import and nothing reports it. So the only correct call site is +// one where no further import on |vkDevCtx| is possible at all: +// VulkanVideoEncoderExtImpl::Deinitialize(), after the encoder has been +// released and every registration retired. +// * NEVER RELEASED, and vkDestroyDevice runs with those VkImage and +// VkDeviceMemory objects still alive on the device +// (VUID-vkDestroyDevice-device-05137). +// +// RETURNS the number of guard imports actually released for |vkDevCtx|'s +// device: 0 when the guard never ran on it (the disabled default, a device it +// does not apply to, no dma-buf import, kill switch set, or a device that +// never imported), and the full guard count when it did. A return value +// rather than only a log line, because an embedder can silence the library's +// narration. +// +// Idempotent and null-safe: the device's entry is taken off the registry +// before anything is destroyed, so a second call for the same device returns 0 +// and destroys nothing, and a VK_NULL_HANDLE device returns 0. +uint32_t VkEncReleaseImportOrdinalGuard(const VulkanDeviceContext& vkDevCtx); + +// --------------------------------------------------------------------------- +// Fault-injection seam for the release-obligation guard. +// +// That mitigation is only as good as its tests: every post-arm early return +// in SubmitRegisteredFrame owes the registration a release, and the sites +// that matter most -- a refused submit AFTER the reference is taken -- are +// reachable on demand only by making the submit backend fail, which no test +// against working hardware can do. So the backend is the seam: a null +// backend that stands in for the encoder on the one call the submit makes to +// it, while every gate, resolution step and bookkeeping path around it stays +// the production code. Null by default; a production session never installs +// one, so the production-path cost is a null pointer test on the gates that +// must admit a backend-less session. +// +// Build placement, stated honestly: the five seam functions below are +// compiled into the production library objects. The usual discipline for a +// seam -- a test-only build target no production target may depend on -- is +// not applied here; nothing in production references these symbols (only +// the library test TUs do), but that gate is convention, not the build +// system. Applying +// the testonly discipline means extracting the impl class definition into +// a src/-private header and moving these five definitions into a TU that +// only test targets compile, on both build systems; until then, this +// header's include list is the boundary to review. + +class VulkanVideoEncoderExt; + +struct VkEncNullBackendState { + // What the submit backend answers. VK_SUCCESS enqueues the pending + // entry through the same bookkeeping the real backend's success arm + // runs; anything else is returned as that backend failure. + VkResult submitResult = VK_SUCCESS; +}; + +// Install |state| (caller-owned; must outlive the encoder) on a freshly +// created, never-initialized encoder from CreateVulkanVideoEncoderExt -- no +// other object may be passed here. The session then reports initialized +// with no device, no worker threads and -- until VkEncPushCapture installs +// its capture source -- no VkVideoEncoder behind it: +// VK_IMAGE registration skips the device-touching view build (the import +// arms still refuse without a device, at validation) and submits terminate +// at the null backend. VK_ERROR_NOT_PERMITTED_KHR on a session that is +// already initialized: backends stand in for a real session, they never +// replace one. +VkResult VkEncInstallNullBackend(VulkanVideoEncoderExt* encoder, + const VkEncNullBackendState* state); + +// THE IMPORT CONTENT PROBE'S MEASUREMENT SEAM. +// +// Hands back the three plane means the probe measured for |resource|, and -- +// the point of the seam -- lets a test INJECT a measurement without a device. +// +// WHY THIS EXISTS AND WHAT IT IS NOT. The probe's own capture needs a real +// dma-buf import, a real staging command buffer and a real fence; nothing +// device-free can produce one. Everything AFTER the measurement -- the +// predicate, the per-registration latch, the oldest-damaged-first report, the +// drain when the consumer retires the buffer, and the whole ext-side reporting +// path through GetCompletionInfo -- is pure host logic, and without this +// seam it would be exercisable only on a device that exhibits the import +// defect. An observable whose only test cannot fail is not an observable. +// +// It injects a MEASUREMENT, never a verdict: the state the caller then reads +// back is the production predicate's answer, not something the test chose. +// A |resource| that is not armed, or already scored, is a no-op -- the +// once-per-registration rule is the production rule and this does not get to +// bypass it. +// +// Returns VK_ERROR_NOT_PERMITTED_KHR when no registration ever asked for a +// content probe (nothing to inject into). +VkResult VkEncInjectImportContentMeasurement(VulkanVideoEncoderExt* encoder, + VkVideoEncoderResource resource, + uint32_t meanYQ8, + uint32_t meanUQ8, + uint32_t meanVQ8); + +// Which route a registered external image takes to the encoder. +// +// DIRECT the encoder reads the caller's image as it stands. The +// registration's format, colour model, tiling and usage together +// satisfy the direct predicate, and no copy and no conversion is +// built. +// STAGED the image is copied into the library's own input pool. This is +// the route for a registration that is not directly encodable and +// needs no conversion either -- a pure tiling mismatch, say. +// FILTER the frame goes through the preprocess compute filter. Chosen +// when the input classifies ENCODABLE_VIA_FILTER, this session +// built the filter for that input, and the views the filter reads +// were created on this image. +// +// An implementation choice and not a contract, which is why it is declared +// here: a caller negotiates what it may hand in, and the library decides +// what it does with what it is handed. Nothing on the public surface names +// this type or its values. +typedef enum VkVideoEncoderExternalInputPath { + VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT = 0, + VK_VIDEO_EXTERNAL_INPUT_PATH_STAGED = 1, + VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER = 2, +} VkVideoEncoderExternalInputPath; + +// The registration-slot facts the release-obligation tests assert on. A +// dropped (or stranded) in-flight reference is observable only here: the +// public surface deliberately answers a stale id with RESOURCE_UNKNOWN +// whether the slot is retired, pinned, or gone. +struct VkEncResourceProbe { + uint32_t inFlight = 0; + VkBool32 live = VK_FALSE; + VkBool32 retired = VK_FALSE; + // The ROUTING DECISION, which is the whole point of the registration + // gate. The public surface carries no accessor for it -- deliberately: + // it is an implementation choice, not a contract -- and the only other + // trace of it is a VkEncErr line that silenceStdio can null for a whole + // session. Without this field a test can assert that a descriptor + // REGISTERED but not what it registered AS, which makes "routes via the + // compute filter" and "was accepted at all" one observation. + // + // The two view facts are reported beside it because the path is a + // FUNCTION of them: reading only inputPath cannot distinguish "the + // filter clause failed" from "the format clause did", and those are + // different bugs. + VkVideoEncoderExternalInputPath inputPath = + VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + VkBool32 planeStorageViews = VK_FALSE; + VkBool32 storageReadView = VK_FALSE; +}; + +// Fill |outProbe| for the slot |resource| names. RESOURCE_UNKNOWN once the +// slot has been destroyed (its generation advanced) or never existed -- +// which is itself the assertion that a deferred retirement completed. +VkVideoEncoderStatusCode VkEncProbeResource(VulkanVideoEncoderExt* encoder, + VkVideoEncoderResource resource, + VkEncResourceProbe* outProbe); + +// Whether the session currently reports initialized. Deinitialize() -- the +// worker join on the destruction path -- is what flips it, so a release +// thunk that reads VK_FALSE here has proof the teardown already ran. This +// is what the destructor-order test asserts, because the thing the order +// actually protects (a worker still inside an invocation) needs a device +// to exist. +VkBool32 VkEncSessionInitialized(VulkanVideoEncoderExt* encoder); + +// Raise the completion edge exactly as a capture push does -- counter, +// OS-event signal, serialized callback invocation -- from a plain test +// thread. This is what lets the trampoline stress test race ReadyThunk +// invocations against SetCompletionCallback replace/detach without a real +// encode session: the delivery threads the production edge runs on exist +// only with a device. +void VkEncFireCompletionEdge(VulkanVideoEncoderExt* encoder, + uint64_t frameId); + +// Deliver a completion record for |frameId| into the session exactly as a +// completed encode would: pushed through the capture funnel on a source +// installed as the session's encoder, so the pop, the match against the +// pending set and the late-capture accounting all run the production drain +// under the production lock order. Null-backend sessions only +// (VK_ERROR_NOT_PERMITTED_KHR otherwise): injection stands in for a real +// session's completion path, it never runs beside one. The first push +// installs a device-free capture source behind the session -- the one +// departure from the no-VkVideoEncoder shape VkEncInstallNullBackend +// documents -- wired to the completion edge exactly as InitializeExt wires +// a real encoder. +VkResult VkEncPushCapture(VulkanVideoEncoderExt* encoder, + uint64_t frameId, + VkResult status); + +// MID-STREAM CONSTANT-QP OBSERVATION SEAM. +// +// Folds any armed rate-control update and hands back the session +// CONSTANT-QP defaults that result -- the values EncodeFrameCommon copies +// into the next frame it processes. +// +// WHY THIS IS THE RIGHT OBSERVABLE. A DISABLED-mode session has no other +// session-level rate lever: the per-layer bitrates a rate-control command +// carries are dropped outright on such a session, because that mode +// commands layerCount 0. So the only way to tell a real constant-QP +// reconfigure from one that merely returned VK_SUCCESS is to read the +// value the next frame would be encoded with, which is what this reports. +// +// Null-backend sessions only (VK_ERROR_NOT_PERMITTED_KHR otherwise), and +// only once VkEncPushCapture has installed the device-free encoder behind +// the session. What this CANNOT assert is that the driver then honours the +// value -- that needs a GPU and a decoded comparison. +VkResult VkEncApplyAndGetSessionConstQp(VulkanVideoEncoderExt* encoder, + int32_t* pQpIntra, + int32_t* pQpInterP, + int32_t* pQpInterB); + +// MID-STREAM RATE-CONTROL OBSERVATION SEAM. +// +// Folds any armed update and reports what is IN FORCE afterwards, at the +// three depths a rate-control change has to survive: +// +// * layer* -- the live VkVideoEncodeRateControlLayerInfoKHR that +// HandleCtrlCmd copies verbatim into the next control +// command. This is where a maxBitrate of 0 shows up as +// the averageBitrate it was coerced to, and where a +// frameRateNum of 0 shows up as the frame rate that was +// left alone -- the values a caller-visible record of +// the configuration has to agree with. +// * config* -- the session config, where a QP clamp REQUEST lands. +// * resolved* -- the codec rate-control layer struct that request +// resolves to. This is the far end of the library-side +// chain and the struct CodecHandleRateControlCmd chains +// onto the command, so a configMinQp that moved while +// resolvedMinQpI did not is a clamp reaching nothing. +// +// codecRefreshCount counts re-invocations of the codec rate-control fill. +// It separates a real refresh from one that recomputed the same numbers, +// and it is what lets a test assert the NEGATIVE case: a bitrate-only +// update must not cause one. +// +// Null-backend sessions only (VK_ERROR_NOT_PERMITTED_KHR otherwise), and +// only once VkEncPushCapture has installed the device-free encoder behind +// the session. What this CANNOT assert is that the driver then honours any +// of it -- that needs a GPU and a decoded comparison. +typedef struct VkEncRateControlObservation { + uint64_t layerAverageBitrate; + uint64_t layerMaxBitrate; + uint32_t layerFrameRateNumerator; + uint32_t layerFrameRateDenominator; + int32_t constQpIntra; + int32_t constQpInterP; + int32_t constQpInterB; + int32_t configMinQp; + int32_t configMaxQp; + uint32_t configMinQpSet; + uint32_t configMaxQpSet; + uint32_t resolvedUseMinQp; + uint32_t resolvedUseMaxQp; + int32_t resolvedMinQpI; + int32_t resolvedMaxQpI; + uint32_t codecRefreshCount; +} VkEncRateControlObservation; + +VkResult VkEncApplyAndGetRateControl(VulkanVideoEncoderExt* encoder, + VkEncRateControlObservation* pOut); + +// THE RECORD Reconfigure COMPARES AGAINST, read back. +// +// Reconfigure keeps a copy of the configuration in force and refuses a +// later call that changes an immutable field, by comparing against this. +// The copy is also the session's own statement of what it is running, +// which is only worth anything if it agrees with the live rate-control +// state above -- and for a coerced maxBitrate or a dropped frame rate it +// did not. Reading both and comparing them is what makes that assertable +// rather than a matter of inspection. +// +// Null-backend sessions only. +VkResult VkEncGetRecordedConfig(VulkanVideoEncoderExt* encoder, + VkVideoEncoderConfig* pOut); + +// Seed that record directly. +// +// A device-free session never runs InitializeExt, so its record is a +// default-constructed config: codec NONE, rate-control mode DEFAULT. +// Several of Reconfigure's refusals are keyed on what the session WAS +// initialized as -- a QP clamp change is refused on an AV1 session and on +// a constant-QP one -- and without this those branches could only be read, +// not run. Null-backend sessions only; pNext is cleared, as InitializeExt +// clears it. +VkResult VkEncSeedRecordedConfig(VulkanVideoEncoderExt* encoder, + const VkVideoEncoderConfig* pConfig); + +// Declare the device QP window the mid-stream clamp check reads. +// +// A real session records it from the codec capabilities in +// InitEncoderCodec. A device-free one has no capabilities to record, so +// the window would sit at "not established" and the check would be inert +// -- untestable rather than merely unexercised. Null-backend sessions +// only. +VkResult VkEncSetDeviceQpWindow(VulkanVideoEncoderExt* encoder, + int32_t minQp, int32_t maxQp); + +// --------------------------------------------------------------------------- +// Sync-resolution observation seam. +// +// SubmitRegisteredFrame's chained-descriptor walk decides, PER DIRECTION, +// whether the submit is handed the caller's raw semaphore arrays or the ones +// the walk resolved. That decision is invisible from the public surface: the +// submit consumes the arrays and answers a status. So a null-backend session +// records what it was handed, and the three functions below set the chain up +// and read the record back -- with no device, no encoder and no queue. +// +// The rule these exist to pin: a chain naming only waits must leave the +// caller's signal list alone, and a chain naming only signals must leave the +// caller's wait list alone. Overriding both directions whenever either is +// resolved would silently zero the caller's signal semaphores on every +// acquire-fence-only frame. + +// Register |semaphore| device-free and answer the id a FrameSyncDescriptor +// names it by. The public RegisterSemaphore cannot run here: it creates a +// timeline VkSemaphore and imports an OS handle into it, and a null-backend +// session has no device with which to do either -- so without this, a +// FrameSyncDescriptor on such a session can only ever resolve to nothing. +// |semaphore| is stored, compared and handed to the submit, never +// dereferenced; pass a distinct non-null sentinel per call. +// +// MUST be paired with VkEncUninstallTestSemaphore before the session is +// destroyed. Deinitialize destroys every still-live registry entry through the +// device dispatch table, which on a device-free session is unpopulated; the +// pairing is what keeps that loop from ever reaching one of these entries. +// VK_ERROR_NOT_PERMITTED_KHR unless the session is null-backend. +VkResult VkEncInstallTestSemaphore(VulkanVideoEncoderExt* encoder, + VkSemaphore semaphore, + VkVideoEncoderResource* outResource); + +// Retire an id from VkEncInstallTestSemaphore with UnregisterSemaphore's slot +// hygiene -- handle nulled, live cleared, generation bumped -- and without the +// driver destroy the public unregister performs. Null-backend sessions only. +VkVideoEncoderStatusCode VkEncUninstallTestSemaphore( + VulkanVideoEncoderExt* encoder, VkVideoEncoderResource resource); + +// How many entries of each array VkEncSubmitSyncProbe carries. The counts it +// reports are NOT clamped to this, so a truncated record is always +// distinguishable from a short list. +enum { kVkEncSubmitSyncProbeCapacity = 8 }; + +// The two arrays the most recent submit on a null-backend session was handed, +// AFTER the walk applied its per-direction override. The record is taken at +// the null backend: below every gate and below the walk, above the two +// SetExternalInputFrame* call sites -- and those forward these same +// VkVideoEncodeInputFrame fields verbatim, so what is recorded is what either +// site would submit. +struct VkEncSubmitSyncProbe { + // VK_FALSE until a submit reaches the null backend. + VkBool32 recorded = VK_FALSE; + uint32_t waitCount = 0; + uint32_t signalCount = 0; + VkSemaphore waitSemaphores[kVkEncSubmitSyncProbeCapacity] = {}; + uint64_t waitValues[kVkEncSubmitSyncProbeCapacity] = {}; + VkSemaphore signalSemaphores[kVkEncSubmitSyncProbeCapacity] = {}; + uint64_t signalValues[kVkEncSubmitSyncProbeCapacity] = {}; +}; + +VkVideoEncoderStatusCode VkEncProbeLastSubmitSync( + VulkanVideoEncoderExt* encoder, VkEncSubmitSyncProbe* outProbe); + +// --------------------------------------------------------------------------- +// CAPABILITY-PROBE KEY OBSERVATION SEAM. +// +// The capability probe is keyed on (codec, profile, bit depth), because a +// Vulkan capability query is per VkVideoProfileInfoKHR and the depth is part +// of that structure. The public entry points name a profile by the codec +// standard number ALONE, which is enough for H.264 and H.265 -- there the +// number decides the depth -- and is NOT enough for AV1, whose seq_profile 0 +// (Main) carries 8 or 10 bits. These read the probe tables directly so that +// the set of combinations the library can put to a driver is asserted on a +// runner with no encode-capable device, where no capability entry point can +// answer anything but "not present". +// +// They report what the library CAN ASK, never what a device answers. A device +// answer is measured on hardware or not at all. + +// Is (codec, profile, bitDepth) a combination this library probes? bitDepth +// is in bits (8 or 10); any other value is false for every codec. +bool VkEncProbeNamesProfileBitDepth(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t bitDepth); + +// How many probe rows the context snapshot carries for |codec|. Rows are what +// a context build issues one driver query each for; two rows may carry the +// same profile number at different depths. +uint32_t VkEncProbeSnapshotRowCount(VkVideoCodecOperationFlagBitsKHR codec); + +// The (profile, bit depth) of snapshot row |slot| for |codec|. False when the +// codec has no such row. Either out pointer may be null. +bool VkEncProbeSnapshotRowAt(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t slot, + uint32_t* outProfile, + uint32_t* outBitDepth); + +#endif // VULKAN_VIDEO_ENCODER_EXT_INTERNAL_H_ diff --git a/vk_video_encoder/json_config/nvidia/README.md b/vk_video_encoder/json_config/nvidia/README.md index edde1076..daff3e0b 100644 --- a/vk_video_encoder/json_config/nvidia/README.md +++ b/vk_video_encoder/json_config/nvidia/README.md @@ -408,11 +408,11 @@ Use the **ThreadedRenderingVk** renderer's GPU compute filter pipeline to genera ```bash # Generate ALL formats headless (no display needed) — 5 resolutions × 8 formats -cd /data/nvidia/vulkan/samples/ThreadedRenderingVk_Standalone -./scripts/generate_encoder_yuv.sh --output-dir /data/misc/VideoClips/ycbcr --frames 128 --all +cd +./scripts/generate_encoder_yuv.sh --output-dir /ycbcr --frames 128 --all # Generate only 4:2:0 (8-bit + 10-bit) — sufficient for most H.264/H.265 profiles -./scripts/generate_encoder_yuv.sh --output-dir /data/misc/VideoClips/ycbcr --frames 128 +./scripts/generate_encoder_yuv.sh --output-dir /ycbcr --frames 128 ``` **Generated file naming:** `{W}x{H}_{subsampling}_{bitdepth}.yuv` @@ -433,7 +433,7 @@ cd /data/nvidia/vulkan/samples/ThreadedRenderingVk_Standalone ```bash # Size verification + visual playback -python3 scripts/verify_yuv_ffmpeg.py /data/misc/VideoClips/ycbcr --frames 128 --show +python3 scripts/verify_yuv_ffmpeg.py /ycbcr --frames 128 --show ``` **Resolutions generated:** 176×144, 352×288, 720×480, 1280×720, 1920×1080 @@ -465,20 +465,20 @@ Input YUV resolution/bitdepth/chroma are auto-detected from filenames. ```bash # Run ALL profiles × all codecs with auto-detected YUV python3 scripts/run_encoder_profile_tests.py \ - --video-dir /data/misc/VideoClips/ycbcr --local + --video-dir /ycbcr --local # Run only NVIDIA profiles python3 scripts/run_encoder_profile_tests.py \ - --video-dir /data/misc/VideoClips/ycbcr --profile-filter nvidia --local + --video-dir /ycbcr --profile-filter nvidia --local # Single profile, single codec python3 scripts/run_encoder_profile_tests.py \ - --video-dir /data/misc/VideoClips/ycbcr \ + --video-dir /ycbcr \ --profile-filter nvidia/high_quality_p4 --codec h265 --local # Verbose (show commands + output filenames) python3 scripts/run_encoder_profile_tests.py \ - --video-dir /data/misc/VideoClips/ycbcr --local --verbose + --video-dir /ycbcr --local --verbose ``` | Option | Description | @@ -556,16 +556,16 @@ End-to-end workflow from YUV generation through encode, decode, and visual verif ```bash # 1. Generate YUV test content (headless, no display needed) -cd /data/nvidia/vulkan/samples/ThreadedRenderingVk_Standalone -./scripts/generate_encoder_yuv.sh --output-dir /data/misc/VideoClips/ycbcr --frames 32 --all +cd +./scripts/generate_encoder_yuv.sh --output-dir /ycbcr --frames 32 --all # 2. Verify YUV files -python3 scripts/verify_yuv_ffmpeg.py /data/misc/VideoClips/ycbcr --frames 32 +python3 scripts/verify_yuv_ffmpeg.py /ycbcr --frames 32 # 3. Run encoder profiles (all codecs × all profiles) -cd /data/nvidia/android-extra/video-apps/vulkan-video-samples +cd python3 scripts/run_encoder_profile_tests.py \ - --video-dir /data/misc/VideoClips/ycbcr --local --max-frames 30 + --video-dir /ycbcr --local --max-frames 30 # 4. Decode all bitstreams (roundtrip verification) DISPLAY=:0 python3 scripts/run_decoder_roundtrip.py /tmp/vulkan_encoder_profile_tests diff --git a/vk_video_encoder/libs/CMakeLists.txt b/vk_video_encoder/libs/CMakeLists.txt index f10ef81c..47d21b1e 100644 --- a/vk_video_encoder/libs/CMakeLists.txt +++ b/vk_video_encoder/libs/CMakeLists.txt @@ -25,9 +25,47 @@ FetchContent_Declare(simdjson ) FetchContent_MakeAvailable(simdjson) +# THE ENCODER'S OS SEAM, selected here and nowhere else. Two translation +# units carry everything the encoder needs from the host OS -- thread naming +# and DRM format modifier selection, and the waitable completion handle -- and +# each has one implementation per platform, chosen by this block. Every +# implementation begins with an #error guarding its own platform, so a unit +# selected for the wrong target fails at its first line rather than at a +# missing symbol during the link. +# +# An unsupported target is refused HERE, at configure time, with the port +# named: a platform with no seam cannot build the encoder, and saying so at +# configure is cheaper than a preprocessor error partway through. +if(CMAKE_SYSTEM_NAME STREQUAL "Linux") + set(VK_VIDEO_ENCODER_OS_ADAPTER_SOURCE + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.cpp) + set(VK_VIDEO_ENCODER_OS_EVENT_SOURCE + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder_os_event_linux.cpp) +elseif(WIN32) + set(VK_VIDEO_ENCODER_OS_ADAPTER_SOURCE + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderOsAdapterWindows.cpp) + set(VK_VIDEO_ENCODER_OS_EVENT_SOURCE + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder_os_event_windows.cpp) +else() + message(FATAL_ERROR + "The Vulkan video encoder library has no OS seam implementation for " + "${CMAKE_SYSTEM_NAME}. Two translation units implement the seam; a " + "port supplies a sibling of each and selects it in this file:\n" + " VkVideoEncoder/VkVideoEncoderOsAdapter.cpp -- thread naming " + "and DRM format modifier selection\n" + " src/vulkan_video_encoder_os_event_.cpp -- the waitable " + "completion handle\n" + "Configure with -DBUILD_ENCODER=OFF to build the rest of the tree " + "without the encoder.") +endif() + set(LIBVKVIDEOENCODER_SRC - ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder_argv.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder_ext.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../include/vulkan_video_encoder.h + ${VK_VIDEO_ENCODER_OS_EVENT_SOURCE} + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/../src/vulkan_video_encoder_os_event_linux.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkEncoderConfigH264.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkEncoderConfigH264.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkEncoderDpbH264.cpp @@ -47,18 +85,31 @@ set(LIBVKVIDEOENCODER_SRC ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderAV1.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderAV1.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkEncoderConfig.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderHdrMetadata.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoder.cpp + ${VK_VIDEO_ENCODER_OS_ADAPTER_SOURCE} + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoGopStructure.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoGopStructure.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoder.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderPsnr.cpp ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderPsnr.h + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderContentProbe.cpp + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/VkVideoEncoder/VkVideoEncoderContentProbe.h ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT}/json/EncoderConfigJsonLoader.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/YCbCrConvUtilsCpu.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkEncoderStdioLatch.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/YCbCrConvUtilsCpu.h + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/Helpers.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/Helpers.h ${VK_DISPATCH_TABLE_SOURCE} ${VK_DISPATCH_TABLE_HEADER} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkThreadSafeQueue.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkThreadSafeQueue.h + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanBufferPool.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanBufferPool.h + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanEncoderInputFrame.cpp + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanEncoderInputFrame.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.cpp ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanDeviceContext.h ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VulkanShaderCompiler.cpp @@ -128,7 +179,7 @@ if(NV_AQ_GPU_LIB_AVAILABLE AND TARGET ${NV_AQ_GPU_LIB_TARGET}) add_dependencies(${VULKAN_VIDEO_ENCODER_LIB} ${NV_AQ_GPU_LIB_TARGET}) endif() -target_include_directories(${VULKAN_VIDEO_ENCODER_LIB} PUBLIC ${VULKAN_VIDEO_ENCODER_INCLUDE} ${VULKAN_VIDEO_ENCODER_INCLUDE}/../NvVideoParser PRIVATE include) +target_include_directories(${VULKAN_VIDEO_ENCODER_LIB} PUBLIC ${VULKAN_VIDEO_ENCODER_INCLUDE} PRIVATE include ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE}) target_compile_definitions(${VULKAN_VIDEO_ENCODER_LIB} PRIVATE VK_VIDEO_ENCODER_IMPLEMENTATION PUBLIC VK_VIDEO_ENCODER_SHAREDLIB @@ -169,7 +220,7 @@ if(NV_AQ_GPU_LIB_AVAILABLE AND TARGET ${NV_AQ_GPU_LIB_TARGET}) target_link_libraries(${VULKAN_VIDEO_ENCODER_STATIC_LIB} PUBLIC ${NV_AQ_GPU_LIB_TARGET}) add_dependencies(${VULKAN_VIDEO_ENCODER_STATIC_LIB} ${NV_AQ_GPU_LIB_TARGET}) endif() -target_include_directories(${VULKAN_VIDEO_ENCODER_STATIC_LIB} PUBLIC ${VULKAN_VIDEO_ENCODER_INCLUDE} ${VULKAN_VIDEO_ENCODER_INCLUDE}/../NvVideoParser PRIVATE include) +target_include_directories(${VULKAN_VIDEO_ENCODER_STATIC_LIB} PUBLIC ${VULKAN_VIDEO_ENCODER_INCLUDE} PRIVATE include ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE}) # Add NV_AQ_GPU_LIB_SUPPORTED definition and include paths if the library is available if(NV_AQ_GPU_LIB_AVAILABLE) @@ -187,6 +238,35 @@ install(TARGETS ${VULKAN_VIDEO_ENCODER_LIB} ${VULKAN_VIDEO_ENCODER_STATIC_LIB} LIBRARY DESTINATION ${CMAKE_INSTALL_LIBDIR} ) +# The encoder's PUBLIC headers, named one by one. A directory install would +# ship whatever the directory happens to hold; naming them is what makes +# adding a header to the shipped surface a decision someone writes down. +# vulkan_video_encoder_ext.h is not among them: it lives in internal/, which is +# on no consumer's include path, so shipping it would export a surface the +# library does not offer. +install(FILES + ${VULKAN_VIDEO_ENCODER_INCLUDE}/vulkan_video_encoder.h + DESTINATION ${CMAKE_INSTALL_INCLUDEDIR} + ) + +# AND WHAT IT INCLUDES. vulkan_video_encoder.h opens with vulkan_interfaces.h +# and VkCodecUtils/VkVideoRefCountBase.h -- so a prefix carrying only the +# header above is a prefix in which it does not compile. +# Both are declaration-only and self-contained: one turns the beta extensions +# on and includes vulkan/vulkan.h, the other declares the base class and the +# shared-pointer alias the interface hands back. Shipping them adds no source +# file and no dependency the consumer did not already have in the Vulkan +# headers. VkCodecUtils/ is reproduced as a subdirectory because that is the +# spelling vulkan_video_encoder.h uses to reach it. +install(FILES + ${VULKAN_VIDEO_APIS_INCLUDE}/vulkan_interfaces.h + DESTINATION ${CMAKE_INSTALL_INCLUDEDIR} + ) +install(FILES + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT}/VkCodecUtils/VkVideoRefCountBase.h + DESTINATION ${CMAKE_INSTALL_INCLUDEDIR}/VkCodecUtils + ) + if(WIN32) install(TARGETS ${VULKAN_VIDEO_ENCODER_LIB} ${VULKAN_VIDEO_ENCODER_STATIC_LIB} RUNTIME DESTINATION ${CMAKE_INSTALL_PREFIX}/bin diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.cpp index f9785d1b..0f02cd50 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.cpp @@ -15,34 +15,82 @@ */ #include "VkVideoEncoder/VkEncoderConfig.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "VkVideoEncoder/VkEncoderConfigH264.h" #include "VkVideoEncoder/VkEncoderConfigH265.h" #include "VkVideoEncoder/VkEncoderConfigAV1.h" #include "json/EncoderConfigJsonLoader.h" #include #include +#include #include #include #include +#include namespace { + // A PARSE THAT CANNOT SILENTLY CHANGE THE NUMBER IT WAS GIVEN. + // + // Three ways the value the caller typed can differ from the value the + // encoder runs with, all of which report success: + // + // * NARROWING. static_cast of a value too large for T keeps the low + // bits. Into a uint8_t, 256 is 0 and 300 is 44 -- and 0 is a sentinel + // in more than one of these fields, so the truncation does not even + // land on an obviously wrong number. + // * A NEGATIVE INTO AN UNSIGNED. strtoull accepts a leading '-' and + // wraps, so "-1" arrives as the largest value T can hold rather than + // as an error. + // * OVERFLOW. Past the range of the accumulator, strtoull/strtoll + // saturate and set ERANGE, which nothing was reading. + // + // Each is refused here instead. The check that the value survives its own + // narrowing is what lets a caller keep parsing straight into a uint8_t + // field: out-of-range input is rejected rather than folded. template inline bool parseUint(const std::string& str, T& value) { if (str.empty()) return false; + // "-1" IS AN ACCEPTED SPELLING, and means all bits set. + // + // It is the Video Codec SDK convention these fields inherit: an + // infinite GOP length is UINT32_MAX -- see the json_config README and + // the idrPeriod derivation in EncoderConfigH264 -- and -1 is how a + // caller writes it without counting the f's. It yields the maximum + // value of T, which is what the unchecked wrap produced, so every + // command line that already used it means the same thing. + // + // NO OTHER NEGATIVE IS. Those are typos, and wrapping one silently is + // how a mistyped bound becomes a four-billion-frame GOP that the + // encoder honours without comment. + if (str == "-1") { + value = static_cast(~0ull); + return true; + } + // A '-' anywhere else is malformed for an unsigned field, including + // one behind leading whitespace that strtoull would skip past. + if (str.find('-') != std::string::npos) return false; + errno = 0; char* end = nullptr; unsigned long long result = strtoull(str.c_str(), &end, 0); if (end != str.c_str() + str.size()) return false; - value = static_cast(result); + if (errno == ERANGE) return false; + const T narrowed = static_cast(result); + if (static_cast(narrowed) != result) return false; + value = narrowed; return true; } template inline bool parseInt(const std::string& str, T& value) { if (str.empty()) return false; + errno = 0; char* end = nullptr; long long result = strtoll(str.c_str(), &end, 10); if (end != str.c_str() + str.size()) return false; - value = static_cast(result); + if (errno == ERANGE) return false; + const T narrowed = static_cast(result); + if (static_cast(narrowed) != result) return false; + value = narrowed; return true; } @@ -97,8 +145,7 @@ namespace { static void printHelp(VkVideoCodecOperationFlagBitsKHR codec) { - fprintf(stderr, - "Usage : EncodeApp \n\ + VkEncPrintfErr("Usage : EncodeApp \n\ -h, --help provides help\n\ -i, --input .yuv Input YUV File Name (YUV420p 8bpp only) \n\ -o, --output .264/5,ivf Output H264/5/AV1 File Name \n\ @@ -139,8 +186,15 @@ static void printHelp(VkVideoCodecOperationFlagBitsKHR codec) --maxQp : Maximum QP value in the range [0, 51] \n\ --qpMap : select quantization map type : deltaQpMap or emaphasisMap \n\ --qpMapFileName : quantization map file name \n\ - --gopFrameCount : Number of frame in the GOP, default 16\n\ - --idrPeriod : Number of frame between 2 IDR frame, default 60\n\ + --gopFrameCount : Number of frame in the GOP, default 16.\n\ + -1 requests an INFINITE GOP: the count is set to\n\ + UINT32_MAX, so the sequence opens with an IDR and\n\ + no second one is emitted. 0 leaves the length to\n\ + the device's preferred value.\n\ + --idrPeriod : Number of frame between 2 IDR frame, default 60.\n\ + -1 requests an INFINITE IDR period (UINT32_MAX):\n\ + no periodic IDR is emitted after the first. 0 leaves\n\ + the period to the device's preferred value.\n\ --consecutiveBFrameCount : Number of consecutive B frame count in a GOP \n\ --temporalLayerCount : Count of temporal layer \n\ --lastFrameType : Last frame type \n\ @@ -192,7 +246,7 @@ static void printHelp(VkVideoCodecOperationFlagBitsKHR codec) being available after every `intraRefreshCycleDuration + index` frames.\n"); #ifdef NV_AQ_GPU_LIB_SUPPORTED - fprintf(stderr, "\ + VkEncPrintfErr("\ --spatialAQStrength : Spatial AQ strength in range [-1.0, 1.0]\n\ < -1.0 = disabled (default: -2.0)\n\ 0.0 = default/neutral strength\n\ @@ -210,18 +264,17 @@ static void printHelp(VkVideoCodecOperationFlagBitsKHR codec) #endif // NV_AQ_GPU_LIB_SUPPORTED if ((codec == VK_VIDEO_CODEC_OPERATION_NONE_KHR) || (codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR)) { - fprintf(stderr, "\nH264 specific arguments:\n\ + VkEncPrintfErr("\nH264 specific arguments:\n\ --slices : Number of slices to divide the picture into\n"); } if ((codec == VK_VIDEO_CODEC_OPERATION_NONE_KHR) || (codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR)) { - fprintf(stderr, "\nH265 specific arguments:\n\ + VkEncPrintfErr("\nH265 specific arguments:\n\ --slices : Number of slices to divide the picture into\n"); } if ((codec == VK_VIDEO_CODEC_OPERATION_NONE_KHR) || (codec == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR)) { - fprintf(stderr, - "\nAV1 specific arguments:\n\ + VkEncPrintfErr("\nAV1 specific arguments:\n\ --tiles Enable tile configuration\n\ --params Enable custom tile configuration when followed by --tiles option\n\ Otherwise default tile configuration will be used\n\ @@ -274,7 +327,17 @@ static void printHelp(VkVideoCodecOperationFlagBitsKHR codec) int EncoderConfig::LoadFromJsonFile(const char* path) { +#if defined(VK_VIDEO_ENCODER_SKIP_JSON_CONFIG) + // JSON-config support is compiled out by this build (it + // avoids pulling in simdjson + EncoderConfigJsonLoader.cpp + // for a feature only the command line uses). Callers go + // through ParseArguments, which checks the return value, so + // -1 surfaces as a clean parse failure. + (void)path; + return -1; +#else return LoadEncoderConfigFromJson(path, this); +#endif } int EncoderConfig::ParseArguments(int argc, const char *argv[]) @@ -282,7 +345,6 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) int argcount = 0; std::vector arglist; std::vector args(argv, argv + argc); - uint32_t frameCount = 0; appName = args[0]; @@ -292,7 +354,7 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) // --encoderConfig: load JSON first (base config); CLI args below override. Precedence: JSON then CLI. if (args[i] == "--encoderConfig") { if (i + 1 >= argc) { - fprintf(stderr, "--encoderConfig requires a path\n"); + VkEncPrintfErr("--encoderConfig requires a path\n"); return -1; } if (LoadFromJsonFile(args[i + 1].c_str()) != 0) return -1; @@ -301,7 +363,7 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } if (args[i] == "-i" || args[i] == "--input") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } size_t fileSize = inputFileHandler.SetFileName(args[i].c_str()); @@ -310,18 +372,36 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } if (inputFileHandler.ParseY4mHeader(&input.width, &input.height, &frameRateNumerator, &frameRateDenominator)) { if (verbose) { - printf("Y4M file detected: width %d height %d\n", input.width, input.height); + VkEncPrintfOut("Y4M file detected: width %d height %d\n", input.width, input.height); } } } else if (args[i] == "-o" || args[i] == "--output") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } size_t fileSize = outputFileHandler.SetFileName(args[i].c_str()); if (fileSize <= 0) { return (int)fileSize; } + } else if (args[i] == "--disableFileOutput") { + // In-memory bitstream capture: the encoded bytes are returned to + // the caller and NO bitstream file is written, including the + // default out.264 / out.265 / out.ivf. This is the command-line + // spelling of EncoderConfig::disableFileOutput; an embedding host + // sets the field instead. Either way FinalizeConfig() reads it + // before deciding whether to open the default output, so the two + // routes reach the same answer. + // + // Not a flush and not a completion signal: which frames are ready + // is reported the same way in both modes. This decides only where + // the bytes go -- to the caller, or additionally to a file. + // + // Combining it with -o still writes no file: capture wins. A + // process that cannot touch the filesystem, such as the Chromium + // GPU process under sandbox, needs a mode in which no path is + // opened at all rather than one it must remember not to name. + disableFileOutput = true; } else if (args[i] == "-h" || args[i] == "--help") { printHelp(codec); return -1; @@ -335,11 +415,11 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; } else { // Invalid codec - fprintf(stderr, "Invalid codec: %s\n", codec_.c_str()); + VkEncPrintfErr("Invalid codec: %s\n", codec_.c_str()); return -1; } if (verbose) { - printf("Selected codec: %s\n", codec_.c_str()); + VkEncPrintfOut("Selected codec: %s\n", codec_.c_str()); } i++; // Skip the next argument since it's the codec value } else if (args[i] == "--dpbMode") { @@ -350,33 +430,33 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) useDpbArray = true; } else { // Invalid codec - fprintf(stderr, "Invalid DPB mode: %s\n", dpbMode.c_str()); + VkEncPrintfErr("Invalid DPB mode: %s\n", dpbMode.c_str()); return -1; } if (verbose) { - printf("Selected DPB mode: %s\n", dpbMode.c_str()); + VkEncPrintfOut("Selected DPB mode: %s\n", dpbMode.c_str()); } i++; // Skip the next argument since it's the dpbMode value } else if (args[i] == "--inputWidth") { if ((++i >= argc) || !parseUint(args[i], input.width)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--inputHeight") { if ((++i >= argc) || !parseUint(args[i], input.height)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--inputNumPlanes") { if ((++i >= argc) || !parseUint(args[i], input.numPlanes)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } // 1 = packed/interleaved single plane (AYUV, Y410 -- 4:4:4 only), // 2 = semi-planar, 3 = planar. if ((input.numPlanes < 1) || (input.numPlanes > 3)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); - fprintf(stderr, "Supported number of planes are 1 (packed 4:4:4), 2 or 3\n"); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Supported number of planes are 1 (packed 4:4:4), 2 or 3\n"); // Reject rather than merely diagnose: an out-of-range plane count that // reaches VerifyInputs indexes planeLayouts[] past its end. return -1; @@ -393,26 +473,26 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) input.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR; } else { // Invalid chromeSubsampling - fprintf(stderr, "Invalid chromeSubsampling: %s\nValid string values are 400, 420, 422, 444 \n", chromeSubsampling.c_str()); + VkEncPrintfErr("Invalid chromeSubsampling: %s\nValid string values are 400, 420, 422, 444 \n", chromeSubsampling.c_str()); return -1; } i++; // Skip the next argument since it's the chromeSubsampling value } else if (args[i] == "--inputLumaPlanePitch") { uint64_t rowPitch = 0; if ((++i >= argc) || !parseUint(args[i], rowPitch)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } input.planeLayouts[0].rowPitch = rowPitch; } else if (args[i] == "--inputBpp") { if ((++i >= argc) || !parseUint(args[i], input.bpp)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--msbShift") { uint8_t msbShiftVal = 0; if ((++i >= argc) || !parseUint(args[i], msbShiftVal)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } input.msbShift = static_cast(msbShiftVal); @@ -420,96 +500,117 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) preferPackedYcbcr = true; } else if (args[i] == "--startFrame") { if (++i >= argc || !parseUint(args[i], startFrame)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--numFrames") { if (++i >= argc || !parseUint(args[i], numFrames)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--repeatInputFrames") { repeatInputFrames = true; } else if (args[i] == "--encodeOffsetX") { if ((++i >= argc) || !parseUint(args[i], encodeOffsetX)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--encodeOffsetY") { if ((++i >= argc) || !parseUint(args[i], encodeOffsetY)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--encodeWidth") { if ((++i >= argc) || !parseUint(args[i], encodeWidth)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--encodeHeight") { if ((++i >= argc) || !parseUint(args[i], encodeHeight)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--encodeMaxWidth") { if ((++i >= argc) || !parseUint(args[i], encodeMaxWidth)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--encodeMaxHeight") { if ((++i >= argc) || !parseUint(args[i], encodeMaxHeight)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--minQp") { if (++i >= argc || !parseInt(args[i], minQp)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } + minQpSet = 1; } else if (args[i] == "--maxQp") { if (++i >= argc || !parseInt(args[i], maxQp)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } + maxQpSet = 1; // GOP structure } else if (args[i] == "--gopFrameCount") { - uint8_t gopFrameCount = EncoderConfig::DEFAULT_GOP_FRAME_COUNT; + // Parsed at the width the GOP structure stores it. parseUint + // casts without a range check, so a narrower local truncates + // silently while still reporting success: 300 arrives as 44, and + // 256 as 0 -- which the codec configs read as + // ZERO_GOP_FRAME_COUNT and replace with the device's preferred + // count. Both give the caller a GOP it did not ask for and is + // never told about. + // + // Zero stays legal here: it IS that sentinel, and asking for the + // device's preference is a request like any other. + uint32_t gopFrameCount = EncoderConfig::DEFAULT_GOP_FRAME_COUNT; if (++i >= argc || !parseUint(args[i], gopFrameCount)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetGopFrameCount(gopFrameCount); if (verbose) { - printf("Selected gopFrameCount: %d\n", gopFrameCount); + VkEncPrintfOut("Selected gopFrameCount: %u\n", gopFrameCount); } } else if (args[i] == "--idrPeriod") { - int32_t idrPeriod = EncoderConfig::DEFAULT_GOP_IDR_PERIOD; - if (++i >= argc || !parseInt(args[i], idrPeriod)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + // Unsigned, because SetIdrPeriod stores it unsigned. Read as a + // signed value it took every negative, and the conversion at the + // setter turned each one into a period so large that no periodic + // IDR is ever emitted -- so a mistyped -5 asked for an infinite + // IDR period and got it. Parsed here at the width and signedness + // the field actually has, -1 keeps its meaning (all bits set, + // infinite) and other negatives are refused. + uint32_t idrPeriod = EncoderConfig::DEFAULT_GOP_IDR_PERIOD; + if (++i >= argc || !parseUint(args[i], idrPeriod)) { + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetIdrPeriod(idrPeriod); if (verbose) { - printf("Selected idrPeriod: %d\n", idrPeriod); + VkEncPrintfOut("Selected idrPeriod: %u\n", idrPeriod); } } else if (args[i] == "--consecutiveBFrameCount") { uint8_t consecutiveBFrameCount = EncoderConfig::DEFAULT_CONSECUTIVE_B_FRAME_COUNT; if (++i >= argc || !parseUint(args[i], consecutiveBFrameCount)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetConsecutiveBFrameCount(consecutiveBFrameCount); if (verbose) { - printf("Selected consecutiveBFrameCount: %d\n", consecutiveBFrameCount); + VkEncPrintfOut("Selected consecutiveBFrameCount: %u\n", + (unsigned)consecutiveBFrameCount); } } else if (args[i] == "--temporalLayerCount") { uint8_t temporalLayerCount = EncoderConfig::DEFAULT_TEMPORAL_LAYER_COUNT; if (++i >= argc || !parseUint(args[i], temporalLayerCount)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetTemporalLayerCount(temporalLayerCount); if (verbose) { - printf("Selected temporalLayerCount: %d\n", temporalLayerCount); + VkEncPrintfOut("Selected temporalLayerCount: %u\n", + (unsigned)temporalLayerCount); } } else if (args[i] == "--lastFrameType") { VkVideoGopStructure::FrameType lastFrameType = VkVideoGopStructure::FRAME_TYPE_P; @@ -522,24 +623,24 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) lastFrameType = VkVideoGopStructure::FRAME_TYPE_I; } else { // Invalid frameTypeName - fprintf(stderr, "Invalid frameTypeName: %s\n", frameTypeName.c_str()); + VkEncPrintfErr("Invalid frameTypeName: %s\n", frameTypeName.c_str()); return -1; } i++; // Skip the next argument since it's the frameTypeName value gopStructure.SetLastFrameType(lastFrameType); if (verbose) { - printf("Selected frameTypeName: %s\n", gopStructure.GetFrameTypeName(lastFrameType)); + VkEncPrintfOut("Selected frameTypeName: %s\n", gopStructure.GetFrameTypeName(lastFrameType)); } } else if (args[i] == "--closedGop") { gopStructure.SetClosedGop(); } else if (args[i] == "--qualityLevel") { if (++i >= argc || !parseUint(args[i], qualityLevel)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--usageHints") { if (++i >= argc) { - fprintf(stderr, "Invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invalid parameter for %s\n", args[i - 1].c_str()); return -1; } std::string encodeUsageStr = argv[i]; @@ -555,12 +656,12 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } else if (encodeUsageStr == "conferencing") { encodeUsageHints = VK_VIDEO_ENCODE_USAGE_CONFERENCING_BIT_KHR; } else { - fprintf(stderr, "Invalid encodeUsage: %s\n", encodeUsageStr.c_str()); + VkEncPrintfErr("Invalid encodeUsage: %s\n", encodeUsageStr.c_str()); return -1; } } else if (args[i] == "--contentHints") { if (++i >= argc) { - fprintf(stderr, "Invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invalid parameter for %s\n", args[i - 1].c_str()); return -1; } std::string encodeContentStr = argv[i]; @@ -574,12 +675,12 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } else if (encodeContentStr == "rendered") { encodeContentHints = VK_VIDEO_ENCODE_CONTENT_RENDERED_BIT_KHR; } else { - fprintf(stderr, "Invalid encodeContent: %s\n", encodeContentStr.c_str()); + VkEncPrintfErr("Invalid encodeContent: %s\n", encodeContentStr.c_str()); return -1; } } else if (args[i] == "--tuningMode") { if (++i >= argc) { - fprintf(stderr, "Invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invalid parameter for %s\n", args[i - 1].c_str()); return -1; } std::string tuningModeStr = argv[i]; @@ -595,12 +696,12 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } else if (tuningModeStr == "lossless") { tuningMode = VK_VIDEO_ENCODE_TUNING_MODE_LOSSLESS_KHR; } else { - fprintf(stderr, "Invalid tuningMode: %s\n", tuningModeStr.c_str()); + VkEncPrintfErr("Invalid tuningMode: %s\n", tuningModeStr.c_str()); return -1; } } else if (args[i] == "--rateControlMode") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i-1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i-1].c_str()); return -1; } std::string rc = args[i]; @@ -614,38 +715,38 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_BIT_KHR; } else { // Invalid rateControlMode - fprintf(stderr, "Invalid rateControlMode: %s\n", rc.c_str()); + VkEncPrintfErr("Invalid rateControlMode: %s\n", rc.c_str()); return -1; } } else if (args[i] == "--averageBitrate") { if (++i >= argc || !parseUint(args[i], averageBitrate)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--maxBitrate") { if (++i >= argc || !parseUint(args[i], maxBitrate)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--vbvBufferSize") { if (++i >= argc || !parseUint(args[i], vbvBufferSize)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--qpI") { if (++i >= argc || !parseUint(args[i], constQp.qpIntra)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--qpP") { if (++i >= argc || !parseUint(args[i], constQp.qpInterP)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--qpB") { if (++i >= argc || !parseUint(args[i], constQp.qpInterB)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--disableEncodeParameterOptimizations") { @@ -653,31 +754,31 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } else if (args[i] == "--drmFormatModifierIndex") { int32_t idx = -1; if ((++i >= argc) || !parseUint(args[i], idx)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } drmFormatModifierIndex = idx; } else if (args[i] == "--deviceID") { uint32_t deviceIdVal = 0; if ((++i >= argc) || !parseHex(args[i], deviceIdVal)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } deviceId = static_cast(deviceIdVal); } else if (args[i] == "--deviceUuid") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } size_t size = deviceUUID.StringToUUID(args[i].c_str()); if (size != VK_UUID_SIZE) { - fprintf(stderr,"Invalid deviceUuid format used: %s with size: %zu." + VkEncPrintfErr("Invalid deviceUuid format used: %s with size: %zu." "deviceUuid must be represented by 16 hex (32 bytes) values.", args[i].c_str(), args[i].length()); return -1; } } else if (args[i] == "--qpMap") { if (++i >= argc) { - fprintf(stderr, "Invalid paramter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invalid paramter for %s\n", args[i - 1].c_str()); return -1; } if (args[i] == "deltaQpMap") { @@ -685,13 +786,13 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } else if (args[i] == "emphasisMap") { qpMapMode = EMPHASIS_MAP; } else { - fprintf(stderr, "Invalid quntization map mode %s\n", args[i].c_str()); + VkEncPrintfErr("Invalid quntization map mode %s\n", args[i].c_str()); return -1; } enableQpMap = true; } else if (args[i] == "--qpMapFileName") { if (++i >= argc) { - fprintf(stderr, "Invaid paramter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invaid paramter for %s\n", args[i - 1].c_str()); return -1; } size_t fileSize = qpMapFileHandler.SetFileName(args[i].c_str()); @@ -706,34 +807,34 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) asyncAssembly = false; } else if (args[i] == "--assemblyThreads") { if (++i >= argc) { - fprintf(stderr, "Invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invalid parameter for %s\n", args[i - 1].c_str()); return -1; } uint32_t val = 0; if (!parseUint(args[i], val) || val == 0 || val > 16) { - fprintf(stderr, "Invalid value for --assemblyThreads (1..16)\n"); + VkEncPrintfErr("Invalid value for --assemblyThreads (1..16)\n"); return -1; } assemblyThreadCount = val; } else if (args[i] == "--testOutOfOrderRecording") { // Testing only - don't use this feature for production! - fprintf(stdout, "Warning: %s should only be used for testing!\n", args[i].c_str()); + VkEncPrintfOut("Warning: %s should only be used for testing!\n", args[i].c_str()); enableOutOfOrderRecording = true; } else if (args[i] == "--enableDebugEncoderInputDisplay") { - fprintf(stdout, "Warning: %s Enabling the display for testing encoder input frames!\n", args[i].c_str()); + VkEncPrintfOut("Warning: %s Enabling the display for testing encoder input frames!\n", args[i].c_str()); enableFramePresent = true; } else if (args[i] == "--intraRefreshCycleDuration") { if (++i >= argc || !parseUint(args[i], intraRefreshCycleDuration)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetIntraRefreshCycleDuration(intraRefreshCycleDuration); if (verbose) { - printf("Selected intraRefreshCycleDuration: %d\n", intraRefreshCycleDuration); + VkEncPrintfOut("Selected intraRefreshCycleDuration: %d\n", intraRefreshCycleDuration); } } else if (args[i] == "--intraRefreshMode") { if (++i >= argc) { - fprintf(stderr, "Invalid paramter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("Invalid paramter for %s\n", args[i - 1].c_str()); return -1; } @@ -746,34 +847,34 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } else if (args[i] == "blocks") { intraRefreshMode = REFRESH_BLOCKS; } else { - fprintf(stderr, "Invalid intra-refresh mode %s\n", args[i].c_str()); + VkEncPrintfErr("Invalid intra-refresh mode %s\n", args[i].c_str()); return -1; } } else if (args[i] == "--testIntraRefreshMidway") { // Testing only - don't use this feature for production! - fprintf(stdout, "Warning: %s should only be used for testing!\n", args[i].c_str()); + VkEncPrintfOut("Warning: %s should only be used for testing!\n", args[i].c_str()); if (++i >= argc || !parseUint(args[i], intraRefreshCycleRestartIndex)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetIntraRefreshCycleRestartIndex(intraRefreshCycleRestartIndex); } else if (args[i] == "--testSkipIntraRefreshStart") { // Testing only - don't use this feature for production! - fprintf(stdout, "Warning: %s should only be used for testing!\n", args[i].c_str()); + VkEncPrintfOut("Warning: %s should only be used for testing!\n", args[i].c_str()); if (++i >= argc || !parseUint(args[i], intraRefreshSkippedStartIndex)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } gopStructure.SetIntraRefreshSkippedStartIndex(intraRefreshSkippedStartIndex); #ifdef NV_AQ_GPU_LIB_SUPPORTED } else if (args[i] == "--spatialAQStrength") { if (++i >= argc || !parseFloat(args[i], spatialAQStrength)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } // Valid range is [-1.0, 1.0], but values < -1.0 mean disabled if (spatialAQStrength > 1.0f) { - fprintf(stderr, "spatialAQStrength must be <= 1.0 (use < -1.0 to disable)\n"); + VkEncPrintfErr("spatialAQStrength must be <= 1.0 (use < -1.0 to disable)\n"); return -1; } // Only enable if value is in valid range [-1.0, 1.0] @@ -784,12 +885,12 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } } else if (args[i] == "--temporalAQStrength") { if (++i >= argc || !parseFloat(args[i], temporalAQStrength)) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } // Valid range is [-1.0, 1.0], but values < -1.0 mean disabled if (temporalAQStrength > 1.0f) { - fprintf(stderr, "temporalAQStrength must be <= 1.0 (use < -1.0 to disable)\n"); + VkEncPrintfErr("temporalAQStrength must be <= 1.0 (use < -1.0 to disable)\n"); return -1; } // Only enable if value is in valid range [-1.0, 1.0] @@ -800,7 +901,7 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } } else if (args[i] == "--aqDumpDir") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } aqDumpDir = args[i]; @@ -809,11 +910,11 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) enablePsnrMetrics = 1; } else if (args[i] == "--crcInit") { if (++i >= argc) { - fprintf(stderr, "--crcInit requires a comma-separated list of uint32 values\n"); + VkEncPrintfErr("--crcInit requires a comma-separated list of uint32 values\n"); return -1; } if (!parseCommaSeparatedUint32s(args[i], crcInitValue)) { - fprintf(stderr, "Invalid --crcInit value (use comma-separated decimal or 0x hex uint32s): %s\n", + VkEncPrintfErr("Invalid --crcInit value (use comma-separated decimal or 0x hex uint32s): %s\n", args[i].c_str()); return -1; } @@ -827,12 +928,31 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) } } + { + // Derived defaults + validation, shared with the direct-binding path. + const int finalizeResult = FinalizeConfig(); + if (finalizeResult != 0) { + return finalizeResult; + } + } + + return DoParseArguments(argcount, arglist.data()); +} + +int EncoderConfig::FinalizeConfig(const EncoderConfig::DeviceCapabilities* deviceCaps) +{ + // Extracted ParseArguments tail: runs identically after argv parsing and + // after direct field binding (CreateCodecConfigDirect path). + uint32_t frameCount = 0; + // External frame input mode (IPC/service): no -i file, width/height come from caller. - // The encoder library's InitializeExt path sets numFrames=UINT32_MAX and provides - // frames via SetExternalInputFrame/SubmitExternalFrame. + // The encoder library's InitializeExt path sets a large finite numFrames + // (see the ext streaming config) and provides frames via + // SetExternalInputFrame/SubmitExternalFrame -- there is no UINT32_MAX + // sentinel on this path. if (!inputFileHandler.HasFileName()) { if (input.width == 0 || input.height == 0) { - fprintf(stderr, "An input file (-i) or --inputWidth/--inputHeight must be specified\n"); + VkEncPrintfErr("An input file (-i) or --inputWidth/--inputHeight must be specified\n"); return -1; } // External frame mode: skip file handler setup, use provided dimensions @@ -840,12 +960,12 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) // frameCount stays at the value from --numFrames (or default) } else { if (input.width == 0) { - fprintf(stderr, "The input width must be specified\n"); + VkEncPrintfErr("The input width must be specified\n"); return -1; } if (input.height == 0) { - fprintf(stderr, "The input height must specified\n"); + VkEncPrintfErr("The input height must specified\n"); return -1; } @@ -856,7 +976,7 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) if (startFrame > 0) { if (startFrame >= frameCount) { - std::cout << "startFrame " << startFrame + VkEncOut() << "startFrame " << startFrame << " must be inferior to input file max frame count of " << frameCount << ". Reseting startFrame to 0." << std::endl; startFrame = 0; @@ -869,23 +989,29 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) // File-based input: clamp numFrames to actual file frame count if ((repeatInputFrames == false) && ((numFrames == 0) || (numFrames > (frameCount - startFrame)))) { - std::cout << "numFrames " << numFrames + VkEncOut() << "numFrames " << numFrames << " should be different from zero and inferior to input file max frame count of " << frameCount << ". Using input file frame count." << std::endl; numFrames = frameCount; if (numFrames == 0) { - fprintf(stderr, "No frames found in the input file, frame count is zero. Exit."); + VkEncPrintfErr("No frames found in the input file, frame count is zero. Exit."); return -1; } } } - // External frame input: numFrames comes from --numFrames arg (typically UINT32_MAX for streaming) - - if (!outputFileHandler.HasFileName()) { + // External frame input: numFrames comes from the --numFrames arg; the ext + // streaming path passes a large finite count rather than a sentinel. + + // No default output file in capture mode. SetFileName() fopen()s + // immediately, so the guard belongs here, during parsing, rather than + // after CreateCodecConfig returns: otherwise a host that writes no file + // still gets a 0-byte out.264/.265/.ivf in its working directory, which + // a process confined to a sandbox may not be permitted to create at all. + if (!disableFileOutput && !outputFileHandler.HasFileName()) { const char* defaultOutName = (codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR) ? "out.264" : (codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR) ? "out.265" : "out.ivf"; if (verbose) { - fprintf(stdout, "No output file name provided. Using %s.\n", defaultOutName); + VkEncPrintfOut("No output file name provided. Using %s.\n", defaultOutName); } size_t fileSize = outputFileHandler.SetFileName(defaultOutName); if (fileSize <= 0) { @@ -927,57 +1053,107 @@ int EncoderConfig::ParseArguments(int argc, const char *argv[]) if (minQp == -1) { if (verbose) { - fprintf(stdout, "No QP was provided. Using default value: 20.\n"); + VkEncPrintfOut("No QP was provided. Using default value: 20.\n"); } minQp = 20; + // NOT PUSHED INTO RATE CONTROL, and deliberately so. Every consumer of + // this field reads it only under minQpSet -- which stays clear on this + // path, because nobody asked for a QP -- so an unset minQp reaches the + // driver as "no clamp" rather than as 20. Validating or clamping the + // value here against the device would therefore decide nothing: the + // number is a documented default, not a request. A caller that wants a + // QP clamp sets one, and the codec configs check THAT against the + // device window and refuse it rather than narrowing it silently. } - codecBlockAlignment = H264MbSizeAlignment; // H264 + // Carried, not consumed: nothing in the tree reads codecBlockAlignment. It + // holds the H.264 macroblock size for every codec, which is wrong for H.265 + // CTBs and AV1 superblocks, so a reader would have to derive it from the + // device's VkVideoCapabilitiesKHR::pictureAccessGranularity rather than trust + // this. Left as it stands rather than given a per-codec value that nothing + // would check. + codecBlockAlignment = H264MbSizeAlignment; if (enableQpMap && !qpMapFileHandler.HasFileName() && !enableAQ) { - fprintf(stderr, "No qpMap file was provided."); + VkEncPrintfErr("No qpMap file was provided."); return -1; } if ((intraRefreshMode == REFRESH_NONE && intraRefreshCycleDuration > 0) || (intraRefreshMode != REFRESH_NONE && intraRefreshCycleDuration == 0)) { - fprintf(stderr, "Both --intraRefreshMode and --intraRefreshCycleDuration must be " + VkEncPrintfErr("Both --intraRefreshMode and --intraRefreshCycleDuration must be " "specified to enable intra-refresh.\n"); return -1; } enableIntraRefresh = (intraRefreshMode != REFRESH_NONE) && (intraRefreshCycleDuration > 0); + // Intra refresh is a device FEATURE, not a command-line one. Asking for it on + // a device that does not expose it is refused here, where the request is + // still attributable, rather than inside session creation. Without a device + // to ask, the request stands and the session decides. + if (enableIntraRefresh && (deviceCaps != nullptr) && !deviceCaps->intraRefreshSupported) { + VkEncPrintfErr("Intra refresh was requested, but this device does not expose " + "VkPhysicalDeviceVideoEncodeIntraRefreshFeaturesKHR::videoEncodeIntraRefresh.\n"); + return -1; + } + if (!enableIntraRefresh && intraRefreshCycleRestartIndex > 0) { - fprintf(stderr, "Intra-refresh must be enabled when using --testIntraRefreshMidway\n"); + VkEncPrintfErr("Intra-refresh must be enabled when using --testIntraRefreshMidway\n"); return -1; } if (enableIntraRefresh && intraRefreshCycleRestartIndex >= intraRefreshCycleDuration) { - fprintf(stderr, "The value specified for --testIntraRefreshMidway must be in " + VkEncPrintfErr("The value specified for --testIntraRefreshMidway must be in " "the range [0, intraRefreshCycleDuration-1]\n"); return -1; } if (!enableIntraRefresh && intraRefreshSkippedStartIndex > 0) { - fprintf(stderr, "Intra-refresh must be enabled when using --testSkipIntraRefreshStart\n"); + VkEncPrintfErr("Intra-refresh must be enabled when using --testSkipIntraRefreshStart\n"); return -1; } if (enableIntraRefresh && intraRefreshSkippedStartIndex >= intraRefreshCycleDuration) { - fprintf(stderr, "The value specified for --testSkipIntraRefreshStart must be in " + VkEncPrintfErr("The value specified for --testSkipIntraRefreshStart must be in " "the range [0, intraRefreshCycleDuration-1]\n"); return -1; } if (intraRefreshCycleRestartIndex > 0 && intraRefreshSkippedStartIndex > 0) { - fprintf(stderr, "Combining --testIntraRefreshMidway with --testSkipIntraRefreshStart " + VkEncPrintfErr("Combining --testIntraRefreshMidway with --testSkipIntraRefreshStart " "is not supported\n"); return -1; } - return DoParseArguments(argcount, arglist.data()); + return 0; +} + +VkResult EncoderConfig::CreateCodecConfigDirect( + VkVideoCodecOperationFlagBitsKHR codecOperation, + VkSharedBaseObj& encoderConfig) +{ + switch ((uint32_t)codecOperation) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: { + VkSharedBaseObj config(new EncoderConfigH264()); + encoderConfig = config; + } break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: { + VkSharedBaseObj config(new EncoderConfigH265()); + encoderConfig = config; + } break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: { + VkSharedBaseObj config(new EncoderConfigAV1()); + encoderConfig = config; + } break; + default: + VkEncPrintfErr("[EncoderConfig] CreateCodecConfigDirect: unsupported codec 0x%x\n", + (unsigned)codecOperation); + return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + } + encoderConfig->codec = codecOperation; + return VK_SUCCESS; } VkResult EncoderConfig::CreateCodecConfig(int argc, const char *argv[], @@ -999,8 +1175,8 @@ VkResult EncoderConfig::CreateCodecConfig(int argc, const char *argv[], codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; } else { // Invalid codec - fprintf(stderr, "Invalid codec: %s\n", codecStr.c_str()); - fprintf(stderr, "Supported codecs are: avc, hevc and av1\n"); + VkEncPrintfErr("Invalid codec: %s\n", codecStr.c_str()); + VkEncPrintfErr("Supported codecs are: avc, hevc and av1\n"); return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; } } else if (args[i] == "--help" || args[i] == "-h") { @@ -1019,7 +1195,7 @@ VkResult EncoderConfig::CreateCodecConfig(int argc, const char *argv[], argDump += " "; argDump += argv[a]; } - fprintf(stderr, "[EncoderConfig] H264 ParseArguments failed (ret=%d). argc=%d. Args:%s\n", ret, argc, argDump.c_str()); + VkEncPrintfErr("[EncoderConfig] H264 ParseArguments failed (ret=%d). argc=%d. Args:%s\n", ret, argc, argDump.c_str()); fflush(stderr); // Don't assert — return error so caller can handle gracefully return VK_ERROR_INITIALIZATION_FAILED; @@ -1075,7 +1251,7 @@ VkResult EncoderConfig::CreateCodecConfig(int argc, const char *argv[], return VK_SUCCESS; } else { - fprintf(stderr, "Codec type is not selected\n. Please select it with --codec parameters\n"); + VkEncPrintfErr("Codec type is not selected\n. Please select it with --codec parameters\n"); printHelp(codec); return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; } @@ -1085,13 +1261,13 @@ VkResult EncoderConfig::CreateCodecConfig(int argc, const char *argv[], void EncoderConfig::InitVideoProfile() { - if (encodeBitDepthLuma == 0) { - encodeBitDepthLuma = input.bpp; - } - - if (encodeBitDepthChroma == 0) { - encodeBitDepthChroma = encodeBitDepthLuma; - } + // THE ENCODE BIT DEPTHS ARE ALREADY SET, and deriving them here was the + // defect. This function runs at session creation; InitProfileLevel() runs + // from InitializeParameters(), well before it, and reads the same two + // fields through EncoderConfigH265::GetCpbVclFactor(). Defaulting them + // here meant that read saw zero. They are derived from the input in + // InitializeParameters() now, beside encodeChromaSubsampling, which is the + // only ordering under which every reader sees the same value. // Get the codec-specific profile (already set by InitProfileLevel) uint32_t codecProfile = GetCodecProfile(); @@ -1103,7 +1279,7 @@ void EncoderConfig::InitVideoProfile() encodeUsageInfo.videoContentHints = encodeContentHints; encodeUsageInfo.tuningMode = tuningMode; - if (verbose) fprintf(stderr, "[EncoderConfig] VkVideoEncodeUsageInfoKHR:\n" + if (verbose) VkEncPrintfErr("[EncoderConfig] VkVideoEncodeUsageInfoKHR:\n" " videoUsageHints = 0x%x%s\n" " videoContentHints = 0x%x\n" " tuningMode = %d (%s)\n", @@ -1159,3 +1335,309 @@ bool EncoderConfig::InitRateControl() return true; } + + +// =========================================================================== +// Colour contract for the RGBA->YCbCr preprocess filter. == CC-1 == +// =========================================================================== +// +// THIS TABLE IS DUPLICATED, DELIBERATELY, IN +// chromium/media/gpu/vulkan/vulkan_video_encode_accelerator.cc +// beside ValidateConfig's matrix guardrail. Code cannot be shared across the +// two repositories, so the RULE is written out in both and each names the +// other. IF YOU CHANGE ONE, CHANGE BOTH. Each side also has a table-shaped +// test that walks every code point, so a drift shows up as a test diff in a +// review rather than as a behaviour difference on a GPU +// (library: test/encoder-ext-filter, CaseMatrixDispositionTable; +// Chromium: VulkanVeaMatrixTableTest). +// +// THE RULE, in one sentence: an unnamed matrix is a DESCRIPTION, not a +// request -- derive it, apply it, and signal the matrix actually applied; a +// NAMED matrix we cannot apply is refused. +// +// H.273 name disposition on the FILTER lane signalled +// ----- ------------------- ------------------------------ --------- +// absent no colour descr. BT.709 model, write nothing nothing +// 0 Identity / GBR UNNAMED (see note) derived +// 1 BT.709 honour 1 +// 2 Unspecified UNNAMED -> derive from primaries derived +// 3 reserved refuse -- +// 4 FCC refuse -- +// 5 BT.470BG honour (YCBCR_601) 5 +// 6 SMPTE 170M honour (YCBCR_601) 6 +// 7 SMPTE 240M refuse -- +// 8 YCoCg refuse -- +// 9 BT.2020 NCL honour (YCBCR_2020) 9 +// 10 BT.2020 CL honour, approximated as NCL 10 +// 11 SMPTE 2085 refuse -- +// 12-14 chroma-derived/ICtCp refuse -- +// 255 unknown / INVALID refuse -- +// +// NOTE ON 0, and it is the one row where the two surfaces differ in what +// they can even be ASKED. On Chromium's surface +// VideoColorSpace::MatrixID::RGB == 0 is a real code point with a real +// producer -- gfx::ColorSpace::MatrixID::RGB maps to it, and every sRGB +// canvas / WebGL / WebCodecs RGBA frame carries it. On THIS library's ext +// surface 0 is definitionally "not supplied" and is rewritten to 2 before it +// reaches anything (vulkan_video_encoder_ext.cpp, grep +// `0 IS "NOT SUPPLIED" ON THIS SURFACE`). So the `case 0` arm below cannot be +// reached by any producer in this tree, and code point 0 arriving on the +// public field behaves as UNNAMED -- which is the same disposition the table +// gives it. The arm is kept as defence for a future producer, not deleted, +// because a rule with a hole in it is how the two implementations drifted the +// first time. +// +// THE DIRECT LANE (NV12 / P010 / I420) IS NOT SUBJECT TO ANY OF THIS. The +// caller's Y'CbCr passes through untouched, the VUI describes the caller's own +// data, and the matrix is the caller's business -- see +// CaseYcbcrSessionKeepsAMatrixTheFilterCannotProduce. + +// The derivation for an UNNAMED matrix. Its ORIGIN is the in-tree hardware +// precedent this contract was reconciled against -- Chromium's +// media/gpu/windows/format_utils.cc, grep +// GetEncoderOutputColorSpaceFromInputColorSpace, which is what the D3D12 +// video encoder does PER FRAME when a frame's matrix is RGB: +// +// PrimaryID::SMPTE170M -> MatrixID::SMPTE170M (6 -> 6) +// PrimaryID::BT2020 -> MatrixID::BT2020_NCL (9 -> 9) +// everything else -> MatrixID::BT709 (-> 1) +// +// TWO ROWS BELOW ARE EXTENSIONS, NOT THE PRECEDENT, and they are marked +// because "taken verbatim" was written here first and was false: +// +// * primaries 5 (BT.470BG) also derives 6. 5 and 6 are the 625-line and +// 525-line spellings of the same BT.601 family and share one matrix +// (this file's own `case 5: case 6:` arm maps both to YCBCR_601), so +// sending 5 to BT.709 would be the one arm of the family that disagrees +// with the other. D3D12 does not hit this because gfx::ColorSpace's +// PrimaryID::BT470BG is not what its RGB producers carry. +// * primaries 7 (SMPTE 240M) also derives 6. The 240M and 170M PRIMARIES +// are the same chromaticities; only the transfer differs. Deriving 6 is +// truthful because 6 is the matrix the filter will actually build -- +// matrix 7 is REFUSED by this contract, so it is not an option to derive. +// +// H.273 PRIMARIES CODE POINT 10 IS DELIBERATELY NOT IN THE BT.2020 ARM. It is +// SMPTE ST 428-1 (CIE 1931 XYZ), not BT.2020 -- Chromium's own enum agrees +// (VideoColorSpace::PrimaryID::SMPTEST428_1 = 10). Only MATRIX 10 is BT.2020 +// (constant luminance). Anything with XYZ primaries and no named matrix falls +// to BT.709 with the rest of the "we were not told" set. +// +// PRIMARIES AND TRANSFER ARE NOT TOUCHED by the caller of this: the filter +// performs no primaries conversion and applies no transfer function +// ("This filter applies the colour MATRIX ONLY", VulkanFilterYuvCompute.cpp). +// RANGE IS NOT TOUCHED EITHER, and this does NOT diverge from the D3D12 +// precedent, which preserves it: the caller's video_full_range_flag is carried +// through to the filter unchanged. VkVideoEncoder.cpp turns it into the +// conversion's ycbcrRange (ITU_FULL when set, ITU_NARROW when clear); +// VulkanFilterYuvCompute::InitRGBA2YCBCR reads that field back out as +// isLimitedRange and passes it to GenRgbToYCbCrConversion, which folds the +// narrow-range scale and offset into the conversion ONLY when narrow was asked +// for and emits nothing for full range. The filter can therefore be asked for +// full range, and what it emits is what the VUI advertises. Range is simply +// not this function's business: the derivation decides the MATRIX and nothing +// else. +uint8_t EncoderConfig::DeriveMatrixFromPrimaries(uint8_t primaries) +{ + switch (primaries) { + case 9: // BT.2020 + return 9; + case 5: // BT.470BG -- extension, see above + case 6: // SMPTE 170M -- the precedent's own row + case 7: // SMPTE 240M -- extension, see above + return 6; + default: + return 1; + } +} + +bool EncoderConfig::ResolveRgbToYcbcrMatrix( + VkSamplerYcbcrModelConversion* outModel) +{ + // WHY A SWITCH WITH A DEFAULT ARM IS THE WRONG SHAPE HERE. Such an arm logs + // that the caller "names no matrix the RGBA->YCbCr filter can express; + // converting as BT.709", substitutes BT.709 FOR THE FILTER ONLY, leaves + // matrix_coefficients untouched and returns success. A caller declaring 2 + // (Unspecified) or 7 (SMPTE 240M) then gets BT.709 pixels under a non-BT.709 + // label, on the SUCCESS path, with nothing between it and a conforming + // decoder mis-colouring the result but a line on stderr. "Warn and diverge" + // is not a contract. + // + // Every arm below either makes the label TRUE or REFUSES. + // + // NOT DECLARED (color_description_present_flag == 0) + // Nothing is advertised, so nothing can disagree. Convert as BT.709, + // which is the practical default for HD content and therefore the + // matrix an undeclared input is overwhelmingly likely to want. + // + // NOT because a decoder infers it. The SPEC inference for an absent + // colour description is 2, Unspecified -- H.264 E.2.1 and H.265 E.3.1 + // say so, and AV1's colour config left undeclared means the same -- + // so "what an H.26x decoder assumes" was a de facto convention + // described as a normative rule. The CHOICE is unchanged and is + // right; only its justification was wrong. + // matrix_coefficients is deliberately NOT written: raising it without + // the presence flag would put a value in the config that no bitstream + // ever carries, and the ext probe would then report a colour the + // stream does not have. + // + // 1, 5, 6, 9, 10 EXPRESSIBLE. Use it; the label is already true. + // (5 and 6 are the same matrix; 9 and 10 both map to the BT.2020 + // model this filter has -- see the note on 10 below.) + // + // 2 (Unspecified), declared + // DERIVE the matrix from the declared PRIMARIES, apply it, and + // signal what was applied. The caller declined to NAME a matrix, so + // there is no request here to ignore -- but there IS information, and + // this arm used to throw it away by coercing to BT.709 unconditionally. + // A caller supplying colourPrimaries 9 and transferCharacteristics 16 + // -- exactly what an HDR caller supplies, and the case this arm was + // written for -- got BT.709 chroma under BT.2020 primaries. Now it + // gets a consistent BT.2020 NCL declaration and BT.2020 pixels. + // See DeriveMatrixFromPrimaries for the mapping, which rows come from + // the D3D12 precedent and which two are extensions. + // + // 0 (Identity/GBR) + // REFUSE, and UNREACHABLE THROUGH EVERY PRODUCER IN THIS TREE. 0 + // asserts the samples ARE RGB, which is the one thing a filter whose + // entire job is to make them not-RGB cannot deliver -- so the refusal + // is right. It is also defence rather than enforcement: the ext + // binder maps a 0 on its public field to "not supplied" and writes 2, + // and EncoderConfigJsonLoader.cpp never raises + // color_description_present_flag at all, so nothing in this tree can + // present 0 WITH the presence flag up. Kept anyway; see the CC-1 + // block above for why an unreachable arm is not deleted. + // + // 7 (SMPTE 240M), 3, 4, 8, 11..14 and everything else + // REFUSE. They name a real, DIFFERENT matrix that was asked for and + // cannot be produced: honouring it is impossible and overriding it is + // accepted-and-ignored. A contradiction rather than an omission, so + // it ends initialization with a reason. + if (outModel == nullptr) { + return false; + } + + if (!color_description_present_flag) { + // NOTHING WAS DECLARED ABOUT THE BITSTREAM -- but the caller may still + // have declared its INPUT, and the input's primaries are what the + // matrix is a function of. Deriving from them is strictly better + // informed than the BT.709 default, and it is the same derivation the + // declared-description arm runs. + if (inputColourPrimaries != 0) { + const uint8_t derived = + DeriveMatrixFromPrimaries(inputColourPrimaries); + *outModel = (derived == 9) + ? VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_2020 + : (derived == 6) + ? VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_601 + : VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709; + return true; + } + // A GENUINELY UNDECLARED INPUT still falls to BT.709, and that + // fallback is the regression control for the arm above: a caller that + // chains nothing gets exactly what it got before. + *outModel = VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709; + return true; + } + + switch (matrix_coefficients) { + case 5: // BT.601-7 625 (PAL/SECAM) + case 6: // BT.601-7 525 (NTSC) -- the same matrix + *outModel = VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_601; + return true; + case 9: // BT.2020 non-constant luminance + case 10: // BT.2020 constant luminance + // 10 is ACCEPTED AND APPROXIMATED, and that is stated rather than + // hidden: VkSamplerYcbcrModelConversion has one BT.2020 value and it + // is the non-constant-luminance matrix. Constant luminance is a + // different derivation, not a different set of constants, and no + // Vulkan sampler model expresses it. It is not refused because the + // primaries, the transfer function and the range -- everything else + // the label carries -- are unaffected, and because the encoder that + // consumes this has no constant-luminance path to steer a caller to. + *outModel = VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_2020; + return true; + case 1: // BT.709 + *outModel = VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709; + return true; + case 0: // Identity/GBR -- see the CC-1 block; unreachable here. + VkEncPrintfErr("\nEncoderConfig: matrix_coefficients 0 (Identity/GBR) " + "asserts the samples ARE R'G'B'. The RGBA->YCbCr preprocess " + "filter exists to make them not-R'G'B', so this cannot be " + "honoured and must not be silently overridden. NOTE: no " + "producer in this tree can reach this arm -- the ext binder " + "maps 0 on its public field to \"not supplied\" -- so if you " + "are reading this, a NEW producer has appeared.\n"); + return false; + case 2: { // Unspecified -- derive from primaries, signal what was applied. + // THE INPUT'S PRIMARIES WHEN THEY WERE DECLARED; THE BITSTREAM'S + // OTHERWISE. The matrix is a property of the samples being converted, + // so the input side is the right side to read. The fallback to the + // output field is sound only because this library implements no + // primaries conversion, which makes the two necessarily equal; it is + // written this way so that identity is documented rather than implicit, + // and so the day a primaries conversion exists the wrong reading is not + // already in place. + const uint8_t sourcePrimaries = (inputColourPrimaries != 0) + ? inputColourPrimaries + : colour_primaries; + const uint8_t derived = DeriveMatrixFromPrimaries(sourcePrimaries); + *outModel = (derived == 9) + ? VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_2020 + : (derived == 6) + ? VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_601 + : VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709; + matrix_coefficients = derived; + VkEncPrintfErr("\nEncoderConfig: matrix_coefficients 2 (Unspecified) was " + "declared with a colour description and the RGBA->YCbCr " + "filter must pick a matrix; DERIVED %u from the %s primaries " + "%u, converting with it and SIGNALLING it so the bitstream " + "describes the samples it carries.\n", + (unsigned)derived, + (inputColourPrimaries != 0) ? "INPUT's" : "bitstream's", + (unsigned)sourcePrimaries); + return true; + } + default: + VkEncPrintfErr("\nEncoderConfig: matrix_coefficients %u names no matrix the " + "RGBA->YCbCr preprocess filter can produce. Refusing rather " + "than converting as BT.709 under this label: that would put " + "BT.709 pixels in a stream that claims otherwise. Declare 1 " + "(BT.709), 5/6 (BT.601), 9/10 (BT.2020), or 2 (Unspecified, " + "which is DERIVED from colour_primaries and signalled as the " + "matrix actually applied), or submit YCbCr input that needs " + "no colour conversion.\n", + (unsigned)matrix_coefficients); + return false; + } +} + +void EncoderConfig::ApplyPreprocessFilterChromaSiting() +{ + // The filter dispatches one thread per OUTPUT chroma sample and derives + // that sample with VulkanFilterYuvCompute::GenAverageChromaBlock, a 2x2 + // box average of the block's Cb and Cr. The result therefore sits at the + // CENTRE of the 2x2 luma block in both axes -- MPEG-1 / JPEG siting, + // which is MIDPOINT horizontally and MIDPOINT vertically, and which is + // exactly what xChromaOffset / yChromaOffset already default to. + // + // Only two of the four (x, y) combinations name an H.26x code point: + // MIDPOINT / MIDPOINT -> chroma_sample_loc_type 1 (centre) + // COSITED_EVEN / MIDPOINT -> chroma_sample_loc_type 0 (left, MPEG-2) + // Anything else is left UNSIGNALLED on purpose. An unsignalled H.26x + // stream defaults to type 0, and a wrong signal is worse than an absent + // one: it is the same "declared colour that is not the colour written" + // failure this whole change exists to remove. + // + // This encoder emits frame pictures only, never fields, so the top and + // bottom field types describe the same sample position -- see + // EncoderConfigH264/H265::InitVuiParameters, which write both. + if ((xChromaOffset == VK_CHROMA_LOCATION_MIDPOINT) && + (yChromaOffset == VK_CHROMA_LOCATION_MIDPOINT)) { + chroma_loc_info_present_flag = 1; + chroma_sample_loc_type = 1; + } else if ((xChromaOffset == VK_CHROMA_LOCATION_COSITED_EVEN) && + (yChromaOffset == VK_CHROMA_LOCATION_MIDPOINT)) { + chroma_loc_info_present_flag = 1; + chroma_sample_loc_type = 0; + } +} diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.h b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.h index 1a082f35..ac3bba66 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfig.h @@ -35,9 +35,12 @@ #include "VkCodecUtils/VkVideoRefCountBase.h" #include "VkVideoEncoder/VkVideoEncoderDef.h" #include "VkVideoEncoder/VkVideoGopStructure.h" +#include "VkVideoEncoder/VkVideoEncoderHdrMetadata.h" #include "VkVideoCore/VkVideoCoreProfile.h" #include "VkVideoCore/VulkanVideoCapabilities.h" +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED #include "VkCodecUtils/VulkanFilterYuvCompute.h" +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED #undef max @@ -65,6 +68,70 @@ static VkVideoComponentBitDepthFlagBitsKHR GetComponentBitDepthFlagBits(uint32_t return VK_VIDEO_COMPONENT_BIT_DEPTH_INVALID_KHR; }; +// The colour model the input samples are in. An enum rather than a boolean +// because it is one of several models and a new one is a new enumerator, not +// a second flag. +enum class VkEncColorSpace : uint32_t { + kYCbCr = 0, + kRGB = 1, +}; + +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED +// Which conversion the preprocess compute filter has to perform: the colour +// model the input samples are DECLARED to carry, against the format the device +// accepts as an encode source. +// +// The mechanism choice is the library's -- query the device first, use +// hardware if it exists, compute if it does not. +// This is its second half: once the answer is "compute", this says WHICH +// compute. +// +// EACH SIDE IS ANSWERED BY WHAT THE SURFACE MEANS, not by which format table +// happens to place its enumerant. The packed 4:4:4 Y'CbCr layouts have no +// Vulkan format of their own and ride RGBA ones (PackedYcbcrFormatDesc names +// them), so the enumerant alone cannot tell one of them from an ordinary +// R'G'B' image -- on either side: +// +// - the INPUT side reads the declared colour model. That declaration is the +// only thing that separates a packed Y'CbCr frame from an R'G'B' one, and +// carrying it is what EncoderInputImageParameters::colorSpace is for. +// - the ENCODE SOURCE carries no declaration -- it is a format the device +// named -- so it is read from both Y'CbCr format tables, the multi-planar +// one and the packed 4:4:4 one. Asking only the first calls a packed +// encode source R'G'B' and routes a Y'CbCr input through the inverse +// matrix, which writes R, G and B into the channels the encoder reads as +// Cr, Cb and Y. That produces a full-frame wrong picture and no error at +// all, because the inverse conversion's own output format is the same +// enumerant the packed encode source is spelled with. +// +// YCBCRCOPY for a YCbCr->YCbCr pair is the filter's own contract, from +// VulkanFilterYuvCompute.h: YCBCRCOPY is the compute-based copy that performs +// format, plane-count and bit-depth conversion between two YCbCr formats, +// explicitly contrasted there with the XFER_* transfer modes, which "must +// have matching plane counts". A 3-plane I420 source into a 2-plane NV12 +// destination is exactly that contrast, so it is YCBCRCOPY and not a +// transfer. +static inline VulkanFilterYuvCompute::FilterType VkEncDeriveFilterType( + VkEncColorSpace inputColorSpace, VkFormat encodeSourceFormat) +{ + const bool inputIsYcbcr = (inputColorSpace == VkEncColorSpace::kYCbCr); + const bool outputIsYcbcr = + (YcbcrVkFormatInfo(encodeSourceFormat) != nullptr) || + (PackedYcbcrFormatDesc(encodeSourceFormat) != nullptr); + if (!inputIsYcbcr && outputIsYcbcr) { + return VulkanFilterYuvCompute::RGBA2YCBCR; + } + if (inputIsYcbcr && !outputIsYcbcr) { + return VulkanFilterYuvCompute::YCBCR2RGBA; + } + // YCbCr -> YCbCr, including the identity. YCBCRCOPY is a compute pass + // either way; the plane-count and bit-depth handling it carries is what + // the 3-plane -> 2-plane case needs, and the identity is what a file + // input whose layout the device already accepts takes. + return VulkanFilterYuvCompute::YCBCRCOPY; +} +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + struct EncoderInputImageParameters { EncoderInputImageParameters() @@ -77,6 +144,7 @@ struct EncoderInputImageParameters , planeLayouts{} , fullImageSize(0) , vkFormat(VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM) + , colorSpace(VkEncColorSpace::kYCbCr) {} public: @@ -90,6 +158,19 @@ struct EncoderInputImageParameters uint64_t fullImageSize; VkFormat vkFormat; + /** + * @brief The colour model of the input samples. + * + * kRGB means |vkFormat| is AUTHORITATIVE: VerifyInputs() carries it + * through rather than re-deriving it, and lays the image out as one + * 4-byte-per-pixel plane. kYCbCr means |vkFormat| is DERIVED, from + * chroma subsampling, bit depth and plane count. + * + * The caller of this struct decides which; this header knows nothing of + * the input-format taxonomy and only carries the conclusion. + */ + VkEncColorSpace colorSpace; + bool VerifyInputs() { if ((width == 0) || (height == 0)) { @@ -97,6 +178,33 @@ struct EncoderInputImageParameters return false; } + // RGBA: one interleaved plane of 4 bytes per pixel, and a vkFormat the + // caller already chose. Everything below this block describes a Y'CbCr + // image -- planar, semi-planar or packed, with the chroma planes + // subsampled by |chromaSubsampling| -- and none of that describes an + // RGBA image. It must therefore be reached before the single-plane + // arm below, which reads |chromaSubsampling| and would refuse an RGBA + // image for carrying the default 4:2:0 value it never uses. + if (colorSpace == VkEncColorSpace::kRGB) { + if (vkFormat == VK_FORMAT_UNDEFINED) { + fprintf(stderr, "Input marked RGBA but vkFormat is UNDEFINED!"); + return false; + } + numPlanes = 1; + const uint32_t rgbaRowPitch = 4 * width; + if (planeLayouts[0].rowPitch < rgbaRowPitch) { + planeLayouts[0].rowPitch = rgbaRowPitch; + } + if (planeLayouts[0].size < (planeLayouts[0].rowPitch * height)) { + planeLayouts[0].size = planeLayouts[0].rowPitch * height; + } + planeLayouts[1] = VkSubresourceLayout{}; + planeLayouts[2] = VkSubresourceLayout{}; + fullImageSize = (uint64_t)planeLayouts[0].size; + // vkFormat is DELIBERATELY left alone -- see the field comment. + return true; + } + // Packed 4:4:4 (AYUV / Y410) is SINGLE-plane and interleaved: one 32-bit texel // carries A,Y,Cb,Cr for one pixel, so the whole pixel is 4 bytes regardless of // whether the components are 8-bit (AYUV) or 10-bit (Y410, packed 10-in-32). @@ -778,6 +886,14 @@ struct EncoderConfig : public VkVideoRefCountBase { // Prefer the packed 4:4:4 encode-source format (AYUV / Y410) when the driver // advertises both representations for the profile. // + // THE PREMISE IS UNVERIFIED ON THE CURRENT DRIVER, and saying so is the point: + // it may have been true of an older one. On every driver this project has + // measured, each CAPS_OK (codec, profile) pair returns EXACTLY ONE encode-source + // format, so the "lists BOTH" case below has not been observed and the ordering + // claim with it. It is left standing rather than rewritten into "the driver lists + // one format", because THAT is a per-driver fact and not a contract either -- and + // the option has to keep working on a driver that does list both. + // // For a 4:4:4 profile the driver lists BOTH the 2-plane form and the packed form, // and the listing order is not guaranteed, so anything that takes the driver's // first entry -- or that matches only against the input FILE's layout -- can never @@ -821,6 +937,19 @@ struct EncoderConfig : public VkVideoRefCountBase { int32_t minQp; int32_t maxQp; + // Caller-provided markers for the two fields above, set by the direct + // binder and the --minQp/--maxQp args. The codec configs' derived + // VkVideoEncode*QpKHR members are what rate control actually reads; + // InitDeviceCapabilities uses these markers to tell a requested clamp + // from the -1 sentinel / default-20 fallback. + uint32_t minQpSet : 1; + uint32_t maxQpSet : 1; + // The same marker for constQp, set only by the direct binder, which + // resolves all three QPs before handing the config over: there an + // explicit 0 is a lossless request, not an unset field, and + // InitDeviceCapabilities must not substitute preferredConstantQp for + // it. The argv path leaves this down, keeping 0-means-unset semantics. + uint32_t constQpSet : 1; ConstQpSettings constQp; uint32_t enableQpMap : 1; @@ -862,11 +991,46 @@ struct EncoderConfig : public VkVideoRefCountBase { uint8_t max_dec_frame_buffering; uint8_t chroma_sample_loc_type; + // THE INPUT SIDE. The VuiParameters block above states what the BITSTREAM + // advertises; these state what the caller's own samples carry, as bound + // from the chained VkVideoEncoderInputColourInfo. They are separate + // members rather than a reinterpretation of the block above because they + // answer a different question: the RGBA->Y'CbCr filter's matrix is a + // function of the INPUT's primaries, and only the absence of any primaries + // conversion in this library made reading the output field's value give + // the same answer. + // + // inputColourChainPresent is what distinguishes "absent" from "present and + // zero"; no value field can, because 0 is UNDECLARED on every axis. + uint8_t inputColourPrimaries; + uint8_t inputTransferCharacteristics; + uint8_t inputMatrixCoefficients; + // VkVideoEncoderRangeDeclaration, carried as a plain integer so this + // header takes no dependency on the ext one. + uint8_t inputRange; + uint32_t inputColourChainPresent : 1; + + // HDR10 static metadata. Zero-initialized by its own member + // initializers, so a config that never touches it emits no SEI and no + // metadata OBU -- absence is the default and it is a real absence, not a + // mastering display of all zeros. + EncoderHdrStaticMetadata hdrMetadata; + EncoderInputFileHandler inputFileHandler; EncoderOutputFileHandler outputFileHandler; EncoderQpMapFileHandler qpMapFileHandler; +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // WHICH conversion the preprocess compute filter performs. Owned by the + // LIBRARY, not by any caller: VkVideoEncoder::InitEncoder overwrites it + // from input.colorSpace and the encode-source format the device reported, + // immediately before creating the filter (VkEncDeriveFilterType). No CLI + // flag, no JSON key and no embeddable-API field reaches it, deliberately: + // the mechanism choice belongs inside the library, where the device + // capabilities are known. The initialiser below is only + // what an uninitialised config reads as. VulkanFilterYuvCompute::FilterType filterType; +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED // Adaptive Quantization (AQ) parameters // Range: [-1.0, 1.0] valid, 0.0 = default/midpoint, < -1.0 (e.g., -2.0) = disabled @@ -888,7 +1052,13 @@ struct EncoderConfig : public VkVideoRefCountBase { uint32_t enableHwLoadBalancing : 1; uint32_t noDeviceFallback : 1; uint32_t selectVideoWithComputeQueue : 1; + // Skip fwrite to outputFileHandler when set; the + // encoder captures bitstream bytes in m_capturedBitstreams + // for the Ext API to drain via TryPopCapturedBitstream(). + uint32_t disableFileOutput : 1; +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED uint32_t enablePreprocessComputeFilter : 1; +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED uint32_t repeatInputFrames : 1; // enablePictureRowColReplication // 0: row and column replication is disabled; @@ -905,6 +1075,52 @@ struct EncoderConfig : public VkVideoRefCountBase { std::string crcOutputFileName; bool IsPsnrMetricsEnabled() const { return enablePsnrMetrics != 0; } + + // ---- Colour contract for the RGBA->YCbCr preprocess filter ------------ + // + // Both are defined in VkEncoderConfig.cpp, both are IDEMPOTENT, and both + // are deliberately reachable from two layers: the ext config binder, + // which has no device and can therefore refuse before the caller has + // allocated a frame pool, and VkVideoEncoder::InitEncoder, which is the + // only gate the argv/JSON path passes through. ONE implementation, so the + // two layers cannot answer differently. + + // Resolve matrix_coefficients to the sampler-conversion model the filter + // reads its matrix out of. Returns false when the DECLARED code point + // names a matrix this filter cannot produce, having printed the reason; + // the caller must then fail initialization. May REWRITE + // matrix_coefficients when the declared code point NAMES NO MATRIX + // (2 = Unspecified): it derives one from colour_primaries, applies that, + // and writes it back, so the label the bitstream carries matches the + // pixels that were written. It never rewrites a matrix the caller DID + // name -- that is honoured or refused. + // + // Call ONLY when the filter will actually apply an RGB->YCbCr matrix. A + // YCbCr->YCbCr copy applies no matrix at all, so refusing a code point + // there would reject a configuration that is entirely correct. + bool ResolveRgbToYcbcrMatrix(VkSamplerYcbcrModelConversion* outModel); + + // CC-1: the matrix an UNNAMED colour description resolves to, derived + // from the declared primaries. Static and public so the ext-filter suite + // can walk it as a table against the Chromium side's copy of the same + // rule. See the CC-1 block above ResolveRgbToYcbcrMatrix. + static uint8_t DeriveMatrixFromPrimaries(uint8_t primaries); + + // Signal the chroma siting the filter's 2x2 box average actually + // produces, so the H.26x VUI describes the samples that were written + // rather than the decoder's default. Same call-only-for-the-RGB-arm rule. + void ApplyPreprocessFilterChromaSiting(); + + // Compile-safe accessor for the build-gated preprocess-filter flag: + // callers can branch on it without carrying the gate macro themselves + // (the member only exists when the compute filter is compiled in). + bool IsPreprocessComputeFilterEnabled() const { +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + return enablePreprocessComputeFilter != 0; +#else + return false; +#endif + } int32_t drmFormatModifierIndex; // -1 = disabled (OPTIMAL), >= 0 = index into non-linear modifier list uint64_t selectedDrmFormatModifier; // resolved modifier value (set during InitEncoder) @@ -952,6 +1168,9 @@ struct EncoderConfig : public VkVideoRefCountBase { , frameRateDenominator() , minQp(-1) , maxQp(-1) + , minQpSet(0) + , maxQpSet(0) + , constQpSet(0) , constQp() , enableQpMap(false) , qpMapMode(DELTA_QP_MAP) @@ -990,8 +1209,17 @@ struct EncoderConfig : public VkVideoRefCountBase { , max_num_reorder_frames() , max_dec_frame_buffering() , chroma_sample_loc_type() + , inputColourPrimaries() + , inputTransferCharacteristics() + , inputMatrixCoefficients() + , inputRange() + , inputColourChainPresent() , inputFileHandler() +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // Placeholder only -- InitEncoder derives the real value from the input + // and encode-source formats. See the member's declaration. , filterType(VulkanFilterYuvCompute::YCBCRCOPY) +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED , enableAQ(VK_FALSE) , spatialAQStrength(-2.0f) // < -1.0 means disabled , temporalAQStrength(-2.0f) // < -1.0 means disabled @@ -1006,7 +1234,10 @@ struct EncoderConfig : public VkVideoRefCountBase { , enableHwLoadBalancing(false) , noDeviceFallback(false) , selectVideoWithComputeQueue(false) + , disableFileOutput(false) +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED , enablePreprocessComputeFilter(true) +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED , repeatInputFrames(false) , enablePictureRowColReplication(1) , enableOutOfOrderRecording(false) @@ -1041,6 +1272,46 @@ struct EncoderConfig : public VkVideoRefCountBase { int ParseArguments(int argc, const char *argv[]); + // What only a device can answer, for the decisions in FinalizeConfig() that + // are properties of the hardware rather than of the command line. + // + // Passed as a POINTER that may be null, because FinalizeConfig() runs on two + // paths: argv parsing, which happens before any device exists, and the + // embedding host, which has already probed one. Null means "no device yet" + // -- every field below keeps its command-line answer, which is the behaviour + // the demo has. A caller that HAS probed supplies this and the same function + // reaches a device-correct answer, so there is one tail rather than a second + // one that callers must remember to run. + struct DeviceCapabilities { + // VkPhysicalDeviceVideoEncodeIntraRefreshFeaturesKHR::videoEncodeIntraRefresh. + // Intra refresh is a per-device feature, so a configuration that asks for + // it on a device that lacks it is refused here rather than at the point + // the session is created. + // + // Defaulted, because this structure is filled field by field by a + // caller that has probed a device: a member left out of that + // assignment must read as "the device does not have it", not as + // whatever the stack held. + bool intraRefreshSupported = false; + }; + + // Derived-defaults / validation tail shared by ParseArguments and the + // direct-binding (no-argv) configuration path: input-geometry checks, + // default-output handling, encode-size clamps and defaults, minQp + // default, block alignment, qpMap / intra-refresh validation. + // + // |deviceCaps| is optional; see DeviceCapabilities for what changes when it + // is supplied. Returns 0 on success, -1 on a validation failure. + int FinalizeConfig(const DeviceCapabilities* deviceCaps = nullptr); + + // Codec-typed factory WITHOUT argv parsing: creates the codec subclass + // and sets |codec|. The caller assigns fields directly, then runs + // FinalizeConfig() + InitializeParameters() -- the same pipeline + // CreateCodecConfig drives after ParseArguments. + static VkResult CreateCodecConfigDirect( + VkVideoCodecOperationFlagBitsKHR codecOperation, + VkSharedBaseObj& encoderConfig); + // Load base config from JSON file (encoder_config.schema.json). JSON is processed first; // command-line args passed to ParseArguments override. Returns 0 on success, -1 on error. int LoadFromJsonFile(const char* path); @@ -1188,9 +1459,37 @@ struct EncoderConfig : public VkVideoRefCountBase { } } - // Copy chroma subsampling from input to encoder config + // THE ENCODE-SIDE GEOMETRY, DERIVED IN ONE PLACE AND BEFORE ANYTHING + // READS IT. + // + // encodeChromaSubsampling and encodeBitDepthLuma/Chroma describe the + // BITSTREAM, and the input fields describe the caller's buffer. They + // are separate fields so that the two can differ -- a chroma + // resampler or a device-driven depth downgrade is what would make + // them -- and today the encode side is simply derived from the input + // side, here. + // + // THE DEPTH MUST NOT BE DERIVED IN InitVideoProfile(), which runs at + // session creation, LATER than the codec arms' InitProfileLevel() -- + // and InitProfileLevel is where the level and tier are selected. So + // EncoderConfigH265::GetCpbVclFactor(), which reads + // encodeBitDepthLuma/Chroma for ITU-T H.265 Table A.8's depth term, + // read zero at the level-selection call site and the real depth at + // the InitRateControl() call site: one function, two answers, inside + // one configuration. A 10-bit 4:4:4 stream selected its level with + // the 8-bit factor 2000 and then sized its default CPB with 2500. + // + // The zero-means-unset guards are kept: an explicit encode depth, if + // one is ever set before this runs, is a request and not a default. encodeChromaSubsampling = input.chromaSubsampling; + if (encodeBitDepthLuma == 0) { + encodeBitDepthLuma = input.bpp; + } + if (encodeBitDepthChroma == 0) { + encodeBitDepthChroma = encodeBitDepthLuma; + } + if ((encodeWidth == 0) || (encodeWidth > input.width)) { encodeWidth = input.width; } diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.cpp index 55879976..09a1410c 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.cpp @@ -15,19 +15,20 @@ */ #include "VkVideoEncoder/VkEncoderConfigAV1.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include #define READ_PARAM(i, param, type) { \ if (++i >= argc) { \ - fprintf(stderr, "invalid parameter"); \ + VkEncPrintfErr("invalid parameter"); \ return -1; \ } \ char* _end = nullptr; \ long long _val = strtoll(argv[i], &_end, 10); \ if (_end == argv[i]) { \ - fprintf(stderr, "invalid parameter"); \ + VkEncPrintfErr("invalid parameter"); \ return -1; \ } \ param = static_cast(_val); \ @@ -147,7 +148,7 @@ int EncoderConfigAV1::DoParseArguments(int argc, const char* argv[]) } } else if (args[i] == "--profile"){ if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i-1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i-1].c_str()); return -1; } std::string prfl = args[i]; @@ -159,11 +160,11 @@ int EncoderConfigAV1::DoParseArguments(int argc, const char* argv[]) profile = STD_VIDEO_AV1_PROFILE_PROFESSIONAL; } else { // Invalid profile - fprintf(stderr, "Invalid profile: %s\n", prfl.c_str()); + VkEncPrintfErr("Invalid profile: %s\n", prfl.c_str()); return -1; } } else { - fprintf(stderr, "Unrecognized option: %s\n", argv[i]); + VkEncPrintfErr("Unrecognized option: %s\n", argv[i]); //printAV1Help(); return -1; } @@ -188,6 +189,112 @@ bool EncoderConfigAV1::InitSequenceHeader(StdVideoAV1SequenceHeader *seqHdr, seqHdr->flags.enable_cdef = enableCdef ? 1 : 0; seqHdr->flags.enable_restoration = enableLr ? 1 : 0; + // A1 colour wiring, AV1 arm: emit the sequence header's color_config -- + // AV1's counterpart of the H.26x VUI colour description. AV1 uses the + // same ISO/IEC 23091-4 code points as the H.26x VUI fields, so the + // colour half is a copy, not a conversion. + // + // ALWAYS SUPPLIED, never conditional. Gating it on + // color_description_present_flag alone, or on that OR + // video_signal_type_present_flag, is the wrong SHAPE, because + // color_config is not a colour-description struct that happens to carry + // some other members. It is a STRUCTURAL struct -- BitDepth, + // subsampling_x/y, mono_chrome, chroma_sample_position and color_range + // are all members of it, none of them is conditioned on + // color_description_present_flag in the AV1 syntax, and color_range has + // no absent state at all. Every one of those is a property of the SESSION + // that this config knows and the driver would otherwise have to supply. + // + // WHAT DELEGATION MEANT IN PRACTICE. With pColorConfig null the driver + // writes its own color_config from the session, and on the one driver + // this was measured against it wrote high_bitdepth and mono_chrome + // correctly (a 10-bit AV1 session read back as pix_fmt=yuv420p10le with + // colour unknown/unknown/unknown). That is a fact about that driver, not + // a requirement of the Vulkan specification, and it is not a basis for + // surrendering fields we know. Supplying the struct unconditionally makes + // the structural fields OURS on every driver while leaving the colour + // description genuinely absent -- which is the combination a + // non-declaring caller asked for, and the only one that is + // driver-independent. + // + // color_range IS WRITTEN EVEN WHEN NOTHING WAS DECLARED, and that is not + // a fabrication: color_range is unconditional AV1 syntax with no "absent" + // encoding, so SOME value is in every AV1 bitstream whether we write it + // or the driver does. video_full_range_flag is 0 unless a caller raised + // it, and 0 is studio range, which is what this encoder produces. + av1ColorConfig = {}; + av1ColorConfig.flags.color_range = video_full_range_flag; + if (color_description_present_flag) { + av1ColorConfig.flags.color_description_present_flag = 1; + av1ColorConfig.color_primaries = + (StdVideoAV1ColorPrimaries)colour_primaries; + av1ColorConfig.transfer_characteristics = + (StdVideoAV1TransferCharacteristics)transfer_characteristics; + av1ColorConfig.matrix_coefficients = + (StdVideoAV1MatrixCoefficients)matrix_coefficients; + } else { + // color_description_present_flag == 0 does NOT mean "leave the + // three fields zero": zero is CP_BT_709 / TC_BT_709 / MC_IDENTITY + // in AV1's enums, and MC_IDENTITY additionally asserts the + // samples are RGB. The AV1 specification's own default for an + // absent description is UNSPECIFIED (2) in all three, so write + // that. + // + // UNTESTABLE BY DESIGN, said here so nobody builds a gate for it: + // with color_description_present_flag == 0 the AV1 bitstream OMITS + // all three fields, so no decoder and no bitstream analyser can tell + // 0/0/0 from 2/2/2 in this struct. The assertion is a struct-level + // one or it is nothing. + av1ColorConfig.color_primaries = + STD_VIDEO_AV1_COLOR_PRIMARIES_BT_UNSPECIFIED; + av1ColorConfig.transfer_characteristics = + STD_VIDEO_AV1_TRANSFER_CHARACTERISTICS_UNSPECIFIED; + av1ColorConfig.matrix_coefficients = + STD_VIDEO_AV1_MATRIX_COEFFICIENTS_UNSPECIFIED; + } + // The non-colour members are structural and must match the session: the + // SESSION's chroma subsampling, at the configured bit depth. + // + // HARDCODING 4:2:0 HERE CONTRADICTED InitProfileLevel BELOW, which derives + // seq_profile 1 from 4:4:4 input and 2 from 4:2:2 -- and AV1 6.4.1 gives + // seq_profile 1 subsampling_x == subsampling_y == 0 and seq_profile 2 at + // ten bits or fewer subsampling_x == 1, subsampling_y == 0. The pair was an + // invalid sequence header. The derivation is the codec-correct side, so the + // comment that called 4:2:0 "the only chroma format this encoder admits" + // was the stale one: the input taxonomy routes 4:2:2 and 4:4:4 and the + // profile derivation names their seq_profiles. + // + // 4:2:0 -> (1, 1), 4:2:2 -> (1, 0), 4:4:4 -> (0, 0), which is AV1 5.5.2's + // mapping. mono_chrome keeps its zeroed value from the `= {}` above -- + // there is no monochrome input path -- and is named here because it is one + // of the fields this config owns rather than the driver. + // THE SEQUENCE HEADER'S OWN BitDepth SYNTAX ELEMENT, so it reads the + // encode side like the subsampling two lines below it. Reading + // input.bpp here would put two members of ONE struct on opposite sides of the + // input/encode boundary -- and seq_profile, which AV1 6.4.1 defines + // against this very field, is derived from the encode side in + // InitProfileLevel. + av1ColorConfig.BitDepth = encodeBitDepthLuma; + av1ColorConfig.subsampling_x = + (encodeChromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR) + ? 0 : 1; + av1ColorConfig.subsampling_y = + (encodeChromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR) + ? 1 : 0; + // UNKNOWN, and it is not a gap that can be closed. The RGBA + // preprocess filter sites its chroma at the CENTRE of the 2x2 luma + // block (a box average), which H.26x expresses as + // chroma_sample_loc_type 1 and which AV1 CANNOT express at all: its + // chroma_sample_position offers UNKNOWN, VERTICAL (co-sited + // horizontally, between rows -- MPEG-2) and COLOCATED (top-left) + // only. Signalling VERTICAL to look decisive would assert a siting + // half a chroma sample away from the one written. For DIRECT input + // the siting is the caller's content's and this library never learns + // it, so UNKNOWN is right there too. + av1ColorConfig.chroma_sample_position = + STD_VIDEO_AV1_CHROMA_SAMPLE_POSITION_UNKNOWN; + seqHdr->pColorConfig = &av1ColorConfig; + opInfo->seq_level_idx = level; opInfo->seq_tier = tier; @@ -206,21 +313,21 @@ VkResult EncoderConfigAV1::InitDeviceCapabilities(const VulkanDeviceContext* vkD av1QuantizationMapCapabilities, intraRefreshCapabilities); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Could not get Video Encode Capabilities for AV1. VkResult: " << result << " (0x" << std::hex << result << std::dec << ")" << std::endl; return result; } if (verboseMsg) { - std::cout << "\t\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode capabilities: " << std::endl; - std::cout << "\t\t\t" << "minBitstreamBufferOffsetAlignment: " << videoCapabilities.minBitstreamBufferOffsetAlignment << std::endl; - std::cout << "\t\t\t" << "minBitstreamBufferSizeAlignment: " << videoCapabilities.minBitstreamBufferSizeAlignment << std::endl; - std::cout << "\t\t\t" << "pictureAccessGranularity: " << videoCapabilities.pictureAccessGranularity.width << " x " << videoCapabilities.pictureAccessGranularity.height << std::endl; - std::cout << "\t\t\t" << "minExtent: " << videoCapabilities.minCodedExtent.width << " x " << videoCapabilities.minCodedExtent.height << std::endl; - std::cout << "\t\t\t" << "maxExtent: " << videoCapabilities.maxCodedExtent.width << " x " << videoCapabilities.maxCodedExtent.height << std::endl; - std::cout << "\t\t\t" << "maxDpbSlots: " << videoCapabilities.maxDpbSlots << std::endl; - std::cout << "\t\t\t" << "maxActiveReferencePictures: " << videoCapabilities.maxActiveReferencePictures << std::endl; + VkEncOut() << "\t\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode capabilities: " << std::endl; + VkEncOut() << "\t\t\t" << "minBitstreamBufferOffsetAlignment: " << videoCapabilities.minBitstreamBufferOffsetAlignment << std::endl; + VkEncOut() << "\t\t\t" << "minBitstreamBufferSizeAlignment: " << videoCapabilities.minBitstreamBufferSizeAlignment << std::endl; + VkEncOut() << "\t\t\t" << "pictureAccessGranularity: " << videoCapabilities.pictureAccessGranularity.width << " x " << videoCapabilities.pictureAccessGranularity.height << std::endl; + VkEncOut() << "\t\t\t" << "minExtent: " << videoCapabilities.minCodedExtent.width << " x " << videoCapabilities.minCodedExtent.height << std::endl; + VkEncOut() << "\t\t\t" << "maxExtent: " << videoCapabilities.maxCodedExtent.width << " x " << videoCapabilities.maxCodedExtent.height << std::endl; + VkEncOut() << "\t\t\t" << "maxDpbSlots: " << videoCapabilities.maxDpbSlots << std::endl; + VkEncOut() << "\t\t\t" << "maxActiveReferencePictures: " << videoCapabilities.maxActiveReferencePictures << std::endl; } result = VulkanVideoCapabilities::GetPhysicalDeviceVideoEncodeQualityLevelProperties @@ -228,33 +335,33 @@ VkResult EncoderConfigAV1::InitDeviceCapabilities(const VulkanDeviceContext* vkD qualityLevelProperties, av1QualityLevelProperties); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Could not get Video Encode QualityLevel Properties for AV1. VkResult: " << result << " (0x" << std::hex << result << std::dec << "), qualityLevel: " << qualityLevel << std::endl; return result; } if (verboseMsg) { - std::cout << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode quality level properties: " << std::endl; - std::cout << "\t\t\t" << "preferredRateControlMode : " << qualityLevelProperties.preferredRateControlMode << std::endl; - std::cout << "\t\t\t" << "preferredRateControlLayerCount : " << qualityLevelProperties.preferredRateControlLayerCount << std::endl; - std::cout << "\t\t\t" << "preferredRateControlFlags : " << av1QualityLevelProperties.preferredRateControlFlags << std::endl; - std::cout << "\t\t\t" << "preferredGopFrameCount : " << av1QualityLevelProperties.preferredGopFrameCount << std::endl; - std::cout << "\t\t\t" << "preferredKeyFramePeriod : " << av1QualityLevelProperties.preferredKeyFramePeriod << std::endl; - std::cout << "\t\t\t" << "preferredConsecutiveBipredictiveFrameCount : " << av1QualityLevelProperties.preferredConsecutiveBipredictiveFrameCount << std::endl; - std::cout << "\t\t\t" << "preferredTemporalLayerCount : " << av1QualityLevelProperties.preferredTemporalLayerCount << std::endl; - std::cout << "\t\t\t" << "preferredConstantQIndex.intraQIndex : " << av1QualityLevelProperties.preferredConstantQIndex.intraQIndex << std::endl; - std::cout << "\t\t\t" << "preferredConstantQIndex.predictiveQIndex : " << av1QualityLevelProperties.preferredConstantQIndex.predictiveQIndex << std::endl; - std::cout << "\t\t\t" << "preferredConstantQIndex.bipredictiveQIndex : " << av1QualityLevelProperties.preferredConstantQIndex.bipredictiveQIndex << std::endl; - std::cout << "\t\t\t" << "preferredMaxSingleReferenceCount : " << av1QualityLevelProperties.preferredMaxSingleReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredSingleReferenceNameMask : " << av1QualityLevelProperties.preferredSingleReferenceNameMask << std::endl; - std::cout << "\t\t\t" << "preferredMaxUnidirectionalCompoundReferenceCount : " << av1QualityLevelProperties.preferredMaxUnidirectionalCompoundReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredMaxUnidirectionalCompoundGroup1ReferenceCount : " << av1QualityLevelProperties.preferredMaxUnidirectionalCompoundGroup1ReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredUnidirectionalCompoundReferenceNameMask : " << av1QualityLevelProperties.preferredUnidirectionalCompoundReferenceNameMask << std::endl; - std::cout << "\t\t\t" << "preferredMaxBidirectionalCompoundReferenceCount : " << av1QualityLevelProperties.preferredMaxBidirectionalCompoundReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredMaxBidirectionalCompoundGroup1ReferenceCount : " << av1QualityLevelProperties.preferredMaxBidirectionalCompoundGroup1ReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredMaxBidirectionalCompoundGroup2ReferenceCount : " << av1QualityLevelProperties.preferredMaxBidirectionalCompoundGroup2ReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredBidirectionalCompoundReferenceNameMask : " << av1QualityLevelProperties.preferredBidirectionalCompoundReferenceNameMask << std::endl; + VkEncOut() << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode quality level properties: " << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlMode : " << qualityLevelProperties.preferredRateControlMode << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlLayerCount : " << qualityLevelProperties.preferredRateControlLayerCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlFlags : " << av1QualityLevelProperties.preferredRateControlFlags << std::endl; + VkEncOut() << "\t\t\t" << "preferredGopFrameCount : " << av1QualityLevelProperties.preferredGopFrameCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredKeyFramePeriod : " << av1QualityLevelProperties.preferredKeyFramePeriod << std::endl; + VkEncOut() << "\t\t\t" << "preferredConsecutiveBipredictiveFrameCount : " << av1QualityLevelProperties.preferredConsecutiveBipredictiveFrameCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredTemporalLayerCount : " << av1QualityLevelProperties.preferredTemporalLayerCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQIndex.intraQIndex : " << av1QualityLevelProperties.preferredConstantQIndex.intraQIndex << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQIndex.predictiveQIndex : " << av1QualityLevelProperties.preferredConstantQIndex.predictiveQIndex << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQIndex.bipredictiveQIndex : " << av1QualityLevelProperties.preferredConstantQIndex.bipredictiveQIndex << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxSingleReferenceCount : " << av1QualityLevelProperties.preferredMaxSingleReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredSingleReferenceNameMask : " << av1QualityLevelProperties.preferredSingleReferenceNameMask << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxUnidirectionalCompoundReferenceCount : " << av1QualityLevelProperties.preferredMaxUnidirectionalCompoundReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxUnidirectionalCompoundGroup1ReferenceCount : " << av1QualityLevelProperties.preferredMaxUnidirectionalCompoundGroup1ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredUnidirectionalCompoundReferenceNameMask : " << av1QualityLevelProperties.preferredUnidirectionalCompoundReferenceNameMask << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxBidirectionalCompoundReferenceCount : " << av1QualityLevelProperties.preferredMaxBidirectionalCompoundReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxBidirectionalCompoundGroup1ReferenceCount : " << av1QualityLevelProperties.preferredMaxBidirectionalCompoundGroup1ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxBidirectionalCompoundGroup2ReferenceCount : " << av1QualityLevelProperties.preferredMaxBidirectionalCompoundGroup2ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredBidirectionalCompoundReferenceNameMask : " << av1QualityLevelProperties.preferredBidirectionalCompoundReferenceNameMask << std::endl; } if (rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_FLAG_BITS_MAX_ENUM_KHR) { @@ -269,25 +376,51 @@ VkResult EncoderConfigAV1::InitDeviceCapabilities(const VulkanDeviceContext* vkD if (gopStructure.GetConsecutiveBFrameCount() == CONSECUTIVE_B_FRAME_COUNT_MAX_VALUE) { gopStructure.SetConsecutiveBFrameCount(av1QualityLevelProperties.preferredConsecutiveBipredictiveFrameCount); } - if (constQp.qpIntra == 0) { + // The direct binder resolves all three qindices and marks constQpSet: + // an explicit 0 there is lossless, not unset, and must keep its value. + if (!constQpSet && (constQp.qpIntra == 0)) { constQp.qpIntra = av1QualityLevelProperties.preferredConstantQIndex.intraQIndex; } - if (constQp.qpInterP == 0) { + if (!constQpSet && (constQp.qpInterP == 0)) { constQp.qpInterP = av1QualityLevelProperties.preferredConstantQIndex.predictiveQIndex; } - if (constQp.qpInterB == 0) { + if (!constQpSet && (constQp.qpInterB == 0)) { constQp.qpInterB = av1QualityLevelProperties.preferredConstantQIndex.bipredictiveQIndex; } + // A driver that reports NO preference leaves these at 0, and 0 is not a + // neutral default here -- it is the lowest quantizer index, i.e. very + // nearly lossless, with the bitrate that implies. Floor an unexpressed + // preference to the mid-range index instead. 128 is derived, not picked: + // it is the qindex that maps to libaom quantizer 32 of 63 (the midpoint), + // which is what QP 26 is for H.26x on 0..51. + // + // Only reachable when constQpSet is clear, i.e. the caller specified + // nothing -- an explicit qindex, 0 for lossless included, sets constQpSet + // and never arrives here. + if (!constQpSet) { + static const uint32_t kMidRangeQIndex = 128; + if (constQp.qpIntra == 0) constQp.qpIntra = kMidRangeQIndex; + if (constQp.qpInterP == 0) constQp.qpInterP = kMidRangeQIndex; + if (constQp.qpInterB == 0) constQp.qpInterB = kMidRangeQIndex; + } + return VK_SUCCESS; } void EncoderConfigAV1::InitProfileLevel() { // If profile hasn't been specified, determine it based on bit depth and chroma + // + // BOTH TERMS READ THE ENCODE SIDE. AV1 6.4.1 defines seq_profile over the + // SEQUENCE HEADER's BitDepth, mono_chrome and subsampling_x/y, and + // InitSequenceHeader writes all of those from the encode fields. The + // chroma term already read the encode value; the depth term read + // input.bpp, so seq_profile and the BitDepth it is defined against came + // from opposite sides of the boundary. if (profile == STD_VIDEO_AV1_PROFILE_INVALID) { // PROFESSIONAL is required for 12-bit or 422 - if ((input.bpp > 10) || + if ((encodeBitDepthLuma > 10) || (encodeChromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR)) { profile = STD_VIDEO_AV1_PROFILE_PROFESSIONAL; } diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.h b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.h index cc8fab8b..89aa3b34 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigAV1.h @@ -22,7 +22,13 @@ #define FRAME_ID_BITS 15 #define DELTA_FRAME_ID_BITS 14 -#define ORDER_HINT_BITS 7 +// 8, not 7. The reference-order-hint writer casts to uint8_t (an 8-bit mask) +// while the DPB writer masks by this value, so at 7 the two disagree and a +// ref_order_hint of >= 128 can be emitted for a field that cannot hold it. +// 8 is also what stream consumers that mask order hints with 0xFF expect -- +// with 7, order_hint wraps at frame 128 and such a consumer rejects every +// frame from there on. +#define ORDER_HINT_BITS 8 #define BASE_QIDX_INTRA 114 #define BASE_QIDX_INTER_P 131 @@ -228,6 +234,12 @@ struct EncoderConfigAV1 : public EncoderConfig { bool enableLr{}; bool customLrConfig{}; StdVideoAV1LoopRestoration lrConfig{}; + // Sequence-header colour description, populated by InitSequenceHeader() + // from the base-class colour fields. StdVideoAV1SequenceHeader carries + // colour BY POINTER (pColorConfig), so the storage must outlive the + // sequence header; it lives here, on the config that owns the values and + // outlives the encoder session. + StdVideoAV1ColorConfig av1ColorConfig{}; }; #endif /* VKVIDEOENCODER_VKENCODERCONFIG_AV1_H_ */ diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.cpp index 188f29c6..da0f6a55 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.cpp @@ -15,6 +15,7 @@ */ #include "VkVideoEncoder/VkEncoderConfigH264.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include @@ -41,18 +42,18 @@ int EncoderConfigH264::DoParseArguments(int argc, const char* argv[]) for (int32_t i = 0; i < argc; i++) { if (args[i] == "--slices") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } char* end = nullptr; sliceCount = static_cast(strtol(args[i].c_str(), &end, 10)); if (end == args[i].c_str()) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--profile") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } std::string profileStr = args[i]; @@ -67,11 +68,11 @@ int EncoderConfigH264::DoParseArguments(int argc, const char* argv[]) } else if (profileStr == "high444" || profileStr == "3") { profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE; } else { - fprintf(stderr, "Invalid H.264 profile: %s\n", profileStr.c_str()); + VkEncPrintfErr("Invalid H.264 profile: %s\n", profileStr.c_str()); return -1; } } else { - fprintf(stderr, "Unrecognized option: %s\n", argv[i]); + VkEncPrintfErr("Unrecognized option: %s\n", argv[i]); return -1; } } @@ -183,6 +184,13 @@ EncoderConfigH264::InitVuiParameters(StdVideoH264SequenceParameterSetVui *vui, } vui->flags.chroma_loc_info_present_flag = chroma_loc_info_present_flag; + if (!!chroma_loc_info_present_flag) { + // BOTH FIELDS, and the same value in both -- see the identical note + // in EncoderConfigH265::InitVuiParameters. The flag was plumbed and + // the type was not, so the flag could only ever advertise 0. + vui->chroma_sample_loc_type_top_field = chroma_sample_loc_type; + vui->chroma_sample_loc_type_bottom_field = chroma_sample_loc_type; + } if ((frameRateNumerator > 0) && (frameRateDenominator > 0)) { double frameRate = (double)frameRateNumerator / frameRateDenominator; @@ -253,6 +261,38 @@ EncoderConfigH264::InitVuiParameters(StdVideoH264SequenceParameterSetVui *vui, } } +// H.264 Annex A: entropy_coding_mode_flag is not available in the Baseline +// profile. profile_idc 66 covers Baseline AND Constrained Baseline -- they are +// the same profile_idc, narrowed by constraint_set1_flag, which +// InitSpsPpsParameters() sets for 66 -- and neither admits CABAC. Main (77) +// and every High profile do admit it. +// +// The file already asserts this rule itself, one branch away, in +// InitProfileLevel(): "Upgrade to MAIN profile if using B-frames or CABAC +// entropy coding". That upgrade only ever runs when NO profile was requested, +// so it never sees an explicit --profile baseline, which is how a Baseline +// session could reach the PPS writer with CABAC still set. +// +// Clamping the TOOL and keeping the requested PROFILE is the same shape the +// file already uses for the other per-profile tool restriction it enforces, +// transform_8x8_mode_flag below (High and above only). It is the right way +// round here too: the profile is the caller's explicit request, and is what +// the level and the DPB were sized against, whereas the entropy coder is not +// requested by anyone -- there is no command-line switch for it, it is taken +// from the device's preferredStdEntropyCodingModeFlag. The resulting +// parameter sets then describe honestly what was emitted: profile_idc 66, +// constraint_set0_flag/constraint_set1_flag set, entropy_coding_mode_flag 0. +EncoderConfigH264::EntropyCodingMode +EncoderConfigH264::ConformantEntropyCodingMode(StdVideoH264ProfileIdc profile, + EntropyCodingMode requested) +{ + if ((profile == STD_VIDEO_H264_PROFILE_IDC_BASELINE) && + (requested == ENTROPY_CODING_MODE_CABAC)) { + return ENTROPY_CODING_MODE_CAVLC; + } + return requested; +} + bool EncoderConfigH264::InitSpsPpsParameters(StdVideoH264SequenceParameterSet *sps, StdVideoH264PictureParameterSet *pps, StdVideoH264SequenceParameterSetVui* vui) @@ -368,11 +408,13 @@ bool EncoderConfigH264::InitSpsPpsParameters(StdVideoH264SequenceParameterSet *s pps->flags.transform_8x8_mode_flag = true; } - if (entropyCodingMode == ENTROPY_CODING_MODE_CABAC) { - pps->flags.entropy_coding_mode_flag = true; - } else { - pps->flags.entropy_coding_mode_flag = false; - } + // Derive the emitted flag through the profile rule rather than from + // |entropyCodingMode| directly, so that the PPS is conformant with the + // sps->profile_idc written below on EVERY path reaching this writer, + // including one that never ran InitDeviceCapabilities(). + pps->flags.entropy_coding_mode_flag = + (ConformantEntropyCodingMode(profileIdc, entropyCodingMode) == + ENTROPY_CODING_MODE_CABAC); // Always write out deblocking_filter_control_present_flag pps->flags.deblocking_filter_control_present_flag = true; @@ -413,22 +455,22 @@ VkResult EncoderConfigH264::InitDeviceCapabilities(const VulkanDeviceContext* vk h264QuantizationMapCapabilities, intraRefreshCapabilities); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Could not get Video Encode Capabilities for H264. VkResult: " << result << " (0x" << std::hex << result << std::dec << ")" << std::endl; return result; } if (verboseMsg) { - std::cout << "\t\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode capabilities: " << std::endl; - std::cout << "\t\t\t" << "minBitstreamBufferOffsetAlignment: " << videoCapabilities.minBitstreamBufferOffsetAlignment << std::endl; - std::cout << "\t\t\t" << "minBitstreamBufferSizeAlignment: " << videoCapabilities.minBitstreamBufferSizeAlignment << std::endl; - std::cout << "\t\t\t" << "pictureAccessGranularity: " << videoCapabilities.pictureAccessGranularity.width << " x " << videoCapabilities.pictureAccessGranularity.height << std::endl; - std::cout << "\t\t\t" << "minExtent: " << videoCapabilities.minCodedExtent.width << " x " << videoCapabilities.minCodedExtent.height << std::endl; - std::cout << "\t\t\t" << "maxExtent: " << videoCapabilities.maxCodedExtent.width << " x " << videoCapabilities.maxCodedExtent.height << std::endl; - std::cout << "\t\t\t" << "maxDpbSlots: " << videoCapabilities.maxDpbSlots << std::endl; - std::cout << "\t\t\t" << "maxActiveReferencePictures: " << videoCapabilities.maxActiveReferencePictures << std::endl; - std::cout << "\t\t\t" << "maxBPictureL0ReferenceCount: " << h264EncodeCapabilities.maxBPictureL0ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode capabilities: " << std::endl; + VkEncOut() << "\t\t\t" << "minBitstreamBufferOffsetAlignment: " << videoCapabilities.minBitstreamBufferOffsetAlignment << std::endl; + VkEncOut() << "\t\t\t" << "minBitstreamBufferSizeAlignment: " << videoCapabilities.minBitstreamBufferSizeAlignment << std::endl; + VkEncOut() << "\t\t\t" << "pictureAccessGranularity: " << videoCapabilities.pictureAccessGranularity.width << " x " << videoCapabilities.pictureAccessGranularity.height << std::endl; + VkEncOut() << "\t\t\t" << "minExtent: " << videoCapabilities.minCodedExtent.width << " x " << videoCapabilities.minCodedExtent.height << std::endl; + VkEncOut() << "\t\t\t" << "maxExtent: " << videoCapabilities.maxCodedExtent.width << " x " << videoCapabilities.maxCodedExtent.height << std::endl; + VkEncOut() << "\t\t\t" << "maxDpbSlots: " << videoCapabilities.maxDpbSlots << std::endl; + VkEncOut() << "\t\t\t" << "maxActiveReferencePictures: " << videoCapabilities.maxActiveReferencePictures << std::endl; + VkEncOut() << "\t\t\t" << "maxBPictureL0ReferenceCount: " << h264EncodeCapabilities.maxBPictureL0ReferenceCount << std::endl; } result = VulkanVideoCapabilities::GetPhysicalDeviceVideoEncodeQualityLevelProperties @@ -436,27 +478,27 @@ VkResult EncoderConfigH264::InitDeviceCapabilities(const VulkanDeviceContext* vk qualityLevelProperties, h264QualityLevelProperties); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Could not get Video Encode QualityLevel Properties for H264. VkResult: " << result << " (0x" << std::hex << result << std::dec << "), qualityLevel: " << qualityLevel << std::endl; return result; } if (verboseMsg) { - std::cout << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode quality level properties: " << std::endl; - std::cout << "\t\t\t" << "preferredRateControlMode : " << qualityLevelProperties.preferredRateControlMode << std::endl; - std::cout << "\t\t\t" << "preferredRateControlLayerCount : " << qualityLevelProperties.preferredRateControlLayerCount << std::endl; - std::cout << "\t\t\t" << "preferredRateControlFlags : " << h264QualityLevelProperties.preferredRateControlFlags << std::endl; - std::cout << "\t\t\t" << "preferredGopFrameCount : " << h264QualityLevelProperties.preferredGopFrameCount << std::endl; - std::cout << "\t\t\t" << "preferredIdrPeriod : " << h264QualityLevelProperties.preferredIdrPeriod << std::endl; - std::cout << "\t\t\t" << "preferredConsecutiveBFrameCount : " << h264QualityLevelProperties.preferredConsecutiveBFrameCount << std::endl; - std::cout << "\t\t\t" << "preferredTemporalLayerCount : " << h264QualityLevelProperties.preferredTemporalLayerCount << std::endl; - std::cout << "\t\t\t" << "preferredConstantQp.qpI : " << h264QualityLevelProperties.preferredConstantQp.qpI << std::endl; - std::cout << "\t\t\t" << "preferredConstantQp.qpP : " << h264QualityLevelProperties.preferredConstantQp.qpP << std::endl; - std::cout << "\t\t\t" << "preferredConstantQp.qpB : " << h264QualityLevelProperties.preferredConstantQp.qpB << std::endl; - std::cout << "\t\t\t" << "preferredMaxL0ReferenceCount : " << h264QualityLevelProperties.preferredMaxL0ReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredMaxL1ReferenceCount : " << h264QualityLevelProperties.preferredMaxL1ReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredStdEntropyCodingModeFlag : " << h264QualityLevelProperties.preferredStdEntropyCodingModeFlag << std::endl; + VkEncOut() << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode quality level properties: " << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlMode : " << qualityLevelProperties.preferredRateControlMode << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlLayerCount : " << qualityLevelProperties.preferredRateControlLayerCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlFlags : " << h264QualityLevelProperties.preferredRateControlFlags << std::endl; + VkEncOut() << "\t\t\t" << "preferredGopFrameCount : " << h264QualityLevelProperties.preferredGopFrameCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredIdrPeriod : " << h264QualityLevelProperties.preferredIdrPeriod << std::endl; + VkEncOut() << "\t\t\t" << "preferredConsecutiveBFrameCount : " << h264QualityLevelProperties.preferredConsecutiveBFrameCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredTemporalLayerCount : " << h264QualityLevelProperties.preferredTemporalLayerCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQp.qpI : " << h264QualityLevelProperties.preferredConstantQp.qpI << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQp.qpP : " << h264QualityLevelProperties.preferredConstantQp.qpP << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQp.qpB : " << h264QualityLevelProperties.preferredConstantQp.qpB << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxL0ReferenceCount : " << h264QualityLevelProperties.preferredMaxL0ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxL1ReferenceCount : " << h264QualityLevelProperties.preferredMaxL1ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredStdEntropyCodingModeFlag : " << h264QualityLevelProperties.preferredStdEntropyCodingModeFlag << std::endl; } if (rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_FLAG_BITS_MAX_ENUM_KHR) { @@ -471,23 +513,98 @@ VkResult EncoderConfigH264::InitDeviceCapabilities(const VulkanDeviceContext* vk if (gopStructure.GetConsecutiveBFrameCount() == CONSECUTIVE_B_FRAME_COUNT_MAX_VALUE) { gopStructure.SetConsecutiveBFrameCount(h264QualityLevelProperties.preferredConsecutiveBFrameCount); } - if (constQp.qpIntra == 0) { + // The direct binder resolves all three QPs and marks constQpSet: an + // explicit 0 there is lossless, not unset, and must keep its value. + if (!constQpSet && (constQp.qpIntra == 0)) { constQp.qpIntra = h264QualityLevelProperties.preferredConstantQp.qpI; } - if (constQp.qpInterP == 0) { + if (!constQpSet && (constQp.qpInterP == 0)) { constQp.qpInterP = h264QualityLevelProperties.preferredConstantQp.qpP; } - if (constQp.qpInterB == 0) { + if (!constQpSet && (constQp.qpInterB == 0)) { constQp.qpInterB = h264QualityLevelProperties.preferredConstantQp.qpB; } if (rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR) { minQp = h264QualityLevelProperties.preferredConstantQp; maxQp = h264QualityLevelProperties.preferredConstantQp; } + // Caller-requested QP clamps override the quality-level defaults. The + // base-class ints carry the request (marked by minQpSet/maxQpSet); these + // derived VkVideoEncodeH264QpKHR members are what GetRateControlParameters + // reads -- without this hop a caller's minQp/maxQp never reached rate + // control at all. + if (minQpSet) { + minQp.qpI = minQp.qpP = minQp.qpB = EncoderConfig::minQp; + } + if (maxQpSet) { + maxQp.qpI = maxQp.qpP = maxQp.qpB = EncoderConfig::maxQp; + } + // Device QP window check for caller clamps (the binder already enforced + // the syntactic 0..51 range): with the use flags raised, the spec + // requires the clamp values inside the device's [minQp, maxQp] + // capability window. Reject rather than silently narrow the caller's + // request. + if (minQpSet && + ((EncoderConfig::minQp < h264EncodeCapabilities.minQp) || + (EncoderConfig::minQp > h264EncodeCapabilities.maxQp))) { + VkEncErr() << "[EncoderConfigH264] requested minQp " + << EncoderConfig::minQp + << " is outside the device QP window [" + << h264EncodeCapabilities.minQp << ", " + << h264EncodeCapabilities.maxQp << "]" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if (maxQpSet && + ((EncoderConfig::maxQp < h264EncodeCapabilities.minQp) || + (EncoderConfig::maxQp > h264EncodeCapabilities.maxQp))) { + VkEncErr() << "[EncoderConfigH264] requested maxQp " + << EncoderConfig::maxQp + << " is outside the device QP window [" + << h264EncodeCapabilities.minQp << ", " + << h264EncodeCapabilities.maxQp << "]" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } numRefL0 = h264QualityLevelProperties.preferredMaxL0ReferenceCount; numRefL1 = h264QualityLevelProperties.preferredMaxL1ReferenceCount; numRefFrames = numRefL0 + numRefL1; - entropyCodingMode = h264QualityLevelProperties.preferredStdEntropyCodingModeFlag == VK_TRUE ? ENTROPY_CODING_MODE_CABAC : ENTROPY_CODING_MODE_CAVLC; + // The device's preferred entropy coder, clamped to what the selected + // profile permits. + // + // This assignment is the LAST unconditional writer of |entropyCodingMode|, + // which is why the clamp belongs here. InitProfileLevel() has already run + // by the time InitDeviceCapabilities() is called -- InitializeParameters() + // calls it at config-construction time, this runs later, from + // VkVideoEncoder::InitEncoder() -- so a clamp placed alongside the profile + // decision would be overwritten by this line and would be inert. + const EntropyCodingMode devicePreferred = + (h264QualityLevelProperties.preferredStdEntropyCodingModeFlag == VK_TRUE) + ? ENTROPY_CODING_MODE_CABAC + : ENTROPY_CODING_MODE_CAVLC; + entropyCodingMode = ConformantEntropyCodingMode(profileIdc, devicePreferred); + + if (entropyCodingMode != devicePreferred) { + VkEncOut() << "[EncoderConfigH264] H.264 Baseline (profile_idc " + << static_cast(profileIdc) + << ") does not permit CABAC; encoding with CAVLC instead of " + "the device's preferred entropy coder, so that the " + "bitstream conforms to the requested profile." + << std::endl; + + // The one case where the clamp itself may not be expressible: a device + // that cannot emit entropy_coding_mode_flag = 0 cannot produce a + // conformant Baseline stream at all. Reported, not refused -- + // stdSyntaxFlags is advisory and unevenly populated across drivers, + // and turning a working session into a hard initialisation failure on + // an unverified capability bit is the larger risk. If it is real, the + // encode fails downstream carrying the driver's own error. + if ((h264EncodeCapabilities.stdSyntaxFlags & + VK_VIDEO_ENCODE_H264_STD_ENTROPY_CODING_MODE_FLAG_UNSET_BIT_KHR) == 0) { + VkEncErr() << "[EncoderConfigH264] the device does not advertise " + "ENTROPY_CODING_MODE_FLAG_UNSET; CAVLC may be " + "unsupported here, in which case Baseline is not " + "encodable on this device." << std::endl; + } + } return VK_SUCCESS; } @@ -496,6 +613,17 @@ void EncoderConfigH264::InitProfileLevel() { // 8x8 transform is only supported by High profile and above. // Main and Baseline profiles only support 4x4 transform. + // + // adaptiveTransformMode HAS NO SETTER ON ANY SURFACE, and the narration + // below describes a configuration that cannot occur because of it. The + // field has exactly one write -- its ENABLE constructor default in + // VkEncoderConfigH264.h -- so the first branch is always taken, + // use8x8Transform is always true, and the derivation below can never + // reach BASELINE or MAIN: the 8x8 clause overwrites whatever the + // B-frame / CABAC clause chose. The narration is kept rather than + // deleted because it states the INTENT, and deleting it would delete the + // record that the intent is unreachable. Giving the field a setter is + // what would make it reachable; that is a decision, not a cleanup. bool use8x8Transform = false; if (adaptiveTransformMode == ADAPTIVE_TRANSFORM_ENABLE) { @@ -522,20 +650,39 @@ void EncoderConfigH264::InitProfileLevel() profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH; } - if (input.bpp > 8) { + // THE ENCODE SIDE, ON BOTH AXES, AND THAT IS WHAT A PROFILE IS + // DEFINED OVER. ITU-T H.264 Annex A Table A-1 constrains profile_idc + // against the values the BITSTREAM carries -- the SPS's + // chroma_format_idc and bit_depth_luma_minus8 -- and this file writes + // both of those from encodeChromaSubsampling and encodeBitDepthLuma + // (InitializeSpsRefPicSet's caller, sps->chroma_format_idc and + // sps->bit_depth_*_minus8). A derivation that read the INPUT side + // would select the profile for a picture that is not the one the + // syntax describes: on the first change that makes input and encode + // differ -- a chroma resampler, a device-driven depth downgrade -- + // this arm would pick 244 from a 4:4:4 input and write + // chroma_format_idc 1 from the encode value, inside one SPS. + // + // The fields are separate PRECISELY so the two can differ, so the + // reads are what is fixed and never the fields. + if (encodeBitDepthLuma > 8) { profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH_10; } // 4:2:2 needs High 4:2:2 (122). High (100) and below cannot code - // chroma_format_idc == 2 at all, so without this a 4:2:2 request is refused by - // the driver's profile query rather than silently downgraded. - if (input.chromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR) { + // chroma_format_idc == 2 at all, so without this a 4:2:2 stream would be + // silently downgraded. With it the request carries 122 into the device + // question, and on a device with no 4:2:2 encode profile InitializeExt + // refuses it there -- naming the format, its subsampling and this + // profile -- rather than letting it reach the driver's own capability + // query, which would refuse it while naming none of the three. + if (encodeChromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR) { profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH_422; } // Upgrade to HIGH_444_PREDICTIVE for lossless encoding or 4:4:4 chroma if ((tuningMode == VK_VIDEO_ENCODE_TUNING_MODE_LOSSLESS_KHR) || - (input.chromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR)) { + (encodeChromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR)) { profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE; } } @@ -557,7 +704,15 @@ void EncoderConfigH264::InitProfileLevel() int8_t EncoderConfigH264::InitDpbCount() { - dpbCount = 0; // TODO: What is the need for this? + // Need-based DPB sizing: size the DPB by the references this encoder + // will actually use (numRefFrames = numRefL0 + numRefL1, populated from + // the driver's preferred quality-level properties) instead of leaving + // dpbCount at 0, which selected the LEVEL-MAX DPB below -- 16+1 slots at + // Level >= 5.0 for a sliding-window encode that uses ~3 references, + // wasting ~12 full-resolution DPB images. If the quality-level query has + // not populated numRefFrames yet (0), fall through to the legacy + // level-max sizing, which the DpbSequenceStart() clamp keeps safe. + dpbCount = (numRefFrames > 0) ? numRefFrames : 0; uint8_t levelDpbSize = (uint8_t)(((1024 * levelLimits[levelIdc].maxDPB)) / ((pic_width_in_mbs * pic_height_in_map_units) * 384)); @@ -643,6 +798,26 @@ bool EncoderConfigH264::GetRateControlParameters(VkVideoEncodeRateControlInfoKHR } else { pRateControlLayerInfoH264->minQp = minQp; pRateControlLayerInfoH264->maxQp = maxQp; + // A caller's QP clamp is only visible to the driver when the + // matching use flag is raised: useMinQp/useMaxQp default to + // VK_FALSE and the spec lets a conformant implementation ignore + // the values entirely without them. The only code that ever + // raised these flags was InitRateControl(VkCommandBuffer, + // uint32_t), which has no caller. Source the values from the + // base-class request directly, so the flag and the value travel + // together on every path, device-initialized or not. + if (minQpSet) { + pRateControlLayerInfoH264->useMinQp = VK_TRUE; + pRateControlLayerInfoH264->minQp.qpI = EncoderConfig::minQp; + pRateControlLayerInfoH264->minQp.qpP = EncoderConfig::minQp; + pRateControlLayerInfoH264->minQp.qpB = EncoderConfig::minQp; + } + if (maxQpSet) { + pRateControlLayerInfoH264->useMaxQp = VK_TRUE; + pRateControlLayerInfoH264->maxQp.qpI = EncoderConfig::maxQp; + pRateControlLayerInfoH264->maxQp.qpP = EncoderConfig::maxQp; + pRateControlLayerInfoH264->maxQp.qpB = EncoderConfig::maxQp; + } } pRateControlLayersInfo->averageBitrate = averageBitrate; diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.h b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.h index 3f2a9ac9..e26e60ab 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH264.h @@ -156,6 +156,13 @@ struct EncoderConfigH264 : public EncoderConfig { static void SetAspectRatio(StdVideoH264SequenceParameterSetVui *vui, int32_t width, int32_t height, int32_t darWidth, int32_t darHeight); + // H.264 Baseline (profile_idc 66) has no CABAC -- entropy_coding_mode_flag + // must be 0 there. Returns |requested| clamped to what |profile| permits. + // Pure, static and public so the conformance rule has exactly one + // definition and can be tested without a device. + static EntropyCodingMode ConformantEntropyCodingMode(StdVideoH264ProfileIdc profile, + EntropyCodingMode requested); + virtual VkResult InitializeParameters() override { VkResult result = EncoderConfig::InitializeParameters(); diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH265.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH265.cpp index 593ce9e0..4556d8b6 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH265.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderConfigH265.cpp @@ -15,6 +15,7 @@ */ #include /* sqrt */ +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include "VkVideoEncoder/VkEncoderConfigH265.h" @@ -62,7 +63,14 @@ static void SetupAspectRatio(StdVideoH265SequenceParameterSetVui *vui, uint32_t // From Table A.8 uint32_t EncoderConfigH265::GetCpbVclFactor() { - uint32_t chroma_format_idc = encodeChromaSubsampling; + // encodeChromaSubsampling is a VkVideoChromaSubsamplingFlagBitsKHR -- 0x2 + // for 4:2:0, 0x4 for 4:2:2, 0x8 for 4:4:4 -- and the test below is against + // a chroma_format_idc, which is 1, 2 and 3. Assigning the flag straight + // into the variable made the 4:4:4 arm unreachable for every real input. + // This is the conversion the rest of this file already uses to write + // sps.chroma_format_idc. + uint32_t chroma_format_idc = + FastIntLog2(encodeChromaSubsampling) - 1u; uint32_t bit_depth = std::max(encodeBitDepthLuma, encodeBitDepthChroma); uint32_t baseFactor = (chroma_format_idc == 3) ? (bit_depth >= 10) ? 2500 : 2000 : 1000; // NOTE: Assumes chroma_format_idc is either 1 or 3 uint32_t depthFactor = (bit_depth >= 10) ? ((bit_depth - 10) >> 1) * 500 : 0; // +500 for 12-bit, +1000 for 14-bit, +1500 for 16-bit @@ -76,18 +84,18 @@ int EncoderConfigH265::DoParseArguments(int argc, const char* argv[]) for (int32_t i = 0; i < argc; i++) { if (args[i] == "--slices") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } char* end = nullptr; sliceCount = static_cast(strtol(args[i].c_str(), &end, 10)); if (end == args[i].c_str()) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } } else if (args[i] == "--profile") { if (++i >= argc) { - fprintf(stderr, "invalid parameter for %s\n", args[i - 1].c_str()); + VkEncPrintfErr("invalid parameter for %s\n", args[i - 1].c_str()); return -1; } std::string profileStr = args[i]; @@ -102,11 +110,11 @@ int EncoderConfigH265::DoParseArguments(int argc, const char* argv[]) } else if (profileStr == "scc" || profileStr == "4") { profile = STD_VIDEO_H265_PROFILE_IDC_SCC_EXTENSIONS; } else { - fprintf(stderr, "Invalid H.265 profile: %s\n", profileStr.c_str()); + VkEncPrintfErr("Invalid H.265 profile: %s\n", profileStr.c_str()); return -1; } } else { - fprintf(stderr, "Unrecognized option: %s\n", argv[i]); + VkEncPrintfErr("Unrecognized option: %s\n", argv[i]); return -1; } } @@ -126,22 +134,22 @@ VkResult EncoderConfigH265::InitDeviceCapabilities(const VulkanDeviceContext* vk h265QuantizationMapCapabilities, intraRefreshCapabilities); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Could not get Video Encode Capabilities for HEVC. VkResult: " << result << " (0x" << std::hex << result << std::dec << ")" << std::endl; return result; } if (verboseMsg) { - std::cout << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode capabilities: " << std::endl; - std::cout << "\t\t\t" << "minBitstreamBufferOffsetAlignment: " << videoCapabilities.minBitstreamBufferOffsetAlignment << std::endl; - std::cout << "\t\t\t" << "minBitstreamBufferSizeAlignment: " << videoCapabilities.minBitstreamBufferSizeAlignment << std::endl; - std::cout << "\t\t\t" << "pictureAccessGranularity: " << videoCapabilities.pictureAccessGranularity.width << " x " << videoCapabilities.pictureAccessGranularity.height << std::endl; - std::cout << "\t\t\t" << "minExtent: " << videoCapabilities.minCodedExtent.width << " x " << videoCapabilities.minCodedExtent.height << std::endl; - std::cout << "\t\t\t" << "maxExtent: " << videoCapabilities.maxCodedExtent.width << " x " << videoCapabilities.maxCodedExtent.height << std::endl; - std::cout << "\t\t\t" << "maxDpbSlots: " << videoCapabilities.maxDpbSlots << std::endl; - std::cout << "\t\t\t" << "maxActiveReferencePictures: " << videoCapabilities.maxActiveReferencePictures << std::endl; - std::cout << "\t\t\t" << "maxBPictureL0ReferenceCount: " << h265EncodeCapabilities.maxBPictureL0ReferenceCount << std::endl; + VkEncOut() << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode capabilities: " << std::endl; + VkEncOut() << "\t\t\t" << "minBitstreamBufferOffsetAlignment: " << videoCapabilities.minBitstreamBufferOffsetAlignment << std::endl; + VkEncOut() << "\t\t\t" << "minBitstreamBufferSizeAlignment: " << videoCapabilities.minBitstreamBufferSizeAlignment << std::endl; + VkEncOut() << "\t\t\t" << "pictureAccessGranularity: " << videoCapabilities.pictureAccessGranularity.width << " x " << videoCapabilities.pictureAccessGranularity.height << std::endl; + VkEncOut() << "\t\t\t" << "minExtent: " << videoCapabilities.minCodedExtent.width << " x " << videoCapabilities.minCodedExtent.height << std::endl; + VkEncOut() << "\t\t\t" << "maxExtent: " << videoCapabilities.maxCodedExtent.width << " x " << videoCapabilities.maxCodedExtent.height << std::endl; + VkEncOut() << "\t\t\t" << "maxDpbSlots: " << videoCapabilities.maxDpbSlots << std::endl; + VkEncOut() << "\t\t\t" << "maxActiveReferencePictures: " << videoCapabilities.maxActiveReferencePictures << std::endl; + VkEncOut() << "\t\t\t" << "maxBPictureL0ReferenceCount: " << h265EncodeCapabilities.maxBPictureL0ReferenceCount << std::endl; } result = VulkanVideoCapabilities::GetPhysicalDeviceVideoEncodeQualityLevelProperties @@ -149,26 +157,26 @@ VkResult EncoderConfigH265::InitDeviceCapabilities(const VulkanDeviceContext* vk qualityLevelProperties, h265QualityLevelProperties); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Could not get Video Encode QualityLevel Properties for HEVC. VkResult: " << result << " (0x" << std::hex << result << std::dec << "), qualityLevel: " << qualityLevel << std::endl; return result; } if (verboseMsg) { - std::cout << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode quality level properties: " << std::endl; - std::cout << "\t\t\t" << "preferredRateControlMode : " << qualityLevelProperties.preferredRateControlMode << std::endl; - std::cout << "\t\t\t" << "preferredRateControlLayerCount : " << qualityLevelProperties.preferredRateControlLayerCount << std::endl; - std::cout << "\t\t\t" << "preferredRateControlFlags : " << h265QualityLevelProperties.preferredRateControlFlags << std::endl; - std::cout << "\t\t\t" << "preferredGopFrameCount : " << h265QualityLevelProperties.preferredGopFrameCount << std::endl; - std::cout << "\t\t\t" << "preferredIdrPeriod : " << h265QualityLevelProperties.preferredIdrPeriod << std::endl; - std::cout << "\t\t\t" << "preferredConsecutiveBFrameCount : " << h265QualityLevelProperties.preferredConsecutiveBFrameCount << std::endl; - std::cout << "\t\t\t" << "preferredSubLayerCount : " << h265QualityLevelProperties.preferredSubLayerCount << std::endl; - std::cout << "\t\t\t" << "preferredConstantQp.qpI : " << h265QualityLevelProperties.preferredConstantQp.qpI << std::endl; - std::cout << "\t\t\t" << "preferredConstantQp.qpP : " << h265QualityLevelProperties.preferredConstantQp.qpP << std::endl; - std::cout << "\t\t\t" << "preferredConstantQp.qpB : " << h265QualityLevelProperties.preferredConstantQp.qpB << std::endl; - std::cout << "\t\t\t" << "preferredMaxL0ReferenceCount : " << h265QualityLevelProperties.preferredMaxL0ReferenceCount << std::endl; - std::cout << "\t\t\t" << "preferredMaxL1ReferenceCount : " << h265QualityLevelProperties.preferredMaxL1ReferenceCount << std::endl; + VkEncOut() << "\t\t" << VkVideoCoreProfile::CodecToName(codec) << "encode quality level properties: " << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlMode : " << qualityLevelProperties.preferredRateControlMode << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlLayerCount : " << qualityLevelProperties.preferredRateControlLayerCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredRateControlFlags : " << h265QualityLevelProperties.preferredRateControlFlags << std::endl; + VkEncOut() << "\t\t\t" << "preferredGopFrameCount : " << h265QualityLevelProperties.preferredGopFrameCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredIdrPeriod : " << h265QualityLevelProperties.preferredIdrPeriod << std::endl; + VkEncOut() << "\t\t\t" << "preferredConsecutiveBFrameCount : " << h265QualityLevelProperties.preferredConsecutiveBFrameCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredSubLayerCount : " << h265QualityLevelProperties.preferredSubLayerCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQp.qpI : " << h265QualityLevelProperties.preferredConstantQp.qpI << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQp.qpP : " << h265QualityLevelProperties.preferredConstantQp.qpP << std::endl; + VkEncOut() << "\t\t\t" << "preferredConstantQp.qpB : " << h265QualityLevelProperties.preferredConstantQp.qpB << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxL0ReferenceCount : " << h265QualityLevelProperties.preferredMaxL0ReferenceCount << std::endl; + VkEncOut() << "\t\t\t" << "preferredMaxL1ReferenceCount : " << h265QualityLevelProperties.preferredMaxL1ReferenceCount << std::endl; } if (rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_FLAG_BITS_MAX_ENUM_KHR) { @@ -183,18 +191,51 @@ VkResult EncoderConfigH265::InitDeviceCapabilities(const VulkanDeviceContext* vk if (gopStructure.GetConsecutiveBFrameCount() == CONSECUTIVE_B_FRAME_COUNT_MAX_VALUE) { gopStructure.SetConsecutiveBFrameCount(h265QualityLevelProperties.preferredConsecutiveBFrameCount); } - if (constQp.qpIntra == 0) { + // The direct binder resolves all three QPs and marks constQpSet: an + // explicit 0 there is lossless, not unset, and must keep its value. + if (!constQpSet && (constQp.qpIntra == 0)) { constQp.qpIntra = h265QualityLevelProperties.preferredConstantQp.qpI; } - if (constQp.qpInterP == 0) { + if (!constQpSet && (constQp.qpInterP == 0)) { constQp.qpInterP = h265QualityLevelProperties.preferredConstantQp.qpP; } - if (constQp.qpInterB == 0) { + if (!constQpSet && (constQp.qpInterB == 0)) { constQp.qpInterB = h265QualityLevelProperties.preferredConstantQp.qpB; } numRefL0 = h265QualityLevelProperties.preferredMaxL0ReferenceCount; numRefL1 = h265QualityLevelProperties.preferredMaxL1ReferenceCount; + // Caller-requested QP clamps (see the H.264 counterpart): the derived + // VkVideoEncodeH265QpKHR members feed GetRateControlParameters; the base + // ints only carry the request. + if (minQpSet) { + minQp.qpI = minQp.qpP = minQp.qpB = EncoderConfig::minQp; + } + if (maxQpSet) { + maxQp.qpI = maxQp.qpP = maxQp.qpB = EncoderConfig::maxQp; + } + // Device QP window check for caller clamps -- see the H.264 counterpart. + if (minQpSet && + ((EncoderConfig::minQp < h265EncodeCapabilities.minQp) || + (EncoderConfig::minQp > h265EncodeCapabilities.maxQp))) { + VkEncErr() << "[EncoderConfigH265] requested minQp " + << EncoderConfig::minQp + << " is outside the device QP window [" + << h265EncodeCapabilities.minQp << ", " + << h265EncodeCapabilities.maxQp << "]" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if (maxQpSet && + ((EncoderConfig::maxQp < h265EncodeCapabilities.minQp) || + (EncoderConfig::maxQp > h265EncodeCapabilities.maxQp))) { + VkEncErr() << "[EncoderConfigH265] requested maxQp " + << EncoderConfig::maxQp + << " is outside the device QP window [" + << h265EncodeCapabilities.minQp << ", " + << h265EncodeCapabilities.maxQp << "]" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + return VK_SUCCESS; } @@ -297,6 +338,16 @@ EncoderConfigH265::InitVuiParameters(StdVideoH265SequenceParameterSetVui *vuiInf } vuiInfo->flags.chroma_loc_info_present_flag = chroma_loc_info_present_flag; + if (!!chroma_loc_info_present_flag) { + // BOTH FIELDS, and the same value in both. The flag was plumbed here + // and chroma_sample_loc_type was not, so raising the flag advertised + // a siting of 0 (left / MPEG-2) whatever the config said -- and 0 is + // precisely the wrong answer for the centre-sited chroma the RGBA + // preprocess filter produces. This encoder emits frame pictures only, + // so the top and bottom field types describe one sample position. + vuiInfo->chroma_sample_loc_type_top_field = chroma_sample_loc_type; + vuiInfo->chroma_sample_loc_type_bottom_field = chroma_sample_loc_type; + } vuiInfo->flags.neutral_chroma_indication_flag = 0; vuiInfo->flags.field_seq_flag = 0; @@ -357,9 +408,16 @@ EncoderConfigH265::InitVuiParameters(StdVideoH265SequenceParameterSetVui *vuiInf vuiInfo->pHrdParameters = pHrdParameters; } - // FIXME: chroma_sample_loc_type_top_field to be configured from settings. - vuiInfo->chroma_sample_loc_type_top_field = 0; - vuiInfo->chroma_sample_loc_type_bottom_field = 0; + // (chroma_sample_loc_type_top_field / _bottom_field are written above, + // beside chroma_loc_info_present_flag. DO NOT RE-ZERO THEM HERE. An + // unconditional `= 0` at this point sits two hundred lines below the flag + // that decides whether anyone reads them, so a config carrying type 1 and a + // write above putting 1 in the VUI would be undone before the SPS is built. + // Such a defect is close to invisible: every H.265 row in the encode matrix + // takes the DIRECT or YCbCr-copy path, which signals no siting at all, and a + // device-free assertion that reads the CONFIG rather than the VUI cannot see + // it either. VkEncBoundConfigProbe projects what InitVuiParameters actually + // produces, which closes the second gap.) // display_window_flag vuiInfo->def_disp_win_left_offset = 0; vuiInfo->def_disp_win_right_offset = 0; @@ -507,11 +565,19 @@ void EncoderConfigH265::DetermineLevelTier() void EncoderConfigH265::InitProfileLevel() { // If profile hasn't been specified, determine it based on bit depth and chroma + // + // BOTH TERMS READ THE ENCODE SIDE. ITU-T H.265 Annex A defines + // general_profile_idc over what the bitstream carries, and this file + // writes sps.chroma_format_idc and sps.bit_depth_*_minus8 from + // encodeChromaSubsampling and encodeBitDepthLuma/Chroma. The chroma term + // already read the encode value; the depth term read input.bpp, so the + // two halves of one derivation sat on opposite sides of a boundary that + // exists to let them differ. if (profile == STD_VIDEO_H265_PROFILE_IDC_INVALID) { if (encodeChromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR) { - if (input.bpp == 8) { + if (encodeBitDepthLuma == 8) { profile = STD_VIDEO_H265_PROFILE_IDC_MAIN; - } else if (input.bpp <= 10) { + } else if (encodeBitDepthLuma <= 10) { profile = STD_VIDEO_H265_PROFILE_IDC_MAIN_10; } else { profile = STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS; @@ -620,6 +686,22 @@ bool EncoderConfigH265::GetRateControlParameters(VkVideoEncodeRateControlInfoKHR } else { rcLayerInfoH265->minQp = minQp; rcLayerInfoH265->maxQp = maxQp; + // See the H.264 counterpart: without useMinQp/useMaxQp the driver + // is entitled to ignore the clamp values, and the only setter of + // these flags was dead code. Values come from the base-class + // request so flag and value travel together on every path. + if (minQpSet) { + rcLayerInfoH265->useMinQp = VK_TRUE; + rcLayerInfoH265->minQp.qpI = EncoderConfig::minQp; + rcLayerInfoH265->minQp.qpP = EncoderConfig::minQp; + rcLayerInfoH265->minQp.qpB = EncoderConfig::minQp; + } + if (maxQpSet) { + rcLayerInfoH265->useMaxQp = VK_TRUE; + rcLayerInfoH265->maxQp.qpI = EncoderConfig::maxQp; + rcLayerInfoH265->maxQp.qpP = EncoderConfig::maxQp; + rcLayerInfoH265->maxQp.qpB = EncoderConfig::maxQp; + } } return true; @@ -697,7 +779,7 @@ bool EncoderConfigH265::InitParamameters(VpsH265 *vpsInfo, SpsH265 *spsInfo, spsInfo->sps.pic_height_in_luma_samples = picHeightAlignedToMinCbsY; if (verbose) { - std::cout << "sps.pic_width_in_luma_samples: " << spsInfo->sps.pic_width_in_luma_samples + VkEncOut() << "sps.pic_width_in_luma_samples: " << spsInfo->sps.pic_width_in_luma_samples << ", sps.pic_height_in_luma_samples: " << spsInfo->sps.pic_height_in_luma_samples << ", cuSize: " << (uint32_t)cuSize << ", cuMinSize: " << (uint32_t)cuMinSize << std::endl; } @@ -720,7 +802,7 @@ bool EncoderConfigH265::InitParamameters(VpsH265 *vpsInfo, SpsH265 *spsInfo, spsInfo->sps.log2_diff_max_min_pcm_luma_coding_block_size = (uint8_t)(ctbLog2SizeY - minCbLog2SizeY); if (verbose) { - std::cout << "sps.log2_min_luma_coding_block_size_minus3: " << (uint32_t)spsInfo->sps.log2_min_luma_coding_block_size_minus3 + VkEncOut() << "sps.log2_min_luma_coding_block_size_minus3: " << (uint32_t)spsInfo->sps.log2_min_luma_coding_block_size_minus3 << ", sps.log2_diff_max_min_luma_coding_block_size: " << (uint32_t)spsInfo->sps.log2_diff_max_min_luma_coding_block_size << ", sps.log2_min_luma_transform_block_size_minus2: " << (uint32_t)spsInfo->sps.log2_min_luma_transform_block_size_minus2 << ", sps.log2_diff_max_min_luma_transform_block_size: " << (uint32_t)spsInfo->sps.log2_diff_max_min_luma_transform_block_size @@ -742,7 +824,7 @@ bool EncoderConfigH265::InitParamameters(VpsH265 *vpsInfo, SpsH265 *spsInfo, (spsInfo->sps.conf_win_bottom_offset != 0)); if (verbose) { - std::cout << "sps.conf_win_left_offset: " << spsInfo->sps.conf_win_left_offset + VkEncOut() << "sps.conf_win_left_offset: " << spsInfo->sps.conf_win_left_offset << ", sps.conf_win_right_offset: " << spsInfo->sps.conf_win_right_offset << ", sps.conf_win_top_offset: " << spsInfo->sps.conf_win_top_offset << ", sps.conf_win_bottom_offset: " << spsInfo->sps.conf_win_bottom_offset diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH264.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH264.cpp index 706ed7cb..2e6b63c8 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH264.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH264.cpp @@ -147,7 +147,19 @@ int32_t VkEncDpbH264::DpbSequenceStart(int32_t userDpbSize) DpbDeinit(); - m_max_dpb_size = userDpbSize; + // Clamp to the manager's structural capacity: m_DPB[MAX_DPB_SLOTS + 1] + // holds 16 countable entries plus the current-picture scratch entry. + // Callers pass the Vulkan session slot count, which at H.264 Level >= 5.0 + // is 17 (16 refs + 1 setup). Storing 17 unclamped makes IsDpbFull() -- + // which counts occupancy over i < MAX_DPB_SLOTS -- compare 16 countable + // entries against a threshold of 17: the DPB never reports full, H.264 + // Annex-C eviction never runs, and the short-term reference set freezes at + // the first 16 pictures, so emitted streams predict from pictures a + // conforming decoder's sliding window has already evicted (progressive + // drift, recon-clean and decode-corrupt). Mirrors the H.265 path + // (VkEncDpbH265::DpbSequenceStart), which clamps to + // STD_VIDEO_H265_MAX_DPB_SIZE. + m_max_dpb_size = (userDpbSize > MAX_DPB_SLOTS) ? MAX_DPB_SLOTS : userDpbSize; for (i = 0; i < MAX_DPB_SLOTS + 1; i++) { m_DPB[i] = DpbEntryH264(); diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH265.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH265.cpp index 322afd27..4ea736f9 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH265.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkEncoderDpbH265.cpp @@ -15,6 +15,7 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include @@ -306,7 +307,7 @@ void VkEncDpbH265::ApplyReferencePictureSet(const StdVideoEncodeH265PictureInfo } if (numRefPics > (m_dpbSize - 1)) { - printf("too many reference frames (%d, max is %d)\n", numRefPics, (m_dpbSize - 1)); + VkEncPrintfOut("too many reference frames (%d, max is %d)\n", numRefPics, (m_dpbSize - 1)); } assert(numRefPics <= (int32_t)STD_VIDEO_H265_MAX_NUM_LIST_REF); @@ -410,7 +411,7 @@ void VkEncDpbH265::ApplyReferencePictureSet(const StdVideoEncodeH265PictureInfo } } if (pRefPicSet->ltCurr[i] < 0) - printf("long-term reference picture not available (POC=%d)\n", pocLtCurr[i]); + VkEncPrintfOut("long-term reference picture not available (POC=%d)\n", pocLtCurr[i]); } for (int32_t i = 0; i < m_numPocLtFoll; i++) { @@ -454,7 +455,7 @@ void VkEncDpbH265::ApplyReferencePictureSet(const StdVideoEncodeH265PictureInfo } } if (pRefPicSet->stCurrBefore[i] < 0) - printf("short-term reference picture not available (POC=%d)\n", pocStCurrBefore[i]); + VkEncPrintfOut("short-term reference picture not available (POC=%d)\n", pocStCurrBefore[i]); } for (int32_t i = 0; i < m_numPocStCurrAfter; i++) { @@ -466,7 +467,7 @@ void VkEncDpbH265::ApplyReferencePictureSet(const StdVideoEncodeH265PictureInfo } } if (pRefPicSet->stCurrAfter[i] < 0) - printf("short-term reference picture not available (POC=%d)\n", pocStCurrAfter[i]); + VkEncPrintfOut("short-term reference picture not available (POC=%d)\n", pocStCurrAfter[i]); } for (int32_t i = 0; i < m_numPocStFoll; i++) { diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.cpp index 4fad3ad9..8601c195 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.cpp @@ -15,6 +15,8 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" +#include #include #include #include // For PRIu64, PRId64 @@ -27,13 +29,279 @@ #include "VkVideoEncoder/VkEncoderConfigH264.h" #include "VkVideoEncoder/VkEncoderConfigH265.h" #include "VkVideoEncoder/VkEncoderConfigAV1.h" -#include "VkCodecUtils/VkDrmFormatModifierUtils.h" +#include "VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h" #include "VkCodecUtils/YCbCrConvUtilsCpu.h" #include "VkCodecUtils/VkVideoCrc.h" #ifdef NV_AQ_GPU_LIB_SUPPORTED #include "VulkanAqProcessor.h" #endif // NV_AQ_GPU_LIB_SUPPORTED +VkResult VkVideoEncoder::RequestRateControlUpdate(uint64_t averageBitrate, + uint64_t maxBitrate, + uint32_t frameRateNumerator, + uint32_t frameRateDenominator, + int32_t constQpIntra, + int32_t constQpInterP, + int32_t constQpInterB, + int32_t minQp, + int32_t maxQp) +{ + // Producer side of Reconfigure: callable from any thread (the ext + // Reconfigure entry runs on the caller's sequence). Values are folded + // into the live rate-control state on the encoder thread. + if (averageBitrate == 0) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + // THE DEVICE QP WINDOW, checked HERE and not on the encoder thread, + // because this is the last point that can still answer the caller. A + // clamp folded in and then found unusable could only be dropped + // silently, which is the accepted-and-ignored shape this whole line of + // work exists to remove. The init path refuses an out-of-window clamp + // (EncoderConfigH26x::InitDeviceCapabilities) and so does this. + // + // Only a CARRIED clamp is checked. Zero is carried and means "no + // clamp", so it is exempt: no value reaches the driver, and a device + // window that excluded zero would otherwise make the clamp + // unclearable. + if (m_deviceQpWindowMax != 0) { + if ((minQp > 0) && ((minQp < m_deviceQpWindowMin) || + (minQp > m_deviceQpWindowMax))) { + VkEncErr() << "[VkVideoEncoder] mid-stream minQp " << minQp + << " is outside the device QP window [" + << m_deviceQpWindowMin << ", " << m_deviceQpWindowMax + << "]" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if ((maxQp > 0) && ((maxQp < m_deviceQpWindowMin) || + (maxQp > m_deviceQpWindowMax))) { + VkEncErr() << "[VkVideoEncoder] mid-stream maxQp " << maxQp + << " is outside the device QP window [" + << m_deviceQpWindowMin << ", " << m_deviceQpWindowMax + << "]" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + std::lock_guard lock(m_pendingRateControlMutex); + m_pendingRateControlUpdate.averageBitrate = averageBitrate; + m_pendingRateControlUpdate.maxBitrate = + (maxBitrate != 0) ? maxBitrate : averageBitrate; + m_pendingRateControlUpdate.frameRateNumerator = frameRateNumerator; + m_pendingRateControlUpdate.frameRateDenominator = frameRateDenominator; + // MERGED, not overwritten: a second update that names no quantizer + // must not erase one an earlier update armed but the encoder thread + // has not consumed yet. + if (constQpIntra >= 0) { + m_pendingRateControlUpdate.constQpIntra = constQpIntra; + } + if (constQpInterP >= 0) { + m_pendingRateControlUpdate.constQpInterP = constQpInterP; + } + if (constQpInterB >= 0) { + m_pendingRateControlUpdate.constQpInterB = constQpInterB; + } + // Merged on the same rule as the quantizers, for the same reason: a + // later update that carries no clamp must not erase one an earlier + // update armed and the encoder thread has not consumed yet. + if (minQp >= 0) { + m_pendingRateControlUpdate.minQp = minQp; + } + if (maxQp >= 0) { + m_pendingRateControlUpdate.maxQp = maxQp; + } + m_pendingRateControlArmed = true; + return VK_SUCCESS; +} + +VkResult VkVideoEncoder::ApplyAndGetConstQpForTest(int32_t* pQpIntra, + int32_t* pQpInterP, + int32_t* pQpInterB) +{ + ApplyPendingRateControlUpdate(); + if (!m_encoderConfig) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + *pQpIntra = (int32_t)m_encoderConfig->constQp.qpIntra; + *pQpInterP = (int32_t)m_encoderConfig->constQp.qpInterP; + *pQpInterB = (int32_t)m_encoderConfig->constQp.qpInterB; + return VK_SUCCESS; +} + +VkResult VkVideoEncoder::ApplyAndGetRateControlForTest( + RateControlObservation* pOut) +{ + if (pOut == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + ApplyPendingRateControlUpdate(); + if (!m_encoderConfig) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + *pOut = {}; + // THE LIVE LAYER, not the request. HandleCtrlCmd copies exactly these + // members into the next control command, so this is where a coerced + // maxBitrate and a frame rate that was left alone are visible as the + // numbers the session is actually running on. + pOut->layerAverageBitrate = m_rateControlLayersInfo[0].averageBitrate; + pOut->layerMaxBitrate = m_rateControlLayersInfo[0].maxBitrate; + pOut->layerFrameRateNumerator = + m_rateControlLayersInfo[0].frameRateNumerator; + pOut->layerFrameRateDenominator = + m_rateControlLayersInfo[0].frameRateDenominator; + pOut->constQpIntra = (int32_t)m_encoderConfig->constQp.qpIntra; + pOut->constQpInterP = (int32_t)m_encoderConfig->constQp.qpInterP; + pOut->constQpInterB = (int32_t)m_encoderConfig->constQp.qpInterB; + pOut->configMinQp = m_encoderConfig->minQp; + pOut->configMaxQp = m_encoderConfig->maxQp; + pOut->configMinQpSet = m_encoderConfig->minQpSet ? 1u : 0u; + pOut->configMaxQpSet = m_encoderConfig->maxQpSet ? 1u : 0u; + // The far end of the chain this class owns: what the codec fill + // resolved the clamp to. A config field that moved while this did not + // would be a clamp that never reaches a command. + GetResolvedQpClampForTest(&pOut->resolvedUseMinQp, &pOut->resolvedMinQpI, + &pOut->resolvedUseMaxQp, &pOut->resolvedMaxQpI); + pOut->codecRefreshCount = m_codecRateControlRefreshCount; + return VK_SUCCESS; +} + +void VkVideoEncoder::ApplyPendingRateControlUpdate() +{ + PendingRateControlUpdate update; + { + std::lock_guard lock(m_pendingRateControlMutex); + if (!m_pendingRateControlArmed) { + return; + } + update = m_pendingRateControlUpdate; + m_pendingRateControlArmed = false; + // Consumed: clear the quantizers and the clamps so a later update + // that names none does not re-apply these. + m_pendingRateControlUpdate.constQpIntra = -1; + m_pendingRateControlUpdate.constQpInterP = -1; + m_pendingRateControlUpdate.constQpInterB = -1; + m_pendingRateControlUpdate.minQp = -1; + m_pendingRateControlUpdate.maxQp = -1; + } + // The CONSTANT-QP defaults do not ride a control command at all -- + // they are read per frame out of the encoder config. Writing them + // here is what makes a constant-QP session reconfigurable: this runs + // on the ENCODER THREAD, the same thread EncodeFrameCommon copies + // them on, so the write needs no further synchronisation and no + // frame can observe a half-updated triple. + if (m_encoderConfig) { + if (update.constQpIntra >= 0) { + m_encoderConfig->constQp.qpIntra = (uint32_t)update.constQpIntra; + } + if (update.constQpInterP >= 0) { + m_encoderConfig->constQp.qpInterP = (uint32_t)update.constQpInterP; + } + if (update.constQpInterB >= 0) { + m_encoderConfig->constQp.qpInterB = (uint32_t)update.constQpInterB; + } + // THE QP CLAMPS NEED A SECOND STEP, and this is the difference + // between them and the triple above. Writing minQp/minQpSet moves + // only the REQUEST; what a command carries is the codec + // rate-control layer struct, filled from that request once, at + // codec-init, by EncoderConfig::GetRateControlParameters. Without + // re-invoking that fill the write here would be a config field + // read by nobody -- a change that returns VK_SUCCESS and alters + // nothing, which is the exact defect being removed. + // + // ZERO CLEARS THE CLAMP rather than requesting QP 0. That is the + // reading InitializeExt gives an explicit zero, and the ext + // refuses a clamp change on the one mode where the fill would + // ignore it (DISABLED, which sources its clamp from the + // quality-level constant QP instead) and on the one codec that + // has no QP-unit clamp at all (AV1). + const bool clampChanged = (update.minQp >= 0) || (update.maxQp >= 0); + if (update.minQp >= 0) { + m_encoderConfig->minQp = (update.minQp > 0) ? update.minQp : -1; + m_encoderConfig->minQpSet = (update.minQp > 0) ? 1u : 0u; + } + if (update.maxQp >= 0) { + m_encoderConfig->maxQp = (update.maxQp > 0) ? update.maxQp : -1; + m_encoderConfig->maxQpSet = (update.maxQp > 0) ? 1u : 0u; + } + if (clampChanged) { + // THE FILL IS NOT SIDE-EFFECT FREE on the live layer: it + // rewrites layer[0] bitrate and frame rate from the CONFIG, + // which still holds the values the session was built with. + // Left alone that would silently revert every bitrate and + // frame-rate change a previous Reconfigure had already put in + // force -- and the frame rate would not even be repaired by + // the loop below, which leaves the frame rate alone when the + // update names none. Snapshot and restore, so the refresh can + // only affect what it is here for. + const uint64_t liveAverageBitrate = + m_rateControlLayersInfo[0].averageBitrate; + const uint64_t liveMaxBitrate = + m_rateControlLayersInfo[0].maxBitrate; + const uint32_t liveFrameRateNum = + m_rateControlLayersInfo[0].frameRateNumerator; + const uint32_t liveFrameRateDen = + m_rateControlLayersInfo[0].frameRateDenominator; + RefreshCodecRateControlParameters(); + m_codecRateControlRefreshCount++; + m_rateControlLayersInfo[0].averageBitrate = liveAverageBitrate; + m_rateControlLayersInfo[0].maxBitrate = liveMaxBitrate; + m_rateControlLayersInfo[0].frameRateNumerator = liveFrameRateNum; + m_rateControlLayersInfo[0].frameRateDenominator = liveFrameRateDen; + } + } + // AFTER the refresh, so the values this update carries are the last + // word on the layer whatever the fill recomputed. + for (uint32_t i = 0; i < ARRAYSIZE(m_rateControlLayersInfo); i++) { + m_rateControlLayersInfo[i].averageBitrate = update.averageBitrate; + m_rateControlLayersInfo[i].maxBitrate = update.maxBitrate; + if (update.frameRateNumerator != 0) { + m_rateControlLayersInfo[i].frameRateNumerator = + update.frameRateNumerator; + m_rateControlLayersInfo[i].frameRateDenominator = + (update.frameRateDenominator != 0) + ? update.frameRateDenominator + : 1; + } + } + // The next frame-record emits VK_VIDEO_CODING_CONTROL_ENCODE_RATE_CONTROL + // with the refreshed values. From the HandleCtrlCmd call site the consume + // immediately follows this call; from the EncodeFrameCommon call site it + // is the SAME frame's record, reached further down the same function. + // Either way the flag is a member, so nothing is lost in between. + m_sendRateControlCmd = true; +} + +// Backpressure bound for un-drained captured bitstreams; generous relative +// to the 8-deep assembly pipeline, so it only fires when the consumer has +// genuinely stopped draining. +static constexpr size_t kMaxUnclaimedCapturedBitstreams = 64; + +bool VkVideoEncoder::CanAcceptNewInputFrame() const +{ + if (m_asyncAssemblyEnabled && (m_assemblyQueueCapacity > 0)) { + // One submit can flush a whole reordered run into the assembly queue; + // leave room for the burst so the producer-side Push never reaches its + // condition-variable wait. NOTE the qualifier: that promise holds for + // the EXT path, which gates on this function. The CLI/file path calls + // EnqueueFrame directly with no admission check and can still block in + // VkThreadSafeQueue's producer wait -- it has no non-blocking contract. + // + // GetMaxAssemblyBurst() is virtual because AV1 splices an extra + // show_existing_frame node per reordered insert. + const size_t burst = GetMaxAssemblyBurst(); + if ((m_assemblyQueue.Size() + burst) > m_assemblyQueueCapacity) { + return false; + } + } + { + // Bound instead of unbounded deque growth when the consumer stops + // draining captured bitstreams. + std::lock_guard lock(m_capturedBitstreamsMutex); + if (m_capturedBitstreams.size() >= kMaxUnclaimedCapturedBitstreams) { + return false; + } + } + return true; +} + VkResult VkVideoEncoder::CreateVideoEncoder(const VulkanDeviceContext* vkDevCtx, VkSharedBaseObj& encoderConfig, VkSharedBaseObj& encoder) @@ -59,40 +327,19 @@ VkResult VkVideoEncoder::SelectDrmFormatModifier( VkSharedBaseObj& encoderConfig, VkFormat format, VkImageUsageFlags usage, const VkExtent2D& imageExtent) { -#ifdef __linux__ - VkDrmFormatModifierUtils drmUtils(m_vkDevCtx); - - const VkFormatFeatureFlags required = - VK_FORMAT_FEATURE_VIDEO_ENCODE_INPUT_BIT_KHR | VK_FORMAT_FEATURE_TRANSFER_DST_BIT; - drmUtils.DumpAvailableModifiers(format, required); - - int32_t idx = encoderConfig->drmFormatModifierIndex; - uint64_t selected = drmUtils.SelectModifier( - format, required, idx, - VkDrmFormatModifierUtils::BlockHeightPref::PreferSmallest, - VkDrmFormatModifierUtils::CompressionPref::PreferUncompressed); - - if (selected == 0 && idx >= 0) { - // Explicit index was requested but no suitable modifier found - fprintf(stderr, "DRM modifier index %d: no suitable modifier found\n", idx); - return VK_ERROR_INITIALIZATION_FAILED; - } - if (selected == 0) { - fprintf(stderr, "No non-linear DRM modifiers support VIDEO_ENCODE_SRC + TRANSFER_DST\n"); - return VK_ERROR_FORMAT_NOT_SUPPORTED; + // The modifier machinery is OS-conditional, so it lives in the + // separately-compiled OS adapter -- this file carries no OS-specific + // code. A platform with no adapter arm reports + // VK_ERROR_FEATURE_NOT_PRESENT rather than selecting anything. + (void)usage; (void)imageExtent; + uint64_t selected = 0; + VkResult result = vkenc::OsSelectDrmFormatModifier( + m_vkDevCtx, format, encoderConfig->drmFormatModifierIndex, &selected); + if (result != VK_SUCCESS) { + return result; } - encoderConfig->selectedDrmFormatModifier = selected; - printf("\n=== Selected DRM format modifier ===\n"); - VkDrmFormatModifierUtils::PrintModifierInfo(selected); - printf("\n"); - return VK_SUCCESS; -#else - (void)format; (void)usage; (void)imageExtent; - fprintf(stderr, "DRM format modifiers are only supported on Linux\n"); - return VK_ERROR_FEATURE_NOT_PRESENT; -#endif } VkResult VkVideoEncoder::LoadNextQpMapFrameFromFile(VkSharedBaseObj& encodeFrameInfo) @@ -290,8 +537,59 @@ VkResult VkVideoEncoder::StageInputFrameQpMap(VkSharedBaseObj& encodeFrameInfo) { + // BEFORE THE COPY BELOW, and that ordering is the whole point of + // this call site. The constant-QP triple does not ride a control + // command: the codec reads the per-frame copy taken on the next + // line, and EncodeFrame() has already run by the time HandleCtrlCmd() + // folds a queued update further down. Folding only there left + // constQpI/P/B landing one frame LATE while the six other fields + // Reconfigure carries landed on the frame it preceded -- the + // interface promises the NEXT ENCODED FRAME for all seven. + // + // The fold in HandleCtrlCmd stays rather than moving here. One armed + // flag must keep ONE consumer: ApplyPendingRateControlUpdate() + // early-returns when nothing is armed, so this call is a no-op for + // every frame that has no update waiting, and splitting the fold + // into conditional halves would break the merge-and-clear invariant + // RequestRateControlUpdate() depends on. + ApplyPendingRateControlUpdate(); encodeFrameInfo->constQp = m_encoderConfig->constQp; + // A per-frame quantizer replaces the session's constant QP for this frame + // only. This has to happen after the copy above, which is unconditional -- + // a value written anywhere earlier would be silently overwritten, which is + // exactly how this field came to be dead. + // + // Refused outside constant-QP mode: in CBR/VBR the rate controller owns + // QP, and an override there yields a stream fighting its own bitrate + // target rather than the one the caller asked for. + if (encodeFrameInfo->qpOverrideOnInput >= 0) { + if (m_encoderConfig->rateControlMode == + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR) { + const int32_t qp = encodeFrameInfo->qpOverrideOnInput; + // One value names the frame, so it applies whichever slice type + // this frame turns out to be. + encodeFrameInfo->constQp.qpIntra = qp; + encodeFrameInfo->constQp.qpInterP = qp; + encodeFrameInfo->constQp.qpInterB = qp; + } else { + // ONCE PER PROCESS, and "once" has to be true rather + // than likely: independent sessions reach this branch + // concurrently, and a plain check-then-store lets two + // of them both read false and both print. The flag + // arbitrates emission and publishes nothing else, so + // relaxed ordering is the entire requirement. It does + // not rely on stdio locking, on per-session + // serialization, or on the output being suppressed. + static std::atomic warned{false}; + if (!warned.exchange(true, std::memory_order_relaxed)) { + VkEncErr() << "[Encoder] per-frame qpOverride ignored: this " + "session's rate-control mode is not DISABLED, " + "so the encoder owns QP." << std::endl; + } + } + } + assert(encodeFrameInfo); assert(m_encoderConfig); assert(encodeFrameInfo->srcEncodeImageResource); @@ -303,9 +601,18 @@ VkResult VkVideoEncoder::EncodeFrameCommon(VkSharedBaseObjframeEncodeInputOrderNum = m_encodeInputFrameNum++; // GetPositionInGOP() method returns display position of the picture relative to last key frame picture. + // A caller-forced mid-stream IDR (forceIdrOnInput) takes the same + // "start a new IDR sequence" branch as the first frame / a periodic + // idrPeriod boundary: pictureType becomes FRAME_TYPE_IDR and the GOP + // state machine restarts at this frame, so all downstream IDR handling + // (DPB flush, idr_pic_id, deferred-queue preflush, header emission) + // follows the normal IDR path. + const bool startNewIdrSequence = + (encodeFrameInfo->frameEncodeInputOrderNum == 0) || + encodeFrameInfo->forceIdrOnInput; const bool isIdr = m_encoderConfig->gopStructure.GetPositionInGOP(m_gopState, encodeFrameInfo->gopPosition, - (encodeFrameInfo->frameEncodeInputOrderNum == 0), + startNewIdrSequence, uint32_t(m_encoderConfig->numFrames - encodeFrameInfo->frameEncodeInputOrderNum)); if (isIdr) { assert(encodeFrameInfo->gopPosition.pictureType == VkVideoGopStructure::FRAME_TYPE_IDR); @@ -342,6 +649,51 @@ VkResult VkVideoEncoder::EncodeFrameCommon(VkSharedBaseObjencodeInfo.dstBuffer = encodeFrameInfo->outputBitstreamBuffer->GetBuffer(); encodeFrameInfo->encodeInfo.dstBufferOffset = 0; + // Frame infos are POOL-RECYCLED (m_frameInfoBuffersQueue) + // and Reset() clears bitstreamHeaderBufferSize but NOT this encodeInfo + // field -- without this re-zero, a non-IDR frame recycled through a + // pool node that previously carried an IDR would keep debiting the + // stale header bytes from the RC budget with zero actual prepended + // bytes. Zero it unconditionally beside dstBufferOffset; the gate + // below re-fills both for real capture-mode IDRs. + encodeFrameInfo->encodeInfo.precedingExternallyEncodedBytes = 0; + + // Local patch; not in upstream vk_video_samples. Stop lying to the + // driver's rate controller about the app-prepended per-IDR headers. In + // capture mode (disableFileOutput -- the Chromium in-memory bitstream + // path) the codec-specific EncodeFrame() above filled + // bitstreamHeaderBuffer with SPS/PPS (H.264) / VPS/SPS/PPS (H.265) for + // EVERY IDR, and WriteBitstreamToFile() prepends those bytes to the + // emitted chunk CPU-side -- but the RC never saw them, so every IDR + // overshoots its frame budget by the header size (~40-60 B/IDR, + // compounded by short GOPs). Therefore: + // * reserve the header bytes in the bitstream buffer via + // dstBufferOffset (aligned up to the driver's + // minBitstreamBufferOffsetAlignment), and + // * report them via precedingExternallyEncodedBytes so the RC debits + // this frame's budget. + // The encode-feedback query's bitstreamStartOffset is defined RELATIVE + // to dstBufferOffset, so every readback site adds + // encodeInfo.dstBufferOffset back (no-op while the offset is 0). + // Scope nuance: this reservation covers + // app-prepended H.264/H.265 headers only -- AV1's 2-byte temporal + // delimiter is outside it and rides its own codec-specific + // assembly path, so AV1 is deliberately NOT gated here. + if ((m_encoderConfig->disableFileOutput != 0) && + (encodeFrameInfo->bitstreamHeaderBufferSize > 0) && + ((m_encoderConfig->codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR) || + (m_encoderConfig->codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR))) { + const VkDeviceSize headerBytes = encodeFrameInfo->bitstreamHeaderBufferSize; + VkDeviceSize offsetAlignment = + m_encoderConfig->videoCapabilities.minBitstreamBufferOffsetAlignment; + if (offsetAlignment == 0) { + offsetAlignment = 1; + } + encodeFrameInfo->encodeInfo.dstBufferOffset = + ((headerBytes + offsetAlignment - 1) / offsetAlignment) * offsetAlignment; + encodeFrameInfo->encodeInfo.precedingExternallyEncodedBytes = + (uint32_t)headerBytes; + } #ifdef NV_AQ_GPU_LIB_SUPPORTED if (m_aqAnalyzes) { @@ -370,7 +722,7 @@ VkResult VkVideoEncoder::EncodeFrameCommon(VkSharedBaseObj aqPendingTemporalBiDiSlot; // Check if we have a pending slot from a previous frame that needs deferred temporal processing - printf("[ProcessFrame] Calling FindFreeBuffer with flags=0x%x\n", prepareFlags); + VkEncPrintfOut("[ProcessFrame] Calling FindFreeBuffer with flags=0x%x\n", prepareFlags); encodeFrameInfo->aqProcessorSlot = m_aqAnalyzes->FindFreeAqProcessorSlot(prepareFlags, pCtxConfig->codecType, @@ -383,10 +735,10 @@ VkResult VkVideoEncoder::EncodeFrameCommon(VkSharedBaseObjaqProcessorSlot == nullptr) { - printf("[ProcessFrame] ERROR: FindFreeBuffer returned nullptr\n"); + VkEncPrintfOut("[ProcessFrame] ERROR: FindFreeBuffer returned nullptr\n"); return VK_ERROR_OUT_OF_POOL_MEMORY; } - printf("[ProcessFrame] Slot allocated, %p\n", encodeFrameInfo->aqProcessorSlot.get()); + VkEncPrintfOut("[ProcessFrame] Slot allocated, %p\n", encodeFrameInfo->aqProcessorSlot.get()); encodeFrameInfo->aqProcessorSlot->UpdateGop(encodeFrameInfo->frameEncodeInputOrderNum, encodeFrameInfo->gopPosition, isIdr); @@ -441,9 +793,11 @@ VkResult VkVideoEncoder::EncodeFrameCommon(VkSharedBaseObj imageResource; @@ -486,6 +855,22 @@ VkResult VkVideoEncoder::WrapExternalImage( subresRange.layerCount = 1; VkSharedBaseObj imageView; + if (isLinearStagingSource) { + // Transfer-only staging source: an image with only TRANSFER usage is + // not view-compatible (VUID-VkImageViewCreateInfo-image-04441), and + // nothing in the staging copy path (TransitionImageLayout + + // CopyLinearToOptimalImage) consumes a VkImageView -- both use the + // raw VkImage handle. Create a view-less wrapper. + result = VkImageResourceView::Create( + m_vkDevCtx, imageResource, subresRange, imageView); + if (result != VK_SUCCESS) { + VkEncErr() << "[WrapExternalImage] view creation failed: " + << result << std::endl; + return result; + } + return VulkanVideoImagePoolNode::CreateExternal( + m_vkDevCtx, imageView, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, outNode); + } if (mpInfo) { // Multiplanar: the combined NV12 view needs VIDEO_ENCODE_SRC + TRANSFER // (no SAMPLED — that would require a YCbCr conversion). @@ -503,7 +888,7 @@ VkResult VkVideoEncoder::WrapExternalImage( result = VkImageResourceView::Create(m_vkDevCtx, imageResource, subresRange, imageView); } if (result != VK_SUCCESS) { - fprintf(stderr, "[WrapExternalImage] VkImageResourceView::Create failed: %d\n", result); + VkEncPrintfErr("[WrapExternalImage] VkImageResourceView::Create failed: %d\n", result); return result; } if (result != VK_SUCCESS) { @@ -516,17 +901,15 @@ VkResult VkVideoEncoder::WrapExternalImage( return result; } -VkResult VkVideoEncoder::SetExternalInputFrame( +void VkVideoEncoder::StampExternalFrameInfo( VkSharedBaseObj& encodeFrameInfo, - VkImage externalImage, - VkDeviceMemory externalMemory, - VkFormat format, - uint32_t width, uint32_t height, - VkImageTiling tiling, VkImageLayout srcImageCurrentLayout, uint64_t frameId, uint64_t pts, bool isLastFrame, + bool forceIdr, + int32_t qpOverride, + ExternalInputResidency residency, uint32_t waitSemaphoreCount, const VkSemaphore* pWaitSemaphores, const uint64_t* pWaitSemaphoreValues, @@ -535,8 +918,6 @@ VkResult VkVideoEncoder::SetExternalInputFrame( const VkSemaphore* pSignalSemaphores, const uint64_t* pSignalSemaphoreValues) { - assert(encodeFrameInfo); - // ============================================= // 1. Replicate LoadNextFrame() bookkeeping // ============================================= @@ -549,16 +930,38 @@ VkResult VkVideoEncoder::SetExternalInputFrame( // ============================================= encodeFrameInfo->isExternalInput = true; encodeFrameInfo->srcExternalImageLayout = srcImageCurrentLayout; + // Keep the CALLER's frame id on the node; the captured-bitstream + // FIFO is keyed by it (see WriteBitstreamToFile), not by the internal + // encode-input counter, so a partially failed submission or a future + // reordering GOP cannot desynchronize capture routing. + encodeFrameInfo->externalFrameId = frameId; + // Latch the caller's mid-stream IDR request for EncodeFrameCommon. + encodeFrameInfo->forceIdrOnInput = forceIdr; + encodeFrameInfo->qpOverrideOnInput = qpOverride; + // Latch the caller-declared queue-family ownership for + // StageInputFrame's barrier construction. + encodeFrameInfo->externalInputResidency = residency; encodeFrameInfo->inputWaitSemaphores.clear(); encodeFrameInfo->inputWaitSemaphoreValues.clear(); + // With no caller-provided masks this vector stays EMPTY, and each + // submission that injects these waits falls back to the stage of its + // own consuming operation: TRANSFER on the staging-copy submit + // (SubmitStagedInputFrame), VIDEO_ENCODE on the direct encode submit + // (SubmitVideoCodingCmds). A single stored default cannot be right for + // both -- TRANSFER on the encode submit leaves vkCmdEncodeVideoKHR + // outside the wait's scope, so the encode could read the input before + // the producer signaled, and VIDEO_ENCODE is not supported on a staging + // submit routed to a dedicated TRANSFER queue. encodeFrameInfo->inputWaitDstStageMasks.clear(); for (uint32_t i = 0; i < waitSemaphoreCount; i++) { encodeFrameInfo->inputWaitSemaphores.push_back(pWaitSemaphores[i]); encodeFrameInfo->inputWaitSemaphoreValues.push_back( pWaitSemaphoreValues ? pWaitSemaphoreValues[i] : 0); - encodeFrameInfo->inputWaitDstStageMasks.push_back( - pWaitDstStageMasks ? pWaitDstStageMasks[i] : VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR); + if (pWaitDstStageMasks != nullptr) { + encodeFrameInfo->inputWaitDstStageMasks.push_back( + pWaitDstStageMasks[i]); + } } encodeFrameInfo->inputSignalSemaphores.clear(); @@ -568,6 +971,38 @@ VkResult VkVideoEncoder::SetExternalInputFrame( encodeFrameInfo->inputSignalSemaphoreValues.push_back( pSignalSemaphoreValues ? pSignalSemaphoreValues[i] : 0); } +} + +VkResult VkVideoEncoder::SetExternalInputFrame( + VkSharedBaseObj& encodeFrameInfo, + VkImage externalImage, + VkDeviceMemory externalMemory, + VkFormat format, + uint32_t width, uint32_t height, + VkImageTiling tiling, + VkImageLayout srcImageCurrentLayout, + uint64_t frameId, + uint64_t pts, + bool isLastFrame, + bool forceIdr, + int32_t qpOverride, + ExternalInputResidency residency, + uint32_t waitSemaphoreCount, + const VkSemaphore* pWaitSemaphores, + const uint64_t* pWaitSemaphoreValues, + const VkPipelineStageFlags2* pWaitDstStageMasks, + uint32_t signalSemaphoreCount, + const VkSemaphore* pSignalSemaphores, + const uint64_t* pSignalSemaphoreValues) +{ + assert(encodeFrameInfo); + + StampExternalFrameInfo(encodeFrameInfo, srcImageCurrentLayout, frameId, + pts, isLastFrame, forceIdr, qpOverride, residency, + waitSemaphoreCount, pWaitSemaphores, + pWaitSemaphoreValues, pWaitDstStageMasks, + signalSemaphoreCount, pSignalSemaphores, + pSignalSemaphoreValues); // ============================================= // 3. Determine input path @@ -632,6 +1067,43 @@ VkResult VkVideoEncoder::SetExternalInputFrame( // ============================================= // Path A: Direct encode (zero-copy) // ============================================= + + // THE DIRECT SUBMIT'S WAIT CAPACITY, REFUSED WHILE A STATUS CAN + // STILL REACH THE CALLER. + // + // A directly encodable frame skips staging, so the waits it carries + // are assembled into the fixed array in SubmitVideoCodingCmds. That + // assembly refuses an over-capacity frame as well, but it runs on + // the encoder's frame-processing path, and the direct path issues + // its submit from the deferred-GOP flush -- which under B-frame + // reordering is a LATER call than the one that admitted the frame. + // A refusal reached there is a frame that is never submitted, never + // completes and raises no completion edge, while its caller holds a + // success it can only wait on. + // + // So the count is checked HERE, before the image is wrapped and + // before any encoder resource is taken, where the refusal is the + // value the entry point returns. VK_ERROR_TOO_MANY_OBJECTS is what + // it is: a fixed array, named, exceeded. + // + // The bound is the ARRAY, not a smaller number of caller waits. + // Eight caller waits fit exactly and must keep working. + // + // Scoped to the direct path deliberately. The staging lane + // assembles its waits into a growable vector and carries no such + // bound, so refusing a ninth wait there would invent a limit the + // library does not have. + // + // Not exact in one direction: a QP-map command buffer and the + // hardware load-balancing timeline each spend a further slot that + // is not decided yet at this point, so a frame carrying those can + // still be refused by the assembly rather than here. This check + // removes the common case from the silent-failure class without + // pretending to knowledge it does not have. + if (waitSemaphoreCount > kDirectSubmitSemaphoreCapacity) { + return VK_ERROR_TOO_MANY_OBJECTS; + } + // Wrap external image and set directly as srcEncodeImageResource. // No staging, no copy, no filter. VkResult result = WrapExternalImage( @@ -641,6 +1113,9 @@ VkResult VkVideoEncoder::SetExternalInputFrame( if (result != VK_SUCCESS) { return result; } + // The encode will read the caller's imported image directly, so it -- + // not a staging copy -- is what must be acquired from FOREIGN. + encodeFrameInfo->srcEncodeImageIsExternal = true; // Go directly to EncodeFrameCommon (skip StageInputFrame). // Wait/signal semaphores will be injected into SubmitVideoCodingCmds @@ -659,6 +1134,33 @@ VkResult VkVideoEncoder::SetExternalInputFrame( return result; } + // Which rung of the adaptation ladder this frame needs. The legacy + // arm is handed the frame's format directly, so it can answer here: + // a format that differs from the encode-source format the device + // reported has to be CONVERTED, which is the compute tier; a format + // that matches needs at most a re-tile, which is the transfer tier. + // That is the adaptation ladder's ordering applied to one frame -- + // and it preserves today's behaviour exactly for the shape this arm + // actually carries, a LINEAR NV12 host-staged image, which matches + // and so still takes the copy. + // + // "Differs from the encode format" is necessary but NOT sufficient, + // and the second clause is what makes this honest. The filter was + // built for exactly ONE input format -- EncoderConfig::input.vkFormat, + // which is what InitEncoder handed VulkanFilterYuvCompute::Create -- + // and its shader's plane count, bit depth and bindings are fixed to + // it. A frame in some OTHER non-encode format routed here would bind + // its planes into a shader that expects a different layout. The legacy + // arm cannot check any of the facts the registered arm checks (the + // wrapper it just built is view-less for LINEAR, and fabricates create + // flags for OPTIMAL), so it must not claim more than the format + // equality it can actually see; the ext layer refuses the classes this + // leaves unserved (SubmitExternalFrameCommon), rather than silently + // degrading them to the copy. + encodeFrameInfo->externalInputViaFilter = + (format != m_imageInFormat) && + (format == m_encoderConfig->input.vkFormat); + // StageInputFrame will: // - Acquire srcEncodeImageResource from pool // - Record the copy/filter command buffer @@ -668,6 +1170,143 @@ VkResult VkVideoEncoder::SetExternalInputFrame( } } +VkResult VkVideoEncoder::SetExternalInputFrameWithNode( + VkSharedBaseObj& encodeFrameInfo, + VkSharedBaseObj& node, + uint64_t registrationId, + bool directlyEncodable, + bool routeViaFilter, + VkImageLayout srcImageCurrentLayout, + bool srcLayoutIsExplicit, + uint64_t frameId, + uint64_t pts, + bool isLastFrame, + bool forceIdr, + int32_t qpOverride, + ExternalInputResidency residency, + uint32_t waitSemaphoreCount, + const VkSemaphore* pWaitSemaphores, + const uint64_t* pWaitSemaphoreValues, + const VkPipelineStageFlags2* pWaitDstStageMasks, + uint32_t signalSemaphoreCount, + const VkSemaphore* pSignalSemaphores, + const uint64_t* pSignalSemaphoreValues) +{ + assert(encodeFrameInfo); + assert(node); + + StampExternalFrameInfo(encodeFrameInfo, srcImageCurrentLayout, frameId, + pts, isLastFrame, forceIdr, qpOverride, residency, + waitSemaphoreCount, pWaitSemaphores, + pWaitSemaphoreValues, pWaitDstStageMasks, + signalSemaphoreCount, pSignalSemaphores, + pSignalSemaphoreValues); + + // Set AFTER the stamp, which clears the external-input block. Only this + // entry point can carry the distinction: the LEGACY lane has no + // registration default for the sentinel to stand in for, so every layout + // it receives is explicit by construction -- and it also builds a fresh + // node per frame, so it never reads a residual either way. + encodeFrameInfo->srcExternalLayoutIsExplicit = srcLayoutIsExplicit; + // Recorded before the Path-A return below, so a directly-encodable frame + // still carries its registration id -- the probe's own NOT_APPLICABLE + // latch for that case is set at ARM time, but a field that is only + // sometimes populated is the kind of thing a later reader gets wrong. + encodeFrameInfo->externalRegistrationId = registrationId; + + // ===== THE ONE-FRAME STAGED DETOUR FOR A DIRECTLY-ENCODABLE IMPORT ===== + // + // Path A hands the caller's imported image straight to + // vkCmdEncodeVideoKHR, so the producer's pixels never pass through a + // transfer this library records and the content probe has nothing to + // ride. That is what made the probe structurally blind to BLOCK-LINEAR + // imports -- the class the defect appears on -- since block-linear plus + // VIDEO_ENCODE_SRC is exactly what classifies DIRECT. + // + // The fix is deliberately NOT a new command buffer, a new submit, or a + // new barrier program on the encode queue. It is to send the FIRST frame + // of an armed registration down the staged path this library already runs + // for every other import, and let every later frame of that registration + // go DIRECT. NeedsCapture() is false from the moment the capture is + // recorded -- and from the moment the probe latches NOT_APPLICABLE -- so + // the detour is bounded at ONE frame per registration, the same budget + // the readback already has. It is unreachable entirely unless the caller + // chained VkVideoEncoderImportContentInfo onto the registration, which is + // the opt-in and has no other switch. + const bool probeStillOwesACapture = + m_contentProbe && m_contentProbe->NeedsCapture(registrationId); + + if (directlyEncodable && !probeStillOwesACapture) { + // Path A: the registration's node IS the encode source. The encode + // reads the caller's imported image directly, so it -- not a + // staging copy -- is what must be acquired from FOREIGN. + encodeFrameInfo->srcEncodeImageResource = node; + encodeFrameInfo->srcEncodeImageIsExternal = true; + return EncodeFrameCommon(encodeFrameInfo); + } + + // Path B/C: the registration's node is the staged input's source; + // StageInputFrame acquires the pool destination and records either the + // copy or the filter, per the routing the registration resolved. + encodeFrameInfo->srcStagingImageView = node; + // A DETOURED DIRECT FRAME TAKES THE COPY, NEVER THE FILTER. Its format is + // the encode-source format by construction -- that is what made it + // directly encodable -- so there is nothing for the filter to convert, + // and the filter's storage read is not a site the probe rides anyway. + encodeFrameInfo->externalInputViaFilter = + directlyEncodable ? false : routeViaFilter; + return StageInputFrame(encodeFrameInfo); +} + +VulkanDeviceContext::QueueFamilySubmitType +VkVideoEncoder::GetStagedInputSubmitType() const +{ +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // The filter IS m_inputCommandBufferPool when it exists (InitEncoder + // assigns it), and it was created on the compute family. Every staged + // frame on such a session therefore holds a compute-family command + // buffer, whichever branch recorded into it, and a command buffer may + // only be submitted to a queue of its pool's family + // (VUID-vkQueueSubmit2-commandBuffer-03874). + if (m_inputComputeFilter != nullptr) { + return VulkanDeviceContext::COMPUTE; + } +#endif + return ((m_vkDevCtx->GetVideoEncodeQueueFlag() & VK_QUEUE_TRANSFER_BIT) != 0) + ? VulkanDeviceContext::ENCODE + : VulkanDeviceContext::TRANSFER; +} + +uint32_t VkVideoEncoder::GetStagedInputQueueFamilyIdx() const +{ + switch (GetStagedInputSubmitType()) { + case VulkanDeviceContext::COMPUTE: + return (uint32_t)m_vkDevCtx->GetComputeQueueFamilyIdx(); + case VulkanDeviceContext::ENCODE: + return (uint32_t)m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(); + case VulkanDeviceContext::TRANSFER: + default: + return (uint32_t)m_vkDevCtx->GetTransferQueueFamilyIdx(); + } +} + +// The staged-input queue family is what the probe's pool must be created on: +// its two vkCmdCopyImage are recorded into the STAGING command buffer, and a +// pool image created for the wrong family would be a queue-ownership +// violation on the sessions where the staged lane is not the encode queue +// (see GetStagedInputSubmitType, which has three answers, not one). +void VkVideoEncoder::ConfigureContentProbe() +{ + if (!m_contentProbe || (m_vkDevCtx == nullptr) || + (m_contentProbeQueueDepth == 0) || (m_encoderConfig == nullptr)) { + return; + } + m_contentProbe->Configure(m_vkDevCtx, m_contentProbeQueueDepth, + GetStagedInputQueueFamilyIdx(), + m_encoderConfig->encodeWidth, + m_encoderConfig->encodeHeight); +} + VkResult VkVideoEncoder::StageInputFrame(VkSharedBaseObj& encodeFrameInfo) { assert(encodeFrameInfo); @@ -683,6 +1322,14 @@ VkResult VkVideoEncoder::StageInputFrame(VkSharedBaseObj } } + // No arm has run yet, so the library has recorded no barrier on the + // encode-source image for this frame. Seeded here rather than relying on + // Reset()/ClearExternalInputSync() alone, because this function is the + // ONLY writer of the record and a writer that cannot state its own + // starting point leaves the reader unable to tell "not staged" from + // "staged by the previous tenant of this recycled node". + encodeFrameInfo->srcEncodeImageStagedLayout = VK_IMAGE_LAYOUT_MAX_ENUM; + m_inputCommandBufferPool->GetAvailablePoolNode(encodeFrameInfo->inputCmdBuffer); assert(encodeFrameInfo->inputCmdBuffer != nullptr); @@ -694,8 +1341,15 @@ VkResult VkVideoEncoder::StageInputFrame(VkSharedBaseObj beginInfo.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; VkCommandBuffer cmdBuf = encodeFrameInfo->inputCmdBuffer->BeginCommandBufferRecording(beginInfo); + // Holds for an external staging wrapper too, which has an image but no + // view: only the raw VkImage is used below (layout-transition barriers + + // vkCmdCopyImage). VkSharedBaseObj linearInputImageView; encodeFrameInfo->srcStagingImageView->GetImageView(linearInputImageView); + if (linearInputImageView == nullptr) { + assert(!"StageInputFrame: no staging image resource!"); + return VK_ERROR_INITIALIZATION_FAILED; + } VkSharedBaseObj srcEncodeImageView; encodeFrameInfo->srcEncodeImageResource->GetImageView(srcEncodeImageView); @@ -706,33 +1360,666 @@ VkResult VkVideoEncoder::StageInputFrame(VkSharedBaseObj }; VkResult result; - // External input frames (DMA-BUF import) are already in the target format - // from the renderer's filter. Skip the encoder's preprocess compute filter - // — it would do storage reads on the imported DRM modifier image which the - // GPU cannot service on compressed block-linear memory. - if (m_inputComputeFilter == nullptr || encodeFrameInfo->isExternalInput) { - // For external input, use actual layout producer left image in (e.g. GENERAL). - // UNDEFINED would discard contents and produce scrambled encode. - VkImageLayout srcOldLayout = encodeFrameInfo->isExternalInput - ? encodeFrameInfo->srcExternalImageLayout - : VK_IMAGE_LAYOUT_UNDEFINED; + + // Source-side facts BOTH branches need, computed here rather than inside the + // copy branch so the filter branch cannot silently record none of them: an + // acquire the copy performs and the filter does not is not a stylistic + // difference, it is the filter reading memory it does not own. + // + // For external input, use actual layout producer left image in (e.g. GENERAL). + // UNDEFINED would discard contents and produce scrambled encode. + VkImageLayout srcOldLayout; + if (encodeFrameInfo->isExternalInput) { + srcOldLayout = encodeFrameInfo->srcExternalImageLayout; if (srcOldLayout == VK_IMAGE_LAYOUT_UNDEFINED) { srcOldLayout = VK_IMAGE_LAYOUT_GENERAL; // Fallback for compute output } - VkImageLayout linearImgNewLayout = TransitionImageLayout(cmdBuf, linearInputImageView, srcOldLayout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL); - VkImageLayout srcImgNewLayout = TransitionImageLayout(cmdBuf, srcEncodeImageView, VK_IMAGE_LAYOUT_UNDEFINED, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL); + } else { + // THE FILE-INPUT LANE, which has no caller and therefore no + // declaration. It must not fall into the remap above, which would name + // it GENERAL -- the "fallback for compute output" value, which + // describes nothing about this image: on frame 1 the pool image has + // never been in GENERAL, and on the filter arm no barrier is recorded + // at all, so a dispatch reading it that way samples an image the spec + // still considers UNDEFINED. + // + // PREINITIALIZED is the true statement, and it is true because + // m_linearInputImagePool is now CREATED that way (see + // VkVideoEncoder::InitEncoder): LoadNextFrame host-writes the mapped + // image and only then calls this function, so on first use the image + // is exactly what PREINITIALIZED asserts -- host-written, never yet + // moved by any barrier. + // + // It is right ONLY on first use, which is why it is not the final + // word: the residual-layout record below overrides it from frame 2 + // onward, exactly as it does for a reused external registration. + // PREINITIALIZED can be true at most once in an image's life and is + // false the instant our own acquire moves it. + srcOldLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + } + // Local patch; not in upstream vk_video_samples. dma_buf-imported + // external inputs are owned by VK_QUEUE_FAMILY_FOREIGN_EXT; without + // an explicit FOREIGN -> local-queue-family acquire the read + // returns undefined content (observed: solid zeros). + // + // Only dma_buf imports are FOREIGN-owned. The CPU-written staging + // image is locally allocated: its barrier uses HOST stages, and + // HOST + QFOT is invalid + // (VUID-VkImageMemoryBarrier2-srcStageMask-03854). + // + // Prefer the caller-declared residency. The legacy AUTO + // heuristic (FOREIGN iff layout != PREINITIALIZED) only holds for + // FIRST-USE local staging images -- a REUSED local staging image's + // true layout after the previous staging copy is + // TRANSFER_SRC_OPTIMAL, which the heuristic would misclassify as a + // foreign import (wrong QFOT + illegal HOST-stage barrier). Callers + // that pool/reuse input images pass RESIDENCY_LOCAL explicitly. + bool isForeignImport; + switch (encodeFrameInfo->externalInputResidency) { + case EXTERNAL_INPUT_RESIDENCY_LOCAL: + isForeignImport = false; + break; + case EXTERNAL_INPUT_RESIDENCY_FOREIGN: + // Even a declared-FOREIGN frame defers to a PREINITIALIZED + // layout: PREINITIALIZED means host-written staging + // content, and its barrier waits on HOST stages -- + // HOST + QFOT is invalid + // (VUID-VkImageMemoryBarrier2-srcStageMask-03854). The + // declaration routes residency; the layout still decides + // whether a FOREIGN acquire is legal on this pass, exactly + // as the AUTO heuristic below does. + isForeignImport = encodeFrameInfo->isExternalInput && + (encodeFrameInfo->srcExternalImageLayout != + VK_IMAGE_LAYOUT_PREINITIALIZED); + break; + case EXTERNAL_INPUT_RESIDENCY_AUTO: + default: + isForeignImport = encodeFrameInfo->isExternalInput && + (encodeFrameInfo->srcExternalImageLayout != + VK_IMAGE_LAYOUT_PREINITIALIZED); + break; + } + + // THE LIBRARY'S OWN RECORD BEATS A REGISTRATION-TIME DECLARATION. + // + // Everything above computed srcOldLayout from what the CALLER said. For a + // registration that is submitted once that is the only fact available and + // it is the right answer. For a registration that is REUSED it is a + // statement about frame 1 that nothing renews: the acquire below moves the + // image, and from that instant the library -- not the caller -- knows + // where it is. Naming the declaration again on frame 2 is + // VUID-VkImageMemoryBarrier2-oldLayout-01197, once per plane per frame, + // for the life of the registration. + // + // |m_stagedInputResidualLayout| is that knowledge, written at the bottom + // of this function from the value the handback used as its barrier + // newLayout, and unset (MAX_ENUM) until the library has actually moved the + // image. So frame 1 still uses the declaration -- which is the caller's to + // get right and which the library cannot improve on -- and every later + // frame uses a fact. + // + // THREE CONDITIONS, EACH LOad-BEARING: + // + // * NOT isExternalInput, deliberately. The library's own file-input lane + // is exactly the reused-registration shape this record exists for -- 24 + // pool images recycled across a 60-frame file, each carrying the layout + // the previous frame's handback left it in. Excluding it is what made + // its every frame name PREINITIALIZED about an image already in GENERAL. + // The two surviving conditions still fence the external lane the same + // way, and they hold trivially for file input: nothing writes + // srcExternalLayoutIsExplicit off the external path + // (VkVideoEncoder.cpp, SubmitExternalFrameCommon), and isForeignImport + // is false on every arm of the switch above when isExternalInput is. + // + // * !srcExternalLayoutIsExplicit. A caller that fills + // VkVideoEncoderFrameSubmitInfo::currentLayout for THIS frame is saying + // "I moved it", which is exactly the case where our record is stale. + // The public contract already carves this out -- UNDEFINED there means + // "as declared at registration" -- so honouring it is reading the + // documented field, not inventing a rule. + // + // * !isForeignImport. When the image really did go back to a foreign + // owner, an agent outside this library held it between frames and may + // have transitioned it; our record describes only what WE did and is + // not authoritative. Note this is the one condition that also has to + // hold in the other direction, which is why the foreign arms below + // CLEAR the record rather than leaving a stale value for a later local + // frame of the same registration to read. + // + // WHAT THIS DELIBERATELY DOES NOT TOUCH: isForeignImport itself, computed + // above from encodeFrameInfo->srcExternalImageLayout -- the DECLARATION -- + // and never from srcOldLayout. That separation is the whole safety + // argument. On every OS-handle import the ext layer DERIVES residency as + // FOREIGN, so routing falls through to the layout heuristic, and a + // PREINITIALIZED declaration is the only thing keeping Chromium's + // host-written staging lane out of a queue-family acquire it must not + // take (HOST + QFOT is VUID-VkImageMemoryBarrier2-srcStageMask-03854). + // Feeding a residual of GENERAL into that predicate would flip it + // silently. Routing reads the declaration; only the BARRIER reads the + // record. + if (!isForeignImport && + !encodeFrameInfo->srcExternalLayoutIsExplicit && + encodeFrameInfo->srcStagingImageView->HasStagedInputResidualLayout()) { + const VkImageLayout residual = + encodeFrameInfo->srcStagingImageView->GetStagedInputResidualLayout(); + static const bool kDebugLayout = + (getenv("VKENC_DEBUG_LAYOUT") != nullptr); + if (kDebugLayout && (residual != srcOldLayout)) { + VkEncPrintfErr("[LAYOUT-RESIDUAL] img=%p declared=%d -> " + "library-recorded=%d\n", + (void*)linearInputImageView->GetImageResource()->GetImage(), + (int)srcOldLayout, (int)residual); + } + srcOldLayout = residual; + } + + // PER-FRAME routing, replacing "external input never filters". + // + // The predicate it replaces was `m_inputComputeFilter == nullptr || + // isExternalInput`, i.e. a blanket bypass: with the macro on, no external + // frame could ever reach the filter, and with the macro off the arm was + // literally `if (true)`, so no frame of any kind could. That is the + // missing rung of the adaptation ladder -- an input the device cannot + // take directly fell straight to the transfer copy even where a compute + // pass was the only mechanism that could have converted it. + // + // Now: the filter runs when this session HAS one and this FRAME needs it. + // A file-input frame needs it exactly as before (session-level + // enablePreprocessComputeFilter, no per-frame opinion); an external frame + // needs it when whoever admitted the frame said so -- from the + // registration's resolved input path, or from the frame's own format on + // the legacy arm. So one session can carry both kinds of frame, which is + // what a single blanket predicate could not express. + // + // With the macro off this is a compile-time false and the copy arm is the + // only arm, unchanged. Nothing routes a frame here that the copy cannot + // service in that build either: the ext layer refuses a format that needs + // converting when no filter is active, and it is that refusal -- not this + // predicate -- that keeps a 3-plane input away from the copy that hangs + // the GPU on it. + bool useComputeFilter = false; +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + useComputeFilter = (m_inputComputeFilter != nullptr) && + (!encodeFrameInfo->isExternalInput || + encodeFrameInfo->externalInputViaFilter); +#endif + + // ROUTING-DIRECTION GUARD. A frame admitted as FILTER must never silently + // take the copy: the two arms are not fast/slow variants of one another, + // and the case that makes them differ -- a 3-plane source into a 2-plane + // destination -- is CopyLinearToOptimalImage's `assert(vkPlaneFormat[2] == + // VK_FORMAT_UNDEFINED)`, compiled out under NDEBUG, then a 2-region + // vkCmdCopyImage that produces VK_ERROR_DEVICE_LOST, a GPU hang and a + // 0-byte bitstream. + // A returned error is recoverable; a hang is not, and it presents as + // flakiness rather than as a defect. + // + // Reachable whenever the admitting gate and the routing object disagree: + // every ext gate answers from the CONFIG flag, this answers from the + // OBJECT. InitEncoder now hard-fails when those two can diverge, so this + // is the second lock on the same door rather than the only one. + if (encodeFrameInfo->externalInputViaFilter && !useComputeFilter) { + VkEncErr() << "[VkVideoEncoder] frame routed to the preprocess compute " + "filter on a session that has none; refusing rather than " + "falling back to the staging copy" << std::endl; + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // The filter reads an EXTERNAL input through PER-PLANE storage views. A + // wrapper built over an image whose exporter did not declare + // MUTABLE_FORMAT carries none (VkImageResourceView::Create declines them, + // VUID-VkImageViewCreateInfo-image-01762), and + // VulkanFilterYuvCompute::UpdateImageDescriptorSets trims its plane + // bindings by that count -- while ShaderGenerateImagePlaneDescriptors has + // already cleared VK_IMAGE_ASPECT_COLOR_BIT out of m_inputImageAspects for + // a multi-planar input, so there is no combined-view binding to fall back + // to. The result is a push-descriptor set with ZERO input bindings and a + // dispatch that reads unbound STORAGE_IMAGE descriptors. Refuse instead. + if (useComputeFilter && encodeFrameInfo->isExternalInput && + (YcbcrVkFormatInfo(m_encoderConfig->input.vkFormat) != nullptr) && + (linearInputImageView->GetNumberOfPlanes() < 2)) { + VkEncErr() << "[VkVideoEncoder] external input routed to the preprocess " + "compute filter carries no per-plane views (planes=" + << linearInputImageView->GetNumberOfPlanes() + << "); the exporter must declare " + "VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT (with EXTENDED_USAGE) " + "and VK_IMAGE_USAGE_STORAGE_BIT" << std::endl; + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } +#endif + + // THE ROUTING OBSERVABLE, recorded HERE and at no other site. + // + // This is the last point before the copy/filter split and the first point + // after every refusal return above, so a frame counted here is a frame + // whose staging barrier program was actually chosen and recorded -- on + // either arm -- exactly once. The alternative placement, inside the two + // arms' own release/handback pairs, needs four sites and undercounts the + // instant a session mixes filtered and copied frames. + // + // isExternalInput gates it because the library's own file-input lane + // declares no residency at all: counting it would make "every frame was + // local" true of a session that registered nothing, which is precisely + // the reading this channel exists to make falsifiable. + if (encodeFrameInfo->isExternalInput) { + if (isForeignImport) { + m_foreignAcquireCount.fetch_add(1, std::memory_order_relaxed); + } else { + m_localAcquireCount.fetch_add(1, std::memory_order_relaxed); + } + } + + if (!useComputeFilter) { + // The acquire's destination family is the family of the queue this + // batch will be submitted to -- read from the one accessor the submit + // also reads, never re-derived. Re-deriving it here is exactly how + // this site and SubmitStagedInputFrame came to be able to disagree. + const uint32_t stagingQueueFamilyIdx = GetStagedInputQueueFamilyIdx(); + const uint32_t linearSrcQueueFamilyIdx = isForeignImport + ? VK_QUEUE_FAMILY_FOREIGN_EXT + : VK_QUEUE_FAMILY_IGNORED; + const uint32_t linearDstQueueFamilyIdx = isForeignImport + ? stagingQueueFamilyIdx + : VK_QUEUE_FAMILY_IGNORED; + VkImageLayout linearImgNewLayout = TransitionImageLayout(cmdBuf, linearInputImageView, srcOldLayout, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + linearSrcQueueFamilyIdx, linearDstQueueFamilyIdx); + // NO QUEUE-FAMILY OWNERSHIP TRANSFER IS RECORDED FOR THE STAGING + // DESTINATION, AND THAT IS A KNOWN GAP RATHER THAN AN OVERSIGHT. + // Stated here because "nobody wrote it down" and "we decided not to" + // must not look the same to the next reader. + // + // THE SHAPE. m_inputImagePool is created VK_SHARING_MODE_EXCLUSIVE with + // queueFamilyIndexCount = 1, pinned to GetVideoEncodeQueueFamilyIdx() + // (see VulkanVideoImagePool::Configure and the InitEncoder call that + // drives it). This transition and the filter arm's GENERAL transition + // both omit the family arguments, so both default to + // VK_QUEUE_FAMILY_IGNORED. The family-qualified arguments a few lines + // above apply to linearInputImageView -- the SOURCE -- and only when + // isForeignImport. The only family-qualified transition of + // srcEncodeImageView anywhere is the Path-A FOREIGN acquire in + // EncodeFrame, which is gated on srcEncodeImageIsExternal && + // externalInputResidency == FOREIGN and therefore never fires for a + // pool-sourced staged frame. + // + // WHEN IT MATTERS. GetStagedInputSubmitType() returns COMPUTE iff an + // input compute filter OBJECT exists -- keyed on the object, not on + // which arm ran, so it covers this copy arm too. On that path the + // staged batch is recorded and submitted on the COMPUTE family while + // the encode reads the image on the ENCODE family, and for an EXCLUSIVE + // resource the spec makes the contents undefined across that boundary + // without a transfer. The staged family is a COMPUTE|TRANSFER one + // (VulkanDeviceContext prefers a compute-ONLY family) and the encode + // family is a different one. With NO filter the fallback returns + // ENCODE, the same family the pool is pinned to, and nothing is + // owed. + // + // WHY IT IS NOT FIXED HERE, in order of weight: + // + // 1. A DRIVER DEFECT SITS EXACTLY HERE. A driver can lose the + // device on a queue-family ownership RELEASE of an image + // carrying VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR when the + // barrier is recorded off the graphics or optical-flow + // families. This pool carries that usage and the release would + // be recorded on the staged family. The known conjunction also + // requires dstQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT, + // which a real-family release would NOT use, so the exact cell + // this would land in is unknown. Adding the transfer blind + // risks converting a benign spec violation into a hard device + // loss on the primary encode lane. + // + // 2. NO INSTRUMENT IN THIS TREE CAN SEE THE RULE. Core VVL does not + // model "EXCLUSIVE contents become undefined without an ownership + // transfer", and synchronization validation models memory and + // execution hazards, not ownership. This was confirmed rather + // than assumed: with validate_sync and + // syncval_submit_time_validation on, the bars go from 16 and 8 + // READ_AFTER_WRITE hazards to zero once the copy's dependency is + // corrected (see CopyLinearToOptimalImage), while this ownership + // gap is untouched and reported by nothing. So a clean sync run + // must NOT be read as evidence that this is absent or benign. + // + // WHAT CLOSING IT REQUIRES, so the next attempt does not start cold: + // establish first that a release recorded on the staged family with a + // real-family destination, and a mirrored acquire on the encode + // family, does not lose the device. If it survives, record the + // release at the end + // of StageInputFrame on srcEncodeImageView and the matching acquire in + // EncodeFrame immediately before CmdBeginVideoCodingKHR, alongside the + // existing Path-A acquire. Gate BOTH halves on one predicate + // (GetStagedInputQueueFamilyIdx() != GetVideoEncodeQueueFamilyIdx()) + // read once, so a same-family session records neither and the pair can + // never go unbalanced. + // + // TWO FACTS THE RELEASE ABOVE DEPENDS ON, stated so that a future + // attempt does not have to infer either: + // + // * THE OLD LAYOUT DOES NOT DIFFER PER ARM. Both arms hand the image + // over in VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR (the CF-02a/CF-02b + // handling below and on the filter arm), and encodeFrameInfo-> + // srcEncodeImageStagedLayout states that as a fact rather than as + // an inference from which branch ran, so a release can name that + // field and be right on both arms. + // + // * TransitionImageLayout's table HAS a (TRANSFER_DST_OPTIMAL -> + // VIDEO_ENCODE_SRC_KHR) arm, with TRANSFER/TRANSFER_WRITE -> + // ALL_COMMANDS/MEMORY_READ. Note the second scope: a release + // additionally needs the ownership arguments, and the ALL_COMMANDS + // destination is what keeps the arm legal on the compute family + // this branch can be recorded on. + // THE RETURN VALUE IS KEPT, and that is the whole shape of this. Discarding + // it -- `(void)srcImgNewLayout;` -- throws away the library's own statement + // of where it just put the image, so + // the hand-off below had nothing to name as an oldLayout and the + // function's `FIXME - use the real old layout` had no answer for this + // site. This variable IS the real old layout, for the one pair where + // the library itself is the producer. + VkImageLayout srcEncodeImgLayout = TransitionImageLayout(cmdBuf, srcEncodeImageView, VK_IMAGE_LAYOUT_UNDEFINED, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL); (void)linearImgNewLayout; - (void)srcImgNewLayout; CopyLinearToOptimalImage(cmdBuf, linearInputImageView, srcEncodeImageView, copyImageExtent); - } else { + // ===== MECHANISM-C part 2: capture what the PRODUCER handed us ===== + // Same command buffer, one command after the library's own copy read + // this exact image in this exact layout. Splits "the compositor gave us + // a zeroed chroma plane" from "our copy lost it". + if (m_psnr && m_psnr->SrcCaptureEnabled()) { + m_psnr->CaptureImported(cmdBuf, encodeFrameInfo.get(), linearInputImageView.get()); + } + + // ===== THE IMPORT CONTENT PROBE ===== + // + // The SHIPPING sibling of the debug capture above, at the same site + // and for the same physical reason: this is the one instruction + // boundary in the library at which the producer's imported pixels are + // (a) in a layout a transfer can read and (b) not yet mixed with + // anything the library did. The difference is what happens to the + // result -- MECHANISM-C narrates to stderr, which the shipping + // Chromium configuration discards wholesale via silenceStdio, and + // this reports through a chained struct that survives it. + // + // ONCE PER REGISTRATION, not per frame: NeedsCapture() is false from + // the second frame of a buffer on. On the owner's 5125-frame session + // over 5 registered buffers that is 5 readbacks, not 5125. + if (m_contentProbe && + m_contentProbe->NeedsCapture(encodeFrameInfo->externalRegistrationId)) { + const VkSharedBaseObj& probeSrcRes = + linearInputImageView->GetImageResource(); + const VkImageCreateInfo& probeSrcCI = probeSrcRes->GetImageCreateInfo(); + VkExtent2D probeExtent = { probeSrcCI.extent.width, + probeSrcCI.extent.height }; + m_contentProbe->RecordCapture(cmdBuf, + encodeFrameInfo->externalRegistrationId, + probeSrcRes->GetImage(), + probeSrcCI.format, probeExtent, + encodeFrameInfo->contentProbeCapture); + } + + // The OTHER side of the observable. Counted here rather than derived + // as (staged - filtered): a derived count cannot distinguish a frame + // that took the copy from a frame that never reached this function + // at all, and a superset counter has already made a live tier read + // as dead once on this project. + m_stagedCopyCount.fetch_add(1, std::memory_order_relaxed); + + // ===== CF-02a: HAND THE COPY DESTINATION TO THE ENCODER ===== + // + // vkCmdEncodeVideoKHR requires its source picture to be in + // VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR at the time the encode + // executes (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811). Without this + // barrier the only VIDEO_ENCODE_SRC_KHR transition in the + // whole encoder was the Path-A FOREIGN acquire in + // RecordVideoCodingCmd, gated on srcEncodeImageIsExternal && + // externalInputResidency == FOREIGN -- a predicate no staged frame + // can satisfy, because Path A returns straight to EncodeFrameCommon + // and never calls this function. So every frame that reached the + // staging copy was encoded out of TRANSFER_DST_OPTIMAL. + // + // WHY HERE. CopyLinearToOptimalImage is the last write to this image + // in this command buffer, so this is the earliest point at which the + // contents are final; and it is the point where the PRODUCER'S first + // scope is still known to be the transfer that just ran, which is + // what makes TRANSFER/TRANSFER_WRITE an honest availability operation + // rather than a guess. The alternative site -- alongside the Path-A + // acquire in the encode command buffer -- would have to state + // srcStageMask = NONE, because by then the write is in another + // submission. + // + // WHY THIS IS NOT A DOUBLE TRANSITION. The Path-A acquire and this + // barrier are mutually exclusive by construction, not by luck: + // srcEncodeImageIsExternal is set only where srcEncodeImageResource + // IS the caller's imported image, and both of those sites return via + // EncodeFrameCommon without entering StageInputFrame. The acquire is + // therefore preserved untouched and is NOT made redundant by this + // change on any path. + // + // WHAT THIS DOES NOT CLOSE, said plainly so it is not read as more + // than it is: the queue-family OWNERSHIP gap documented at length + // above is untouched. This barrier passes VK_QUEUE_FAMILY_IGNORED on + // both sides -- it changes layout, not ownership -- which is also + // what keeps it clear of the device-loss a VIDEO_ENCODE_SRC + // ownership RELEASE off the graphics engine can provoke. A + // layout-only transition is not that shape. + srcEncodeImgLayout = TransitionImageLayout(cmdBuf, srcEncodeImageView, + srcEncodeImgLayout, + VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR); + // The record RecordVideoCodingCmd reads -- and the exact limit of + // what it can witness. It is + // written from srcEncodeImgLayout, which the statement directly + // above assigned FROM TransitionImageLayout's return, and that + // function has exactly one return -- an unconditional + // `return newLayout` echoing its own by-value argument, which the + // body never reassigns. The value recorded here is therefore the + // layout the CALL ASKED FOR, never an observation of what reached + // cmdBuf. + // + // Delete this call and the running variable keeps its old value, so + // the record says TRANSFER_DST_OPTIMAL and the reader in + // RecordVideoCodingCmd goes red -- which is the only mutation it can + // catch. Suppress only the CmdPipelineBarrier2KHR inside + // TransitionImageLayout, leaving the call and its return in place, + // and the reader stays green while no hand-off barrier is recorded + // at all. + // + // So this reader gates the CALL SITE, not the barrier record. Do not + // cite it as evidence that a barrier reached the command buffer. + encodeFrameInfo->srcEncodeImageStagedLayout = srcEncodeImgLayout; + + // Queue-family RELEASE -- the missing half of the acquire above. + // CopyLinearToOptimalImage is the LAST use of the imported image in + // this command buffer, so this is the earliest correct point. Inside + // this branch on purpose: it is the only scope where + // linearDstQueueFamilyIdx -- the acquire's OWN destination family -- + // is still live, so the release cannot name a family the acquire did + // not. Gated on the same isForeignImport, so a frame that never + // acquired can never release. + // + // oldLayout is the LITERAL the acquire named as its newLayout, which + // is what our copy actually left the image in + // (VUID-VkImageMemoryBarrier2-oldLayout-01197). Naming srcOldLayout -- + // the PRE-acquire producer layout -- would break that VUID. + // + // newLayout is srcOldLayout -- the SAME value the next acquire of + // this registration will name as ITS oldLayout -- so the handover + // round-trips exactly, whatever the producer declares. + // + // A CONSTANT WOULD NOT DO, and GENERAL specifically would be a + // regression for one producer: a caller that declares + // TRANSFER_SRC_OPTIMAL round-trips, and would then find its own + // declaration contradicted, which is VUID-...-oldLayout-01197 on the + // next frame. + // + // "ROUND-TRIPS" DEPENDS ON THE (TRANSFER_SRC_OPTIMAL -> + // TRANSFER_SRC_OPTIMAL) ARMS. The handback hands the image back in + // TRANSFER_SRC_OPTIMAL, and the NEXT frame's acquire then presents the pair + // (TRANSFER_SRC_OPTIMAL -> TRANSFER_SRC_OPTIMAL); with no arm for it such a + // caller does not round-trip, it aborts (standalone build) or takes a + // silently wrong barrier (Chromium). Covered by encoder-ext-input-residency + // --local-tso-opaque-fd. + // above) and cannot be PREINITIALIZED (isForeignImport excludes it), + // so it is always a legal newLayout under VUID-...-newLayout-01198. + // + // For the in-tree Chromium CPU dma-buf lane this evaluates to GENERAL, + // which is also the layout that lane needs on other grounds: it + // declares RESIDENCY_FOREIGN and then host-writes the buffer through + // an mmap between frames, and host access to image memory is + // well defined only for a LINEAR image currently in GENERAL or + // PREINITIALIZED. That second argument is real but narrower than this + // one -- it does not hold for an OPTIMAL or DRM-modifier import + // reaching the same release -- so the round-trip is the reason, and + // host-writability is a property of the answer rather than its + // justification. + // + // What does NOT decide it: the release/acquire layout-equality rule. + // A release runs on a queue of the SOURCE family and its acquire on + // the DESTINATION family, so our release (local -> FOREIGN) and our + // next acquire (FOREIGN -> local) are opposite-direction transfers and + // no VUID binds their layouts. Two contradictory assertions about one + // instant are still worth not making. + if (isForeignImport) { + ReleaseImageToForeignQueue(cmdBuf, linearInputImageView, + VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + srcOldLayout, + linearDstQueueFamilyIdx, + VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR, + VK_ACCESS_2_TRANSFER_READ_BIT_KHR); + // The image is now owned by VK_QUEUE_FAMILY_FOREIGN_EXT and a + // foreign agent may transition it before we see it again, so + // this library has no record worth keeping. Cleared rather + // than left alone: a registration can reach this arm on one + // frame and the local arm on the next (residency AUTO/FOREIGN + // defers to the per-frame layout), and a residual written by an + // earlier local frame would then be read after a foreign owner + // had the image. + encodeFrameInfo->srcStagingImageView->SetStagedInputResidualLayout( + VK_IMAGE_LAYOUT_MAX_ENUM); + } else { + // THE SAME HANDBACK, for a LOCAL registration. Everything the + // release above argues about newLayout applies unchanged: the + // image is handed back in the layout the next acquire of this + // registration will declare, so the round-trip closes. The only + // difference is that no ownership changes hands, so there is no + // queue-family transfer and the destination scope is the caller's + // host access rather than a foreign agent's unknown one. + // + // An else-if on the SAME condition, not a second if: exactly one + // of {foreign release, local restore} can ever fire, which is the + // structural version of the claim that a frame which never + // acquired can never release. + // + // NOT GATED ON isExternalInput. The file-input lane owns its linear + // pool image outright and declares nothing, which would argue for + // skipping the restore as a barrier no caller can observe. It does + // not, because the pool is created PREINITIALIZED and srcOldLayout + // above states that fact, so the + // library DOES have something to record: without this handback the + // node's residual is never written, so frame 2 would name + // PREINITIALIZED about an image its own frame-1 acquire had already + // moved to TRANSFER_SRC_OPTIMAL -- VUID-...-oldLayout-01197. + // + // STATED PLAINLY: WHICH FILE-INPUT FRAMES REACH THIS HALF. The + // copy arm requires useComputeFilter == false, which for a + // file-input frame reduces to m_inputComputeFilter == nullptr, + // which requires EncoderConfig::enablePreprocessComputeFilter == + // false. That field is constructed true (VkEncoderConfig.h) and + // has exactly one writer in the tree -- + // vulkan_video_encoder_ext.cpp, on the ext layer, which only ever + // produces EXTERNAL frames. There is no CLI flag and no JSON + // schema key for it. In a build with the compute filter compiled + // in, therefore, no file-input frame reaches this line; the arm + // is written for consistency with the filter arm beside it. + // + // oldLayout is the literal TRANSFER_SRC_OPTIMAL for the same + // reason the release names it: it is what the acquire above + // transitioned to and what CopyLinearToOptimalImage left behind. + // + // AND THE RECORD. The helper returns the layout it actually + // left the image in -- the declaration, or GENERAL where the + // declaration was not a legal barrier destination -- so the + // barrier and the record cannot disagree: there is no second + // expression of the same fact to keep in step. + const VkImageLayout handedBackLayout = + RestoreStagedInputLayout(cmdBuf, linearInputImageView, + VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + srcOldLayout, + VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR, + VK_ACCESS_2_TRANSFER_READ_BIT_KHR); + encodeFrameInfo->srcStagingImageView->SetStagedInputResidualLayout( + handedBackLayout); + } + } +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + else { VkVideoPictureResourceInfoKHR srcPictureResourceInfo(*encodeFrameInfo->srcStagingImageView->GetPictureResourceInfo()); VkVideoPictureResourceInfoKHR dstPictureResourceInfo(*encodeFrameInfo->srcEncodeImageResource->GetPictureResourceInfo()); srcPictureResourceInfo.codedExtent = copyImageExtent; + // The barriers the copy branch beside this one has always recorded + // and this branch never did. + // + // NO LONGER GATED ON isExternalInput. That gate was written on the + // reading that external input "is the only kind of frame that arrives + // owned by another queue family or in a layout this encoder did not + // choose". The FAMILY half holds, and is why the family indices below + // are conditional. The LAYOUT half does not hold for the library's own + // file-input frames: the linear pool image arrives in whatever the + // previous frame left it in, or PREINITIALIZED on first use, and + // NEITHER of those is the GENERAL that VulkanFilterYuvCompute's + // STORAGE_IMAGE descriptors demand. + // + // With the gate in place the dispatch reads an image the validation + // layer reports as UNDEFINED where GENERAL is required + // (VUID-vkCmdDraw-None-09600), once per input plane plus twice for + // the encode-input image, on every frame and on both the 3-plane and + // the 2-plane shape. The dispatch is reading an image the spec + // permits the driver to have discarded, and it survives only because + // a LINEAR host-coherent allocation on this vendor happens not to + // be. + // + // Destination family is the COMPUTE family, because the compute + // filter is what consumes this image and a queue-family acquire must + // execute on a queue of its DESTINATION family. That is the same + // rule the copy branch obeys by naming the transfer/encode family -- + // and it is why this must land before the branch is reachable: + // acquiring into the encode family and then submitting the batch on + // the compute queue is not a slow path, it is a wedged queue. + // This arm's running record of where srcEncodeImageView actually is, + // for the same reason and with the same mutation property as the copy + // arm's. Declared out here because the acquire block below closes + // before the filter has run, and the hand-off that reads it is after + // the dispatch. + VkImageLayout srcEncodeImgLayout = VK_IMAGE_LAYOUT_UNDEFINED; + { + // Same accessor the copy arm and the submit read. On any session + // that reached this arm it answers the COMPUTE family, because + // the filter IS the input command-buffer pool -- but reading it + // rather than naming the compute family directly is what keeps + // the three sites from ever describing different queues. + const uint32_t filterQueueFamilyIdx = GetStagedInputQueueFamilyIdx(); + const uint32_t filterSrcQueueFamilyIdx = isForeignImport + ? VK_QUEUE_FAMILY_FOREIGN_EXT + : VK_QUEUE_FAMILY_IGNORED; + const uint32_t filterDstQueueFamilyIdx = isForeignImport + ? filterQueueFamilyIdx + : VK_QUEUE_FAMILY_IGNORED; + // Input -> GENERAL: the filter binds it as a STORAGE_IMAGE, and + // a storage descriptor admits no other layout. + TransitionImageLayout(cmdBuf, linearInputImageView, + srcOldLayout, VK_IMAGE_LAYOUT_GENERAL, + filterSrcQueueFamilyIdx, + filterDstQueueFamilyIdx); + // Output -> GENERAL, discarding: the destination is this + // encoder's own pool image and the filter overwrites every + // texel, exactly as the copy branch discards into + // TRANSFER_DST_OPTIMAL. + srcEncodeImgLayout = + TransitionImageLayout(cmdBuf, srcEncodeImageView, + VK_IMAGE_LAYOUT_UNDEFINED, + VK_IMAGE_LAYOUT_GENERAL); + } + if (m_encoderConfig->enablePictureRowColReplication == 1) { // replicate the last row and column to the padding area dstPictureResourceInfo.codedExtent.width = m_encoderConfig->encodeAlignedWidth; @@ -779,7 +2066,171 @@ VkResult VkVideoEncoder::StageInputFrame(VkSharedBaseObj if (result != VK_SUCCESS) { return result; } + + // What ACTUALLY ran, for SubmitStagedInputFrame. Recorded after the + // record succeeded, so a filter that failed to record leaves the + // frame described as unfiltered rather than as filtered-and-broken. + encodeFrameInfo->inputFilterRecorded = true; + // ...and the same fact made readable from OUTSIDE the library, on + // the same line and under the same success condition, so the + // observable can never drift from the routing flag it mirrors. + m_inputFilterDispatchCount.fetch_add(1, std::memory_order_relaxed); + + // ===== CF-02b: HAND THE FILTER'S OUTPUT TO THE ENCODER ===== + // + // The filter writes this image through STORAGE_IMAGE descriptors, so + // the acquire above put it in GENERAL and it has to be there for the + // dispatch. The only barrier VulkanFilterYuvCompute records after its + // dispatch is a GENERAL -> GENERAL availability operation on this + // same output image -- it makes the shader writes available and + // deliberately does not change the layout, because the filter has no + // opinion about what its consumer needs. This library does: the + // consumer is vkCmdEncodeVideoKHR, and GENERAL satisfies + // VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 only when the + // unifiedImageLayoutsVideo feature is enabled. It is enabled nowhere + // -- zero occurrences across this library, media/gpu/, gpu/vulkan/ + // and the DEPS-pinned submodule -- so without this barrier the filter arm + // encodes out of GENERAL, in violation of the spec. + // + // AFTER THE RECORD, NOT INSIDE THE BLOCK ABOVE: the dispatch is what + // fills the image, so a transition placed with the acquire would + // transition an empty image and then let the dispatch write it in a + // layout its own descriptors reject. Placed here it is ordered after + // RecordCommandBuffer's own trailing barrier, which is the correct + // reading of "the filter has finished with its output". + // + // MASKS: see the (GENERAL -> VIDEO_ENCODE_SRC_KHR) arm. COMPUTE_SHADER + // /SHADER_WRITE is the dispatch that produced the contents and is + // legal here because this arm only runs on a session whose input + // command buffer IS the filter's compute-family pool; the second + // scope is ALL_COMMANDS/MEMORY_READ because that same family need + // not carry VK_QUEUE_VIDEO_ENCODE_BIT_KHR, and the encode's + // visibility arrives through the input->encode semaphore. + // + // Both families are VK_QUEUE_FAMILY_IGNORED: layout only, no + // ownership transfer, so this is not the shape that can lose the + // device. + srcEncodeImgLayout = TransitionImageLayout(cmdBuf, srcEncodeImageView, + srcEncodeImgLayout, + VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR); + // Same record, same limit as the copy arm above: written from + // TransitionImageLayout's return, so it witnesses that the CALL was + // made, not that a barrier was recorded. See the note there. + encodeFrameInfo->srcEncodeImageStagedLayout = srcEncodeImgLayout; + + // Queue-family RELEASE -- the missing half of the filter acquire. + // The filter's dispatch is the LAST use of the INPUT image; the only + // barrier it records afterwards targets the OUTPUT. The input is + // therefore still in VK_IMAGE_LAYOUT_GENERAL here, which is the + // literal the acquire named. + // + // The acquire's family locals are out of scope by now, so the + // accessor is re-read: it is pure and answers the same family the + // acquire named on any session that reached this arm. + // + // srcAccessMask is SHADER_READ, deliberately not SHADER_WRITE. The + // filter READS this image; the GENERAL->GENERAL table arm supplies + // WRITE, which is right for the acquire direction and would be an + // availability operation on a write that never happened here. + if (isForeignImport) { + // isForeignImport alone, not (isExternalInput && isForeignImport): + // every arm of the residency switch that can set isForeignImport + // conjoins isExternalInput already, so the second test was + // redundant -- and dropping it is what makes the else below able + // to mean "every frame that did not release to a foreign owner", + // which is the set the local handback is for. + // + // oldLayout is GENERAL because that is what the filter acquire + // above transitioned this image to, and the filter records no + // further barrier on its INPUT (the one it does record after the + // dispatch targets the output image). + // + // Not because "a sampled descriptor admits no other layout" -- it + // does: a COMBINED_IMAGE_SAMPLER, which is what the Y'CbCr arm + // binds when a conversion sampler exists, admits others; only a + // STORAGE_IMAGE is confined to GENERAL + // (VUID-VkDescriptorImageInfo-imageView-06711). What settles it is + // that VulkanFilterYuvCompute declares ONE input layout for every + // arm, and that layout is GENERAL. GENERAL is right on every lane + // the ENCODER can reach for a different reason: a multi-planar + // input has no aspect-0 binding so every plane descriptor is + // forced to GENERAL, and the RGBA storage-read arm overrides back + // to GENERAL explicitly. Stated precisely because a blanket claim + // here reads as a licence to skip the check. + // + // newLayout is srcOldLayout, NOT a second GENERAL. The copy arm + // hands back the layout the next acquire will declare, and this + // arm must too: the helper's own rule is that newLayout should be + // what the NEXT acquire of the same resource names, so the two + // barriers do not assert different things about one instant, and + // BOTH acquires read the same srcOldLayout computed once above. + // Hardcoding GENERAL was only correct when srcOldLayout happened + // to be GENERAL, which is not pinned: isForeignImport excludes + // PREINITIALIZED and UNDEFINED is remapped to GENERAL, but + // TRANSFER_SRC_OPTIMAL remains reachable from a producer's + // declaration, and the layout table deliberately keeps an arm for + // exactly that producer. + // + // VIDEO_ENCODE_SRC_KHR is reachable as a declaration too, but do + // NOT read this as saying that lane is wired: the only + // (VIDEO_ENCODE_SRC_KHR -> GENERAL) arm in the table is documented + // for the filter's OUTPUT image and supplies SHADER_WRITE + // visibility, so a producer declaring it on a filter INPUT would + // get an acquire with no read visibility for the dispatch about to + // sample it. That arm is owed before such a producer is + // supported. No lane regresses: on the host-mmap lane srcOldLayout + // IS GENERAL, so this is identical there. + ReleaseImageToForeignQueue(cmdBuf, linearInputImageView, + VK_IMAGE_LAYOUT_GENERAL, + srcOldLayout, + GetStagedInputQueueFamilyIdx(), + VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT_KHR, + VK_ACCESS_2_SHADER_READ_BIT_KHR); + // Cleared for the same reason the copy arm's release clears it: + // a foreign owner has the image between frames. + encodeFrameInfo->srcStagingImageView->SetStagedInputResidualLayout( + VK_IMAGE_LAYOUT_MAX_ENUM); + } else { + // The LOCAL handback on the filter arm. Same argument as the copy + // arm beside it, with this arm's residual layout: the filter's + // dispatch is the last use of the INPUT image and records no + // further barrier on it, so the image is still in the GENERAL the + // filter acquire named. + // + // srcStageMask COMPUTE_SHADER is legal HERE AND ONLY HERE, and not + // by assumption: this arm is reachable only on a session that has + // an input compute filter, and GetStagedInputSubmitType() returns + // COMPUTE for exactly that session, so the batch is submitted on + // the compute family. The copy arm beside it therefore must not + // and does not name this stage. + // + // SHADER_READ not SHADER_WRITE, for the same reason the release + // above states: the filter READS this image. + // + // REACHABILITY. A consumer that declares GENERAL -- which equals + // this arm's residual -- makes the helper's equal-layout early + // return fire, so no barrier is recorded and this call is a + // no-op. A filter-routed registration declaring + // TRANSFER_SRC_OPTIMAL instead, the ext layer's own legacy-wrap + // default, is the shape that exercises it. + // + // AND IT IS LOAD-BEARING, not merely reached. Suppressing ONLY + // the CmdPipelineBarrier2KHR inside RestoreStagedInputLayout, + // leaving the return value alone so the registration's residual + // record still claims TRANSFER_SRC_OPTIMAL while the image is + // really still in GENERAL, makes frame 2's acquire name a layout + // the image is not in. + const VkImageLayout handedBackLayout = + RestoreStagedInputLayout(cmdBuf, linearInputImageView, + VK_IMAGE_LAYOUT_GENERAL, + srcOldLayout, + VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT_KHR, + VK_ACCESS_2_SHADER_READ_BIT_KHR); + encodeFrameInfo->srcStagingImageView->SetStagedInputResidualLayout( + handedBackLayout); + } } +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED // Stage QPMap if it needs staging. Reuse the same command buffer used for staging of the input image if (m_encoderConfig->enableQpMap && (m_qpMapTiling != VK_IMAGE_TILING_LINEAR)) { @@ -794,8 +2245,22 @@ VkResult VkVideoEncoder::StageInputFrame(VkSharedBaseObj return result; } - // Now submit the staged input to the queue - SubmitStagedInputFrame(encodeFrameInfo); + // Now submit the staged input to the queue. + // + // A rejected staging submit means the encode source was never written and + // the semaphore the encode submit waits on will never be signalled. + // Encoding anyway produces a frame from whatever the destination image + // happened to hold and queues a wait nothing can satisfy, so the driver's + // own VkResult is returned here rather than discarded. + // + // The frame's images, imported waits and registrations are deliberately + // NOT torn down on this path. A rejected batch is not in flight, but the + // caller's input may still be referenced by work that did land, and the + // pending record is what accounts for it. + result = SubmitStagedInputFrame(encodeFrameInfo); + if (result != VK_SUCCESS) { + return result; + } // and encode the input frame with the encoder next return EncodeFrameCommon(encodeFrameInfo); @@ -840,7 +2305,11 @@ VkResult VkVideoEncoder::SubmitStagedQpMap(VkSharedBaseObjqpMapCmdBuffer->SetCommandBufferSubmitted(); + // As for the input staging buffer above: a rejected batch is not in + // flight and must not be recorded as if it were. + if (result == VK_SUCCESS) { + encodeFrameInfo->qpMapCmdBuffer->SetCommandBufferSubmitted(); + } bool syncCpuAfterStaging = false; if (syncCpuAfterStaging) { encodeFrameInfo->qpMapCmdBuffer->SyncHostOnCmdBuffComplete(false, "encoderStagedInputFence"); @@ -903,7 +2372,7 @@ void VkVideoEncoder::CopyYCbCrPlanesDirectCPU( } else if (chromaHorzRatio == 2 && chromaVertRatio == 1) { subsamplingDesc = "4:2:2"; } - printf("YCbCr copy with %s subsampling (chromaHorzRatio=%d, chromaVertRatio=%d), %d-bit\n", + VkEncPrintfOut("YCbCr copy with %s subsampling (chromaHorzRatio=%d, chromaVertRatio=%d), %d-bit\n", subsamplingDesc, chromaHorzRatio, chromaVertRatio, bitDepth); } @@ -987,6 +2456,25 @@ VkResult VkVideoEncoder::SubmitStagedInputFrame(VkSharedBaseObjinputFilterRecorded + ? VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT + : VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + const uint32_t MAX_SIGNAL_SEMAPHORES = 2; uint32_t signalSemaphoreCount = 0; VkSemaphoreSubmitInfoKHR signalSemaphoreInfos[MAX_SIGNAL_SEMAPHORES]{}; @@ -996,7 +2484,7 @@ VkResult VkVideoEncoder::SubmitStagedInputFrame(VkSharedBaseObjinputWaitSemaphores[i]; waitInfo.value = (i < encodeFrameInfo->inputWaitSemaphoreValues.size()) ? encodeFrameInfo->inputWaitSemaphoreValues[i] : 0; + // Same reasoning as the signal side, in the other direction: the + // default names the stage that CONSUMES the producer's image, and + // on the filter branch that is the compute dispatch, not a + // transfer. A TRANSFER-only wait leaves the dispatch outside the + // second synchronization scope, i.e. reading the producer's + // dma-buf before the acquire semaphore is signalled. waitInfo.stageMask = (i < encodeFrameInfo->inputWaitDstStageMasks.size()) ? encodeFrameInfo->inputWaitDstStageMasks[i] - : VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + : stagedInputSignalStage; waitInfo.deviceIndex = 0; waitSemaphoreInfos.push_back(waitInfo); } @@ -1044,7 +2538,11 @@ VkResult VkVideoEncoder::SubmitStagedInputFrame(VkSharedBaseObjinputSignalSemaphores[i]; signalInfo.value = (i < encodeFrameInfo->inputSignalSemaphoreValues.size()) ? encodeFrameInfo->inputSignalSemaphoreValues[i] : 0; - signalInfo.stageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + // The RELEASE timeline: this is what the embedder's release-fence + // SYNC_FD is exported from, so signalling it at TRANSFER on a + // compute batch tells the producer "input released" while the + // filter is still sampling its dma-buf. + signalInfo.stageMask = stagedInputSignalStage; signalInfo.deviceIndex = 0; allSignalSemaphoreInfos.push_back(signalInfo); } @@ -1062,10 +2560,30 @@ VkResult VkVideoEncoder::SubmitStagedInputFrame(VkSharedBaseObjinputCmdBuffer->GetFence(); assert(VK_NOT_READY == m_vkDevCtx->GetFenceStatus(*m_vkDevCtx, queueCompleteFence)); + // THE SUBMIT QUEUE IS KEYED OFF THE POOL'S FAMILY, NOT OFF THE BRANCH THAT + // RAN, and that is forced rather than preferred. Both branches record into a + // command buffer from m_inputCommandBufferPool, and InitEncoder creates that + // pool on ONE family per session -- the compute family when the filter + // exists, because the filter IS the pool. A command buffer may only be + // submitted to a queue of its pool's family + // (VUID-vkQueueSubmit2-commandBuffer-03874), so a copy-branch frame on a + // filter-bearing session cannot legally be sent to the transfer queue + // whatever its barriers say. Keying the submit off the branch would trade a + // wedged queue for an invalid submit. + // + // So the dependency points the other way: both barrier sites read the + // family from GetStagedInputQueueFamilyIdx(), and this submit reads the + // queue from GetStagedInputSubmitType() -- the same fact, twice. The + // assertion below is the invariant stated where it can fail loudly. const VulkanDeviceContext::QueueFamilySubmitType submitType = - (m_inputComputeFilter != nullptr) ? VulkanDeviceContext::COMPUTE : - (((m_vkDevCtx->GetVideoEncodeQueueFlag() & VK_QUEUE_TRANSFER_BIT) != 0) ? - VulkanDeviceContext::ENCODE : VulkanDeviceContext::TRANSFER); + GetStagedInputSubmitType(); +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // A recorded filter dispatch implies a compute-family batch. If this ever + // fires, a filter ran on a session whose input pool is not the filter's, + // and the acquires recorded above name a family this submit is not on. + assert(!encodeFrameInfo->inputFilterRecorded || + (submitType == VulkanDeviceContext::COMPUTE)); +#endif VkResult result = m_vkDevCtx->MultiThreadedQueueSubmit(submitType, 0, // queueIndex @@ -1076,7 +2594,15 @@ VkResult VkVideoEncoder::SubmitStagedInputFrame(VkSharedBaseObjinputCmdBuffer->SetCommandBufferSubmitted(); + // Only a submit the driver ACCEPTED puts this node in flight. Marking a + // rejected batch submitted tells the release-fence export that a signal + // operation is pending execution when none was ever queued, and leaves + // the node claiming a fence that will never be signalled. Rejected + // commands stay Recorded, which is what they are, and reset/reuse + // proceeds from there. + if (result == VK_SUCCESS) { + encodeFrameInfo->inputCmdBuffer->SetCommandBufferSubmitted(); + } bool syncCpuAfterStaging = false; if (syncCpuAfterStaging) { encodeFrameInfo->inputCmdBuffer->SyncHostOnCmdBuffComplete(false, "encoderStagedInputFence"); @@ -1120,7 +2646,7 @@ VkResult VkVideoEncoder::AssembleBitstreamData(VkSharedBaseObjEnabled()) { + if (m_psnr && (m_psnr->Enabled() || m_psnr->SrcCaptureEnabled())) { m_psnr->ComputeFramePsnr(encodeFrameInfo.get()); } + // POST-FENCE, and that is load-bearing: the probe's readback lands in + // HOST_VISIBLE memory written by the staging command buffer, which this + // frame's encode command buffer is ordered after. Scoring it before the + // fence would read whatever the mapping happened to hold. ScoreCapture + // is a no-op for the frames that carry no capture, which is all of them + // but the first of each armed registration. + if (m_contentProbe) { + m_contentProbe->ScoreCapture(encodeFrameInfo->contentProbeCapture); + } + if (m_crc.Enabled()) { m_crc.SignalFrameEnd((uint32_t)(encodeFrameInfo->gopPosition.inputOrder)); } @@ -1166,15 +2730,26 @@ VkResult VkVideoEncoder::ReadbackBitstreamData( VkResult result = encodeFrameInfo->encodeCmdBuffer->SyncHostOnCmdBuffComplete( false, "asyncAssemblyFence"); if (result != VK_SUCCESS) { - fprintf(stderr, "\nAsync assembly: fence wait failed with result 0x%x.\n", result); + VkEncPrintfErr("\nAsync assembly: fence wait failed with result 0x%x.\n", result); return result; } - if (m_psnr && m_psnr->Enabled()) { + if (m_psnr && (m_psnr->Enabled() || m_psnr->SrcCaptureEnabled())) { std::lock_guard psnrLock(m_assemblyFileMutex); m_psnr->ComputeFramePsnr(encodeFrameInfo.get()); } + // The ASYNC assembly lane's copy of the post-fence score above. Both + // sites are needed and neither is redundant: a session runs one lane or + // the other, and wiring only the synchronous one is exactly how the + // async lane silently loses an observable (this tree has shipped that + // mistake once already, on the completion-record path). No + // m_assemblyFileMutex here -- the probe carries its own lock and touches + // no file output. + if (m_contentProbe) { + m_contentProbe->ScoreCapture(encodeFrameInfo->contentProbeCapture); + } + uint32_t querySlotId = (uint32_t)-1; VkQueryPool queryPool = encodeFrameInfo->encodeCmdBuffer->GetQueryPool(querySlotId); @@ -1190,7 +2765,7 @@ VkResult VkVideoEncoder::ReadbackBitstreamData( VK_QUERY_RESULT_WITH_STATUS_BIT_KHR | VK_QUERY_RESULT_WAIT_BIT); if (result != VK_SUCCESS || encodeResult.status != VK_QUERY_RESULT_STATUS_COMPLETE_KHR) { - fprintf(stderr, "\nAsync assembly: query failed (0x%x, status=0x%x).\n", + VkEncPrintfErr("\nAsync assembly: query failed (0x%x, status=0x%x).\n", result, encodeResult.status); return (result != VK_SUCCESS) ? result : VK_INCOMPLETE; } @@ -1206,13 +2781,77 @@ VkResult VkVideoEncoder::WriteBitstreamToFile( VkSharedBaseObj& encodeFrameInfo, uint32_t frameIdx, uint32_t ofTotalFrames, BitstreamReadback& readback) +{ + // Every frame that reaches assembly publishes exactly one completion + // record, in both output modes; only the payload differs. In capture + // mode (disableFileOutput) the record carries the bytes the caller + // retrieves; in file-output mode the bytes go to the file and the + // record carries metadata plus the file-write result, so an ext + // consumer at the config default still gets a truthful per-frame + // completion instead of a deadline-synthesized VK_TIMEOUT drop. + // + // Key the record by the CALLER's frame id when this frame + // came through SetExternalInputFrame(). frameEncodeInputOrderNum is + // only a fallback for the file-based path -- it coincides with the + // caller's ids solely in the no-error, no-reorder case and drifts + // permanently after any partially failed submission. + CapturedBitstream cap; + cap.frameId = (encodeFrameInfo->externalFrameId != uint64_t(-1)) + ? encodeFrameInfo->externalFrameId + : encodeFrameInfo->frameEncodeInputOrderNum; + cap.isIdr = (encodeFrameInfo->gopPosition.pictureType == + VkVideoGopStructure::FRAME_TYPE_IDR); + cap.pictureType = static_cast( + encodeFrameInfo->gopPosition.pictureType); + + VkResult result = VK_SUCCESS; + if (m_encoderConfig && m_encoderConfig->disableFileOutput) { + if (encodeFrameInfo->bitstreamHeaderBufferSize > 0) { + const uint8_t* hdr = + encodeFrameInfo->bitstreamHeaderBuffer + + encodeFrameInfo->bitstreamHeaderOffset; + cap.bytes.insert( + cap.bytes.end(), hdr, + hdr + encodeFrameInfo->bitstreamHeaderBufferSize); + } + if (readback.readbackDone && readback.bitstreamSize > 0) { + const uint8_t* src; + if (!readback.bitstreamCopy.empty()) { + src = readback.bitstreamCopy.data(); + } else { + VkDeviceSize maxSize; + // bitstreamStartOffset is relative to + // encodeInfo.dstBufferOffset (header reservation). + src = encodeFrameInfo->outputBitstreamBuffer-> + GetDataPtr(0, maxSize) + + encodeFrameInfo->encodeInfo.dstBufferOffset + + readback.bitstreamStartOffset; + } + cap.bytes.insert(cap.bytes.end(), src, + src + readback.bitstreamSize); + } + } else { + result = WriteBitstreamToFileOutput(encodeFrameInfo, readback); + cap.status = result; // VK_SUCCESS, or the file-write failure code + } + PushCapturedBitstream(std::move(cap)); + return result; +} + +// File-output arm of WriteBitstreamToFile: writes the non-VCL header, then the +// coded payload described by readback, which ReadbackBitstreamData() has +// already fetched from the feedback query pool. +// Private and non-virtual: it must never grow a second completion publish. +VkResult VkVideoEncoder::WriteBitstreamToFileOutput( + VkSharedBaseObj& encodeFrameInfo, + BitstreamReadback& readback) { if(encodeFrameInfo->bitstreamHeaderBufferSize > 0) { size_t nonVcl = WriteDataToFile(encodeFrameInfo->bitstreamHeaderBuffer + encodeFrameInfo->bitstreamHeaderOffset, encodeFrameInfo->bitstreamHeaderBufferSize); if (m_encoderConfig->verboseFrameStruct) { - std::cout << " == Non-Vcl data " << (nonVcl ? "SUCCESS" : "FAIL") + VkEncOut() << " == Non-Vcl data " << (nonVcl ? "SUCCESS" : "FAIL") << " File Output non-VCL data with size: " << encodeFrameInfo->bitstreamHeaderBufferSize << ", Input Order: " << encodeFrameInfo->gopPosition.inputOrder << ", Encode Order: " << encodeFrameInfo->gopPosition.encodeOrder @@ -1239,14 +2878,14 @@ VkResult VkVideoEncoder::WriteBitstreamToFile( size_t remaining = readback.bitstreamSize - totalBytesWritten; size_t written = WriteDataToFile(src + totalBytesWritten, remaining); if (written == 0) { - fprintf(stderr, "Error writing VCL data\n"); + VkEncPrintfErr("Error writing VCL data\n"); return VK_ERROR_OUT_OF_HOST_MEMORY; } totalBytesWritten += written; } if (m_encoderConfig->verboseFrameStruct) { - std::cout << " == Output VCL data " << ((totalBytesWritten == readback.bitstreamSize) ? "SUCCESS" : "FAIL") << " with size: " << readback.bitstreamSize + VkEncOut() << " == Output VCL data " << ((totalBytesWritten == readback.bitstreamSize) ? "SUCCESS" : "FAIL") << " with size: " << readback.bitstreamSize << " and offset: " << readback.bitstreamStartOffset << ", Input Order: " << encodeFrameInfo->gopPosition.inputOrder << ", Encode Order: " << encodeFrameInfo->gopPosition.encodeOrder << std::endl << std::flush; @@ -1258,8 +2897,14 @@ VkResult VkVideoEncoder::WriteBitstreamToFile( void VkVideoEncoder::AssemblyWorkerThread(int threadId) { + { + char threadName[16]; + snprintf(threadName, sizeof(threadName), "VkEncAsm%d", threadId); + vkenc::OsSetCurrentThreadName(threadName); + } + if (m_encoderConfig->verbose) { - std::cout << "[AsyncAssembly] Worker " << threadId << " started" << std::endl; + VkEncOut() << "[AsyncAssembly] Worker " << threadId << " started" << std::endl; } while (true) { @@ -1275,12 +2920,38 @@ void VkVideoEncoder::AssemblyWorkerThread(int threadId) VkResult result = ReadbackBitstreamData(frame, item.readback); if (result != VK_SUCCESS) { - fprintf(stderr, "[AsyncAssembly] Worker %d: readback failed (0x%x) " + VkEncPrintfErr("[AsyncAssembly] Worker %d: readback failed (0x%x) " "seq=%lu\n", threadId, result, (unsigned long)item.sequenceNumber); m_assemblyErrorCount++; + // Deliver the per-frame failure through the completion funnel + // in BOTH output modes -- an empty record with the failure + // VkResult (e.g. VK_INCOMPLETE for a non-COMPLETE query status + // such as INSUFFICIENT_BITSTREAM_BUFFER_RANGE). A frame dropped + // here without a record never surfaces at the Ext caller's + // retrieval, so the stream stalls with no diagnosis instead of + // an actionable per-frame error. { - std::lock_guard lock(m_assemblyFileMutex); + std::unique_lock lock(m_assemblyFileMutex); + // Wait for this frame's turn before advancing the hand-off. + // The success path below waits on exactly this predicate, so + // advancing out of turn -- as this path used to -- steps past a + // lower-numbered worker's slot and strands it forever: the + // condition variable has no timeout, so the whole assembly + // pipeline wedges with nothing logged. Taking the turn also + // keeps failed captures in submission order with successful + // ones, which the consumer's FIFO assumes. + m_assemblyOrderCV.wait(lock, [&] { + return item.sequenceNumber == m_nextWriteSequence.load(); + }); + CapturedBitstream cap; + cap.frameId = (frame->externalFrameId != uint64_t(-1)) + ? frame->externalFrameId + : frame->frameEncodeInputOrderNum; + cap.isIdr = false; + cap.pictureType = 0; + cap.status = result; + PushCapturedBitstream(std::move(cap)); m_nextWriteSequence++; } m_assemblyOrderCV.notify_all(); @@ -1288,11 +2959,14 @@ void VkVideoEncoder::AssemblyWorkerThread(int threadId) continue; } - // The threaded (ext streaming) assembly path does not go through - // AssembleBitstreamData, so PSNR / recon capture is done here. - if (m_psnr && m_psnr->Enabled()) { - m_psnr->ComputeFramePsnr(frame.get()); - } + // PSNR / recon capture for the threaded path happens inside + // ReadbackBitstreamData (locked, right after the fence wait). A + // second call here read the same frame's recon state without holding + // that lock, concurrently with the worker that does. In steady state + // it was a no-op -- ComputeFramePsnr nulls its staging image, so the + // second call found nothing to measure -- which is why nothing + // visibly broke; the unsynchronised read is the reason it is gone, + // not a miscount. { std::unique_lock lock(m_assemblyFileMutex); @@ -1305,7 +2979,7 @@ void VkVideoEncoder::AssemblyWorkerThread(int threadId) (uint32_t)item.sequenceNumber + 1, item.readback); if (result != VK_SUCCESS) { - fprintf(stderr, "[AsyncAssembly] Worker %d: write failed (0x%x) " + VkEncPrintfErr("[AsyncAssembly] Worker %d: write failed (0x%x) " "seq=%lu\n", threadId, result, (unsigned long)item.sequenceNumber); m_assemblyErrorCount++; @@ -1319,27 +2993,43 @@ void VkVideoEncoder::AssemblyWorkerThread(int threadId) } if (m_encoderConfig->verbose) { - std::cout << "[AsyncAssembly] Worker " << threadId << " exiting" << std::endl; + VkEncOut() << "[AsyncAssembly] Worker " << threadId << " exiting" << std::endl; } } +// Hands a deferred-frame chain to the assembly workers, and shortens |frames| +// by exactly the frames it hands over: on return |frames| is the part of the +// chain the encoder still owns -- empty when every frame was queued, and the +// unqueued remainder when a push was refused. A queued frame is released by +// the worker that finishes it, so this is what lets the caller release what is +// left without reaching a frame a worker is already assembling. VkResult VkVideoEncoder::QueueFramesForAssembly( VkSharedBaseObj& frames, uint32_t numFrames) { - VkSharedBaseObj current = frames; - while (current != nullptr) { + while (frames != nullptr) { + // Read the link while this thread is still the frame's only owner. + // From the moment the item is on the queue a worker may take it and + // clear the frame's links, so the chain cannot be walked across a push. + VkSharedBaseObj next = frames->dependantFrames; + AssemblyWorkItem item; - item.frameInfo = current; - item.sequenceNumber = m_assemblySequenceCounter++; + item.frameInfo = frames; + // Assign the number, consume it only on a successful push. An + // incremented-then-abandoned number (Push fails only when the queue + // is flushing) would never take its turn, and every later item would + // wait forever on the timeout-less ordering condition variable. + // Safe unlocked: this method is session-serial (submit thread only). + item.sequenceNumber = m_assemblySequenceCounter; bool pushed = m_assemblyQueue.Push(item); if (!pushed) { - fprintf(stderr, "[AsyncAssembly] Failed to push to assembly queue\n"); + VkEncPrintfErr("[AsyncAssembly] Failed to push to assembly queue\n"); return VK_ERROR_OUT_OF_HOST_MEMORY; } + m_assemblySequenceCounter++; - VkSharedBaseObj next = current->dependantFrames; - current = next; + // Ownership of this frame has moved to the queued item. + frames = next; } return VK_SUCCESS; } @@ -1357,13 +3047,43 @@ void VkVideoEncoder::ReleaseAssemblyItem(AssemblyWorkItem& item) } } +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED +// Map the RESOLVED sampler model to the primaries-constant selector. +// +// File-local. nvidia_utils/vulkan/ycbcr_utils.h ships this mapping as +// VkYcbcrModelToYcbcrBtStandard(), but that function sits behind `#ifdef +// VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709`, which names an ENUMERATOR and +// never a macro, so the block compiles in no translation unit. The other +// users carry their own copy for the same reason +// (common/libs/VkCodecUtils/pattern.cpp and VulkanFilterYuvCompute.cpp, both +// spelled GetYcbcrPrimariesConstantsId). Un-gating the header would impose a +// Vulkan header on every includer, which ycbcr_utils.h deliberately avoids. +// +// The input is the model this file already RESOLVED, never +// matrix_coefficients, so the constants and the model cannot name different +// matrices. +static YcbcrBtStandard VkEncModelToBtStandard( + VkSamplerYcbcrModelConversion modelConversion) +{ + switch (modelConversion) { + case VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_601: return YcbcrBtStandardBt601Ebu; + case VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_2020: return YcbcrBtStandardBt2020; + case VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709: return YcbcrBtStandardBt709; + default: break; + } + // Unreachable: ResolveRgbToYcbcrMatrix only ever emits the three above. + return YcbcrBtStandardBt709; +} + +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConfig) { if (!VulkanVideoCapabilities::IsCodecTypeSupported(m_vkDevCtx, m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(), encoderConfig->codec)) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "The video codec " << VkVideoCoreProfile::CodecToName(encoderConfig->codec) << " is not supported!" << std::endl; return VK_ERROR_INITIALIZATION_FAILED; @@ -1380,7 +3100,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf VkResult result = encoderConfig->InitDeviceCapabilities(m_vkDevCtx); if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "InitDeviceCapabilities() failed. VkResult: " << result << " (0x" << std::hex << result << std::dec << ")" << " - The video profile/format may not be supported by the driver." << std::endl; @@ -1388,7 +3108,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf } if (encoderConfig->qualityLevel >= encoderConfig->videoEncodeCapabilities.maxQualityLevels) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Quality level " << encoderConfig->qualityLevel << " is greater than the maximum supported quality level " << (encoderConfig->videoEncodeCapabilities.maxQualityLevels - 1) << std::endl; @@ -1397,21 +3117,21 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf if (encoderConfig->useDpbArray == false && (encoderConfig->videoCapabilities.flags & VK_VIDEO_CAPABILITY_SEPARATE_REFERENCE_IMAGES_BIT_KHR) == 0) { - std::cout << "Separate DPB was requested, but the implementation does not support it!" << std::endl; - std::cout << "Fallback to layered DPB!" << std::endl; + VkEncOut() << "Separate DPB was requested, but the implementation does not support it!" << std::endl; + VkEncOut() << "Fallback to layered DPB!" << std::endl; encoderConfig->useDpbArray = true; } if (m_encoderConfig->enableQpMap) { if ((m_encoderConfig->qpMapMode == EncoderConfig::DELTA_QP_MAP) && ((m_encoderConfig->videoEncodeCapabilities.flags & VK_VIDEO_ENCODE_CAPABILITY_QUANTIZATION_DELTA_MAP_BIT_KHR) == 0)) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Delta QP Map was requested, but the implementation does not support it!" << std::endl; return VK_ERROR_INITIALIZATION_FAILED; } if ((m_encoderConfig->qpMapMode == EncoderConfig::EMPHASIS_MAP) && ((m_encoderConfig->videoEncodeCapabilities.flags & VK_VIDEO_ENCODE_CAPABILITY_EMPHASIS_MAP_BIT_KHR) == 0)) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " << "Emphasis Map was requested, but the implementation does not support it!" << std::endl; return VK_ERROR_INITIALIZATION_FAILED; } @@ -1422,7 +3142,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf const char* modeString = nullptr; if (!VulkanVideoCapabilities::IsVideoEncodeIntraRefreshSupported(m_vkDevCtx)) { - std::cout << "Intra-refresh has been requested, but the implementation does not support it." << std::endl; + VkEncOut() << "Intra-refresh has been requested, but the implementation does not support it." << std::endl; return VK_ERROR_INITIALIZATION_FAILED; } @@ -1448,13 +3168,13 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf } if ((mode & m_encoderConfig->intraRefreshCapabilities.intraRefreshModes) == 0) { - std::cout << modeString << " intra-refresh was requested, but the implementation does not support it." << std::endl; + VkEncOut() << modeString << " intra-refresh was requested, but the implementation does not support it." << std::endl; return VK_ERROR_INITIALIZATION_FAILED; } if (m_encoderConfig->intraRefreshCycleDuration > m_encoderConfig->intraRefreshCapabilities.maxIntraRefreshCycleDuration) { - std::cout << "The requested intra-refresh cycle duration is greater than the maximum (" + VkEncOut() << "The requested intra-refresh cycle duration is greater than the maximum (" << m_encoderConfig->intraRefreshCapabilities.maxIntraRefreshCycleDuration << ") supported by the implementation" << std::endl; return VK_ERROR_INITIALIZATION_FAILED; @@ -1471,8 +3191,8 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf m_encoderConfig->gopStructure.Init(m_encoderConfig->numFrames); if (encoderConfig->GetMaxBFrameCount() < m_encoderConfig->gopStructure.GetConsecutiveBFrameCount()) { if (m_encoderConfig->verbose) { - std::cout << "Max consecutive B frames: " << (uint32_t)encoderConfig->GetMaxBFrameCount() << " lower than the configured one: " << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() << std::endl; - std::cout << "Fallback to the max value: " << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() << std::endl; + VkEncOut() << "Max consecutive B frames: " << (uint32_t)encoderConfig->GetMaxBFrameCount() << " lower than the configured one: " << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() << std::endl; + VkEncOut() << "Fallback to the max value: " << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() << std::endl; } m_encoderConfig->gopStructure.SetConsecutiveBFrameCount(encoderConfig->GetMaxBFrameCount()); } @@ -1482,19 +3202,44 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf (m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() != 0)) { if (m_encoderConfig->verbose) { - std::cout << "Use of B-frames / compound prediction is not supported when intra-refresh is enabled" << std::endl; - std::cout << "Setting the count of Consecutive B-frames to 0" << std::endl; + VkEncOut() << "Use of B-frames / compound prediction is not supported when intra-refresh is enabled" << std::endl; + VkEncOut() << "Setting the count of Consecutive B-frames to 0" << std::endl; } m_encoderConfig->gopStructure.SetConsecutiveBFrameCount(0); } } + // AV1 CAPTURE CANNOT REORDER IN THIS RELEASE. + // + // This is the definitive check, placed after gopStructure.Init(), after + // the device-maximum clamp and after the intra-refresh adjustment, so it + // reads the count that will actually be encoded rather than the one that + // was requested -- the driver-preferred sentinel in particular only + // resolves in InitDeviceCapabilities. It runs before pools and workers + // start, so a refused session leaves nothing running. + // + // Reordering AV1 emits show-existing-frame headers, and the temporal unit + // is assembled by the file writer rather than by the capture path. A + // captured reordered stream is therefore missing those headers and its + // frame identity cannot be reconstructed. File output keeps B-frames; + // capture keeps B=0, which is what Chromium uses. + if ((m_encoderConfig->codec == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) && + (m_encoderConfig->disableFileOutput != 0) && + (m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() > 0)) { + VkEncErr() << "[VkVideoEncoder] AV1 in-memory capture does not support " + "B-frames in this release (effective consecutive B " + "frames: " + << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() + << "); use file output, or request 0" << std::endl; + return VK_ERROR_FEATURE_NOT_PRESENT; + } + if (m_encoderConfig->verbose) { - std::cout << std::endl << "GOP frame count: " << (uint32_t)m_encoderConfig->gopStructure.GetGopFrameCount(); - std::cout << ", IDR period: " << (uint32_t)m_encoderConfig->gopStructure.GetIdrPeriod(); - std::cout << ", Consecutive B frames: " << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount(); - m_encoderConfig->gopStructure.IsClosedGop() ? std::cout << ", Closed GOP" : std::cout << ", Open GOP"; - std::cout << std::endl; + VkEncOut() << std::endl << "GOP frame count: " << (uint32_t)m_encoderConfig->gopStructure.GetGopFrameCount(); + VkEncOut() << ", IDR period: " << (uint32_t)m_encoderConfig->gopStructure.GetIdrPeriod(); + VkEncOut() << ", Consecutive B frames: " << (uint32_t)m_encoderConfig->gopStructure.GetConsecutiveBFrameCount(); + m_encoderConfig->gopStructure.IsClosedGop() ? VkEncOut() << ", Closed GOP" : VkEncOut() << ", Open GOP"; + VkEncOut() << std::endl; const uint64_t maxFramesToDump = std::min(m_encoderConfig->numFrames, m_encoderConfig->gopStructure.GetGopFrameCount() + 19); m_encoderConfig->gopStructure.PrintGopStructure(maxFramesToDump); @@ -1523,8 +3268,22 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf } - // The required num of DPB images - m_maxDpbPicturesCount = encoderConfig->InitDpbCount(); + // The required num of DPB images. + // Defense-in-depth cap. At H.264 Level >= 5.0 the legacy level-max + // sizing in InitDpbCount() returned 17 Vulkan slots (16 refs + 1 setup; + // the driver's maxDpbSlots=17 / maxActiveReferencePictures=16 + // advertisement is spec-correct). A 17 here breaks the H.264 + // DPB manager's eviction accounting (VkEncDpbH264::IsDpbFull counts 16 + // entries against a threshold of 17 -> eviction never runs -> the + // reference set freezes -> progressive drift). The root fixes are the + // DpbSequenceStart() clamp and need-based InitDpbCount() sizing; this + // cap remains as defense-in-depth for any config path that still yields + // >16, and additionally keeps clear of a driver slot-index-16 + // limitation: the driver hangs the encode engine (fence waits time out + // with VK_ERROR_DEVICE_LOST) when a reference is bound at DPB slot + // index 16. The cap is defense-in-depth on top of the DpbSequenceStart() + // clamp, so that binding is never exercised by this encoder. + m_maxDpbPicturesCount = std::min(encoderConfig->InitDpbCount(), 16u); encoderConfig->InitRateControl(); @@ -1536,7 +3295,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf formatCount, supportedDpbFormats); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to get desired video format for the DPB.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to get desired video format for the DPB.\n"); return result; } @@ -1554,7 +3313,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf formatCount, supportedInFormats); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to get desired video format for input images.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to get desired video format for input images.\n"); return result; } @@ -1563,14 +3322,53 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf // Select the encode-source format that MATCHES the request, rather than taking // whatever the driver happened to list first. // - // The driver returns every format compatible with the profile, and both the set and - // the listing order are its choice. A 4:4:4 profile can advertise the semi-planar - // and the packed form together (packed 4:4:4 rides an RGBA alias and is never a DPB - // format), and nothing says which of them comes first, so element [0] can be either. - // Even among the semi-planar formats [0] is only correct by luck -- the profile - // filter narrows by chroma and bit depth, but nothing guarantees a unique survivor. + // The driver returns every format compatible with the profile, and the order is + // its choice. UNVERIFIED ON THE CURRENT DRIVER: every CAPS_OK (codec, profile) + // pair this project has measured returns exactly ONE encode-source format, so the + // two-entry ordering below has not been observed here. It is stated as the + // assumption the selection is built against rather than rewritten into a + // one-format claim, which would be a per-driver fact and no more of a contract. + // For 4:4:4 it advertises the semi-planar form first and the packed + // form second (packed 4:4:4 rides an RGBA alias and can never be a DPB format), + // so taking element [0] makes a packed input unreachable by construction. Even + // for the semi-planar formats [0] is only correct by luck -- the profile filter + // narrows by chroma and bit depth, but nothing guarantees a unique survivor. m_imageInFormat = VK_FORMAT_UNDEFINED; - const VkFormat requestedInFormat = encoderConfig->input.vkFormat; + // THE REQUEST IS AN ENCODE-SOURCE REQUEST ONLY ON THE DIRECT LANE. + // + // On the filter lane input.vkFormat describes the FILTER's input -- a + // 3-plane I420, a depth-converted P010 source, an RGBA surface -- and no + // driver advertises those as encode sources, nor ever will: they are + // precisely the formats that exist to be converted. Matching against them + // therefore fails by construction and every healthy filtered session + // announced a chroma mismatch about a chroma that matched. + // + // The old gate tested input.colorSpace == kRGB, which caught only the RGB + // half of the filter lane and let the Y'CbCr half -- 3-plane and + // depth-converted -- fall through and print. The FILTER FLAG is the right + // gate on both halves. + // + // AND THE ARGV PATH IS THE PART THAT NEEDS SAYING, because it looks wrong + // and is not: EncoderConfig constructs enablePreprocessComputeFilter TRUE + // and only the ext binder ever writes it down, so on the argv path this + // reads true always. That is correct there, because an argv session is a + // file session and a file input is converted on every frame whatever its + // format (see the preprocessFilterWritesEncodeSource comment below). Do + // not "fix" this by conjoining a colorSpace test: that is what put the + // Y'CbCr filter lane back on the wrong side of the gate. + // + // The packed 4:4:4 encode sources ride RGBA format enums (AYUV on + // R8G8B8A8_UNORM, Y410 on A2B10G10R10_UNORM_PACK32), which are also the + // enums a genuine RGBA session declares. Matching those by enum alone + // would give an RGBA session a packed encode source, and + // SetExternalInputFrame would then find the format in its + // directly-encodable switch and hand the encoder the caller's red, green + // and blue bytes as if they were luma and chroma. An RGBA session is on + // the filter lane, so the gate below excludes it too. + const VkFormat requestedInFormat = + encoderConfig->IsPreprocessComputeFilterEnabled() + ? VK_FORMAT_UNDEFINED + : encoderConfig->input.vkFormat; // --preferPackedYcbcr asks for the packed 4:4:4 encode source (AYUV / Y410) whenever // the driver offers one. It deliberately outranks the input-format match below: the @@ -1587,7 +3385,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf } } if ((m_imageInFormat == VK_FORMAT_UNDEFINED) && encoderConfig->verbose) { - printf("--preferPackedYcbcr: no packed format advertised for this profile; " + VkEncPrintfOut("--preferPackedYcbcr: no packed format advertised for this profile; " "using the normal selection.\n"); } } @@ -1606,30 +3404,85 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf // how a 4:4:4 request ends up encoded as 4:2:0. m_imageInFormat = supportedInFormats[0]; if (requestedInFormat != VK_FORMAT_UNDEFINED) { - fprintf(stderr, - "\nInitEncoder Warning: requested encode-source format %d is not " - "advertised by the driver for this profile; falling back to %d. " - "The encoded chroma format will NOT match the request.\n", + // LOUDER, NOT QUIETER. Narrowing the gate above removed the only + // signal a real mismatch had, so the case that survives it has to + // say what actually happens rather than make a chroma claim it + // cannot support. The surviving case is the DIRECT lane: no + // conversion is configured, so SetExternalInputFrame hands the + // encoder the caller's bytes as they lie, and the encoder reads + // them under the substituted layout. That is a correctness + // problem, not a quality one, and the chroma may well match. + VkEncPrintfErr("\nInitEncoder Error(non-fatal): this session declared " + "encode-source format %d on the DIRECT lane -- no " + "conversion is configured -- and the driver does not " + "advertise it for this profile. Substituting %d. The " + "encoder will read the caller's bytes under a DIFFERENT " + "layout than the one declared.\n", (int)requestedInFormat, (int)m_imageInFormat); // Dump what the driver DOES offer. Without this the fallback tells you only // that your request was refused, not what to ask for instead -- and the set // is profile-dependent, so it cannot be inferred from a static table. - fprintf(stderr, "InitEncoder: driver advertises %u encode-source format(s) " + VkEncPrintfErr("InitEncoder: driver advertises %u encode-source format(s) " "for this profile:", formatCount); for (uint32_t fmtIdx = 0; fmtIdx < formatCount; fmtIdx++) { - fprintf(stderr, " %d", (int)supportedInFormats[fmtIdx]); + VkEncPrintfErr(" %d", (int)supportedInFormats[fmtIdx]); } - fprintf(stderr, "\n"); + VkEncPrintfErr("\n"); } } + // Without the preprocess filter the input image IS the encode source, so ANY + // requested format the driver does not advertise for this profile cannot be + // encoded -- there is nothing left to convert it. Which formats those are is a + // property of the driver and the profile, not of a particular layout or + // subsampling; the filter is what makes the rest reachable. Refuse here, where + // the caller can still be told why and what to ask for instead. + // + // Without this check the fallback above substitutes the driver's first + // advertised format while frames keep arriving in the requested one, and the + // mismatch costs the device rather than the call: VK_ERROR_DEVICE_LOST and a + // 0-byte bitstream, several hundred lines from its cause. + // + // Reachable by default rather than only on an unusual request, which is why it + // is checked here at all: EncoderConfig::input.vkFormat defaults to a format no + // driver advertises as an encode source. + // + // Not supporting a format is a legitimate configuration; losing the device + // over it is not. +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + const bool preprocessFilterAvailable = + (encoderConfig->enablePreprocessComputeFilter != 0); +#else + // Filter compiled out: EncoderConfig has no enablePreprocessComputeFilter + // field to read and there is no filter to turn on, so the refusal below + // is unconditional. + const bool preprocessFilterAvailable = false; +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + + if (!preprocessFilterAvailable && + (requestedInFormat != VK_FORMAT_UNDEFINED) && + (requestedInFormat != m_imageInFormat)) { + VkEncPrintfErr("\nInitEncoder Error: encode-source format %d was requested with the " + "preprocess compute filter disabled, but the driver does not advertise " + "it for this profile. Without the filter the input image is the encode " + "source, so no conversion is possible. Either enable the filter " + "(EncoderConfig::enablePreprocessComputeFilter) or supply one of the " + "%u advertised format(s):", + (int)requestedInFormat, formatCount); + for (uint32_t fmtIdx = 0; fmtIdx < formatCount; fmtIdx++) { + VkEncPrintfErr(" %d", (int)supportedInFormats[fmtIdx]); + } + VkEncPrintfErr("\n"); + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + // State the encode-source format that was actually chosen. Without this the choice is // unobservable from outside: a packed and a 2-plane 4:4:4 source both produce a // yuv444p bitstream, so a --preferPackedYcbcr that silently did nothing would look // exactly like one that worked. if (encoderConfig->verbose) { const VkPackedYcbcrFormatDesc* pPacked = PackedYcbcrFormatDesc(m_imageInFormat); - printf("InitEncoder: encode-source format %d (%s)%s\n", + VkEncPrintfOut("InitEncoder: encode-source format %d (%s)%s\n", (int)m_imageInFormat, (pPacked != nullptr) ? pPacked->debugName : "planar/semi-planar", encoderConfig->preferPackedYcbcr ? " [--preferPackedYcbcr]" : ""); @@ -1648,7 +3501,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf true, supportedQpMapTexelSize); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to get desired video format for qpMap images.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to get desired video format for qpMap images.\n"); return result; } @@ -1661,7 +3514,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf encoderConfig->input.height, m_qpMapTexelSize); if (qpMapFrameCount < encoderConfig->numFrames) { - std::cerr << "Number of QP maps (" << qpMapFrameCount << ") in the input QP map file " + VkEncErr() << "Number of QP maps (" << qpMapFrameCount << ") in the input QP map file " << "is less than the number of frames (" << encoderConfig->numFrames << ") to be encoded." << std::endl; return VK_ERROR_INITIALIZATION_FAILED; @@ -1687,7 +3540,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf // is checked against the reported capability rather than a fixed limit. if ((requestedW > encoderConfig->videoCapabilities.maxCodedExtent.width) || (requestedH > encoderConfig->videoCapabilities.maxCodedExtent.height)) { - fprintf(stderr, "[CAPS] ERROR: requested %ux%u exceeds this profile's maximum " + VkEncPrintfErr("[CAPS] ERROR: requested %ux%u exceeds this profile's maximum " "coded extent %ux%u; refusing to encode a cropped picture\n", requestedW, requestedH, encoderConfig->videoCapabilities.maxCodedExtent.width, @@ -1711,7 +3564,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf encoderConfig->encodeAlignedWidth = vk::alignedSize (encoderConfig->encodeWidth, encoderConfig->videoCapabilities.pictureAccessGranularity.width); encoderConfig->encodeAlignedHeight = vk::alignedSize (encoderConfig->encodeHeight, encoderConfig->videoCapabilities.pictureAccessGranularity.height); - fprintf(stderr, "[CAPS] encode=%ux%u range=[%ux%u..%ux%u] granularity=%ux%u", + VkEncPrintfErr("[CAPS] encode=%ux%u range=[%ux%u..%ux%u] granularity=%ux%u", encoderConfig->encodeWidth, encoderConfig->encodeHeight, encoderConfig->videoCapabilities.minCodedExtent.width, encoderConfig->videoCapabilities.minCodedExtent.height, @@ -1720,11 +3573,11 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf encoderConfig->videoCapabilities.pictureAccessGranularity.width, encoderConfig->videoCapabilities.pictureAccessGranularity.height); if (encoderConfig->encodeWidth != requestedW || encoderConfig->encodeHeight != requestedH) - fprintf(stderr, " (clamped from %ux%u)", requestedW, requestedH); + VkEncPrintfErr(" (clamped from %ux%u)", requestedW, requestedH); if (encoderConfig->encodeAlignedWidth != encoderConfig->encodeWidth || encoderConfig->encodeAlignedHeight != encoderConfig->encodeHeight) - fprintf(stderr, " (aligned to %ux%u)", encoderConfig->encodeAlignedWidth, encoderConfig->encodeAlignedHeight); - fprintf(stderr, "\n"); + VkEncPrintfErr(" (aligned to %ux%u)", encoderConfig->encodeAlignedWidth, encoderConfig->encodeAlignedHeight); + VkEncPrintfErr("\n"); const uint32_t maxActiveReferencePicturesCount = encoderConfig->videoCapabilities.maxActiveReferencePictures; const uint32_t maxDpbPicturesCount = std::min(m_maxDpbPicturesCount, encoderConfig->videoCapabilities.maxDpbSlots); @@ -1800,17 +3653,81 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf sessionCreateInfoChain, m_videoSession); + // RELEASE-VISIBLE, and it has to be. Chromium builds this library + // with NDEBUG, where the assert below compiles to nothing -- so without + // this check a failed vkCreateVideoSessionKHR is not merely + // unhandled, it was never examined, and InitEncoder ran on to report + // VK_SUCCESS. VulkanVideoSession::Create leaves its out-parameter + // untouched on every one of its early error returns, so m_videoSession + // keeps its PREVIOUS value: null on a first init, which + // InitEncoderCodec then dereferences as *m_videoSession while building + // the session parameters; or, when this branch was entered because + // IsCompatible() said no, the stale incompatible session, which is + // worse, because it encodes against a profile the caller never + // negotiated. + // + // The assert is kept alongside the check on purpose: in a debug build + // a failure here should still be fatal at the point of failure. + if (result != VK_SUCCESS) { + VkEncPrintfErr("\nInitEncoder Error: Failed to create the video session (0x%x).\n", result); + assert(result == VK_SUCCESS); + return result; + } + // after creating a new video session, we need a codec reset. + // Success path only: this records that a NEW session needs a codec + // reset, and after a failed Create there is no new session. The flag + // is write-only in this tree today -- nothing reads it -- so the + // placement is currently inert and is correct for when a reader lands. m_resetEncoder = true; - assert(result == VK_SUCCESS); } + // THE ENCODE-SOURCE POOL ASKS ONLY FOR THE USAGE IT PERFORMS, and + // VK_IMAGE_USAGE_STORAGE_BIT is the one bit that has to be earned. + // + // The preprocess compute filter is its only consumer here: the filter + // writes its output through per-plane views of the encode-source image + // bound as VK_DESCRIPTOR_TYPE_STORAGE_IMAGE. Every other producer reaches + // the same image through vkCmdCopyImage, and a directly encodable + // registration is not staged into it at all. + // + // Declaring it on a session that never filters costs profile + // compatibility, which is not a diagnostic detail but the encode itself: + // vkCmdEncodeVideoKHR requires its source image to be compatible with the + // bound session's video profile + // (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-08206). These images carry no + // VkVideoProfileListInfoKHR -- the pool creates them + // VK_IMAGE_CREATE_VIDEO_PROFILE_INDEPENDENT_BIT_KHR instead, so that one + // pool can serve whatever profile the session negotiates -- and a + // profile-independent image is compatible with a profile only while every + // usage it declares is one vkGetPhysicalDeviceVideoFormatPropertiesKHR + // reports for that profile. STORAGE is not among the usages reported for + // an encode-source format, so an unconditional request presents every + // frame of a non-filtering session to the encoder through an image the + // profile does not admit. + // + // Two session shapes can carry a filtered frame, and both are known here: + // * a file input, which is converted on every frame whatever its + // format; and + // * an external registration in the session's FILTER-INPUT format, + // which is by construction a different format from the encode source + // -- a registration that already matches the encode source is either + // encoded from the caller's own image or staged into this pool with + // a copy, and neither path binds a storage view. + const bool preprocessFilterWritesEncodeSource = + encoderConfig->IsPreprocessComputeFilterEnabled() && + (encoderConfig->inputFileHandler.HasFileName() || + (encoderConfig->input.vkFormat != m_imageInFormat)); + const VkImageUsageFlags inImageUsage = ( VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | - VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_STORAGE_BIT | + VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT | - VK_IMAGE_USAGE_TRANSFER_DST_BIT); + VK_IMAGE_USAGE_TRANSFER_DST_BIT | + (preprocessFilterWritesEncodeSource + ? VK_IMAGE_USAGE_STORAGE_BIT + : 0) ); const VkImageUsageFlags dpbImageUsage = VK_IMAGE_USAGE_VIDEO_ENCODE_DPB_BIT_KHR; // Linear staging pool — only needed for file-based input (CPU upload). @@ -1818,7 +3735,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf if (!encoderConfig->repeatInputFrames) { result = VulkanVideoImagePool::Create(m_vkDevCtx, m_linearInputImagePool); if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create linearInputImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create linearInputImagePool.\n"); return result; } @@ -1833,16 +3750,30 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(), (VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT | VK_MEMORY_PROPERTY_HOST_CACHED_BIT), - nullptr, VK_IMAGE_ASPECT_COLOR_BIT, false, false, true); + nullptr, VK_IMAGE_ASPECT_COLOR_BIT, false, false, true, + 0 /* drmFormatModifier */, + // PREINITIALIZED, not the UNDEFINED default. VkVideoEncoder:: + // LoadNextFrame host-writes this image through a persistent + // mapping and only THEN calls StageInputFrame, so the host write + // precedes every barrier this library records on it. UNDEFINED + // says the opposite -- that the contents may be discarded -- and + // left the image with no layout at all for the filter's + // STORAGE_IMAGE descriptors to match, which is + // VUID-vkCmdDraw-None-09600 ("expects GENERAL -- instead, current + // layout is UNDEFINED") once per input plane per frame. + // + // PREINITIALIZED is legal here on both counts the spec attaches to + // it: the tiling is LINEAR and the memory is HOST_VISIBLE. + VK_IMAGE_LAYOUT_PREINITIALIZED); if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure linearInputImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure linearInputImagePool.\n"); return result; } } result = VulkanVideoImagePool::Create(m_vkDevCtx, m_inputImagePool); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create inputImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create inputImagePool.\n"); return result; } @@ -1855,17 +3786,39 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf if (encoderConfig->drmFormatModifierIndex >= 0) { result = SelectDrmFormatModifier(encoderConfig, m_imageInFormat, inImageUsage, imageExtent); if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to select DRM format modifier.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to select DRM format modifier.\n"); return result; } } + // WHICH FAMILY WILL WRITE THIS POOL. + // + // Deliberately not GetStagedInputQueueFamilyIdx(): that answers from + // m_inputComputeFilter, which InitEncoder does not create until several + // hundred lines below here. It would say ENCODE or TRANSFER for every + // session and be wrong for precisely the sessions this matters to. The + // routing decision is made here from the config predicate the filter's + // creation is gated on -- the same one the ext admission gates read. + uint32_t stagedInputQueueFamilyIdx = + ((m_vkDevCtx->GetVideoEncodeQueueFlag() & VK_QUEUE_TRANSFER_BIT) != 0) + ? (uint32_t)m_vkDevCtx->GetVideoEncodeQueueFamilyIdx() + : (uint32_t)m_vkDevCtx->GetTransferQueueFamilyIdx(); +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + if (encoderConfig->IsPreprocessComputeFilterEnabled()) { + stagedInputQueueFamilyIdx = + (uint32_t)m_vkDevCtx->GetComputeQueueFamilyIdx(); + } +#endif + const std::vector inputPoolQueueFamilies = { + (uint32_t)m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(), + stagedInputQueueFamilyIdx }; + result = m_inputImagePool->Configure( m_vkDevCtx, encoderConfig->numInputImages, m_imageInFormat, imageExtent, inImageUsage, - m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(), + inputPoolQueueFamilies, VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT, nullptr, VK_IMAGE_ASPECT_COLOR_BIT, @@ -1875,7 +3828,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf encoderConfig->selectedDrmFormatModifier ); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure inputImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure inputImagePool.\n"); return result; } @@ -1896,7 +3849,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf if (encoderConfig->enableHwLoadBalancing) { if (m_vkDevCtx->GetVideoEncodeNumQueues() < 2) { - std::cout << "\t WARNING: Enabling HW Load Balancing for a device with only " << + VkEncOut() << "\t WARNING: Enabling HW Load Balancing for a device with only " << m_vkDevCtx->GetVideoEncodeNumQueues() << " queue!!!" << std::endl; } @@ -1916,7 +3869,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf if (result == VK_SUCCESS) { m_currentVideoQueueIndx = 0; // start with index zero } - std::cout << "\t Enabling HW Load Balancing for device with " + VkEncOut() << "\t Enabling HW Load Balancing for device with " << m_vkDevCtx->GetVideoEncodeNumQueues() << " queues" << std::endl; } @@ -1927,7 +3880,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf // If the linear tiling is not supported, we need to stage the image result = VulkanVideoImagePool::Create(m_vkDevCtx, m_linearQpMapImagePool); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create linearQpMapImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create linearQpMapImagePool.\n"); return result; } @@ -1952,13 +3905,13 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf true // useLinear ); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure linearQpMapImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure linearQpMapImagePool.\n"); return result; } } result = VulkanVideoImagePool::Create(m_vkDevCtx, m_qpMapImagePool); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create inputImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create inputImagePool.\n"); return result; } @@ -2007,14 +3960,14 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf qpMapMemoryUseLinear // useLinear ); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure qpMapImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure qpMapImagePool.\n"); return result; } } result = VulkanVideoImagePool::Create(m_vkDevCtx, m_dpbImagePool); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create dpbImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create dpbImagePool.\n"); return result; } @@ -2035,7 +3988,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf false // useLinear ); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure inputImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure inputImagePool.\n"); return result; } @@ -2063,7 +4016,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf nullptr, 0, bitstreamBuffer); assert(result == VK_SUCCESS); if (result != VK_SUCCESS) { - fprintf(stderr, "\nERROR: VulkanBitstreamBufferImpl::Create() result: 0x%x\n", result); + VkEncPrintfErr("\nERROR: VulkanBitstreamBufferImpl::Create() result: 0x%x\n", result); break; } @@ -2093,7 +4046,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf if (!m_aqAnalyzes) { - std::cerr << "Failed to create AQ processor (API may not be available in this library)" << std::endl; + VkEncErr() << "Failed to create AQ processor (API may not be available in this library)" << std::endl; return VK_ERROR_INITIALIZATION_FAILED; } @@ -2125,7 +4078,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf config.resourceFlags = nvenc_aq::EncodeAqAnalyzes::AQConfig::AQ_RESOURCE_CPU_UPLOAD; config.maxQueueSlots = encoderConfig->numInputImages; - printf("DEBUG: chromaFormat=%u\n", config.chromaFormat); + VkEncPrintfOut("DEBUG: chromaFormat=%u\n", config.chromaFormat); switch (m_encoderConfig->codec) { case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: @@ -2138,7 +4091,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf config.codecType = nvenc_aq::EncodeAqAnalyzes::AQConfig::AQ_CODEC_AV1; break; default: - std::cerr << "Unknown codec: " << m_encoderConfig->codec << ", defaulting to H.264" << std::endl; + VkEncErr() << "Unknown codec: " << m_encoderConfig->codec << ", defaulting to H.264" << std::endl; config.codecType = nvenc_aq::EncodeAqAnalyzes::AQConfig::AQ_CODEC_H264; break; } @@ -2184,17 +4137,85 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf int result = m_aqAnalyzes->Configure(config); if (result != 0) { assert(!"Failed to configure AQ processor!!!"); - std::cerr << "Failed to configure AQ processor: " << result << std::endl; + VkEncErr() << "Failed to configure AQ processor: " << result << std::endl; return VK_ERROR_INITIALIZATION_FAILED; } } #endif // NV_AQ_GPU_LIB_SUPPORTED +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED if (encoderConfig->enablePreprocessComputeFilter) { - const VkSamplerYcbcrRange ycbcrRange = VK_SAMPLER_YCBCR_RANGE_ITU_FULL; // FIXME - const VkSamplerYcbcrModelConversion ycbcrModelConversion = VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_2020; // FIXME - const YcbcrPrimariesConstants ycbcrPrimariesConstants = GetYcbcrPrimariesConstants(YcbcrBtStandardBt2020); // FIXME + // THE CONVERSION KIND IS DECIDED FIRST, because the colour + // parameters below are meaningful for exactly one of them. + // + // The mechanism choice belongs inside the library, where the device + // capabilities are known, rather than in the embedder. + // m_imageInFormat is not assumed here: it + // was read out of vkGetPhysicalDeviceVideoFormatPropertiesKHR above, + // so a future device that accepts something other than NV12 as an + // encode source changes this derivation without changing a line. + // + // It is decided here rather than at the Create() call below because + // the matrix contract in between REFUSES some code points, and + // refusing them on a YCbCr->YCbCr copy -- which applies no matrix at + // all -- would reject configurations that are entirely correct. + encoderConfig->filterType = + VkEncDeriveFilterType(encoderConfig->input.colorSpace, + m_imageInFormat); + const bool filterAppliesRgbToYcbcrMatrix = + (encoderConfig->filterType == VulkanFilterYuvCompute::RGBA2YCBCR); + + // Colour conversion parameters for the RGBA->YCbCr preprocess filter, + // derived from the VUI the caller asked for. + // + // These were hardcoded to BT.2020 / full range behind three FIXMEs, and + // they are not cosmetic. VulkanFilterYuvCompute::InitRGBA2YCBCR reads + // exactly these two fields back out of the sampler-conversion info to + // choose the shader's matrix (from ycbcrModel) and its range mapping + // (from ycbcrRange). Hardcoding them converted every RGBA frame at + // BT.2020 full range while the bitstream's VUI advertised whatever the + // caller set -- so a conforming decoder was required to mis-colour the + // result. A Chromium session is the concrete case: its config builder + // defaults matrixCoefficients to 1 (BT.709) and videoFullRange to + // VK_FALSE, i.e. the two values furthest from what was being applied. + // + // The VUI fields are the right source precisely because they are what + // the bitstream will advertise: deriving from them makes the conversion + // and the advertisement agree by construction rather than by luck. + // + // WHY THERE IS NO FALLBACK. Falling back to BT.709 for an unexpressible + // matrix value, even announced -- on the argument that silently substituting + // a matrix is how this class of bug is born -- is not enough: the + // substitution applies to the FILTER only and leaves matrix_coefficients + // naming the matrix that was NOT applied, on the + // success path. EncoderConfig::ResolveRgbToYcbcrMatrix() decides instead, + // and it either honours the label, DERIVES one when none was + // named, or refuses -- never diverges. The per-code-point contract + // (CC-1) and its reasons live with that function. + VkSamplerYcbcrModelConversion ycbcrModelConversion = + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709; + if (filterAppliesRgbToYcbcrMatrix) { + if (!encoderConfig->ResolveRgbToYcbcrMatrix(&ycbcrModelConversion)) { + // The reason is printed by the resolver, which is the only + // place that knows which refusal class fired. + return VK_ERROR_INITIALIZATION_FAILED; + } + // Describe the siting this conversion produces, in the same + // branch that decides the matrix, so the two cannot be applied + // to different sessions. + encoderConfig->ApplyPreprocessFilterChromaSiting(); + } + + const VkSamplerYcbcrRange ycbcrRange = encoderConfig->video_full_range_flag ? + VK_SAMPLER_YCBCR_RANGE_ITU_FULL : + VK_SAMPLER_YCBCR_RANGE_ITU_NARROW; + // From the RESOLVED model, not from matrix_coefficients again, so the + // constants and the model can never name different matrices. + const YcbcrBtStandard ycbcrBtStandard = + VkEncModelToBtStandard(ycbcrModelConversion); + const YcbcrPrimariesConstants ycbcrPrimariesConstants = + GetYcbcrPrimariesConstants(ycbcrBtStandard); const VkSamplerYcbcrConversionCreateInfo ycbcrConversionCreateInfo { VK_STRUCTURE_TYPE_SAMPLER_YCBCR_CONVERSION_CREATE_INFO, @@ -2207,8 +4228,25 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf VK_COMPONENT_SWIZZLE_IDENTITY, VK_COMPONENT_SWIZZLE_IDENTITY }, - VK_CHROMA_LOCATION_MIDPOINT, // FIXME - VK_CHROMA_LOCATION_MIDPOINT, // FIXME + // CHROMA SITING, and these are no longer FIXMEs. + // + // MIDPOINT/MIDPOINT (the config default these now read) + // is not a placeholder: it is what the shader produces. + // InitRGBA2YCBCR dispatches one thread per output chroma + // sample and calls GenAverageChromaBlock, a 2x2 box + // average, so the written sample sits at the centre of the + // 2x2 luma block in both axes -- MPEG-1 / JPEG siting. + // + // Nothing on the RGBA arm READS these two: the input is + // fetched with texelFetch through a SAMPLED_IMAGE that + // binds no sampler, and InitRGBA2YCBCR takes only + // ycbcrModel and ycbcrRange back out of this struct. They + // are a DECLARATION. What makes the declaration + // load-bearing is that the same two values now drive + // EncoderConfig::ApplyPreprocessFilterChromaSiting(), + // which puts the siting in the H.26x VUI. + encoderConfig->xChromaOffset, + encoderConfig->yChromaOffset, VK_FILTER_LINEAR, false }; @@ -2227,6 +4265,21 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf // VulkanFilterYuvCompute now supports subsampling uint32_t filterFlags = VulkanFilterYuvCompute::FLAG_NONE; if (encoderConfig->input.msbShift > 0) { + // msbShift > 0 says the source samples are LSB-aligned: a 10-bit + // value occupies 0..1023 of a 16-bit container, so the + // VK_FORMAT_R16_UNORM view the filter reads normalizes it to + // pixel / 65535 -- 1/64 of the intended magnitude at 10-bit. + // GenApplyBlockOutputShift multiplies the written sample by + // 2^msbShift, which is exactly what restores it. + // + // The input MSB-to-LSB shift is that transform's inverse and + // belongs to an already-MSB-aligned source (P010-style, which + // DetectInputMsbShift reports as msbShift == 0). Such a source + // normalizes to (pixel << 6) / 65535 ~= pixel / 1023 on its own and + // needs no shift in either direction. Setting both flags from this + // one condition would cancel them and leave every sample at 1/64 + // scale: a near-black frame that encodes to a structurally valid + // bitstream a fraction of the expected size. filterFlags |= VulkanFilterYuvCompute::FLAG_OUTPUT_LSB_TO_MSB_SHIFT; } #ifdef NV_AQ_GPU_LIB_SUPPORTED @@ -2237,7 +4290,11 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf #endif // NV_AQ_GPU_LIB_SUPPORTED // Enable row/column replication filterFlags |= VulkanFilterYuvCompute::FLAG_ENABLE_ROW_COLUMN_REPLICATION_ALL; - + + // encoderConfig->filterType is derived at the top of this block, + // ahead of the colour parameters: the matrix contract there is scoped + // to RGBA2YCBCR and needs the answer before it can be applied. + result = VulkanFilterYuvCompute::Create(m_vkDevCtx, m_vkDevCtx->GetComputeQueueFamilyIdx(), 0, // queueIndex @@ -2250,6 +4307,52 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf &ycbcrPrimariesConstants, &samplerInfo, m_inputComputeFilter); + + // FATAL, and it has to be. Every ext-path gate that admits a frame + // for conversion -- SupportsFormat, ValidateImageDescriptor, + // RegisterImageResource's VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER -- + // answers from the CONFIG flag (IsPreprocessComputeFilterEnabled), + // while StageInputFrame routes on the OBJECT + // (m_inputComputeFilter != nullptr). The `else` arm below overwrites + // |result| with the command-buffer pool's own VK_SUCCESS, so without + // this return a failed VulkanFilterYuvCompute::Create (push-descriptor + // layout unsupported on a borrowed device, runtime GLSL compile, + // pipeline creation -- Create leaves its out-parameter null and + // returns the error) would leave InitEncoder reporting SUCCESS on a + // session the ext layer believes can convert. A 3-plane frame then + // reaches CopyLinearToOptimalImage, whose 2-region copy from a + // 3-plane source is a VK_ERROR_DEVICE_LOST / GPU hang / 0-byte + // bitstream. An init failure + // with a reason is recoverable; that is not. + if ((result != VK_SUCCESS) || (m_inputComputeFilter == nullptr)) { + VkEncPrintfErr("\nInitEncoder Error: enablePreprocessComputeFilter is set " + "but the preprocess compute filter could not be created " + "(%d).\n", result); + return (result != VK_SUCCESS) ? result : VK_ERROR_INITIALIZATION_FAILED; + } + + // Filter-dispatch observable: record WHICH conversion was built. + // After the fatal check above, so the kind is only ever published + // for a filter that actually exists. Translated to the public + // taxonomy here rather than exposing VulkanFilterYuvCompute's enum, + // which is an internal header the ext consumer does not include. + switch (encoderConfig->filterType) { + case VulkanFilterYuvCompute::RGBA2YCBCR: + m_inputFilterKind.store(INPUT_FILTER_RGBA_TO_YCBCR, + std::memory_order_relaxed); + break; + case VulkanFilterYuvCompute::YCBCR2RGBA: + m_inputFilterKind.store(INPUT_FILTER_YCBCR_TO_RGBA, + std::memory_order_relaxed); + break; + default: + // VkEncDeriveFilterType collapses every YCbCr->YCbCr pair, + // including 3-plane I420 -> 2-plane NV12 and the identity, + // onto YCBCRCOPY. + m_inputFilterKind.store(INPUT_FILTER_YCBCR_COPY, + std::memory_order_relaxed); + break; + } } if ((result == VK_SUCCESS) && (m_inputComputeFilter != nullptr) ) { @@ -2261,7 +4364,7 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf // Allocate subsampled image pool for the new filter capability result = VulkanVideoImagePool::Create(m_vkDevCtx, m_inputSubsampledImagePool); if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create inputSubsampledImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create inputSubsampledImagePool.\n"); return result; } @@ -2300,16 +4403,18 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf ); if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure inputSubsampledImagePool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure inputSubsampledImagePool.\n"); return result; } } #endif // NV_AQ_GPU_LIB_SUPPORTED - } else { + } else +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + { result = VulkanCommandBufferPool::Create(m_vkDevCtx, m_inputCommandBufferPool); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create m_inputCommandBufferPool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create m_inputCommandBufferPool.\n"); return result; } @@ -2326,13 +4431,13 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf } if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure m_inputCommandBufferPool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure m_inputCommandBufferPool.\n"); return result; } result = VulkanCommandBufferPool::Create(m_vkDevCtx, m_encodeCommandBufferPool); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create m_encodeCommandBufferPool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create m_encodeCommandBufferPool.\n"); return result; } @@ -2352,13 +4457,13 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf true // createFences ); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to Configure m_encodeCommandBufferPool.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to Configure m_encodeCommandBufferPool.\n"); return result; } result = CreateFrameInfoBuffersQueue(encoderConfig->numInputImages); if(result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create FrameInfoBuffersQueue.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create FrameInfoBuffersQueue.\n"); return result; } @@ -2370,42 +4475,126 @@ VkResult VkVideoEncoder::InitEncoder(VkSharedBaseObj& encoderConf m_encoderQueueConsumerThread = std::thread(&VkVideoEncoder::ConsumerThread, this); } - if (encoderConfig->IsPsnrMetricsEnabled()) { + // The dma-buf import content probe, if the ext layer has already injected + // one. There is no env var and no build flag gating it, because the + // OPT-IN IS THE CALLER CHAINING VkVideoEncoderImportContentInfo onto a + // registration -- and no allocation happens here either way: the probe's + // device-memory pool is configured lazily on the first capture actually + // recorded, so a session that arms nothing allocates nothing. + m_contentProbeQueueDepth = maxEncodeQueueDepth; + ConfigureContentProbe(); + + // MECHANISM-C shares the PSNR helper's pool machinery but is gated on its own + // env var, so the encoder-input capture can run WITHOUT the PSNR path's + // per-frame host sync. + static const bool kDebugDumpSrc = + (getenv("VKENC_DEBUG_DUMP_SRC") != nullptr); + if (encoderConfig->IsPsnrMetricsEnabled() || kDebugDumpSrc) { if (!m_psnr) { result = VkVideoEncoderPsnr::Create(m_psnr); if (result != VK_SUCCESS) { - fprintf(stderr, "\nInitEncoder Error: Failed to create PSNR helper.\n"); + VkEncPrintfErr("\nInitEncoder Error: Failed to create PSNR helper.\n"); return result; } } result = m_psnr->Configure(m_vkDevCtx, encoderConfig, maxEncodeQueueDepth, m_imageDpbFormat, imageExtent, - m_vkDevCtx->GetVideoEncodeQueueFamilyIdx()); + m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(), + m_imageInFormat); if (result != VK_SUCCESS) { return result; } } - if (m_encoderConfig->asyncAssembly) { - m_asyncAssemblyEnabled = true; - m_assemblySequenceCounter = 0; - m_nextWriteSequence = 0; - m_assemblyErrorCount = 0; - m_assemblyQueue.SetMaxPendingQueueNodes( - encoderConfig->numBitstreamBuffersToPreallocate); - for (uint32_t i = 0; i < m_encoderConfig->assemblyThreadCount; i++) { - m_assemblyThreads.emplace_back( - &VkVideoEncoder::AssemblyWorkerThread, this, (int)i); - } - std::cout << "[AsyncAssembly] Started " << m_encoderConfig->assemblyThreadCount - << " assembly worker threads (queue capacity=" - << (int)encoderConfig->numBitstreamBuffersToPreallocate << ")" - << std::endl; - } + // Zeroed here and ONLY here: they stay monotonic for the life of the + // session, across any number of non-terminal drains (see + // StartAssemblyThreads). + m_assemblySequenceCounter = 0; + m_nextWriteSequence = 0; + m_assemblyErrorCount = 0; + m_frameProcessingErrorCount = 0; + // A false return here means asyncAssembly is off (--syncAssembly), which + // is a configuration and not a failure. The synchronous fallback's own + // guard in ProcessOrderedFrames covers the case where that configuration + // cannot serve the caller's completion needs. + (void)StartAssemblyThreads(); return VK_SUCCESS; } +// THE single site that brings the assembly workers up: called from +// InitEncoder and, for a non-terminal drain, from DrainAndRestartThreads(). +// Idempotent; returns false and does nothing when the session is configured +// for synchronous assembly. +// +// The sequence counters are deliberately NOT reset here. InitEncoder zeroes +// them once, before the first call, and they stay monotonic for the session. +// Resetting them on a restart would happen to be safe -- a joined pipeline +// leaves m_nextWriteSequence == m_assemblySequenceCounter -- but leaving them +// alone is correct WITHOUT depending on that, and a restart that somehow ran +// with work still in flight then wedges on the ordering condition variable +// instead of silently putting two frames on the same turn. +bool VkVideoEncoder::StartAssemblyThreads() +{ + if (!m_encoderConfig || !m_encoderConfig->asyncAssembly) { + return false; + } + if (m_asyncAssemblyEnabled) { + return true; // already up + } + if (!m_assemblyThreads.empty()) { + VkEncPrintfErr("[AsyncAssembly] refusing to start: %u worker(s) " + "still present; the previous pipeline was not joined\n", + (uint32_t)m_assemblyThreads.size()); + return false; + } + // Clears the sticky flush latch a previous SetFlushAndExit() raised. On + // the InitEncoder call the latch was never raised and this is a no-op; on + // a restart it is the step without which every Push() below would be + // refused and QueueFramesForAssembly would fail on the first frame. + if (!m_assemblyQueue.ClearFlushAndReuse()) { + VkEncPrintfErr("[AsyncAssembly] refusing to start: the assembly " + "queue is not drained\n"); + return false; + } + // THE CAPACITY SIDE OF THE INEQUALITY, WHICH MUST NOT BE IGNORED. + // CanAcceptNewInputFrame() refuses while (Size() + burst) > capacity. With + // capacity pinned to numBitstreamBuffersToPreallocate (default 8) any burst + // above 8 makes that test false FOREVER -- Size() 0 does not help -- so the + // ext session answers VK_NOT_READY on every frame and never makes + // progress. That is a hang, not backpressure, and the header permits + // 1..254 consecutive B frames. Take the max so the queue can always hold + // one full burst. + // + // Only the QUEUE capacity moves. numBitstreamBuffersToPreallocate is left + // alone deliberately: it also feeds numInputImages, the async-assembly + // slack and the preallocation check, and growing those would cost real + // image memory for a problem that lives in this queue. + const uint32_t assemblyCapacity = + std::max(m_encoderConfig->numBitstreamBuffersToPreallocate, + (uint32_t)GetMaxAssemblyBurst()); + m_assemblyQueue.SetMaxPendingQueueNodes(assemblyCapacity); + m_assemblyQueueCapacity = assemblyCapacity; + // THE INVARIANT, stated where it can be checked: at the instant + // CanAcceptNewInputFrame() returns true, one EnqueueFrame() must not be + // able to push more than (capacity - Size()) items. Checkable at init as + // capacity >= burst. It holds by construction after the max() above -- + // m_assemblyQueueCapacity is uint32_t and the burst tops out at 257 -- so + // this asserts the construction, not the configuration. + assert(m_assemblyQueueCapacity >= GetMaxAssemblyBurst()); + m_asyncAssemblyEnabled = true; + for (uint32_t i = 0; i < m_encoderConfig->assemblyThreadCount; i++) { + m_assemblyThreads.emplace_back( + &VkVideoEncoder::AssemblyWorkerThread, this, (int)i); + } + VkEncOut() << "[AsyncAssembly] Started " << m_encoderConfig->assemblyThreadCount + << " assembly worker threads (queue capacity=" + << (int)assemblyCapacity << ", burst=" + << (int)GetMaxAssemblyBurst() << ")" + << std::endl; + return true; +} + VkDeviceSize VkVideoEncoder::GetBitstreamBuffer(VkSharedBaseObj& bitstreamBuffer) { VkDeviceSize newSize = m_streamBufferSize; @@ -2429,11 +4618,11 @@ VkDeviceSize VkVideoEncoder::GetBitstreamBuffer(VkSharedBaseObjMemsetData(0x0, copySize, newSize - copySize); #endif if (debugBitstreamBufferDumpAlloc) { - std::cout << "\t\tFrom bitstream buffer pool with size " << newSize << " B, " << + VkEncOut() << "\t\tFrom bitstream buffer pool with size " << newSize << " B, " << newSize/1024 << " KB, " << newSize/1024/1024 << " MB" << std::endl; - std::cout << "\t\t\t FreeNodes " << m_bitstreamBuffersQueue.GetFreeNodesNumber(); - std::cout << " of MaxNodes " << m_bitstreamBuffersQueue.GetMaxNodes(); - std::cout << ", AvailableNodes " << m_bitstreamBuffersQueue.GetAvailableNodesNumber(); - std::cout << std::endl; + VkEncOut() << "\t\t\t FreeNodes " << m_bitstreamBuffersQueue.GetFreeNodesNumber(); + VkEncOut() << " of MaxNodes " << m_bitstreamBuffersQueue.GetMaxNodes(); + VkEncOut() << ", AvailableNodes " << m_bitstreamBuffersQueue.GetAvailableNodesNumber(); + VkEncOut() << std::endl; } } bitstreamBuffer = newBitstreamBuffer; if (newSize > m_streamBufferSize) { - std::cout << "\tAllocated bitstream buffer with size " << newSize << " B, " << + VkEncOut() << "\tAllocated bitstream buffer with size " << newSize << " B, " << newSize/1024 << " KB, " << newSize/1024/1024 << " MB" << std::endl; m_streamBufferSize = (size_t)newSize; } @@ -2472,7 +4661,9 @@ VkDeviceSize VkVideoEncoder::GetBitstreamBuffer(VkSharedBaseObj& imageView, - VkImageLayout oldLayout, VkImageLayout newLayout) + VkImageLayout oldLayout, VkImageLayout newLayout, + uint32_t srcQueueFamilyIndex, + uint32_t dstQueueFamilyIndex) { uint32_t baseArrayLayer = 0; VkImageMemoryBarrier2KHR imageBarrier = { @@ -2485,8 +4676,8 @@ VkImageLayout VkVideoEncoder::TransitionImageLayout(VkCommandBuffer cmdBuf, VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR, // VkAccessFlags dstAccessMask oldLayout, // VkImageLayout oldLayout // FIXME - use the real old layout newLayout, // VkImageLayout newLayout - VK_QUEUE_FAMILY_IGNORED, // uint32_t srcQueueFamilyIndex - VK_QUEUE_FAMILY_IGNORED, // uint32_t dstQueueFamilyIndex + srcQueueFamilyIndex, // uint32_t srcQueueFamilyIndex + dstQueueFamilyIndex, // uint32_t dstQueueFamilyIndex imageView->GetImageResource()->GetImage(), // VkImage image; { // VkImageSubresourceRange subresourceRange @@ -2508,21 +4699,262 @@ VkImageLayout VkVideoEncoder::TransitionImageLayout(VkCommandBuffer cmdBuf, imageBarrier.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; imageBarrier.srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT; imageBarrier.dstStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT; + } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && + (newLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL) && + (srcQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT)) { + // Local patch; not in upstream vk_video_samples. + // + // The staging copy's FOREIGN acquire, split out from the arm below. + // An acquire's FIRST synchronisation and access scopes are IGNORED -- + // the matching release supplies them -- so naming a stage here is + // meaningless, and naming VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT (which + // this pair carried while it served both callers) is worse than + // meaningless: it is the stage that is illegal on this queue family + // the moment the same pair is reached WITHOUT an acquire. + // + // Empty first scope, stated deliberately -- the identical idiom, for + // the identical reason, as the (VIDEO_ENCODE_SRC_KHR -> + // VIDEO_ENCODE_SRC_KHR) FOREIGN acquire further down. + imageBarrier.srcAccessMask = 0; + imageBarrier.dstAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_NONE_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && (newLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)) { - imageBarrier.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT; - imageBarrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; - imageBarrier.srcStageMask = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT; - imageBarrier.dstStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT; + // Local patch; not in upstream vk_video_samples. + // + // THE SAME PAIR, NOT AN ACQUIRE: the staging copy's source-side + // transition for an image this device already owns. Every producer + // that reaches it writes the LINEAR staging image from the HOST or + // with a TRANSFER, never with a compute dispatch: + // + // * the library's own file-input lane, which memcpy's the frame + // into a persistently mapped linear image (LoadNextFrame -> + // CopyYCbCrPlanesDirectCPU) and reaches this pair because a + // non-external frame's srcOldLayout is remapped UNDEFINED -> + // GENERAL before the acquire; + // * an external LOCAL registration that declares GENERAL, which is + // what a caller reusing a host-written staging image must declare. + // + // NOT VK_ACCESS_SHADER_WRITE_BIT / VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + // which would describe compute-filter output. That is wrong in two + // independent ways on this lane, both of them reachable on the file-input + // path: the host's stores + // got NO availability operation before the transfer read (no + // validation error, potentially wrong pixels), and COMPUTE_SHADER is + // not a stage the staging queue family necessarily supports + // (VUID-vkCmdPipelineBarrier2-srcStageMask-09675 -- which exempts + // acquires by its own wording, which is exactly why splitting the + // acquire out above is what makes this arm safe to state correctly). + // + // HOST and TRANSFER together, rather than HOST alone: a first + // synchronisation scope wider than the producer needs is never + // incorrect, and covering both spares the next producer the + // rediscovery. Both are legal on every family this batch can be + // submitted to -- HOST requires no queue capability at all, and + // TRANSFER is implied by COMPUTE and by VIDEO_ENCODE. COMPUTE_SHADER + // is deliberately NOT in the union: it is the one stage that could be + // rejected here, and the only compute producer that can reach this + // pair is a foreign one, which takes the acquire arm above. + imageBarrier.srcAccessMask = VK_ACCESS_2_HOST_WRITE_BIT_KHR | + VK_ACCESS_2_TRANSFER_WRITE_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_HOST_BIT_KHR | + VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; } else if ((oldLayout == VK_IMAGE_LAYOUT_UNDEFINED) && (newLayout == VK_IMAGE_LAYOUT_GENERAL)) { imageBarrier.srcAccessMask = 0; imageBarrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT; imageBarrier.srcStageMask = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT; imageBarrier.dstStageMask = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT; - } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR)) { - imageBarrier.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT; + } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && + (newLayout == VK_IMAGE_LAYOUT_GENERAL) && + (srcQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT)) { + // Local patch; not in upstream vk_video_samples. + // + // The preprocess compute filter's input acquire, FOREIGN half. An + // external producer that hands over an image leaves it in GENERAL, + // and GENERAL is also what the filter reads it in (a STORAGE_IMAGE + // descriptor admits no other layout), so the transition is a no-op in + // layout terms and entirely real in ownership and visibility terms. + // + // Split from the non-acquire arm below by the same rule, and with the + // same idiom, as (GENERAL -> TRANSFER_SRC_OPTIMAL) and + // (TRANSFER_SRC_OPTIMAL -> TRANSFER_SRC_OPTIMAL) above: an acquire's + // FIRST synchronisation and access scopes are IGNORED, the matching + // release supplies them, so the first scope is stated empty rather + // than invented. The VK_ACCESS_SHADER_WRITE_BIT / + // VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT on a pair serving BOTH callers would + // describe a compute producer that only the foreign side could have -- and + // on the foreign side it is ignored. + imageBarrier.srcAccessMask = 0; + imageBarrier.dstAccessMask = VK_ACCESS_2_SHADER_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_NONE_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && (newLayout == VK_IMAGE_LAYOUT_GENERAL)) { + // Local patch; not in upstream vk_video_samples. + // + // THE SAME PAIR, NOT AN ACQUIRE: the filter's input transition for an + // image this device already owns. This is the arm the shipping + // Chromium filter lane actually takes, and the scopes it used to + // carry were wrong for every producer that can reach it. + // + // WHO REACHES IT. The gate at the filter call site is + // isExternalInput, NOT isForeignImport, so a LOCAL registration lands + // here with both families IGNORED. Two producers reach it and + // NEITHER is a compute dispatch: + // + // * the HOST, through a persistent mapping. This is Chromium's + // shipping CPU/shmem staging tier on an I420 or RGBA session: it + // host-writes the image every frame through a coherent mmap, + // declares RESIDENCY_LOCAL, and leaves currentLayout UNDEFINED so + // srcOldLayout is remapped to GENERAL. Its stores got NO + // availability operation before the filter's SHADER_READ. + // * a TRANSFER, which is what the in-tree suite above does with + // vkCmdCopyBufferToImage. + // + // WHY IT IS SILENT, and why it is the copy arm's defect class rather + // than a new one: the layers track + // LAYOUTS, and this arm passes GENERAL -> GENERAL through verbatim, + // so a completely wrong first scope is spec-clean. Worse here than on + // the copy arm in one respect -- there the defect began at frame 2, + // here the UNDEFINED -> GENERAL remap puts frame 1 on it too. + // + // COMPUTE_SHADER IS SAFE IN THIS UNION, and only here. Naming it on + // the copy arm would risk + // VUID-vkCmdPipelineBarrier2-srcStageMask-09675, which is why that + // arm deliberately omits it. This arm is reachable ONLY on a session + // that has an input compute filter, and GetStagedInputSubmitType() + // returns COMPUTE for exactly that session, so the batch is submitted + // on a family that supports it. That is the same argument the local + // filter handback beside it already makes for its own COMPUTE_SHADER. + // It is retained rather than dropped because a genuinely + // compute-written LOCAL producer is a shape a caller may still have. + imageBarrier.srcAccessMask = VK_ACCESS_2_HOST_WRITE_BIT_KHR | + VK_ACCESS_2_TRANSFER_WRITE_BIT_KHR | + VK_ACCESS_2_SHADER_WRITE_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_SHADER_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_HOST_BIT_KHR | + VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR | + VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL) && (newLayout == VK_IMAGE_LAYOUT_GENERAL)) { + // Same acquire, for a producer that hands the image over after a + // transfer read. It does not describe the library's own staged input: + // the staging release hands the image back in the layout the next + // acquire declares, so a reused registration is left in srcOldLayout + // rather than TRANSFER_SRC_OPTIMAL. Kept because an external producer + // may still hand over in TRANSFER_SRC_OPTIMAL. + imageBarrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + imageBarrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT; + } else if ((oldLayout == VK_IMAGE_LAYOUT_PREINITIALIZED) && (newLayout == VK_IMAGE_LAYOUT_GENERAL)) { + // Host-written staging content read by the filter instead of by a + // copy. HOST stages, and therefore never combined with a queue-family + // transfer (VUID-VkImageMemoryBarrier2-srcStageMask-03854) -- the + // callers apply that rule, the same way they already do for the + // PREINITIALIZED -> TRANSFER_SRC_OPTIMAL arm. + imageBarrier.srcAccessMask = VK_ACCESS_HOST_WRITE_BIT; + imageBarrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_HOST_BIT; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT; + } else if ((oldLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR) && (newLayout == VK_IMAGE_LAYOUT_GENERAL)) { + // The filter's OUTPUT image, taken from the encoder's own input pool, + // which hands its nodes out declared VIDEO_ENCODE_SRC_KHR. + imageBarrier.srcAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_SHADER_WRITE_BIT; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT; + } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && + (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR) && + (srcQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT)) { + // Local patch; not in upstream vk_video_samples. + // + // THE FIFTH ARM OF THE SPLIT, and the one the previous four missed. + // Path A's acquire in VkVideoEncoder::RecordVideoCodingCmd names + // pathAProducerLayout as its oldLayout, and that value is whatever the + // producer declared -- which for a caller that leaves + // VkVideoEncoderExternalImageDescriptor::defaultLayout alone is + // GENERAL. So the Path-A acquire lands on THIS pair, not on the + // FOREIGN-guarded (VIDEO_ENCODE_SRC_KHR -> VIDEO_ENCODE_SRC_KHR) arm + // below. Without this arm it falls into the non-acquire arm that follows + // and goes out carrying that arm's compute-producer first scope. + // + // Shared with the non-acquire arm, the Path-A acquire went out + // naming VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT and + // VK_ACCESS_SHADER_WRITE_BIT on a queue family that reports + // TRANSFER|SPARSE|VIDEO_ENCODE and no COMPUTE at all. + // + // Empty first scope, for the same reason and with the same idiom as + // the four sibling arms: an acquire's FIRST synchronisation and access + // scopes are IGNORED -- the matching release supplies them -- so + // naming a stage here is meaningless, and naming a stage the recording + // family does not support is what makes it a latent defect rather than + // merely noise. The exemption that keeps it quiet today is explicit in + // VUID-vkCmdPipelineBarrier2-srcStageMask-09675, which constrains + // srcStageMask to the recording family's stages only when the barrier + // does NOT specify an acquire operation. Share the arm with a + // non-acquire caller -- which is exactly what was happening -- and the + // exemption is gone. + // + // WHAT THIS DOES NOT FIX: the Path-A device loss. The acquire is + // correct in isolation; the RELEASE is the trigger, and the trigger + // is a driver defect. + imageBarrier.srcAccessMask = 0; imageBarrier.dstAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; - imageBarrier.srcStageMask = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_NONE_KHR; imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_GENERAL) && (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR)) { + // THE SAME PAIR, NOT AN ACQUIRE: the compute filter's OUTPUT image on + // its way into the encode. That caller passes no queue families (both + // default to VK_QUEUE_FAMILY_IGNORED), so it keeps the compute-producer + // first scope below, which is correct FOR IT and only for it. + // + // CF-02b. This arm was written for that caller and then had no + // caller: until StageInputFrame's filter arm was taught to hand its + // output over (below), NOTHING in the tree produced this pair with + // IGNORED families, and the encode read the filter's output in + // GENERAL. That is legal only with the unifiedImageLayoutsVideo + // feature, which has zero occurrences anywhere in this library, in + // media/gpu/ or in gpu/vulkan/ -- so it was a real violation of + // VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811, invisible for the + // reason VkVideoEncodeFrameInfo::srcEncodeImageStagedLayout + // documents. + // + // FIRST SCOPE, unchanged and correct: the filter's dispatch WROTE + // this image, and this arm is only ever recorded into the input + // command buffer of a session that HAS a filter -- which is the + // filter's own pool, created on the COMPUTE family -- so + // COMPUTE_SHADER is a stage that family supports. + // + // SECOND SCOPE, CORRECTED from (VIDEO_ENCODE, VIDEO_ENCODE_READ) to + // (ALL_COMMANDS, MEMORY_READ), and this is the whole reason the arm + // could not simply be called as it stood. + // VUID-vkCmdPipelineBarrier2-dstStageMask-09676 requires every stage + // in dstStageMask to be valid for the queue family the command pool + // was created on. A device may well expose a compute family with + // NO VK_QUEUE_VIDEO_ENCODE_BIT_KHR at all and an encode family with + // no COMPUTE, and on such a device no single barrier can name + // COMPUTE_SHADER as its source and VIDEO_ENCODE as its destination + // and be legal anywhere: the two stages have no queue family in + // common. + // Naming VIDEO_ENCODE here would have traded a silent layout + // violation for a loud, and equally real, barrier violation. + // + // ALL_COMMANDS is not a cop-out and it is not the fallback's + // resignation. It carries no queue-capability requirement, so it is + // legal on every family this batch can be recorded on -- and the + // actual reader, vkCmdEncodeVideoKHR, is in a DIFFERENT SUBMISSION + // ordered by the input->encode binary semaphore. A semaphore signal + // makes all prior writes available and its wait makes them visible, + // so the encode's visibility does not come from this barrier's second + // scope in any case; what this barrier owes is the LAYOUT TRANSITION + // and an execution dependency on the dispatch that produced the + // contents. Both are supplied here. + imageBarrier.srcAccessMask = VK_ACCESS_2_SHADER_WRITE_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_MEMORY_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT_KHR; } else if ((oldLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_DPB_KHR) && (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_DPB_KHR)) { imageBarrier.srcAccessMask = VK_ACCESS_2_VIDEO_ENCODE_WRITE_BIT_KHR; imageBarrier.dstAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; @@ -2548,6 +4980,109 @@ VkImageLayout VkVideoEncoder::TransitionImageLayout(VkCommandBuffer cmdBuf, imageBarrier.dstAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; imageBarrier.srcStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT; imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL) && (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR)) { + // Local patch; not in upstream vk_video_samples. + // + // CF-02a, AND THE ARM THE COPY BRANCH'S OWN COMMENT ASKED FOR BY + // NAME. StageInputFrame's copy branch discarded the pool image into + // TRANSFER_DST_OPTIMAL, ran vkCmdCopyImage into it, and then recorded + // NOTHING -- so the encode read its source in TRANSFER_DST_OPTIMAL, + // which VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 forbids. The + // branch is now taught to hand the image over, and this is the arm + // that hand-off lands on. Without it the call would have fallen into + // the total ALL_COMMANDS fallback below: still functionally safe, + // still an actual transition, but a full pipeline stall plus an + // unconditional per-frame "MISSING ARM" line on stderr. + // + // NOTE THE ASYMMETRY WITH THE ARM DIRECTLY ABOVE, because it is the + // one thing a reader is most likely to get wrong when adding the next + // arm: that one is TRANSFER_SRC (a producer that READ the image, so + // TRANSFER_READ) and this one is TRANSFER_DST (vkCmdCopyImage WROTE + // it, so TRANSFER_WRITE). Only a WRITE needs an availability + // operation. Copying the sibling's masks would have produced a + // barrier that makes nothing available and is silent about it -- + // every arm passes oldLayout/newLayout through verbatim, so the + // layers stay happy and a wrong first scope raises no VUID at all. + // + // SECOND SCOPE IS ALL_COMMANDS/MEMORY_READ RATHER THAN + // VIDEO_ENCODE/VIDEO_ENCODE_READ, for the reason spelled out on the + // GENERAL -> VIDEO_ENCODE_SRC_KHR arm above and one more that is + // specific to this pair. The copy branch is NOT the no-filter branch: + // useComputeFilter is (m_inputComputeFilter != nullptr) && (this + // FRAME needs it), so a filter-equipped session routing a frame that + // needs no conversion takes THIS branch while + // m_inputCommandBufferPool is still the filter's compute-family pool. + // A compute family without VK_QUEUE_VIDEO_ENCODE_BIT_KHR makes + // VIDEO_ENCODE in dstStageMask trip + // VUID-vkCmdPipelineBarrier2-dstStageMask-09676 on exactly that + // configuration -- and on no other, which is the shape of a defect + // that ships. ALL_COMMANDS has no queue-capability requirement and is + // legal on all three families this batch can be recorded on + // (COMPUTE, TRANSFER, ENCODE); the encode's visibility comes from the + // input->encode binary semaphore, not from here. + imageBarrier.srcAccessMask = VK_ACCESS_2_TRANSFER_WRITE_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_MEMORY_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL) && + (newLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL) && + (srcQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT)) { + // Local patch; not in upstream vk_video_samples. + // + // THE STAGING COPY'S ACQUIRE FOR A PRODUCER THAT DECLARES + // TRANSFER_SRC_OPTIMAL, foreign half. The copy arm's acquire + // transitions TO TRANSFER_SRC_OPTIMAL, so a caller whose declared + // input layout IS TRANSFER_SRC_OPTIMAL produces an equal-layout pair + // -- and equal layouts are not a no-op: this is still an ownership + // transfer and still the only memory dependency between the + // producer's writes and vkCmdCopyImage's read. + // + // Split from the non-acquire arm below for exactly the reason, and + // with exactly the idiom, that (GENERAL -> TRANSFER_SRC_OPTIMAL) is + // split above: an acquire's FIRST synchronisation and access scopes + // are IGNORED -- the matching release supplies them -- so the first + // scope is stated empty rather than invented. + imageBarrier.srcAccessMask = 0; + imageBarrier.dstAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_NONE_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL) && + (newLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)) { + // Local patch; not in upstream vk_video_samples. + // + // THE SAME PAIR, NOT AN ACQUIRE. This is the arm the library had been + // documenting for two releases without having: VkVideoEncoder.cpp's + // staging release (:1290-1294) and its filter release (:1513-1517) + // both promise, in as many words, that "a caller that declares + // TRANSFER_SRC_OPTIMAL round-trips today". It did not round-trip. It + // reached the terminal else, which under __cpp_exceptions -- the + // standalone CMake build -- is an uncaught throw from a function with + // no handler anywhere on its call stack, i.e. std::terminate, and + // under Chromium's -fno-exceptions build is a SILENT fall-through + // that leaves the struct defaults: a VIDEO_ENCODE second scope in + // front of a vkCmdCopyImage TRANSFER read, with no diagnostic. + // + // The declaration is not exotic. It is what a caller that pools a + // staging image and last used it as a copy source must state to be + // truthful, it is what the ext layer's own legacy wrap uses as its + // default, and the library's residual-layout record hands this exact + // value back to such a caller on every frame. + // + // SCOPES: mirror the non-acquire (GENERAL -> TRANSFER_SRC_OPTIMAL) + // arm above verbatim, because the producer is the same producer -- + // the host through a persistent mapping, or a transfer -- and only + // the layout it chose to name differs. In particular COMPUTE_SHADER + // is deliberately absent: it is the one stage the staging queue + // family may not support + // (VUID-vkCmdPipelineBarrier2-srcStageMask-09675), and the only + // compute producer that can reach this pair is a foreign one, which + // takes the acquire arm above. + imageBarrier.srcAccessMask = VK_ACCESS_2_HOST_WRITE_BIT_KHR | + VK_ACCESS_2_TRANSFER_WRITE_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_HOST_BIT_KHR | + VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; } else if ((oldLayout == VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL) && (newLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)) { imageBarrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; imageBarrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; @@ -2558,10 +5093,136 @@ VkImageLayout VkVideoEncoder::TransitionImageLayout(VkCommandBuffer cmdBuf, imageBarrier.dstAccessMask = 0; imageBarrier.srcStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT; imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_BOTTOM_OF_PIPE_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_PREINITIALIZED) && (newLayout == VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL)) { + // Local patch; not in upstream vk_video_samples. + // A host-written LINEAR external input. Chromium's VulkanVideoEncode- + // Accelerator shmem-staging path host-fills the NV12 planes through a + // persistent mapping and then SUBMITS the frame with + // currentLayout=PREINITIALIZED -- forwarded as desc.defaultLayout, and + // reaching this table as srcExternalImageLayout. That SUBMIT-time + // layout is what this arm keys on. It is NOT the staging image's + // create-time layout: that is initialLayout=UNDEFINED, as + // VUID-VkImageCreateInfo-pNext-01443 requires of an external-memory + // image (vulkan_video_encode_accelerator.cc:1604). This comment used to + // say Chromium created the image PREINITIALIZED; that stopped being + // true when Chromium moved to UNDEFINED, and behaviour never depended + // on it. Chromium hands the frame to SetExternalInputFrame; the library + // stages it to an OPTIMAL encode image via CopyLinearToOptimalImage. The + // upstream transition table only covers producer-left GENERAL images, so + // PREINITIALIZED fell through to the (exceptions-off) empty else and kept + // the default VIDEO_ENCODE-stage barrier -> wrong sync for the following + // transfer read. Make the host writes available to the copy. + imageBarrier.srcAccessMask = VK_ACCESS_HOST_WRITE_BIT; + imageBarrier.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_HOST_BIT; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_TRANSFER_BIT; + } else if ((oldLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR) && + (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR) && + (srcQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT)) { + // Local patch; not in upstream vk_video_samples. + // + // THE PATH-A ACQUIRE ALREADY TAKES THIS PAIR AND HAD NO ARM. A + // producer that hands over an encode-source image declares + // VIDEO_ENCODE_SRC_KHR, and the acquire transitions it to + // VIDEO_ENCODE_SRC_KHR -- an ownership transfer with no layout + // change. With no arm it fell to the else below, which in Chromium's + // -fno-exceptions build silently kept the struct defaults and in the + // standalone CMake build THROWS. It has been surviving in Chromium + // only because those defaults happen to suit an acquire feeding a + // video-encode read; stating it makes that luck a contract. + imageBarrier.srcAccessMask = 0; + imageBarrier.dstAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_NONE_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + } else if ((oldLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR) && + (newLayout == VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR)) { + // Same pair, NOT an acquire. The arm above deliberately leaves the + // first synchronisation scope empty because a release operation + // supplies it and an acquire's is ignored; an intra-queue barrier with + // those masks would be ordered against nothing. This table dispatches + // on the layout pair and would otherwise hand the acquire's scopes to + // any future caller. No such caller exists today -- the only site that + // produces this pair is the Path-A FOREIGN acquire -- so this arm + // exists to keep the next one from inheriting the wrong scope. + imageBarrier.srcAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; } else { -#ifdef __cpp_exceptions - throw std::invalid_argument("unsupported layout transition!"); -#endif + // Local patch; not in upstream vk_video_samples. + // + // THE TOTAL FALLBACK, replacing a construct that was the worst of + // both worlds: `throw std::invalid_argument` under __cpp_exceptions + // and an EMPTY BODY without it. That is precisely inverted with + // respect to which build ships. The standalone CMake build -- which + // defines __cpp_exceptions, has no handler anywhere on + // StageInputFrame's call stack, and is a test harness -- got + // std::terminate. Chromium, which compiles this file -fno-exceptions + // and is the build that actually ships, fell THROUGH the empty else + // and recorded the barrier with the struct defaults set at the top of + // this function: srcStageMask NONE, srcAccessMask 0, dstStageMask + // VIDEO_ENCODE, dstAccessMask VIDEO_ENCODE_READ. In front of a + // vkCmdCopyImage that is the wrong second scope, and it is silent -- + // the layers cannot object, because oldLayout/newLayout are passed + // through verbatim so the layout tracker stays consistent and no VUID + // is violated. A wrong barrier with no diagnostic is the single + // hardest defect class in this file to find; two of the arms above + // exist because it was found the hard way. + // + // WHY NOT KEEP THE THROW: a library that terminates the embedder's + // process because a caller named a legal VkImageLayout this table has + // not been taught yet is not a library. And it terminated only in the + // build that cannot ship, so it bought no shipping safety at all. + // + // WHY NOT RETURN AN ERROR: three of this function's ten call sites + // discard the return value entirely and two more cast it to (void). + // Threading a status out would be a far wider change than the defect + // warrants and would give the discarding sites nothing. + // + // WHY THESE MASKS ARE SAFE ON EVERY QUEUE THIS CAN RECORD ON: + // VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT carries no queue-capability + // requirement, so unlike COMPUTE_SHADER or VIDEO_ENCODE it can + // trip neither VUID-vkCmdPipelineBarrier2-srcStageMask-09675 nor + // -dstStageMask-09676 on the transfer, compute or encode family this + // batch may be submitted to. MEMORY_READ|MEMORY_WRITE expand to every + // access type the stages support. + // + // THE COST, STATED: on an unhandled pair this degenerates to a full + // pipeline stall for this image. That is the correct price for "the + // library does not know what your producer did", and it is paid only + // on a pair no arm claims. + imageBarrier.srcStageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT_KHR; + imageBarrier.srcAccessMask = VK_ACCESS_2_MEMORY_WRITE_BIT_KHR; + imageBarrier.dstStageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT_KHR; + imageBarrier.dstAccessMask = VK_ACCESS_2_MEMORY_READ_BIT_KHR | + VK_ACCESS_2_MEMORY_WRITE_BIT_KHR; + // ALL_COMMANDS does NOT include VK_PIPELINE_STAGE_2_HOST_BIT, and a + // host-written producer is the most likely unhandled caller to arrive + // here, so its stores would otherwise get no availability operation. + // Added ONLY when there is no ownership transfer: HOST stages + // combined with a queue-family transfer is + // VUID-VkImageMemoryBarrier2-srcStageMask-03854. An acquire or a + // release names a real family on one side, so either one reaching + // this fallback takes it without HOST, whatever layout pair carried + // it in. + if ((srcQueueFamilyIndex == VK_QUEUE_FAMILY_IGNORED) && + (dstQueueFamilyIndex == VK_QUEUE_FAMILY_IGNORED)) { + imageBarrier.srcStageMask |= VK_PIPELINE_STAGE_2_HOST_BIT_KHR; + imageBarrier.srcAccessMask |= VK_ACCESS_2_HOST_WRITE_BIT_KHR; + } + // UNCONDITIONAL, not behind VKENC_DEBUG_LAYOUT. Reaching this branch + // means the table is incomplete for a caller that exists, which is a + // fact the next reader of a log needs whether or not they knew to ask + // for it. It names the pair and both families, because "an unhandled + // pair occurred" is not actionable and a COUNT of them cannot even + // distinguish one recurring pair from several different ones. + VkEncErr() << "[VkVideoEncoder] TransitionImageLayout: no arm for (" + << (int)oldLayout << " -> " << (int)newLayout + << ") srcQueueFamily=" << (int)srcQueueFamilyIndex + << " dstQueueFamily=" << (int)dstQueueFamilyIndex + << "; recording the conservative ALL_COMMANDS fallback. " + "This is a MISSING ARM, not a supported shape." + << std::endl; } const VkDependencyInfoKHR dependencyInfo = { @@ -2577,9 +5238,282 @@ VkImageLayout VkVideoEncoder::TransitionImageLayout(VkCommandBuffer cmdBuf, }; m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &dependencyInfo); + // THE BARRIER PROGRAM THIS TABLE ACTUALLY SELECTED, under the same env + // var as the residual/restore probes beside it. + // + // WHY THIS EXISTS: most of what this function decides is invisible to the + // validation layers. The layers track LAYOUTS, and every arm here passes + // oldLayout/newLayout through verbatim, so a completely wrong pair of + // synchronisation scopes is spec-clean and silent -- which is how a + // wrong pair of scopes survives review. A test that wants to assert the + // library made the right + // memory dependency has no other way to see it: no public observable + // reports a barrier's masks, and a COUNT of barriers cannot distinguish a + // right one from a wrong one. + static const bool kDebugLayout = + (getenv("VKENC_DEBUG_LAYOUT") != nullptr); + if (kDebugLayout) { + VkEncPrintfErr("[LAYOUT-BARRIER] img=%p old=%d new=%d srcQF=%d " + "dstQF=%d srcStage=0x%llx srcAccess=0x%llx " + "dstStage=0x%llx dstAccess=0x%llx\n", + (void*)imageView->GetImageResource()->GetImage(), + (int)oldLayout, (int)newLayout, + (int)srcQueueFamilyIndex, (int)dstQueueFamilyIndex, + (unsigned long long)imageBarrier.srcStageMask, + (unsigned long long)imageBarrier.srcAccessMask, + (unsigned long long)imageBarrier.dstStageMask, + (unsigned long long)imageBarrier.dstAccessMask); + } + return newLayout; } +void VkVideoEncoder::ReleaseImageToForeignQueue(VkCommandBuffer cmdBuf, + VkSharedBaseObj& imageView, + VkImageLayout oldLayout, + VkImageLayout newLayout, + uint32_t srcQueueFamilyIndex, + VkPipelineStageFlags2KHR srcStageMask, + VkAccessFlags2KHR srcAccessMask) +{ + assert(imageView); + // A release is defined only on a queue of its SOURCE family, and an + // ownership transfer needs a real family on at least one side. If the + // caller could not name one there was no acquire either, so record + // nothing rather than a one-sided transfer. + if ((srcQueueFamilyIndex == VK_QUEUE_FAMILY_IGNORED) || + (srcQueueFamilyIndex == VK_QUEUE_FAMILY_FOREIGN_EXT)) { + assert(!"ReleaseImageToForeignQueue: no local source queue family"); + return; + } + // HOST stage + a queue-family transfer is invalid + // (VUID-VkImageMemoryBarrier2-srcStageMask-03854). Unreachable by + // construction -- every isForeignImport arm already excludes + // PREINITIALIZED -- but asserted so a later caller cannot reintroduce it. + assert((srcStageMask & VK_PIPELINE_STAGE_2_HOST_BIT_KHR) == 0); + + uint32_t baseArrayLayer = 0; + const VkImageMemoryBarrier2KHR imageBarrier = { + VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2_KHR, // VkStructureType sType + nullptr, // const void* pNext + srcStageMask, // VkPipelineStageFlags2KHR srcStageMask -- USED by a release + srcAccessMask, // VkAccessFlags2KHR srcAccessMask -- USED by a release + // IGNORED for a release; the spec says to set these to 0. + VK_PIPELINE_STAGE_2_NONE_KHR, // VkPipelineStageFlags2KHR dstStageMask + 0, // VkAccessFlags2KHR dstAccessMask + // + // WHY THE CALLER CHOOSES BOTH LAYOUTS, and what does NOT decide it. + // + // An earlier version of this comment argued that a transition here + // is "unsatisfiable" because the spec requires a release and its + // acquire to repeat identical layouts and a + // VK_QUEUE_FAMILY_FOREIGN_EXT consumer records no Vulkan acquire. + // THAT ARGUMENT WAS WRONG on both halves. A transition is defined + // by any barrier whose layouts differ, independently of pairing; + // the equality rule exists only so a transition submitted twice + // executes once. And the rule never reaches this barrier anyway: a + // release runs on a queue of the SOURCE family and its acquire on + // the DESTINATION family, so our release (local -> FOREIGN) and + // our own next acquire of the same image (FOREIGN -> local) are + // two OPPOSITE-DIRECTION transfers, not two halves of one. Nothing + // in the spec binds their layouts together. + // + // What DOES decide it is the caller: |newLayout| should be the + // layout the NEXT acquire of the same resource will declare, so + // the two barriers do not assert different things about one + // instant. Lanes whose producer re-declares the layout it handed + // over pass oldLayout == newLayout, which defines no transition + // and is the cheapest correct thing to say. + // + // NOTE WHAT THIS DOES NOT CLAIM. The spec sequences a queue-family + // layout transition to happen-after the release and happen-before + // the acquire, and no acquire is ever recorded by a FOREIGN + // consumer -- so the library can NAME a layout here but cannot + // promise the foreign agent observes it, and contents are + // undefined after a release regardless. This is internal + // self-consistency, not a guarantee to the consumer. + oldLayout, // VkImageLayout oldLayout + newLayout, // VkImageLayout newLayout + srcQueueFamilyIndex, // uint32_t srcQueueFamilyIndex + VK_QUEUE_FAMILY_FOREIGN_EXT, // uint32_t dstQueueFamilyIndex + imageView->GetImageResource()->GetImage(), // VkImage image + { + // Must match the acquire's subresource range exactly. + VK_IMAGE_ASPECT_COLOR_BIT, // VkImageAspectFlags aspectMask + 0, // uint32_t baseMipLevel + 1, // uint32_t levelCount + baseArrayLayer, // uint32_t baseArrayLayer + 1, // uint32_t layerCount + }, + }; + + const VkDependencyInfoKHR dependencyInfo = { + VK_STRUCTURE_TYPE_DEPENDENCY_INFO_KHR, + nullptr, + // 0, not BY_REGION: by-region has meaning only inside a render pass + // instance, and an ownership transfer is not a per-region operation. + 0, + 0, + nullptr, + 0, + nullptr, + 1, + &imageBarrier, + }; + m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &dependencyInfo); + + // Record-time observable. A missing release is invisible to the validation + // layers -- that is exactly why it survived -- so counting is the only way + // to prove it ran, and to prove it does NOT run on the lanes that never + // acquired. + static const bool kDebugQfot = + (getenv("VKENC_DEBUG_QFOT") != nullptr); + if (kDebugQfot) { + // Both layouts: oldLayout is the one VUID-...-oldLayout-01197 + // constrains, and printing only one makes the copy and filter lanes + // indistinguishable in a log when they agree on it. + VkEncPrintfErr("[QFOT-REL] img=%p old=%d new=%d srcFamily=%u -> " + "FOREIGN srcStage=0x%llx srcAccess=0x%llx\n", + (void*)imageView->GetImageResource()->GetImage(), + (int)oldLayout, (int)newLayout, srcQueueFamilyIndex, + (unsigned long long)srcStageMask, + (unsigned long long)srcAccessMask); + } +} + +VkImageLayout VkVideoEncoder::RestoreStagedInputLayout(VkCommandBuffer cmdBuf, + VkSharedBaseObj& imageView, + VkImageLayout residualLayout, + VkImageLayout declaredLayout, + VkPipelineStageFlags2KHR srcStageMask, + VkAccessFlags2KHR srcAccessMask) +{ + assert(imageView); + + // SUBSTITUTION FIRST, so that everything below -- including the + // equal-layout early return -- reasons about the layout the image will + // ACTUALLY be left in rather than about the one that was asked for. + // + // Neither UNDEFINED nor PREINITIALIZED is a legal barrier destination + // (VUID-VkImageMemoryBarrier2-newLayout-01198), and PREINITIALIZED is + // unrestorable by construction: it asserts "never yet in any other + // layout since creation", which can be true at most once in an image's + // life and is false the moment our own acquire moves it. GENERAL is the + // only other layout in which host access to a LINEAR image is defined, + // which is what such a caller does between frames, and it is a legal + // destination. + // + // This used to record nothing here. See the header for why that was + // right then and wrong now: the return value is stored on the + // registration's node and named as the NEXT acquire's oldLayout, so the + // substituted layout is not a second false declaration -- it is the one + // true statement in the sequence. + VkImageLayout targetLayout = declaredLayout; + if ((declaredLayout == VK_IMAGE_LAYOUT_UNDEFINED) || + (declaredLayout == VK_IMAGE_LAYOUT_PREINITIALIZED)) { + targetLayout = VK_IMAGE_LAYOUT_GENERAL; + static const bool kDebugLayout = + (getenv("VKENC_DEBUG_LAYOUT") != nullptr); + if (kDebugLayout) { + VkEncPrintfErr("[LAYOUT-RESTORE] SUBSTITUTED img=%p residual=%d " + "declared=%d -> GENERAL (declared layout is not a " + "legal barrier destination; the library records " + "GENERAL and names it on the next acquire)\n", + (void*)imageView->GetImageResource()->GetImage(), + (int)residualLayout, (int)declaredLayout); + } + } + + // Our arm already leaves the image where the next acquire will name it. + // Record nothing -- a barrier here would be a no-op transition whose only + // effect is to construct a pair the layout table has no arm for. The + // RETURN VALUE is still |targetLayout|, so the caller's record is right + // on this branch too. + if (residualLayout == targetLayout) { + return targetLayout; + } + + // HOST is the destination scope because the only consumer between this + // handback and the caller's next acquire is the caller writing the LINEAR + // staging image through a persistent mapping -- which is also why the + // layout being restored to is one in which host access is defined. + // + // HOST is additionally the one destination stage that CANNOT be rejected + // for the recording queue: it requires no queue capability, so it cannot + // trip VUID-vkCmdPipelineBarrier2-dstStageMask-09676. That matters + // concretely -- an earlier attempt at this fix reached for the layout + // table's compute-filter scopes here and emitted + // VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT on a queue family that reports + // TRANSFER|SPARSE_BINDING|VIDEO_ENCODE, producing 24 new validation + // errors. The shipping PREINITIALIZED -> TRANSFER_SRC_OPTIMAL arm already + // names HOST stages on that exact family every frame with none. + // + // No queue-family transfer (both families IGNORED), so + // VUID-VkImageMemoryBarrier2-srcStageMask-03854 -- which forbids HOST + // stages combined with an ownership transfer, and which + // ReleaseImageToForeignQueue asserts against for that reason -- does not + // apply here. + uint32_t baseArrayLayer = 0; + const VkImageMemoryBarrier2KHR imageBarrier = { + VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2_KHR, // VkStructureType sType + nullptr, // const void* pNext + srcStageMask, // VkPipelineStageFlags2KHR srcStageMask + srcAccessMask, // VkAccessFlags2KHR srcAccessMask + VK_PIPELINE_STAGE_2_HOST_BIT_KHR, // VkPipelineStageFlags2KHR dstStageMask + VK_ACCESS_2_HOST_WRITE_BIT_KHR | + VK_ACCESS_2_HOST_READ_BIT_KHR, // VkAccessFlags2KHR dstAccessMask + // oldLayout: the literal our arm named as its acquire newLayout, + // which is what VUID-VkImageMemoryBarrier2-oldLayout-01197 + // constrains. + residualLayout, // VkImageLayout oldLayout + // newLayout: what the NEXT acquire of this registration will + // name as its oldLayout -- which is this function's return + // value, recorded by the caller on the registration's node. + // Equal to the caller's declaration whenever that declaration + // is a legal barrier destination, and GENERAL when it is not. + targetLayout, // VkImageLayout newLayout + VK_QUEUE_FAMILY_IGNORED, // uint32_t srcQueueFamilyIndex + VK_QUEUE_FAMILY_IGNORED, // uint32_t dstQueueFamilyIndex + imageView->GetImageResource()->GetImage(), // VkImage image + { + // Must match the acquire's subresource range exactly. + VK_IMAGE_ASPECT_COLOR_BIT, // VkImageAspectFlags aspectMask + 0, // uint32_t baseMipLevel + 1, // uint32_t levelCount + baseArrayLayer, // uint32_t baseArrayLayer + 1, // uint32_t layerCount + }, + }; + + const VkDependencyInfoKHR dependencyInfo = { + VK_STRUCTURE_TYPE_DEPENDENCY_INFO_KHR, + nullptr, + // 0, not BY_REGION: by-region has meaning only inside a render pass + // instance. Same reasoning as ReleaseImageToForeignQueue. + 0, + 0, + nullptr, + 0, + nullptr, + 1, + &imageBarrier, + }; + m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &dependencyInfo); + + static const bool kDebugLayout = + (getenv("VKENC_DEBUG_LAYOUT") != nullptr); + if (kDebugLayout) { + VkEncPrintfErr("[LAYOUT-RESTORE] img=%p old=%d new=%d " + "srcStage=0x%llx srcAccess=0x%llx\n", + (void*)imageView->GetImageResource()->GetImage(), + (int)residualLayout, (int)targetLayout, + (unsigned long long)srcStageMask, + (unsigned long long)srcAccessMask); + } + + return targetLayout; +} + VkResult VkVideoEncoder::CopyLinearToOptimalImage(VkCommandBuffer& commandBuffer, VkSharedBaseObj& srcImageView, VkSharedBaseObj& dstImageView, @@ -2646,10 +5580,53 @@ VkResult VkVideoEncoder::CopyLinearToOptimalImage(VkCommandBuffer& commandBuffer (uint32_t)2, copyRegion); { + // MAKE THE COPY'S WRITE VISIBLE TO THE ENCODE'S READ. + // + // It names the copy's WRITE, not its read, and a real destination stage. + // VK_ACCESS_TRANSFER_READ_BIT as the source and + // VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT as the destination stage are both + // wrong for what this has to accomplish: + // + // * the interesting access this copy performed on the image the + // ENCODE will read is the WRITE to the destination, not the read + // of the source, so TRANSFER_READ made none of the written data + // available; + // * BOTTOM_OF_PIPE as a DESTINATION stage makes nothing visible to + // anything -- it is the end of the pipeline, so there is no + // subsequent stage for the dstAccessMask to apply to. + // + // The consequence is a real read-after-write hazard on the staged + // input lane, not a theoretical one: vkCmdEncodeVideoKHR reads a + // resource vkCmdCopyImage wrote, while the dependency as written + // allows all accesses at VK_PIPELINE_STAGE_2_ALL_TRANSFER_BIT rather + // than VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR at + // VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR. The hazard spans two + // submits, so only submit-time synchronization validation can see + // it. + // + // Naming TRANSFER_WRITE as the + // source access and ALL_COMMANDS as the destination stage covers the + // video-encode stage and its read access without this file having to + // depend on the synchronization2 / video pipeline-stage enums, which + // the surrounding code does not use. + // + // THIS IS A WIDENING, which is why it is safe to make on the shipping + // staged lane: it adds synchronization rather than removing it, so it + // cannot introduce a race that was not already there. It costs a + // stricter dependency at the end of a staging copy that is already + // fenced against the encode submit. + // + // NOT THE QUEUE-FAMILY OWNERSHIP QUESTION, which is separate and is + // NOT addressed here: this pool image is VK_SHARING_MODE_EXCLUSIVE and + // pinned to the encode family, and when a compute filter exists the + // staged batch is recorded and submitted on the COMPUTE family with no + // ownership transfer either way. Synchronization validation does not + // model ownership, so a clean sync run says nothing about it. See the + // note at the staged transitions in StageInputFrame. VkMemoryBarrier memoryBarrier = {VK_STRUCTURE_TYPE_MEMORY_BARRIER}; - memoryBarrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + memoryBarrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; memoryBarrier.dstAccessMask = VK_ACCESS_MEMORY_READ_BIT; - m_vkDevCtx->CmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, + m_vkDevCtx->CmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 1, &memoryBarrier, 0, 0, 0, 0); } @@ -2698,9 +5675,25 @@ VkResult VkVideoEncoder::CopyLinearToLinearImage(VkCommandBuffer& commandBuffer, { VkMemoryBarrier memoryBarrier = {VK_STRUCTURE_TYPE_MEMORY_BARRIER}; - memoryBarrier.srcAccessMask = VK_ACCESS_TRANSFER_READ_BIT; + // Same defect, same fix, as the one corrected in + // CopyLinearToOptimalImage: this copy WRITES the destination, so + // naming its READ made none of the written data available, and + // BOTTOM_OF_PIPE as a DESTINATION stage makes nothing visible to + // anything. The destination here is the QP map, which ProcessQpMap + // chains onto encodeInfo as quantizationMapInfo.quantizationMap and + // vkCmdEncodeVideoKHR then reads -- and with useDedicatedCommandBuf it + // is a SEPARATE submit, i.e. the cross-submit case. + // + // NOT PROVEN. Unlike its sibling this carries no RED: no harness under + // vk_video_encoder/test/ enables qpMap at all, so every "0 hazards" + // result in this tree is silent on this lane by construction rather + // than by cleanliness. Landed anyway because it is the identical + // two-value strict widening of both scopes and therefore cannot + // introduce a race. To prove it, add a qpMap arm (the CLI reaches it + // via --qpMapFileName) and run it with the validation layer FORCED. + memoryBarrier.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; memoryBarrier.dstAccessMask = VK_ACCESS_MEMORY_READ_BIT; - m_vkDevCtx->CmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, + m_vkDevCtx->CmdPipelineBarrier(commandBuffer, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, 1, &memoryBarrier, 0, 0, 0, 0); } @@ -2744,6 +5737,15 @@ void VkVideoEncoder::FillIntraRefreshInfo(VkSharedBaseObj& encodeFrameInfo) { + // Fold any queued mid-stream rate-control update BEFORE the master-gate + // check. The gate is raised at session init only (frame 0 carries + // RESET + RATE_CONTROL + QUALITY_LEVEL); a mid-stream update must raise + // it itself, or the armed update is never consumed and the session runs + // at its initial rate forever. + ApplyPendingRateControlUpdate(); + if (m_sendRateControlCmd) { + m_sendControlCmd = true; + } if (m_sendControlCmd == 0) { return false; } @@ -2800,7 +5802,15 @@ bool VkVideoEncoder::HandleCtrlCmd(VkSharedBaseObj& enco encodeFrameInfo->rateControlInfo.pLayers = encodeFrameInfo->rateControlLayersInfo; encodeFrameInfo->rateControlInfo.layerCount = 1; } - m_beginRateControlInfo = encodeFrameInfo->rateControlInfo; + // NOTE: the begin-coding cache is deliberately NOT written here. + // Upstream this assignment was dead -- the unconditional wipe in + // RecordVideoCodingCmd clobbered it -- and gating that wipe on RESET + // (as the spec requires) brought it back to life, where it declares + // the newly commanded values with only the BASE chain. Begin-coding + // must describe the session's rate-control state in full, codec + // struct included, or VUID-vkCmdBeginVideoCodingKHR-pBeginInfo-08254 + // fires on the update frame. The post-control refresh below installs + // the complete chain; leave the cache to it. if (pNext != nullptr) { vk::ChainNextVkStruct(encodeFrameInfo->rateControlInfo, *pNext); @@ -2866,21 +5876,129 @@ VkResult VkVideoEncoder::RecordVideoCodingCmd(VkSharedBaseObjCmdResetQueryPool(cmdBuf, queryPool, querySlotId, numQuerySamples); - if (encodeFrameInfo->controlCmd != VkVideoCodingControlFlagsKHR()) + if (encodeFrameInfo->controlCmd & VK_VIDEO_CODING_CONTROL_RESET_BIT_KHR) { + // A session RESET returns rate control to the initial (default) + // state, and the begin-coding info must describe exactly that. For + // a mid-stream control command (rate-control update WITHOUT reset) + // the begin info must keep describing the CURRENT state -- the + // control command inside this coding scope then changes it, and the + // cache refresh below picks the new state up for later frames. m_beginRateControlInfo = {VK_STRUCTURE_TYPE_VIDEO_ENCODE_RATE_CONTROL_INFO_KHR, NULL}; - m_beginCodecRateControlInfoValid = false; } encodeBeginInfo.pNext = &m_beginRateControlInfo; - if (getenv("VKENC_DEBUG_PSNR")) { - fprintf(stderr, "[BEGINRC] picType=%d controlCmd=0x%x beginRCmode=%d\n", + static const bool kDebugPsnr = + (getenv("VKENC_DEBUG_PSNR") != nullptr); + if (kDebugPsnr) { + VkEncPrintfErr("[BEGINRC] picType=%d controlCmd=0x%x beginRCmode=%d\n", (int)encodeFrameInfo->gopPosition.pictureType, (unsigned)encodeFrameInfo->controlCmd, (int)m_beginRateControlInfo.rateControlMode); } + // Path-A queue-family acquire. A dma_buf-imported image is owned by + // VK_QUEUE_FAMILY_FOREIGN_EXT; without acquiring it into the encode family + // the encode reads memory it does not own, which is undefined by the spec + // (the staging path's equivalent acquire records having observed solid + // zeros without it). Recorded on the acquiring queue and outside the video + // coding scope. The layout does not change -- the producer already hands + // the image over in VIDEO_ENCODE_SRC_KHR -- so this is purely the ownership + // half of the transfer. Honours the CALLER-DECLARED residency: a caller + // that pools its own images passes RESIDENCY_LOCAL and gets no barrier. + // + // THE PRODUCER LAYOUT, COMPUTED ONCE FOR BOTH HALVES, and at this scope + // deliberately: as a local inside the acquire block below it is out of scope + // for the release at the bottom of this function, which would then have to + // hardcode a literal. Computing it here is what lets the two + // barriers agree. + // + // NEITHER SENTINEL SURVIVES INTO A BARRIER. This value is named as the + // release's newLayout at the bottom of this function, and + // VUID-VkImageMemoryBarrier2-newLayout-01198 forbids BOTH + // VK_IMAGE_LAYOUT_UNDEFINED and VK_IMAGE_LAYOUT_PREINITIALIZED there. + // They arrive for different reasons -- UNDEFINED is the absent + // declaration, PREINITIALIZED is a real one, a producer stating that the + // image is exactly as created and has never been transitioned -- and + // neither is legal in that position. VIDEO_ENCODE_SRC_KHR stands in for + // both. It is the layout the acquire below targets in any case, so the + // substituted pair is an ownership transfer with no layout change, and + // the two halves still name one value. + // + // A caller that needs the image left in a particular layout between + // frames states a legal one per frame; the two the spec reserves for + // image creation cannot be honoured as a handback. + const VkImageLayout declaredInputLayout = + encodeFrameInfo->srcExternalImageLayout; + const VkImageLayout pathAProducerLayout = + ((declaredInputLayout != VK_IMAGE_LAYOUT_UNDEFINED) && + (declaredInputLayout != VK_IMAGE_LAYOUT_PREINITIALIZED)) + ? declaredInputLayout + : VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR; + + if (encodeFrameInfo->srcEncodeImageIsExternal && + (encodeFrameInfo->externalInputResidency == + EXTERNAL_INPUT_RESIDENCY_FOREIGN)) { + VkSharedBaseObj srcEncodeImageView; + if (encodeFrameInfo->srcEncodeImageResource->GetImageView( + srcEncodeImageView)) { + TransitionImageLayout(cmdBuf, srcEncodeImageView, + pathAProducerLayout, + VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR, + VK_QUEUE_FAMILY_FOREIGN_EXT, + (uint32_t)m_vkDevCtx->GetVideoEncodeQueueFamilyIdx()); + } + } + + // ===== THE READER OF THE STAGED-INPUT LAYOUT RECORD ===== + // + // The last point before the video coding scope opens, and therefore the + // last point at which the encoder can still say something about the image + // vkCmdEncodeVideoKHR is about to read. + // + // THIS IS NOT DECORATION, AND IT IS NOT A MIRROR OF THE BARRIER. The + // layer check that should catch this -- VUID-vkCmdEncodeVideoKHR- + // pEncodeInfo-10811 -- reads the image-layout map of THIS command buffer, + // and the staging barriers are recorded into a different one, so the map + // has no entry for this image and ValidateVideoImageLayout returns true + // without comparing anything. The submit-time sweep keys off the same + // registry and is equally blind. So on the staged paths there is no + // instrument in the tree that can see the rule at all, and this record -- + // written by the staging arm, read here by the consumer, in a different + // function and a different command buffer -- is the only thing that can. + // + // MAX_ENUM means the frame was never staged: Path A / RESIDENCY_LOCAL, + // where Chromium declares defaultLayout = VK_IMAGE_LAYOUT_VIDEO_ENCODE_ + // SRC_KHR and the frame is conformant by declaration, and the AV1 + // show-existing pseudo-frames, which have no source picture at all. + // Judging those would be judging a fact this library did not record. + // + // A DIAGNOSTIC, NOT A REFUSAL: by the time this is reachable the frame is + // recorded, submitted and in flight, and there is no correct way to + // unwind it here. VkEncErr rather than assert alone, so it survives the + // Release build that ships -- an NDEBUG-only check on a defect class + // whose entire history is "silent in the build that ships" would be the + // same mistake again. + if ((encodeFrameInfo->srcEncodeImageStagedLayout != VK_IMAGE_LAYOUT_MAX_ENUM) && + (encodeFrameInfo->srcEncodeImageStagedLayout != VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR)) { + VkEncErr() << "[VkVideoEncoder] the staged encode-source image is in " + "layout " << (int)encodeFrameInfo->srcEncodeImageStagedLayout + << ", not VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR (" + << (int)VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR + << "); vkCmdEncodeVideoKHR requires VIDEO_ENCODE_SRC_KHR " + "(VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811). A " + "StageInputFrame arm is missing its hand-off barrier." + << std::endl; + assert(!"staged encode-source image is not in VIDEO_ENCODE_SRC_KHR"); + } + + // ===== MECHANISM-C: capture the ENCODER INPUT, one command before the + // encode reads it, from the encode command buffer itself. ===== + if (m_psnr && m_psnr->SrcCaptureEnabled()) { + m_psnr->CaptureSource(cmdBuf, encodeFrameInfo.get()); + } + vkDevCtx->CmdBeginVideoCodingKHR(cmdBuf, &encodeBeginInfo); if (encodeFrameInfo->controlCmd != VkVideoCodingControlFlagsKHR()) { @@ -2890,58 +6008,101 @@ VkResult VkVideoEncoder::RecordVideoCodingCmd(VkSharedBaseObjcontrolCmd}; vkDevCtx->CmdControlVideoCodingKHR(cmdBuf, &renderControlInfo); - // Cache the new session rate-control state for subsequent frames' BeginCoding. - // The chain head is the codec-specific RC struct (e.g. - // VkVideoEncodeH265RateControlInfoKHR), so walk the chain for the base RC - // struct — casting the head read gopFrameCount as rateControlMode. + // Cache the new session rate-control state for subsequent frames' + // BeginCoding. The state the control command establishes includes + // the codec-specific RC struct and the per-layer codec structs, and + // VUID-vkCmdBeginVideoCodingKHR-pBeginInfo-08254 requires the + // begin-info chain to match that state in FULL -- a base-only cache + // trips validation on every subsequent frame. Everything is + // snapshotted by VALUE: the frame's copies are pool-recycled and + // m_rateControlLayersInfo is mutated by ApplyPendingRateControl- + // Update before the next control command records, so neither may + // back the cache. + const VkBaseInStructure* baseRc = nullptr; + const VkBaseInStructure* codecRc = nullptr; for (const VkBaseInStructure* p = reinterpret_cast(encodeFrameInfo->pControlCmdChain); p != nullptr; p = p->pNext) { - // Capture the codec-specific RC struct too: the BeginCoding chain has to - // match the session state that this CmdControlVideoCodingKHR establishes, - // or vkCmdBeginVideoCodingKHR reports VUID-...-pBeginInfo-08254. The walk - // stops at the base struct, so a codec-specific struct is picked up only - // when it precedes the base one -- which is how the codec encoders build - // the chain. - if ((p->sType == VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_RATE_CONTROL_INFO_KHR) || - (p->sType == VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_RATE_CONTROL_INFO_KHR) || - (p->sType == VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_RATE_CONTROL_INFO_KHR)) { - switch (p->sType) { + switch ((uint32_t)p->sType) { + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_RATE_CONTROL_INFO_KHR: + baseRc = p; + break; + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_RATE_CONTROL_INFO_KHR: + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_RATE_CONTROL_INFO_KHR: + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_RATE_CONTROL_INFO_KHR: + codecRc = p; + break; + default: + break; + } + } + if (baseRc != nullptr) { + m_beginRateControlInfo = + *reinterpret_cast(baseRc); + m_beginRateControlInfo.pNext = nullptr; + if (codecRc != nullptr) { + switch ((uint32_t)codecRc->sType) { case VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_RATE_CONTROL_INFO_KHR: - m_beginCodecRateControlInfo.h264 = - *reinterpret_cast(p); + m_beginRateControlInfoH264 = + *reinterpret_cast(codecRc); + m_beginRateControlInfoH264.pNext = nullptr; + m_beginRateControlInfo.pNext = &m_beginRateControlInfoH264; break; case VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_RATE_CONTROL_INFO_KHR: - m_beginCodecRateControlInfo.h265 = - *reinterpret_cast(p); + m_beginRateControlInfoH265 = + *reinterpret_cast(codecRc); + m_beginRateControlInfoH265.pNext = nullptr; + m_beginRateControlInfo.pNext = &m_beginRateControlInfoH265; + break; + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_RATE_CONTROL_INFO_KHR: + m_beginRateControlInfoAV1 = + *reinterpret_cast(codecRc); + m_beginRateControlInfoAV1.pNext = nullptr; + m_beginRateControlInfo.pNext = &m_beginRateControlInfoAV1; break; default: - m_beginCodecRateControlInfo.av1 = - *reinterpret_cast(p); break; } - m_beginCodecRateControlInfo.base.pNext = nullptr; - m_beginCodecRateControlInfoValid = true; - continue; } - - if (p->sType == VK_STRUCTURE_TYPE_VIDEO_ENCODE_RATE_CONTROL_INFO_KHR) { - m_beginRateControlInfo = *reinterpret_cast(p); - m_beginRateControlInfo.pNext = nullptr; - // The frame's rateControlLayersInfo array is pool-recycled; point the - // cached copy at the encoder's persistent layer storage instead. - if (m_beginRateControlInfo.layerCount > 0) { - m_beginRateControlInfo.pLayers = m_rateControlLayersInfo; + if (m_beginRateControlInfo.layerCount > 0) { + for (uint32_t li = 0; li < ARRAYSIZE(m_beginRateControlLayersInfo); li++) { + m_beginRateControlLayersInfo[li] = + encodeFrameInfo->rateControlLayersInfo[li]; + m_beginRateControlLayersInfo[li].pNext = nullptr; + const VkBaseInStructure* layerExt = + reinterpret_cast( + encodeFrameInfo->rateControlLayersInfo[li].pNext); + if (layerExt != nullptr) { + switch ((uint32_t)layerExt->sType) { + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_RATE_CONTROL_LAYER_INFO_KHR: + m_beginRateControlLayersInfoH264[li] = + *reinterpret_cast(layerExt); + m_beginRateControlLayersInfoH264[li].pNext = nullptr; + m_beginRateControlLayersInfo[li].pNext = + &m_beginRateControlLayersInfoH264[li]; + break; + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_RATE_CONTROL_LAYER_INFO_KHR: + m_beginRateControlLayersInfoH265[li] = + *reinterpret_cast(layerExt); + m_beginRateControlLayersInfoH265[li].pNext = nullptr; + m_beginRateControlLayersInfo[li].pNext = + &m_beginRateControlLayersInfoH265[li]; + break; + case VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_RATE_CONTROL_LAYER_INFO_KHR: + m_beginRateControlLayersInfoAV1[li] = + *reinterpret_cast(layerExt); + m_beginRateControlLayersInfoAV1[li].pNext = nullptr; + m_beginRateControlLayersInfo[li].pNext = + &m_beginRateControlLayersInfoAV1[li]; + break; + default: + break; + } + } } - break; + m_beginRateControlInfo.pLayers = m_beginRateControlLayersInfo; } } - - // Re-link after the walk: both halves are encoder-owned storage, so the chain - // stays valid for every later frame that reuses the cached state. - if (m_beginCodecRateControlInfoValid) { - m_beginRateControlInfo.pNext = &m_beginCodecRateControlInfo; - } } if (m_videoMaintenance1FeaturesSupported) @@ -2976,6 +6137,74 @@ VkResult VkVideoEncoder::RecordVideoCodingCmd(VkSharedBaseObjCmdEndVideoCodingKHR(cmdBuf, &encodeEndInfo); + // Queue-family RELEASE -- the missing half of the Path-A acquire made + // just before CmdBeginVideoCodingKHR. Placed immediately after the coding + // scope closes, mirroring the acquire's placement immediately before it + // opened. In-scope would also be legal, but out-of-scope keeps the two + // halves symmetric and cannot interleave with the PSNR readback below, + // which touches only the reconstructed picture. + // + // The gate is a copy of the acquire's: a frame that did not acquire must + // never release. + if (encodeFrameInfo->srcEncodeImageIsExternal && + (encodeFrameInfo->externalInputResidency == + EXTERNAL_INPUT_RESIDENCY_FOREIGN)) { + VkSharedBaseObj srcEncodeImageViewRel; + if (encodeFrameInfo->srcEncodeImageResource->GetImageView( + srcEncodeImageViewRel)) { + // oldLayout is the LITERAL the acquire above named as its + // newLayout, and nothing transitions this image between the encode + // and here, so the literal IS the current layout -- which is what + // VUID-VkImageMemoryBarrier2-oldLayout-01197 constrains. + // + // newLayout is pathAProducerLayout -- the SAME value the NEXT + // frame's acquire will name as ITS oldLayout -- so the handover + // round-trips the producer's declaration wherever that + // declaration is one a barrier may name as a destination, and the + // substitute computed for it at the top of this function where it + // is not. A hardcoded + // VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR on both sides is correct + // only for a producer that declares VIDEO_ENCODE_SRC_KHR: a + // registration that leaves + // VkVideoEncoderExternalImageDescriptor::defaultLayout alone + // declares GENERAL, so frame 1 would acquire + // (GENERAL -> VIDEO_ENCODE_SRC_KHR) and release + // (VIDEO_ENCODE_SRC_KHR -> VIDEO_ENCODE_SRC_KHR), leaving the + // image in VIDEO_ENCODE_SRC_KHR while frame 2's acquire declares + // GENERAL about it. + // + // THIS DOES NOT MAKE PATH A WORK, and the reason is not in this + // function. The release below is a spec-legal queue-family + // ownership release to VK_QUEUE_FAMILY_FOREIGN_EXT, and a driver + // can lose the device executing one whenever the image carries + // VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR (or the _DPB_ bit) and + // the barrier is recorded on any queue family except graphics or + // optical flow. For a RELEASE that is forced, not a choice: a + // release must run on a queue of its SOURCE family, and the only + // family that owns this image is the encode family. It presents + // as a device loss with an EMPTY VK_EXT_device_fault record, and + // the validation layer reports nothing either way. + // + // No re-expression of this barrier avoids it. The layout pair, + // the stage masks, the external-memory-ness, the video coding + // scope and even the barrier API generation are all irrelevant + // to the outcome. + // + // NOTHING IS SUPPRESSED HERE, DELIBERATELY. Dropping this release + // makes the row encode, but it would leave the acquire above + // permanently unmatched and would silently change what the + // library promises a real dma_buf consumer. That is a design + // decision, not a workaround to slip in under a driver bug. + ReleaseImageToForeignQueue( + cmdBuf, srcEncodeImageViewRel, + VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR, + pathAProducerLayout, + (uint32_t)m_vkDevCtx->GetVideoEncodeQueueFamilyIdx(), + VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR, + VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR); + } + } + if (m_psnr && m_psnr->Enabled() && (encodeFrameInfo->setupImageResource != nullptr)) { VkSharedBaseObj setupEncodeImageView; encodeFrameInfo->setupImageResource->GetImageView(setupEncodeImageView); @@ -3020,22 +6249,126 @@ VkResult VkVideoEncoder::SubmitVideoCodingCmds(VkSharedBaseObjaqProcessorSlot) { + requiredWaits++; + } else +#else + if (encodeFrameInfo->inputCmdBuffer) { + requiredWaits++; + } +#endif // NV_AQ_GPU_LIB_SUPPORTED + if (encodeFrameInfo->qpMapCmdBuffer) { + requiredWaits++; + } + if (encodeFrameInfo->isExternalInput && !encodeFrameInfo->inputCmdBuffer) { + requiredWaits += (uint32_t)encodeFrameInfo->inputWaitSemaphores.size(); + } + // Reserved, not optional: the HW load-balancing pair below appends + // unconditionally on count. + if (m_hwLoadBalancingTimelineSemaphore != VK_NULL_HANDLE) { + requiredWaits++; + } + if (requiredWaits > waitSemaphoreMaxCount) { + VkEncPrintfErr("\nEncoder Error: this frame needs %u wait semaphores but the " + "direct submit array holds %u. Refusing the submit rather than " + "dropping a wait: a dropped wait lets vkCmdEncodeVideoKHR read " + "the input image before the producer that wait names has " + "finished writing it.\n", + requiredWaits, waitSemaphoreMaxCount); + assert(!"direct submit wait array would overflow"); + return VK_ERROR_TOO_MANY_OBJECTS; + } + + // Upper bound on the signal side. The external block's release-timeline + // entry only fires at a queue flush point and the pass-through loop then + // starts at index 1, so that block contributes at most + // inputSignalSemaphores.size(); counting the full size is conservative + // and never refuses a shape that would have been assembled correctly. + uint32_t requiredSignals = 0; + if (frameCompleteSemaphore != VK_NULL_HANDLE) { + requiredSignals++; + } + if (encodeFrameInfo->isExternalInput && !encodeFrameInfo->inputCmdBuffer && + !encodeFrameInfo->inputSignalSemaphores.empty()) { + requiredSignals += (uint32_t)encodeFrameInfo->inputSignalSemaphores.size(); + } + if ((m_completionTimelineSemaphore != VK_NULL_HANDLE) && + (encodeFrameInfo->externalFrameId != uint64_t(-1))) { + requiredSignals++; + } + if (m_hwLoadBalancingTimelineSemaphore != VK_NULL_HANDLE) { + requiredSignals++; + } + if (requiredSignals > signalSemaphoreMaxCount) { + VkEncPrintfErr("\nEncoder Error: this frame needs %u signal semaphores but the " + "direct submit array holds %u. Refusing the submit rather than " + "dropping a signal: a dropped signal is a semaphore nobody ever " + "signals, and whoever waits on it waits forever.\n", + requiredSignals, signalSemaphoreMaxCount); + assert(!"direct submit signal array would overflow"); + return VK_ERROR_TOO_MANY_OBJECTS; + } + } + #ifdef NV_AQ_GPU_LIB_SUPPORTED if (encodeFrameInfo->aqProcessorSlot) { uint64_t inputSeqNumber = encodeFrameInfo->aqProcessorSlot->GetInputSeqNumber(); assert(encodeFrameInfo->frameEncodeInputOrderNum == inputSeqNumber); AqProcessor::SlotState slotState = encodeFrameInfo->aqProcessorSlot->GetState(); VkVideoGopStructure::GopPosition gopPosition = encodeFrameInfo->aqProcessorSlot->GetGopPosition(); - printf("Submitting AQ qpMap inputSeqNumber %" PRIu64 ", type: %s, state: %s\n", inputSeqNumber, + VkEncPrintfOut("Submitting AQ qpMap inputSeqNumber %" PRIu64 ", type: %s, state: %s\n", inputSeqNumber, VkVideoGopStructure::GetFrameTypeName(gopPosition.pictureType), AqProcessor::GetSlotStateDisplayName(slotState)); assert((slotState == AqProcessor::SlotState::GRAPH_COMPLETED) || @@ -3063,7 +6396,17 @@ VkResult VkVideoEncoder::SubmitVideoCodingCmds(VkSharedBaseObjinputCmdBuffer->GetSemaphore(); waitSemaphoreInfos[waitSemaphoreCount].value = 0; // Binary semaphore // Use transfer bit since these semaphores come from transfer operations - waitSemaphoreInfos[waitSemaphoreCount].stageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; + // -- except when the staged input was produced by the preprocess + // COMPUTE filter, which SubmitStagedInputFrame then signals with an + // ALL_COMMANDS scope. Waiting at TRANSFER on that signal would let + // vkCmdEncodeVideoKHR read a pool image the filter has only partially + // written. (The TRANSFER default on the copy branch is left as-is: + // that is pre-existing behaviour on the shipped path, not something + // this change makes reachable.) + waitSemaphoreInfos[waitSemaphoreCount].stageMask = + encodeFrameInfo->inputFilterRecorded + ? VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT + : VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR; waitSemaphoreInfos[waitSemaphoreCount].deviceIndex = 0; waitSemaphoreCount++; } @@ -3149,7 +6492,7 @@ VkResult VkVideoEncoder::SubmitVideoCodingCmds(VkSharedBaseObjexternalFrameId != uint64_t(-1))) { + const uint64_t completionValue = encodeFrameInfo->externalFrameId + 1; + if (completionValue > m_maxSubmittedCompletionValue) { + m_maxSubmittedCompletionValue = completionValue; + } + // Flush point = last EXTERNAL frame of the ordered batch; the chain + // may end with codec pseudo-frames -- the same scan as the release + // block above, for the same reason. + bool completionFlushPoint = true; + for (const VkVideoEncodeFrameInfo* next = encodeFrameInfo->dependantFrames.get(); + next != nullptr; next = next->dependantFrames.get()) { + if (next->isExternalInput) { + completionFlushPoint = false; + break; + } + } + static const bool completionDebug = (getenv("VKENC_COMPLETION_DEBUG") != nullptr); + if (completionDebug) { + VkEncPrintfErr("[CMPDBG] submit id=%llu val=%llu max=%llu last=%llu tail=%d\n", + (unsigned long long)encodeFrameInfo->externalFrameId, + (unsigned long long)completionValue, + (unsigned long long)m_maxSubmittedCompletionValue, + (unsigned long long)m_lastSignaledCompletionValue, + (int)completionFlushPoint); + } + // The capacity guard degrades by SKIPPING the signal -- a later + // flush point catches up via the running max -- rather than + // overflowing the array. + if (completionFlushPoint && + (m_maxSubmittedCompletionValue > m_lastSignaledCompletionValue) && + (signalSemaphoreCount < signalSemaphoreMaxCount)) { + signalSemaphoreInfos[signalSemaphoreCount].sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO_KHR; + signalSemaphoreInfos[signalSemaphoreCount].semaphore = m_completionTimelineSemaphore; + signalSemaphoreInfos[signalSemaphoreCount].value = m_maxSubmittedCompletionValue; + signalSemaphoreInfos[signalSemaphoreCount].stageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + signalSemaphoreInfos[signalSemaphoreCount].deviceIndex = 0; + signalSemaphoreCount++; + m_lastSignaledCompletionValue = m_maxSubmittedCompletionValue; + } + } + if (m_hwLoadBalancingTimelineSemaphore != VK_NULL_HANDLE) { if (m_verbose) { uint64_t currSemValue = 0; VkResult semResult = m_vkDevCtx->GetSemaphoreCounterValue(*m_vkDevCtx, m_hwLoadBalancingTimelineSemaphore, &currSemValue); - std::cout << "\t TL semaphore value: " << currSemValue << ", status: " << semResult << std::endl; + VkEncOut() << "\t TL semaphore value: " << currSemValue << ", status: " << semResult << std::endl; } + // CAPACITY GUARD -- this pair had none. + // + // Every other appender in this function tests before it writes (the + // frameCompleteSemaphore assert, the release-timeline guard, the + // signal pass-through bound, the completion-timeline guard). These two + // did not. With the arrays already full at 8 this wrote a 48-byte + // VkSemaphoreSubmitInfoKHR one past the end of an 8-element STACK + // array -- memory corruption, not a bad frame -- and then set + // waitSemaphoreInfoCount to 9 so the driver read the overrun entry + // too. + // + // The capacity check at the top of this function now reserves a slot + // for this pair, so this cannot fire. It is kept anyway: "cannot fire" + // is a property of code fifty lines above that a later edit can remove + // without ever touching this block, and the cost of being wrong here + // is a stack smash. + assert(waitSemaphoreCount < waitSemaphoreMaxCount); + assert(signalSemaphoreCount < signalSemaphoreMaxCount); + if ((waitSemaphoreCount >= waitSemaphoreMaxCount) || + (signalSemaphoreCount >= signalSemaphoreMaxCount)) { + VkEncPrintfErr("\nEncoder Error: no room for the HW load-balancing timeline " + "semaphore (waits %u/%u, signals %u/%u). Refusing the submit " + "rather than writing past the end of the submit arrays.\n", + waitSemaphoreCount, waitSemaphoreMaxCount, + signalSemaphoreCount, signalSemaphoreMaxCount); + return VK_ERROR_TOO_MANY_OBJECTS; + } waitSemaphoreInfos[waitSemaphoreCount].sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO_KHR; waitSemaphoreInfos[waitSemaphoreCount].pNext = nullptr; waitSemaphoreInfos[waitSemaphoreCount].semaphore = m_hwLoadBalancingTimelineSemaphore; @@ -3239,18 +6662,18 @@ VkResult VkVideoEncoder::SubmitVideoCodingCmds(VkSharedBaseObjGetSemaphoreCounterValue(*m_vkDevCtx, m_hwLoadBalancingTimelineSemaphore, &currSemValue); - std::cout << "\t TL semaphore value ater submit: " << currSemValue << ", status: " << semResult << std::endl; + VkEncOut() << "\t TL semaphore value ater submit: " << currSemValue << ", status: " << semResult << std::endl; const bool waitOnTlSemaphore = false; if (waitOnTlSemaphore) { uint64_t value = encodeFrameInfo->frameEncodeEncodeOrderNum + 1; // wait on the future frameEncodeEncodeOrderNum VkSemaphoreWaitInfo waitInfo = { VK_STRUCTURE_TYPE_SEMAPHORE_WAIT_INFO, nullptr, VK_SEMAPHORE_WAIT_ANY_BIT, 1, &m_hwLoadBalancingTimelineSemaphore, &value }; - std::cout << "\t TL semaphore wait for value: " << value << std::endl; + VkEncOut() << "\t TL semaphore wait for value: " << value << std::endl; semResult = m_vkDevCtx->WaitSemaphores(*m_vkDevCtx, &waitInfo, 1000 * 1000 * 1000 /* 1000 mSec */); semResult = m_vkDevCtx->GetSemaphoreCounterValue(*m_vkDevCtx, m_hwLoadBalancingTimelineSemaphore, &currSemValue); - std::cout << "\t TL semaphore value: " << currSemValue << ", status: " << semResult << std::endl; + VkEncOut() << "\t TL semaphore value: " << currSemValue << ", status: " << semResult << std::endl; } } @@ -3285,12 +6708,30 @@ VkResult VkVideoEncoder::PushOrderedFrames() // Testing only - don't use for production! result = ProcessOutOfOrderFrames(m_lastDeferredFrame, m_numDeferredFrames); } - if (m_asyncAssemblyEnabled) { - m_lastDeferredFrame = nullptr; - } else { - VkVideoEncodeFrameInfo::ResetAndReleaseFrames(m_lastDeferredFrame); - assert(m_lastDeferredFrame == nullptr); + if (result != VK_SUCCESS) { + // The deferred chain is emptied immediately below whether + // the push succeeded or not -- released there when the frames + // never reached the assembly workers, and by the worker that + // finishes each frame when they did -- so no part of the + // failure survives in the frames themselves and this counter + // is the trace the drain reads afterwards. Counted here + // because every push -- the per-frame ones the enqueue makes + // and the final one the drain makes -- arrives through this + // single path. + m_frameProcessingErrorCount++; } + // Whatever is still on the deferred chain is still the + // encoder's: QueueFramesForAssembly removes the frames it hands + // to the assembly workers, and each of those is released by the + // worker that finishes it. What is left never reached the workers + // -- a record or submit failure fails ahead of the hand-off, and + // the synchronous path does not hand over at all -- and is still + // holding its images, bitstream buffer and command buffers. Frame + // nodes come from a pool and run no destructor when the last + // reference to one drops, so releasing those resources has to be + // explicit or they stay pinned inside the pooled node. + VkVideoEncodeFrameInfo::ResetAndReleaseFrames(m_lastDeferredFrame); + assert(m_lastDeferredFrame == nullptr); } m_numDeferredFrames = 0; m_numDeferredRefFrames = 0; @@ -3308,6 +6749,28 @@ VkResult VkVideoEncoder::ProcessOrderedFrames(VkSharedBaseObj& frame, uint32_t frameIdx, uint32_t ofTotalFrames) { return AssembleBitstreamData(frame, frameIdx, ofTotalFrames); }} ); @@ -3321,7 +6784,7 @@ VkResult VkVideoEncoder::ProcessOrderedFrames(VkSharedBaseObjverbose) { const std::string& description = pair.first; - std::cout << "====== Total number of frames processed by " << description << ": " << processedFramesCount << " : " << result << std::endl; + VkEncOut() << "====== Total number of frames processed by " << description << ": " << processedFramesCount << " : " << result << std::endl; } if (result != VK_SUCCESS) { @@ -3338,6 +6801,70 @@ VkResult VkVideoEncoder::ProcessOrderedFrames(VkSharedBaseObj& frames, uint32_t numFrames) { + // This path assembles synchronously and never queues for the assembly + // workers, unlike ProcessOrderedFrames which chooses between the two on + // m_asyncAssemblyEnabled. With async assembly ON, every frame reaching + // here would be assembled off-thread of the capture queue and never + // published -- lost to the consumer, silently. + // + // REACHABILITY. The dispatch gate is enableOutOfOrderRecording, NOT the + // B-frame count: both call sites choose this function on + // `m_encoderConfig->enableOutOfOrderRecording`, which only the CLI flag + // --testOutOfOrderRecording sets (VkEncoderConfig.cpp) and which nothing + // in the ext layer ever writes. So the file-based CLI reaches this today, + // while the ext path -- the only one that registers a completion + // subscriber -- cannot. consecutiveBFrames does not appear in this + // decision at any point. + // + // Because it only ever assembles synchronously, this function must refuse + // BOTH shapes in which a frame would be encoded and never reported: async + // assembly on (immediately below) and a completion subscriber with async + // off (after it). Fail loudly rather than drop frames quietly -- the same + // treatment ProcessOrderedFrames gives its own unpublishable branch. + if (m_asyncAssemblyEnabled) { + VkEncPrintfErr("\nProcessOutOfOrderFrames reached with async assembly " + "enabled: this path has no queue-for-assembly branch and " + "would drop %u frame(s) without publishing them. Give it the " + "ProcessOrderedFrames async branch before enabling B-frames " + "on the ext path.\n", + numFrames); + return VK_ERROR_UNKNOWN; + } + + // The mirror of ProcessOrderedFrames' subscriber guard, and the reason it + // is needed HERE only became true when AssembleBitstreamData stopped + // publishing: before that, a synchronous out-of-order assembly with a + // subscriber attached would publish a record too EARLY -- ahead of the + // EnqueuePendingFrame() that creates its PendingFrame -- so + // DrainCapturesLocked() would discard it and count it in + // m_lateCaptures: wrong, but LOUD. With the publish correctly absent + // from the synchronous path, that same configuration encodes every + // frame and reports none of them SILENTLY, which is strictly harder to + // diagnose and is exactly the class ProcessOrderedFrames refuses. + // + // Unreachable from any CLI or ext configuration today, because the + // subscriber and enableOutOfOrderRecording come from surfaces that never + // overlap (see REACHABILITY above). It is a guard against a future + // enablement -- wiring out-of-order recording or B-frames onto the ext + // path. + // + // It is NOT untestable, and an earlier version of this comment claimed it + // was. m_asyncAssemblyEnabled defaults false, SetOnBitstreamCaptured is + // public, and this guard returns before anything touches a device -- so + // test/encoder-sync-assembly pins it directly, device-free, in + // CaseOutOfOrderSubscriberGuard. Neutralise the branch and the call falls + // through into the callback sequence and answers + // VK_ERROR_FEATURE_NOT_PRESENT instead of VK_ERROR_UNKNOWN, so the + // coverage discriminates rather than merely executing the line. + if (HasCompletionSubscriber()) { + VkEncPrintfErr("\nProcessOutOfOrderFrames reached with a completion " + "subscriber registered and async assembly off: this path " + "assembles synchronously and publishes no completion record, " + "so %u frame(s) would be encoded and never reported.\n", + numFrames); + return VK_ERROR_UNKNOWN; + } + const std::vector&, uint32_t, uint32_t)>>> callbacksSeq = { {true, [this](VkSharedBaseObj& frame, uint32_t frameIdx, uint32_t ofTotalFrames) { return StartOfVideoCodingEncodeOrder(frame, frameIdx, ofTotalFrames); }}, {true, [this](VkSharedBaseObj& frame, uint32_t frameIdx, uint32_t ofTotalFrames) { return ProcessDpb(frame, frameIdx, ofTotalFrames); }}, @@ -3373,7 +6900,7 @@ void VkVideoEncoder::DumpStateInfo(const char* stageName, uint32_t ident, VkSharedBaseObj& encodeFrameInfo, int32_t frameIdx, uint32_t ofTotalFrames) const { - std::cout << std::string(ident, ' ') << "===> " + VkEncOut() << std::string(ident, ' ') << "===> " << VkVideoCoreProfile::CodecToName(m_encoderConfig->codec) << ": " << stageName << " [" << frameIdx << " of " << ofTotalFrames << "]" << " type " << VkVideoGopStructure::GetFrameTypeName(encodeFrameInfo->gopPosition.pictureType) @@ -3387,7 +6914,10 @@ void VkVideoEncoder::DumpStateInfo(const char* stageName, uint32_t ident, bool VkVideoEncoder::WaitForThreadsToComplete() { - PushOrderedFrames(); + // The last deferred frame is encoded here, so its result belongs to this + // drain as much as the workers do -- on a synchronous session it is the + // only place a frame is processed at all. + const VkResult pushResult = PushOrderedFrames(); if (m_enableEncoderThreadQueue) { m_encoderThreadQueue.SetFlushAndExit(); @@ -3404,12 +6934,97 @@ bool VkVideoEncoder::WaitForThreadsToComplete() m_assemblyThreads.clear(); m_asyncAssemblyEnabled = false; if (m_assemblyErrorCount > 0) { - fprintf(stderr, "[AsyncAssembly] Completed with %u errors\n", + VkEncPrintfErr("[AsyncAssembly] Completed with %u errors\n", m_assemblyErrorCount.load()); } } - return true; + // The verdict on the whole session, and the only one a caller that + // watched none of the run can reach. Every arm is a frame that did not + // become bitstream. + return (pushResult == VK_SUCCESS) && + (m_assemblyErrorCount.load() == 0) && + (m_frameProcessingErrorCount.load() == 0); +} + +// The NON-TERMINAL drain. See the header for why this exists separately from +// WaitForThreadsToComplete(). +// +// The join above is exactly what makes the restart safe: on return from +// WaitForThreadsToComplete() every work item has been popped and has taken +// its ordering turn, no worker thread is alive, and the assembly queue is +// empty -- which is the precondition ClearFlushAndReuse() checks for. +// +// The ENCODER-QUEUE consumer thread is deliberately not restarted. It is +// joined by the same call and its queue carries the same sticky latch, but +// m_enableEncoderThreadQueue is false for every session in this tree, so +// there is no restart path here that could be tested. If it is ever enabled, +// this is where the second half belongs -- so this says so, loudly, rather +// than bringing half the pipeline back and leaving the caller to find out +// which half. +bool VkVideoEncoder::DrainAndRestartThreads() +{ + WaitForThreadsToComplete(); + + if (m_enableEncoderThreadQueue) { + VkEncPrintfErr("\nDrainAndRestartThreads with the encoder thread queue " + "enabled: its consumer thread was joined and is NOT restarted " + "by this path. Give it a restart before enabling " + "m_enableEncoderThreadQueue.\n"); + return false; + } + + return StartAssemblyThreads(); +} + +// Ext currency 3 (see the ext header): one library-owned timeline per +// session, GPU-signaled at queue flush points in SubmitVideoCodingCmds. +// Idempotent; runs on the session-serial thread before any submit. +VkResult VkVideoEncoder::CreateCompletionTimelineSemaphore() +{ + if (m_completionTimelineSemaphore != VK_NULL_HANDLE) { + return VK_SUCCESS; + } + + VkSemaphoreTypeCreateInfo timelineCreateInfo{VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO}; + timelineCreateInfo.semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + timelineCreateInfo.initialValue = 0; + + // Exportability is a physical-device property: ask, never assume. The + // query chain uses a SEPARATE type-info instance so the create chain's + // pNext is not aliased. When the answer is no -- or the dispatch entry + // is absent -- the semaphore is created plain and stays fully usable + // in-process; only the export arm is refused, by type, at the ext + // layer. OPAQUE_FD is the only export type this library ships (the + // Win32 semaphore-export arm is reserved), and querying a handle-type + // bit is portable Vulkan, so this file stays free of OS-specific code. + VkExportSemaphoreCreateInfo exportInfo{VK_STRUCTURE_TYPE_EXPORT_SEMAPHORE_CREATE_INFO}; + if (m_vkDevCtx->GetPhysicalDeviceExternalSemaphoreProperties != nullptr) { + VkSemaphoreTypeCreateInfo queryTypeInfo = timelineCreateInfo; + VkPhysicalDeviceExternalSemaphoreInfo extInfo{VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_SEMAPHORE_INFO}; + extInfo.pNext = &queryTypeInfo; + extInfo.handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT; + VkExternalSemaphoreProperties extProps{VK_STRUCTURE_TYPE_EXTERNAL_SEMAPHORE_PROPERTIES}; + m_vkDevCtx->GetPhysicalDeviceExternalSemaphoreProperties( + m_vkDevCtx->getPhysicalDevice(), &extInfo, &extProps); + if ((extProps.externalSemaphoreFeatures & + VK_EXTERNAL_SEMAPHORE_FEATURE_EXPORTABLE_BIT) != 0) { + exportInfo.handleTypes = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT; + timelineCreateInfo.pNext = &exportInfo; + m_completionSemaphoreExportable = true; + } + } + + VkSemaphoreCreateInfo createInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + createInfo.pNext = &timelineCreateInfo; + VkResult result = m_vkDevCtx->CreateSemaphore(*m_vkDevCtx, &createInfo, + NULL, + &m_completionTimelineSemaphore); + if (result != VK_SUCCESS) { + m_completionTimelineSemaphore = VK_NULL_HANDLE; + m_completionSemaphoreExportable = false; + } + return result; } int32_t VkVideoEncoder::DeinitEncoder() @@ -3432,6 +7047,19 @@ int32_t VkVideoEncoder::DeinitEncoder() m_hwLoadBalancingTimelineSemaphore = VK_NULL_HANDLE; } + // Completion timeline (ext currency 3): the same lifecycle discipline + // as the load-balancing timeline above -- WaitForThreadsToComplete() + // and the ENCODE queue wait-idle have already run, so no submitted + // batch can still reference it + // (VUID-vkDestroySemaphore-semaphore-01137). + if (m_completionTimelineSemaphore != VK_NULL_HANDLE) { + m_vkDevCtx->DestroySemaphore(*m_vkDevCtx, m_completionTimelineSemaphore, NULL); + m_completionTimelineSemaphore = VK_NULL_HANDLE; + } + m_completionSemaphoreExportable = false; + m_maxSubmittedCompletionValue = 0; + m_lastSignaledCompletionValue = 0; + m_linearInputImagePool = nullptr; m_inputImagePool = nullptr; m_dpbImagePool = nullptr; @@ -3446,24 +7074,30 @@ int32_t VkVideoEncoder::DeinitEncoder() const double psnrY = m_psnr->GetAveragePsnrY(); const double psnrU = m_psnr->GetAveragePsnrU(); const double psnrV = m_psnr->GetAveragePsnrV(); - printf("Average PSNR (dB): Y=%.2f", psnrY); + VkEncPrintfOut("Average PSNR (dB): Y=%.2f", psnrY); if (psnrU >= 0.0) { - printf(" U=%.2f", psnrU); + VkEncPrintfOut(" U=%.2f", psnrU); } if (psnrV >= 0.0) { - printf(" V=%.2f", psnrV); + VkEncPrintfOut(" V=%.2f", psnrV); } - printf("\n"); + VkEncPrintfOut("\n"); fflush(stdout); } else { - fprintf(stderr, "PSNR was requested (--psnr) but metrics are unavailable (initialization may have failed).\n"); + VkEncPrintfErr("PSNR was requested (--psnr) but metrics are unavailable (initialization may have failed).\n"); fflush(stderr); } } + if (m_contentProbe) { + m_contentProbe->Deinit(); + } + if (m_psnr) { m_psnr->Deinit(); } +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED m_inputComputeFilter = nullptr; +#endif m_inputCommandBufferPool = nullptr; m_encodeCommandBufferPool = nullptr; @@ -3479,12 +7113,14 @@ int32_t VkVideoEncoder::DeinitEncoder() void VkVideoEncoder::ConsumerThread() { - std::cout << "ConsumerThread is stating now.\n" << std::endl; + vkenc::OsSetCurrentThreadName("VkEncConsumer"); + + VkEncOut() << "ConsumerThread is stating now.\n" << std::endl; do { VkSharedBaseObj encodeFrameInfo; bool success = m_encoderThreadQueue.WaitAndPop(encodeFrameInfo); if (success) { // 5 seconds in nanoseconds - std::cout << "==>>>> Consumed: " << (uint32_t)encodeFrameInfo->gopPosition.inputOrder + VkEncOut() << "==>>>> Consumed: " << (uint32_t)encodeFrameInfo->gopPosition.inputOrder << ", Order: " << (uint32_t)encodeFrameInfo->gopPosition.encodeOrder << std::endl << std::flush; VkResult result; @@ -3494,25 +7130,23 @@ void VkVideoEncoder::ConsumerThread() // Testing only - don't use for production! result = ProcessOutOfOrderFrames(encodeFrameInfo, 0); } - if (m_asyncAssemblyEnabled) { - // Frames are owned by the async-assembly queue items now. - VkVideoEncodeFrameInfo::ReleaseChildrenFrames(encodeFrameInfo); - } else { - VkVideoEncodeFrameInfo::ResetAndReleaseFrames(encodeFrameInfo); - } + // Only the frames the assembly workers did not take are still + // here, and those are the encoder's to release. + VkVideoEncodeFrameInfo::ResetAndReleaseFrames(encodeFrameInfo); assert(encodeFrameInfo == nullptr); if (result != VK_SUCCESS) { - std::cout << "Error processing frames from the frame thread!" << std::endl; + VkEncOut() << "Error processing frames from the frame thread!" << std::endl; + m_frameProcessingErrorCount++; m_encoderThreadQueue.SetFlushAndExit(); } } else { bool shouldExit = m_encoderThreadQueue.ExitQueue(); - std::cout << "Thread should exit: " << (shouldExit ? "Yes" : "No") << std::endl; + VkEncOut() << "Thread should exit: " << (shouldExit ? "Yes" : "No") << std::endl; } } while (!m_encoderThreadQueue.ExitQueue()); - std::cout << "ConsumerThread is exiting now.\n" << std::endl; + VkEncOut() << "ConsumerThread is exiting now.\n" << std::endl; } size_t VkVideoEncoder::WriteDataToFile(const uint8_t* data, size_t size) @@ -3524,10 +7158,42 @@ size_t VkVideoEncoder::WriteDataToFile(const uint8_t* data, size_t size) if (m_crc.Enabled()) { m_crc.UpdateCrc(data, size); } + // Skip the fwrite when disableFileOutput is set. + // WriteBitstreamToFile captures the bytes into + // m_capturedBitstreams separately. Returning size signals + // success to the existing callers. + if (m_encoderConfig && m_encoderConfig->disableFileOutput) { + return size; + } size_t bytesWritten = fwrite(data, 1, size, m_encoderConfig->outputFileHandler.GetFileHandle()); return bytesWritten; } +// Pop the oldest completion record from the in-memory FIFO populated by +// PushCapturedBitstream whenever a drain-capable consumer exists (the +// encoded bytes are carried only in capture mode). Returns true if a +// record was popped, false if the queue is empty. Used by +// VulkanVideoEncoderExtImpl to route the records into the matching +// PendingFrame entries. +bool VkVideoEncoder::TryPopCapturedBitstream( + uint64_t* out_frame_id, std::vector* out_bytes, + bool* out_is_idr, uint32_t* out_picture_type, + VkResult* out_status) +{ + std::lock_guard lock(m_capturedBitstreamsMutex); + if (m_capturedBitstreams.empty()) { + return false; + } + CapturedBitstream& front = m_capturedBitstreams.front(); + if (out_frame_id) *out_frame_id = front.frameId; + if (out_bytes) *out_bytes = std::move(front.bytes); + if (out_is_idr) *out_is_idr = front.isIdr; + if (out_picture_type) *out_picture_type = front.pictureType; + if (out_status) *out_status = front.status; + m_capturedBitstreams.pop_front(); + return true; +} + size_t VkVideoEncoder::GetCrcValues(uint32_t* pCrcValues, size_t buffSize) const { return m_crc.GetCrcValues(pCrcValues, buffSize); diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.h index f7d8de33..d0bde820 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoder.h @@ -18,8 +18,10 @@ #define _VKVIDEOENCODER_VKVIDEOENCODER_H_ #include +#include #include #include +#include #include #include "VkCodecUtils/VkVideoRefCountBase.h" #include "VkVideoEncoderDef.h" @@ -31,6 +33,7 @@ #include "VkCodecUtils/VulkanBufferPool.h" #include "VkCodecUtils/VulkanCommandBufferPool.h" #include "VkCodecUtils/VulkanVideoReferenceCountedPool.h" +#include "VkVideoEncoder/VkVideoEncoderContentProbe.h" #include "VkCodecUtils/VkBufferResource.h" #include "VkCodecUtils/VulkanBistreamBufferImpl.h" #include "VkCodecUtils/VkThreadSafeQueue.h" @@ -55,10 +58,133 @@ class VkVideoEncoder : public VkVideoRefCountBase { public: + // ------------------------------------------------------------------ + // FILTER-DISPATCH OBSERVABLE + // + // What ACTUALLY ran, readable from outside the library. The session's + // input format says whether a filter was BUILT; it does not say what + // happened to any given frame. A session that has a filter can still + // route every frame down the staging copy -- StageInputFrame's + // `useComputeFilter` is a PER-FRAME predicate, not a session property -- + // and the copy arm is exactly the arm that hangs the GPU on a 3-plane + // source, so "did the filter run" cannot be inferred from the config. + // + // Counted at the RECORD SITE, immediately after + // VulkanFilter::RecordCommandBuffer() returns VK_SUCCESS, which is the + // same site the hardware proof of this filter counted at. A filter that + // failed to record does not count. + // + // BOTH ARMS ARE COUNTED SEPARATELY, on purpose. A single total plus a + // subtraction is how a live tier already read as dead once on this + // project (staging_frames_submitted_ is a superset of the cpu-dmabuf + // count). These two never overlap: they are the two sides of one `if`. + // + // The counters are cumulative for the life of the encoder object and + // are NOT reset by DeinitEncoder(), matching the diagnostic channel's + // documented "never resets" rule -- a teardown-time read is the one + // most likely to matter. + enum InputFilterKind { + INPUT_FILTER_NONE = 0, + INPUT_FILTER_YCBCR_COPY = 1, // VulkanFilterYuvCompute::YCBCRCOPY + INPUT_FILTER_RGBA_TO_YCBCR = 2, // VulkanFilterYuvCompute::RGBA2YCBCR + INPUT_FILTER_YCBCR_TO_RGBA = 3, // VulkanFilterYuvCompute::YCBCR2RGBA + }; + + // The filter OBJECT exists on this session. Deliberately distinct from + // the config flag: InitEncoder hard-fails when those two can diverge, + // and this is what proves that hard-fail is holding. + bool HasInputComputeFilter() const { +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + return (m_inputComputeFilter != nullptr); +#else + return false; +#endif + } + uint32_t GetInputFilterKind() const { + return m_inputFilterKind.load(std::memory_order_relaxed); + } + uint64_t GetInputFilterDispatchCount() const { + return m_inputFilterDispatchCount.load(std::memory_order_relaxed); + } + uint64_t GetStagedCopyCount() const { + return m_stagedCopyCount.load(std::memory_order_relaxed); + } + // The dma-buf import content probe, INJECTED by the ext layer -- see the + // note on m_contentProbe for why it is not owned here. Null on a session + // whose caller never opted in, and inert (allocating no device memory) + // until a capture is actually recorded. + const VkSharedBaseObj& GetContentProbe() const { + return m_contentProbe; + } + // Idempotent. Safe to call before or after InitEncoder: whichever runs + // second performs the Configure, so the ext layer does not have to know + // which order CreateVideoEncoder left things in. + void SetContentProbe(const VkSharedBaseObj& probe) { + m_contentProbe = probe; + ConfigureContentProbe(); + } + // The two sides of the staging ACQUIRE decision, reported through + // VkVideoEncoderInputResidencyInfo in vulkan_video_encoder_ext_internal.h. + // See it for why a validation count cannot substitute for these: both + // barrier programs are spec-clean and leave the image in the same layout. + uint64_t GetForeignAcquireCount() const { + return m_foreignAcquireCount.load(std::memory_order_relaxed); + } + uint64_t GetLocalAcquireCount() const { + return m_localAcquireCount.load(std::memory_order_relaxed); + } + + // THE STAGED-INPUT SUBMIT FAMILY, reported through + // VkVideoEncoderInputResidencyInfo in vulkan_video_encoder_ext_internal.h. + // Public wrappers over the two + // protected accessors declared further down (GetStagedInputSubmitType / + // GetStagedInputQueueFamilyIdx) so an out-of-library caller can read the + // fact those two exist to keep in agreement, WITHOUT re-deriving it. + // + // WHY THIS IS OBSERVABLE AT ALL, stated here because "it encoded" is not + // the question these answer. GetStagedInputSubmitType() returns COMPUTE + // whenever a preprocess filter OBJECT exists on the session -- a session + // property, not a per-frame one -- so a frame that dispatches NO filter + // and takes the plain-copy arm still has its acquire, its + // ReleaseImageToForeignQueue and its submit moved onto the compute family + // the moment the session gains a filter. Whether that move happened is + // invisible in a bitstream, invisible in a byte count, and invisible to + // the validation layer. It is exactly the fact that has to be read to + // know whether widening a session's declared input format moved the NV12 + // lane off the encode engine, which matters because a FOREIGN release + // recorded off the wrong engine can lose the device (see + // ReleaseImageToForeignQueue). + // + // Returned as the raw VK_QUEUE_* flag bit rather than the internal enum + // so no ext-layer translation table can drift from it. + uint32_t GetStagedInputSubmitTypeFlag() const { + return (uint32_t)GetStagedInputSubmitType(); + } + uint32_t GetStagedInputSubmitQueueFamilyIdx() const { + return GetStagedInputQueueFamilyIdx(); + } + using VulkanBitstreamBufferPool = VulkanVideoRefCountedPool; enum { MAX_IMAGE_REF_RESOURCES = 17 }; /* List of reference pictures 16 + 1 for current */ - enum { MAX_BITSTREAM_HEADER_BUFFER_SIZE = 256 }; + // 256 held VPS/SPS/PPS with room to spare and nothing else was ever + // appended. The HDR10 SEI NAL is up to 47 bytes (4 start code + 2 NAL + // header + 26 mastering display + 6 content light + 1 trailing, plus + // emulation-prevention bytes), and the AV1 metadata OBU pair is 36. The + // headroom is doubled rather than computed exactly because the driver + // owns the parameter-set half of this buffer and its size is not ours to + // predict; the appenders check the remaining capacity and FAIL rather + // than silently dropping a payload. + enum { MAX_BITSTREAM_HEADER_BUFFER_SIZE = 512 }; + + // Queue-family ownership of an external input image. Internal + // mirror of the public VkVideoEncoderInputResidency (the Ext API maps + // one onto the other); AUTO keeps the legacy layout-based inference. + enum ExternalInputResidency { + EXTERNAL_INPUT_RESIDENCY_AUTO = 0, + EXTERNAL_INPUT_RESIDENCY_LOCAL = 1, + EXTERNAL_INPUT_RESIDENCY_FOREIGN = 2, + }; struct BitstreamReadback { uint32_t bitstreamStartOffset{0}; @@ -246,6 +372,15 @@ class VkVideoEncoder : public VkVideoRefCountBase { VkSharedBaseObj qpMapCmdBuffer; /** Per-frame PSNR capture/recon data (only used when PSNR is enabled). */ VkVideoEncoderPsnr::FrameData psnrFrameData; + /** The dma-buf import content probe's readback for THIS frame, held + * from the record site in StageInputFrame to the post-fence score. + * Empty on all but the first frame of an armed registration. */ + VkVideoEncoderContentProbe::FrameCapture contentProbeCapture; + /** The registration this frame was submitted against, 0 for the + * unregistered (legacy direct-image) arm. Carried because the + * content probe's unit is the REGISTRATION, not the frame: it is + * what latches "this buffer has already been looked at". */ + uint64_t externalRegistrationId = 0; #ifdef NV_AQ_GPU_LIB_SUPPORTED std::shared_ptr aqProcessorSlot; #endif // NV_AQ_GPU_LIB_SUPPORTED @@ -261,6 +396,154 @@ class VkVideoEncoder : public VkVideoRefCountBase { bool isExternalInput{false}; VkImageLayout srcExternalImageLayout{VK_IMAGE_LAYOUT_UNDEFINED}; + // Did the CALLER state srcExternalImageLayout for THIS FRAME, or is + // it the registration-time default standing in? + // + // The public contract makes the distinction, and the encoder core + // cannot see it unaided: VkVideoEncoderFrameSubmitInfo:: + // currentLayout is documented as "the layout the producer left the + // image in. VK_IMAGE_LAYOUT_UNDEFINED means 'as declared at + // registration'", and the ext layer collapses both cases into one + // value before calling down here. An EXPLICIT per-frame declaration + // is the caller saying "I moved it", and it must beat the library's + // own record of where it last left the image + // (VulkanVideoImagePoolNode::GetStagedInputResidualLayout). + // + // false on the LEGACY SubmitExternalFrame lane and on the library's + // own file-input lane, neither of which has a registration default + // to stand in for -- and neither of which reads the residual either, + // so the value is inert there. + bool srcExternalLayoutIsExplicit{false}; + + // True when srcEncodeImageResource IS the caller's imported image + // (Path A, zero-copy) rather than a library pool image that a staging + // copy filled (Path B/C). Only Path A needs a queue-family acquire at + // encode time: on B/C the encode reads library-owned memory that the + // staging copy already acquired. + bool srcEncodeImageIsExternal{false}; + + // THE LAYOUT StageInputFrame LEFT THE ENCODE-SOURCE IMAGE IN, as a + // fact about a barrier this library recorded -- read once, by + // RecordVideoCodingCmd, immediately before it opens the video coding + // scope that will read that image. + // + // WHY THIS FIELD EXISTS AT ALL, when the barrier three lines away + // already does the work: no validation layer in this tree can see the + // rule it guards. VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 is + // checked against the per-command-buffer image-layout map of the + // command buffer the encode is recorded into, and the staging + // barriers are recorded into a DIFFERENT command buffer + // (m_inputCommandBufferPool, which is the compute filter itself on a + // filter-equipped session). The encode command buffer therefore has + // no map entry for this image and the check silently returns true. + // That is why CF-02a and CF-02b survived a run that was celebrated as + // "144 -> 0 validation messages": zero was the right count for a + // check that never executed. + // + // ONE WRITER, ONE READER, deliberately -- the same discipline as + // VulkanVideoImagePoolNode::SetStagedInputResidualLayout. The writer + // is the END of each StageInputFrame arm: the arm seeds a running + // variable with the layout the copy/filter needs and re-reads it + // from the hand-off TransitionImageLayout's RETURN. Delete the + // hand-off call and the record keeps saying TRANSFER_DST_OPTIMAL + // (copy arm) or GENERAL (filter arm), so the reader goes RED. That + // mutation is how this check was proven able to fail. + // + // AND THAT IS THE WHOLE OF WHAT IT PROVES. TransitionImageLayout + // has exactly ONE return -- an unconditional `return newLayout`, an + // echo of its own by-value argument, which the body never reassigns + // -- so this field records the layout the CALL ASKED FOR and never + // an observation of the barrier. Suppress the CmdPipelineBarrier2KHR + // inside TransitionImageLayout and this field still reads + // VIDEO_ENCODE_SRC_KHR. This reader gates the CALL SITE, not the + // barrier record. + // + // VK_IMAGE_LAYOUT_MAX_ENUM means "this frame was not staged", which + // is the Path-A / RESIDENCY_LOCAL zero-copy case and the AV1 + // show-existing pseudo-frames. Those declare their own layout through + // VkVideoEncoderExternalImageDescriptor::defaultLayout and are + // conformant by declaration, so the reader must not judge them. + VkImageLayout srcEncodeImageStagedLayout{VK_IMAGE_LAYOUT_MAX_ENUM}; + + // ROUTING INTENT for this frame: does its input need the preprocess + // compute filter to become encodable? Decided by whichever + // SetExternalInputFrame* entry point admitted the frame -- from the + // registration's resolved input path on the registered arm, from the + // frame's own format on the legacy arm -- and read by + // StageInputFrame. Per FRAME, not per session: with one session, + // registrations in the session's encode format take the direct path + // and registrations in its filter-input format take the filter, and + // the blanket "external input never filters" predicate this replaces + // could express neither. + // + // Never true for a file-input frame; those reach the filter through + // the session-level enablePreprocessComputeFilter as they always did. + bool externalInputViaFilter{false}; + + // WHAT ACTUALLY RAN, written by StageInputFrame after it has taken a + // branch and read by SubmitStagedInputFrame to pick the queue. These + // two facts must never be inferred from each other: the submit used + // to key off the filter OBJECT existing, so the moment a session had + // a filter, EVERY staged frame -- including the ones that took the + // plain copy, whose queue-family acquire names the transfer/encode + // family -- was submitted on the compute queue. A queue-family + // acquire executed on a queue outside its destination family is not a + // slowdown; it is a wedged queue, and it presents as a hang, so a + // timing-out test reads as flakiness rather than as this defect. + bool inputFilterRecorded{false}; + + // The CALLER's frame identifier (VkVideoEncodeInputFrame:: + // frameId), stored verbatim at SetExternalInputFrame() time and used + // to key the captured-bitstream FIFO (CapturedBitstream::frameId). + // Previously the capture was keyed by frameEncodeInputOrderNum, the + // library-internal encode-input counter. The two coincide only while + // every accepted SubmitExternalFrame() reaches EncodeFrameCommon() + // exactly once AND there is no B-frame reordering; any partially + // failed submission (EncodeFrameCommon increments the counter, then + // EncodeFrame errors and the caller drops the frame) desynchronizes + // the counter from the caller's ids PERMANENTLY, after which every + // capture is routed to the wrong (or no) pending frame. uint64_t(-1) + // == "not an external frame" (file-based path) -- the capture then + // falls back to frameEncodeInputOrderNum. + uint64_t externalFrameId{uint64_t(-1)}; + + // Caller-requested mid-stream IDR (VkVideoEncodeInputFrame:: + // forceIDR). Consumed by EncodeFrameCommon(), which passes it as + // GetPositionInGOP()'s "start a new IDR sequence" trigger -- the + // SAME path a periodic idrPeriod boundary takes, so the forced + // frame gets FRAME_TYPE_IDR, resets the GOP state machine (the GOP + // cadence restarts at this frame, matching VAAPI keyframe + // semantics), flushes the deferred queue and flows through the + // codec's normal IDR handling (idr_pic_id, DPB flush, header + // emission). + bool forceIdrOnInput{false}; + + // Caller-requested per-frame quantizer (VkVideoEncodeInputFrame:: + // qpOverride), -1 for "use the session's configured constQp". + // + // Consumed by EncodeFrameCommon(), which applies it AFTER copying + // the config's constQp -- that copy is unconditional, so a value + // written anywhere earlier would be silently overwritten. + // + // Units are the codec's own QP units, the same ones the config's + // constQp uses: 0..51 for H.264/H.265, qindex 0..255 for AV1. There + // is no second convention to learn. + // + // Honoured ONLY when the session's rate-control mode is DISABLED. + // In CBR/VBR the encoder owns QP and a per-frame override would + // fight its rate controller, so it is refused there rather than + // half-applied. + int32_t qpOverrideOnInput{-1}; + + // Caller-declared queue-family ownership of the external + // input image. Consumed by StageInputFrame() to decide whether the + // pre-copy barrier is a FOREIGN_EXT -> staging-family acquire (QFOT) + // or a plain (HOST-stage-legal) transition. AUTO falls back to the + // legacy "PREINITIALIZED means local" layout heuristic, which is + // wrong for REUSED local staging images (their true layout after the + // first staging copy is TRANSFER_SRC_OPTIMAL). + ExternalInputResidency externalInputResidency{EXTERNAL_INPUT_RESIDENCY_AUTO}; + // Wait semaphores: the staging/encode command buffer will wait on these // before accessing the external input image. Typically this is the // producer's graph timeline semaphore. @@ -276,6 +559,19 @@ class VkVideoEncoder : public VkVideoRefCountBase { void ClearExternalInputSync() { isExternalInput = false; + srcExternalLayoutIsExplicit = false; + srcEncodeImageIsExternal = false; + // Cleared with the rest of the per-frame input state, so a pool + // node recycled from a staged frame into a non-staged role (a + // Path-A frame, an AV1 show-existing pseudo-frame) cannot carry + // the previous tenant's record into a frame that never staged. + srcEncodeImageStagedLayout = VK_IMAGE_LAYOUT_MAX_ENUM; + externalInputViaFilter = false; + inputFilterRecorded = false; + externalFrameId = uint64_t(-1); + forceIdrOnInput = false; + qpOverrideOnInput = -1; + externalInputResidency = EXTERNAL_INPUT_RESIDENCY_AUTO; inputWaitSemaphores.clear(); inputWaitSemaphoreValues.clear(); inputWaitDstStageMasks.clear(); @@ -575,8 +871,6 @@ class VkVideoEncoder : public VkVideoRefCountBase { , m_minStreamBufferSize(2 * 1024 * 1024) , m_streamBufferSize(m_minStreamBufferSize) , m_rateControlInfo{ VK_STRUCTURE_TYPE_VIDEO_ENCODE_RATE_CONTROL_INFO_KHR } - , m_beginCodecRateControlInfo{} - , m_beginCodecRateControlInfoValid(false) , m_rateControlLayersInfo{{ VK_STRUCTURE_TYPE_VIDEO_ENCODE_RATE_CONTROL_LAYER_INFO_KHR }} , m_picIdxToDpb{} , m_gopState() @@ -615,6 +909,10 @@ class VkVideoEncoder : public VkVideoRefCountBase { #endif // VIDEO_DISPLAY_QUEUE_SUPPORT , m_hwLoadBalancingTimelineSemaphore() , m_currentVideoQueueIndx(-1) + , m_completionTimelineSemaphore() + , m_completionSemaphoreExportable(false) + , m_maxSubmittedCompletionValue(0) + , m_lastSignaledCompletionValue(0) , m_imageQpMapFormat() , m_qpMapTexelSize() , m_qpMapTiling() @@ -645,6 +943,40 @@ class VkVideoEncoder : public VkVideoRefCountBase { #endif // VIDEO_DISPLAY_QUEUE_SUPPORT virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes) = 0; + // True when SubmitExternalFrame can accept another input frame + // without the submit path blocking on the assembly queue's producer + // condition variable and without unbounded captured-bitstream + // growth. Submission is session-serial, so between this check and + // the enqueue the queue can only drain; the check is conservative. + bool CanAcceptNewInputFrame() const; + + // Upper bound on the number of items ONE EnqueueFrame() can push into + // m_assemblyQueue. The chain at rest is a run of non-reference frames + // (postFlushQueue fires on every reference frame, and both counters reset + // on every flush), so the bound is that run plus the reference frame whose + // insertion flushes it. Reads the generator's sub-GOP CYCLE, not the + // configured B count: the two can disagree, and the cycle is what places + // reference frames. + virtual size_t GetMaxAssemblyBurst() const { + if (!m_encoderConfig) { return 1u; } + const uint8_t cycle = m_encoderConfig->gopStructure.GetGopFrameCycle(); + return (size_t)((cycle > 0) ? cycle : 1u); + } + + // Create the completion timeline semaphore (ext currency 3; idempotent). + // Called by the ext layer at initialization, on the session-serial + // thread, before any submit. Exportability is a physical-device + // property, queried rather than assumed; when the answer is no, the + // semaphore is created plain and export is refused by type at the ext + // layer. + VkResult CreateCompletionTimelineSemaphore(); + VkSemaphore GetCompletionTimelineSemaphore() const { + return m_completionTimelineSemaphore; + } + bool IsCompletionSemaphoreExportable() const { + return m_completionSemaphoreExportable; + } + virtual bool GetAvailablePoolNode(VkSharedBaseObj& encodeFrameInfo) = 0; virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig) = 0; // Must be implemented by the codec @@ -666,6 +998,12 @@ class VkVideoEncoder : public VkVideoRefCountBase { // frameId: unique frame identifier (passed through to output) // pts: presentation timestamp // isLastFrame: set true for the final frame (triggers EOS) + // forceIdr: encode this frame as an IDR and restart the GOP + // sequence at it (mid-stream keyframe request) + // qpOverride: per-frame quantizer in the codec's own units, -1 for + // none. Honoured only when rate control is DISABLED. + // residency: queue-family ownership of the input image + // (AUTO = legacy layout heuristic) // waitSemaphore/signalSemaphore arrays: injected into SubmitStagedInputFrame VkResult SetExternalInputFrame( VkSharedBaseObj& encodeFrameInfo, @@ -678,6 +1016,9 @@ class VkVideoEncoder : public VkVideoRefCountBase { uint64_t frameId, uint64_t pts, bool isLastFrame, + bool forceIdr, + int32_t qpOverride, + ExternalInputResidency residency, uint32_t waitSemaphoreCount, const VkSemaphore* pWaitSemaphores, const uint64_t* pWaitSemaphoreValues, @@ -686,7 +1027,78 @@ class VkVideoEncoder : public VkVideoRefCountBase { const VkSemaphore* pSignalSemaphores, const uint64_t* pSignalSemaphoreValues); - // Helper: wrap an external VkImage as a VulkanVideoImagePoolNode + // Registered-path twin of SetExternalInputFrame: the wrap already + // happened, once, at registration (see the ext layer's + // BuildRegisteredViewLocked). Performs the same bookkeeping and routing + // but creates NO Vulkan object and NO wrapper: |node| is shared by + // every frame of its registration and is read-only on the encode path + // (GetPictureResourceInfo / GetImageView are the only consumers). + // |directlyEncodable| is the registration-time routing predicate + // (encodable format, non-LINEAR tiling, encode-capable usage), + // replacing the legacy arm's per-frame tiling check. + VkResult SetExternalInputFrameWithNode( + VkSharedBaseObj& encodeFrameInfo, + VkSharedBaseObj& node, + // The registration id this frame names, purely so the staging arm can + // tell the content probe WHICH BUFFER it is looking at. 0 means "not + // a registered submit", which the probe treats as never-armed. + uint64_t registrationId, + bool directlyEncodable, + // The registration's resolved rung of the adaptation ladder, for the + // frames |directlyEncodable| refuses: true routes the staged frame + // through the preprocess compute filter, false through the transfer + // copy. Decided once at registration by the layer that knows the + // descriptor's format and the session's, and passed down rather than + // re-derived here, so routing and the registration's reported input + // path cannot disagree. + bool routeViaFilter, + VkImageLayout srcImageCurrentLayout, + // True when |srcImageCurrentLayout| came from the FRAME's own + // currentLayout field and false when it is the registration's + // defaultLayout standing in. Only the ext layer can tell those + // apart -- it is the one that applies the UNDEFINED sentinel -- so + // it is passed down rather than re-derived. Decides whether the + // staged-input acquire may prefer the library's own residual + // record over the declaration. + bool srcLayoutIsExplicit, + uint64_t frameId, + uint64_t pts, + bool isLastFrame, + bool forceIdr, + int32_t qpOverride, + ExternalInputResidency residency, + uint32_t waitSemaphoreCount, + const VkSemaphore* pWaitSemaphores, + const uint64_t* pWaitSemaphoreValues, + const VkPipelineStageFlags2* pWaitDstStageMasks, + uint32_t signalSemaphoreCount, + const VkSemaphore* pSignalSemaphores, + const uint64_t* pSignalSemaphoreValues); + + // Shared bookkeeping of the two external-input entry points (the + // "replicate LoadNextFrame" and "store external sync info" sections), + // factored so the legacy and registered arms cannot drift. + void StampExternalFrameInfo( + VkSharedBaseObj& encodeFrameInfo, + VkImageLayout srcImageCurrentLayout, + uint64_t frameId, + uint64_t pts, + bool isLastFrame, + bool forceIdr, + int32_t qpOverride, + ExternalInputResidency residency, + uint32_t waitSemaphoreCount, + const VkSemaphore* pWaitSemaphores, + const uint64_t* pWaitSemaphoreValues, + const VkPipelineStageFlags2* pWaitDstStageMasks, + uint32_t signalSemaphoreCount, + const VkSemaphore* pSignalSemaphores, + const uint64_t* pSignalSemaphoreValues); + + // Helper: wrap an external VkImage as a VulkanVideoImagePoolNode. + // Legacy per-frame wrap for SubmitExternalFrame only; registered + // submissions carry a node built once at registration. Slated for + // removal with the M5 consumer migration. VkResult WrapExternalImage( VkImage image, VkDeviceMemory memory, VkFormat format, uint32_t width, uint32_t height, @@ -723,6 +1135,23 @@ class VkVideoEncoder : public VkVideoRefCountBase { VkResult RecordVideoCodingCmds(VkSharedBaseObj& encodeFrameInfo, uint32_t numFrames); + // Capacity of the DIRECT (zero-copy) submit's wait and signal arrays. + // + // SubmitVideoCodingCmds assembles that submit into fixed stack arrays of + // this many entries, and a frame whose assembled list would not fit is + // REFUSED with VK_ERROR_TOO_MANY_OBJECTS rather than submitted with an + // entry dropped: a dropped wait lets the encode read the input image + // while the producer named by that wait is still writing it, and a + // dropped signal is a semaphore nobody ever signals. The staged path + // assembles into a growable vector and carries no such bound. + // + // One number, named once, because the entry points that refuse an + // over-capacity frame early -- while a status can still be returned to + // the caller -- have to refuse at exactly the count the arrays hold. A + // second literal that drifted from this one would reopen the gap it + // exists to close. + static constexpr uint32_t kDirectSubmitSemaphoreCapacity = 8; + virtual VkResult SubmitVideoCodingCmds(VkSharedBaseObj& encodeFrameInfo, uint32_t frameIdx, uint32_t ofTotalFrames); @@ -730,7 +1159,38 @@ class VkVideoEncoder : public VkVideoRefCountBase { uint32_t frameIdx, uint32_t ofTotalFrames); + // Returns BYTES ACCOUNTED FOR, which is not the same as bytes fwritten. + // + // When disableFileOutput is set the fwrite is skipped and this returns + // `size` anyway. That is deliberate and load-bearing, not an oversight: + // both loop-driving callers -- VkVideoEncoder::WriteBitstreamToFileOutput() + // and VkVideoEncoderAV1::FlushBatchedTemporalUnit() -- drive a partial-write + // loop + // while (written < total) { n = WriteDataToFile(...); if (!n) fail; } + // and read 0 as a hard failure. Returning 0 under the suppression flag + // therefore turns the CLI's own documented discard mode + // (--disableFileOutput, "suppress ALL bitstream file output", see + // VkEncoderConfig.cpp) into an error: every frame would report "Error + // writing VCL data" and AssembleBitstreamData would return + // VK_ERROR_OUT_OF_HOST_MEMORY, while the run still exited 0. + // + // So 0 means "the write was attempted and came up short" and nothing + // else. Callers must not read a non-zero return as evidence that bytes + // reached a file; in capture mode they reach m_capturedBitstreams + // instead, and under `--syncAssembly --disableFileOutput` they reach + // neither -- that combination is encode-and-discard by construction, + // which is what those two flags jointly ask for. size_t WriteDataToFile(const uint8_t* data, size_t size); + +private: + // File-output arm of WriteBitstreamToFile: writes the non-VCL header, then + // the coded payload described by readback, which ReadbackBitstreamData() + // has already fetched from the feedback query pool. + // Private and non-virtual: it must never grow a second completion publish. + VkResult WriteBitstreamToFileOutput(VkSharedBaseObj& encodeFrameInfo, + BitstreamReadback& readback); + +public: virtual VkResult ReadbackBitstreamData(VkSharedBaseObj& encodeFrameInfo, BitstreamReadback& readback); @@ -792,7 +1252,135 @@ class VkVideoEncoder : public VkVideoRefCountBase { uint32_t numPlanes, VkFormat format); + // Drain the pipeline and join every worker. Returns true when the + // session completed everything it was given, and FALSE when it did not -- + // a frame the encoder thread could not process, a bitstream the assembly + // workers could not read back or write, or a deferred frame that could + // not be pushed. The threads report each failure as it happens, but a + // process that only ever sees the end of the run has no other place to + // learn that one occurred, and a bitstream is not evidence: a session + // that failed on its first frame leaves a file of zero bytes behind. bool WaitForThreadsToComplete(); + // Queue a mid-stream rate-control update. Thread-safe producer; + // applied on the encoder thread by the fold at the TOP of + // EncodeFrameCommon, so every field this call carries is in force for + // the next frame the encoder processes -- the frame the caller placed + // this call in front of, not the one after it. + // + // THE CONSTANT-QP TRIPLE IS CARRIED HERE TOO, and it is the only rate + // lever a DISABLED-mode session has: on such a session the per-layer + // bitrates above are dropped outright, because that mode commands + // layerCount 0. A NEGATIVE member means the caller did not name that + // quantizer and it is left alone -- never 0, which is a valid + // (lossless) QP. The values land on m_encoderConfig->constQp, from + // where EncodeFrameCommon takes a PER-FRAME COPY. That copy is + // unconditional, and it is the first thing the function does, so the + // fold has to precede it in the same function -- which is why + // EncodeFrameCommon folds before it copies. A fold that ran only + // later, at the control-command point, would leave this triple + // landing one frame after the six fields around it. + // + // THE QP CLAMPS TRAVEL DIFFERENTLY FROM THE CONSTANT-QP TRIPLE, and + // that difference is the whole reason they need their own arguments. + // constQp reaches a frame as that per-frame copy, so writing the + // config is the whole of the update -- provided the write lands + // before the copy is taken. minQp/maxQp are read once, by the + // codec-specific EncoderConfig::GetRateControlParameters, into the + // codec rate-control layer struct that CodecHandleRateControlCmd + // chains onto the next ENCODE_RATE_CONTROL command -- so writing the + // config alone changes nothing until that fill is RE-INVOKED. The + // re-invocation is RefreshCodecRateControlParameters() below, and it + // is why these two arguments exist rather than a caller simply + // editing the config. + // + // NEGATIVE means the update carries no clamp. Zero and above are + // carried LITERALLY, and zero means "no clamp", exactly as it does at + // InitializeExt. Deliberate: a clamp that could be set but never + // cleared would be a different contract from the one the config field + // already documents. + // + // REFUSES the update (VK_ERROR_INITIALIZATION_FAILED) when a carried + // clamp falls outside the device QP window this session recorded at + // codec-init. With useMinQp/useMaxQp raised the spec requires the + // value to lie inside that window, and the init path already refuses + // one that does not; an unchecked mid-stream write would be a hole + // straight past that check. + VkResult RequestRateControlUpdate(uint64_t averageBitrate, + uint64_t maxBitrate, + uint32_t frameRateNumerator, + uint32_t frameRateDenominator, + int32_t constQpIntra = -1, + int32_t constQpInterP = -1, + int32_t constQpInterB = -1, + int32_t minQp = -1, + int32_t maxQp = -1); + + // Test observation seam. Folds any armed update, then reports the + // session CONSTANT-QP defaults -- the exact members EncodeFrameCommon + // copies into the next frame it processes. Reporting the value that + // frame would be encoded with is the point: a mid-stream update that + // only returned VK_SUCCESS would be the defect this seam exists to + // catch. VK_ERROR_NOT_PERMITTED_KHR when the session has no config. + // + // WHAT IT STRUCTURALLY CANNOT SEE is WHEN the fold happens relative to + // that copy, because it forces a fold itself and then reads the + // config rather than a frame. An ordering regression -- the fold + // moving back after the copy -- leaves this seam reading the new + // value and every assertion built on it green. Only a device leg that + // reads the quantizer out of an encoded picture can witness that. + VkResult ApplyAndGetConstQpForTest(int32_t* pQpIntra, + int32_t* pQpInterP, + int32_t* pQpInterB); + + // What a mid-stream rate-control update actually PUT IN FORCE, as + // opposed to what the caller passed. Three layers of it, because a + // change is only real if it survives all three: + // + // * the LIVE rate-control layer, which HandleCtrlCmd copies verbatim + // into the next control command. This is where a coerced maxBitrate + // and a frame rate left alone become visible as the numbers the + // session is running on -- which is what a caller-visible record of + // the configuration has to agree with, and did not; + // * the session config, where the QP clamp request lands; and + // * the RESOLVED codec rate-control layer struct, the end of the + // chain this class owns and the struct CodecHandleRateControlCmd + // chains onto the command. + // + // codecRefreshCount counts re-invocations of the codec fill, so a test + // can tell a real refresh from one that happened to recompute the same + // numbers, and can assert that a bitrate-only update does NOT cause + // one. + // + // WHAT THIS CANNOT SHOW is whether the driver then honours any of it. + // That needs a device and a decoded comparison. + struct RateControlObservation { + uint64_t layerAverageBitrate; + uint64_t layerMaxBitrate; + uint32_t layerFrameRateNumerator; + uint32_t layerFrameRateDenominator; + int32_t constQpIntra; + int32_t constQpInterP; + int32_t constQpInterB; + int32_t configMinQp; + int32_t configMaxQp; + uint32_t configMinQpSet; + uint32_t configMaxQpSet; + uint32_t resolvedUseMinQp; + uint32_t resolvedUseMaxQp; + int32_t resolvedMinQpI; + int32_t resolvedMaxQpI; + uint32_t codecRefreshCount; + }; + VkResult ApplyAndGetRateControlForTest(RateControlObservation* pOut); + + // Declare the device QP window directly. Test-only: a device-free + // session never runs InitEncoderCodec, so it has no other way to + // stand up the window the clamp check reads, and without one that + // check would be untestable rather than merely inert. + void SetDeviceQpWindowForTest(int32_t minQp, int32_t maxQp) { + m_deviceQpWindowMin = minQp; + m_deviceQpWindowMax = maxQp; + } protected: @@ -801,9 +1389,136 @@ class VkVideoEncoder : public VkVideoRefCountBase { VkDeviceSize GetBitstreamBuffer(VkSharedBaseObj& bitstreamBuffer); + // Local patch; not in upstream vk_video_samples. Optional queue-family- + // ownership transfer for imported (VK_QUEUE_FAMILY_FOREIGN_EXT) + // external images. VkImageLayout TransitionImageLayout(VkCommandBuffer cmdBuf, VkSharedBaseObj& imageView, - VkImageLayout oldLayout, VkImageLayout newLayout); + VkImageLayout oldLayout, VkImageLayout newLayout, + uint32_t srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED, + uint32_t dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED); + + // Local patch; not in upstream vk_video_samples. + // + // The RELEASE half of the FOREIGN_EXT ownership transfers the acquire + // sites perform. Deliberately NOT TransitionImageLayout: + // + // (a) That function picks its four barrier masks from an + // (oldLayout,newLayout) if/else chain, and an unhandled pair still + // gets a barrier -- now a deliberate ALL_COMMANDS/MEMORY_* fallback + // with a loud diagnostic, rather than the struct defaults. + // + // THIS SUB-ARGUMENT USED TO REST ON A BUILD DIVERGENCE THAT NO + // LONGER EXISTS, and is corrected here rather than quietly left to + // rot. The chain's final else was + // `#ifdef __cpp_exceptions throw ... #endif`, which meant the + // standalone CMake build (exceptions ON) terminated on an unhandled + // pair while Chromium's -fno-exceptions build silently kept the + // defaults -- so a ctest could not reproduce the shipping failure + // mode. Both halves were replaced by one total fallback that behaves + // identically in both builds, so that particular reason to avoid + // this table is gone. + // + // What survives, and is why this helper still exists: the fallback + // is correct-but-conservative for an ACQUIRE and still not right for + // a RELEASE, for reason (b) below. + // (b) For a RELEASE the spec USES srcStageMask/srcAccessMask and + // IGNORES the dst pair -- the exact inverse of an acquire. So the + // defaults that let an unhandled ACQUIRE survive by luck turn an + // unhandled RELEASE into an ownership transfer with an EMPTY first + // synchronisation scope: ordered against nothing, and invisible to + // the validation layers, since VK_PIPELINE_STAGE_2_NONE is + // trivially queue-valid. + // (c) The table dispatches on the layout pair alone and ignores the + // queue-family arguments, so one table cannot serve both + // directions regardless. + // + // The caller therefore states the src masks explicitly. + // |oldLayout| must be the layout our last use actually left the image in + // (VUID-VkImageMemoryBarrier2-oldLayout-01197). |newLayout| is the layout + // the FOREIGN consumer will find it in; it must not be UNDEFINED or + // PREINITIALIZED (VUID-...-newLayout-01198). + void ReleaseImageToForeignQueue(VkCommandBuffer cmdBuf, + VkSharedBaseObj& imageView, + VkImageLayout oldLayout, + VkImageLayout newLayout, + uint32_t srcQueueFamilyIndex, + VkPipelineStageFlags2KHR srcStageMask, + VkAccessFlags2KHR srcAccessMask); + + // Local patch; not in upstream vk_video_samples. + // + // The LOCAL-residency counterpart of ReleaseImageToForeignQueue: it hands + // a staged input image back in a layout the NEXT frame's acquire can name + // truthfully, and RETURNS that layout so the caller can record it. + // + // WHY THIS EXISTS. A registration that is reused across frames is + // declared once, and StageInputFrame records that declaration as its + // acquire oldLayout. Nothing restored the image afterwards unless it was + // a FOREIGN import, so for a LOCAL registration the declaration was true + // on frame 1 only: the copy arm leaves the image in TRANSFER_SRC_OPTIMAL + // and the filter arm leaves it in GENERAL, and every later frame asserted + // a layout the image was not in + // (VUID-VkImageMemoryBarrier2-oldLayout-01197). + // + // NOT TransitionImageLayout: the destination here comes from the CALLER, + // so routing it through a table keyed on the layout pair would turn an + // unusual but legal declaration into the table's terminal throw. This + // helper is total. + // + // THE UNRESTORABLE DECLARATION IS SUBSTITUTED, NOT SKIPPED, and that is a + // reversal of this helper's original contract. UNDEFINED and + // PREINITIALIZED are not legal barrier DESTINATIONS + // (VUID-VkImageMemoryBarrier2-newLayout-01198), and PREINITIALIZED is + // unrestorable BY CONSTRUCTION -- it means "never yet in any other layout + // since creation", which can be true at most once in an image's life. So + // this helper hands such a registration back in VK_IMAGE_LAYOUT_GENERAL: + // the only other layout in which host access to a LINEAR image is + // defined, and a legal barrier destination. + // + // The original contract recorded NOTHING in that case, on the ground that + // substituting GENERAL would "replace one false declaration with a + // different false declaration". That objection was correct while the + // library kept USING the caller's declaration as the next acquire's + // oldLayout, and it is dissolved now that it does not: StageInputFrame + // stores this function's RETURN VALUE on the registration's pool node and + // names it -- not the declaration -- on the next acquire. The library no + // longer trusts the declaration; it makes its own statement true. Nothing + // false is left over for the substitution to add to. + // + // WHAT THE CALLER LOSES, stated plainly: a caller that declared + // PREINITIALIZED and reused the registration finds its image in GENERAL + // from the second frame on rather than in a layout it named. It could not + // have been left in PREINITIALIZED by any legal barrier, so there is no + // behaviour it could have relied on -- and GENERAL is strictly better for + // the only thing such a caller does between frames, which is host-write a + // LINEAR image through a persistent mapping, which + // TRANSFER_SRC_OPTIMAL does not permit at all. + // + // NO LAYOUT TRACKING THROUGH m_currentImageLayout. |residualLayout| is + // still a LITERAL supplied by whichever arm recorded the work -- the same + // literal that arm named as its acquire newLayout -- never a stored + // field. The node field this function's return value feeds + // (m_stagedInputResidualLayout) is a DIFFERENT field from + // m_currentImageLayout, with one writer and one reader; reading + // m_currentImageLayout, which three unrelated producers write, is what + // forced the previous attempt at this fix to be reverted. + // + // ONE CASE STILL RECORDS NOTHING: residualLayout already equals the + // (possibly substituted) target, so our arm left the image exactly where + // the next acquire will name it. This is what keeps an (X -> X) pair, + // which the layout table has no arm for, from ever being constructed, + // without consulting any state. The return value is the target either + // way, so the caller's record is correct in both. + // + // Returns the layout the image is left in. + VkImageLayout RestoreStagedInputLayout(VkCommandBuffer cmdBuf, + VkSharedBaseObj& imageView, + VkImageLayout residualLayout, + VkImageLayout declaredLayout, + VkPipelineStageFlags2KHR srcStageMask, + VkAccessFlags2KHR srcAccessMask); + VkResult CopyLinearToOptimalImage(VkCommandBuffer& commandBuffer, VkSharedBaseObj& srcImageView, @@ -842,22 +1557,93 @@ class VkVideoEncoder : public VkVideoRefCountBase { int32_t DeinitEncoder(); - bool EnqueueFrame(VkSharedBaseObj& encodeFrameInfo, - bool isIdrFrame, bool isReferenceFrame) { + // Pop the next completion record (FIFO) from + // m_capturedBitstreams. Returns true if a record was popped. Used by + // VulkanVideoEncoderExtImpl to drain the in-memory completion queue + // in both output modes (bytes are carried only in capture mode). + // *out_status carries the per-frame result -- VK_SUCCESS for a + // normal capture; the readback failure code (e.g. VK_INCOMPLETE for a + // non-COMPLETE query status such as INSUFFICIENT_BITSTREAM_BUFFER_RANGE) + // for a frame whose assembly failed, with empty bytes. + bool TryPopCapturedBitstream(uint64_t* out_frame_id, + std::vector* out_bytes, + bool* out_is_idr, + uint32_t* out_picture_type, + VkResult* out_status); + + // Quiesce the pipeline the way WaitForThreadsToComplete() does -- push + // the deferred GOP tail, join every worker, so that everything submitted + // so far is encoded AND has published its completion record -- and then + // bring the assembly workers back up so the session can keep doing both. + // + // This is what a NON-TERMINAL drain has to be. WaitForThreadsToComplete() + // on its own is terminal for the COMPLETION SURFACE, not for the encoder: + // it clears m_asyncAssemblyEnabled, ProcessOrderedFrames then falls back + // to the synchronous AssembleBitstreamData, and that path publishes no + // CapturedBitstream -- so every later frame encodes correctly and is + // never reported. Callers that really are tearing down (Flush, + // Deinitialize) keep calling WaitForThreadsToComplete() directly. + // + // Returns false if the workers could not be restarted; the drain itself + // has still happened. + bool DrainAndRestartThreads(); + + // True when someone has registered for the completion edge (the Ext + // layer does, at InitializeExt). Used to tell "nobody is listening, so + // publishing is pointless" apart from "somebody is listening and a + // dropped record is a contract violation". + bool HasCompletionSubscriber() const { + std::lock_guard lock(m_capturedBitstreamsMutex); + return (m_onBitstreamCaptured != nullptr); + } + + // Record this frame into the deferred-GOP queue and flush that queue + // around it. Returns VK_SUCCESS when every flush this call made + // succeeded, and otherwise the first failure one reported. + // + // A flush is where a frame is recorded and submitted, so a frame whose + // commands could not be recorded or whose submit was refused fails HERE, + // on the caller thread. This return is the route by which that failure + // reaches the caller's own status; the drain at the end of the session + // reports it too, but only once every remaining frame has been given + // away. + VkResult EnqueueFrame(VkSharedBaseObj& encodeFrameInfo, + bool isIdrFrame, bool isReferenceFrame) { + + VkResult result = VK_SUCCESS; const bool preFlushQueue = isIdrFrame; if (preFlushQueue) { - PushOrderedFrames(); + result = PushOrderedFrames(); } InsertOrdered(encodeFrameInfo, isReferenceFrame); + // Local patch; not in upstream vk_video_samples. + // Flush per-frame when there is no B-frame reordering. The deferred-GOP + // queue is otherwise drained only on lastFrame / IDR-preflush / a full + // reference window. A Chromium stream never signals lastFrame, and with + // consecutiveBFrames=0 no reordering is needed, so without this trigger + // every frame piles into m_lastDeferredFrame, the encode-image pool + // exhausts (~m_holdRefFramesInQueue), RecordVideoCodingCmd never runs, and + // the bitstream is empty. With no reordering, input order == encode order, + // so recording each frame immediately is correct and low-latency. + const bool noReorderingNeeded = + (m_encoderConfig->gopStructure.GetConsecutiveBFrameCount() == 0); const bool postFlushQueue = (encodeFrameInfo->lastFrame || + noReorderingNeeded || (isReferenceFrame && (m_numDeferredRefFrames == m_holdRefFramesInQueue))); if (postFlushQueue) { - PushOrderedFrames(); + // Both flushes are made whatever the first reported: this frame + // is in the queue by now, and the queue must not be left holding + // it because an earlier frame failed. The FIRST failure is the + // one returned -- it is the one with a cause behind it. + const VkResult postResult = PushOrderedFrames(); + if (result == VK_SUCCESS) { + result = postResult; + } } - return true; + return result; } void ConsumerThread(); @@ -941,24 +1727,114 @@ class VkVideoEncoder : public VkVideoRefCountBase { VkVideoEncodeQualityLevelInfoKHR m_qualityLevelInfo; VkVideoEncodeRateControlInfoKHR m_rateControlInfo; VkVideoEncodeRateControlInfoKHR m_beginRateControlInfo; - // The codec-specific rate-control info that goes with m_beginRateControlInfo. - // - // vkCmdBeginVideoCodingKHR's rate-control chain must MATCH the state configured on - // the session (VUID-vkCmdBeginVideoCodingKHR-pBeginInfo-08254). That state is set by - // CmdControlVideoCodingKHR with the codec-specific struct chained on, so a chain - // that carries only the base struct disagrees with the session on every member the - // codec-specific struct sets -- gopFrameCount among them. Both halves are cached and - // re-linked, so every frame that reuses the cached state matches the session. - // Cache it codec-agnostically -- the base class does not know which codec it is. - union CodecRateControlInfo { - VkBaseInStructure base; - VkVideoEncodeH264RateControlInfoKHR h264; - VkVideoEncodeH265RateControlInfoKHR h265; - VkVideoEncodeAV1RateControlInfoKHR av1; - }; - CodecRateControlInfo m_beginCodecRateControlInfo; - bool m_beginCodecRateControlInfoValid; VkVideoEncodeRateControlLayerInfoKHR m_rateControlLayersInfo[1]; + // Value snapshot backing m_beginRateControlInfo.pLayers: the last + // COMMANDED layer state, immune to pending-update mutation of + // m_rateControlLayersInfo between control commands. + VkVideoEncodeRateControlLayerInfoKHR m_beginRateControlLayersInfo[1] = + {{ VK_STRUCTURE_TYPE_VIDEO_ENCODE_RATE_CONTROL_LAYER_INFO_KHR }}; + // Codec-specific halves of the cached begin-coding rate-control state. + // The session state a control command establishes includes the codec RC + // struct and per-layer codec structs, and + // VUID-vkCmdBeginVideoCodingKHR-pBeginInfo-08254 requires the begin-info + // chain to match that state in FULL -- so they are snapshotted together + // with the base struct and layer values. + VkVideoEncodeH264RateControlInfoKHR m_beginRateControlInfoH264{}; + VkVideoEncodeH265RateControlInfoKHR m_beginRateControlInfoH265{}; + VkVideoEncodeAV1RateControlInfoKHR m_beginRateControlInfoAV1{}; + VkVideoEncodeH264RateControlLayerInfoKHR m_beginRateControlLayersInfoH264[1] = {}; + VkVideoEncodeH265RateControlLayerInfoKHR m_beginRateControlLayersInfoH265[1] = {}; + VkVideoEncodeAV1RateControlLayerInfoKHR m_beginRateControlLayersInfoAV1[1] = {}; + // Pending mid-stream rate-control update. Produced by any + // thread via RequestRateControlUpdate() (the ext Reconfigure entry); + // applied ON THE ENCODER THREAD by ApplyPendingRateControlUpdate(), + // which runs at the TOP of EncodeFrameCommon, before that frame takes + // its copy of the constant-QP triple and before HandleCtrlCmd builds + // the frame record -- so the refreshed values ride THAT frame's + // VK_VIDEO_CODING_CONTROL_ENCODE_RATE_CONTROL command, not the one + // after it. HandleCtrlCmd keeps a fold of its own, which is a no-op + // whenever EncodeFrameCommon has already consumed the arm. Both + // fields are guarded by m_pendingRateControlMutex. + struct PendingRateControlUpdate { + uint64_t averageBitrate; + uint64_t maxBitrate; + uint32_t frameRateNumerator; + uint32_t frameRateDenominator; + // The constant-QP defaults this update carries. NEGATIVE means + // the update does not name that quantizer, and it MUST be the + // default: 0 is a valid lossless QP, so a zero-initialized + // pending update would silently rewrite the session to lossless + // the first time any rate change was armed. + int32_t constQpIntra = -1; + int32_t constQpInterP = -1; + int32_t constQpInterB = -1; + // The QP clamps this update carries. NEGATIVE means it carries + // none. Zero is NOT that: at InitializeExt zero already means "no + // clamp reaches the driver", so zero has to keep meaning the same + // thing here, or a clamp would become settable and not clearable. + int32_t minQp = -1; + int32_t maxQp = -1; + }; + void ApplyPendingRateControlUpdate(); + + // Re-run the codec-specific EncoderConfig::GetRateControlParameters + // fill against the CURRENT config, refreshing the codec rate-control + // structs that CodecHandleRateControlCmd chains onto the next + // ENCODE_RATE_CONTROL command. Runs ON THE ENCODER THREAD, from + // ApplyPendingRateControlUpdate, and only when the update actually + // changed something the fill reads. + // + // THE BASE IS A NO-OP ON PURPOSE, and AV1 keeps it. This fill is the + // only route a QP clamp has to the driver, and AV1 rate control is + // quantizer-index based: EncoderConfigAV1::GetRateControlParameters + // reads its own minQIndex/maxQIndex, which are derived from the + // DEVICE capability limits and never from the QP-unit clamps. There + // is nothing on AV1 for a refresh to carry, which is why the ext + // refuses a QP-unit clamp on an AV1 session rather than calling this + // and changing nothing. + // + // NOT A GENERAL REFRESH-EVERYTHING HOOK. The fill also recomputes the + // GOP counts and the virtual buffer size out of the config. Those are + // only safe to recompute because the config fields behind them are + // held immutable across Reconfigure, so the recomputation lands on + // the values already in force. Making one of them mutable means + // revisiting this, not just adding a call site. + virtual void RefreshCodecRateControlParameters() {} + + // The QP clamp as the codec rate-control layer struct now resolves + // it. Test observation only; the base reports "no clamp", which is + // the truth for an arm that carries none. + virtual void GetResolvedQpClampForTest(uint32_t* pUseMinQp, + int32_t* pMinQpI, + uint32_t* pUseMaxQp, + int32_t* pMaxQpI) const + { + *pUseMinQp = 0; + *pMinQpI = 0; + *pUseMaxQp = 0; + *pMaxQpI = 0; + } + + // The device QP window the codec arm records at codec-init, so the + // caller-thread clamp check can read it without touching + // m_encoderConfig -- whose lifetime is only guaranteed on the encoder + // thread, and dereferencing which off that thread is the hazard the + // note on m_computeFilterActive in the ext exists to forbid. Written + // once, before any Reconfigure can run. + // + // A MAXIMUM OF ZERO MEANS NOT ESTABLISHED, and the check is skipped. + // No H.26x device reports a maximum QP of zero: the syntactic range + // is 0..51 and a device admitting only QP 0 could encode nothing but + // lossless. So zero is unambiguous, and it is exactly what an arm + // that never recorded a window leaves behind. + int32_t m_deviceQpWindowMin = 0; + int32_t m_deviceQpWindowMax = 0; + // Counts RefreshCodecRateControlParameters() invocations. Read by the + // test observation seam only. + uint32_t m_codecRateControlRefreshCount = 0; + std::mutex m_pendingRateControlMutex; + bool m_pendingRateControlArmed = false; + PendingRateControlUpdate m_pendingRateControlUpdate = {}; int8_t m_picIdxToDpb[17]; // MAX_DPB_SLOTS + 1 VkVideoGopStructure::GopState m_gopState; uint32_t m_dpbSlotsMask; @@ -1000,8 +1876,60 @@ class VkVideoEncoder : public VkVideoRefCountBase { #ifdef NV_AQ_GPU_LIB_SUPPORTED VkSharedBaseObj m_inputSubsampledImagePool; #endif // NV_AQ_GPU_LIB_SUPPORTED +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED VkSharedBaseObj m_inputComputeFilter; +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // Filter-dispatch observable; see the accessors at the top of the class. + // Declared OUTSIDE the compute-filter ifdef so the member layout of this + // class does not depend on that macro -- the counters simply stay 0 in a + // build with no filter, which is the honest answer there. + // + // Atomic because the recording site runs on whatever thread drives + // EncodeFrame while GetCompletionInfo() is documented threading class + // (c) -- callable concurrently. Relaxed ordering: these are counters + // read for reporting, and they order nothing. + std::atomic m_inputFilterKind{0}; + std::atomic m_inputFilterDispatchCount{0}; + std::atomic m_stagedCopyCount{0}; + // Atomic and relaxed for the same reason the three counters above are: + // written on whatever thread drives EncodeFrame, read by + // GetCompletionInfo() which is documented threading class (c). + std::atomic m_foreignAcquireCount{0}; + std::atomic m_localAcquireCount{0}; + // See GetContentProbe(). NOT owned here: the ext layer creates it and + // injects it, because arming and reporting must work on a null-backend + // session that has no encoder object at all. This class supplies the one + // thing the ext layer cannot -- a command buffer in which the imported + // image is readable -- and nothing else. + VkSharedBaseObj m_contentProbe; + // Captured in InitEncoder so SetContentProbe() can Configure whenever it + // is called. Zero until InitEncoder has run. + uint32_t m_contentProbeQueueDepth = 0; + void ConfigureContentProbe(); VkSharedBaseObj m_inputCommandBufferPool; + + // The ONE queue the staged-input batch runs on, and the queue FAMILY that + // queue belongs to. Read by StageInputFrame (for the FOREIGN -> local + // acquire's destination family, on BOTH branches) and by + // SubmitStagedInputFrame (for the queue). Two reads of one fact, because + // the two must never be able to disagree: a queue-family acquire executes + // on a queue of its DESTINATION family or it is invalid, and the failure + // presents as a wedged queue -- a hang, which a timing-out test reads as + // flakiness rather than as a defect. + // + // The queue is a SESSION property, not a per-frame one, and it has to be: + // both branches take their command buffer from m_inputCommandBufferPool, + // which InitEncoder creates on ONE family -- the compute family when the + // preprocess filter exists (the filter IS that pool), the transfer or + // encode family otherwise -- and a command buffer may only be submitted + // to a queue of the family its pool was created for + // (VUID-vkQueueSubmit2-commandBuffer-03874). So the branch that ran + // cannot select the queue independently; what it must not do is name a + // different family in its barriers than the one this submit uses, which + // is precisely what these two accessors prevent. + VulkanDeviceContext::QueueFamilySubmitType GetStagedInputSubmitType() const; + uint32_t GetStagedInputQueueFamilyIdx() const; + VkSharedBaseObj m_encodeCommandBufferPool; VulkanBitstreamBufferPool m_bitstreamBuffersQueue; #ifdef VIDEO_DISPLAY_QUEUE_SUPPORT @@ -1012,6 +1940,14 @@ class VkVideoEncoder : public VkVideoRefCountBase { VkSharedBaseObj m_lastDeferredFrame; VkSemaphore m_hwLoadBalancingTimelineSemaphore; int32_t m_currentVideoQueueIndx; + // Completion timeline (ext API currency 3): created by the ext layer at + // init, before any submit -- which is what lets the submit path read + // the handle without a lock; GPU-signaled at queue flush points with + // max(externalFrameId)+1. GPU ordering only -- see the ext header. + VkSemaphore m_completionTimelineSemaphore; + bool m_completionSemaphoreExportable; + uint64_t m_maxSubmittedCompletionValue; + uint64_t m_lastSignaledCompletionValue; VkFormat m_imageQpMapFormat; VkExtent2D m_qpMapTexelSize; @@ -1026,14 +1962,123 @@ class VkVideoEncoder : public VkVideoRefCountBase { std::shared_ptr m_aqAnalyzes; #endif // NV_AQ_GPU_LIB_SUPPORTED + // THE single site that brings the assembly workers up. InitEncoder used + // to inline this; it is a function because a non-terminal drain has to + // run it a second time, and two copies of "how the assembly pipeline is + // started" is how the two drift apart. + bool StartAssemblyThreads(); + bool m_asyncAssemblyEnabled{false}; + uint32_t m_assemblyQueueCapacity{0}; AssemblyQueue m_assemblyQueue; std::vector m_assemblyThreads; + + // In-memory completion-record queue. On the ASSEMBLY-WORKER path every + // completed frame publishes a record here through PushCapturedBitstream + // when a drain-capable consumer exists (capture mode, or a registered + // completion subscriber); the encoded bytes are carried only in capture + // mode (disableFileOutput) -- in file-output mode the payload went to the + // file and the record is metadata plus the per-frame result. Drained + // by VulkanVideoEncoderExtImpl via TryPopCapturedBitstream(). + // + // The SYNCHRONOUS assembly path is deliberately OUTSIDE that sentence and + // publishes nothing: AssembleBitstreamData writes through the non- + // publishing WriteBitstreamToFileOutput and reaches no PushCapturedBitstream + // at all. That is precisely why ProcessOrderedFrames and + // ProcessOutOfOrderFrames both hard-refuse to run it while a completion + // subscriber is registered -- otherwise frames would be encoded correctly + // and reported to nobody. +public: + // Single completion edge (M6): raised from PushCapturedBitstream and + // nowhere else, after every frame completion on the ASSEMBLY-WORKER path, + // in both output modes. Set once by the Ext layer. The synchronous + // assembly path raises no edge either -- see the queue comment above; the + // encoder-sync-assembly test asserts exactly that, in both output modes. + // + // THREADING, and read this before writing a callback: the callback runs + // with m_capturedBitstreamsMutex RELEASED (so it may re-enter the + // thread-safe retrieval methods) but with the ASSEMBLY ORDERING LOCK + // (m_assemblyFileMutex) STILL HELD, because both capture paths publish + // from inside their ordering turn. The callback therefore MUST NOT BLOCK + // and MUST NOT wait on anything an assembly worker could produce -- doing + // so stalls every worker and wedges the pipeline. Chromium's callback is a + // post-to-sequence trampoline, which satisfies this by construction. + void SetOnBitstreamCaptured(std::function callback) { + std::lock_guard lock(m_capturedBitstreamsMutex); + m_onBitstreamCaptured = std::move(callback); + } + + void NotifyBitstreamCaptured(uint64_t frameId) { + std::function callback; + { + std::lock_guard lock(m_capturedBitstreamsMutex); + callback = m_onBitstreamCaptured; + } + if (callback) { + callback(frameId); + } + } + +protected: + struct CapturedBitstream { + uint64_t frameId; + std::vector bytes; + bool isIdr; + uint32_t pictureType; + // Per-frame result. VK_SUCCESS for a normal capture; the + // readback/assembly failure code (with empty bytes) otherwise, so + // the Ext caller gets an actionable per-frame error instead of a + // frame that silently never becomes ready. + VkResult status = VK_SUCCESS; + }; + std::function m_onBitstreamCaptured; + mutable std::mutex m_capturedBitstreamsMutex; + std::deque m_capturedBitstreams; + + // THE single point where a frame's completion becomes visible to the + // consumer. Every codec path must publish through here, in BOTH output + // modes -- the completion edge does not depend on disableFileOutput. + // What is STORED depends on who can drain it: + // - capture mode (disableFileOutput): the record carries the encoded + // bytes; the FIFO is the payload channel. + // - file-output mode with a completion subscriber (the Ext layer + // registers one at InitializeExt): the record carries metadata only + // (empty bytes -- the payload went to the file); the subscriber + // drains it on this very edge, so the FIFO never accumulates. + // - file-output mode with no subscriber (the file-based CLI apps): + // nothing is stored, because nothing would ever drain it, and the + // notify below degenerates to a no-op. + // The insert and the edge-raise stay paired in one place precisely so + // that a path added later cannot complete a frame and silently omit the + // edge. The captured-bitstreams lock is dropped before notifying, so the + // callback may re-enter the retrieval methods; both capture paths hold + // the assembly ordering lock across this call, so the callback runs + // under it (see SetOnBitstreamCaptured -- it must not block). + void PushCapturedBitstream(CapturedBitstream&& cap) { + const uint64_t frameId = cap.frameId; + { + std::lock_guard lock(m_capturedBitstreamsMutex); + const bool captureMode = + (m_encoderConfig && (m_encoderConfig->disableFileOutput != 0)); + if (captureMode || (m_onBitstreamCaptured != nullptr)) { + m_capturedBitstreams.push_back(std::move(cap)); + } + } + NotifyBitstreamCaptured(frameId); + } std::atomic m_assemblySequenceCounter{0}; std::atomic m_nextWriteSequence{0}; std::mutex m_assemblyFileMutex; std::condition_variable m_assemblyOrderCV; std::atomic m_assemblyErrorCount{0}; + // Frames that could not be processed. The assembly counter above speaks + // only for the async-assembly workers; a failure raised while the frame + // was being recorded or submitted happens in PushOrderedFrames, which + // counts it there. That is what lets the drain speak for the frames + // already pushed and released during the run, and not only for the last + // one. Both counters are monotonic for the life of the session and are + // zeroed with the rest of the per-session counters. + std::atomic m_frameProcessingErrorCount{0}; }; VkResult CreateVideoEncoderH264(const VulkanDeviceContext* vkDevCtx, diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.cpp index cb0e8837..31e98585 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.cpp @@ -15,6 +15,7 @@ */ #include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include "VkVideoEncoder/VkVideoEncoderAV1.h" @@ -86,7 +87,7 @@ VkResult VkVideoEncoderAV1::InitEncoderCodec(VkSharedBaseObj& enc VkResult result = InitEncoder(encoderConfig); if (result != VK_SUCCESS) { - fprintf(stderr, "\nERROR: InitEncoder() failed with ret(%d)\n", result); + VkEncPrintfErr("\nERROR: InitEncoder() failed with ret(%d)\n", result); return result; } @@ -95,7 +96,7 @@ VkResult VkVideoEncoderAV1::InitEncoderCodec(VkSharedBaseObj& enc encodeCaps.maxSingleReferenceCount < 2 && encodeCaps.maxUnidirectionalCompoundReferenceCount == 0 && encodeCaps.maxBidirectionalCompoundReferenceCount == 0) { - std::cout << "B-frames were requested but the implementation does not support multiple reference frames!" << std::endl; + VkEncOut() << "B-frames were requested but the implementation does not support multiple reference frames!" << std::endl; assert(!"B-frames not supported"); return VK_ERROR_INITIALIZATION_FAILED; } @@ -121,14 +122,14 @@ VkResult VkVideoEncoderAV1::InitEncoderCodec(VkSharedBaseObj& enc nullptr, &sessionParameters); if (result != VK_SUCCESS) { - fprintf(stderr, "\nEncodeFrame Error: Failed to get create video session parameters.\n"); + VkEncPrintfErr("\nEncodeFrame Error: Failed to get create video session parameters.\n"); return result; } result = VulkanVideoSessionParameters::Create(m_vkDevCtx, m_videoSession, sessionParameters, m_videoSessionParameters); if (result != VK_SUCCESS) { - fprintf(stderr, "\nEncodeFrame Error: Failed to get create video session object.\n"); + VkEncPrintfErr("\nEncodeFrame Error: Failed to get create video session object.\n"); return result; } @@ -169,6 +170,35 @@ VkResult VkVideoEncoderAV1::EncodeVideoSessionParameters(VkSharedBaseObjbitstreamHeaderBufferSize = bufferSize; + // HDR10 STATIC METADATA, appended to the sequence header OBU. + // + // Both output arms consume this buffer and both put it in the right + // place: BuildFrameObuSequence (file arm) copies it in first, and the + // capture arm of WriteBitstreamToFile inserts it after the temporal + // delimiter and before the frame OBU. So the temporal unit reads + // TD, sequence header, metadata, frame -- which is the order a decoder + // needs and the order that makes the metadata apply to the frames that + // follow it. + // + if (m_encoderConfig->hdrMetadata.Any()) { + bool truncated = false; + const size_t used = encodeFrameInfo->bitstreamHeaderOffset + + encodeFrameInfo->bitstreamHeaderBufferSize; + const size_t obuBytes = VkEncBuildAv1HdrMetadataObus( + m_encoderConfig->hdrMetadata, + encodeFrameInfo->bitstreamHeaderBuffer + used, + sizeof(encodeFrameInfo->bitstreamHeaderBuffer) - used, + &truncated); + if (truncated) { + VkEncPrintfErr("\nEncodeVideoSessionParameters Error: the HDR10 metadata " + "OBUs do not fit in the %zu-byte non-VCL header buffer " + "after %zu bytes of sequence header.\n", + sizeof(encodeFrameInfo->bitstreamHeaderBuffer), used); + return VK_ERROR_OUT_OF_HOST_MEMORY; + } + encodeFrameInfo->bitstreamHeaderBufferSize += obuBytes; + } + return result; } @@ -513,7 +543,7 @@ VkResult VkVideoEncoderAV1::EncodeFrame(VkSharedBaseObj& DumpStateInfo("input", 1, encodeFrameInfo); if (encodeFrameInfo->lastFrame) { - std::cout << "#### It is the last frame: " << encodeFrameInfo->frameInputOrderNum + VkEncOut() << "#### It is the last frame: " << encodeFrameInfo->frameInputOrderNum << " of type " << VkVideoGopStructure::GetFrameTypeName(encodeFrameInfo->gopPosition.pictureType) << " ###" << std::endl << std::flush; @@ -651,7 +681,11 @@ void VkVideoEncoderAV1::InitializeFrameHeader(StdVideoAV1SequenceHeader* pSequen for (uint32_t bufIdx = 0; bufIdx < STD_VIDEO_AV1_NUM_REF_FRAMES; bufIdx++) { int32_t dpbIdx = m_dpbAV1->GetRefBufDpbId(bufIdx); assert(dpbIdx != VkEncDpbAV1::INVALID_IDX); - pStdPictureInfo->ref_order_hint[bufIdx] = (uint8_t)m_dpbAV1->GetPicOrderCntVal(dpbIdx); + // Masked by the advertised width rather than by the cast, so + // this holds for whatever width the sequence header states. + pStdPictureInfo->ref_order_hint[bufIdx] = + (uint8_t)(m_dpbAV1->GetPicOrderCntVal(dpbIdx) % + (1 << ORDER_HINT_BITS)); } } } @@ -817,7 +851,7 @@ void VkVideoEncoderAV1::BuildFrameObuSequence(uint32_t frameIdx, bitstream.insert(bitstream.end(), seqHdrData, seqHdrData + encodeFrameInfo->bitstreamHeaderBufferSize); if (m_encoderConfig->verboseFrameStruct) { - std::cout << " == Non-VCL data SUCCESS" + VkEncOut() << " == Non-VCL data SUCCESS" << " Non-VCL data with size: " << encodeFrameInfo->bitstreamHeaderBufferSize << ", Input Order: " << (uint32_t)encodeFrameInfo->gopPosition.inputOrder << ", Encode Order: " << (uint32_t)encodeFrameInfo->gopPosition.encodeOrder @@ -849,7 +883,7 @@ void VkVideoEncoderAV1::BuildFrameObuSequence(uint32_t frameIdx, } if (m_encoderConfig->verboseFrameStruct) { - std::cout << " == Output VCL data SUCCESS for " << frameIdx << " with size: " << frameSize + VkEncOut() << " == Output VCL data SUCCESS for " << frameIdx << " with size: " << frameSize << ", Input Order: " << (uint32_t)encodeFrameInfo->gopPosition.inputOrder << ", Encode Order: " << (uint32_t)encodeFrameInfo->gopPosition.encodeOrder << std::endl << std::flush; @@ -882,13 +916,13 @@ VkResult VkVideoEncoderAV1::FlushBatchedTemporalUnit(VkSharedBaseObjverboseFrameStruct) { - std::cout << ">>>>>> Assembly VCL index " << curFrameIdx << " has size: " << m_bitstream[curFrameIdx].size() + VkEncOut() << ">>>>>> Assembly VCL index " << curFrameIdx << " has size: " << m_bitstream[curFrameIdx].size() << std::endl << std::flush; } } if (m_encoderConfig->verboseFrameStruct) { - std::cout << ">>>>>> Assembly total VCL data is: " + VkEncOut() << ">>>>>> Assembly total VCL data is: " << tuSize - sizeof(tdObu) << std::endl << std::flush; } @@ -914,7 +948,7 @@ VkResult VkVideoEncoderAV1::FlushBatchedTemporalUnit(VkSharedBaseObj 0) { const size_t bytesWritten = WriteDataToFile(writeData, remainingBytes); if (bytesWritten == 0) { - std::cerr << "Failed to write bitstream data for frame " << curFrameIdx << std::endl; + VkEncErr() << "Failed to write bitstream data for frame " << curFrameIdx << std::endl; return VK_ERROR_OUT_OF_HOST_MEMORY; } @@ -944,7 +978,7 @@ VkResult VkVideoEncoderAV1::AssembleBitstreamData(VkSharedBaseObj& encodeFrameInfo, uint32_t frameIdx, uint32_t ofTotalFrames, BitstreamReadback& readback) +{ + // Every frame that reaches assembly publishes exactly one completion + // record, in both output modes -- including deferred (non-shown) frames + // and show-existing calls, which are per-frame calls like any other. In + // file-output mode a deferred frame's record carries its own assembly + // turn's result even though its bytes reach the file at the batch + // flush: the edge means the frame was committed to the output channel, + // matching the per-frame pairing the pending-frame model requires. + CapturedBitstream cap; + cap.frameId = (encodeFrameInfo->externalFrameId != uint64_t(-1)) + ? encodeFrameInfo->externalFrameId + : encodeFrameInfo->frameEncodeInputOrderNum; + cap.isIdr = (encodeFrameInfo->gopPosition.pictureType == + VkVideoGopStructure::FRAME_TYPE_IDR); + cap.pictureType = static_cast( + encodeFrameInfo->gopPosition.pictureType); + + VkResult result = VK_SUCCESS; + // Browser (in-memory) path. Capture the AV1 temporal + // unit into the completion record (the FIFO the VEA drains) rather than + // muxing IVF to a file. Empirically the NVIDIA driver does NOT emit the + // leading Temporal-Delimiter OBU, so prepend it -- as the standalone + // IVF writer does -- else the OBU stream cannot be split into temporal + // units by a decoder. + if (m_encoderConfig && m_encoderConfig->disableFileOutput) { + static const uint8_t kAv1TdObu[2] = { 0x12, 0x00 }; + cap.bytes.insert(cap.bytes.end(), kAv1TdObu, kAv1TdObu + 2); + if (encodeFrameInfo->bitstreamHeaderBufferSize > 0) { + const uint8_t* hdr = encodeFrameInfo->bitstreamHeaderBuffer + + encodeFrameInfo->bitstreamHeaderOffset; + cap.bytes.insert(cap.bytes.end(), hdr, + hdr + encodeFrameInfo->bitstreamHeaderBufferSize); + } + if (readback.readbackDone && readback.bitstreamSize > 0) { + const uint8_t* src; + if (!readback.bitstreamCopy.empty()) { + src = readback.bitstreamCopy.data(); + } else { + VkDeviceSize maxSize; + src = encodeFrameInfo->outputBitstreamBuffer->GetDataPtr(0, maxSize) + + readback.bitstreamStartOffset; + } + cap.bytes.insert(cap.bytes.end(), src, src + readback.bitstreamSize); + } + } else { + result = WriteBitstreamToFileOutput(encodeFrameInfo, frameIdx, readback); + cap.status = result; // VK_SUCCESS, or the file-write failure code + } + PushCapturedBitstream(std::move(cap)); + return result; +} + +// File-output arm of the AV1 override: the show-existing header write, the +// per-frame OBU staging (BuildFrameObuSequence) and deferred (non-shown) frame +// batching, then the flush of the completed temporal unit. The IVF mux itself +// lives in FlushBatchedTemporalUnit(). Private and non-virtual: it must never +// grow a second completion publish. +VkResult VkVideoEncoderAV1::WriteBitstreamToFileOutput( + VkSharedBaseObj& encodeFrameInfo, + uint32_t frameIdx, BitstreamReadback& readback) { VkVideoEncodeFrameInfoAV1* pFrameInfo = GetEncodeFrameInfoAV1(encodeFrameInfo); @@ -1118,8 +1212,17 @@ void VkVideoEncoderAV1::InsertOrdered(VkSharedBaseObj& c // For out of order frames, insert display-frameheader in display order if (node->dependantFrames != nullptr) { VkSharedBaseObj showExistingFrameInfo; - GetAvailablePoolNode(showExistingFrameInfo); - assert(showExistingFrameInfo); + // CHECKED, not asserted. This is the SECOND pool node this insert + // needs -- the ext layer reserves exactly one per admitted input + // frame -- so a miss is reachable, and assert() compiles out. In a + // release build the miss left the handle null and the + // GetEncodeFrameInfoAV1() below dereferenced it. + if (!GetAvailablePoolNode(showExistingFrameInfo) || + !showExistingFrameInfo) { + VkEncPrintfErr("[EncoderAV1] no pool node for the show_existing_frame " + "companion; emitting the reordered frame without it\n"); + return; + } VkVideoEncodeFrameInfoAV1* pCurrentFrameInfo = GetEncodeFrameInfoAV1(showExistingFrameInfo); pCurrentFrameInfo->bOverlayFrame = true; diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.h index 5104bedd..ccb3dfaf 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderAV1.h @@ -134,13 +134,13 @@ class VkVideoEncoderAV1 : public VkVideoEncoder , m_numBFramesToEncode() { } - virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig); - virtual VkResult InitRateControl(VkCommandBuffer cmdBuf, uint32_t qp); + virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig) override; + virtual VkResult InitRateControl(VkCommandBuffer cmdBuf, uint32_t qp) override; virtual VkResult EncodeVideoSessionParameters(VkSharedBaseObj& encodeFrameInfo); virtual VkResult ProcessDpb(VkSharedBaseObj& encodeFrameInfo, - uint32_t frameIdx, uint32_t ofTotalframes); - virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes); - virtual bool GetAvailablePoolNode(VkSharedBaseObj& encodeFrameInfo) { + uint32_t frameIdx, uint32_t ofTotalframes) override; + virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes) override; + virtual bool GetAvailablePoolNode(VkSharedBaseObj& encodeFrameInfo) override{ VkSharedBaseObj encodeFrameInfoAV1; bool success = m_frameInfoBuffersQueue->GetAvailablePoolNode(encodeFrameInfoAV1); if (success) { @@ -149,25 +149,50 @@ class VkVideoEncoderAV1 : public VkVideoEncoder return success; } - virtual VkResult StartOfVideoCodingEncodeOrder(VkSharedBaseObj& encodeFrameInfo, uint32_t frameIdx, uint32_t ofTotalFrames); + virtual VkResult StartOfVideoCodingEncodeOrder(VkSharedBaseObj& encodeFrameInfo, uint32_t frameIdx, uint32_t ofTotalFrames) override; virtual VkResult RecordVideoCodingCmd(VkSharedBaseObj& encodeFrameInfo, - uint32_t frameIdx, uint32_t ofTotalFrames); + uint32_t frameIdx, uint32_t ofTotalFrames) override; virtual VkResult SubmitVideoCodingCmds(VkSharedBaseObj& encodeFrameInfo, - uint32_t frameIdx, uint32_t ofTotalFrames); + uint32_t frameIdx, uint32_t ofTotalFrames) override; virtual VkResult AssembleBitstreamData(VkSharedBaseObj& encodeFrameInfo, - uint32_t frameIdx, uint32_t ofTotalFrames); + uint32_t frameIdx, uint32_t ofTotalFrames) override; virtual VkResult ReadbackBitstreamData(VkSharedBaseObj& encodeFrameInfo, - BitstreamReadback& readback); + BitstreamReadback& readback) override; virtual VkResult WriteBitstreamToFile(VkSharedBaseObj& encodeFrameInfo, uint32_t frameIdx, uint32_t ofTotalFrames, - BitstreamReadback& readback); + BitstreamReadback& readback) override; void WriteShowExistingFrameHeader(VkSharedBaseObj& encodeFrameInfo); +private: + // File-output arm of the AV1 WriteBitstreamToFile override: show-existing + // header write, per-frame OBU staging and deferred-frame batching, then the + // temporal-unit flush. Private and non-virtual: it must never grow a second + // completion publish. + VkResult WriteBitstreamToFileOutput(VkSharedBaseObj& encodeFrameInfo, + uint32_t frameIdx, BitstreamReadback& readback); + +public: + virtual void InsertOrdered(VkSharedBaseObj& current, VkSharedBaseObj& prev, - VkSharedBaseObj& node); + VkSharedBaseObj& node) override; + + // InsertOrdered() splices ONE show_existing_frame node into the chain per + // reordered insert and counts it, and QueueFramesForAssembly walks it like + // any other node -- so an AV1 burst is one larger than the base. Exactly + // one per mini-GOP: only the reference-frame insert lands ahead of + // existing nodes. + // + // `override` IS REQUIRED. This header omits it elsewhere by house style, + // and CanAcceptNewInputFrame() is const: a cv-qualifier or signature + // mismatch here would SILENTLY SHADOW rather than override, the base + // version would be called through the base pointer, and the AV1 +1 would + // be lost with no compile error at all. + virtual size_t GetMaxAssemblyBurst() const override { + return VkVideoEncoder::GetMaxAssemblyBurst() + 1u; + } void AppendShowExistingFrame(VkSharedBaseObj& prev, VkSharedBaseObj& node); @@ -190,8 +215,8 @@ class VkVideoEncoderAV1 : public VkVideoEncoder } // Must be called from VkVideoEncoder::EncodeFrameCommon only - virtual VkResult EncodeFrame(VkSharedBaseObj& encodeFrameInfo); - virtual VkResult CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo); + virtual VkResult EncodeFrame(VkSharedBaseObj& encodeFrameInfo) override; + virtual VkResult CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo) override; private: VkVideoEncodeFrameInfoAV1* GetEncodeFrameInfoAV1(VkSharedBaseObj& encodeFrameInfo) { diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderContentProbe.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderContentProbe.cpp new file mode 100644 index 00000000..a6e2a66a --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderContentProbe.cpp @@ -0,0 +1,498 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "VkVideoEncoder/VkVideoEncoderContentProbe.h" + +#include +#include + +#include "VkCodecUtils/VkImageResource.h" +#include "VkCodecUtils/VulkanDeviceContext.h" + +VkResult VkVideoEncoderContentProbe::Create( + VkSharedBaseObj& probe) +{ + probe.reset(new VkVideoEncoderContentProbe()); + return (probe != nullptr) ? VK_SUCCESS : VK_ERROR_OUT_OF_HOST_MEMORY; +} + +VkVideoEncoderContentProbe::~VkVideoEncoderContentProbe() +{ + Deinit(); +} + +void VkVideoEncoderContentProbe::Configure(const VulkanDeviceContext* vkDevCtx, + uint32_t poolDepth, + uint32_t queueFamilyIndex, + uint32_t encodeWidth, + uint32_t encodeHeight) +{ + std::lock_guard lock(m_mutex); + m_vkDevCtx = vkDevCtx; + // +2 over the queue depth for the same reason the PSNR capture pool takes + // it: a node is held from the record site until the post-fence score, so + // the in-flight set can briefly exceed the encode queue depth. + m_poolDepth = poolDepth + 2u; + m_queueFamilyIndex = queueFamilyIndex; + m_encodeExtent.width = encodeWidth; + m_encodeExtent.height = encodeHeight; +} + +bool VkVideoEncoderContentProbe::IsProbeableFormat(VkFormat format) +{ + return IsProbeableFormatLocked(format); +} + +bool VkVideoEncoderContentProbe::IsProbeableFormatLocked(VkFormat format) +{ + // 8-bit 2-plane 420 only. The predicate is stated over Y, U and V plane + // means, so a format with no chroma planes to score (RGBA) has no verdict + // to give, and a 10/12-bit packed format stores its samples in the high + // bits of 16-bit words -- the byte-wise mean below would be reading the + // wrong half. Widening this means teaching the scorer the word layout, + // not just adding a case here, and NOT_APPLICABLE is the honest answer + // until someone does. + return (format == VK_FORMAT_G8_B8R8_2PLANE_420_UNORM); +} + +void VkVideoEncoderContentProbe::ArmRegistration(uint64_t registrationId, + CaptureSite captureSite) +{ + if (registrationId == 0) { + return; + } + std::lock_guard lock(m_mutex); + Registration& reg = m_registrations[registrationId]; + // A re-arm of an already-scored registration must not erase its verdict. + // + // A registration id is reused only after the generation counter has + // invalidated the old one, and ForgetRegistration has already run by that + // point. So a re-arm arriving for an id that still holds a verdict is a + // REPEAT of a live registration, not a new one, and the verdict it + // reached must survive it. + if ((reg.state == STATE_CLEAN) || (reg.state == STATE_DAMAGED_CHROMA) || + (reg.state == STATE_DAMAGED_ALL)) { + return; + } + // See the header: this is "can the probe's readback ride this + // registration", NOT "is this buffer directly encodable". Latched here + // rather than discovered per frame so the caller gets the answer in its + // registration echo. + reg.state = (captureSite == CaptureSite::kReachable) ? STATE_ARMED + : STATE_NOT_APPLICABLE; + reg.capturePending = false; +} + +void VkVideoEncoderContentProbe::ForgetRegistration(uint64_t registrationId) +{ + std::lock_guard lock(m_mutex); + m_registrations.erase(registrationId); + m_verdicts.erase(registrationId); + m_damagedOrder.erase( + std::remove(m_damagedOrder.begin(), m_damagedOrder.end(), registrationId), + m_damagedOrder.end()); + // m_probedCount / m_damagedCount are SESSION TOTALS and are deliberately + // not decremented: "three of the buffers this session imported were + // damaged" stays true after those three are retired, and a consumer + // reading a falling total would conclude the damage went away. +} + +VkVideoEncoderContentProbe::State +VkVideoEncoderContentProbe::GetRegistrationState(uint64_t registrationId) const +{ + std::lock_guard lock(m_mutex); + auto it = m_registrations.find(registrationId); + return (it == m_registrations.end()) ? STATE_NOT_EVALUATED : it->second.state; +} + +bool VkVideoEncoderContentProbe::IsArmed() const +{ + std::lock_guard lock(m_mutex); + return !m_registrations.empty(); +} + +bool VkVideoEncoderContentProbe::NeedsCapture(uint64_t registrationId) const +{ + std::lock_guard lock(m_mutex); + auto it = m_registrations.find(registrationId); + return (it != m_registrations.end()) && (it->second.state == STATE_ARMED) && + !it->second.capturePending; +} + +bool VkVideoEncoderContentProbe::RecordCapture(VkCommandBuffer cmdBuf, + uint64_t registrationId, + VkImage srcImage, + VkFormat srcFormat, + const VkExtent2D& srcExtent, + FrameCapture& outCapture) +{ + std::lock_guard lock(m_mutex); + if ((m_vkDevCtx == nullptr) || (cmdBuf == VK_NULL_HANDLE) || + (srcImage == VK_NULL_HANDLE)) { + return false; + } + auto it = m_registrations.find(registrationId); + if ((it == m_registrations.end()) || (it->second.state != STATE_ARMED) || + it->second.capturePending) { + return false; + } + if (!IsProbeableFormatLocked(srcFormat)) { + // Latched, not returned bare: without this the caller would ask again + // on every frame of a session whose format can never be scored. + it->second.state = STATE_NOT_APPLICABLE; + return false; + } + + const uint32_t w = std::min(srcExtent.width, m_encodeExtent.width); + const uint32_t h = std::min(srcExtent.height, m_encodeExtent.height); + if ((w == 0) || (h == 0)) { + it->second.state = STATE_NOT_APPLICABLE; + return false; + } + + // Lazily configured, so a session that arms nothing allocates nothing. + // The extent is the first probed registration's; a later registration + // larger than the pool is refused rather than truncated, because a + // truncated read scores a region the producer may legitimately not have + // written and the whole point is that the score is trustworthy. + if (m_pool == nullptr) { + m_poolFormat = srcFormat; + m_poolExtent.width = srcExtent.width; + m_poolExtent.height = srcExtent.height; + VkResult r = VulkanVideoImagePool::Create(m_vkDevCtx, m_pool); + if (r == VK_SUCCESS) { + r = m_pool->Configure( + m_vkDevCtx, m_poolDepth, m_poolFormat, m_poolExtent, + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, + m_queueFamilyIndex, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, + nullptr, VK_IMAGE_ASPECT_COLOR_BIT, false, false, true); + } + if (r != VK_SUCCESS) { + m_pool = nullptr; + it->second.state = STATE_NOT_APPLICABLE; + return false; + } + } + if ((srcFormat != m_poolFormat) || + (srcExtent.width > m_poolExtent.width) || + (srcExtent.height > m_poolExtent.height)) { + it->second.state = STATE_NOT_APPLICABLE; + return false; + } + + VkSharedBaseObj node; + if (!m_pool->GetAvailableImage(node, VK_IMAGE_LAYOUT_UNDEFINED)) { + // Transient: every node is still held by an unscored capture. Leave + // the registration ARMED so the next frame retries -- this is the one + // refusal that is not a latch. + return false; + } + VkSharedBaseObj dstView; + node->GetImageView(dstView); + if (!dstView) { + return false; + } + const VkImage dstImage = dstView->GetImageResource()->GetImage(); + + // ORDERING AND LAYOUT, and why only the destination needs a barrier. + // This is recorded into the staging command buffer immediately after + // CopyLinearToOptimalImage read |srcImage|, so the source is already in + // TRANSFER_SRC_OPTIMAL and already owned by this queue family. What is + // needed is (a) the destination brought to TRANSFER_DST_OPTIMAL from + // UNDEFINED, and (b) a TRANSFER->TRANSFER execution dependency so this + // copy is ordered after the library's own read of the same image. + VkImageMemoryBarrier2KHR bar = {}; + bar.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2_KHR; + bar.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + bar.srcAccessMask = VK_ACCESS_2_NONE; + bar.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + bar.dstAccessMask = VK_ACCESS_2_TRANSFER_WRITE_BIT; + bar.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + bar.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + bar.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + bar.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + bar.image = dstImage; + bar.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1 }; + VkDependencyInfoKHR dep = {}; + dep.sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO_KHR; + dep.imageMemoryBarrierCount = 1; + dep.pImageMemoryBarriers = &bar; + m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &dep); + + VkImageCopy luma = { { VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { w, h, 1 } }; + m_vkDevCtx->CmdCopyImage(cmdBuf, srcImage, + VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dstImage, + VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &luma); + VkImageCopy chroma = { { VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { (w + 1) / 2, (h + 1) / 2, 1 } }; + m_vkDevCtx->CmdCopyImage(cmdBuf, srcImage, + VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dstImage, + VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &chroma); + + it->second.capturePending = true; + outCapture.image = node; + outCapture.registrationId = registrationId; + outCapture.width = w; + outCapture.height = h; + return true; +} + +void VkVideoEncoderContentProbe::ScoreCapture(FrameCapture& capture) +{ + if (capture.image == nullptr) { + return; + } + // Released on EVERY exit below, including the failure ones: a node that + // is never released is a pool slot gone for the life of the session, and + // the pool is sized for the number of concurrent probes and nothing more. + struct Release { + FrameCapture* c; + ~Release() { c->image = nullptr; } + } release{ &capture }; + + std::lock_guard lock(m_mutex); + if (m_vkDevCtx == nullptr) { + return; + } + auto it = m_registrations.find(capture.registrationId); + if (it == m_registrations.end()) { + // Unregistered between the record and the fence. Nothing to report to. + return; + } + it->second.capturePending = false; + if (it->second.state != STATE_ARMED) { + return; + } + + VkSharedBaseObj view; + capture.image->GetImageView(view); + if (!view) { + return; + } + const VkSharedBaseObj& res = view->GetImageResource(); + VkDevice device = m_vkDevCtx->getDevice(); + void* mapped = nullptr; + if ((m_vkDevCtx->MapMemory(device, res->GetDeviceMemory(), + res->GetImageDeviceMemoryOffset(), + res->GetImageDeviceMemorySize(), 0, + &mapped) != VK_SUCCESS) || + (mapped == nullptr)) { + return; + } + + const uint8_t* base = static_cast(mapped); + const uint32_t w = capture.width; + const uint32_t h = capture.height; + + VkImage img = res->GetImage(); + VkImageSubresource sub = {}; + VkSubresourceLayout ly = {}; + VkSubresourceLayout lc = {}; + sub.aspectMask = VK_IMAGE_ASPECT_PLANE_0_BIT; + m_vkDevCtx->GetImageSubresourceLayout(device, img, &sub, &ly); + sub.aspectMask = VK_IMAGE_ASPECT_PLANE_1_BIT; + m_vkDevCtx->GetImageSubresourceLayout(device, img, &sub, &lc); + + uint32_t meanY = 0, meanU = 0, meanV = 0; + ScorePlanesQ8(base, ly, lc, w, h, m_scratch, meanY, meanU, meanV); + m_vkDevCtx->UnmapMemory(device, res->GetDeviceMemory()); + + if ((w == 0) || (h == 0)) { + return; + } + ApplyVerdictLocked(capture.registrationId, meanY, meanU, meanV); +} + +void VkVideoEncoderContentProbe::ScorePlanesQ8( + const uint8_t* base, const VkSubresourceLayout& lumaLayout, + const VkSubresourceLayout& chromaLayout, uint32_t width, uint32_t height, + std::vector& scratch, uint32_t& outMeanYQ8, uint32_t& outMeanUQ8, + uint32_t& outMeanVQ8) +{ + outMeanYQ8 = 0; + outMeanUQ8 = 0; + outMeanVQ8 = 0; + if ((base == nullptr) || (width == 0) || (height == 0)) { + return; + } + const uint32_t cw = (width + 1) / 2; + const uint32_t ch = (height + 1) / 2; + + // See kRowStride's comment: the bulk memcpy into cached scratch plus a + // strided row set is what keeps this off the ~200 fps -> 3.6 fps cliff a + // byte-wise walk of the same memory already fell off once in this tree. + scratch.resize((size_t)std::max(width, 2u * cw) + 64u); + uint8_t* row = scratch.data(); + + uint64_t sumY = 0, nY = 0; + for (uint32_t y = 0; y < height; y += kRowStride) { + memcpy(row, base + lumaLayout.offset + ((size_t)y * lumaLayout.rowPitch), + width); + for (uint32_t x = 0; x < width; x++) { + sumY += row[x]; + } + nY += width; + } + uint64_t sumU = 0, sumV = 0, nC = 0; + for (uint32_t y = 0; y < ch; y += kRowStride) { + memcpy(row, + base + chromaLayout.offset + ((size_t)y * chromaLayout.rowPitch), + (size_t)2 * cw); + for (uint32_t x = 0; x < cw; x++) { + sumU += row[(2 * x) + 0]; + sumV += row[(2 * x) + 1]; + } + nC += cw; + } + if (nY != 0) { + outMeanYQ8 = (uint32_t)((sumY * 256u) / nY); + } + if (nC != 0) { + outMeanUQ8 = (uint32_t)((sumU * 256u) / nC); + outMeanVQ8 = (uint32_t)((sumV * 256u) / nC); + } +} + +// ===== THE PREDICATE ===== +// +// The UNION of the two measured damage modes, and it is a union on purpose. +// A chroma-only scorer -- which is what this tree's PSNR readback has, +// `if ((u == 0) && (v == 0)) zeroUV++` with no luma term at all -- reports an +// ALL_ZERO buffer and a CHROMA_ZERO buffer identically, and that conflation +// has produced a wrong conclusion on this defect twice. The two need the same +// reaction and different explanations, so they get different states. +// +// THE UNCOVERED QUADRANT, named rather than left to be discovered: luma dead +// with LIVE chroma scores CLEAN here. That is deliberate. It is not one of the +// two modes this driver defect has ever produced (the chroma plane is either +// dead with luma alive, or everything is dead), and a frame with a +// legitimately black luma plane over live chroma is a real thing a producer +// can send. Adding it would trade a false negative nobody has observed for a +// false positive that reroutes a working buffer. +// +// AND A FULLY LEGAL BLACK FRAME DOES NOT TRIP IT, which is the whole reason a +// content test is admissible here: black is U = V = 128, so meanU and meanV +// are 32768 in these units -- sixty-four times the threshold. +VkVideoEncoderContentProbe::State +VkVideoEncoderContentProbe::ClassifyPlaneMeansQ8(uint32_t meanYQ8, + uint32_t meanUQ8, + uint32_t meanVQ8) +{ + const bool yDead = (meanYQ8 < kDeadPlaneMeanQ8); + const bool uDead = (meanUQ8 < kDeadPlaneMeanQ8); + const bool vDead = (meanVQ8 < kDeadPlaneMeanQ8); + if (yDead && uDead && vDead) { + return STATE_DAMAGED_ALL; + } + if (!yDead && (uDead || vDead)) { + return STATE_DAMAGED_CHROMA; + } + return STATE_CLEAN; +} + +void VkVideoEncoderContentProbe::ApplyVerdict(uint64_t registrationId, + uint32_t meanYQ8, + uint32_t meanUQ8, + uint32_t meanVQ8) +{ + std::lock_guard lock(m_mutex); + ApplyVerdictLocked(registrationId, meanYQ8, meanUQ8, meanVQ8); +} + +void VkVideoEncoderContentProbe::ApplyVerdictLocked(uint64_t registrationId, + uint32_t meanYQ8, + uint32_t meanUQ8, + uint32_t meanVQ8) +{ + auto it = m_registrations.find(registrationId); + if ((it == m_registrations.end()) || (it->second.state != STATE_ARMED)) { + // Not armed, already scored, or retired between the record and the + // fence. A second verdict for one registration must never move the + // session counters: the probe's unit is the buffer, not the frame. + return; + } + Verdict verdict; + verdict.registrationId = registrationId; + verdict.meanY = meanYQ8; + verdict.meanU = meanUQ8; + verdict.meanV = meanVQ8; + verdict.state = ClassifyPlaneMeansQ8(meanYQ8, meanUQ8, meanVQ8); + + it->second.state = verdict.state; + m_verdicts[registrationId] = verdict; + m_probedCount++; + if (verdict.state == STATE_CLEAN) { + m_lastCleanVerdict = verdict; + } else { + m_damagedCount++; + m_damagedOrder.push_back(registrationId); + } +} + +void VkVideoEncoderContentProbe::GetSnapshot(Verdict& outVerdict, + uint32_t& outProbedCount, + uint32_t& outDamagedCount, + uint32_t& outArmedCount) const +{ + std::lock_guard lock(m_mutex); + outProbedCount = m_probedCount; + outDamagedCount = m_damagedCount; + // COUNTED HERE, NOT TRACKED AS A RUNNING TALLY. Every transition out of + // STATE_ARMED already exists in three places (ApplyVerdictLocked's + // verdict, RecordCapture's two NOT_APPLICABLE latches) and + // ForgetRegistration removes entries outright; a counter incremented and + // decremented across four sites is a counter that drifts. m_registrations + // is small by construction -- it is one entry per REGISTERED BUFFER, five + // to eleven on the owner's measured sessions, not one per frame -- so + // walking it under a lock already held costs nothing worth naming. + outArmedCount = 0; + for (const auto& entry : m_registrations) { + if (entry.second.state == STATE_ARMED) { + outArmedCount++; + } + } + // OLDEST STILL-REGISTERED DAMAGED FIRST. A caller reacts by retiring that + // registration, ForgetRegistration drops it from m_damagedOrder, and the + // next poll surfaces the next one. A caller that does NOT react sees the + // same entry again -- so a verdict is never consumed by being read, and + // two buffers damaged between two polls both get reported. "Most recent" + // would have lost the older one permanently, because a registration is + // probed exactly once and there is no second look coming. + for (uint64_t id : m_damagedOrder) { + auto v = m_verdicts.find(id); + if (v != m_verdicts.end()) { + outVerdict = v->second; + return; + } + } + outVerdict = m_lastCleanVerdict; +} + +void VkVideoEncoderContentProbe::Deinit() +{ + std::lock_guard lock(m_mutex); + m_pool = nullptr; + m_registrations.clear(); + m_verdicts.clear(); + m_damagedOrder.clear(); + m_scratch.clear(); + m_scratch.shrink_to_fit(); +} diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderContentProbe.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderContentProbe.h new file mode 100644 index 00000000..555a97ff --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderContentProbe.h @@ -0,0 +1,324 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef _VKVIDEOENCODER_VKVIDEOENCODERCONTENTPROBE_H_ +#define _VKVIDEOENCODER_VKVIDEOENCODERCONTENTPROBE_H_ + +#include +#include +#include +#include + +#include "VkCodecUtils/VkVideoRefCountBase.h" +#include "VkCodecUtils/VulkanVideoImagePool.h" +#include "vulkan/vulkan.h" + +class VulkanDeviceContext; + +// The dma-buf IMPORT CONTENT PROBE. +// +// WHAT IT IS FOR, in one sentence: a dma-buf import can come back bound to +// memory the producer's writes never reach, and this is the only +// place in the library that can SEE that, because it is the only place that +// reads imported pixels on the host. +// +// The contract, the exact predicate, the false-positive budget and the +// latency cost are all stated at VkVideoEncoderImportContentInfo, in the +// descriptor API's own header. Named by symbol rather than by file: this +// class sits below that layer and must not depend on where it is declared. +// This header does not restate the contract; it describes the mechanism. +// +// THE MECHANISM, and why it is shaped this way. +// +// * The capture is TWO vkCmdCopyImage into the caller's OWN staging command +// buffer, one command after VkVideoEncoder::StageInputFrame's +// CopyLinearToOptimalImage has read the same image in the same layout. +// No extra submit, no extra fence, no extra queue. The imported image is +// already in TRANSFER_SRC_OPTIMAL at that point -- the staging arm put it +// there -- so only the destination needs a barrier. +// +// * The scoring is host-side, off a HOST_VISIBLE|HOST_COHERENT LINEAR pool +// image, and MUST run only after that command buffer's fence has been +// waited. VkVideoEncoder calls it from the same two post-fence sites the +// PSNR readback uses. +// +// * ROW STRIDE. Only every kRowStride-th row is touched, and that is a +// requirement rather than a micro-optimisation: a byte-wise walk of this +// same 3.1 MB of HOST_VISIBLE memory costs VkVideoEncoderPsnr the +// difference between ~200 fps and 3.6 fps. A strided mean is ample for +// "is this plane dead", which is the only question asked. +// +// * ONCE PER REGISTRATION. Not once per frame. The defect is a property of +// the IMPORT, is permanent for the life of that buffer (a damaged import +// is not recoverable, and re-importing at the next ordinal rescues it in +// 0 of 8 attempts), and 100% of a damaged buffer's frames carry it -- so a +// second look costs a readback and can learn nothing. On a 5125-frame +// session with 5 registered buffers this is 5 probes, not 5125. +// +// WHAT IT DELIBERATELY DOES NOT DO. It does not write to stderr. A host that +// silences stdio, or that runs the encoder in a child process whose stderr it +// never reads, would lose every finding -- and a diagnostic whose only signal +// can be discarded by the caller it exists for tells that caller nothing. +// Every answer leaves through the chained struct, which the caller must read. +class VkVideoEncoderContentProbe : public VkVideoRefCountBase { +public: + // Mirrors VkVideoEncoderImportContentState, in + // vulkan_video_encoder_ext_internal.h. Kept as a separate enum because + // this class sits BELOW the ext layer and must not depend on it; + // vulkan_video_encoder_ext.cpp maps one onto the other, and a + // static_assert there pins the mapping. + enum State { + STATE_NOT_EVALUATED = 0, + STATE_NOT_APPLICABLE = 1, + STATE_ARMED = 2, + STATE_CLEAN = 3, + STATE_DAMAGED_CHROMA = 4, + STATE_DAMAGED_ALL = 5, + }; + + // Whether the probe's readback can ride a given registration at all. + // + // A SCOPED ENUM AND NOT A BOOL, on purpose. Its true-ish value means "DO + // arm", which is the opposite sense of the `bool directlyEncodable` a caller + // reaching for the obvious spelling would pass, whose true means "do not + // arm". A bool lets a call site compile with its meaning inverted -- arming + // what should be refused and refusing what should be armed -- and says + // nothing. An enum class makes each of those a compile error. + enum class CaptureSite { + kUnreachable = 0, + kReachable = 1, + }; + + // The predicate's threshold in Q8 (plane mean * 256): strictly below + // 2.0/255. Must equal VK_VIDEO_ENCODER_IMPORT_CONTENT_DEAD_PLANE_MEAN_Q8; + // vulkan_video_encoder_ext.cpp static_asserts that. + static constexpr uint32_t kDeadPlaneMeanQ8 = 512u; + + // Travels on VkVideoEncodeFrameInfo from the record site (submit thread) + // to the score site (assembly thread, post-fence). + struct FrameCapture { + VkSharedBaseObj image; + uint64_t registrationId = 0; + uint32_t width = 0; + uint32_t height = 0; + }; + + struct Verdict { + State state = STATE_NOT_EVALUATED; + uint64_t registrationId = 0; + uint32_t meanY = 0; // Q8 + uint32_t meanU = 0; // Q8 + uint32_t meanV = 0; // Q8 + }; + + static VkResult Create(VkSharedBaseObj& probe); + + // Idempotent; called from VkVideoEncoder::InitEncoder. Allocates nothing: + // the image pool is configured lazily on the first capture, so a session + // that arms no registration pays no memory at all. + void Configure(const VulkanDeviceContext* vkDevCtx, + uint32_t poolDepth, + uint32_t queueFamilyIndex, + uint32_t encodeWidth, + uint32_t encodeHeight); + + // ---- Arming, from the registration thread ------------------------- + // + // |captureSiteReachable| is the ONE fact that decides whether a + // registration can be probed, and it is deliberately NOT "is this buffer + // directly encodable". Coupling the two would make the probe + // structurally blind to the class of buffer the driver defect appears + // on: a BLOCK-LINEAR import that also carries VIDEO_ENCODE_SRC + // classifies encodeCapable -- because encodeCapable's middle clause is + // (tiling != VK_IMAGE_TILING_LINEAR) -- and encodeCapable would latch + // NOT_APPLICABLE here, while block-linear imports are exactly the class + // that gets poisoned. Tiling plays no part in this decision. + // + // What replaces it is the only thing that actually decides it: does a + // transfer-readable copy of the PRODUCER'S pixels pass through this + // library for this registration. The DIRECT path now makes that true for + // itself with a one-frame staged detour, taken only while a capture is + // still owed (VkVideoEncoder::SetExternalInputFrameWithNode). + // + // A registration whose capture site is genuinely unreachable -- a + // FILTER-routed one, which is sampled and never copied, or an import + // without TRANSFER_SRC, out of which no copy may legally be recorded -- + // is armed as NOT_APPLICABLE rather than refused: the caller asked, and + // "your buffer never takes the path this probe is hooked to" is an + // answer, not an error. + void ArmRegistration(uint64_t registrationId, CaptureSite captureSite); + + // Whether the SCORER can read this format at all -- a pure function of + // the format, holding no state and needing no lock. + // + // PUBLIC BECAUSE THE ARM DECISION HAS TO ASK IT. Consulting it only from + // RecordCapture, on the first frame, lets a registration in a format the + // scorer cannot read echo ARMED at import and then be downgraded to + // NOT_APPLICABLE a frame later, which leaves a session indistinguishable from + // one that probed every buffer and found them all clean. An ARM is a promise + // of a verdict; a promise the scorer cannot keep must be refused where it is + // made. + static bool IsProbeableFormat(VkFormat format); + // Drop everything remembered about a registration. Called from + // UnregisterImageResource, which is also what makes the damaged-list + // report in GetSnapshot() drain as a caller reacts to it. + void ForgetRegistration(uint64_t registrationId); + State GetRegistrationState(uint64_t registrationId) const; + bool IsArmed() const; + + // ---- Capture, from the submit thread ------------------------------ + // + // True only while this registration is armed and has neither been + // captured nor scored. Cheap enough to call per frame. + bool NeedsCapture(uint64_t registrationId) const; + // Records the readback into |cmdBuf| and hands back the pool node in + // |outCapture|. Returns false without touching |cmdBuf| when the probe + // cannot run (unsupported format, pool exhausted, not configured), and + // latches NOT_APPLICABLE for an unsupported format so the caller is not + // asked again every frame. + bool RecordCapture(VkCommandBuffer cmdBuf, + uint64_t registrationId, + VkImage srcImage, + VkFormat srcFormat, + const VkExtent2D& srcExtent, + FrameCapture& outCapture); + + // ---- Scoring, from the assembly thread, POST-FENCE ---------------- + // + // Consumes |capture| (releases the pool node) whether or not it scores. + void ScoreCapture(FrameCapture& capture); + + // The two halves of the scoring, split out as PURE FUNCTIONS -- and the + // split exists so they can be tested at all. + // + // Everything else about a probe needs a Vulkan device: a real import, a + // real staging command buffer, a real fence. The DECISION does not, and + // it is the part that has to be right. A driver-workaround predicate that + // is only exercised on the one host that has the broken driver is a + // predicate nobody can regression-test, and this project has already + // shipped observables whose test could not fail. These two take bytes and + // numbers, and test/encoder-ext-import-content drives them over + // synthetic NV12 buffers -- chroma-zeroed, all-zero, legal black and + // ordinary content -- on any host, with no GPU. + + // Reads a HOST-VISIBLE, LINEAR, 8-bit 2-plane 420 image and returns the + // three plane means in Q8 (mean * 256). |base| is the mapped memory, + // |lumaLayout| and |chromaLayout| the VkSubresourceLayout of PLANE_0 and + // PLANE_1 -- offsets and row pitches included, because a real image is + // padded and a scorer that assumes width == rowPitch reads the padding. + // |scratch| is caller-owned reusable storage; see kRowStride. + static void ScorePlanesQ8(const uint8_t* base, + const VkSubresourceLayout& lumaLayout, + const VkSubresourceLayout& chromaLayout, + uint32_t width, uint32_t height, + std::vector& scratch, + uint32_t& outMeanYQ8, + uint32_t& outMeanUQ8, + uint32_t& outMeanVQ8); + + // THE PREDICATE. The union of the two measured damage modes; see the + // definition for the full argument, including the quadrant it does not + // cover and why. + static State ClassifyPlaneMeansQ8(uint32_t meanYQ8, uint32_t meanUQ8, + uint32_t meanVQ8); + + // Latch a measurement against a registration: classify it, record the + // verdict, and move the session counters. Public because ScoreCapture is + // not the only legitimate caller -- it is the second half of scoring, and + // separating "get the numbers" from "act on the numbers" is what lets the + // latch, the damaged ordering and the drain-on-retirement be exercised + // without a device. A no-op for a registration that is not ARMED. + void ApplyVerdict(uint64_t registrationId, uint32_t meanYQ8, + uint32_t meanUQ8, uint32_t meanVQ8); + + // ---- Reporting, from any thread ----------------------------------- + // + // Reports the OLDEST STILL-REGISTERED damaged verdict when there is one, + // and otherwise the most recent CLEAN verdict. See + // VkVideoEncoderImportContentInfo, in vulkan_video_encoder_ext_internal.h, + // for why that ordering, and not "most recent", is the one a caller can + // act on without losing a verdict between two polls. + // + // |outArmedCount| is THE ANSWER TO "DID ANY OF THIS ACTUALLY RUN", and it + // exists because without it this probe had the same disguised-inertness + // shape it was built to expose. A registration that arms and is never + // captured stays STATE_ARMED forever; GetSnapshot then reports the + // NOT_EVALUATED default verdict with probed=0 damaged=0 -- which is + // BYTE-IDENTICAL to a session that armed nothing at all, and only one + // step away from "armed everything and every buffer was clean". A probe + // whose failure mode reads as success is not a detector. + // + // It is a COUNT OF LIVE REGISTRATIONS IN STATE_ARMED, not a session + // total, and that is the useful direction: it falls to zero as captures + // land, so "still non-zero at the end of a session" is the exact + // statement "these buffers were promised a verdict and never got one". + // The one-frame staged detour in VkVideoEncoder::SetExternalInputFrame- + // WithNode is what drives it to zero for a DIRECT registration, so this + // is also the number that goes wrong first if that detour ever stops + // firing -- which is the regression C1b in + // test/encoder-ext-import-content could not otherwise catch. + void GetSnapshot(Verdict& outVerdict, + uint32_t& outProbedCount, + uint32_t& outDamagedCount, + uint32_t& outArmedCount) const; + + void Deinit(); + + ~VkVideoEncoderContentProbe() override; + +private: + // Every kRowStride-th row of each plane is read. See the class comment. + static constexpr uint32_t kRowStride = 8u; + + struct Registration { + State state = STATE_ARMED; + // Set while a capture for this registration is recorded but not yet + // scored, so a second frame of the same registration does not queue a + // second readback behind the first. + bool capturePending = false; + }; + + // Callers hold m_mutex. + static bool IsProbeableFormatLocked(VkFormat format); + void ApplyVerdictLocked(uint64_t registrationId, uint32_t meanYQ8, + uint32_t meanUQ8, uint32_t meanVQ8); + + const VulkanDeviceContext* m_vkDevCtx = nullptr; + uint32_t m_poolDepth = 0; + uint32_t m_queueFamilyIndex = 0; + VkExtent2D m_encodeExtent = {}; + + mutable std::mutex m_mutex; + VkSharedBaseObj m_pool; + VkFormat m_poolFormat = VK_FORMAT_UNDEFINED; + VkExtent2D m_poolExtent = {}; + + std::unordered_map m_registrations; + // Damaged registration ids in SCORING order. ForgetRegistration erases + // from it; GetSnapshot reports its front. + std::vector m_damagedOrder; + // Verdicts, kept so a report can carry the arithmetic behind the state. + std::unordered_map m_verdicts; + Verdict m_lastCleanVerdict; + uint32_t m_probedCount = 0; + uint32_t m_damagedCount = 0; + // Reused across scorings; guarded by m_mutex like everything else. One + // row of the widest plane, so the per-row memcpy has a fixed destination + // and the mapped HOST_VISIBLE memory is read exactly once, sequentially. + std::vector m_scratch; +}; + +#endif /* _VKVIDEOENCODER_VKVIDEOENCODERCONTENTPROBE_H_ */ diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderDef.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderDef.h index c9a35166..e4ba11c3 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderDef.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderDef.h @@ -23,6 +23,9 @@ #include "vk_video/vulkan_video_codec_h264std_encode.h" #include "vk_video/vulkan_video_codec_h265std.h" #include "vk_video/vulkan_video_codec_h265std_encode.h" +#include +#include +#include #if !defined(VK_USE_PLATFORM_WIN32_KHR) #ifndef ARRAYSIZE @@ -99,4 +102,11 @@ struct ConstQpSettings uint32_t qpIntra; }; + +// Process-wide stdio-silence gate: relocated to VkCodecUtils so common- +// library code (VulkanDeviceContext) can use it without a layering +// dependency on vk_video_encoder include paths. Included here so every +// existing VkEncOut()/VkEncErr() call site keeps compiling unchanged. +#include "VkCodecUtils/VkEncoderStdioLatch.h" + #endif /* _VKVIDEOENCODERDEF_H_ */ diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.cpp index 24131b11..36476048 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.cpp @@ -15,6 +15,7 @@ */ #include "VkVideoEncoder/VkVideoEncoderH264.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "VkVideoCore/VulkanVideoCapabilities.h" VkResult CreateVideoEncoderH264(const VulkanDeviceContext* vkDevCtx, @@ -48,7 +49,7 @@ VkResult VkVideoEncoderH264::InitEncoderCodec(VkSharedBaseObj& en VkResult result = InitEncoder(encoderConfig); if (result != VK_SUCCESS) { - fprintf(stderr, "\nERROR: InitEncoder() failed with ret(%d)\n", result); + VkEncPrintfErr("\nERROR: InitEncoder() failed with ret(%d)\n", result); return result; } @@ -57,6 +58,14 @@ VkResult VkVideoEncoderH264::InitEncoderCodec(VkSharedBaseObj& en assert(m_dpb264); m_dpb264->DpbSequenceStart(m_maxDpbPicturesCount); + // The device QP window, recorded where the capabilities are known to + // be populated -- InitEncoder above is what runs + // EncoderConfigH264::InitDeviceCapabilities. A mid-stream clamp is + // checked against this on the caller thread, which cannot safely + // reach the config. + m_deviceQpWindowMin = m_encoderConfig->h264EncodeCapabilities.minQp; + m_deviceQpWindowMax = m_encoderConfig->h264EncodeCapabilities.maxQp; + m_encoderConfig->GetRateControlParameters(&m_rateControlInfo, m_rateControlLayersInfo, &m_h264.m_rateControlInfoH264, m_h264.m_rateControlLayersInfoH264); m_encoderConfig->InitSpsPpsParameters(&m_h264.m_spsInfo, &m_h264.m_ppsInfo, @@ -79,14 +88,14 @@ VkResult VkVideoEncoderH264::InitEncoderCodec(VkSharedBaseObj& en nullptr, &sessionParameters); if(result != VK_SUCCESS) { - fprintf(stderr, "\nEncodeFrame Error: Failed to get create video session parameters.\n"); + VkEncPrintfErr("\nEncodeFrame Error: Failed to get create video session parameters.\n"); return result; } result = VulkanVideoSessionParameters::Create(m_vkDevCtx, m_videoSession, sessionParameters, m_videoSessionParameters); if(result != VK_SUCCESS) { - fprintf(stderr, "\nEncodeFrame Error: Failed to get create video session object.\n"); + VkEncPrintfErr("\nEncodeFrame Error: Failed to get create video session object.\n"); return result; } @@ -356,19 +365,43 @@ VkResult VkVideoEncoderH264::ProcessDpb(VkSharedBaseObj& } } - // It's not entirely correct to have two separate loops below, one for L0 - // and the other for L1. In each loop, elements are added to referenceSlotsInfo[] - // without checking for duplication. Duplication could occur if the same - // picture appears in both L0 and L1; AFAIK, we don't have a situation - // today like that so the two loops work fine. - // TODO: create a set out of the ref lists and then iterate over that to - // build referenceSlotsInfo[]. + // L0 AND L1 ARE LISTS; referenceSlotsInfo[] IS A SET. The same picture may + // hold a position in both reference lists, and on a B frame whose DPB + // carries a single reference picture it always does: that one picture is + // L0[0] and L1[0] alike. referenceSlotsInfo[] is not a reference list -- + // it is the set of DPB slots the recorded commands BIND -- and Vulkan + // requires each picture resource named in it to be unique + // (VUID-VkVideoBeginCodingInfoKHR-pPictureResource-07238, + // VUID-vkCmdEncodeVideoKHR-pPictureResource-08220) and each DPB frame to + // be used at most once across it and the setup slot + // (VUID-vkCmdEncodeVideoKHR-dpbFrameUseCount-08221). So both lists are + // walked and each slot is admitted at most once. + // + // This does not touch the Std reference lists. Those name DPB slots by + // index, carry their own ordering, and a slot appearing in both of them + // is what the bitstream describes. + const uint32_t firstReferenceSlot = numReferenceSlots; for (uint32_t listNum = 0; listNum < 2; listNum++) { for (uint32_t i = 0; i < refLists.refPicListCount[listNum]; i++) { int8_t slotIndex = refLists.refPicList[listNum][i]; + + // The scan starts at the first entry these loops filled: + // referenceSlotsInfo[0] is reserved for the setup slot and its + // slotIndex is not written until after them. + bool slotAlreadyBound = false; + for (uint32_t bound = firstReferenceSlot; bound < numReferenceSlots; bound++) { + if (pFrameInfo->referenceSlotsInfo[bound].slotIndex == slotIndex) { + slotAlreadyBound = true; + break; + } + } + if (slotAlreadyBound) { + continue; + } + bool refPicAvailable = m_dpb264->GetRefPicture(slotIndex, pFrameInfo->dpbImageResources[numReferenceSlots]); assert(refPicAvailable); if (!refPicAvailable) { @@ -535,7 +568,7 @@ VkResult VkVideoEncoderH264::EncodeFrame(VkSharedBaseObj DumpStateInfo("input", 1, encodeFrameInfo); if (encodeFrameInfo->lastFrame) { - std::cout << "#### It is the last frame: " << encodeFrameInfo->frameInputOrderNum + VkEncOut() << "#### It is the last frame: " << encodeFrameInfo->frameInputOrderNum << " of type " << VkVideoGopStructure::GetFrameTypeName(encodeFrameInfo->gopPosition.pictureType) << " ###" << std::endl << std::flush; @@ -598,8 +631,15 @@ VkResult VkVideoEncoderH264::EncodeFrame(VkSharedBaseObj m_IDRPicId++; } + // In capture mode (disableFileOutput -- the Chromium in-memory + // bitstream path) EVERY IDR chunk must be independently decodable: the + // VEA hands keyframe chunks to consumers (WebCodecs, muxers, + // validators) that expect in-band SPS/PPS on each keyframe, including + // mid-stream forced IDRs. The file-based sample keeps the original + // headers-once-at-stream-start behavior. if ((encodeFrameInfo->gopPosition.pictureType == VkVideoGopStructure::FRAME_TYPE_IDR) && - (encodeFrameInfo->frameEncodeInputOrderNum == 0)) { + ((encodeFrameInfo->frameEncodeInputOrderNum == 0) || + (m_encoderConfig->disableFileOutput != 0))) { VkResult result = EncodeVideoSessionParameters(encodeFrameInfo); if (result != VK_SUCCESS) { return result; @@ -641,6 +681,61 @@ VkResult VkVideoEncoderH264::EncodeFrame(VkSharedBaseObj return VK_SUCCESS; } +void VkVideoEncoderH264::RefreshCodecRateControlParameters() +{ + // READ THROUGH THE BASE CONFIG POINTER, not the H.264 member. This + // runs from VkVideoEncoder::ApplyPendingRateControlUpdate, which just + // wrote the new clamp on VkVideoEncoder::m_encoderConfig; reading the + // same pointer is what guarantees the fill sees that write. In a + // device-initialized session the two name one object anyway -- + // InitEncoderCodec builds the H.264 member as an aliasing handle on + // the very config InitEncoder stores -- so this is the same object by + // a route that also holds for a session that never ran + // InitEncoderCodec. + if (!VkVideoEncoder::m_encoderConfig) { + return; + } + EncoderConfigH264* config = + VkVideoEncoder::m_encoderConfig->GetEncoderConfigh264(); + if (config == nullptr) { + return; + } + // RESET THE LAYER STRUCT TO ITS CODEC-INIT STATE FIRST, because the + // fill only ever RAISES useMinQp/useMaxQp -- it has no else branch that + // lowers them. At codec-init that is harmless: the struct arrives + // zero-initialised but for its sType, so an unset clamp leaves the flag + // down. Re-invoked in place it is not: a clamp that was set once and is + // then cleared would keep its flag raised and go on clamping at a value + // the caller withdrew, which is the accepted-and-ignored shape in + // reverse and worse. The brace-init below is the state the constructor + // gives this member, and the fill plus CodecHandleRateControlCmd are + // its only other writers, so nothing else is lost by rebuilding it. + for (uint32_t layerIndx = 0; + layerIndx < ARRAYSIZE(m_h264.m_rateControlLayersInfoH264); + layerIndx++) { + m_h264.m_rateControlLayersInfoH264[layerIndx] = + VkVideoEncodeH264RateControlLayerInfoKHR{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_RATE_CONTROL_LAYER_INFO_KHR}; + } + config->GetRateControlParameters(&m_rateControlInfo, + m_rateControlLayersInfo, + &m_h264.m_rateControlInfoH264, + m_h264.m_rateControlLayersInfoH264); +} + +void VkVideoEncoderH264::GetResolvedQpClampForTest(uint32_t* pUseMinQp, + int32_t* pMinQpI, + uint32_t* pUseMaxQp, + int32_t* pMaxQpI) const +{ + const VkVideoEncodeH264RateControlLayerInfoKHR& layer = + m_h264.m_rateControlLayersInfoH264[0]; + *pUseMinQp = (layer.useMinQp == VK_TRUE) ? 1u : 0u; + *pMinQpI = layer.minQp.qpI; + *pUseMaxQp = (layer.useMaxQp == VK_TRUE) ? 1u : 0u; + *pMaxQpI = layer.maxQp.qpI; +} + VkResult VkVideoEncoderH264::CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo) { VkVideoEncodeFrameInfoH264* pFrameInfo = GetEncodeFrameInfoH264(encodeFrameInfo); diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.h index fa4affe5..82153509 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH264.h @@ -113,13 +113,18 @@ class VkVideoEncoderH264 : public VkVideoEncoder { , m_dpb264() { } - virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig); - virtual VkResult InitRateControl(VkCommandBuffer cmdBuf, uint32_t qp); + virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig) override; + virtual VkResult InitRateControl(VkCommandBuffer cmdBuf, uint32_t qp) override; + // DECLARES rather than overrides -- there is no base-class + // EncodeVideoSessionParameters -- so it is the one virtual here that + // correctly carries no 'override', and marking it would not compile. virtual VkResult EncodeVideoSessionParameters(VkSharedBaseObj& encodeFrameInfo); virtual VkResult ProcessDpb(VkSharedBaseObj& encodeFrameInfo, - uint32_t frameIdx, uint32_t ofTotalFrames); - virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes); - virtual bool GetAvailablePoolNode(VkSharedBaseObj& encodeFrameInfo) + uint32_t frameIdx, + uint32_t ofTotalFrames) override; + virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes) override; + virtual bool GetAvailablePoolNode( + VkSharedBaseObj& encodeFrameInfo) override { VkSharedBaseObj encodeFrameInfoH264; bool success = m_frameInfoBuffersQueue->GetAvailablePoolNode(encodeFrameInfoH264); @@ -147,8 +152,19 @@ class VkVideoEncoderH264 : public VkVideoEncoder { } // Must be called from VkVideoEncoder::EncodeFrameCommon only - virtual VkResult EncodeFrame(VkSharedBaseObj& encodeFrameInfo); - virtual VkResult CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo); + virtual VkResult EncodeFrame(VkSharedBaseObj& encodeFrameInfo) override; + virtual VkResult CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo) override; + + // The H.264 arm of the mid-stream rate-control refresh: the same fill + // InitEncoderCodec runs once, re-run against the current config so a + // QP clamp changed by Reconfigure reaches + // m_h264.m_rateControlLayersInfoH264 -- which is the struct + // CodecHandleRateControlCmd above chains onto the next command. + virtual void RefreshCodecRateControlParameters() override; + virtual void GetResolvedQpClampForTest(uint32_t* pUseMinQp, + int32_t* pMinQpI, + uint32_t* pUseMaxQp, + int32_t* pMaxQpI) const override; private: diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.cpp index e568c2e0..ea377d36 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.cpp @@ -15,8 +15,27 @@ */ #include "VkVideoEncoder/VkVideoEncoderH265.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include "VkVideoCore/VulkanVideoCapabilities.h" +namespace { + +// Whether |dpbIndex| already occupies one of the entries |slots|[|first|, +// |end|) holds. Both reference-list walks admit a slot through this, so a +// picture that L0 and L1 both name is bound once. +bool SlotAlreadyBound(const VkVideoReferenceSlotInfoKHR* slots, + uint32_t first, uint32_t end, uint8_t dpbIndex) +{ + for (uint32_t bound = first; bound < end; bound++) { + if (slots[bound].slotIndex == (int32_t)dpbIndex) { + return true; + } + } + return false; +} + +} // namespace + VkResult CreateVideoEncoderH265(const VulkanDeviceContext* vkDevCtx, VkSharedBaseObj& encoderConfig, VkSharedBaseObj& encoder) @@ -48,7 +67,7 @@ VkResult VkVideoEncoderH265::InitEncoderCodec(VkSharedBaseObj& en VkResult result = InitEncoder(encoderConfig); if (result != VK_SUCCESS) { - fprintf(stderr, "\nERROR: InitEncoder() failed with ret(%d)\n", result); + VkEncPrintfErr("\nERROR: InitEncoder() failed with ret(%d)\n", result); return result; } @@ -56,10 +75,16 @@ VkResult VkVideoEncoderH265::InitEncoderCodec(VkSharedBaseObj& en m_dpb.DpbSequenceStart(m_maxDpbPicturesCount, (m_encoderConfig->numRefL0 > 0) || (m_encoderConfig->numRefL1 > 0)); if (m_encoderConfig->verbose) { - std::cout << ", numRefL0: " << (uint32_t)m_encoderConfig->numRefL0 + VkEncOut() << ", numRefL0: " << (uint32_t)m_encoderConfig->numRefL0 << ", numRefL1: " << (uint32_t)m_encoderConfig->numRefL1 << std::endl; } + // The device QP window; see the H.264 counterpart. InitEncoder above + // is what runs EncoderConfigH265::InitDeviceCapabilities, so the + // capabilities are populated by here. + m_deviceQpWindowMin = m_encoderConfig->h265EncodeCapabilities.minQp; + m_deviceQpWindowMax = m_encoderConfig->h265EncodeCapabilities.maxQp; + m_encoderConfig->GetRateControlParameters(&m_rateControlInfo, m_rateControlLayersInfo, &m_rateControlInfoH265, m_rateControlLayersInfoH265); m_encoderConfig->InitParamameters(&m_vps, &m_sps, &m_pps, @@ -100,14 +125,14 @@ VkResult VkVideoEncoderH265::InitEncoderCodec(VkSharedBaseObj& en nullptr, &sessionParameters); if(result != VK_SUCCESS) { - fprintf(stderr, "\nEncodeFrame Error: Failed to get create video session parameters.\n"); + VkEncPrintfErr("\nEncodeFrame Error: Failed to get create video session parameters.\n"); return result; } result = VulkanVideoSessionParameters::Create(m_vkDevCtx, m_videoSession, sessionParameters, m_videoSessionParameters); if(result != VK_SUCCESS) { - fprintf(stderr, "\nEncodeFrame Error: Failed to get create video session object.\n"); + VkEncPrintfErr("\nEncodeFrame Error: Failed to get create video session object.\n"); return result; } @@ -256,6 +281,23 @@ VkResult VkVideoEncoderH265::ProcessDpb(VkSharedBaseObj& } } + // L0 AND L1 ARE LISTS; referenceSlotsInfo[] IS A SET. The same picture may + // hold a position in both reference lists, and on a B frame whose DPB + // carries a single reference picture it always does: that one picture is + // L0[0] and L1[0] alike. referenceSlotsInfo[] is not a reference list -- + // it is the set of DPB slots the recorded commands BIND -- and Vulkan + // requires each picture resource named in it to be unique + // (VUID-VkVideoBeginCodingInfoKHR-pPictureResource-07238, + // VUID-vkCmdEncodeVideoKHR-pPictureResource-08220) and each DPB frame to + // be used at most once across it and the setup slot + // (VUID-vkCmdEncodeVideoKHR-dpbFrameUseCount-08221). So both lists are + // walked and each slot is admitted at most once. + // + // This does not touch the Std reference lists. Those name DPB slots by + // index, carry their own ordering, and a slot appearing in both of them + // is what the bitstream describes. + const uint32_t firstReferenceSlot = numReferenceSlots; + if ((encodeFrameInfo->gopPosition.pictureType == VkVideoGopStructure::FRAME_TYPE_P) || (encodeFrameInfo->gopPosition.pictureType == VkVideoGopStructure::FRAME_TYPE_B)) { @@ -263,6 +305,16 @@ VkResult VkVideoEncoderH265::ProcessDpb(VkSharedBaseObj& uint8_t dpbIndex = pFrameInfo->stdReferenceListsInfo.RefPicList0[i]; + // The scan spans only the entries these two loops filled. + // referenceSlotsInfo[0] holds the setup slot, whose slotIndex is + // the slot the CURRENT picture reconstructs into and is replaced + // by -1 once both loops have run; it is not a reference and is + // not a candidate for a duplicate. + if (SlotAlreadyBound(pFrameInfo->referenceSlotsInfo, firstReferenceSlot, + numReferenceSlots, dpbIndex)) { + continue; + } + bool refPicAvailable = m_dpb.GetRefPicture(dpbIndex, pFrameInfo->dpbImageResources[numReferenceSlots]); assert(refPicAvailable); if (!refPicAvailable) { @@ -295,12 +347,16 @@ VkResult VkVideoEncoderH265::ProcessDpb(VkSharedBaseObj& } pFrameInfo->numDpbImageResources = numReferenceSlots; - // TODO: iterate over L1 when coding B-frames if (encodeFrameInfo->gopPosition.pictureType == VkVideoGopStructure::FRAME_TYPE_B) { for (uint32_t i = 0; i <= pFrameInfo->stdReferenceListsInfo.num_ref_idx_l1_active_minus1; i++) { uint8_t dpbIndex = pFrameInfo->stdReferenceListsInfo.RefPicList1[i]; + if (SlotAlreadyBound(pFrameInfo->referenceSlotsInfo, firstReferenceSlot, + numReferenceSlots, dpbIndex)) { + continue; + } + bool refPicAvailable = m_dpb.GetRefPicture(dpbIndex, pFrameInfo->dpbImageResources[numReferenceSlots]); assert(refPicAvailable); if (!refPicAvailable) { @@ -394,6 +450,51 @@ VkResult VkVideoEncoderH265::EncodeVideoSessionParameters(VkSharedBaseObjbitstreamHeaderBufferSize = bufferSize; + // HDR10 STATIC METADATA, appended to the parameter sets the driver just + // wrote. + // + // HERE, and not on the per-frame path, for two reasons. The access-unit + // order a decoder requires is VPS, SPS, PPS, prefix SEI, slice -- so the + // bytes belong immediately after what this function produced. And this + // function runs for EVERY IDR (see the note in + // VkVideoEncoder::EncodeFrame about bitstreamHeaderBufferSize), so the + // colour volume repeats at every random-access point instead of once at + // the head of the stream where a seek or a mid-stream join would miss it. + // + // Vulkan Video has no std structure for either payload and + // GetEncodedVideoSessionParametersKHR writes parameter sets only, so the + // NAL is built by hand -- see VkVideoEncoderHdrMetadata.cpp. + if (m_encoderConfig->hdrMetadata.Any()) { + bool truncated = false; + const size_t used = encodeFrameInfo->bitstreamHeaderOffset + + encodeFrameInfo->bitstreamHeaderBufferSize; + const size_t seiBytes = VkEncBuildH265HdrSeiNal( + m_encoderConfig->hdrMetadata, + encodeFrameInfo->bitstreamHeaderBuffer + used, + sizeof(encodeFrameInfo->bitstreamHeaderBuffer) - used, + &truncated); + if (truncated) { + // FATAL. A caller that asked for HDR10 and got a stream without + // it has no way to notice: every counter, every completion edge + // and every byte count is identical. Refusing is the only signal + // this failure has. + VkEncPrintfErr("\nEncodeVideoSessionParameters Error: the HDR10 SEI does " + "not fit in the %zu-byte non-VCL header buffer after %zu " + "bytes of parameter sets.\n", + sizeof(encodeFrameInfo->bitstreamHeaderBuffer), used); + return VK_ERROR_OUT_OF_HOST_MEMORY; + } + // AND THE RATE CONTROLLER IS TOLD. bitstreamHeaderBufferSize is + // what VkVideoEncoder::EncodeFrameCommon reserves via + // dstBufferOffset and reports as precedingExternallyEncodedBytes, + // and it runs AFTER this function (the codec arm fills the buffer + // first). So growing the count here debits the SEI's bytes from the + // IDR's frame budget exactly as the parameter sets' bytes already + // were -- rather than leaving the RC to overshoot by another ~47 + // bytes on every IDR. + encodeFrameInfo->bitstreamHeaderBufferSize += seiBytes; + } + return result; } @@ -423,7 +524,7 @@ VkResult VkVideoEncoderH265::EncodeFrame(VkSharedBaseObj DumpStateInfo("input", 1, encodeFrameInfo); if (encodeFrameInfo->lastFrame) { - std::cout << "#### It is the last frame: " << encodeFrameInfo->frameInputOrderNum + VkEncOut() << "#### It is the last frame: " << encodeFrameInfo->frameInputOrderNum << " of type " << VkVideoGopStructure::GetFrameTypeName(encodeFrameInfo->gopPosition.pictureType) << " ###" << std::endl << std::flush; @@ -447,8 +548,15 @@ VkResult VkVideoEncoderH265::EncodeFrame(VkSharedBaseObj VkResult result = VK_SUCCESS; + // In capture mode (disableFileOutput -- the Chromium in-memory + // bitstream path) EVERY IDR chunk must be independently decodable: the + // VEA hands keyframe chunks to consumers that expect in-band VPS/SPS/PPS + // on each keyframe, including mid-stream forced IDRs. The file-based + // sample keeps the original headers-once-at-stream-start behavior. if ((encodeFrameInfo->gopPosition.pictureType == VkVideoGopStructure::FRAME_TYPE_IDR) && - (encodeFrameInfo->frameEncodeInputOrderNum == 0 /*|| pEncodeConfigH265->repeatSPSPPS || m_bReconfigForcedIDR*/)) { + ((encodeFrameInfo->frameEncodeInputOrderNum == 0) || + (m_encoderConfig->disableFileOutput != 0) + /*|| pEncodeConfigH265->repeatSPSPPS || m_bReconfigForcedIDR*/)) { result = EncodeVideoSessionParameters(encodeFrameInfo); if (result != VK_SUCCESS ) { @@ -525,13 +633,13 @@ VkResult VkVideoEncoderH265::EncodeFrame(VkSharedBaseObj pFrameInfo->naluSliceSegmentInfo[i].constantQp = constantQp; } if (getenv("VKENC_DEBUG_PSNR")) { - fprintf(stderr, "[QPDBG] picType=%d constantQp=%d (qpI=%d qpP=%d qpB=%d) rcMode=%d\n", + VkEncPrintfErr("[QPDBG] picType=%d constantQp=%d (qpI=%d qpP=%d qpB=%d) rcMode=%d\n", (int)encodeFrameInfo->gopPosition.pictureType, constantQp, encodeFrameInfo->constQp.qpIntra, encodeFrameInfo->constQp.qpInterP, encodeFrameInfo->constQp.qpInterB, (int)m_rateControlInfo.rateControlMode); } } else if (getenv("VKENC_DEBUG_PSNR")) { - fprintf(stderr, "[QPDBG] rcMode=%d NOT DISABLED (picType=%d) -> QP not forced\n", + VkEncPrintfErr("[QPDBG] rcMode=%d NOT DISABLED (picType=%d) -> QP not forced\n", (int)m_rateControlInfo.rateControlMode, (int)encodeFrameInfo->gopPosition.pictureType); } @@ -551,6 +659,48 @@ VkResult VkVideoEncoderH265::EncodeFrame(VkSharedBaseObj return result; } +void VkVideoEncoderH265::RefreshCodecRateControlParameters() +{ + // Through the base config pointer, for the reason spelled out on the + // H.264 counterpart: that is the pointer + // ApplyPendingRateControlUpdate just wrote the clamp on. + if (!VkVideoEncoder::m_encoderConfig) { + return; + } + EncoderConfigH265* config = + VkVideoEncoder::m_encoderConfig->GetEncoderConfigh265(); + if (config == nullptr) { + return; + } + // Reset to the codec-init state first; see the H.264 counterpart. The + // fill raises useMinQp/useMaxQp and never lowers them, so re-invoking + // it in place could not clear a clamp that had once been set. + for (uint32_t layerIndx = 0; + layerIndx < ARRAYSIZE(m_rateControlLayersInfoH265); + layerIndx++) { + m_rateControlLayersInfoH265[layerIndx] = + VkVideoEncodeH265RateControlLayerInfoKHR{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_RATE_CONTROL_LAYER_INFO_KHR}; + } + config->GetRateControlParameters(&m_rateControlInfo, + m_rateControlLayersInfo, + &m_rateControlInfoH265, + m_rateControlLayersInfoH265); +} + +void VkVideoEncoderH265::GetResolvedQpClampForTest(uint32_t* pUseMinQp, + int32_t* pMinQpI, + uint32_t* pUseMaxQp, + int32_t* pMaxQpI) const +{ + const VkVideoEncodeH265RateControlLayerInfoKHR& layer = + m_rateControlLayersInfoH265[0]; + *pUseMinQp = (layer.useMinQp == VK_TRUE) ? 1u : 0u; + *pMinQpI = layer.minQp.qpI; + *pUseMaxQp = (layer.useMaxQp == VK_TRUE) ? 1u : 0u; + *pMaxQpI = layer.maxQp.qpI; +} + VkResult VkVideoEncoderH265::CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo) { VkVideoEncodeFrameInfoH265* pFrameInfo = GetEncodeFrameInfoH265(encodeFrameInfo); diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.h index 92cd7d4e..c8c986db 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderH265.h @@ -115,13 +115,18 @@ class VkVideoEncoderH265 : public VkVideoEncoder { , m_dpb{} { } - virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig); - virtual VkResult InitRateControl(VkCommandBuffer cmdBuf, uint32_t qp); + virtual VkResult InitEncoderCodec(VkSharedBaseObj& encoderConfig) override; + virtual VkResult InitRateControl(VkCommandBuffer cmdBuf, uint32_t qp) override; + // DECLARES rather than overrides -- there is no base-class + // EncodeVideoSessionParameters -- so it is the one virtual here that + // correctly carries no 'override', and marking it would not compile. virtual VkResult EncodeVideoSessionParameters(VkSharedBaseObj& encodeFrameInfo); virtual VkResult ProcessDpb(VkSharedBaseObj& encodeFrameInfo, - uint32_t frameIdx, uint32_t ofTotalFrames); - virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes); - virtual bool GetAvailablePoolNode(VkSharedBaseObj& encodeFrameInfo) + uint32_t frameIdx, + uint32_t ofTotalFrames) override; + virtual VkResult CreateFrameInfoBuffersQueue(uint32_t numPoolNodes) override; + virtual bool GetAvailablePoolNode( + VkSharedBaseObj& encodeFrameInfo) override { VkSharedBaseObj encodeFrameInfoH265; bool success = m_frameInfoBuffersQueue->GetAvailablePoolNode(encodeFrameInfoH265); @@ -142,8 +147,18 @@ class VkVideoEncoderH265 : public VkVideoEncoder { } // Must be called from VkVideoEncoder::EncodeFrameCommon only - virtual VkResult EncodeFrame(VkSharedBaseObj& encodeFrameInfo); - virtual VkResult CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo); + virtual VkResult EncodeFrame(VkSharedBaseObj& encodeFrameInfo) override; + virtual VkResult CodecHandleRateControlCmd(VkSharedBaseObj& encodeFrameInfo) override; + + // The H.265 arm of the mid-stream rate-control refresh. See the H.264 + // counterpart: the same fill InitEncoderCodec runs once, re-run so a + // QP clamp changed by Reconfigure reaches the struct + // CodecHandleRateControlCmd chains onto the next command. + virtual void RefreshCodecRateControlParameters() override; + virtual void GetResolvedQpClampForTest(uint32_t* pUseMinQp, + int32_t* pMinQpI, + uint32_t* pUseMaxQp, + int32_t* pMaxQpI) const override; private: VkVideoEncodeFrameInfoH265* GetEncodeFrameInfoH265(VkSharedBaseObj& encodeFrameInfo) { diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderHdrMetadata.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderHdrMetadata.cpp new file mode 100644 index 00000000..004c590b --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderHdrMetadata.cpp @@ -0,0 +1,234 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "VkVideoEncoder/VkVideoEncoderHdrMetadata.h" + +#include +#include + +namespace { + +void PutU16(std::vector& v, uint16_t x) +{ + v.push_back((uint8_t)(x >> 8)); + v.push_back((uint8_t)(x & 0xFF)); +} + +void PutU32(std::vector& v, uint32_t x) +{ + v.push_back((uint8_t)(x >> 24)); + v.push_back((uint8_t)((x >> 16) & 0xFF)); + v.push_back((uint8_t)((x >> 8) & 0xFF)); + v.push_back((uint8_t)(x & 0xFF)); +} + +// AV1 leb128(), 7 payload bits per byte, low group first, continuation in +// bit 7. Every length here is far below 128 so this emits one byte in +// practice; it is written in full because a hand-rolled "just one byte" is +// how a 128-byte payload silently becomes a corrupt stream. +void PutLeb128(std::vector& v, uint64_t value) +{ + do { + uint8_t byte = (uint8_t)(value & 0x7F); + value >>= 7; + if (value != 0) { + byte |= 0x80; + } + v.push_back(byte); + } while (value != 0); +} + +// ST 2086 chromaticity (x50000) -> AV1 0.16 fixed point (x65536). +// Rounded to nearest; clamped because coordinate 1.0 is 50000 in ST 2086 and +// 65536 in 0.16, and 65536 does not fit the f(16) field. Real primaries are +// well under 0.8 so the clamp is a guard, not a path. +uint16_t St2086ChromaToAv1(uint16_t v) +{ + const uint64_t scaled = (((uint64_t)v * 65536ull) + 25000ull) / 50000ull; + return (scaled > 0xFFFFull) ? (uint16_t)0xFFFF : (uint16_t)scaled; +} + +// ST 2086 luminance (x10000 cd/m^2) -> AV1 luminance_max, 24.8 fixed point. +uint32_t St2086LumaMaxToAv1(uint32_t v) +{ + const uint64_t scaled = (((uint64_t)v * 256ull) + 5000ull) / 10000ull; + return (scaled > 0xFFFFFFFFull) ? 0xFFFFFFFFu : (uint32_t)scaled; +} + +// ST 2086 luminance (x10000 cd/m^2) -> AV1 luminance_min, 18.14 fixed point. +uint32_t St2086LumaMinToAv1(uint32_t v) +{ + const uint64_t scaled = (((uint64_t)v * 16384ull) + 5000ull) / 10000ull; + return (scaled > 0xFFFFFFFFull) ? 0xFFFFFFFFu : (uint32_t)scaled; +} + +size_t Emit(const std::vector& bytes, uint8_t* out, size_t capacity, + bool* outTruncated) +{ + if (outTruncated != nullptr) { + *outTruncated = false; + } + if (bytes.empty()) { + return 0; + } + if (bytes.size() > capacity) { + if (outTruncated != nullptr) { + *outTruncated = true; + } + return 0; + } + std::memcpy(out, bytes.data(), bytes.size()); + return bytes.size(); +} + +} // namespace + +size_t VkEncBuildH265HdrSeiNal(const EncoderHdrStaticMetadata& metadata, + uint8_t* out, size_t capacity, + bool* outTruncated) +{ + if (outTruncated != nullptr) { + *outTruncated = false; + } + if (!metadata.Any() || (out == nullptr)) { + return 0; + } + + // nal_unit_header(): forbidden_zero_bit 0, nal_unit_type 39 + // (PREFIX_SEI_NUT), nuh_layer_id 0, nuh_temporal_id_plus1 1. The two + // bytes are part of the emulation-prevention scan window, so they go into + // the same buffer rather than being written around it. + std::vector nal; + nal.push_back((uint8_t)(39 << 1)); // 0x4E + nal.push_back(0x01); + + // sei_message() is { payloadType, payloadSize, payload }, each of the two + // sizes ff-coded. Both types (137, 144) and both sizes (24, 4) are below + // 255, so each is one byte -- asserted by construction here rather than + // by a chained-0xFF loop that could never run. + if (metadata.masteringDisplayPresent != 0) { + nal.push_back(137); // mastering_display_colour_volume + nal.push_back(24); + // D.3.28 interleaves the pairs: (x[c], y[c]) for c = 0,1,2, i.e. + // green, blue, red -- NOT all three x then all three y. + for (int c = 0; c < 3; c++) { + PutU16(nal, metadata.displayPrimaryX[c]); + PutU16(nal, metadata.displayPrimaryY[c]); + } + PutU16(nal, metadata.whitePointX); + PutU16(nal, metadata.whitePointY); + PutU32(nal, metadata.maxDisplayMasteringLuminance); + PutU32(nal, metadata.minDisplayMasteringLuminance); + } + if (metadata.contentLightLevelPresent != 0) { + nal.push_back(144); // content_light_level_info + nal.push_back(4); + PutU16(nal, metadata.maxContentLightLevel); + PutU16(nal, metadata.maxFrameAverageLightLevel); + } + nal.push_back(0x80); // rbsp_trailing_bits() + + // Byte stream NAL unit: a 4-byte start code -- the same length the driver + // writes ahead of VPS/SPS/PPS in this buffer -- then the NAL with + // emulation_prevention_three_byte inserted wherever two zero bytes are + // followed by 0x00..0x03. min_display_mastering_luminance is very often + // a small value like 1 (0x00000001), so this is a live path and not a + // formality. + std::vector stream; + stream.push_back(0x00); + stream.push_back(0x00); + stream.push_back(0x00); + stream.push_back(0x01); + int zeroRun = 0; + for (size_t i = 0; i < nal.size(); i++) { + const uint8_t b = nal[i]; + if ((zeroRun >= 2) && (b <= 0x03)) { + stream.push_back(0x03); + zeroRun = 0; + } + stream.push_back(b); + zeroRun = (b == 0x00) ? (zeroRun + 1) : 0; + } + + return Emit(stream, out, capacity, outTruncated); +} + +size_t VkEncBuildAv1HdrMetadataObus(const EncoderHdrStaticMetadata& metadata, + uint8_t* out, size_t capacity, + bool* outTruncated) +{ + if (outTruncated != nullptr) { + *outTruncated = false; + } + if (!metadata.Any() || (out == nullptr)) { + return 0; + } + + // obu_header(): obu_forbidden_bit 0, obu_type OBU_METADATA (5), + // obu_extension_flag 0, obu_has_size_field 1, obu_reserved_1bit 0 + // 0 0101 0 1 0 -> 0x2A + // The size field is set because these OBUs are appended to a stream whose + // other OBUs carry one; a length-delimited container is not available on + // the capture path this library takes. + const uint8_t kObuMetadataHeader = 0x2A; + + std::vector obus; + + if (metadata.masteringDisplayPresent != 0) { + std::vector payload; + PutLeb128(payload, 2); // METADATA_TYPE_HDR_MDCV + // AV1 ORDERS THE PRIMARIES RED, GREEN, BLUE -- NOT ST 2086's GREEN, + // BLUE, RED. This is the sharpest trap in the file: the two payloads + // carry the same three numbers and permute them differently, so a + // straight copy produces a stream that parses, validates, and names + // green as red. + // + static const int kSt2086IndexForAv1[3] = {2, 0, 1}; // R, G, B + for (int i = 0; i < 3; i++) { + const int c = kSt2086IndexForAv1[i]; + PutU16(payload, St2086ChromaToAv1(metadata.displayPrimaryX[c])); + PutU16(payload, St2086ChromaToAv1(metadata.displayPrimaryY[c])); + } + PutU16(payload, St2086ChromaToAv1(metadata.whitePointX)); + PutU16(payload, St2086ChromaToAv1(metadata.whitePointY)); + PutU32(payload, + St2086LumaMaxToAv1(metadata.maxDisplayMasteringLuminance)); + PutU32(payload, + St2086LumaMinToAv1(metadata.minDisplayMasteringLuminance)); + // trailing_bits(): the payload is byte-aligned, so this is one byte + // with the stop bit set. It is REQUIRED -- metadata_obu() ends with + // trailing_bits() whenever obu_has_size_field is 1. + payload.push_back(0x80); + + obus.push_back(kObuMetadataHeader); + PutLeb128(obus, payload.size()); + obus.insert(obus.end(), payload.begin(), payload.end()); + } + + if (metadata.contentLightLevelPresent != 0) { + std::vector payload; + PutLeb128(payload, 1); // METADATA_TYPE_HDR_CLL + PutU16(payload, metadata.maxContentLightLevel); + PutU16(payload, metadata.maxFrameAverageLightLevel); + payload.push_back(0x80); + + obus.push_back(kObuMetadataHeader); + PutLeb128(obus, payload.size()); + obus.insert(obus.end(), payload.begin(), payload.end()); + } + + return Emit(obus, out, capacity, outTruncated); +} diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderHdrMetadata.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderHdrMetadata.h new file mode 100644 index 00000000..d44681a5 --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderHdrMetadata.h @@ -0,0 +1,110 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef VKVIDEOENCODER_VKVIDEOENCODERHDRMETADATA_H_ +#define VKVIDEOENCODER_VKVIDEOENCODERHDRMETADATA_H_ + +#include +#include + +//============================================================================= +// HDR10 STATIC METADATA -- the two payloads, and the two bitstream shapes. +// +// This is the only place in the encoder that writes bits by hand, and that +// is deliberate: neither payload has a Vulkan Video std structure. There is +// no StdVideoEncodeH265SeiMastering..., no AV1 metadata OBU entry point, and +// GetEncodedVideoSessionParametersKHR emits parameter sets only. Every +// mastering-display / content-light symbol elsewhere in this tree is +// decoder-side or lives in scripts/vk.xml. So the bytes are built here and +// appended to the non-VCL header the driver produced. +// +// UNITS. The struct below carries SMPTE ST 2086 units, which is what the +// H.265 SEI codes directly and what a producer already has (Chromium's +// gfx::HdrMetadataSmpteSt2086, ffmpeg's AVMasteringDisplayMetadata): +// +// chromaticity increments of 0.00002 (value = coordinate * 50000) +// luminance increments of 0.0001 cd/m^2 (value = nits * 10000) +// MaxCLL/MaxFALL cd/m^2, integral +// +// AV1 DOES NOT USE THOSE UNITS, and that mismatch is the sharpest trap in +// this file. metadata_hdr_mdcv() codes +// +// primary/white-point chromaticity 0.16 fixed point (value = coord * 65536) +// luminance_max 24.8 fixed point (value = nits * 256) +// luminance_min 18.14 fixed point (value = nits * 16384) +// +// so the AV1 builder converts and the H.265 builder does not. +// +// AV1 ALSO PERMUTES THE PRIMARIES: metadata_hdr_mdcv() indexes them RED, +// GREEN, BLUE where ST 2086 (and therefore the H.265 SEI) indexes them GREEN, +// BLUE, RED. Copying the array straight across produces a stream that parses +// cleanly and calls green red. +// +// Golden vectors pinning both payloads byte for byte live in +// test/encoder-ext-filter. +//============================================================================= + +struct EncoderHdrStaticMetadata { + // Each payload is independently present. A caller that knows MaxCLL and + // nothing about the mastering display gets exactly one SEI message and + // exactly one metadata OBU -- rather than a mastering display of all + // zeros, which is a positive claim that the display is black. + uint32_t masteringDisplayPresent = 0; + uint32_t contentLightLevelPresent = 0; + + // ST 2086 ORDER: GREEN, BLUE, RED. That is the H.265 SEI's own order + // (D.3.28: "c equal to 0, 1 and 2 corresponding to the green, blue and + // red colour primary"), so the H.265 builder writes this array straight + // through. The AV1 builder does NOT -- metadata_hdr_mdcv() indexes red, + // green, blue, and it permutes. + uint16_t displayPrimaryX[3] = {0, 0, 0}; + uint16_t displayPrimaryY[3] = {0, 0, 0}; + uint16_t whitePointX = 0; + uint16_t whitePointY = 0; + uint32_t maxDisplayMasteringLuminance = 0; + uint32_t minDisplayMasteringLuminance = 0; + + uint16_t maxContentLightLevel = 0; + uint16_t maxFrameAverageLightLevel = 0; + + bool Any() const { + return (masteringDisplayPresent != 0) || + (contentLightLevelPresent != 0); + } +}; + +// Build the H.265 PREFIX SEI NAL unit carrying mastering_display_colour_volume +// (payloadType 137) and/or content_light_level_info (payloadType 144), with a +// 4-byte start code and emulation-prevention bytes inserted, ready to be +// appended to the VPS/SPS/PPS the driver wrote. +// +// Returns the number of bytes written, or 0 if there was nothing to write. +// Returns 0 AND SETS *outTruncated if the payload did not fit in |capacity|: +// the caller must treat that as an error rather than encode without it. A +// dropped colour volume is invisible in every other observable the encoder +// has, which is exactly how this class of defect survives. +size_t VkEncBuildH265HdrSeiNal(const EncoderHdrStaticMetadata& metadata, + uint8_t* out, size_t capacity, + bool* outTruncated); + +// Build the AV1 metadata OBU(s): METADATA_TYPE_HDR_MDCV (2) and/or +// METADATA_TYPE_HDR_CLL (1), each a complete OBU with obu_has_size_field set, +// ready to be appended to the sequence header OBU. Same return contract. +size_t VkEncBuildAv1HdrMetadataObus(const EncoderHdrStaticMetadata& metadata, + uint8_t* out, size_t capacity, + bool* outTruncated); + +#endif // VKVIDEOENCODER_VKVIDEOENCODERHDRMETADATA_H_ diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.cpp new file mode 100644 index 00000000..48fecc6f --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.cpp @@ -0,0 +1,89 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Linux implementation of the encoder core OS seam declared in + * VkVideoEncoderOsAdapterLinux.h. + * + * This translation unit is platform-specific by construction. Another + * platform is served by another translation unit implementing the same + * declarations, selected by the build; it is never served by an #else arm in + * this one. + */ +#include "VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" + +#if !defined(__linux__) +#error "VkVideoEncoderOsAdapterLinux.cpp implements the encoder OS seam for Linux only. Port the seam in a sibling translation unit and select it in the build." +#endif + +#include + +#include + +#include "VkCodecUtils/VkDrmFormatModifierUtils.h" + +namespace vkenc { + +void OsSetCurrentThreadName(const char* name) +{ + // The result is checked and reported: a thread that silently failed to be + // named is indistinguishable in a crash dump from one that was never + // named at all, and the report is what tells those two apart. + const int rc = pthread_setname_np(pthread_self(), name); + if (rc != 0) { + VkEncPrintfErr("[VkVideoEncoder] could not name thread \"%s\" (errno %d); " + "it will appear unnamed in crash dumps\n", + name, rc); + } +} + +VkResult OsSelectDrmFormatModifier(const VulkanDeviceContext* vkDevCtx, + VkFormat format, + int32_t requestedModifierIndex, + uint64_t* pSelectedModifier) +{ + VkDrmFormatModifierUtils drmUtils(vkDevCtx); + + const VkFormatFeatureFlags required = + VK_FORMAT_FEATURE_VIDEO_ENCODE_INPUT_BIT_KHR | VK_FORMAT_FEATURE_TRANSFER_DST_BIT; + drmUtils.DumpAvailableModifiers(format, required); + + const int32_t idx = requestedModifierIndex; + uint64_t selected = drmUtils.SelectModifier( + format, required, idx, + VkDrmFormatModifierUtils::BlockHeightPref::PreferSmallest, + VkDrmFormatModifierUtils::CompressionPref::PreferUncompressed); + + if (selected == 0 && idx >= 0) { + // Explicit index was requested but no suitable modifier found + VkEncPrintfErr("DRM modifier index %d: no suitable modifier found\n", idx); + return VK_ERROR_INITIALIZATION_FAILED; + } + if (selected == 0) { + VkEncPrintfErr("No non-linear DRM modifiers support VIDEO_ENCODE_SRC + TRANSFER_DST\n"); + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + *pSelectedModifier = selected; + VkEncPrintfOut("\n=== Selected DRM format modifier ===\n"); + VkDrmFormatModifierUtils::PrintModifierInfo(selected); + VkEncPrintfOut("\n"); + + return VK_SUCCESS; +} + +} // namespace vkenc diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h new file mode 100644 index 00000000..29c879e1 --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h @@ -0,0 +1,61 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * OS adapter for the encoder core. + * + * The encoder core carries no OS-specific code -- no , no + * , no poll/epoll. Everything OS-conditional it needs is declared + * here in platform-neutral terms, and is the core-layer counterpart of the + * ext layer's vulkan_video_encoder_os_event_linux seam. + * + * These declarations are portable; the implementations are not. Exactly one + * implementation translation unit is compiled per platform and the build + * selects it; the Linux one is VkVideoEncoderOsAdapterLinux.cpp. A platform + * with no implementation stops the build at configure time, so an unported + * platform is a build error and never a run-time no-op. + */ +#ifndef VK_VIDEO_ENCODER_OS_ADAPTER_LINUX_H_ +#define VK_VIDEO_ENCODER_OS_ADAPTER_LINUX_H_ + +#include + +#include + +class VulkanDeviceContext; + +namespace vkenc { + +// Give the calling thread a name the OS will report. Linux caps this at +// 15 characters plus NUL and fails silently past it, so keep the names +// short. A named thread is what lets a stack inside the encoder be +// attributed to this library in a crash dump. +void OsSetCurrentThreadName(const char* name); + +// Select a DRM format modifier suitable for VIDEO_ENCODE_SRC + TRANSFER_DST +// and write it to *pSelectedModifier. requestedModifierIndex >= 0 demands +// that entry of the reported list and fails with +// VK_ERROR_INITIALIZATION_FAILED if it is not suitable; a negative index +// lets the utility choose, failing with VK_ERROR_FORMAT_NOT_SUPPORTED when +// nothing qualifies. +VkResult OsSelectDrmFormatModifier(const VulkanDeviceContext* vkDevCtx, + VkFormat format, + int32_t requestedModifierIndex, + uint64_t* pSelectedModifier); + +} // namespace vkenc + +#endif // VK_VIDEO_ENCODER_OS_ADAPTER_LINUX_H_ diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterWindows.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterWindows.cpp new file mode 100644 index 00000000..fc1da72e --- /dev/null +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderOsAdapterWindows.cpp @@ -0,0 +1,85 @@ +/* + * Copyright 2025 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "VkVideoEncoder/VkVideoEncoderOsAdapterLinux.h" +#include "VkCodecUtils/VkEncoderStdioLatch.h" + +#if !defined(_WIN32) +#error "VkVideoEncoderOsAdapterWindows.cpp implements the encoder OS seam for Windows only." +#endif + +#include +#include +#include + +namespace vkenc { + +void OsSetCurrentThreadName(const char* name) +{ + // SetThreadDescription takes UTF-16 and is the only naming API that + // survives into a crash dump; the older exception-based trick is visible + // to a debugger only while one is attached. + // + // Resolved at run time rather than linked: it arrives in Windows 10 1607, + // and a hard import would stop the library loading on anything earlier + // for a facility that only decorates a dump. + using SetThreadDescriptionFn = HRESULT(WINAPI*)(HANDLE, PCWSTR); + static const SetThreadDescriptionFn setThreadDescription = []() { + HMODULE kernelBase = ::GetModuleHandleW(L"kernelbase.dll"); + return kernelBase ? reinterpret_cast( + reinterpret_cast(::GetProcAddress( + kernelBase, "SetThreadDescription"))) + : nullptr; + }(); + + if (setThreadDescription == nullptr) { + return; // Pre-1607: the thread stays unnamed, which is not an error. + } + + const int wide = ::MultiByteToWideChar(CP_UTF8, 0, name, -1, nullptr, 0); + if (wide <= 0) { + return; + } + std::wstring wideName(static_cast(wide), L'\0'); + if (::MultiByteToWideChar(CP_UTF8, 0, name, -1, wideName.data(), wide) <= 0) { + return; + } + const HRESULT hr = setThreadDescription(::GetCurrentThread(), wideName.c_str()); + if (FAILED(hr)) { + VkEncPrintfErr("[VkVideoEncoder] could not name thread \"%s\" (hr 0x%08lx); " + "it will appear unnamed in crash dumps\n", + name, static_cast(hr)); + } +} + +VkResult OsSelectDrmFormatModifier(const VulkanDeviceContext* /*vkDevCtx*/, + VkFormat /*format*/, + int32_t /*requestedModifierIndex*/, + uint64_t* pSelectedModifier) +{ + // REFUSED, NOT STUBBED TO ZERO. DRM format modifiers are a Linux/DRM + // concept: there is no Windows equivalent to select, and returning + // "modifier 0" would read as DRM_FORMAT_MOD_LINEAR -- a specific, wrong + // answer that an importer would then act on. A caller that asked for a + // modifier on this platform asked for something that does not exist, and + // is told so. + if (pSelectedModifier != nullptr) { + *pSelectedModifier = 0; + } + return VK_ERROR_FORMAT_NOT_SUPPORTED; +} + +} // namespace vkenc diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.cpp index 54ff8439..6de43ef4 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.cpp @@ -14,9 +14,13 @@ * limitations under the License. */ +#include +#include "VkCodecUtils/VkEncoderStdioLatch.h" #include #include #include +#include +#include #include #include "VkVideoEncoder/VkVideoEncoder.h" #include "VkVideoEncoder/VkVideoEncoderPsnr.h" @@ -40,9 +44,11 @@ VkResult VkVideoEncoderPsnr::Configure(const VulkanDeviceContext* vkDevCtx, uint32_t maxEncodeQueueDepth, VkFormat imageDpbFormat, const VkExtent2D& imageExtent, - uint32_t encodeQueueFamilyIndex) + uint32_t encodeQueueFamilyIndex, + VkFormat imageInFormat) { if (m_vkDevCtx == nullptr) { + m_imageInFormat = (imageInFormat != VK_FORMAT_UNDEFINED) ? imageInFormat : imageDpbFormat; m_vkDevCtx = vkDevCtx; m_encoderConfig = encoderConfig; m_maxEncodeQueueDepth = maxEncodeQueueDepth; @@ -70,7 +76,7 @@ VkResult VkVideoEncoderPsnr::Configure(const VulkanDeviceContext* vkDevCtx, false, true); if (result != VK_SUCCESS) { - fprintf(stderr, "\nVkVideoEncoderPsnr: Failed to Configure psnrReconImagePool.\n"); + VkEncPrintfErr("\nVkVideoEncoderPsnr: Failed to Configure psnrReconImagePool.\n"); m_psnrReconImagePool = nullptr; m_initResult = result; } else { @@ -80,6 +86,41 @@ VkResult VkVideoEncoderPsnr::Configure(const VulkanDeviceContext* vkDevCtx, } } + // ===== MECHANISM-C: encoder-INPUT capture pool ===== + // Independent of IsPsnrMetricsEnabled() on purpose: the PSNR path forces a + // per-frame host wait on the encode fence, which is exactly the kind of + // serialisation that can hide a submission-ordering race. + if ((m_capSrcImagePool == nullptr) && (getenv("VKENC_DEBUG_DUMP_SRC") != nullptr)) { + const char* maxFilesEnv = getenv("VKENC_DEBUG_DUMP_SRC_FILES"); + m_capSrcMaxFiles = (maxFilesEnv != nullptr) ? (uint32_t)atoi(maxFilesEnv) : 0; + const char* strideEnv = getenv("VKENC_DEBUG_DUMP_SRC_STRIDE"); + m_capStride = (strideEnv != nullptr) ? (uint32_t)std::max(1, atoi(strideEnv)) : 8u; + m_capSrcEnabled = true; + VkResult capRes = VulkanVideoImagePool::Create(m_vkDevCtx, m_capSrcImagePool); + if (capRes == VK_SUCCESS) { + capRes = m_capSrcImagePool->Configure(m_vkDevCtx, + m_maxEncodeQueueDepth + 2, + m_imageInFormat, + m_imageExtent, + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, + m_encodeQueueFamilyIndex, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, + nullptr, + VK_IMAGE_ASPECT_COLOR_BIT, + false, + false, + true); + } + if (capRes != VK_SUCCESS) { + VkEncPrintfErr("[CAPSRC] pool configure FAILED (0x%x); capture disabled\n", capRes); + m_capSrcImagePool = nullptr; + } else { + VkEncPrintfErr("[CAPSRC] enabled: fmt=%d %ux%u nodes=%u maxFiles=%u\n", + (int)m_imageInFormat, m_imageExtent.width, m_imageExtent.height, + m_maxEncodeQueueDepth + 2, m_capSrcMaxFiles); + } + } + if ((m_encoderConfig != nullptr) && m_encoderConfig->IsPsnrMetricsEnabled() && (m_psnrReconImagePool == nullptr)) { return m_initResult; } @@ -134,6 +175,341 @@ void VkVideoEncoderPsnr::CaptureInput(void* encodeFrameInfoVoid, const uint8_t* } } + +// ===================== MECHANISM-C: ENCODER-INPUT CAPTURE ===================== +// +// WHAT IT MEASURES. The pixels of encodeFrameInfo->srcEncodeImageResource -- the +// image named as pSrcPictureResource of vkCmdEncodeVideoKHR -- read out of the +// ENCODE command buffer, one command before vkCmdBeginVideoCodingKHR. That places +// the copy in the same submission, on the same queue, behind the same (or absent) +// cross-queue dependency as the encode's own read. A capture that shows zero +// chroma therefore says the encode read zero chroma too; the corruption is at or +// before the encode's input, not inside the encode. +// +// WHY THE BARRIERS CANNOT LAUNDER THE ANSWER. Both barriers below name +// VK_QUEUE_FAMILY_IGNORED and scope their source at ALL_COMMANDS *within this +// command buffer's submission*. A pipeline barrier cannot create a dependency on +// work submitted to a different queue, and cannot substitute for a semaphore +// between submissions. So if the staged copy has not landed when the encode runs, +// it has not landed when this capture runs either. +bool VkVideoEncoderPsnr::CaptureSource(VkCommandBuffer cmdBuf, void* encodeFrameInfoVoid) +{ + if ((encodeFrameInfoVoid == nullptr) || (m_capSrcImagePool == nullptr) || (m_vkDevCtx == nullptr)) { + return false; + } + VkVideoEncoder::VkVideoEncodeFrameInfo& encodeFrameInfo = + *static_cast(encodeFrameInfoVoid); + if (encodeFrameInfo.srcEncodeImageResource == nullptr) { + return false; + } + const bool srcIs2Plane = (m_imageInFormat == VK_FORMAT_G8_B8R8_2PLANE_420_UNORM); + if (!srcIs2Plane) { + // Once per process, and once has to be true rather than likely -- + // see the equivalent claims in the encoder core. Debug capture is + // opt-in, but two sessions that opt in reach this together. + static std::atomic warned{false}; + if (!warned.exchange(true, std::memory_order_relaxed)) { + VkEncPrintfErr("[CAPSRC] input format %d is not 2-plane NV12; capture skipped\n", + (int)m_imageInFormat); + } + return false; + } + if (!m_capSrcImagePool->GetAvailableImage(encodeFrameInfo.psnrFrameData.capSrcImage, + VK_IMAGE_LAYOUT_UNDEFINED)) { + m_capSrcMissCount++; + return false; + } + VkSharedBaseObj srcView; + encodeFrameInfo.srcEncodeImageResource->GetImageView(srcView); + VkSharedBaseObj dstView; + encodeFrameInfo.psnrFrameData.capSrcImage->GetImageView(dstView); + if (!srcView || !dstView) { + encodeFrameInfo.psnrFrameData.capSrcImage = nullptr; + return false; + } + VkImage srcImage = srcView->GetImageResource()->GetImage(); + VkImage dstImage = dstView->GetImageResource()->GetImage(); + + // The layout the staging arm recorded, or the spec-required encode layout when + // the frame was never staged (Path A / RESIDENCY_LOCAL). + const VkImageLayout srcLayout = + (encodeFrameInfo.srcEncodeImageStagedLayout != VK_IMAGE_LAYOUT_MAX_ENUM) + ? encodeFrameInfo.srcEncodeImageStagedLayout + : VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR; + + VkImageMemoryBarrier2KHR bars[2] = {}; + bars[0].sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2_KHR; + bars[0].srcStageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT; + bars[0].srcAccessMask = VK_ACCESS_2_NONE; + bars[0].dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + bars[0].dstAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT; + bars[0].oldLayout = srcLayout; + bars[0].newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + bars[0].srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + bars[0].dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + bars[0].image = srcImage; + bars[0].subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1 }; + bars[1] = bars[0]; + bars[1].srcAccessMask = VK_ACCESS_2_NONE; + bars[1].dstAccessMask = VK_ACCESS_2_TRANSFER_WRITE_BIT; + bars[1].oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + bars[1].newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + bars[1].image = dstImage; + + VkDependencyInfoKHR depInfo = {}; + depInfo.sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO_KHR; + depInfo.imageMemoryBarrierCount = 2; + depInfo.pImageMemoryBarriers = bars; + m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &depInfo); + + const uint32_t w = m_encoderConfig->encodeWidth; + const uint32_t h = m_encoderConfig->encodeHeight; + VkImageCopy r0 = { { VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0, 1 }, { 0, 0, 0 }, { w, h, 1 } }; + m_vkDevCtx->CmdCopyImage(cmdBuf, srcImage, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + dstImage, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &r0); + VkImageCopy r1 = { { VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { (w + 1) / 2, (h + 1) / 2, 1 } }; + m_vkDevCtx->CmdCopyImage(cmdBuf, srcImage, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + dstImage, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &r1); + + // Hand the encode source back in exactly the layout it was handed to us in, + // so the encode's own VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 still holds. + VkImageMemoryBarrier2KHR back = bars[0]; + back.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + back.srcAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT; + back.dstStageMask = VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + back.dstAccessMask = VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; + back.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + back.newLayout = srcLayout; + depInfo.imageMemoryBarrierCount = 1; + depInfo.pImageMemoryBarriers = &back; + m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &depInfo); + return true; +} + +// Records (imported linear image -> host-visible LINEAR image) into the STAGING +// command buffer, immediately after CopyLinearToOptimalImage has read the same +// image. This is the pixels the PRODUCER handed us, before the library's own +// copy can be blamed for them. Pool is configured lazily because the imported +// image's format and extent are not known until the first frame arrives. +bool VkVideoEncoderPsnr::CaptureImported(VkCommandBuffer cmdBuf, void* encodeFrameInfoVoid, + void* linearImageViewVoid) +{ + if (!m_capSrcEnabled || (encodeFrameInfoVoid == nullptr) || + (linearImageViewVoid == nullptr) || (m_vkDevCtx == nullptr)) { + return false; + } + VkVideoEncoder::VkVideoEncodeFrameInfo& encodeFrameInfo = + *static_cast(encodeFrameInfoVoid); + VkImageResourceView* linearView = static_cast(linearImageViewVoid); + const VkSharedBaseObj& srcRes = linearView->GetImageResource(); + const VkImageCreateInfo& ci = srcRes->GetImageCreateInfo(); + + if (m_capImpImagePool == nullptr) { + if (ci.format != VK_FORMAT_G8_B8R8_2PLANE_420_UNORM) { + // As for the capture-source claim above. + static std::atomic warned{false}; + if (!warned.exchange(true, std::memory_order_relaxed)) { + VkEncPrintfErr("[CAPIMP] imported format %d is not 2-plane NV12; capture disabled\n", + (int)ci.format); + } + return false; + } + m_capImpFormat = ci.format; + m_capImpExtent.width = ci.extent.width; + m_capImpExtent.height = ci.extent.height; + VkResult r = VulkanVideoImagePool::Create(m_vkDevCtx, m_capImpImagePool); + if (r == VK_SUCCESS) { + r = m_capImpImagePool->Configure(m_vkDevCtx, m_maxEncodeQueueDepth + 2, + m_capImpFormat, m_capImpExtent, + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT, + m_encodeQueueFamilyIndex, + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT, + nullptr, VK_IMAGE_ASPECT_COLOR_BIT, + false, false, true); + } + if (r != VK_SUCCESS) { + VkEncPrintfErr("[CAPIMP] pool configure FAILED (0x%x); capture disabled\n", r); + m_capImpImagePool = nullptr; + return false; + } + VkEncPrintfErr("[CAPIMP] enabled: fmt=%d %ux%u tiling=%d\n", + (int)m_capImpFormat, m_capImpExtent.width, m_capImpExtent.height, (int)ci.tiling); + } + if (!m_capImpImagePool->GetAvailableImage(encodeFrameInfo.psnrFrameData.capImpImage, + VK_IMAGE_LAYOUT_UNDEFINED)) { + m_capImpMissCount++; + return false; + } + VkSharedBaseObj dstView; + encodeFrameInfo.psnrFrameData.capImpImage->GetImageView(dstView); + if (!dstView) { + encodeFrameInfo.psnrFrameData.capImpImage = nullptr; + return false; + } + VkImage srcImage = srcRes->GetImage(); + VkImage dstImage = dstView->GetImageResource()->GetImage(); + encodeFrameInfo.psnrFrameData.capSeq = m_capSeq.fetch_add(1) + 1; + encodeFrameInfo.psnrFrameData.capImpImageId = (uint64_t)srcImage; + encodeFrameInfo.psnrFrameData.capImpMemId = (uint64_t)srcRes->GetDeviceMemory(); + + // The imported image is already in TRANSFER_SRC_OPTIMAL here -- the staging + // arm put it there and CopyLinearToOptimalImage just read it -- so only the + // destination needs a barrier, plus a TRANSFER->TRANSFER execution dependency + // so this copy is ordered after the library's own. + VkImageMemoryBarrier2KHR bar = {}; + bar.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2_KHR; + bar.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + bar.srcAccessMask = VK_ACCESS_2_NONE; + bar.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + bar.dstAccessMask = VK_ACCESS_2_TRANSFER_WRITE_BIT; + bar.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; + bar.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + bar.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + bar.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + bar.image = dstImage; + bar.subresourceRange = { VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1 }; + VkDependencyInfoKHR dep = {}; + dep.sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO_KHR; + dep.imageMemoryBarrierCount = 1; + dep.pImageMemoryBarriers = &bar; + m_vkDevCtx->CmdPipelineBarrier2KHR(cmdBuf, &dep); + + const uint32_t w = std::min(m_capImpExtent.width, m_encoderConfig->encodeWidth); + const uint32_t h = std::min(m_capImpExtent.height, m_encoderConfig->encodeHeight); + VkImageCopy r0 = { { VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0, 1 }, { 0, 0, 0 }, { w, h, 1 } }; + m_vkDevCtx->CmdCopyImage(cmdBuf, srcImage, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + dstImage, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &r0); + VkImageCopy r1 = { { VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0, 1 }, { 0, 0, 0 }, + { (w + 1) / 2, (h + 1) / 2, 1 } }; + m_vkDevCtx->CmdCopyImage(cmdBuf, srcImage, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, + dstImage, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, 1, &r1); + return true; +} + +// Host-side reader, shared by both captures. Must be called only after the +// ENCODE command buffer's fence has been waited on. +// +// COST DISCIPLINE: v1 of this walked the whole 3.1 MB of HOST_VISIBLE image +// memory byte-wise and dropped the pipeline from ~200 fps to 3.6 fps. Rows are +// now bulk-memcpy'd into cached scratch and only every m_capStride-th row is +// touched, which is ample for "is this plane all zeros". +void VkVideoEncoderPsnr::DumpCapturedNode(VkSharedBaseObj& node, + const char* tag, uint32_t frameIdx, + uint32_t w, uint32_t h, bool writeFile, + uint64_t seq, uint64_t imgId) +{ + if (node == nullptr) { + return; + } + VkSharedBaseObj view; + node->GetImageView(view); + if (!view) { node = nullptr; return; } + const VkSharedBaseObj& res = view->GetImageResource(); + VkDevice device = m_vkDevCtx->getDevice(); + void* mapped = nullptr; + if ((m_vkDevCtx->MapMemory(device, res->GetDeviceMemory(), + res->GetImageDeviceMemoryOffset(), + res->GetImageDeviceMemorySize(), 0, &mapped) != VK_SUCCESS) || + (mapped == nullptr)) { + node = nullptr; + return; + } + const uint8_t* base = static_cast(mapped); + const uint32_t cw = (w + 1) / 2; + const uint32_t ch = (h + 1) / 2; + const uint32_t stride = writeFile ? 1u : m_capStride; + VkImage img = res->GetImage(); + VkImageSubresource sub = {}; + VkSubresourceLayout ly = {}; + VkSubresourceLayout lc = {}; + sub.aspectMask = VK_IMAGE_ASPECT_PLANE_0_BIT; + m_vkDevCtx->GetImageSubresourceLayout(device, img, &sub, &ly); + sub.aspectMask = VK_IMAGE_ASPECT_PLANE_1_BIT; + m_vkDevCtx->GetImageSubresourceLayout(device, img, &sub, &lc); + + m_capScratch.resize((size_t)std::max(w, 2u * cw) + 64); + uint8_t* row = m_capScratch.data(); + + uint64_t sumY = 0; uint64_t nY = 0; + for (uint32_t y = 0; y < h; y += stride) { + memcpy(row, base + ly.offset + ((size_t)y * ly.rowPitch), w); + for (uint32_t x = 0; x < w; x++) { sumY += row[x]; } + nY += w; + } + uint64_t sumU = 0, sumV = 0, nC = 0, zeroUV = 0; + for (uint32_t y = 0; y < ch; y += stride) { + memcpy(row, base + lc.offset + ((size_t)y * lc.rowPitch), (size_t)2 * cw); + for (uint32_t x = 0; x < cw; x++) { + const uint8_t u = row[(2 * x) + 0]; + const uint8_t v = row[(2 * x) + 1]; + sumU += u; sumV += v; + if ((u == 0) && (v == 0)) { zeroUV++; } + } + nC += cw; + } + + if (writeFile) { + char filename[256]; + snprintf(filename, sizeof(filename), "%s_frame_%05u_%ux%u.yuv", tag, frameIdx, w, h); + std::ofstream out(filename, std::ios::binary); + if (out) { + std::vector line(w); + for (uint32_t y = 0; y < h; y++) { + memcpy(line.data(), base + ly.offset + ((size_t)y * ly.rowPitch), w); + out.write(reinterpret_cast(line.data()), w); + } + std::vector up(cw), vp(cw), pair((size_t)2 * cw); + for (uint32_t pl = 0; pl < 2; pl++) { + for (uint32_t y = 0; y < ch; y++) { + memcpy(pair.data(), base + lc.offset + ((size_t)y * lc.rowPitch), (size_t)2 * cw); + for (uint32_t x = 0; x < cw; x++) { up[x] = pair[(2 * x) + pl]; } + out.write(reinterpret_cast(up.data()), cw); + } + } + (void)vp; + } + } + m_vkDevCtx->UnmapMemory(device, res->GetDeviceMemory()); + + VkEncPrintfErr("[%s] seq=%llu gopf=%u img=0x%llx Ymean=%.3f Umean=%.3f Vmean=%.3f zeroUVpct=%.2f\n", + tag, (unsigned long long)seq, frameIdx, (unsigned long long)imgId, + nY ? (double)sumY / (double)nY : -1.0, + nC ? (double)sumU / (double)nC : -1.0, + nC ? (double)sumV / (double)nC : -1.0, + nC ? 100.0 * (double)zeroUV / (double)nC : -1.0); + node = nullptr; +} + +void VkVideoEncoderPsnr::DumpCapturedSource(void* encodeFrameInfoVoid) +{ + VkVideoEncoder::VkVideoEncodeFrameInfo& encodeFrameInfo = + *static_cast(encodeFrameInfoVoid); + const uint32_t inputOrder = (uint32_t)encodeFrameInfo.gopPosition.inputOrder; + const bool writeFile = (m_capSrcFilesWritten < m_capSrcMaxFiles); + if (encodeFrameInfo.psnrFrameData.capImpImage != nullptr) { + DumpCapturedNode(encodeFrameInfo.psnrFrameData.capImpImage, "CAPIMP", inputOrder, + std::min(m_capImpExtent.width, m_encoderConfig->encodeWidth), + std::min(m_capImpExtent.height, m_encoderConfig->encodeHeight), + writeFile, + encodeFrameInfo.psnrFrameData.capSeq, + encodeFrameInfo.psnrFrameData.capImpImageId); + } + if (encodeFrameInfo.psnrFrameData.capSrcImage != nullptr) { + DumpCapturedNode(encodeFrameInfo.psnrFrameData.capSrcImage, "CAPSRC", inputOrder, + m_encoderConfig->encodeWidth, m_encoderConfig->encodeHeight, + writeFile, + encodeFrameInfo.psnrFrameData.capSeq, + encodeFrameInfo.psnrFrameData.capImpMemId); + if (writeFile) { m_capSrcFilesWritten++; } + } + fflush(stderr); +} + bool VkVideoEncoderPsnr::CaptureOutput(VkCommandBuffer cmdBuf, void* encodeFrameInfoVoid) { if ((encodeFrameInfoVoid == nullptr) || (m_psnrReconImagePool == nullptr) || (m_encoderConfig == nullptr)) { @@ -206,10 +582,13 @@ void VkVideoEncoderPsnr::ComputeFramePsnr(void* encodeFrameInfoVoid) return; } VkVideoEncoder::VkVideoEncodeFrameInfo& encodeFrameInfo = *static_cast(encodeFrameInfoVoid); - if (encodeFrameInfo.psnrFrameData.psnrStagingImage == nullptr) { - return; - } - if (encodeFrameInfo.setupImageResource == nullptr) { + // MECHANISM-C runs without the PSNR pool, so a frame carrying only a source + // capture must not be turned away by the PSNR preconditions. + const bool haveSrcCapture = (encodeFrameInfo.psnrFrameData.capSrcImage != nullptr) || + (encodeFrameInfo.psnrFrameData.capImpImage != nullptr); + const bool havePsnrCapture = (encodeFrameInfo.psnrFrameData.psnrStagingImage != nullptr) && + (encodeFrameInfo.setupImageResource != nullptr); + if (!haveSrcCapture && !havePsnrCapture) { return; } const uint32_t width = std::min(m_encoderConfig->encodeWidth, m_encoderConfig->input.width); @@ -244,7 +623,16 @@ void VkVideoEncoderPsnr::ComputeFramePsnr(void* encodeFrameInfoVoid) } VkResult syncResult = encodeFrameInfo.encodeCmdBuffer->SyncHostOnCmdBuffComplete(false, "encoderEncodeFence"); if (syncResult != VK_SUCCESS) { - fprintf(stderr, "\nPSNR: wait on encoder fence failed (0x%x), skipping frame PSNR.\n", syncResult); + VkEncPrintfErr("\nPSNR: wait on encoder fence failed (0x%x), skipping frame PSNR.\n", syncResult); + return; + } + + // MECHANISM-C readback. The encode fence is signalled, so the capture copy + // recorded ahead of the coding scope in this same command buffer is done. + if (haveSrcCapture) { + DumpCapturedSource(encodeFrameInfoVoid); + } + if (!havePsnrCapture) { return; } diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.h index 96fedd6c..40b814af 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoEncoderPsnr.h @@ -17,6 +17,7 @@ #ifndef _VKVIDEOENCODER_VKVIDEOENCODERPSNR_H_ #define _VKVIDEOENCODER_VKVIDEOENCODERPSNR_H_ +#include #include #include "VkCodecUtils/VkVideoRefCountBase.h" #include "VkCodecUtils/VulkanVideoImagePool.h" @@ -38,6 +39,15 @@ class VkVideoEncoderPsnr : public VkVideoRefCountBase { std::vector psnrInputU; std::vector psnrInputV; VkSharedBaseObj psnrStagingImage; + // MECHANISM-C: host-visible LINEAR copy of the ENCODER INPUT image, + // recorded in the encode command buffer just before the coding scope. + VkSharedBaseObj capSrcImage; + // MECHANISM-C part 2: host-visible LINEAR copy of the IMPORTED image + // (the producer's dma-buf), recorded in the STAGING command buffer. + VkSharedBaseObj capImpImage; + uint64_t capSeq = 0; // 1-based, monotonic + uint64_t capImpImageId = 0; // VkImage of the import + uint64_t capImpMemId = 0; // VkDeviceMemory of the import }; static VkResult Create(VkSharedBaseObj& psnr); @@ -48,9 +58,19 @@ class VkVideoEncoderPsnr : public VkVideoRefCountBase { uint32_t maxEncodeQueueDepth, VkFormat imageDpbFormat, const VkExtent2D& imageExtent, - uint32_t encodeQueueFamilyIndex); + uint32_t encodeQueueFamilyIndex, + VkFormat imageInFormat = VK_FORMAT_UNDEFINED); void CaptureInput(void* encodeFrameInfo, const uint8_t* pInputFrameData); + // MECHANISM-C. Records (encode-source image -> host-visible LINEAR image) + // into the ENCODE command buffer, ahead of vkCmdBeginVideoCodingKHR, so the + // capture observes the image through the same submission and the same + // dependency chain the encode itself does. Returns false when disabled. + bool CaptureSource(VkCommandBuffer cmdBuf, void* encodeFrameInfo); + bool SrcCaptureEnabled() const { return m_capSrcEnabled; } + // Records (imported linear image -> host-visible LINEAR image) into the + // STAGING command buffer. linearImageViewVoid is a VkImageResourceView*. + bool CaptureImported(VkCommandBuffer cmdBuf, void* encodeFrameInfo, void* linearImageViewVoid); bool CaptureOutput(VkCommandBuffer cmdBuf, void* encodeFrameInfo); void ComputeFramePsnr(void* encodeFrameInfo); double GetAveragePsnrY() const; @@ -73,6 +93,25 @@ class VkVideoEncoderPsnr : public VkVideoRefCountBase { uint32_t m_encodeQueueFamilyIndex = 0; VkSharedBaseObj m_psnrReconImagePool; + // MECHANISM-C state. + VkSharedBaseObj m_capSrcImagePool; + VkSharedBaseObj m_capImpImagePool; + bool m_capSrcEnabled = false; + VkFormat m_capImpFormat = VK_FORMAT_UNDEFINED; + VkExtent2D m_capImpExtent = {}; + uint32_t m_capImpMissCount = 0; + uint32_t m_capStride = 8; + std::atomic m_capSeq{0}; + std::vector m_capScratch; + void DumpCapturedNode(VkSharedBaseObj& node, + const char* tag, uint32_t frameIdx, + uint32_t w, uint32_t h, bool writeFile, + uint64_t seq, uint64_t imgId); + VkFormat m_imageInFormat = VK_FORMAT_UNDEFINED; + uint32_t m_capSrcMissCount = 0; + uint32_t m_capSrcFilesWritten = 0; + uint32_t m_capSrcMaxFiles = 0; + void DumpCapturedSource(void* encodeFrameInfoVoid); double m_psnrSum = 0.0; double m_psnrSumU = 0.0; double m_psnrSumV = 0.0; diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.cpp b/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.cpp index f4d54531..1042da94 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.cpp +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.cpp @@ -17,7 +17,7 @@ #include "VkVideoGopStructure.h" #include -VkVideoGopStructure::VkVideoGopStructure(uint8_t gopFrameCount, +VkVideoGopStructure::VkVideoGopStructure(uint32_t gopFrameCount, int32_t idrPeriod, uint8_t consecutiveBFrameCount, uint8_t temporalLayerCount, @@ -42,7 +42,12 @@ VkVideoGopStructure::VkVideoGopStructure(uint8_t gopFrameCount, bool VkVideoGopStructure::Init(uint64_t maxNumFrames) { - m_gopFrameCycle = (uint8_t)(m_consecutiveBFrameCount + 1); + // Same sentinel guard as SetConsecutiveBFrameCount(): while the count is + // still UINT8_MAX ("driver preferred", unresolved) a +1 would wrap to 0 and + // the sub-GOP modulo would divide by zero. + m_gopFrameCycle = (m_consecutiveBFrameCount == UINT8_MAX) + ? m_gopFrameCycle + : (uint8_t)(m_consecutiveBFrameCount + 1); return true; } diff --git a/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.h b/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.h index dedbeb38..68bcddd7 100644 --- a/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.h +++ b/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.h @@ -81,7 +81,7 @@ class VkVideoGopStructure { {} }; - VkVideoGopStructure(uint8_t gopFrameCount = 8, + VkVideoGopStructure(uint32_t gopFrameCount = 8, int32_t idrPeriod = 60, uint8_t consecutiveBFrameCount = 2, uint8_t temporalLayerCount = 1, @@ -117,18 +117,56 @@ class VkVideoGopStructure { // If it is set to 0, the rate control algorithm may assume an // implementation-dependent GOP length. If it is set to UINT32_MAX, // the GOP length is treated as infinite. + // + // UINT32_MAX is the Video Codec SDK's infinite-GOP value, and the two + // spellings a caller reaches it by are worth naming together: the command + // line takes -1, and the JSON configuration takes 4294967295. An infinite + // GOP is also what EncoderConfigH264::GetRateControlParameters keys on + // when it drops idrPeriod to 0 -- an IDR interval shorter than the GOP is + // not expressible, so a GOP that never closes leaves no period to state. void SetGopFrameCount(uint32_t gopFrameCount) { m_gopFrameCount = gopFrameCount; } uint32_t GetGopFrameCount() const { return m_gopFrameCount; } - // idrPeriod is the interval, in terms of number of frames, between two IDR frames (see IDR period). - // If it is set to 0, the rate control algorithm may assume an implementation-dependent IDR period. - // If it is set to UINT8_MAX, the IDR period is treated as infinite. + // idrPeriod is the interval, in terms of number of frames, between two IDR + // frames (see IDR period). + // + // 0 MEANS A DIFFERENT THING AT EACH LAYER, and both are reachable: + // * To EncoderConfig it is the "unset" sentinel -- EncoderConfigH264, + // H265 and AV1 each replace a 0 with the device's preferred IDR + // period before this structure is configured. + // * To this structure it applies no IDR-period bound at all, so nothing + // forces a periodic IDR. A caller that sets this directly, past the + // codec config, gets that second meaning. + // + // UINT32_MAX makes the period infinite explicitly, and is not rewritten by + // the codec configs the way 0 is: it says "never emit a periodic IDR" + // rather than "choose one for me". It is the same all-bits-set convention + // gopFrameCount uses, and the command line spells it -1 for a field of any + // width -- 255 into a uint8_t field, 4294967295 into a uint32_t one. void SetIdrPeriod(uint32_t idrPeriod) { m_idrPeriod = idrPeriod; } uint32_t GetIdrPeriod() const { return m_idrPeriod; } // consecutiveBFrameCount is the number of consecutive B frames between I and/or P frames within the GOP. - void SetConsecutiveBFrameCount(uint8_t consecutiveBFrameCount) { m_consecutiveBFrameCount = consecutiveBFrameCount; } + // m_gopFrameCycle is what actually PLACES reference frames (see the + // `(gopPos.inGop % m_gopFrameCycle) == 0` sub-GOP test below), so it has + // to move with the count, and not only from Init(): Init() runs BEFORE the + // GetMaxBFrameCount() clamp in VkVideoEncoder::InitEncoder, so a cycle + // written only there leaves that clamp INERT -- the generator emits runs of + // the REQUESTED length while GetConsecutiveBFrameCount() reports the clamped + // one. UINT8_MAX is the "driver preferred" sentinel and a cycle of 0 would + // make the modulo a division by zero, so hold the cycle until the sentinel + // is resolved by InitDeviceCapabilities(). + void SetConsecutiveBFrameCount(uint8_t consecutiveBFrameCount) { + m_consecutiveBFrameCount = consecutiveBFrameCount; + if (consecutiveBFrameCount != UINT8_MAX) { + m_gopFrameCycle = (uint8_t)(consecutiveBFrameCount + 1); + } + } uint8_t GetConsecutiveBFrameCount() const { return m_consecutiveBFrameCount; } + // The generator's real sub-GOP period. Equals GetConsecutiveBFrameCount()+1 + // once Init() or the setter has run; read THIS, not the count, when you + // need to bound the longest run of non-reference frames. + uint8_t GetGopFrameCycle() const { return m_gopFrameCycle; } void SetIntraRefreshCycleDuration(uint32_t intraRefreshCycleDuration) { m_intraRefreshCycleDuration = intraRefreshCycleDuration; } @@ -363,7 +401,9 @@ class VkVideoGopStructure { uint8_t m_consecutiveBFrameCount; uint8_t m_gopFrameCycle; uint8_t m_temporalLayerCount; - uint32_t m_idrPeriod; // 0 means unlimited GOP with no IDRs. + // 0 here applies no IDR-period bound; see SetIdrPeriod for why that is + // not the same as 0 in EncoderConfig. + uint32_t m_idrPeriod; FrameType m_lastFrameType; FrameType m_preClosedGopAnchorFrameType; uint32_t m_closedGop : 1; diff --git a/vk_video_encoder/libs/json/EncoderConfigJsonLoader.cpp b/vk_video_encoder/libs/json/EncoderConfigJsonLoader.cpp index ef4d7de1..c5e4f01e 100644 --- a/vk_video_encoder/libs/json/EncoderConfigJsonLoader.cpp +++ b/vk_video_encoder/libs/json/EncoderConfigJsonLoader.cpp @@ -154,5 +154,67 @@ int LoadEncoderConfigFromJson(const char* path, void* encoderConfig) { } } } + + // ---- Colour description: RAISE THE PRESENCE FLAGS ------------------- + // + // WHAT THIS FIXES, and it is an A1-doctrine violation rather than a + // missing feature: the four keys above wrote colour_primaries, + // transfer_characteristics, matrix_coefficients and video_full_range_flag + // and NOTHING ELSE. Every VUI writer in this library gates on the + // PRESENCE flags -- VkEncoderConfigH264.cpp and VkEncoderConfigH265.cpp + // both open their colour block with `if (!!color_description_present_flag)` + // and their range block with video_signal_type_present_flag, and + // EncoderConfigAV1 keys its own description on the same flag -- so a JSON + // config that declared BT.2020 was ACCEPTED AND SILENTLY IGNORED. The CLI + // reported success and the bitstream carried no colour description at all. + // + // It is also the only route by which matrix_coefficients could ever have + // been non-zero WITHOUT the presence flag, which is the one state + // EncoderConfig::ResolveRgbToYcbcrMatrix's `case 0` arm is written to + // defend against. Closing it here is what makes that arm's "unreachable + // through every producer in this tree" comment true. + // + // THE RULE IS THE EXT BINDER'S, DELIBERATELY, so the two entry points + // cannot disagree about what a JSON file and a chained struct mean by the + // same numbers (vulkan_video_encoder_ext.cpp, grep + // `0 IS "NOT SUPPLIED" ON THIS SURFACE`): + // * 0 in any of the three idc fields means NOT SUPPLIED, not + // Reserved/Identity. It costs the ability to REQUEST code point 0 + // through JSON, which is deliberate: this encoder has no Identity + // path to offer. + // * an unsupplied field becomes 2 (Unspecified) -- the code point that + // MEANS "not stated" in H.264, H.265 and AV1 alike -- so a partially + // supplied declaration stays truthful in every field instead of + // asserting Reserved by omission. + // * video_signal_type travels ALONE when only the range was declared, + // because on H.26x the range lives under a different presence flag. + // * video_format 5 = "unspecified" (Rec. ITU-T H.264 Table E-2), the + // correct value when a caller communicates colorimetry only. + // + // AFTER THE KEY LOOP, not inside it, because "was anything supplied" is a + // question about the whole object and the keys may arrive in any order. + { + const bool anyColourIdcSupplied = (config->colour_primaries != 0) || + (config->transfer_characteristics != 0) || + (config->matrix_coefficients != 0); + const uint8_t kColourIdcUnspecified = 2; + if (anyColourIdcSupplied || (config->video_full_range_flag != 0)) { + config->video_format = 5; + config->video_signal_type_present_flag = 1; + } + if (anyColourIdcSupplied) { + if (config->colour_primaries == 0) { + config->colour_primaries = kColourIdcUnspecified; + } + if (config->transfer_characteristics == 0) { + config->transfer_characteristics = kColourIdcUnspecified; + } + if (config->matrix_coefficients == 0) { + config->matrix_coefficients = kColourIdcUnspecified; + } + config->color_description_present_flag = 1; + } + } + return 0; } diff --git a/vk_video_encoder/src/encoder_argv_completion.h b/vk_video_encoder/src/encoder_argv_completion.h new file mode 100644 index 00000000..fc57e3b3 --- /dev/null +++ b/vk_video_encoder/src/encoder_argv_completion.h @@ -0,0 +1,94 @@ +/* +* Copyright 2026 NVIDIA Corporation. +* +* Licensed under the Apache License, Version 2.0 (the "License"); +* you may not use this file except in compliance with the License. +* You may obtain a copy of the License at +* +* http://www.apache.org/licenses/LICENSE-2.0 +* +* Unless required by applicable law or agreed to in writing, software +* distributed under the License is distributed on an "AS IS" BASIS, +* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +* See the License for the specific language governing permissions and +* limitations under the License. +*/ + +#ifndef _VK_VIDEO_ENCODER_ARGV_COMPLETION_H_ +#define _VK_VIDEO_ENCODER_ARGV_COMPLETION_H_ + +#include +#include + +#include "vulkan/vulkan.h" + +//============================================================================= +// The argv wrapper's completion verdict. +// +// PRIVATE: no public ABI, no queried role, not installed. It exists so the +// wrapper can tell the truth about a completion it previously reported as +// VK_SUCCESS whatever happened -- it discarded the thread join's bool and +// never asked whether the buffered bitstream had reached the file. +// +// Both steps are supplied as callables so the decision can be tested without +// an encoder or a device, and so the test exercises this code rather than a +// copy of its condition. +//============================================================================= +class ArgvCompletionState { +public: + // ORDER IS LOAD-BEARING: wait, then check the output, then let the caller + // release the owner. A flush that runs before the workers have joined can + // report success for a file those workers are still writing. + // + // |flushOutput| is only consulted when file output is enabled; a capture + // session has no file to flush and must not be failed for the absence of + // one. + // + // Cached. A repeated call neither joins nor flushes again, and never turns + // a recorded failure into a success. + VkResult Complete(bool fileOutputEnabled, + const std::function& waitForThreads, + const std::function& flushOutput) + { + if (m_completed) { + return m_result; + } + m_completed = true; + + const bool waited = waitForThreads ? waitForThreads() : false; + + bool flushed = true; + if (fileOutputEnabled) { + flushed = flushOutput ? flushOutput() : false; + } + + // The lower bool supplies no precise cause, so this does not invent a + // Vulkan reason it cannot support. + m_result = (waited && flushed) ? VK_SUCCESS : VK_ERROR_UNKNOWN; + return m_result; + } + + bool Completed() const { return m_completed; } + VkResult Result() const { return m_result; } + + // Flush and check one stdio stream. Separate so the wrapper binds it and a + // test can substitute a failing stream. + static bool FlushFileOutput(FILE* file) + { + if (file == nullptr) { + // File output was enabled and there is no handle: the output the + // caller asked for does not exist. + return false; + } + if (fflush(file) != 0) { + return false; + } + return ferror(file) == 0; + } + +private: + bool m_completed = false; + VkResult m_result = VK_SUCCESS; +}; + +#endif /* _VK_VIDEO_ENCODER_ARGV_COMPLETION_H_ */ diff --git a/vk_video_encoder/src/vulkan_video_encoder.cpp b/vk_video_encoder/src/vulkan_video_encoder.cpp index 90997884..966f9037 100644 --- a/vk_video_encoder/src/vulkan_video_encoder.cpp +++ b/vk_video_encoder/src/vulkan_video_encoder.cpp @@ -1,5 +1,5 @@ /* - * Copyright 2024 NVIDIA Corporation. + * Copyright 2026 NVIDIA Corporation. * * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. @@ -14,243 +14,2011 @@ * limitations under the License. */ +// The encoder interface over the encoder implementation. +// +// This file is a translation layer and holds no encoder logic: every method +// forwards to the descriptor API and converts the result. The translation is +// deliberately dull, because the value of the layer is that the rules it +// enforces -- which formats resolve to which input path, which codecs carry +// HDR metadata, which reconfigurations are refusable -- are stated once here +// rather than in each caller. +// +// A role is offered only when the underlying implementation can serve it, so +// Query() returning null means the same thing to a caller whether the interface +// is the implementation or a translation onto one. + #include "vulkan_video_encoder.h" +#include "vulkan_video_encoder_ext.h" +#include "vulkan_video_encoder_ext_internal.h" +#include "vulkan_video_encoder_os_event_linux.h" + +#include +#include +#include +#include +#include +#include + +namespace vk { +namespace video { +namespace enc { + +namespace { + +//============================================================================= +// VOCABULARY TRANSLATION +//============================================================================= + +VkVideoCodecOperationFlagBitsKHR ToVkCodec(Codec codec) +{ + switch (codec) { + case Codec::H264: return VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + case Codec::H265: return VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + case Codec::AV1: return VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + } + return VK_VIDEO_CODEC_OPERATION_NONE_KHR; +} + +uint32_t ToExtProfile(Profile profile) +{ + switch (profile) { + case Profile::Default: return VK_VIDEO_ENCODER_PROFILE_DEFAULT; + case Profile::H264Baseline: return VK_VIDEO_ENCODER_PROFILE_H264_BASELINE; + case Profile::H264Main: return VK_VIDEO_ENCODER_PROFILE_H264_MAIN; + case Profile::H264High: return VK_VIDEO_ENCODER_PROFILE_H264_HIGH; + case Profile::H264High10: return VK_VIDEO_ENCODER_PROFILE_H264_HIGH_10; + case Profile::H265Main: return VK_VIDEO_ENCODER_PROFILE_H265_MAIN; + case Profile::H265Main10: return VK_VIDEO_ENCODER_PROFILE_H265_MAIN10; + case Profile::H265MainStillPicture: return VK_VIDEO_ENCODER_PROFILE_H265_MAIN_STILL_PICTURE; + case Profile::H265Rext: return VK_VIDEO_ENCODER_PROFILE_H265_FORMAT_RANGE_EXTENSIONS; + case Profile::AV1Main: return VK_VIDEO_ENCODER_PROFILE_AV1_MAIN; + case Profile::AV1High: return VK_VIDEO_ENCODER_PROFILE_AV1_HIGH; + case Profile::AV1Professional: return VK_VIDEO_ENCODER_PROFILE_AV1_PROFESSIONAL; + } + return VK_VIDEO_ENCODER_PROFILE_DEFAULT; +} + +// Which codec a profile belongs to. Default belongs to whichever codec it is +// paired with, so it is not answered here. +bool ProfileMatchesCodec(Codec codec, Profile profile) +{ + switch (profile) { + case Profile::Default: + return true; + case Profile::H264Baseline: + case Profile::H264Main: + case Profile::H264High: + case Profile::H264High10: + return codec == Codec::H264; + case Profile::H265Main: + case Profile::H265Main10: + case Profile::H265MainStillPicture: + case Profile::H265Rext: + return codec == Codec::H265; + case Profile::AV1Main: + case Profile::AV1High: + case Profile::AV1Professional: + return codec == Codec::AV1; + } + return false; +} + +VkVideoEncoderColorModel ToExtColorModel(ColorModel model) +{ + switch (model) { + case ColorModel::FromFormat: return VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT; + case ColorModel::Rgb: return VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + case ColorModel::YCbCr: return VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + } + return VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT; +} + +VkVideoEncodeRateControlModeFlagBitsKHR ToExtRateControl(RateControlMode mode) +{ + switch (mode) { + case RateControlMode::Default: return VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR; + case RateControlMode::Disabled: return VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + case RateControlMode::Cbr: return VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + case RateControlMode::Vbr: return VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_BIT_KHR; + } + return VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR; +} + +VkVideoEncodeTuningModeKHR ToExtTuning(TuningMode mode) +{ + switch (mode) { + case TuningMode::Default: return VK_VIDEO_ENCODE_TUNING_MODE_DEFAULT_KHR; + case TuningMode::HighQuality: return VK_VIDEO_ENCODE_TUNING_MODE_HIGH_QUALITY_KHR; + case TuningMode::LowLatency: return VK_VIDEO_ENCODE_TUNING_MODE_LOW_LATENCY_KHR; + case TuningMode::UltraLowLatency: return VK_VIDEO_ENCODE_TUNING_MODE_ULTRA_LOW_LATENCY_KHR; + case TuningMode::Lossless: return VK_VIDEO_ENCODE_TUNING_MODE_LOSSLESS_KHR; + } + return VK_VIDEO_ENCODE_TUNING_MODE_DEFAULT_KHR; +} + +FrameState FromExtFrameState(VkVideoEncoderFrameState state) +{ + switch (state) { + case VK_VIDEO_ENCODER_FRAME_STATE_PENDING: return FrameState::Pending; + case VK_VIDEO_ENCODER_FRAME_STATE_READY: return FrameState::Ready; + case VK_VIDEO_ENCODER_FRAME_STATE_ACQUIRED: return FrameState::Acquired; + default: return FrameState::Unknown; + } +} + +PictureType FromExtPictureType(VkVideoEncoderPictureType type) +{ + switch (type) { + case VK_VIDEO_ENCODER_PICTURE_TYPE_P: return PictureType::Predicted; + case VK_VIDEO_ENCODER_PICTURE_TYPE_B: return PictureType::Bidirectional; + default: return PictureType::Intra; + } +} + +VkVideoEncoderInputResidency ToExtResidency(Residency residency, ExternalHandleType handleType) +{ + switch (residency) { + case Residency::Local: return VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + case Residency::Foreign: return VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN; + case Residency::Auto: + break; + } + // An image already resident on the encoder's device needs no ownership + // transfer, and the library cannot infer that from the handle alone. + // Every other handle type is an import, where AUTO is the right answer. + return (handleType == ExternalHandleType::VkImageHandle) + ? VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL + : VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO; +} -#include "VkVideoEncoder/VkEncoderConfig.h" -#include "VkVideoEncoder/VkVideoEncoder.h" +// The driver's own identity for a physical device. Read through the loader +// the caller supplies, so the answer comes from the same driver the encoder +// talks to rather than whichever one is first on the path. +bool VkEncQueryDeviceUuid(VkInstance instance, VkPhysicalDevice physicalDevice, + PFN_vkGetInstanceProcAddr getInstanceProcAddr, + uint8_t outUuid[VK_UUID_SIZE]) +{ + if ((instance == VK_NULL_HANDLE) || (physicalDevice == VK_NULL_HANDLE) || + (getInstanceProcAddr == nullptr) || (outUuid == nullptr)) { + return false; + } + PFN_vkGetPhysicalDeviceProperties2 getProps2 = + reinterpret_cast( + getInstanceProcAddr(instance, "vkGetPhysicalDeviceProperties2")); + if (getProps2 == nullptr) { + return false; + } + VkPhysicalDeviceIDProperties idProps{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_ID_PROPERTIES}; + VkPhysicalDeviceProperties2 props{VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2}; + props.pNext = &idProps; + getProps2(physicalDevice, &props); + std::copy(std::begin(idProps.deviceUUID), std::end(idProps.deviceUUID), outUuid); + return true; +} -class VulkanVideoEncoderImpl : public VulkanVideoEncoder { +VkVideoEncoderHandleOwnership ToExtOwnership(HandleOwnership ownership) +{ + return (ownership == HandleOwnership::Borrow) + ? VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_BORROW + : VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_TRANSFER; +} + +VkVideoEncoderExternalHandleType ToExtHandleType(ExternalHandleType type) +{ + switch (type) { + case ExternalHandleType::NoHandle: return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_NONE; + case ExternalHandleType::OpaqueFd: return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD; + case ExternalHandleType::DmaBuf: return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF; + case ExternalHandleType::OpaqueWin32: return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_WIN32; + case ExternalHandleType::D3D11Texture: return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_D3D11_TEXTURE; + case ExternalHandleType::VkImageHandle: return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + } + return VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_NONE; +} + +// The descriptor API speaks two error currencies; this interface speaks one. A VkResult +// that has no closer meaning becomes InternalError rather than being dropped, +// and the detail string keeps the distinction visible in a log. +Result FromVkResult(VkResult result, const char* detail) +{ + switch (result) { + case VK_SUCCESS: return Result(); + case VK_NOT_READY: return Result(ResultCode::NotReady, detail); + case VK_TIMEOUT: return Result(ResultCode::Timeout, detail); + case VK_ERROR_FORMAT_NOT_SUPPORTED: return Result(ResultCode::UnsupportedFormat, detail); + case VK_ERROR_FEATURE_NOT_PRESENT: return Result(ResultCode::UnsupportedFeature, detail); + case VK_ERROR_DEVICE_LOST: return Result(ResultCode::DeviceLost, detail); + case VK_ERROR_OUT_OF_HOST_MEMORY: + case VK_ERROR_OUT_OF_DEVICE_MEMORY: return Result(ResultCode::OutOfResources, detail); + case VK_ERROR_INITIALIZATION_FAILED: return Result(ResultCode::InternalError, detail); + default: return Result(ResultCode::InternalError, detail); + } +} + +Result FromExtStatusCode(VkVideoEncoderStatusCode code, const char* detail) +{ + if (code == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return Result(); + } + switch (code) { + case VK_VIDEO_ENCODER_STATUS_NOT_READY: + return Result(ResultCode::NotReady, detail); + case VK_VIDEO_ENCODER_STATUS_ERROR_FORMAT_UNSUPPORTED: + case VK_VIDEO_ENCODER_STATUS_ERROR_MODIFIER_UNSUPPORTED: + case VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED: + return Result(ResultCode::UnsupportedFormat, detail); + case VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED: + case VK_VIDEO_ENCODER_STATUS_ERROR_EXTENSION_MISSING: + case VK_VIDEO_ENCODER_STATUS_ERROR_SHARING_MODE_UNSUPPORTED: + case VK_VIDEO_ENCODER_STATUS_ERROR_API_VERSION_UNSUPPORTED: + return Result(ResultCode::UnsupportedFeature, detail); + case VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN: + case VK_VIDEO_ENCODER_STATUS_ERROR_PLANE_LAYOUT_INVALID: + case VK_VIDEO_ENCODER_STATUS_ERROR_ALLOCATION_SIZE_INVALID: + case VK_VIDEO_ENCODER_STATUS_ERROR_EXTENT_INVALID: + case VK_VIDEO_ENCODER_STATUS_ERROR_USAGE_INSUFFICIENT: + case VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN: + case VK_VIDEO_ENCODER_STATUS_ERROR_DEVICE_MISMATCH: + case VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED: + return Result(ResultCode::InvalidArgument, detail); + case VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT: + return Result(ResultCode::OutOfResources, detail); + case VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED: + return Result(ResultCode::NotConfigured, detail); + default: + return Result(ResultCode::InternalError, detail); + } +} + +//============================================================================= +// EXTERNAL IMAGE +//============================================================================= + +// Fill the descriptor the implementation expects. The OS handle travels +// separately because the descriptor API takes it as its own argument. +void ToExtImageDescriptor(const ExternalImage& image, + VkVideoEncoderExternalImageDescriptor& out, + uint64_t& outOsHandle) +{ + out = VkVideoEncoderExternalImageDescriptor(); + out.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + out.pNext = nullptr; + + out.handleType = ToExtHandleType(image.handleType); + out.format = image.format; + out.width = image.width; + out.height = image.height; + out.imageType = VK_IMAGE_TYPE_2D; + out.mipLevels = 1; + out.arrayLayers = 1; + out.samples = VK_SAMPLE_COUNT_1_BIT; + out.tiling = image.tiling; + out.imageUsage = image.usage; + out.imageFlags = image.createFlags; + out.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + out.residency = ToExtResidency(image.residency, image.handleType); + out.ownership = ToExtOwnership(image.ownership); + out.defaultLayout = image.layout; + out.colorModel = ToExtColorModel(image.colorModel); + out.existingImage = image.existingImage; + + out.hasDrmFormatModifier = image.hasDrmFormatModifier ? VK_TRUE : VK_FALSE; + out.drmFormatModifier = image.drmFormatModifier; + + out.planeCount = image.planeCount; + for (uint32_t i = 0; i < image.planeCount && i < VK_VIDEO_ENCODER_MAX_PLANES; ++i) { + out.planeLayouts[i].offset = image.planeLayouts[i].offset; + out.planeLayouts[i].rowPitch = image.planeLayouts[i].rowPitch; + out.planeLayouts[i].size = image.planeLayouts[i].size; + } + + std::copy(std::begin(image.deviceUUID), std::end(image.deviceUUID), + std::begin(out.deviceUUID)); + std::copy(std::begin(image.driverUUID), std::end(image.driverUUID), + std::begin(out.driverUUID)); + std::copy(std::begin(image.deviceLUID), std::end(image.deviceLUID), + std::begin(out.deviceLUID)); + out.deviceLUIDValid = image.deviceLuidValid ? VK_TRUE : VK_FALSE; + + out.allocationSize = image.allocationSize; + out.memoryTypeBits = image.memoryTypeBits; + out.memoryTypeIndex = image.memoryTypeIndex; + + switch (image.handleType) { + case ExternalHandleType::OpaqueFd: + case ExternalHandleType::DmaBuf: + outOsHandle = static_cast(static_cast(image.fd)); + break; + case ExternalHandleType::OpaqueWin32: + case ExternalHandleType::D3D11Texture: + outOsHandle = reinterpret_cast(image.win32Handle); + break; + default: + outOsHandle = 0; + break; + } +} + +} // namespace + +//============================================================================= +// CAPABILITIES +//============================================================================= + +class CapsImpl final : public IEncoderCaps { public: - virtual VkResult Initialize(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, - int argc, const char** argv); - virtual int64_t GetNumberOfFrames() + CapsImpl(VulkanVideoEncoderContext* ctx, uint32_t deviceIndex) + : m_ctx(ctx), m_deviceIndex(deviceIndex) { } + + void* QueryInterface(std::string_view id) override { - return m_encoderConfig->numFrames; + if (id == IEncoderCaps::kId) { + return static_cast(this); + } + return nullptr; } - virtual VkResult EncodeNextFrame(int64_t& frameNumEncoded); - virtual VkResult GetBitstream() { - if (m_encoder) { - m_encoder->WaitForThreadsToComplete(); + + ArrayView Codecs() const override + { + std::lock_guard lock(m_mutex); + if (m_codecs.empty()) { + for (Codec codec : {Codec::H264, Codec::H265, Codec::AV1}) { + VkVideoEncoderCapabilities caps{}; + caps.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES; + if (VkEncGetEncodeCapabilities(m_ctx, m_deviceIndex, ToVkCodec(codec), + VK_VIDEO_ENCODER_PROFILE_DEFAULT, + &caps) == VK_SUCCESS) { + m_codecs.push_back(codec); + } + } } - return VK_SUCCESS; + return ArrayView(m_codecs); } - VulkanVideoEncoderImpl() - : m_vkDevCtxt() - , m_encoderConfig() - , m_encoder() - , m_lastFrameIndex(0) - { } + ArrayView Profiles(Codec codec) const override + { + std::lock_guard lock(m_mutex); + auto it = m_profiles.find(codec); + if (it == m_profiles.end()) { + std::vector supported; + static const Profile kAll[] = { + Profile::H264Baseline, Profile::H264Main, Profile::H264High, Profile::H264High10, + Profile::H265Main, Profile::H265Main10, Profile::H265MainStillPicture, + Profile::H265Rext, + Profile::AV1Main, Profile::AV1High, Profile::AV1Professional, + }; + for (Profile profile : kAll) { + if (!ProfileMatchesCodec(codec, profile)) { + continue; + } + VkVideoEncoderCapabilities caps{}; + caps.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES; + if (VkEncGetEncodeCapabilities(m_ctx, m_deviceIndex, ToVkCodec(codec), + ToExtProfile(profile), &caps) == VK_SUCCESS) { + supported.push_back(profile); + } + } + it = m_profiles.emplace(codec, std::move(supported)).first; + } + return ArrayView(it->second); + } - virtual ~VulkanVideoEncoderImpl() { } + ArrayView InputFormats(Profile profile) const override + { + std::lock_guard lock(m_mutex); + auto it = m_inputFormats.find(profile); + if (it != m_inputFormats.end()) { + return ArrayView(it->second); + } - void Deinitialize() + const VkVideoCodecOperationFlagBitsKHR codec = CodecForProfile(profile); + std::vector formats; + + uint32_t count = 0; + if (VkEncEnumerateInputFormats(m_ctx, m_deviceIndex, codec, ToExtProfile(profile), + &count, nullptr) == VK_SUCCESS && count > 0) { + std::vector props(count); + if (VkEncEnumerateInputFormats(m_ctx, m_deviceIndex, codec, ToExtProfile(profile), + &count, props.data()) == VK_SUCCESS) { + formats.reserve(count); + for (uint32_t i = 0; i < count; ++i) { + InputFormat f; + f.format = props[i].format; + f.colorModel = ColorModel::FromFormat; + f.isOptimal = (props[i].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL); + // A format the encoder can take unchanged needs no + // conversion; one whose session format differs is reached + // by a copy, or by the filter when the samples themselves + // must change. + f.path = (props[i].format == props[i].encodeFormat) + ? InputPath::Direct + : InputPath::Copy; + formats.push_back(f); + } + } + } + it = m_inputFormats.emplace(profile, std::move(formats)).first; + return ArrayView(it->second); + } + + ArrayView DrmModifiers(Profile profile, VkFormat format) const override { - m_encoder->WaitForThreadsToComplete(); + (void)profile; + std::lock_guard lock(m_mutex); + auto it = m_modifiers.find(format); + if (it != m_modifiers.end()) { + return ArrayView(it->second); + } + + const VkImageUsageFlags usage = + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + std::vector modifiers; + uint32_t count = 0; + if (VkEncEnumerateDrmModifiers(m_ctx, m_deviceIndex, format, usage, + &count, nullptr) == VK_SUCCESS && count > 0) { + modifiers.resize(count); + if (VkEncEnumerateDrmModifiers(m_ctx, m_deviceIndex, format, usage, + &count, modifiers.data()) != VK_SUCCESS) { + modifiers.clear(); + } + } + it = m_modifiers.emplace(format, std::move(modifiers)).first; + return ArrayView(it->second); + } + + RateControlCaps RateControl(Profile profile) const override + { + RateControlCaps out; + VkVideoEncoderCapabilities caps{}; + if (!GetCaps(profile, caps)) { + return out; + } + out.supportsCbr = (caps.supportedRateControlModes & + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR) != 0; + out.supportsVbr = (caps.supportedRateControlModes & + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_BIT_KHR) != 0; + out.supportsConstantQp = (caps.supportedRateControlModes & + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR) != 0; + out.maxQualityLevels = caps.maxQualityLevels; + out.maxBitrate = caps.maxBitrate; + return out; + } + + VkExtent2D MinCodedExtent(Profile profile) const override + { + VkVideoEncoderCapabilities caps{}; + if (!GetCaps(profile, caps)) { + return VkExtent2D{0, 0}; + } + return caps.minCodedExtent; + } - if (m_encoderConfig->verbose) { - std::cout << "Done processing " << m_lastFrameIndex << " input frames!" << std::endl - << "Encoded file's location is at " << m_encoderConfig->outputFileHandler.GetFileName() - << std::endl; + VkExtent2D MaxCodedExtent(Profile profile) const override + { + VkVideoEncoderCapabilities caps{}; + if (!GetCaps(profile, caps)) { + return VkExtent2D{0, 0}; } + return caps.maxCodedExtent; + } - m_encoder = nullptr; - m_encoderConfig = nullptr; + bool Supports(Feature feature) const override + { + VkVideoEncoderCapabilities caps{}; + switch (feature) { + case Feature::IntraRefresh: + return GetCaps(Profile::Default, caps) && caps.supportsIntraRefresh; + case Feature::HdrMetadata: + // Carried by H.265 and AV1; H.264 has no such SEI. + return true; + case Feature::DmaBufImport: + case Feature::ExternalSemaphores: + case Feature::RegisteredResources: + case Feature::Reconfigure: + return true; + case Feature::BFrames: + return GetCaps(Profile::Default, caps) && caps.maxDpbSlots > 2; + } + return false; } private: - VulkanDeviceContext m_vkDevCtxt; - VkSharedBaseObj m_encoderConfig; - VkSharedBaseObj m_encoder; - uint32_t m_lastFrameIndex; + VkVideoCodecOperationFlagBitsKHR CodecForProfile(Profile profile) const + { + for (Codec codec : {Codec::H264, Codec::H265, Codec::AV1}) { + if (profile != Profile::Default && ProfileMatchesCodec(codec, profile)) { + return ToVkCodec(codec); + } + } + return VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + } + + bool GetCaps(Profile profile, VkVideoEncoderCapabilities& caps) const + { + caps.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES; + caps.pNext = nullptr; + return VkEncGetEncodeCapabilities(m_ctx, m_deviceIndex, CodecForProfile(profile), + ToExtProfile(profile), &caps) == VK_SUCCESS; + } + + VulkanVideoEncoderContext* m_ctx; + uint32_t m_deviceIndex; + + mutable std::mutex m_mutex; + mutable std::vector m_codecs; + mutable std::map> m_profiles; + mutable std::map> m_inputFormats; + mutable std::map> m_modifiers; }; -VkResult VulkanVideoEncoderImpl::Initialize(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, - int argc, const char** argv) -{ - VkResult result = EncoderConfig::CreateCodecConfig(argc, argv, m_encoderConfig); - if (VK_SUCCESS != result) { - return result; +//============================================================================= +// DEVICE BINDING +//============================================================================= + +class DeviceBindingImpl final : public IDeviceBinding { +public: + static constexpr std::string_view kId = IDeviceBinding::kId; + + DeviceBindingImpl(const PlatformCreateInfo& info) : m_info(info) { } + + void* QueryInterface(std::string_view id) override + { + if (id == IDeviceBinding::kId) { + return static_cast(this); + } + return nullptr; } - static const char* const requiredInstanceLayers[] = { - "VK_LAYER_KHRONOS_validation", - nullptr - }; + // Once a session exists it owns the authoritative handles, including the + // ones a library-created device only has after initialisation. + void BindSession(const VkSharedBaseObj& encoder) + { + m_encoder = encoder; + } - static const char* const requiredInstanceExtensions[] = { - VK_EXT_DEBUG_REPORT_EXTENSION_NAME, - nullptr - }; + VkInstance Instance() const override + { + return m_encoder ? m_encoder->GetVkInstance() : m_info.instance; + } + VkPhysicalDevice PhysicalDevice() const override + { + return m_encoder ? m_encoder->GetVkPhysicalDevice() : m_info.physicalDevice; + } + VkDevice Device() const override + { + return m_encoder ? m_encoder->GetVkDevice() : m_info.device; + } + uint32_t EncodeQueueFamilyIndex() const override { return m_info.encodeQueueFamilyIndex; } + uint32_t ComputeQueueFamilyIndex() const override { return m_info.computeQueueFamilyIndex; } - static const char* const requiredDeviceExtension[] = { -#if defined(__linux) || defined(__linux__) || defined(linux) - VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME, - VK_KHR_EXTERNAL_FENCE_FD_EXTENSION_NAME, -#endif - VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME, - VK_KHR_VIDEO_QUEUE_EXTENSION_NAME, - VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME, - VK_KHR_TIMELINE_SEMAPHORE_EXTENSION_NAME, - nullptr - }; + PFN_vkGetInstanceProcAddr GetInstanceProcAddr() const override + { + return m_encoder ? m_encoder->GetVkGetInstanceProcAddr() : nullptr; + } - static const char* const optinalDeviceExtension[] = { - VK_EXT_YCBCR_2PLANE_444_FORMATS_EXTENSION_NAME, - VK_EXT_DESCRIPTOR_BUFFER_EXTENSION_NAME, - VK_KHR_BUFFER_DEVICE_ADDRESS_EXTENSION_NAME, - VK_KHR_PUSH_DESCRIPTOR_EXTENSION_NAME, - VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, - nullptr - }; + bool DeviceUuid(uint8_t outUuid[VK_UUID_SIZE]) const override + { + return VkEncQueryDeviceUuid(Instance(), PhysicalDevice(), + GetInstanceProcAddr(), outUuid); + } + +private: + PlatformCreateInfo m_info; + VkSharedBaseObj m_encoder; +}; + +//============================================================================= +// CONFIGURATION +//============================================================================= + +// How a session reaches the descriptor-API configuration behind an +// IEncoderConfig. +// +// This library and Chromium both build with -fno-rtti, so dynamic_cast is not +// available to check that a caller handed back a configuration this library +// made. QueryInterface already is a typed downcast that needs no RTTI, so it +// serves here too and there is only one mechanism to understand. The id is not +// in the public header: it is a private channel between two classes in this +// file, reached through the public mechanism. +class IConfigAccess : public IObject { +public: + static constexpr std::string_view kId = "vk.video.enc.internal.IConfigAccess/1"; + + virtual const VkVideoEncoderConfig& ExtConfig() const = 0; + virtual bool RequiresImportExtensions() const = 0; + virtual bool HasHdrMetadata() const = 0; + virtual const HdrMetadata& GetHdrMetadata() const = 0; +}; + +// Accumulates a VkVideoEncoderConfig, and answers the questions the caller +// would otherwise have to answer for itself. +// +// The HDR role is offered only for codecs that can carry the metadata, so an +// H.264 caller learns that HDR is unavailable from a null Query rather than +// from a refused session. +class ConfigImpl final : public IEncoderConfig, + public IHdrMetadataConfig, + public IConfigAccess { +public: + ConfigImpl(Codec codec, Profile profile, const PlatformCreateInfo& platform) + : m_codec(codec), m_profile(profile) + { + // The struct carries default member initialisers, so it is + // value-initialised and then overwritten field by field. Blanking it + // would replace each documented default with whatever zero means for + // that field. + m_config = VkVideoEncoderConfig(); + m_config.pNext = nullptr; + m_config.codec = ToVkCodec(codec); + m_config.profile = ToExtProfile(profile); + + m_config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR; + m_config.tuningMode = VK_VIDEO_ENCODE_TUNING_MODE_DEFAULT_KHR; + m_config.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT; + + // An embedding host has nowhere to write and no reason to. A file is + // written only once SetOutputPath names one. + m_config.disableFileOutput = VK_TRUE; + m_config.outputPath = nullptr; - if (m_encoderConfig->validate) { - m_vkDevCtxt.AddReqInstanceLayers(requiredInstanceLayers); - m_vkDevCtxt.AddReqInstanceExtensions(requiredInstanceExtensions); + // externalInstance and externalPhysicalDevice ARE DELIBERATELY NOT SET + // from |platform|, even when the caller supplied them. Every session + // this interface creates is created ON the platform's context, and a + // context already carries the instance and the physical device it was + // built against. Naming them a second time here is a contradiction the + // encoder refuses outright ("the context already supplies both"), + // which would make an adopted-instance platform unable to create any + // session at all. + // + // externalDevice and the queue families below are a different case and + // do pass through: a context never creates a VkDevice, so a caller + // supplying one is adding something the context does not have. + m_config.externalDevice = platform.device; + m_config.externalEncodeQueueFamilyIndex = platform.encodeQueueFamilyIndex; + m_config.externalComputeQueueFamilyIndex = platform.computeQueueFamilyIndex; + + m_config.deviceId = platform.deviceIndex; + m_config.validate = platform.enableValidation ? VK_TRUE : VK_FALSE; + m_config.silenceStdio = platform.silenceStdio ? VK_TRUE : VK_FALSE; + m_config.verbose = VK_FALSE; + if (platform.gpuUuidValid) { + std::copy(std::begin(platform.gpuUuid), std::end(platform.gpuUuid), + std::begin(m_config.gpuUUID)); + } + + // The three bitstream code points default to "unspecified" (2) rather + // than to a guess about the caller's content. + m_config.colourPrimaries = 2; + m_config.transferCharacteristics = 2; + m_config.matrixCoefficients = 2; + // The input's transfer function defaults to 0, which is a different + // statement: it claims nothing, leaving the input in whatever + // transferCharacteristics ends up declaring. 2 here would be a claim + // of "unspecified" that conflicts with every bitstream declaring a + // real transfer function. + m_config.inputTransferCharacteristics = 0; } - m_vkDevCtxt.AddReqDeviceExtensions(requiredDeviceExtension); - m_vkDevCtxt.AddOptDeviceExtensions(optinalDeviceExtension); + void* QueryInterface(std::string_view id) override + { + if (id == IEncoderConfig::kId) { + return static_cast(this); + } + if (id == IHdrMetadataConfig::kId && CodecCarriesHdrMetadata()) { + return static_cast(this); + } + if (id == IConfigAccess::kId) { + return static_cast(this); + } + return nullptr; + } - result = m_vkDevCtxt.InitVulkanDevice(m_encoderConfig->appName.c_str(), VK_NULL_HANDLE, - m_encoderConfig->verbose); - if (result != VK_SUCCESS) { - printf("Could not initialize the Vulkan device!\n"); - return result; + IEncoderConfig& SetCodedExtent(uint32_t width, uint32_t height) override + { + m_config.encodeWidth = width; + m_config.encodeHeight = height; + if (m_config.inputWidth == 0) { + m_config.inputWidth = width; + m_config.inputHeight = height; + } + return *this; } - result = m_vkDevCtxt.InitDebugReport(m_encoderConfig->validate, - m_encoderConfig->validateVerbose); - if (result != VK_SUCCESS) { - return result; + IEncoderConfig& SetInputExtent(uint32_t width, uint32_t height) override + { + m_config.inputWidth = width; + m_config.inputHeight = height; + return *this; } - VkQueueFlags requestVideoEncodeQueueMask = VK_QUEUE_VIDEO_ENCODE_BIT_KHR; + IEncoderConfig& SetFrameRate(uint32_t numerator, uint32_t denominator) override + { + m_config.frameRateNum = numerator; + // A zero denominator with a real numerator means whole frames per + // second, which is what a host that tracks only a rate passes. + m_config.frameRateDen = + (denominator == 0 && numerator != 0) ? 1 : denominator; + return *this; + } - if (m_encoderConfig->selectVideoWithComputeQueue) { - requestVideoEncodeQueueMask |= VK_QUEUE_COMPUTE_BIT; + IEncoderConfig& SetInputFormat(VkFormat format, ColorModel colorModel) override + { + m_config.inputFormat = format; + m_config.inputColorModel = ToExtColorModel(colorModel); + return *this; } - VkQueueFlags requestVideoComputeQueueMask = 0; - if (m_encoderConfig->enablePreprocessComputeFilter == VK_TRUE) { - requestVideoComputeQueueMask = VK_QUEUE_COMPUTE_BIT; + IEncoderConfig& SetRateControl(const RateControl& rc) override + { + m_config.rateControlMode = ToExtRateControl(rc.mode); + m_config.averageBitrate = rc.averageBitrate; + m_config.maxBitrate = rc.maxBitrate; + m_config.vbvBufferSize = rc.vbvBufferSize; + m_config.constQpI = rc.constQpIntra; + m_config.constQpP = rc.constQpPredicted; + m_config.constQpB = rc.constQpBidirectional; + m_config.minQp = rc.minQp; + m_config.maxQp = rc.maxQp; + m_config.qualityLevel = rc.qualityLevel; + m_config.tuningMode = ToExtTuning(rc.tuning); + return *this; } - // No display presentation and no decoder - just the encoder - result = m_vkDevCtxt.InitPhysicalDevice(m_encoderConfig->deviceId, m_encoderConfig->deviceUUID, - ( requestVideoComputeQueueMask | - requestVideoEncodeQueueMask | - VK_QUEUE_TRANSFER_BIT), - nullptr, - 0, - VK_VIDEO_CODEC_OPERATION_NONE_KHR, - requestVideoEncodeQueueMask, - videoCodecOperation); - if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " - << "InitVulkanDevice() failed - video codec may not be supported. VkResult: " << result - << " (0x" << std::hex << result << std::dec << ")" << std::endl; - return result; + IEncoderConfig& SetGop(const GopStructure& gop) override + { + m_config.gopLength = gop.gopLength; + m_config.idrPeriod = gop.idrPeriod; + m_config.consecutiveBFrames = gop.consecutiveBFrames; + m_config.closedGop = gop.closedGop ? VK_TRUE : VK_FALSE; + return *this; } - const int32_t numEncodeQueues = ((m_encoderConfig->queueId != 0) || - (m_encoderConfig->enableHwLoadBalancing != 0)) ? - -1 : // all available HW encoders - 1; // only one HW encoder instance - - result = m_vkDevCtxt.CreateVulkanDevice(0, // num decode queues - numEncodeQueues, // num encode queues - videoCodecOperation, - // If no graphics or compute queue is requested, only video queues - // will be created. Not all implementations support transfer on video queues, - // so request a separate transfer queue for such implementations. - ((m_vkDevCtxt.GetVideoEncodeQueueFlag() & VK_QUEUE_TRANSFER_BIT) == 0), // createTransferQueue - false, // createGraphicsQueue - false, // createDisplayQueue - ((m_encoderConfig->selectVideoWithComputeQueue == 1) || // createComputeQueue - (m_encoderConfig->enablePreprocessComputeFilter == VK_TRUE)) - ); - if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " - << "CreateVulkanDevice() failed. VkResult: " << result - << " (0x" << std::hex << result << std::dec << ")" << std::endl; - return result; + IEncoderConfig& SetColourInfo(const ColourInfo& colour) override + { + m_config.colourPrimaries = colour.colourPrimaries; + m_config.transferCharacteristics = colour.transferCharacteristics; + m_config.matrixCoefficients = colour.matrixCoefficients; + m_config.videoFullRange = colour.fullRange ? VK_TRUE : VK_FALSE; + m_config.inputTransferCharacteristics = colour.inputTransferCharacteristics; + return *this; } - result = VkVideoEncoder::CreateVideoEncoder(&m_vkDevCtxt, m_encoderConfig, m_encoder); - if (result != VK_SUCCESS) { - std::cerr << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " - << "CreateVideoEncoder() failed. VkResult: " << result - << " (0x" << std::hex << result << std::dec << ")" << std::endl; - return result; + IEncoderConfig& RequireImportExtensions(bool require) override + { + m_requireImportExtensions = require; + return *this; } - return result; -} + IEncoderConfig& SetOutputPath(const char* path) override + { + if (path != nullptr && path[0] != '\0') { + m_outputPath = path; + m_config.outputPath = m_outputPath.c_str(); + m_config.disableFileOutput = VK_FALSE; + } else { + m_outputPath.clear(); + m_config.outputPath = nullptr; + m_config.disableFileOutput = VK_TRUE; + } + return *this; + } -VkResult VulkanVideoEncoderImpl::EncodeNextFrame(int64_t& frameNumEncoded) -{ - if (m_lastFrameIndex >= m_encoderConfig->numFrames) { - return VK_ERROR_TOO_MANY_OBJECTS; + Codec GetCodec() const override { return m_codec; } + Profile GetProfile() const override { return m_profile; } + + // The rule, in one place: a copy engine moves samples the encoder can + // already read; a compute filter runs only when the samples themselves + // must change. A caller asks instead of deciding. + InputPath ResolveInputPath() const override + { + if (m_config.inputFormat == VK_FORMAT_UNDEFINED) { + return InputPath::Copy; + } + const VkEncInputFormatClass cls = + VkEncClassifyInput(m_config.inputFormat, m_config.inputColorModel); + if (cls == VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER) { + return InputPath::ComputeFilter; + } + if (cls == VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + return InputPath::ComputeFilter; + } + // Encodable as-is: a copy still runs when the geometry or the tiling + // differs, but no sample is rewritten. + return InputPath::Copy; } - if (m_encoderConfig->verboseFrameStruct) { - std::cout << "####################################################################################" << std::endl - << "Start processing current input frame index: " << m_lastFrameIndex << std::endl; + VkFormat ResolveSessionFormat() const override + { + return m_config.inputFormat; } - VkSharedBaseObj encodeFrameInfo; - m_encoder->GetAvailablePoolNode(encodeFrameInfo); - assert(encodeFrameInfo); - // load frame data from the file - VkResult result = m_encoder->LoadNextFrame(encodeFrameInfo); - if (result != VK_SUCCESS) { - std::cout << "ERROR processing input frame index: " << m_lastFrameIndex << std::endl; - return result; + Result Validate() const override + { + if (m_config.encodeWidth == 0 || m_config.encodeHeight == 0) { + return Result(ResultCode::InvalidArgument, "coded extent is unset"); + } + if (m_config.inputFormat == VK_FORMAT_UNDEFINED) { + return Result(ResultCode::InvalidArgument, "input format is unset"); + } + // NO FRAME-RATE REQUIREMENT. The encoder has a default and coerces a + // zero denominator to one, so refusing here would refuse a + // configuration the encoder accepts -- and Validate exists to catch + // what the encoder would refuse, not to invent rules of its own. A + // check stricter than the implementation turns a working host into a + // silent no-output. + if (!ProfileMatchesCodec(m_codec, m_profile)) { + return Result(ResultCode::UnsupportedProfile, "profile does not belong to this codec"); + } + if (VkEncClassifyInput(m_config.inputFormat, m_config.inputColorModel) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + return Result(ResultCode::UnsupportedFormat, + "the encoder cannot take this input format on any path"); + } + const bool constantQp = + m_config.rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + if (!constantQp && + m_config.rateControlMode != VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR && + m_config.averageBitrate == 0) { + return Result(ResultCode::InvalidArgument, "a bitrate mode needs an average bitrate"); + } + return Result(); } - frameNumEncoded = encodeFrameInfo->frameInputOrderNum; + void SetHdrMetadata(const HdrMetadata& metadata) override + { + m_hdr = metadata; + m_hasHdr = true; + } - if (m_encoderConfig->verboseFrameStruct) { - std::cout << "End processing current input frame index: " << m_lastFrameIndex << std::endl; + // Consumed by the session: the descriptor API takes HDR metadata as a + // chained struct, and the chain is built only at that point so nothing + // dangles on a configuration the caller keeps. + const VkVideoEncoderConfig& ExtConfig() const override { return m_config; } + bool RequiresImportExtensions() const override + { + return m_requireImportExtensions; } + bool HasHdrMetadata() const override { return m_hasHdr; } - m_lastFrameIndex++; + const HdrMetadata& GetHdrMetadata() const override { return m_hdr; } - return result; -} +private: + bool CodecCarriesHdrMetadata() const + { + // H.264 defines no mastering-display or content-light SEI. + return m_codec != Codec::H264; + } -VK_VIDEO_ENCODER_EXPORT -VkResult CreateVulkanVideoEncoder(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, - int argc, const char** argv, - VkSharedBaseObj& vulkanVideoEncoder) -{ - switch((uint32_t)videoCodecOperation) + bool m_requireImportExtensions = false; + Codec m_codec; + Profile m_profile; + VkVideoEncoderConfig m_config; + std::string m_outputPath; + HdrMetadata m_hdr; + bool m_hasHdr = false; +}; + +//============================================================================= +// SESSION +//============================================================================= + +// One object implementing the session and every role it can serve. The roles +// are separate interfaces so a caller depends on the four methods it uses; a +// single implementation behind them is an implementation detail. +class SessionImpl final : public IEncoderSession, + public IFrameSubmitter, + public IBitstreamSource, + public ICompletionSignal, + public IResourceRegistry, + public IDeviceBinding, + public IDiagnostics, + public IShutdownDiagnostics, + public std::enable_shared_from_this { +public: + SessionImpl(const VkSharedBaseObj& encoder, + const VkVideoEncoderConfig& config) + : m_encoder(encoder), m_config(config) { } + + // Detach before the encoder reference goes. + // + // Dropping m_encoder is not the same as destroying the encoder: an + // aliased role can still hold it, and its workers can still be draining. + // Whatever survives holds a callback registration this session installed, + // so the registration has to come down here rather than be left to the + // encoder's own teardown. + // + // The detach is a quiesce point -- it returns only once any in-flight + // invocation has returned -- and it hands the cookie back to + // TrampolineRelease, which frees it and runs the client release. No lock + // is held across it, for the reason SetCallback explains. + ~SessionImpl() override { - case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: - case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: - case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: { + std::lock_guard lock(m_callbackMutex); + m_closing = true; + } + if (m_encoder) { + // A refused detach cannot be reported from here and cannot be + // retried by abandoning the destructor. It is survivable rather + // than silent: the cookie belongs to the encoder either way, so + // refusing to detach leaks a registration, not a dangling one. + (void)m_encoder->SetCompletionCallback(nullptr, nullptr, nullptr); + } + } + void* QueryInterface(std::string_view id) override + { + if (id == IEncoderSession::kId) return static_cast(this); + if (id == IFrameSubmitter::kId) return static_cast(this); + if (id == IBitstreamSource::kId) return static_cast(this); + if (id == ICompletionSignal::kId) return static_cast(this); + if (id == IResourceRegistry::kId) return static_cast(this); + if (id == IDeviceBinding::kId) return static_cast(this); + if (id == IDiagnostics::kId) return static_cast(this); + if (id == IShutdownDiagnostics::kId) + return static_cast(this); + return nullptr; + } + + //---- IEncoderSession ---------------------------------------------------- + + // Apply what the session can carry and name what it cannot. + // + // Only rate control and frame rate can move mid-stream. Everything else is + // settled in the sequence header written once, or in the input routing the + // session was built around, so it needs a new session. The library holds + // the configuration the session was created with and compares against it, + // which is what lets a caller hand back either the object it built the + // session from or a fresh one carrying only the change. + // + // A field left at its unset value means "leave this alone" rather than + // "set this to zero" -- otherwise a caller moving only the frame rate + // would have to restate the resolution, the format and the bitrate to + // avoid being refused for changing them. + Result Reconfigure(const Ref& config) override + { + if (!config) { + return Fail(Result(ResultCode::InvalidArgument, "no configuration")); } - break; + Ref access = Query(config); + if (!access) { + return Fail(Result(ResultCode::InvalidArgument, "foreign configuration object")); + } + const VkVideoEncoderConfig& in = access->ExtConfig(); - default: - assert(!"Unsupported codec type!!!\n"); - return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + std::lock_guard lock(m_configMutex); + + if (const char* immutable = FirstImmutableChange(in)) { + RecordDetail(immutable); + return Fail(Result(ResultCode::UnsupportedFeature, + "this field cannot change without a new session")); + } + + // Start from what the session is running, and overlay only what moves. + VkVideoEncoderConfig next = m_config; + next.pNext = nullptr; // a chain is refused; the session already has its metadata + + if (in.averageBitrate != 0) { + next.averageBitrate = in.averageBitrate; + } + // Zero here is coerced to the average rather than meaning "no cap", so + // an unstated maximum carries forward instead of silently capping the + // session at its own average. + if (in.maxBitrate != 0) { + next.maxBitrate = in.maxBitrate; + } + if (in.frameRateNum != 0) { + next.frameRateNum = in.frameRateNum; + next.frameRateDen = (in.frameRateDen != 0) ? in.frameRateDen : 1; + } + // A negative constant-QP member names no quantizer; zero is a valid + // (lossless) one, so the two cannot be collapsed. + if (in.constQpI >= 0) { next.constQpI = in.constQpI; } + if (in.constQpP >= 0) { next.constQpP = in.constQpP; } + if (in.constQpB >= 0) { next.constQpB = in.constQpB; } + next.minQp = in.minQp; + next.maxQp = in.maxQp; + + if (next.averageBitrate == 0 && + next.rateControlMode != VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR) { + return Fail(Result(ResultCode::InvalidArgument, + "a rate-controlled session needs a non-zero average bitrate")); + } + + const VkResult result = m_encoder->Reconfigure(next); + if (result != VK_SUCCESS) { + return Fail(FromVkResult(result, "the session refused this reconfiguration")); + } + m_config = next; + return Result(); } - VkSharedBaseObj vulkanVideoEncoderObj( new VulkanVideoEncoderImpl()); - if (!vulkanVideoEncoderObj) { - return VK_ERROR_OUT_OF_HOST_MEMORY; + Result Drain() override + { + return Fail(FromVkResult(m_encoder->DrainPendingFrames(), + "the drain did not complete")); } - VkResult result = vulkanVideoEncoderObj->Initialize(videoCodecOperation, argc, argv); + Result Finish() override + { + // The underlying Flush is the terminal one: it releases the encoder on + // the way out, which is why Drain above is a different call and not a + // parameter of this one. + return Fail(FromVkResult(m_encoder->Flush(), "the stream could not be ended")); + } + + Expected AbandonAll() override + { + uint32_t abandoned = 0; + const VkResult result = m_encoder->AbandonAllFrames(&abandoned); + if (result != VK_SUCCESS) { + return Fail(FromVkResult(result, "abandon failed")); + } + return abandoned; + } + + //---- IFrameSubmitter ---------------------------------------------------- + + Result SubmitFrame(const FrameSubmit& frame) override + { + // The descriptor API takes parallel arrays; this interface takes one array of + // pairs, which is the shape that cannot go out of step. + std::vector waitSems, signalSems; + std::vector waitVals, signalVals; + Unpack(frame.waitSemaphores, waitSems, waitVals); + Unpack(frame.signalSemaphores, signalSems, signalVals); + + if (frame.registeredImage != kNoResource) { + return SubmitRegistered(frame, waitSems, waitVals, signalSems, signalVals); + } + + // Same fence pair as the registered path: an inline image has the same + // producer-ordering problem, and answering it on only one of the two + // paths would make the choice of path change the semantics. + VkVideoEncoderFrameFenceDescriptor fences; + fences.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR; + fences.pNext = nullptr; + fences.acquireFenceFd = frame.acquireFenceFd; + fences.pReleaseFenceFd = frame.releaseFenceFd; + const bool wantFences = + (frame.acquireFenceFd >= 0) || (frame.releaseFenceFd != nullptr); + + VkVideoEncodeInputFrame in{}; + in.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME; + in.pNext = wantFences ? &fences : nullptr; + in.image = frame.image.existingImage; + in.format = frame.image.format; + in.width = frame.image.width; + in.height = frame.image.height; + in.imageTiling = frame.image.tiling; + in.currentLayout = frame.image.layout; + in.frameId = frame.frameId; + in.pts = frame.pts; + in.forceIDR = frame.forceIdr ? VK_TRUE : VK_FALSE; + in.isLastFrame = frame.isLastFrame ? VK_TRUE : VK_FALSE; + in.qpOverride = LowerQpOverride(frame); + + in.waitSemaphoreCount = static_cast(waitSems.size()); + in.pWaitSemaphores = waitSems.empty() ? nullptr : waitSems.data(); + in.pWaitSemaphoreValues = waitVals.empty() ? nullptr : waitVals.data(); + in.signalSemaphoreCount = static_cast(signalSems.size()); + in.pSignalSemaphores = signalSems.empty() ? nullptr : signalSems.data(); + in.pSignalSemaphoreValues = signalVals.empty() ? nullptr : signalVals.data(); + + const VkResult result = m_encoder->SubmitExternalFrame(in, nullptr); + if (result == VK_SUCCESS) { + ++m_framesSubmitted; + } + return Fail(FromVkResult(result, "frame submission failed")); + } + + // A frame whose image was registered earlier. The registration already + // holds the import, so this path costs no image creation -- which is the + // reason a compositor registers at all. + Result SubmitRegistered(const FrameSubmit& frame, + const std::vector& waitSems, + const std::vector& waitVals, + const std::vector& signalSems, + const std::vector& signalVals) + { + // The fence pair is a chained structure on this API. It lives for the + // duration of the call, which is all the library needs: it consumes + // the acquire fd and writes the release fd synchronously. + VkVideoEncoderFrameFenceDescriptor fences; + fences.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR; + fences.pNext = nullptr; + fences.acquireFenceFd = frame.acquireFenceFd; + fences.pReleaseFenceFd = frame.releaseFenceFd; + const bool wantFences = + (frame.acquireFenceFd >= 0) || (frame.releaseFenceFd != nullptr); + + VkVideoEncoderFrameSubmitInfo info; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = static_cast(frame.registeredImage); + info.frameId = frame.frameId; + info.pts = frame.pts; + info.forceIDR = frame.forceIdr ? VK_TRUE : VK_FALSE; + info.isLastFrame = frame.isLastFrame ? VK_TRUE : VK_FALSE; + info.qpOverride = LowerQpOverride(frame); + // UNDEFINED asks the encoder to use the layout declared at + // registration, which is what a caller that has not moved the image + // wants. + info.currentLayout = frame.image.layout; + info.pNext = wantFences ? &fences : nullptr; + + info.waitSemaphoreCount = static_cast(waitSems.size()); + info.pWaitSemaphores = waitSems.empty() ? nullptr : waitSems.data(); + info.pWaitSemaphoreValues = waitVals.empty() ? nullptr : waitVals.data(); + info.signalSemaphoreCount = static_cast(signalSems.size()); + info.pSignalSemaphores = signalSems.empty() ? nullptr : signalSems.data(); + info.pSignalSemaphoreValues = signalVals.empty() ? nullptr : signalVals.data(); + + const VkVideoEncoderStatusCode code = + m_encoder->SubmitRegisteredFrame(info, nullptr); + if (code == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + ++m_framesSubmitted; + } + RecordCode("registered frame submission failed", code); + return Fail(FromExtStatusCode(code, "registered frame submission failed")); + } + + Result CancelFrame(uint64_t frameId) override + { + return Fail(FromVkResult(m_encoder->CancelFrame(frameId), "frame is not cancellable")); + } + + FrameState GetFrameState(uint64_t frameId) const override + { + return FromExtFrameState(m_encoder->GetFrameStatus(frameId)); + } + + //---- IBitstreamSource --------------------------------------------------- + + Expected AcquireNext() override + { + VkVideoEncodeResult out{}; + out.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_ENCODE_RESULT; + const VkResult result = m_encoder->AcquireNextEncodedFrame(out); + if (result != VK_SUCCESS) { + return Fail(FromVkResult(result, "no encoded frame is ready")); + } + ++m_framesEncoded; + return FromExtResult(out); + } + + Expected Acquire(uint64_t frameId) override + { + VkVideoEncodeResult out{}; + out.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_ENCODE_RESULT; + const VkResult result = m_encoder->AcquireEncodedFrame(frameId, out); + if (result != VK_SUCCESS) { + return Fail(FromVkResult(result, "that frame is not ready")); + } + ++m_framesEncoded; + return FromExtResult(out); + } + + void Release(uint64_t frameId) override + { + m_encoder->ReleaseEncodedFrame(frameId); + } + + //---- ICompletionSignal -------------------------------------------------- + + Result SetCallback(CompletionCallback callback, void* userData, + UserDataRelease release) override + { + // NEVER HOLD m_callbackMutex ACROSS THE LIBRARY CALL. Installing or + // clearing a callback is a quiesce point: it returns only once any + // in-flight invocation has returned. If that invocation were waiting + // on a lock held here, the two would wait on each other. The + // trampolines below take no adapter lock at all, which is what keeps + // that true even while a replacement is in progress. + // + // Keep this session and its encoder alive for the whole method. A + // client release runs synchronously inside the lower call, and a + // release handler that drops the caller's last Ref would otherwise + // destroy the object executing this line. + const Ref self = shared_from_this(); + const VkSharedBaseObj encoder = m_encoder; + { + std::lock_guard lock(m_callbackMutex); + if (m_closing) { + return Fail(Result(ResultCode::NotConfigured, + "the session is shutting down")); + } + } + + if (callback == nullptr) { + // Detach at the library. Once it returns, no trampoline can be + // running and the cookie has been handed back to + // TrampolineRelease -- which is what makes the promise that a + // caller may destroy whatever userData referenced once this + // returns. + // + // A refused detach leaves the OLD registration installed. Do not + // record the session as callback-free in that case: the caller + // must be told its cookie is still reachable. + const Result detached = Fail(FromVkResult( + encoder->SetCompletionCallback(nullptr, nullptr, nullptr), + "the completion callback could not be cleared")); + if (detached.ok()) { + std::lock_guard lock(m_callbackMutex); + m_hasCallback = false; + } + return detached; + } + + // The cookie is a state object the lower layer owns, never this + // session. Allocate it before the detach so a failed allocation + // cannot leave the session with no callback at all. + CallbackState* state = new (std::nothrow) + CallbackState(callback, userData, release); + if (state == nullptr) { + return Fail(Result(ResultCode::OutOfResources, + "the completion callback state could not be " + "allocated")); + } + + // Replacing one callback with another goes through a detach, so no + // in-flight invocation can straddle the change and deliver the new + // cookie to the old callback. The library sees the same trampoline + // pointer either way and would not quiesce on its own. + bool hadCallback = false; + { + std::lock_guard lock(m_callbackMutex); + hadCallback = m_hasCallback; + } + if (hadCallback) { + // A refused detach means the old registration is still live. + // Installing over it would strand the old cookie, whose release + // the encoder would then never call. Refuse, and destroy the + // state this call allocated rather than the one still in use. + const Result detached = Fail(FromVkResult( + encoder->SetCompletionCallback(nullptr, nullptr, nullptr), + "the previous completion callback could not be replaced")); + if (!detached.ok()) { + delete state; + return detached; + } + std::lock_guard lock(m_callbackMutex); + m_hasCallback = false; + } + + // ALWAYS install the release thunk, even when the client supplied no + // release of its own: the thunk is what frees the state object. With + // nullptr here the encoder would drop the cookie on the floor at + // detach and every SetCallback would leak one. + const Result installed = Fail(FromVkResult( + encoder->SetCompletionCallback(&SessionImpl::TrampolineCallback, + state, + &SessionImpl::TrampolineRelease), + "the completion callback could not be installed")); + if (!installed.ok()) { + // Ownership transfers only on success. A refused installation + // never reached the encoder, so nothing else will free this. + delete state; + return installed; + } + { + std::lock_guard lock(m_callbackMutex); + m_hasCallback = true; + } + return installed; + } + + uint64_t CompletedCount() const override + { + return m_encoder->GetCompletionCounter(); + } + + //---- IShutdownDiagnostics ----------------------------------------------- + + ShutdownSnapshot Shutdown() const override + { + VkVideoEncoderShutdownInfo info{}; + m_encoder->GetShutdownInfo(&info); + + ShutdownSnapshot out; + switch (info.disposition) { + case VK_VIDEO_ENCODER_SHUTDOWN_IDLE: + out.disposition = ShutdownDisposition::Idle; + break; + case VK_VIDEO_ENCODER_SHUTDOWN_LOST_DEVICE_RETIRED: + out.disposition = ShutdownDisposition::LostDeviceRetired; + break; + default: + out.disposition = ShutdownDisposition::Unproven; + break; + } + out.firstError = FromVkResult(info.firstError, "").code(); + out.workersJoined = (info.workersJoined != VK_FALSE); + out.callbackDetached = (info.callbackDetached != VK_FALSE); + out.deviceLostObserved = (info.deviceLostObserved != VK_FALSE); + out.complete = (info.shutdownComplete != VK_FALSE); + return out; + } + + VkSemaphore CompletionSemaphore() const override + { + return m_encoder->GetCompletionSemaphore(); + } + + Expected ExportCompletionHandle() override + { + uint64_t handle = 0; + const VkVideoEncoderStatusCode code = m_encoder->GetCompletionEventHandle(&handle); + if (code != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return Fail(FromExtStatusCode(code, "no completion handle is available")); + } + // The lower layer's handle is BORROWED: the encoder signals through it + // and closes it at teardown. The interface hands the caller a close + // obligation, so each export must be a handle of its own. Returning the + // borrowed one makes a caller that honours the documented contract close + // the encoder's live event -- after which completions are written to a + // descriptor the process has since reused, and teardown closes it twice. + const uint64_t duplicate = vkenc::OsCompletionEventDuplicate(handle); + if (duplicate == vkenc::kOsCompletionEventNone) { + return Fail(Result(ResultCode::OutOfResources, + "the completion handle could not be duplicated for export")); + } + return static_cast(duplicate); + } + + //---- IResourceRegistry -------------------------------------------------- + + Expected RegisterImage(const ExternalImage& image, + bool* outHandleConsumed) override + { + if (outHandleConsumed != nullptr) { + *outHandleConsumed = false; + } + VkVideoEncoderExternalImageDescriptor desc; + uint64_t osHandle = 0; + ToExtImageDescriptor(image, desc, osHandle); + + // The echo is written on every return, so it is read on every return. + VkVideoEncoderStatus echo; + echo.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; + echo.pNext = nullptr; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode code = + m_encoder->RegisterImageResource(desc, osHandle, &resource, &echo); + + // Written before the refusal below, not after it: the echo's whole + // purpose is to be true on a path that returns no value. + if (outHandleConsumed != nullptr) { + *outHandleConsumed = (echo.handlesConsumed != VK_FALSE); + } + + if (code != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + RecordCode("image could not be registered", code); + return Fail(FromExtStatusCode(code, "image could not be registered")); + } + return static_cast(resource); + } + + Result UnregisterImage(ResourceId id) override + { + if (id == kNoResource) { + return Fail(Result(ResultCode::InvalidArgument, "no such registered image")); + } + return Fail(FromExtStatusCode( + m_encoder->UnregisterImageResource(static_cast(id)), + "image could not be unregistered")); + } + + Expected QueryImageSupport(const ExternalImage& image) override + { + VkVideoEncoderExternalImageDescriptor desc; + uint64_t osHandle = 0; + ToExtImageDescriptor(image, desc, osHandle); + + VkVideoEncoderImageSupport support{}; + support.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT; + const VkVideoEncoderStatusCode code = + m_encoder->QueryImageSupport(desc, &support); + if (code != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return Fail(FromExtStatusCode(code, "this image cannot be imported")); + } + // TWO ANSWERS, AND BOTH MEAN NO. The call reports whether it could be + // answered; the reply reports whether the image is usable. A query + // that was answered "no" returns SUCCESS, so reading only the status + // turns a refusal into an acceptance. + if (support.supported != VK_TRUE) { + return Fail(Result(ResultCode::UnsupportedFormat, + "the encoder cannot take this image")); + } + const VkEncInputFormatClass cls = + VkEncClassifyInput(image.format, ToExtColorModel(image.colorModel)); + return (cls == VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT) ? InputPath::Copy + : InputPath::ComputeFilter; + } + + Expected RegisterSemaphore(const ExternalSemaphore& semaphore) override + { + VkVideoEncoderSemaphoreDescriptor desc; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_SEMAPHORE_DESCRIPTOR; + desc.pNext = nullptr; + desc.handleType = ToExtHandleType(semaphore.handleType); + // Timeline only: a binary semaphore cannot report a completion to a + // consumer that was not already waiting when it was signalled. + desc.semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + desc.ownership = ToExtOwnership(semaphore.ownership); + + uint64_t osHandle = 0; + switch (semaphore.handleType) { + case ExternalHandleType::OpaqueFd: + osHandle = static_cast(static_cast(semaphore.fd)); + break; + case ExternalHandleType::OpaqueWin32: + osHandle = reinterpret_cast(semaphore.win32Handle); + break; + default: + return Fail(Result(ResultCode::InvalidArgument, + "a semaphore needs an importable handle type")); + } + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode code = + m_encoder->RegisterSemaphore(desc, osHandle, &resource, nullptr); + if (code != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return Fail(FromExtStatusCode(code, "semaphore could not be registered")); + } + return static_cast(resource); + } + + Result UnregisterSemaphore(ResourceId id) override + { + if (id == kNoResource) { + return Fail(Result(ResultCode::InvalidArgument, "no such registered semaphore")); + } + return Fail(FromExtStatusCode( + m_encoder->UnregisterSemaphore(static_cast(id)), + "semaphore could not be unregistered")); + } + + //---- IDeviceBinding ----------------------------------------------------- + // + // A session runs on a device, so it can say which. The platform answers + // the same question for a caller that has not created a session yet; a + // caller that has one should not have to keep the platform alive, or hold + // a second object, to ask what device its own session is on. + + VkInstance Instance() const override { return m_encoder->GetVkInstance(); } + VkPhysicalDevice PhysicalDevice() const override + { + return m_encoder->GetVkPhysicalDevice(); + } + VkDevice Device() const override { return m_encoder->GetVkDevice(); } + + // The queue families are the session's own business and it does not + // publish them; a caller that supplied them already knows what it gave. + uint32_t EncodeQueueFamilyIndex() const override { return UINT32_MAX; } + uint32_t ComputeQueueFamilyIndex() const override { return UINT32_MAX; } + + PFN_vkGetInstanceProcAddr GetInstanceProcAddr() const override + { + return m_encoder->GetVkGetInstanceProcAddr(); + } + + bool DeviceUuid(uint8_t outUuid[VK_UUID_SIZE]) const override + { + return VkEncQueryDeviceUuid(m_encoder->GetVkInstance(), + m_encoder->GetVkPhysicalDevice(), + m_encoder->GetVkGetInstanceProcAddr(), + outUuid); + } + + //---- IDiagnostics ------------------------------------------------------- + + RuntimeInfo GetRuntimeInfo() const override + { + RuntimeInfo out; + VkVideoEncoderRuntimeInfo info{}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_RUNTIME_INFO; + if (m_encoder->GetRuntimeInfo(&info) != VK_SUCCESS) { + return out; + } + // The source is a fixed-width field that need not be terminated, so + // one byte is reserved for the terminator this side guarantees. + const size_t room = sizeof(out.implementationName) - 1; + std::copy(info.implementationName, info.implementationName + room, + out.implementationName); + out.implementationName[room] = '\0'; + out.isHardwareAccelerated = info.isHardwareAccelerated != VK_FALSE; + out.supportsNativeHandle = info.supportsNativeHandle != VK_FALSE; + out.trustedRateController = info.trustedRateController != VK_FALSE; + out.supportsSimulcast = info.supportsSimulcast != VK_FALSE; + out.supportsFrameSizeChange = info.supportsFrameSizeChange != VK_FALSE; + out.reportsAverageQp = info.reportsAverageQp != VK_FALSE; + out.applyAlignmentToAllSimulcastLayers = + info.applyAlignmentToAllSimulcastLayers != VK_FALSE; + out.resolutionAlignmentWidth = + info.requestedResolutionAlignmentWidth ? info.requestedResolutionAlignmentWidth : 1; + out.resolutionAlignmentHeight = + info.requestedResolutionAlignmentHeight ? info.requestedResolutionAlignmentHeight : 1; + return out; + } + + const char* LastErrorDetail() const override + { + std::lock_guard lock(m_errorMutex); + return m_lastError.c_str(); + } + + Expected GetCompletionStats() const override + { + VkVideoEncoderCompletionInfo info; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + + // The diagnostics ride in on the chain here so the caller receives one + // value rather than having to know that two structures exist. + VkVideoEncoderDiagnosticInfo diagnostics; + diagnostics.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_DIAGNOSTIC_INFO; + diagnostics.pNext = nullptr; + diagnostics.diagnosticCount = 0; + diagnostics.lastDiagnostic[0] = '\0'; + info.pNext = &diagnostics; + + const VkResult result = m_encoder->GetCompletionInfo(&info); + if (result != VK_SUCCESS) { + return Fail(FromVkResult(result, "the session has no completion accounting")); + } + + CompletionStats stats; + stats.completionCounter = info.completionCounter; + stats.framesTimedOut = info.framesTimedOut; + stats.lateCaptures = info.lateCaptures; + stats.framesCancelled = info.framesCancelled; + stats.framesPending = info.framesPending; + stats.framesReady = info.framesReady; + stats.framesAcquired = info.framesAcquired; + stats.diagnosticCount = diagnostics.diagnosticCount; + + const size_t room = sizeof(stats.lastDiagnostic) - 1; + const size_t take = (VK_VIDEO_ENCODER_MAX_DIAGNOSTIC_CHARS - 1 < room) + ? VK_VIDEO_ENCODER_MAX_DIAGNOSTIC_CHARS - 1 + : room; + std::copy(diagnostics.lastDiagnostic, diagnostics.lastDiagnostic + take, + stats.lastDiagnostic); + stats.lastDiagnostic[take] = '\0'; + return stats; + } + + uint64_t FramesSubmitted() const override { return m_framesSubmitted; } + uint64_t FramesEncoded() const override { return m_framesEncoded; } + +private: + // The first field that is both stated by the caller and different from + // what the session runs, or null when nothing immutable moves. An unset + // field is not a change: it is the absence of one. + const char* FirstImmutableChange(const VkVideoEncoderConfig& in) const + { + if (in.codec != m_config.codec) { return "codec"; } + if (in.profile != m_config.profile) { return "profile"; } + if (in.encodeWidth != 0 && in.encodeWidth != m_config.encodeWidth) { + return "coded width"; + } + if (in.encodeHeight != 0 && in.encodeHeight != m_config.encodeHeight) { + return "coded height"; + } + if (in.inputFormat != VK_FORMAT_UNDEFINED && + in.inputFormat != m_config.inputFormat) { + return "input format"; + } + if (in.rateControlMode != VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR && + in.rateControlMode != m_config.rateControlMode) { + return "rate control mode"; + } + // 2 is "unspecified" in every one of these code points, and is what an + // untouched configuration carries. + if (in.colourPrimaries != 2 && in.colourPrimaries != m_config.colourPrimaries) { + return "colour primaries"; + } + if (in.transferCharacteristics != 2 && + in.transferCharacteristics != m_config.transferCharacteristics) { + return "transfer characteristics"; + } + if (in.matrixCoefficients != 2 && + in.matrixCoefficients != m_config.matrixCoefficients) { + return "matrix coefficients"; + } + if (in.gopLength != 0 && in.gopLength != m_config.gopLength) { + return "GOP length"; + } + if (in.qualityLevel != 0 && in.qualityLevel != m_config.qualityLevel) { + return "quality level"; + } + return nullptr; + } + + void RecordDetail(const char* what) const + { + std::lock_guard lock(m_errorMutex); + m_lastError = std::string("cannot change ") + what + " without a new session"; + m_lastErrorPinned = true; + } + + // The encoder hands back the cookie it was given, which is a + // CallbackState and never this session. The lower layer owns it from a + // successful SetCompletionCallback until it hands it here, exactly once, + // after any in-flight invocation has returned -- on replacement, on + // detach, or at encoder destruction. This is the only place it is freed. + static void TrampolineRelease(void* userData) VK_VIDEO_ENCODER_CB_NOEXCEPT + { + CallbackState* state = static_cast(userData); + if (state == nullptr) { + return; + } + if (state->release != nullptr) { + state->release(state->userData); + } + delete state; + } + + // Reads immutable state through a pointer the encoder guarantees is live + // for the duration of the call. It deliberately touches no session member + // and takes no session lock: a session destroyed while the encoder is + // still draining must not turn a completion into a use-after-free, and a + // trampoline that waited on an adapter lock would deadlock against the + // quiesce point in SetCallback. + static void TrampolineCallback(uint64_t frameId, void* userData) + VK_VIDEO_ENCODER_CB_NOEXCEPT + { + const CallbackState* state = static_cast(userData); + if (state != nullptr && state->callback != nullptr) { + state->callback(frameId, state->userData); + } + } + + // The ONE place the public flag/value pair becomes a lower-layer field. + // + // -1 is the lower layer's "no override". Zero is not: it is a legitimate + // constant QP a caller may ask for deliberately, which is exactly why the + // public contract carries a separate hasQpOverride flag. The two arms + // disagreed here -- the inline arm sent 0 for an unset override, so a + // caller that never asked for one had every inline frame encoded at QP 0 + // while the same submission through the registered arm behaved correctly. + // + // The sentinel is documented at this boundary and nowhere else: the + // public default stays zero, and an intentional QP of zero still reaches + // the encoder as zero. + static int32_t LowerQpOverride(const FrameSubmit& frame) + { + return frame.hasQpOverride ? frame.qpOverride : -1; + } + + static void Unpack(const ArrayView& in, + std::vector& sems, + std::vector& values) + { + sems.reserve(in.size()); + values.reserve(in.size()); + for (size_t i = 0; i < in.size(); ++i) { + sems.push_back(in[i].semaphore); + values.push_back(in[i].value); + } + } + + static EncodedFrame FromExtResult(const VkVideoEncodeResult& in) + { + EncodedFrame out; + out.frameId = in.frameId; + out.pts = in.pts; + out.dts = in.dts; + out.bitstream = ArrayView(in.pBitstreamData, in.bitstreamSize); + out.pictureType = FromExtPictureType(in.pictureType); + out.isIdr = in.isIDR != VK_FALSE; + out.temporalLayerId = in.temporalLayerId; + // A deadline drop is delivered like any other frame and carries no + // bitstream, so the outcome travels with it rather than being inferred + // from an empty view. + out.outcome = FromVkResult(in.status, "").code(); + return out; + } + + // The status code a Result cannot carry -- its detail is a literal -- + // recorded where IDiagnostics can hand it back. This is the split the + // interface promises: literal in Result, runtime context in IDiagnostics. + void RecordCode(const char* what, VkVideoEncoderStatusCode code) const + { + if (code == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return; + } + std::lock_guard lock(m_errorMutex); + m_lastError = std::string(what) + " (encoder status " + + std::to_string(static_cast(code)) + ")"; + m_lastErrorPinned = true; + } + + // Every failing Result also lands in LastErrorDetail, which is where the + // runtime context Result deliberately omits can be recovered from. + const Result& Fail(const Result& status) const + { + if (!status.ok()) { + std::lock_guard lock(m_errorMutex); + if (m_lastErrorPinned) { + // RecordCode already wrote the fuller message for this + // failure; do not overwrite it with the literal. + m_lastErrorPinned = false; + } else { + m_lastError = status.detail(); + } + } + return status; + } + + VkSharedBaseObj m_encoder; + + // What the session is running. Reconfigure overlays onto this rather than + // replacing it, so a caller may hand back a configuration carrying only + // the change. + std::mutex m_configMutex; + VkVideoEncoderConfig m_config; + + mutable std::mutex m_errorMutex; + mutable std::string m_lastError; + mutable bool m_lastErrorPinned = false; + + // What the lower layer holds while a callback is installed. Immutable + // once constructed, so the trampolines read it without a lock; owned by + // the encoder, so it outlives this session whenever the encoder does. + struct CallbackState { + CallbackState(CompletionCallback cb, void* data, UserDataRelease rel) + : callback(cb), userData(data), release(rel) { } + + const CompletionCallback callback; + void* const userData; + const UserDataRelease release; + }; + + // Guards only the two bookkeeping bits below. The callback itself is not + // reachable from here on purpose -- see TrampolineCallback. + std::mutex m_callbackMutex; + bool m_hasCallback = false; + bool m_closing = false; + + uint64_t m_framesSubmitted = 0; + uint64_t m_framesEncoded = 0; +}; + +//============================================================================= +// PLATFORM +//============================================================================= + +class PlatformImpl final : public IEncoderPlatform, + public IPlatformLifetime { +public: + PlatformImpl(const PlatformCreateInfo& info, + const VkSharedBaseObj& context) + : m_info(info) + , m_context(context) + , m_caps(std::make_shared(context.get(), + info.deviceIndex < 0 ? 0u + : static_cast(info.deviceIndex))) + , m_binding(std::make_shared(info)) + { } + + void* QueryInterface(std::string_view id) override + { + if (id == IEncoderPlatform::kId) { + return static_cast(this); + } + if (id == IPlatformLifetime::kId) { + return static_cast(this); + } + return nullptr; + } + + // Releases the process-wide floor reference, not this platform: the + // instance being destroyed is shared by every platform built for this + // device, which is why retiring it is a request a caller has to make + // rather than something a single platform's destructor may do. + uint32_t Retire() override { return VkEncRetireOwnContexts(); } + + Ref Caps() const override { return m_caps; } + Ref DeviceBinding() const override { return m_binding; } + + Expected> CreateConfig(Codec codec, Profile profile) override + { + if (!ProfileMatchesCodec(codec, profile)) { + return Result(ResultCode::UnsupportedProfile, + "profile does not belong to this codec"); + } + Ref config = std::make_shared(codec, profile, m_info); + return config; + } + + Expected> CreateSession(const Ref& config) override + { + if (!config) { + return Result(ResultCode::InvalidArgument, "no configuration"); + } + Ref access = Query(config); + if (!access) { + return Result(ResultCode::InvalidArgument, "foreign configuration object"); + } + if (Result valid = config->Validate(); !valid) { + return valid; + } + + // ON THE PLATFORM'S CONTEXT, not beside it. The platform IS the device + // scope: it holds the context that answered every capability question + // above, and a session created independently would build a SECOND + // Vulkan instance and device of its own. + // + // In a process that already owns Vulkan -- a compositor, a renderer, + // anything that draws what it encodes -- that second device is not + // merely wasteful. It is another instance in a process that has one, + // and initialisation fails. + VkSharedBaseObj encoder; + const uint32_t deviceIndex = + (m_info.deviceIndex < 0) ? 0u : static_cast(m_info.deviceIndex); + if (VkResult result = + CreateVulkanVideoEncoderExtOnContext(m_context, deviceIndex, encoder); + result != VK_SUCCESS) { + return FromVkResult(result, "the encoder could not be created"); + } + + // The HDR chain is built here and lives only for the duration of the + // call, so nothing dangles on a configuration the caller keeps. + VkVideoEncoderConfig extConfig = access->ExtConfig(); + // deviceId stays at its "no selection" value: the context already + // chose the physical device, and a session built on one refuses a + // second selector rather than letting a config field silently outrank + // the device whose capabilities the caller queried. + VkVideoEncoderHdrMetadataInfo hdr; + if (access->HasHdrMetadata()) { + const HdrMetadata& src = access->GetHdrMetadata(); + hdr = VkVideoEncoderHdrMetadataInfo(); + hdr.masteringDisplayPresent = src.masteringDisplayPresent ? VK_TRUE : VK_FALSE; + for (int i = 0; i < 3; ++i) { + hdr.displayPrimaryX[i] = src.displayPrimaryX[i]; + hdr.displayPrimaryY[i] = src.displayPrimaryY[i]; + } + hdr.whitePointX = src.whitePointX; + hdr.whitePointY = src.whitePointY; + hdr.maxDisplayMasteringLuminance = src.maxLuminance; + hdr.minDisplayMasteringLuminance = src.minLuminance; + hdr.contentLightLevelPresent = src.contentLightLevelPresent ? VK_TRUE : VK_FALSE; + hdr.maxContentLightLevel = src.maxContentLightLevel; + hdr.maxFrameAverageLightLevel = src.maxFrameAverageLightLevel; + extConfig.pNext = &hdr; + } + + // Chained onto a copy that lives only for this call: the retained + // baseline a reconfiguration overlays onto must carry no chain. + VkVideoEncoderValidationInfo validation; + if (access->RequiresImportExtensions()) { + validation = VkVideoEncoderValidationInfo(); + validation.flags = VK_VIDEO_ENCODER_VALIDATE_EXTENSIONS_BIT; + validation.pNext = extConfig.pNext; + extConfig.pNext = &validation; + } + + if (VkResult result = encoder->InitializeExt(extConfig); result != VK_SUCCESS) { + return FromVkResult(result, "the encoder refused this configuration"); + } + + m_binding->BindSession(encoder); + + // The session keeps the configuration it was created with, so + // Reconfigure can overlay onto it. The chain is NOT kept: it points at + // a local that dies with this call, and the metadata it carried is + // already written into the session's parameter sets. + VkVideoEncoderConfig retained = extConfig; + retained.pNext = nullptr; + Ref session = std::make_shared(encoder, retained); + return session; + } + +private: + PlatformCreateInfo m_info; + VkSharedBaseObj m_context; + Ref m_caps; + Ref m_binding; +}; + +} // namespace enc +} // namespace video +} // namespace vk + +//============================================================================= +// FORMAT CLASSIFICATION +// +// The four questions a host asks about a format before it allocates -- can you +// take this, by which route, what does the filter need of my image, and what +// will the session run at -- answered once, from the library's own routing +// tables rather than from a copy of them in the caller. +//============================================================================= + +extern "C" VK_ENC_EXPORT +VkResult VkEncClassifyFormat(VkFormat format, + vk::video::enc::ColorModel colorModel, + vk::video::enc::FormatRouting* outRouting) +{ + using namespace vk::video::enc; + + if (outRouting == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + *outRouting = FormatRouting(); + + const VkVideoEncoderColorModel model = ToExtColorModel(colorModel); + const VkEncInputFormatClass cls = VkEncClassifyInput(format, model); + + if (cls == VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + outRouting->supported = true; + + if (cls == VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT) { + // The encoder reads these samples as they are. A copy may still move + // them, but nothing rewrites one. + outRouting->path = InputPath::Copy; + outRouting->filterAccess = FilterAccess::NoFilter; + outRouting->sessionFormat = format; + return VK_SUCCESS; + } + + outRouting->path = InputPath::ComputeFilter; + + // How the filter must reach the source. A planar source is addressed one + // plane at a time; a single-plane source -- packed YCbCr or RGB -- is read + // as one storage image. The distinction decides what usage and what view + // formats the caller's image needs, which is why it is answered here + // rather than inferred by each host from a format list of its own. + outRouting->filterAccess = (VkEncInputFormatPlaneCount(format) > 1) + ? FilterAccess::PlaneStorage + : FilterAccess::StorageRead; + + // Without a device list the conversion target is knowable only where it + // does not depend on one. It does for RGB, whose session takes the + // device's first advertised encode-source format. + outRouting->sessionFormat = + VkEncConversionTargetFormat(format, nullptr, 0); + + return VK_SUCCESS; +} + +//============================================================================= +// PLATFORM CREATION -- THE SECOND EXPORTED SYMBOL +//============================================================================= + +extern "C" VK_ENC_EXPORT +VkResult VkEncCreatePlatform(const vk::video::enc::PlatformCreateInfo& createInfo, + vk::video::enc::Ref& outPlatform) +{ + using namespace vk::video::enc; + + VkVideoEncoderContextCreateInfo contextInfo{}; + contextInfo.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONTEXT_CREATE_INFO; + contextInfo.pNext = nullptr; + + // Adopting the caller's instance is what keeps the encoder on the device + // that already owns the images; creating one is for the standalone tools. + const bool adopt = (createInfo.instance != VK_NULL_HANDLE); + contextInfo.mode = adopt ? VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT + : VK_VIDEO_ENCODER_CONTEXT_MODE_OWN; + contextInfo.adoptInstance = createInfo.instance; + contextInfo.adoptPhysicalDevice = createInfo.physicalDevice; + contextInfo.silenceStdio = createInfo.silenceStdio ? VK_TRUE : VK_FALSE; + if (createInfo.gpuUuidValid) { + std::copy(std::begin(createInfo.gpuUuid), std::end(createInfo.gpuUuid), + std::begin(contextInfo.gpuUUID)); + } + + VkSharedBaseObj context; + const VkResult result = CreateVulkanVideoEncoderContext(&contextInfo, context); if (result != VK_SUCCESS) { - vulkanVideoEncoderObj = nullptr; - } else { - vulkanVideoEncoder = vulkanVideoEncoderObj; + return result; } - return result; + outPlatform = std::make_shared(createInfo, context); + return VK_SUCCESS; } diff --git a/vk_video_encoder/src/vulkan_video_encoder_argv.cpp b/vk_video_encoder/src/vulkan_video_encoder_argv.cpp new file mode 100644 index 00000000..6db7854d --- /dev/null +++ b/vk_video_encoder/src/vulkan_video_encoder_argv.cpp @@ -0,0 +1,274 @@ +/* + * Copyright 2024 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "vulkan_video_encoder_argv.h" + +#include "encoder_argv_completion.h" +#include "VkVideoEncoder/VkEncoderConfig.h" +#include "VkVideoEncoder/VkVideoEncoder.h" + +class VulkanVideoEncoderImpl : public VulkanVideoEncoder { +public: + virtual VkResult Initialize(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, + int argc, const char** argv); + virtual int64_t GetNumberOfFrames() + { + return m_encoderConfig->numFrames; + } + virtual VkResult EncodeNextFrame(int64_t& frameNumEncoded); + virtual VkResult GetBitstream() { + // Was: join, discard the verdict, answer VK_SUCCESS. A caller had no + // way to learn that the threads had not completed, or that the + // buffered bitstream never reached the file, and the quality harness + // read the zero exit that followed as a pass. + // + // Wait, then check the output, then let the caller release the owner. + // Cached, so a repeated call neither joins nor flushes twice and never + // turns a recorded failure into a success. + return m_completion.Complete( + (m_encoderConfig != nullptr) && !m_encoderConfig->disableFileOutput, + [this]() -> bool { + return m_encoder ? m_encoder->WaitForThreadsToComplete() : false; + }, + [this]() -> bool { + return ArgvCompletionState::FlushFileOutput( + m_encoderConfig->outputFileHandler.GetFileHandle()); + }); + } + + VulkanVideoEncoderImpl() + : m_vkDevCtxt() + , m_encoderConfig() + , m_encoder() + , m_lastFrameIndex(0) + { } + + virtual ~VulkanVideoEncoderImpl() { } + + void Deinitialize() + { + m_encoder->WaitForThreadsToComplete(); + + if (m_encoderConfig->verbose) { + VkEncOut() << "Done processing " << m_lastFrameIndex << " input frames!" << std::endl + << "Encoded file's location is at " << m_encoderConfig->outputFileHandler.GetFileName() + << std::endl; + } + + m_encoder = nullptr; + m_encoderConfig = nullptr; + } + +private: + VulkanDeviceContext m_vkDevCtxt; + VkSharedBaseObj m_encoderConfig; + VkSharedBaseObj m_encoder; + uint32_t m_lastFrameIndex; + // The completion verdict, cached so repeated calls agree. + ArgvCompletionState m_completion; +}; + +VkResult VulkanVideoEncoderImpl::Initialize(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, + int argc, const char** argv) +{ + VkResult result = EncoderConfig::CreateCodecConfig(argc, argv, m_encoderConfig); + if (VK_SUCCESS != result) { + return result; + } + + static const char* const requiredInstanceLayers[] = { + "VK_LAYER_KHRONOS_validation", + nullptr + }; + + static const char* const requiredInstanceExtensions[] = { + VK_EXT_DEBUG_REPORT_EXTENSION_NAME, + nullptr + }; + + static const char* const requiredDeviceExtension[] = { +#if defined(__linux) || defined(__linux__) || defined(linux) + VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME, + VK_KHR_EXTERNAL_FENCE_FD_EXTENSION_NAME, +#endif + VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME, + VK_KHR_VIDEO_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME, + VK_KHR_TIMELINE_SEMAPHORE_EXTENSION_NAME, + nullptr + }; + + static const char* const optinalDeviceExtension[] = { + VK_EXT_YCBCR_2PLANE_444_FORMATS_EXTENSION_NAME, + VK_EXT_DESCRIPTOR_BUFFER_EXTENSION_NAME, + VK_KHR_BUFFER_DEVICE_ADDRESS_EXTENSION_NAME, + VK_KHR_PUSH_DESCRIPTOR_EXTENSION_NAME, + VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, + nullptr + }; + + if (m_encoderConfig->validate) { + m_vkDevCtxt.AddReqInstanceLayers(requiredInstanceLayers); + m_vkDevCtxt.AddReqInstanceExtensions(requiredInstanceExtensions); + } + + m_vkDevCtxt.AddReqDeviceExtensions(requiredDeviceExtension); + m_vkDevCtxt.AddOptDeviceExtensions(optinalDeviceExtension); + + result = m_vkDevCtxt.InitVulkanDevice(m_encoderConfig->appName.c_str(), VK_NULL_HANDLE, + m_encoderConfig->verbose); + if (result != VK_SUCCESS) { + if (!IsVkEncoderStdioSilenced()) { + printf("Could not initialize the Vulkan device!\n"); + } + return result; + } + + result = m_vkDevCtxt.InitDebugReport(m_encoderConfig->validate, + m_encoderConfig->validateVerbose); + if (result != VK_SUCCESS) { + return result; + } + + VkQueueFlags requestVideoEncodeQueueMask = VK_QUEUE_VIDEO_ENCODE_BIT_KHR; + + if (m_encoderConfig->selectVideoWithComputeQueue) { + requestVideoEncodeQueueMask |= VK_QUEUE_COMPUTE_BIT; + } + + VkQueueFlags requestVideoComputeQueueMask = 0; + if (m_encoderConfig->IsPreprocessComputeFilterEnabled()) { + requestVideoComputeQueueMask = VK_QUEUE_COMPUTE_BIT; + } + + // No display presentation and no decoder - just the encoder + result = m_vkDevCtxt.InitPhysicalDevice(m_encoderConfig->deviceId, m_encoderConfig->deviceUUID, + ( requestVideoComputeQueueMask | + requestVideoEncodeQueueMask | + VK_QUEUE_TRANSFER_BIT), + nullptr, + 0, + VK_VIDEO_CODEC_OPERATION_NONE_KHR, + requestVideoEncodeQueueMask, + videoCodecOperation); + if (result != VK_SUCCESS) { + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + << "InitVulkanDevice() failed - video codec may not be supported. VkResult: " << result + << " (0x" << std::hex << result << std::dec << ")" << std::endl; + return result; + } + + const int32_t numEncodeQueues = ((m_encoderConfig->queueId != 0) || + (m_encoderConfig->enableHwLoadBalancing != 0)) ? + -1 : // all available HW encoders + 1; // only one HW encoder instance + + result = m_vkDevCtxt.CreateVulkanDevice(0, // num decode queues + numEncodeQueues, // num encode queues + videoCodecOperation, + // If no graphics or compute queue is requested, only video queues + // will be created. Not all implementations support transfer on video queues, + // so request a separate transfer queue for such implementations. + ((m_vkDevCtxt.GetVideoEncodeQueueFlag() & VK_QUEUE_TRANSFER_BIT) == 0), // createTransferQueue + false, // createGraphicsQueue + false, // createDisplayQueue + ((m_encoderConfig->selectVideoWithComputeQueue == 1) || // createComputeQueue + m_encoderConfig->IsPreprocessComputeFilterEnabled()) + ); + if (result != VK_SUCCESS) { + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + << "CreateVulkanDevice() failed. VkResult: " << result + << " (0x" << std::hex << result << std::dec << ")" << std::endl; + return result; + } + + result = VkVideoEncoder::CreateVideoEncoder(&m_vkDevCtxt, m_encoderConfig, m_encoder); + if (result != VK_SUCCESS) { + VkEncErr() << "ERROR [" << __FILE__ << ":" << __LINE__ << "]: " + << "CreateVideoEncoder() failed. VkResult: " << result + << " (0x" << std::hex << result << std::dec << ")" << std::endl; + return result; + } + + return result; +} + +VkResult VulkanVideoEncoderImpl::EncodeNextFrame(int64_t& frameNumEncoded) +{ + if (m_lastFrameIndex >= m_encoderConfig->numFrames) { + return VK_ERROR_TOO_MANY_OBJECTS; + } + + if (m_encoderConfig->verboseFrameStruct) { + VkEncOut() << "####################################################################################" << std::endl + << "Start processing current input frame index: " << m_lastFrameIndex << std::endl; + } + + VkSharedBaseObj encodeFrameInfo; + m_encoder->GetAvailablePoolNode(encodeFrameInfo); + assert(encodeFrameInfo); + // load frame data from the file + VkResult result = m_encoder->LoadNextFrame(encodeFrameInfo); + if (result != VK_SUCCESS) { + VkEncOut() << "ERROR processing input frame index: " << m_lastFrameIndex << std::endl; + return result; + } + + frameNumEncoded = encodeFrameInfo->frameInputOrderNum; + + if (m_encoderConfig->verboseFrameStruct) { + VkEncOut() << "End processing current input frame index: " << m_lastFrameIndex << std::endl; + } + + m_lastFrameIndex++; + + return result; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoder(VkVideoCodecOperationFlagBitsKHR videoCodecOperation, + int argc, const char** argv, + VkSharedBaseObj& vulkanVideoEncoder) +{ + switch((uint32_t)videoCodecOperation) + { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + { + + } + break; + + default: + assert(!"Unsupported codec type!!!\n"); + return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + } + + VkSharedBaseObj vulkanVideoEncoderObj( new VulkanVideoEncoderImpl()); + if (!vulkanVideoEncoderObj) { + return VK_ERROR_OUT_OF_HOST_MEMORY; + } + + VkResult result = vulkanVideoEncoderObj->Initialize(videoCodecOperation, argc, argv); + if (result != VK_SUCCESS) { + vulkanVideoEncoderObj = nullptr; + } else { + vulkanVideoEncoder = vulkanVideoEncoderObj; + } + + return result; +} diff --git a/vk_video_encoder/src/vulkan_video_encoder_ext.cpp b/vk_video_encoder/src/vulkan_video_encoder_ext.cpp index 95c0012f..43f7433b 100644 --- a/vk_video_encoder/src/vulkan_video_encoder_ext.cpp +++ b/vk_video_encoder/src/vulkan_video_encoder_ext.cpp @@ -14,15 +14,61 @@ * limitations under the License. */ +// THE COUPLING THIS FILE DEPENDS ON, MADE LOUD. Every Win32 arm below is +// spelled `defined(_WIN32)` while the symbols inside them -- HANDLE, +// VkImportSemaphoreWin32HandleInfoKHR, VK_KHR_EXTERNAL_*_WIN32_EXTENSION_NAME +// -- come from vulkan_win32.h, which is gated on VK_USE_PLATFORM_WIN32_KHR. +// Both builds that compile this file define it whenever the target is Windows +// (the CMake build sets it directly; the GN build gets it from +// third_party/vulkan-headers). That is a real guarantee, not an accident, but +// nothing enforced it -- so a third build could have flipped every Win32 arm +// to a wall of undeclared identifiers. It cannot now. +#if defined(_WIN32) && !defined(VK_USE_PLATFORM_WIN32_KHR) +#error "This file spells its Win32 arms `defined(_WIN32)` but uses \ +vulkan_win32.h symbols inside them. Define VK_USE_PLATFORM_WIN32_KHR when \ +building for Windows, or convert every _WIN32 arm in this file to \ +VK_USE_PLATFORM_WIN32_KHR." +#endif + #include #include +#include "vulkan_video_encoder_os_event_linux.h" + +#include #include +#include +#include #include +#include +#include + +#include "vulkan_video_encoder_ext_internal.h" +#include #include +#include + +// POSIX, not Linux. is where close(2) is declared on every POSIX +// target, and the acquire fence this file consumes is a POSIX fd on all of +// them -- gfx::GpuFenceHandle::ScopedPlatformFence is base::ScopedFD under +// BUILDFLAG(IS_POSIX), not under IS_LINUX. MSVC ships no , which is +// why _WIN32 gets its own arm below rather than an unguarded include. +#if !defined(_WIN32) +#include +#include // F_DUPFD_CLOEXEC: the BORROW mode's private duplicate +#include // close(2): the fd-consumption rule (VkEncConsumeOsHandle) +#endif #include "vulkan_video_encoder_ext.h" #include "VkVideoEncoder/VkEncoderConfig.h" +#include "VkVideoEncoder/VkEncoderConfigH264.h" +#include "VkVideoEncoder/VkEncoderConfigH265.h" +#include "VkVideoEncoder/VkEncoderConfigAV1.h" #include "VkVideoEncoder/VkVideoEncoder.h" +// The device-free capture backend below stands in for a real session +// by BEING one of the codec encoders rather than imitating it, so the +// H.264 arm of the mid-stream rate-control refresh is the code a test +// drives, not a second copy of it. +#include "VkVideoEncoder/VkVideoEncoderH264.h" // YcbcrVkFormatInfo() / GetBitsPerChannel() -- used to derive the input bit depth, // chroma subsampling and plane count from VkVideoEncoderConfig::inputFormat. #include "nvidia_utils/vulkan/ycbcrvkinfo.h" @@ -42,10 +88,67 @@ class VulkanVideoEncoderExtImpl : public VulkanVideoEncoderExt { , m_encoder() , m_initialized(false) , m_framesSubmitted(0) + , m_rateControlMode(VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR) { } virtual ~VulkanVideoEncoderExtImpl() { + // ENFORCED, like the session-serial methods -- but they can + // return VK_ERROR_NOT_PERMITTED_KHR from inside the completion + // callback and a destructor cannot. Dropping the last encoder + // reference inside pfnFrameReady tears this state down under + // the invoking edge's feet (it still writes its thread-id + // bookkeeping after the callback returns) and, on a real + // session, Deinitialize() would join the very delivery thread + // running the callback. Both failure modes are silent -- a + // use-after-free with no report, or a self-join deadlock whose + // stack names nobody -- so the enforcement is the diagnosed + // abort, the same shape as the noexcept-escape abort on the + // invocation path. The message goes to stderr; under + // silenceStdio the abort itself is still the diagnosis. + if (IsInCompletionCallback()) { + VkEncErr() << "[EncoderExt] encoder destroyed from inside " + "the completion callback: the last reference " + "must not be dropped from pfnFrameReady -- " + "aborting" << std::endl; + std::abort(); + } + // Workers first: Deinitialize() IS the join, and until it returns + // an assembly worker can still be inside OnBitstreamCaptured -- + // invoking the callback with m_completionUserData and writing to + // the completion event. The cookie is released AFTER the workers are + // joined, not before: releasing first inverts the header's promise (the + // release thunk fires + // "only after any in-flight completion invocation has returned") + // for exactly the consumer the release exists for: one that + // destroys without detaching. Both shipping consumers happen to + // detach or quiesce before destruction today; the order is fixed + // so the contract stops depending on that. Deinitialize(); + // The cookie goes back after the join: a caller that transferred + // ownership must get it back even on the path where nothing ever + // detached, and only once nothing can still be invoking it. + if (m_completionUserDataRelease != nullptr) { + PFN_vkVideoEncoderUserDataRelease release = + m_completionUserDataRelease; + void* userData = m_completionUserData; + m_completionUserDataRelease = nullptr; + m_completionUserData = nullptr; + m_completionCallback = nullptr; + release(userData); + } + // Destroyed last, and here rather than in Deinitialize: the + // handle's lifetime is the encoder object, not the session. A + // consumer that obtained it and then saw a Deinitialize must not + // have it closed underneath it -- the number gets recycled, and + // the next thing to open a file inherits a waiter. After the join + // above, nothing can be writing to it either, which the previous + // order (close before join) did not guarantee. + // Relaxed is sufficient here and only here: Deinitialize() above is + // the worker join, so no other thread can be loading this member. + vkenc::OsCompletionEventDestroy( + m_completionEventHandle.load(std::memory_order_relaxed)); + m_completionEventHandle.store(vkenc::kOsCompletionEventNone, + std::memory_order_relaxed); } @@ -67,21 +170,165 @@ class VulkanVideoEncoderExtImpl : public VulkanVideoEncoderExt { // VulkanVideoEncoderExt (extended interface - external frame input) //========================================================================= VkResult InitializeExt(const VkVideoEncoderConfig& config) override; - VkResult SubmitExternalFrame(const VkVideoEncodeInputFrame& frame, - VkSemaphore* pStagingCompleteSemaphore = nullptr) override; - VkResult PollEncodeComplete(uint64_t frameId) override; + // Public legacy entry point: a thin wrapper over + // SubmitExternalFrameCommon (the one submit implementation both public + // entry points share) with no prepared node, so the per-frame wrap + // happens inside the encoder exactly as before registration existed. + VkResult SubmitExternalFrame( + const VkVideoEncodeInputFrame& frame, + VkSemaphore* pStagingCompleteSemaphore = nullptr) override; + VkResult SetCompletionCallback( + PFN_vkVideoEncoderCompletionCallback callback, void* pUserData, + PFN_vkVideoEncoderUserDataRelease releaseUserData) override; + uint64_t GetCompletionCounter() override; + VkResult GetCompletionInfo(VkVideoEncoderCompletionInfo* pInfo) override; + VkVideoEncoderStatusCode GetCompletionEventHandle(uint64_t* outHandle) override; + VkSemaphore GetCompletionSemaphore() const override; + VkVideoEncoderStatusCode ExportCompletionSemaphoreHandle( + VkVideoEncoderExternalHandleType handleType, + uint64_t* outHandle) override; + VkResult CancelFrame(uint64_t frameId) override; + VkResult AbandonAllFrames(uint32_t* pAbandonedCount) override; + VkResult AcquireNextEncodedFrame(VkVideoEncodeResult& result) override; + VkResult AcquireEncodedFrame(uint64_t frameId, + VkVideoEncodeResult& result) override; + VkVideoEncoderFrameState GetFrameStatus(uint64_t frameId) override; VkResult GetEncodedFrame(VkVideoEncodeResult& result) override; void ReleaseEncodedFrame(uint64_t frameId) override; - VkFence GetEncodeFence(uint64_t frameId) override; VkResult Flush() override; + void GetShutdownInfo(VkVideoEncoderShutdownInfo* pInfo) const override; + // Records a device loss wherever it is first observed; returns |result| + // unchanged so it can wrap a return expression. + VkResult NoteDeviceResult(VkResult result); + VkVideoEncoderShutdownDisposition ClassifyShutdownLocked() const; + VkResult WaitWholeDeviceLocked(); + VkResult DrainPendingFrames() override; VkResult Reconfigure(const VkVideoEncoderConfig& config) override; - VkBool32 SupportsFormat(VkFormat inputFormat) const override; - uint32_t GetMaxWidth() const override; - uint32_t GetMaxHeight() const override; + // The library-internal format predicate, asked of a DECLARED colour + // model. A descriptor states the model its samples carry, and the + // packed 4:4:4 layouts make that statement load-bearing: they ride RGBA + // format enumerants, so the format alone cannot say which of the two a + // surface is. VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT declares nothing + // and falls back to the session's model, which is what the callers with + // no declaration of their own to pass hand in. + VkBool32 SupportsFormat(VkFormat inputFormat, + VkVideoEncoderColorModel declaredColorModel) const; + // Whether THIS session can perform the compute-tier conversion: the + // filter compiled in and the caller having asked for it. The rung-2 + // half of the adaptation ladder's "query the device first" rule. + bool ComputeFilterActive() const; + // Stricter: does this session's filter take THIS DECLARED PAIR -- + // format and colour model -- as its input? + bool ComputeFilterTakesFormat(VkFormat inputFormat, + VkVideoEncoderColorModel colorModel) const; + // The colour model to read |inputFormat| under, from this session's point + // of view: the caller's declaration for the session's OWN input format, + // and the format's own answer for anything else. A session declares one + // input; a question about some other format is not covered by that + // declaration and must not silently borrow it. + VkVideoEncoderColorModel SessionColorModel(VkFormat inputFormat) const; + // Whether the DEVICE could STORAGE-READ an image with this descriptor's + // format and tiling -- the fact the registration gate consults for the + // single-plane arm of the preprocess filter. + bool DeviceCanStorageRead( + const VkVideoEncoderExternalImageDescriptor& desc) const; + VkResult GetRuntimeInfo(VkVideoEncoderRuntimeInfo* outInfo) const override; + + // One predicate, shared by RegisterImageResource and QueryImageSupport. + // Pure with respect to the handle: it never consumes an fd, so the + // caller owns that decision and the query can run without one. + VkVideoEncoderStatusCode ValidateImageDescriptor( + const VkVideoEncoderExternalImageDescriptor& descriptor) const; + VkVideoEncoderStatusCode QueryImageSupport( + const VkVideoEncoderExternalImageDescriptor& descriptor, + VkVideoEncoderImageSupport* outSupport) override; + VkVideoEncoderStatusCode RegisterImageResource( + const VkVideoEncoderExternalImageDescriptor& descriptor, + uint64_t osHandle, + VkVideoEncoderResource* outResource, + VkVideoEncoderStatus* pStatus) override; + VkVideoEncoderStatusCode UnregisterImageResource( + VkVideoEncoderResource resource) override; + VkVideoEncoderStatusCode RegisterSemaphore( + const VkVideoEncoderSemaphoreDescriptor& descriptor, + uint64_t osHandle, + VkVideoEncoderResource* outResource, + VkVideoEncoderStatus* pStatus) override; + VkVideoEncoderStatusCode UnregisterSemaphore( + VkVideoEncoderResource resource) override; + // VK_NULL_HANDLE for a stale, untagged, or never-registered id. + VkSemaphore ResolveSemaphore(VkVideoEncoderResource resource); + VkVideoEncoderStatusCode SubmitRegisteredFrame( + const VkVideoEncoderFrameSubmitInfo& info, + VkSemaphore* pStagingCompleteSemaphore) override; + // Drop a frame's reference on its registration, retiring the + // registration if an Unregister was deferred waiting for it. + void ReleaseResourceReference(VkVideoEncoderResource resource); VkDevice GetVkDevice() const override { return m_vkDevCtx; } VkPhysicalDevice GetVkPhysicalDevice() const override { return m_vkDevCtx.getPhysicalDevice(); } VkInstance GetVkInstance() const override { return m_vkDevCtx.getInstance(); } + PFN_vkGetInstanceProcAddr GetVkGetInstanceProcAddr() const override { + // VkInterfaceFunctions::GetInstanceProcAddr, dlsym'd from the loader + // handle this context keeps mapped. Null until InitVulkanDevice runs. + return m_vkDevCtx.GetInstanceProcAddr; + } + + // Fault-injection seam (vulkan_video_encoder_ext_internal.h): exactly + // these functions may reach the impl's private state, and only for + // objects minted by CreateVulkanVideoEncoderExt. + friend VkResult VkEncInjectImportContentMeasurement( + VulkanVideoEncoderExt*, VkVideoEncoderResource, uint32_t, uint32_t, + uint32_t); + friend VkResult VkEncInstallNullBackend(VulkanVideoEncoderExt*, + const VkEncNullBackendState*); + friend VkVideoEncoderStatusCode VkEncProbeResource(VulkanVideoEncoderExt*, + VkVideoEncoderResource, + VkEncResourceProbe*); + friend VkBool32 VkEncSessionInitialized(VulkanVideoEncoderExt*); + friend void VkEncFireCompletionEdge(VulkanVideoEncoderExt*, uint64_t); + friend VkResult VkEncApplyAndGetSessionConstQp(VulkanVideoEncoderExt*, + int32_t*, int32_t*, + int32_t*); + friend VkResult VkEncApplyAndGetRateControl( + VulkanVideoEncoderExt*, VkEncRateControlObservation*); + friend VkResult VkEncGetRecordedConfig(VulkanVideoEncoderExt*, + VkVideoEncoderConfig*); + friend VkResult VkEncSeedRecordedConfig(VulkanVideoEncoderExt*, + const VkVideoEncoderConfig*); + friend VkResult VkEncSetDeviceQpWindow(VulkanVideoEncoderExt*, + int32_t, int32_t); + friend VkResult VkEncPushCapture(VulkanVideoEncoderExt*, uint64_t, + VkResult); + friend VkResult VkEncInstallTestSemaphore(VulkanVideoEncoderExt*, + VkSemaphore, + VkVideoEncoderResource*); + friend VkVideoEncoderStatusCode VkEncUninstallTestSemaphore( + VulkanVideoEncoderExt*, VkVideoEncoderResource); + friend VkVideoEncoderStatusCode VkEncProbeLastSubmitSync( + VulkanVideoEncoderExt*, VkEncSubmitSyncProbe*); + + // Bind this session to the context it was created on. Called ONLY by + // CreateVulkanVideoEncoderExtOnContext, before the session is visible to + // anyone else, so there is nothing to synchronise: these are written once + // and read on the init path. + // + // The instance and physical device are RESOLVED BY THE CALLER and cached + // here as plain values rather than re-read from the context during init. + // Two reasons, and the first is a hard one: VulkanVideoEncoderContext is + // an incomplete type at InitVulkanDevice's definition in this translation + // unit, so the accessors cannot be called there at all. The second is that + // caching is sound -- a context is immutable after construction, which the + // header states as a contract and which is what lets unrelated sequences + // share one context with no locking. + void SetContext(const VkSharedBaseObj& context, + VkInstance instance, + VkPhysicalDevice physicalDevice) + { + m_context = context; + m_contextInstance = instance; + m_contextPhysDevice = physicalDevice; + } private: void Deinitialize(); @@ -95,20 +342,593 @@ class VulkanVideoEncoderExtImpl : public VulkanVideoEncoderExt { VkResult InitVulkanDevice(VkVideoCodecOperationFlagBitsKHR codecOp, const VkVideoEncoderConfig& config); + // Non-null only for a session built by CreateVulkanVideoEncoderExtOnContext. + // + // WHAT THIS REFERENCE DOES, AND WHAT IT CANNOT DO. It keeps the CONTEXT + // OBJECT -- and with it the capability snapshot the caller selected from -- + // alive for as long as the session. It does NOT keep the borrowed + // VkInstance alive, and no reference here could: in ADOPT the instance + // belongs to the embedder and the context destroys nothing on release + // (header context rule 3), and in OWN the library's floor registry already + // holds the context above zero for the process lifetime (rule 2). An + // embedder that destroys its VkInstance under a live session has a + // use-after-free either way; that hazard is the embedder's to avoid, and it + // is identical on the config path. An earlier version of this comment + // claimed the reference closed it. It does not. + // + // Declared BEFORE |m_vkDevCtx| so it is destroyed AFTER it. Given the + // above, that ordering is not load-bearing today -- it is kept because it + // is the order that stays correct if a session ever holds something the + // context genuinely owns. + VkSharedBaseObj m_context; + // The context's instance and the chosen physical device, resolved at + // creation. See SetContext for why they are cached rather than re-read. + VkInstance m_contextInstance = VK_NULL_HANDLE; + VkPhysicalDevice m_contextPhysDevice = VK_NULL_HANDLE; + VulkanDeviceContext m_vkDevCtx; VkSharedBaseObj m_encoderConfig; VkSharedBaseObj m_encoder; - bool m_initialized; + // Written at init/teardown, read by the lock-free class-(c) queries. + std::atomic m_initialized; + // Test seam (VkEncInstallNullBackend, internal header): null in every + // production session. When set, the session reports initialized with no + // VkVideoEncoder behind it (until VkEncPushCapture installs its + // device-free capture source), VK_IMAGE registration skips the + // device-touching view build, and SubmitExternalFrameCommon terminates + // at this backend instead of m_encoder. + // Per-frame acquire fences imported as binary semaphores (design 3.5). + // The IMPORT is temporary -- SYNC_FD is copy-transference, so the payload + // dies with the first wait -- but the VkSemaphore OBJECT is the library's + // and has vkDestroySemaphore's ordinary precondition, so it retires with + // the frame that waited on it exactly like the release fence below. It + // destroyed here rather than parked in a session-lifetime vector: with no + // destroy call anywhere, one driver semaphore leaks per fenced frame for + // the life of the session. + VkSemaphore ImportAcquireFenceLocked(int fd); + + // === Per-frame RELEASE fences (design 3.5, the export half) ============ + // + // The other direction of the same entry point: a library-owned BINARY + // semaphore appended to the frame's signal list, so the submission that + // consumes the input image signals it, then exported as a SYNC_FD for + // the caller to feed into its own end-of-read-access obligation. + // + // Binary and per-frame, NOT the registered timeline: registration is + // timeline-only by design and a binary semaphore cannot be registered + // once and named repeatedly. These two paths stay separate. + // + // Two graveyards, because "may I destroy this VkSemaphore" is exactly + // "has every batch referring to it completed", and only one of the + // retirement paths can prove that. |m_releaseFenceRetired| holds + // semaphores whose frame retired WITH a real capture -- the capture is + // published only after the encode command buffer's fence wait, and the + // encode submit waits on the staging submit, so both submissions that + // could signal a release fence have completed. They are destroyed on the + // next submit, off the lock. |m_unprovenSemaphores| holds the rest + // (abandoned, timed out, failed, delivered with a non-SUCCESS status, or + // still in flight at Deinitialize) plus every imported ACQUIRE semaphore + // that took one of those same routes. Destroyed behind a device wait-idle + // -- at teardown, and in-session once the vector crosses + // kUnprovenSemaphoreBound, because the routes that feed it are all + // repeatable in-session operations (a deadline drop, a CancelFrame, an + // AbandonAllFrames, a failed capture) and a graveyard only teardown + // empties is not a graveyard, it is a leak with a comment. + std::vector m_releaseFenceRetired; + std::vector m_unprovenSemaphores; + // Above this many entries the unproven graveyard is swept on the submit + // path behind a DeviceWaitIdle. Well above any steady-state depth: in a + // healthy session every acquire and release fence retires with its frame + // and this vector stays empty, so the sweep is a backstop for the + // abnormal routes and never a per-frame cost. + static const size_t kUnprovenSemaphoreBound = 64; + // Exportability is a physical-device property: asked once, never + // assumed -- creating a semaphore with an unsupported export handle + // type is itself invalid usage, so the query has to precede the create. + bool m_releaseFenceExportProbed = false; + bool m_releaseFenceExportable = false; + // Create a SYNC_FD-exportable binary semaphore, or VK_NULL_HANDLE when + // this device/platform cannot provide one. Never touches a lock. + VkSemaphore CreateReleaseFenceSemaphore(); + // Export |semaphore|'s pending signal as a SYNC_FD. -1 is a legal answer + // (already signalled, or the driver declined); it is never an error the + // caller has to handle as one. + int ExportReleaseFenceFd(VkSemaphore semaphore); + // Destroy everything in m_releaseFenceRetired. Takes m_pendingMutex to + // swap the vector out, then destroys OUTSIDE it -- the same one-sided + // lock discipline Flush and Deinitialize use for driver calls. + void DrainRetiredReleaseFences(); + // Destroy everything in m_unprovenSemaphores when it has crossed + // kUnprovenSemaphoreBound. Unlike the retired vector these carry no + // completion proof of their own, so the destroy precondition has to be + // manufactured -- DeviceWaitIdle, the same instrument Deinitialize uses, + // for the same reason (a staging batch can ride the TRANSFER queue, which + // an encode-queue wait does not cover). Takes m_pendingMutex to swap out, + // waits and destroys outside it. + void DrainUnprovenSemaphores(); + + const VkEncNullBackendState* m_nullBackend = nullptr; + // Observation seam (VkEncProbeLastSubmitSync, internal header): the + // wait/signal arrays the last submit was handed after the + // chained-descriptor walk. Written on the NULL-BACKEND arm only, so a + // production session never touches either member and pays nothing. + std::mutex m_lastSubmitSyncMutex; + VkEncSubmitSyncProbe m_lastSubmitSync; + void RecordSubmitSyncForTest(const VkVideoEncodeInputFrame& frame); + uint64_t m_framesSubmitted; + // Rate-control mode (VkVideoEncoderConfig::rateControlMode) + // stashed at InitializeExt time so GetRuntimeInfo() can derive + // trustedRateController without reaching into EncoderConfig internals. + // The rate-control MODE is session-fixed; Reconfigure() updates rates only. + std::atomic m_rateControlMode; + // The preprocess compute filter's session state, snapshotted so the + // lock-free readers never dereference m_encoderConfig: ComputeFilterActive + // is reached from SupportsFormat, which ValidateImageDescriptor calls + // WITHOUT taking m_pendingMutex -- while Deinitialize nulls and + // then destroys m_encoderConfig under that mutex. Reading the + // shared_ptr there is both a use-after-free window and an unsynchronised + // read of a shared_ptr instance another thread is storing to. + // + // The FORMAT is snapshotted too, not just the flag: "this session has a + // filter" is not the same question as "this session's filter takes THIS + // format", and answering the second with the first over-promises to a + // producer that then allocates a pool RegisterImageResource refuses. + std::atomic m_computeFilterActive{false}; + std::atomic m_computeFilterInputFormat{VK_FORMAT_UNDEFINED}; + // The session's own input declaration -- the format and the colour model + // the caller stated for it -- snapshotted for the same lock-free readers. + std::atomic m_sessionInputFormat{VK_FORMAT_UNDEFINED}; + std::atomic m_sessionInputColorModel{ + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT}; + // What the session was initialized with, so Reconfigure can tell a + // change it can carry from one it can only pretend to. + VkVideoEncoderConfig m_initConfig{}; + std::atomic m_capsGranularityW{0}; + std::atomic m_capsGranularityH{0}; + // R-5 telemetry: how often import memory-type selection ran WITHOUT its + // authoritative constraint (query unavailable, or opaque import without + // the exporter's index), and how often an exporter-supplied index was + // overridden by the mask. "Zero fallbacks" was the design's named + // removal condition for the transition quirk; these are what make that + // number observable. + std::atomic m_importMemTypeHeuristicSelections{0}; + std::atomic m_importMemTypeExporterOverrides{0}; + void SnapshotCaps(); // Tracking submitted frames for async retrieval struct PendingFrame { - uint64_t frameId; - uint64_t pts; + uint64_t frameId = 0; + uint64_t pts = 0; VkSharedBaseObj encodeFrameInfo; + // In-memory bitstream captured for this frame. + // Populated by GetEncodedFrame() draining + // VkVideoEncoder::TryPopCapturedBitstream(). Empty until a + // corresponding capture arrives. + std::vector bytes; + bool isIdr = false; + uint32_t pictureType = 0; + // Readiness + per-frame result. hasCapture (not + // bytes-non-empty) is the readiness signal so both failed frames + // (empty bytes + error status) and legitimate 0-byte drop-frames + // are DELIVERED instead of wedging the FIFO forever. + bool hasCapture = false; + VkResult status = VK_SUCCESS; + // Retrieval state: |acquired| marks delivery (pBitstreamData stays + // valid until ReleaseEncodedFrame); |timedOut| marks a + // deadline-synthesized drop whose late capture, if one ever + // arrives, must be discarded; |submitTime| anchors the deadline. + bool acquired = false; + bool timedOut = false; + // The registration this frame was submitted against, so its + // reference can be dropped when the frame leaves the queue. + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + // Unique per reservation. Frame ids are consumer-chosen and may + // recur while an earlier frame carrying the same id is still + // resident, so commit and rollback address a record by this and + // never by frameId -- which would otherwise let a submit roll back + // somebody else's frame. + uint64_t admissionToken = 0; + // False between reservation and the submit returning. The entry is + // in the queue from the reservation onwards so an early worker + // capture has somewhere to land; what only the submit knows is + // filled at commit. + bool admitted = false; + std::chrono::steady_clock::time_point submitTime; + // The per-frame release fence's semaphore, if this frame asked for + // one. The SYNC_FD handed to the caller was exported from it and is + // an independent kernel object from that moment on, so this handle + // exists only to be destroyed once the submissions referring to it + // have completed -- see the two graveyards above. + VkSemaphore releaseFenceSemaphore = VK_NULL_HANDLE; + // The acquire half of the same entry point: the binary semaphores + // this frame's imported acquire fds were imported into. Owned here + // for the same reason and disposed on the same proof -- the + // submission that WAITS on them is the submission that signals the + // release fence, so the two have an identical destroy precondition. + std::vector acquireFenceSemaphores; }; std::deque m_pendingFrames; - std::mutex m_pendingMutex; + // Monotonic, never reused, guarded by m_pendingMutex. Zero is "no + // reservation", which is what a rollback for a refused reservation + // carries. + uint64_t m_nextAdmissionToken = 1; + // Completion order (design 3.4). Frame ids in the order their + // outcomes became deliverable -- a real capture, a deadline drop, or a + // cancellation -- which is NOT submit order. AcquireNextEncodedFrame + // drains from the front, so a frame that never completes cannot strand + // the frames behind it. Entries are skipped lazily when the frame was + // taken first by the keyed AcquireEncodedFrame, and pruned on release. + std::deque m_readyOrder; + // Frame ids ReleaseEncodedFrame or AbandonAllFrames erased while they + // were still PENDING (no capture, no deadline drop, no cancellation): + // their completion records are still in flight and will pop with no + // entry to match. DrainCapturesLocked consumes one record per such pop + // and discards it silently; an unmatched pop with no record here is + // the late class m_lateCaptures counts. Guarded by m_pendingMutex, + // like the queues above. Self-draining in steady state (one insert, + // one pop); cleared where its pops can no longer arrive -- Flush() + // where m_encoder is dropped, and Deinitialize() with the pending set. + // A multiset because frame ids are consumer-chosen: an id released + // while pending may be submitted and released again before its first + // capture pops. + std::unordered_multiset m_releasedWhilePending; + + // Registered semaphores. Separate from the image registry, and ids carry + // kSemaphoreTag so an image id used where a semaphore is expected is + // rejected rather than resolved against the wrong table -- that mistake + // would produce a wait on an unrelated object, which presents as a hang + // rather than as an error. + struct RegisteredSemaphore { + VkSemaphore semaphore = VK_NULL_HANDLE; + uint32_t generation = 1; + bool live = false; + }; + static constexpr uint64_t kSemaphoreTag = 1ull << 63; + std::mutex m_semaphoreMutex; + std::deque m_semaphores; + // mutable: GetCompletionSemaphore() is const but class (c), and the + // state it reads (m_encoder) is cleared under this lock by Flush() + // and teardown -- a const method that skipped the lock would race + // them (the defect this comment replaces). + mutable std::mutex m_pendingMutex; + + // Completion deadline (ns; clamped at InitializeExt) + the two + // observability counters that make a deadline/fence-cap collision or a + // stalling pipeline visible instead of silent. + uint64_t m_frameTimeoutNs = 8000000000ull; + uint64_t m_framesTimedOut = 0; + uint64_t m_lateCaptures = 0; + + void OnBitstreamCaptured(uint64_t frameId); + bool IsInCompletionCallback() const { + return m_callbackThreadId.load(std::memory_order_relaxed) == + std::this_thread::get_id(); + } + void CancelFrameLocked(PendingFrame& frame); + + // M6 completion currency. Counter is the coalescing-safe drain target. + // The callback pointer pair is guarded by m_pendingMutex and is only + // READ FOR INVOCATION while m_callbackMutex is held; invocations are + // serialized by m_callbackMutex, with the invoking thread recorded so + // session-serial methods can reject re-entry from inside the callback. + // (Captures are already serialized by the core encoder's assembly + // ordering lock, so concurrent invocation is not reachable today; + // m_callbackMutex keeps the contract true independently of that.) + // SetCompletionCallback takes BOTH locks (callback -> pending, the + // OnBitstreamCaptured order), making it a quiesce point: when it + // returns, no invocation of the previously installed callback is in + // flight or can start -- the caller may then destroy the old pUserData. + PFN_vkVideoEncoderCompletionCallback m_completionCallback = nullptr; + void* m_completionUserData = nullptr; + std::mutex m_callbackMutex; + std::atomic m_callbackThreadId{}; + std::atomic m_completionCounter{0}; + + // Non-null when the caller transferred ownership of m_completionUserData + // to us. Invoked exactly once, when the cookie is displaced or the + // encoder is destroyed. + PFN_vkVideoEncoderUserDataRelease m_completionUserDataRelease{nullptr}; + + // OS-handle completion currency (M7). kOsCompletionEventNone until a + // caller asks for it, so a consumer that never wants one pays nothing. + // Atomic because creation (under m_pendingMutex, published with a + // release store) races the capture-path signal, which is a bare acquire + // load plus write() and must not take a lock on the capture path; the + // pairing makes the eventfd's creation happen-before any signal + // through it. + std::atomic m_completionEventHandle{ + vkenc::kOsCompletionEventNone}; + uint64_t m_framesCancelled = 0; + + // Diagnosability that survives silenceStdio (A10.2): the stderr + // misuse lines are also recorded here -- count plus most recent + // text -- and read back through the GetCompletionInfo pNext chain. + // Guarded by m_pendingMutex, which every recording site already + // holds. + uint64_t m_diagnosticCount = 0; + char m_lastDiagnostic[VK_VIDEO_ENCODER_MAX_DIAGNOSTIC_CHARS] = {}; + + // The most recent verdict from a registration the dma-buf + // import-ordinal guard EVALUATED, readable through the + // GetCompletionInfo pNext chain (VkVideoEncoderImportGuardInfo). A + // registration the guard does not apply to leaves it UNCHANGED, so a + // VK_IMAGE registration cannot erase the COMPLETE a dma-buf one + // established. Guarded by m_pendingMutex, like the diagnostic pair + // above and for the same reason: that is the lock GetCompletionInfo + // reads under. + VkEncImportOrdinalGuardReport m_importGuardReport{}; + + // This session's silence request, held for its whole life -- see + // VkEncoderStdioLatch.h. Declared before the members whose teardown can + // still print, so it is destroyed after them. + VkEncoderStdioSilenceScope m_stdioSilence; + + // --- Terminal shutdown --- + // + // Serializes Flush() against itself so a repeated Finish never drains + // twice or restarts workers, and guards every field of m_shutdown. Never + // taken while m_pendingMutex or m_callbackMutex is held: Flush joins + // library threads, and those threads take both. + mutable std::mutex m_shutdownMutex; + VkVideoEncoderShutdownInfo m_shutdown{ + VK_VIDEO_ENCODER_SHUTDOWN_UNPROVEN, + VK_SUCCESS, + VK_NOT_READY, + VK_FALSE, + VK_FALSE, + VK_FALSE, + VK_FALSE}; + + // Acceptance stops ONCE, before the joins, and stays stopped even if the + // shutdown ends Unproven and keeps the encoder alive. Checked without + // m_shutdownMutex because the submit path must not serialize behind a + // shutdown that is joining threads. + std::atomic m_shutdownStarted{false}; + + // --- Handle exchange --- + struct RegisteredImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + // Bumped on every retirement, so a stale id names a generation that + // no longer exists and is rejected instead of matching a recycled + // slot. This is the whole reason ids are not pointers. + uint32_t generation = 1; + // Frames submitted against this registration that the GPU may still + // be reading. Retirement waits for zero: refcounted, never + // timeline-driven, because a timeline eviction either stalls on a + // device wait or frees an image mid-read. + uint32_t inFlight = 0; + bool live = false; + bool retired = false; + // False for VK_IMAGE registrations: the caller's image, not ours. + bool ownsImage = false; + // Kept so submit can apply the residency rule: an EXPLICIT + // declaration is honoured on ANY handle type; AUTO (and a + // zero-initialised field) is derived as FOREIGN for an import. + // The rule is stated on the field it describes, not inferred from the + // handle type. + VkVideoEncoderExternalHandleType handleType = + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_NONE; + VkVideoEncoderInputResidency residency = + VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO; + VkFormat format = VK_FORMAT_UNDEFINED; + uint32_t width = 0; + uint32_t height = 0; + VkImageLayout defaultLayout = VK_IMAGE_LAYOUT_UNDEFINED; + VkImageTiling tiling = VK_IMAGE_TILING_OPTIMAL; + // Input routing decided once, at registration, from the SAME + // predicate that routes the registered submit (encodeCapable below) + // -- so path and routing can never disagree. All three values are + // produced: FILTER is assigned when the registration needs the + // preprocess compute filter and this session can actually run it, + // and the submit path reads this field back to route the frame. + VkVideoEncoderExternalInputPath inputPath = + VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + // Reserved scratch slot for the A2 filter path: the library-owned + // OPTIMAL image the compute filter would write and the encode would + // read. Never populated today; destruction is + // wired in DestroyResourceLocked so a future population cannot leak. + VkImage filterScratchImage = VK_NULL_HANDLE; + VkDeviceMemory filterScratchMemory = VK_NULL_HANDLE; + VkImageView filterScratchView = VK_NULL_HANDLE; + // The import's / caller's true usage. For an import this is the + // usage the image was created with here (never fabricated); for a + // VK_IMAGE registration it is the caller's declaration, 0 = unknown + // (the view then assumes the legacy set -- once, at registration). + VkImageUsageFlags imageUsage = 0; + // Routing predicate, computed once at registration: encodable + // format, non-LINEAR tiling, and the image actually carries + // VIDEO_ENCODE_SRC usage. Replaces the per-frame tiling check on + // the registered submit path. + bool encodeCapable = false; + // Whether this registration's view carries the per-plane STORAGE + // views the preprocess compute filter binds. Computed once, from the + // DESCRIPTOR (MUTABLE_FORMAT + STORAGE + a multi-planar format) and + // then confirmed against the view that was actually built, so a + // driver that declined a plane view cannot leave this claiming one + // exists. Distinct from |encodeCapable|: this registration's image is + // the filter's SOURCE, and only the filter's OUTPUT is an encode + // source (PROCESSING_GRAPHS section 3.1). + bool planeStorageViews = false; + // The SINGLE-PLANE parallel of the field above, and a different + // question about the same wrapper: not "how many per-plane views + // were built" but "is there a combined view, and may it be bound as + // a STORAGE_IMAGE". The compute filter's RGBA arm binds exactly one + // VK_DESCRIPTOR_TYPE_STORAGE_IMAGE over that combined view + // (VulkanFilterYuvCompute: m_inputImageAspects = COLOR_BIT, aspect 0 + // bound through GetImageView()), so a slot claiming this must have + // one. Computed the same way |planeStorageViews| is -- from the view + // that was actually BUILT, never from the descriptor's request -- + // because those two can disagree and only the built one is what the + // filter will bind. + bool storageReadView = false; + // The allocation size the import actually used, recorded so the + // refcounted wrapper's bookkeeping matches the real allocation. + VkDeviceSize importedAllocSize = 0; + // Created ONCE at registration, so that a submit performs no + // allocation at all. The node is shared by every frame + // submitted against this registration -- safe because the encode + // path only reads it (GetPictureResourceInfo / GetImageView are the + // only consumers; SetNewLayout has no call site in the encoder + // libs). A frame in flight holds |node| via VkSharedBaseObj, so + // retirement can drop these refs without freeing memory the GPU is + // still reading: the wrapper's refcount is the final arbiter. + VkSharedBaseObj imageView; + VkSharedBaseObj node; + }; + std::vector m_resources; + std::mutex m_resourceMutex; + + static VkVideoEncoderResource MakeResourceId(size_t index, + uint32_t generation) { + return ((VkVideoEncoderResource)generation << 32) | + (VkVideoEncoderResource)(index + 1); + } + static size_t ResourceIndex(VkVideoEncoderResource r) { + return (size_t)((r & 0xFFFFFFFFull) - 1); + } + static uint32_t ResourceGeneration(VkVideoEncoderResource r) { + return (uint32_t)(r >> 32); + } + // Caller holds m_resourceMutex. + RegisteredImage* LookupResourceLocked(VkVideoEncoderResource resource); + void DestroyResourceLocked(RegisteredImage& slot); + VkVideoEncoderStatusCode ImportImageLocked( + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t osHandle, RegisteredImage& slot); + // Fold the calling thread's import-guard verdict into the session + // snapshot, and record an INCOMPLETE one in the diagnostic channel. + // MUST NOT be called with m_resourceMutex held -- see the definition. + void PublishImportGuardVerdict(); + // The dma-buf import CONTENT probe. Owned HERE and injected into the + // encoder, not the other way round: a null-backend session has no + // encoder object, and arming plus reporting must still work there or the + // carrier is provable only on hardware. Created lazily by the first + // registration that asks for it, so a caller that never chains + // VkVideoEncoderImportContentInfo pays nothing -- not even the object. + VkSharedBaseObj m_contentProbe; + // Both MUST NOT be called with m_resourceMutex held: they take + // m_pendingMutex to reach m_encoder, and this file's only established + // order is pending -> resource. Both callers run them from a scope guard + // declared ahead of their m_resourceMutex lock guard. + VkVideoEncoderImportContentState ArmImportContentProbe( + VkVideoEncoderResource resource, + VkVideoEncoderContentProbe::CaptureSite captureSite); + void ForgetImportContentProbe(VkVideoEncoderResource resource); + // Build the once-per-registration wrapper + combined view + pool node + // for |slot| (see the definition for the full contract). Caller holds + // m_resourceMutex. + VkVideoEncoderStatusCode BuildRegisteredViewLocked( + const VkVideoEncoderExternalImageDescriptor& desc, + RegisteredImage& slot); + // Device's DRM format modifiers + their 32-bit tiling features for + // |format| (two-call vkGetPhysicalDeviceFormatProperties2 with + // VkDrmFormatModifierPropertiesListEXT chained). Empty when the + // session is not initialized or the modifier extension is unavailable. + std::vector EnumerateDrmModifiers( + VkFormat format) const; + // Would an image with |desc|'s shape register if it were DRM-tiled + // with |modifier| and imported from |desc|'s handle type? + // vkGetPhysicalDeviceImageFormatProperties2 with the modifier, + // external-handle and (under MUTABLE_FORMAT) format-list inputs + // mirroring ImportImageLocked's vkCreateImage, accepted only when the + // handle type is IMPORTABLE and every image-creation limit the query + // returns admits the descriptor (VkEncDescriptorWithinCreationLimits) + // -- the limits that create call's validity is defined against. + // Mirrored inputs alone are not agreement: the create call is bound by + // the query's OUTPUTS too, and an acceptance that reads fewer of them + // admits a descriptor the import must then refuse. What the query + // cannot see -- the explicit plane layouts, allocation failure -- stays + // the import's to judge, and a refusal there is a genuine + // IMPORT_FAILED, not a misclassified MODIFIER_UNSUPPORTED. + bool ModifierWouldRegister( + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t modifier) const; + // Fill a chained VkVideoEncoderImageSupportDetails. Best-effort: leaves + // the zeroed defaults wherever the answer is unknowable. + void FillImageSupportDetails( + const VkVideoEncoderExternalImageDescriptor& desc, + VkVideoEncoderImageSupportDetails* details) const; + // The one submit implementation behind both public entry points. With + // |preparedNode| == nullptr this is the legacy arm, byte-for-byte: the + // per-frame wrap happens inside SetExternalInputFrame. With a node it + // is the registered arm: the wrap already happened at registration and + // this creates no Vulkan object for the input. + // + // |releaseFenceSemaphore| (optional) is the caller's per-frame release + // fence semaphore, already appended to |frame|'s signal list by the + // caller of this function. This function owns exporting it, because the + // only place the answer to "was the input-consuming submit issued?" is + // observable is right here, on the frame-info node, immediately after + // the Set*Frame call returns. |pReleaseFenceFd| receives the exported fd + // or -1; the semaphore itself always ends up owned by the PendingFrame. + VkResult SubmitExternalFrameCommon( + const VkVideoEncodeInputFrame& frame, + VkSharedBaseObj* preparedNode, + bool encodeCapable, + // The registration's resolved ladder rung for the frames + // |encodeCapable| refuses. Passed down rather than re-derived in the + // encoder core, so the path this registration REPORTS through + // VK_VIDEO_EXTERNAL_INPUT_PATH_* and the path its frames actually + // take are one decision, made once. + bool routeViaFilter, + VkVideoEncoderResource resource, + VkSemaphore* pStagingCompleteSemaphore, + VkSemaphore releaseFenceSemaphore = VK_NULL_HANDLE, + int* pReleaseFenceFd = nullptr, + const std::vector* acquireFenceSemaphores = nullptr, + // Did |frame.currentLayout| come from the FRAME (explicit) or from + // the registration's defaultLayout standing in for the UNDEFINED + // sentinel? Only SubmitRegisteredFrame, which applies that + // sentinel, can answer, and the encoder core needs the answer to + // decide whether its own recorded residual layout may supersede + // the declaration. Defaults false, which is both correct and + // inert for the LEGACY lane: it has no registration default, and + // its per-frame node never carries a residual. + bool srcLayoutIsExplicit = false); + // Create the PendingFrame entry for a successfully submitted frame, + // handing it ownership of |resource|'s in-flight reference at creation + // (NULL for the legacy arm). Takes m_pendingMutex. + void EnqueuePendingFrame( + const VkVideoEncodeInputFrame& frame, + VkSharedBaseObj& + encodeFrameInfo, + VkVideoEncoderResource resource, + VkSemaphore releaseFenceSemaphore = VK_NULL_HANDLE, + const std::vector* acquireFenceSemaphores = nullptr); + + // Reserve/commit/rollback around the core submission. + // + // Reserve inserts the entry with everything known before the encoder is + // touched, and returns its admission token. It takes and RELEASES + // m_pendingMutex: no core or queue operation may run under that lock, + // because the assembly workers take it on the capture path. + uint64_t ReservePendingFrame( + const VkVideoEncodeInputFrame& frame, + VkVideoEncoderResource resource, + VkSemaphore releaseFenceSemaphore, + const std::vector* acquireFenceSemaphores); + // Finish admission for a submit that succeeded: attach the frame info + // the submit produced and count the frame once. + void CommitPendingFrame( + uint64_t admissionToken, + VkSharedBaseObj& + encodeFrameInfo); + // Undo a reservation whose submit was refused. Removes the entry only + // when no capture arrived for it; a reservation that DID capture + // describes work that landed, and its resources stay owned by the + // pending record rather than being torn down under in-flight use. + void RollbackPendingFrame( + uint64_t admissionToken, + VkSharedBaseObj& + encodeFrameInfo); + + void DrainCapturesLocked(); + void FillResultLocked(PendingFrame& frame, VkVideoEncodeResult& result); + bool SynthesizeTimeoutLocked(PendingFrame& frame); + void MarkFrameReadyLocked(PendingFrame& frame); + bool TryPopReadyLocked(VkVideoEncodeResult& result); }; //============================================================================= @@ -130,194 +950,1694 @@ static VkVideoCodecOperationFlagBitsKHR MapCodecOperation( //============================================================================= // Build EncoderConfig from structured config //============================================================================= +// Size guard for the config binder below. This does NOT pin an ABI -- the +// struct is versioned by sType and consumers are rebuilt against the header. +// It exists so that ADDING A FIELD cannot silently do nothing: the binder +// copies fields one at a time, a new field that nobody wired up simply never +// reaches the encoder, and that failure is invisible at runtime. Tripping this +// assert means: bind the new field here, then update the number. +// The public profile constants ARE the codec standard's numbers, which is +// what lets the binder below cast rather than translate. The std enums carry +// the same numbers, so this pins the two together at compile time instead of +// leaving a table to drift. +static_assert((uint32_t)VK_VIDEO_ENCODER_PROFILE_H264_BASELINE == + (uint32_t)STD_VIDEO_H264_PROFILE_IDC_BASELINE && + (uint32_t)VK_VIDEO_ENCODER_PROFILE_H264_MAIN == + (uint32_t)STD_VIDEO_H264_PROFILE_IDC_MAIN && + (uint32_t)VK_VIDEO_ENCODER_PROFILE_H264_HIGH == + (uint32_t)STD_VIDEO_H264_PROFILE_IDC_HIGH, + "H.264 profile constants must be profile_idc"); +static_assert((uint32_t)VK_VIDEO_ENCODER_PROFILE_H265_MAIN == + (uint32_t)STD_VIDEO_H265_PROFILE_IDC_MAIN && + (uint32_t)VK_VIDEO_ENCODER_PROFILE_H265_MAIN10 == + (uint32_t)STD_VIDEO_H265_PROFILE_IDC_MAIN_10, + "H.265 profile constants must be general_profile_idc"); +static_assert((uint32_t)VK_VIDEO_ENCODER_PROFILE_AV1_MAIN == + (uint32_t)STD_VIDEO_AV1_PROFILE_MAIN, + "AV1 profile constants must be seq_profile"); + +#if defined(__LP64__) || defined(_LP64) || defined(_WIN64) +static_assert(sizeof(VkVideoEncoderConfig) == 208, + "VkVideoEncoderConfig changed size -- a field was added, removed " + "or reordered. Bind it in BuildEncoderConfig before updating " + "this number."); +#endif // 64-bit: sizeof moves with pointer width + + +// Declared here rather than in the internal header: its signature names +// EncoderConfig, and that header deliberately stays clear of the library's +// private types so a test can include it without them. +// |requestedEncodeBitDepth| is the encode side stated rather than derived; see +// the internal header's note on VkEncBuildAndProbeConfig for why anything +// needs to state it. Zero is the ordinary path. +VkResult VkEncBuildEncoderConfig(const VkVideoEncoderConfig& extConfig, + VkVideoCodecOperationFlagBitsKHR codecOp, + VkSharedBaseObj& outConfig, + uint32_t requestedEncodeBitDepth = 0); + +// The quantizer range |codecOp| admits, in the units the caller states +// constQpI/P/B in. +// +// CODEC-DEPENDENT BECAUSE THE UNIT IS. H.264 and H.265 carry a QP on 0..51. +// AV1 has no QP at all: it carries a quantizer INDEX on 0..255. 52 is a +// legal AV1 quantizer index and an illegal H.26x QP, so one range applied to +// both would either refuse three quarters of the AV1 scale or admit an H.26x +// value the codec has no syntax for. +static int32_t VkEncMaxConstQpForCodec(VkVideoCodecOperationFlagBitsKHR codecOp) +{ + return (codecOp == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) ? 255 : 51; +} + +// Refuse a constant quantizer the codec cannot express, instead of letting it +// wrap. +// +// WHY THE REFUSAL IS HERE AND NOT WHERE THE VALUE BREAKS. Nothing between +// this boundary and the bitstream narrows it: the ext fields are int32_t, +// ConstQpSettings holds uint32_t (VkVideoEncoderDef.h), and the single +// narrowing is the (uint8_t) cast in VkVideoEncoderAV1::EncodeFrame that +// writes pictureInfo.constantQIndex and, from it, stdQuantInfo.base_q_idx. +// So an out-of-range value was refused nowhere -- it was TRUNCATED there, +// modulo 256, with VK_SUCCESS answered to the caller and no diagnostic +// anywhere: constQpI 300 encoded at quantizer index 44. A value silently +// accepted and altered is the same defect as a VK_SUCCESS that changes +// nothing, and it is worse for being invisible at the call that caused it. +// +// NEGATIVE IS NOT OUT OF RANGE. It is this API's spelling of "this config +// names no quantizer", read that way on both entry paths, so it is left to +// them; only an upper bound is enforced here. +// +// THE DEVICE'S OWN QUANTIZER WINDOW IS NOT CONSULTED, and reusing the one +// this library already records would not have covered the defect. That +// window (VkVideoEncoder::m_deviceQpWindowMin/Max, checked in +// RequestRateControlUpdate) is written only by the H.264 and H.265 arms at +// codec init, from h26xEncodeCapabilities.minQp/maxQp. The AV1 arm never +// writes it, so it reads 0 and the guard keyed on it is inert on exactly the +// codec whose scale wraps. AV1 device limits do exist, but in the other unit +// and on another object (EncoderConfigAV1::minQIndex/maxQIndex). This is the +// SYNTACTIC range, refused device-free; the device window stays the later +// and separate gate it already was. +static VkResult VkEncValidateConstQpRange( + int32_t constQpI, int32_t constQpP, int32_t constQpB, + VkVideoCodecOperationFlagBitsKHR codecOp, const char* where) +{ + const int32_t maxQuantizer = VkEncMaxConstQpForCodec(codecOp); + const struct { + const char* name; + int32_t value; + } named[] = { + {"constQpI", constQpI}, + {"constQpP", constQpP}, + {"constQpB", constQpB}, + }; + for (const auto& quantizer : named) { + if (quantizer.value > maxQuantizer) { + VkEncErr() << "[EncoderExt] " << where << quantizer.name << " " + << quantizer.value << " is outside the " + << ((codecOp == + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) + ? "AV1 quantizer-index range 0..255" + : "H.26x QP range 0..51") + << ". The unit is the codec's own, and the value is " + "refused rather than truncated." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + return VK_SUCCESS; +} + VkResult VulkanVideoEncoderExtImpl::BuildEncoderConfig( const VkVideoEncoderConfig& extConfig, VkVideoCodecOperationFlagBitsKHR codecOp, VkSharedBaseObj& outConfig) { - // Create the codec-specific config via the static factory - // We'll build argc/argv from the structured config for now. - // This is a bridge until EncoderConfig supports direct field assignment. + return VkEncBuildEncoderConfig(extConfig, codecOp, outConfig); +} - std::vector argStrings; - argStrings.push_back("encoder"); // argv[0] +// Section 4.2 layer 3: run the binder and flatten what it produced. Kept +// beside the binder so a new BOUND field is projected here in the same edit +// that binds it, and the conformance test then has something to assert on. +VkResult VkEncBuildAndProbeConfig(const VkVideoEncoderConfig& extConfig, + VkVideoCodecOperationFlagBitsKHR codecOp, + VkEncBoundConfigProbe* outProbe, + uint32_t requestedEncodeBitDepth) +{ + if (outProbe == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VkSharedBaseObj cfg; + const VkResult result = VkEncBuildEncoderConfig(extConfig, codecOp, cfg, + requestedEncodeBitDepth); + if ((result != VK_SUCCESS) || !cfg) { + return (result != VK_SUCCESS) ? result : VK_ERROR_INITIALIZATION_FAILED; + } + *outProbe = {}; + outProbe->encodeWidth = cfg->encodeWidth; + outProbe->encodeHeight = cfg->encodeHeight; + outProbe->inputWidth = cfg->input.width; + outProbe->inputHeight = cfg->input.height; + outProbe->inputBpp = cfg->input.bpp; + outProbe->rateControlMode = (uint32_t)cfg->rateControlMode; + outProbe->averageBitrate = (uint32_t)cfg->averageBitrate; + outProbe->maxBitrate = (uint32_t)cfg->maxBitrate; + outProbe->vbvBufferSize = (uint32_t)cfg->vbvBufferSize; + outProbe->constQpIntra = (uint32_t)cfg->constQp.qpIntra; + outProbe->constQpInterP = (uint32_t)cfg->constQp.qpInterP; + outProbe->constQpInterB = (uint32_t)cfg->constQp.qpInterB; + outProbe->minQp = cfg->minQp; + outProbe->maxQp = cfg->maxQp; + outProbe->minQpSet = cfg->minQpSet ? 1u : 0u; + outProbe->maxQpSet = cfg->maxQpSet ? 1u : 0u; + outProbe->constQpSet = cfg->constQpSet ? 1u : 0u; + outProbe->gopFrameCount = cfg->gopStructure.GetGopFrameCount(); + outProbe->idrPeriod = (uint32_t)cfg->gopStructure.GetIdrPeriod(); + outProbe->consecutiveBFrames = + cfg->gopStructure.GetConsecutiveBFrameCount(); + outProbe->closedGop = cfg->gopStructure.IsClosedGop() ? 1u : 0u; + outProbe->frameRateNumerator = cfg->frameRateNumerator; + outProbe->frameRateDenominator = cfg->frameRateDenominator; + outProbe->qualityLevel = (uint32_t)cfg->qualityLevel; + outProbe->tuningMode = (uint32_t)cfg->tuningMode; + outProbe->colourPrimaries = cfg->colour_primaries; + outProbe->transferCharacteristics = cfg->transfer_characteristics; + outProbe->matrixCoefficients = cfg->matrix_coefficients; + outProbe->videoFullRangeFlag = cfg->video_full_range_flag; + outProbe->colorDescriptionPresent = cfg->color_description_present_flag; + outProbe->videoSignalTypePresent = cfg->video_signal_type_present_flag; + outProbe->chromaLocInfoPresent = cfg->chroma_loc_info_present_flag; + outProbe->chromaSampleLocType = cfg->chroma_sample_loc_type; + outProbe->inputColourChainPresent = cfg->inputColourChainPresent; + outProbe->inputColourPrimaries = cfg->inputColourPrimaries; + outProbe->inputTransferCharacteristics = + cfg->inputTransferCharacteristics; + outProbe->inputMatrixCoefficients = cfg->inputMatrixCoefficients; + outProbe->inputRange = cfg->inputRange; + outProbe->hdrMasteringPresent = cfg->hdrMetadata.masteringDisplayPresent; + outProbe->hdrContentLightPresent = cfg->hdrMetadata.contentLightLevelPresent; + outProbe->hdrMaxDisplayMasteringLuminance = + cfg->hdrMetadata.maxDisplayMasteringLuminance; + outProbe->hdrMinDisplayMasteringLuminance = + cfg->hdrMetadata.minDisplayMasteringLuminance; + outProbe->hdrMaxContentLightLevel = cfg->hdrMetadata.maxContentLightLevel; + outProbe->hdrMaxFrameAverageLightLevel = + cfg->hdrMetadata.maxFrameAverageLightLevel; + outProbe->hdrGreenPrimaryX = cfg->hdrMetadata.displayPrimaryX[0]; + outProbe->hdrGreenPrimaryY = cfg->hdrMetadata.displayPrimaryY[0]; + outProbe->verbose = cfg->verbose ? 1u : 0u; + outProbe->validate = cfg->validate ? 1u : 0u; + outProbe->disableFileOutput = cfg->disableFileOutput ? 1u : 0u; + // Not a public config field: it is where the library's own + // preprocess-conversion decision lands. Read through the compile-safe + // accessor so the projection exists under both build gates -- and so a + // build without the filter honestly reports 0 for an input that would + // have needed one, which is the same build whose binder refuses that + // input outright. + outProbe->preprocessComputeFilter = + cfg->IsPreprocessComputeFilterEnabled() ? 1u : 0u; + // The only observable trace of inputFormat's plane layout: EncoderConfig + // derives input.vkFormat from this rather than storing the format. + outProbe->inputNumPlanes = cfg->input.numPlanes; + // The other half of that derivation, and what the encode profile is + // picked from: a 4:4:4 input must have moved this off the 4:2:0 default. + outProbe->inputChromaSubsampling = + (uint32_t)cfg->input.chromaSubsampling; + outProbe->inputVkFormat = (uint32_t)cfg->input.vkFormat; + outProbe->encodeChromaSubsampling = + (uint32_t)cfg->encodeChromaSubsampling; + outProbe->encodeBitDepthLuma = cfg->encodeBitDepthLuma; + outProbe->encodeBitDepthChroma = cfg->encodeBitDepthChroma; + // Per arm through virtual dispatch: the profile the session-creation + // path consumes (EncoderConfig::InitVideoProfile reads GetCodecProfile). + // Never INVALID here -- the binder either wrote the caller's profile or + // InitProfileLevel derived one inside InitializeParameters above. + outProbe->codecProfile = cfg->GetCodecProfile(); - // Codec (required by CreateCodecConfig to select H264/H265/AV1 subclass) + // Per-codec-arm projections. Everything here is device-free: the H.26x + // GetRateControlParameters overloads and the AV1 InitSequenceHeader read + // only config state the binder + FinalizeConfig already produced, so the + // conformance test can assert per-arm EFFECT, not just per-field storage. switch ((uint32_t)codecOp) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: { + EncoderConfigH264* h264 = cfg->GetEncoderConfigh264(); + if (h264 != nullptr) { + VkVideoEncodeRateControlInfoKHR rcInfo{}; + VkVideoEncodeRateControlLayerInfoKHR rcLayer{}; + VkVideoEncodeH264RateControlInfoKHR rcInfoH264{}; + VkVideoEncodeH264RateControlLayerInfoKHR rcLayerH264{}; + h264->GetRateControlParameters(&rcInfo, &rcLayer, + &rcInfoH264, &rcLayerH264); + outProbe->rcUseMinQp = (rcLayerH264.useMinQp == VK_TRUE) ? 1u : 0u; + outProbe->rcUseMaxQp = (rcLayerH264.useMaxQp == VK_TRUE) ? 1u : 0u; + outProbe->rcMinQpI = rcLayerH264.minQp.qpI; + outProbe->rcMaxQpI = rcLayerH264.maxQp.qpI; + + StdVideoH264SequenceParameterSetVui vui{}; + StdVideoH264HrdParameters hrd{}; + h264->InitVuiParameters(&vui, &hrd); + outProbe->vuiChromaLocInfoPresent = + vui.flags.chroma_loc_info_present_flag; + outProbe->vuiChromaSampleLocTypeTop = + vui.chroma_sample_loc_type_top_field; + outProbe->vuiChromaSampleLocTypeBottom = + vui.chroma_sample_loc_type_bottom_field; + } + } break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: { + EncoderConfigH265* h265 = cfg->GetEncoderConfigh265(); + if (h265 != nullptr) { + VkVideoEncodeRateControlInfoKHR rcInfo{}; + VkVideoEncodeRateControlLayerInfoKHR rcLayer{}; + VkVideoEncodeH265RateControlInfoKHR rcInfoH265{}; + VkVideoEncodeH265RateControlLayerInfoKHR rcLayerH265{}; + h265->GetRateControlParameters(&rcInfo, &rcLayer, + &rcInfoH265, &rcLayerH265); + outProbe->rcUseMinQp = (rcLayerH265.useMinQp == VK_TRUE) ? 1u : 0u; + outProbe->rcUseMaxQp = (rcLayerH265.useMaxQp == VK_TRUE) ? 1u : 0u; + outProbe->rcMinQpI = rcLayerH265.minQp.qpI; + outProbe->rcMaxQpI = rcLayerH265.maxQp.qpI; + outProbe->h265CpbVclFactor = h265->GetCpbVclFactor(); + outProbe->h265LevelIdc = (uint32_t)h265->levelIdc; + outProbe->h265GeneralTierFlag = h265->general_tier_flag; + + StdVideoH265SequenceParameterSetVui vui{}; + StdVideoH265HrdParameters hrd{}; + StdVideoH265SubLayerHrdParameters subHrd{}; + h265->InitVuiParameters(&vui, &hrd, &subHrd); + outProbe->vuiChromaLocInfoPresent = + vui.flags.chroma_loc_info_present_flag; + outProbe->vuiChromaSampleLocTypeTop = + vui.chroma_sample_loc_type_top_field; + outProbe->vuiChromaSampleLocTypeBottom = + vui.chroma_sample_loc_type_bottom_field; + } + } break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: { + EncoderConfigAV1* av1 = cfg->GetEncoderConfigAV1(); + if (av1 != nullptr) { + StdVideoAV1SequenceHeader seqHdr{}; + StdVideoEncodeAV1OperatingPointInfo opInfo{}; + av1->InitSequenceHeader(&seqHdr, &opInfo); + outProbe->av1ColorConfigPresent = + (seqHdr.pColorConfig != nullptr) ? 1u : 0u; + if (seqHdr.pColorConfig != nullptr) { + outProbe->av1ColorDescriptionPresent = + seqHdr.pColorConfig->flags.color_description_present_flag; + outProbe->av1ColorPrimaries = + (uint32_t)seqHdr.pColorConfig->color_primaries; + outProbe->av1TransferCharacteristics = + (uint32_t)seqHdr.pColorConfig->transfer_characteristics; + outProbe->av1MatrixCoefficients = + (uint32_t)seqHdr.pColorConfig->matrix_coefficients; + outProbe->av1ColorRange = + seqHdr.pColorConfig->flags.color_range; + outProbe->av1BitDepth = seqHdr.pColorConfig->BitDepth; + outProbe->av1ChromaSamplePosition = + (uint32_t)seqHdr.pColorConfig->chroma_sample_position; + outProbe->av1SubsamplingX = + seqHdr.pColorConfig->subsampling_x; + outProbe->av1SubsamplingY = + seqHdr.pColorConfig->subsampling_y; + } + } + } break; + default: + break; + } + return VK_SUCCESS; +} + +// The chroma subsamplings and the maximum component bit depth a codec profile +// admits, per the codec standard. +// +// THE STANDARD'S RULE AND ONLY THE STANDARD'S. Not this device's: a device may +// refuse H.264 High 4:4:4 Predictive above 8 bits while H.264 Table A-1 admits +// up to 14, and which of the two a caller has hit is answered by a device query +// and not from here. Not this library's binding set either -- but every row is +// now reachable through it: the guard below binds 66, 77, 100, 110, 122 and 244 +// for H.264, 1, 2, 3, 4 and 9 for H.265, and 0, 1 and 2 for AV1, which is every +// number this table states. +// +// TWO CALLERS, ONE TABLE, and that is why it is a table rather than a pair of +// literals at the two sites. The explicit-profile guard below refuses a named +// profile that cannot carry the declared input; VkEncQueryInputFormatSupport +// answers the same question before a session exists. Stated separately the two +// could drift, and the shape of that drift is a query that promises what +// InitializeExt then refuses -- the accepted-then-refused failure the input +// taxonomy exists to prevent. +// +// SUBSAMPLING IS STATED OVER {4:2:0, 4:2:2, 4:4:4} AND NOTHING ELSE. +// Monochrome is a chroma_format_idc that H.264 100/110/122/244 and H.265 4 +// all admit, and its absence is deliberate rather than an oversight: +// EncoderConfig::input.chromaSubsampling is DERIVED from the input VkFormat +// further down this file and that derivation has no monochrome arm, so a +// monochrome bit here would be a claim nothing can put to it. +struct VkEncProfileInputLimits { + uint32_t maxBpp; + VkVideoChromaSubsamplingFlagsKHR subsamplings; +}; + +// False when |profile| is not a number this table states for |codec|. That is +// NOT "the standard does not define it": it means nothing about the number +// should be inferred from here, and each caller decides what the silence +// means -- the guard falls through to its own bindability refusal, and the +// point query reports that it cannot answer. +static bool VkEncGetProfileInputLimits(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkEncProfileInputLimits& out) +{ + const VkVideoChromaSubsamplingFlagsKHR only420 = + VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + const VkVideoChromaSubsamplingFlagsKHR upTo422 = + only420 | VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR; + const VkVideoChromaSubsamplingFlagsKHR upTo444 = + upTo422 | VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR; + + switch ((uint32_t)codec) { case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: - argStrings.push_back("-c"); argStrings.push_back("h264"); break; + // ITU-T H.264 Annex A, Table A-1. + switch (profile) { + case STD_VIDEO_H264_PROFILE_IDC_BASELINE: + case STD_VIDEO_H264_PROFILE_IDC_MAIN: + case STD_VIDEO_H264_PROFILE_IDC_HIGH: + out.maxBpp = 8; out.subsamplings = only420; return true; + case STD_VIDEO_H264_PROFILE_IDC_HIGH_10: + out.maxBpp = 10; out.subsamplings = only420; return true; + case STD_VIDEO_H264_PROFILE_IDC_HIGH_422: + out.maxBpp = 10; out.subsamplings = upTo422; return true; + case STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE: + out.maxBpp = 14; out.subsamplings = upTo444; return true; + default: + return false; + } case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: - argStrings.push_back("-c"); argStrings.push_back("h265"); break; + // ITU-T H.265 Annex A.3. + switch (profile) { + case STD_VIDEO_H265_PROFILE_IDC_MAIN: + case STD_VIDEO_H265_PROFILE_IDC_MAIN_STILL_PICTURE: + out.maxBpp = 8; out.subsamplings = only420; return true; + case STD_VIDEO_H265_PROFILE_IDC_MAIN_10: + out.maxBpp = 10; out.subsamplings = only420; return true; + case STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS: + case STD_VIDEO_H265_PROFILE_IDC_SCC_EXTENSIONS: + out.maxBpp = 16; out.subsamplings = upTo444; return true; + default: + return false; + } case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: - argStrings.push_back("-c"); argStrings.push_back("av1"); break; - default: break; - } - - // No -i flag: external frame input mode. ParseArguments handles this - // by skipping file handler setup when inputFileHandler.HasFileName() is false. - - // Resolution - argStrings.push_back("--inputWidth"); - argStrings.push_back(std::to_string(extConfig.inputWidth)); - argStrings.push_back("--inputHeight"); - argStrings.push_back(std::to_string(extConfig.inputHeight)); - argStrings.push_back("--encodeWidth"); - argStrings.push_back(std::to_string(extConfig.encodeWidth)); - argStrings.push_back("--encodeHeight"); - argStrings.push_back(std::to_string(extConfig.encodeHeight)); - - // Input geometry: derive bit depth, chroma subsampling AND plane count from - // the VkFormat. - // - // This is the only consumer of VkVideoEncoderConfig::inputFormat, so all three - // properties have to be read off it here. A derivation that sets only --inputBpp - // leaves chromaSubsampling and numPlanes at their 4:2:0 defaults and pins every - // caller of this library to 4:2:0 regardless of the format it asked for. - // - // Nothing else needs to change for that: EncoderConfig parses - // --inputChromaSubsampling and --inputNumPlanes, and CodecGetVkFormat covers 422/444 - // at 8/10/12-bit, so the encode side is generic once the geometry arrives correct. + // AV1 6.4.1 and A.2. Main is 4:2:0 at 8 or 10 bits, High is 4:4:4 + // at the same two, and Professional is the one that reaches 12. + switch (profile) { + case STD_VIDEO_AV1_PROFILE_MAIN: + out.maxBpp = 10; out.subsamplings = only420; return true; + case STD_VIDEO_AV1_PROFILE_HIGH: + out.maxBpp = 10; + out.subsamplings = VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR; + return true; + case STD_VIDEO_AV1_PROFILE_PROFESSIONAL: + out.maxBpp = 12; out.subsamplings = upTo444; return true; + default: + return false; + } + default: + return false; + } +} + +// The subsampling a refusal names, so a caller reads back what it declared +// rather than a flag value. The three the input derivation can produce, and a +// fallback that derivation cannot reach. +static const char* VkEncChromaSubsamplingName( + VkVideoChromaSubsamplingFlagBitsKHR subsampling) +{ + switch ((uint32_t)subsampling) { + case VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR: return "4:2:0"; + case VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR: return "4:2:2"; + case VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR: return "4:4:4"; + // Unreachable from the input derivation, which produces only the three + // above -- and named rather than left as a pronoun because this + // function has two callers now and one of them prints it beside the + // depth, where "its" would read as a missing word rather than a + // fallback. + default: return "an unnamed chroma subsampling"; + } +} + +// VK_SUCCESS when |profile| can carry an input at |bpp| bits and +// |subsampling|, and otherwise the refusal, having already reported WHICH +// term of the standard's rule it failed and what to do instead. +// +// CALLED ONLY FROM THE ARMS THAT CAN BIND THE NUMBER, and the ordering is the +// point. A profile this library cannot bind at all -- AV1 High (1), say -- +// must be refused as unbindable, because that is the caller's actual problem; +// telling it instead that AV1 High does not admit 4:2:0 is true, and useless, +// since no input format would make the request succeed. So bindability is +// settled first and this runs inside the arms that survived it. +// +// |bpp| is taken as uint32_t deliberately: EncoderConfig::input.bpp is a +// uint8_t and streams as a CHARACTER, which silently emptied the number out +// of this diagnostic when it was written against the field's own type. +static VkResult VkEncRefuseIfProfileCannotCarryInput( + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t bpp, + VkVideoChromaSubsamplingFlagBitsKHR subsampling) +{ + VkEncProfileInputLimits limits = {}; + if (!VkEncGetProfileInputLimits(codec, profile, limits)) { + // The table states nothing about this number. It is not this + // function's place to invent a constraint, and the arm that called it + // has already decided the number is bindable. + return VK_SUCCESS; + } + if (bpp > limits.maxBpp) { + VkEncErr() << "[EncoderExt] profile " << profile + << " does not admit " << bpp + << "-bit input: the codec standard gives it at most " + << limits.maxBpp + << " bits per component. Submit frames at a depth it " + "admits, or use VK_VIDEO_ENCODER_PROFILE_DEFAULT, which " + "derives the profile from the input." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if ((limits.subsamplings & subsampling) == 0) { + VkEncErr() << "[EncoderExt] profile " << profile + << " does not admit " + << VkEncChromaSubsamplingName(subsampling) + << " input: the codec standard gives it" + << (((limits.subsamplings & + VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR) != 0) + ? " 4:2:0" : "") + << (((limits.subsamplings & + VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR) != 0) + ? " 4:2:2" : "") + << (((limits.subsamplings & + VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR) != 0) + ? " 4:4:4" : "") + << " only. The subsampling is read off the input format, so " + "submit frames at an admitted one, or use " + "VK_VIDEO_ENCODER_PROFILE_DEFAULT, which derives the " + "profile from the input's own subsampling -- 4:2:2 " + "derives H.264 High 4:2:2 (122) and 4:4:4 derives H.264 " + "High 4:4:4 Predictive (244) or H.265 Range Extensions " + "(4)." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + return VK_SUCCESS; +} + +// Free function: reads only its arguments, so a test can drive it with no +// device and no encoder -- VkEncBuildAndProbeConfig above is that entry. +// The layer-3 binder suite (VulkanVideoEncoderConfigBinderTest in Chromium's +// vulkan_video_encode_accelerator_unittest.cc) drives it per codec arm and +// asserts effect-or-explicit-rejection for every BOUND field in the section +// 4.2 field table except outputPath, which has no probe projection: it +// binds as a conditional file-open (see the disableFileOutput interplay +// below) and no test exercises it -- the Chromium consumer nulls the field +// by design. +VkResult VkEncBuildEncoderConfig( + const VkVideoEncoderConfig& extConfig, + VkVideoCodecOperationFlagBitsKHR codecOp, + VkSharedBaseObj& outConfig, + uint32_t requestedEncodeBitDepth) +{ + // Direct binder: create the codec-typed config, assign the fields on it, + // then run the same derived tail (FinalizeConfig) + InitializeParameters + // the ParseArguments pipeline runs. No argv round-trip -- a field no + // longer needs a CLI flag to exist, so nothing here can be silently + // dropped by the parser. + VkResult result = EncoderConfig::CreateCodecConfigDirect(codecOp, outConfig); + if ((result != VK_SUCCESS) || !outConfig) { + return (result != VK_SUCCESS) ? result : VK_ERROR_INITIALIZATION_FAILED; + } + EncoderConfig* cfg = outConfig.get(); + + // THE ENCODE DEPTH, WHEN IT IS STATED RATHER THAN DERIVED. Written here, + // before InitializeParameters, because that is where its zero-means-unset + // guard reads it: set, the derivation from input.bpp does not run and the + // encode side is the request. Zero leaves the derivation in charge, which + // is every caller but the internal probe. + if (requestedEncodeBitDepth != 0) { + cfg->encodeBitDepthLuma = (uint8_t)requestedEncodeBitDepth; + } + + // A2 clause 2: reject an unencodable input format at INIT, not only + // per frame. SubmitExternalFrame already rejects it -- but only after the + // producer has allocated an entire frame pool in a format this encoder + // will never take. Init is the last point at which it can still choose + // differently, which is why the finding asked for both. + // + // The colour-model DECLARATION is judged first, and separately, because + // the two refusals are about different fields. VkEncResolveColorModel + // answers VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT for a + // (format, colorModel) pair that cannot be reconciled -- RGB declared + // over a Y'CbCr format, Y'CbCr declared over an RGBA layout that carries + // no packed 4:4:4 reading, or a value the enumeration does not define -- + // and there is no route to choose from an answer that means "the caller + // stated two things that cannot both be true". This is the same + // predicate, and the same refusal, the registration gate applies to a + // descriptor, so one contradiction is answered once. + // + // Separated from the format refusal below because the FORMAT in such a + // pair is usually the half that is not wrong: NV12 declared RGB reaches + // here and NV12 is directly encodable, and the format message would + // answer it by naming the submitted format among the ones it accepts. + // The field the caller has to change is inputColorModel, and the + // refusal says so. + // + // FROM_FORMAT -- what a zero-initialised config declares, and the + // ordinary case -- reads the model off the format and passes wherever + // the format does, as does a declaration that agrees with its format. + if (VkEncResolveColorModel(extConfig.inputFormat, + extConfig.inputColorModel) == + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) { + VkEncErr() << "[EncoderExt] declared inputColorModel " + << (uint32_t)extConfig.inputColorModel + << " contradicts inputFormat " + << (uint32_t)extConfig.inputFormat + << "; the two cannot both be true and nothing here can " + "know which was meant. Declare the colour model the " + "format carries, or leave the field " + "VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT to read it " + "off the format." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + const VkEncInputFormatClass inputFormatClass = + VkEncClassifyInput(extConfig.inputFormat, extConfig.inputColorModel); + if (inputFormatClass == VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + // THE SET IS WALKED, NOT NAMED. A sentence listing the accepted + // formats is a third statement of the routable set, beside the + // classifier and the enumeration, and it is the one nothing tests: + // this message named "NV12, P010, NV24 and S410 ... P012, I420 and + // its 10/12-bit siblings" and was already narrower than the set the + // classifier answered for. Printing the derived list cannot go stale. + uint32_t routableCount = 0; + const VkFormat* const routable = + VkEncRoutableInputFormats(routableCount); + std::ostringstream routableText; + for (uint32_t i = 0; i < routableCount; i++) { + routableText << ((i == 0) ? "" : ", ") << (uint32_t)routable[i]; + } + VkEncErr() << "[EncoderExt] inputFormat " + << (uint32_t)extConfig.inputFormat + << " is not encodable. The VkFormat values this library " + "routes, directly or through the preprocess compute " + "filter, are: " + << routableText.str() + << ". Two more are reached only by DECLARING " + "VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR over an RGBA " + "enumerant -- the packed 4:4:4 layouts AYUV " + "(R8G8B8A8_UNORM) and Y410 (A2B10G10R10_UNORM_PACK32) " + "-- so they are not on that list and their absence " + "from it is not a refusal. Convert before submitting. " + "Note that Y416 (R16G16B16A16_UNORM) is not taken: 16 " + "bits per component is not an encode component bit " + "depth. Note also that the _SRGB " + "spellings are deliberately NOT accepted -- the filter " + "binds its RGBA input as a storage image, and no _SRGB " + "format carries VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT, so " + "an sRGB view can never be that descriptor; submit the " + "_UNORM spelling of the same format instead." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // THE FILTER DECISION, AND IT IS THE LIBRARY'S. A format that is + // encodable only THROUGH the preprocess compute filter gets the filter; a + // directly encodable one does not. There is no request to honour: the + // class is a pure function of the declared input pair, inputFormat and + // inputColorModel, both of which state what the frames ARE rather than + // ask for a conversion. That pair is why this is decided here, in a + // binder with no device, and is answerable on the query surface before a + // producer allocates a frame pool. + const bool needsPreprocessFilter = + (inputFormatClass == VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER); + + // Refusing rather than dropping the conversion: without the filter the + // frames that needed converting would fall to the staging copy, and for a + // plane-count or colour-model mismatch that copy does not encode slowly, + // it encodes wrongly -- from a three-plane source it hangs the GPU. So + // the answer to an input that needs converting is a conversion or an + // error, never a quiet no. + // + // This is the half a device-free binder CAN check: that the filter is + // compiled into this build. The session-level half -- that the device has + // a compute queue to run it on, which under a caller-supplied VkDevice + // cannot be assumed -- is checked in InitializeExt, after the device + // exists. Both halves must hold. +#ifndef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + if (needsPreprocessFilter) { + VkEncErr() << "[EncoderExt] inputFormat " + << (uint32_t)extConfig.inputFormat + << " is encodable only through the preprocess compute " + "filter, which is not compiled into this build " + "(VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED is " + "undefined; the CMake option is " + "BUILD_ENCODER_COMPUTE_FILTER). Convert before " + "submitting, or submit a semi-planar 4:2:0 format." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } +#endif + + // Resolution. + cfg->input.width = extConfig.inputWidth; + cfg->input.height = extConfig.inputHeight; + cfg->encodeWidth = extConfig.encodeWidth; + cfg->encodeHeight = extConfig.encodeHeight; + + // Input geometry: derive the bit depth, chroma subsampling AND plane count + // from the input VkFormat. + // + // This is the only consumer of VkVideoEncoderConfig::inputFormat. Deriving + // the bit depth alone and leaving chromaSubsampling and numPlanes at their + // 4:2:0 defaults pins every caller of this library to 4:2:0 whatever format + // it asked for: a 4:4:4 or 4:2:2 request then encodes as 4:2:0 and reports + // success. CodecGetVkFormat covers 4:2:2 and 4:4:4 at 8, 10 and 12 bits, so + // the encode side is already generic; only the derivation is needed here. const VkMpFormatInfo* mpInfo = YcbcrVkFormatInfo(extConfig.inputFormat); if (mpInfo != nullptr) { const uint32_t bpp = GetBitsPerChannel(mpInfo->planesLayout); if (bpp > 8) { - argStrings.push_back("--inputBpp"); - argStrings.push_back(std::to_string(bpp)); + cfg->input.bpp = bpp; } // secondaryPlaneSubsampledX/Y are 1-bit flags: 0 = full rate, 1 = halved. // 4:2:0 -> X=1, Y=1 4:2:2 -> X=1, Y=0 4:4:4 -> X=0, Y=0 const bool subX = (mpInfo->planesLayout.secondaryPlaneSubsampledX != 0); const bool subY = (mpInfo->planesLayout.secondaryPlaneSubsampledY != 0); - const char* chroma = (!subX && !subY) ? "444" : (subX && !subY) ? "422" : "420"; - argStrings.push_back("--inputChromaSubsampling"); - argStrings.push_back(chroma); + cfg->input.chromaSubsampling = + (!subX && !subY) ? VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR : + (subX && !subY) ? VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR : + VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; // numberOfExtraPlanes: 1 for semi-planar (2-plane), 2 for 3-plane planar. const uint32_t numPlanes = mpInfo->planesLayout.numberOfExtraPlanes + 1u; if ((numPlanes == 2) || (numPlanes == 3)) { - argStrings.push_back("--inputNumPlanes"); - argStrings.push_back(std::to_string(numPlanes)); + cfg->input.numPlanes = numPlanes; } } - // else: not a YCbCr VkFormat. Either genuine RGB input destined for the - // RGBA->YCbCr filter, or one of the packed 4:4:4 aliases (AYUV on - // R8G8B8A8_UNORM, Y410 on A2B10G10R10_UNORM_PACK32) which have no YCbCr - // VkFormat of their own. These two cases are INDISTINGUISHABLE from the format - // alone -- separating them requires an explicit input colour model - // (VkSamplerYcbcrModelConversion: RGB_IDENTITY vs YCBCR_601/709/2020), which - // VkVideoEncoderConfig does not yet carry. Until it does, leave the geometry at - // its defaults rather than guessing. + // else: not a Y'CbCr VkFormat. Either RGB input destined for the + // RGBA->Y'CbCr preprocess filter, or one of the packed 4:4:4 aliases (AYUV + // on R8G8B8A8_UNORM, Y410 on A2B10G10R10_UNORM_PACK32) which have no Y'CbCr + // VkFormat of their own. The two are indistinguishable from the VkFormat + // alone, so the input-format taxonomy is what separates them: the two arms + // immediately below read it. A format the taxonomy does not place keeps the + // default geometry rather than a guess. + + // RGBA is the one input family whose format cannot be RECONSTRUCTED from + // (subsampling, bit depth, plane count) -- CodecGetVkFormat() only spells + // Y'CbCr. So for RGBA the format has to be carried through literally, and + // input.colorSpace is what tells VerifyInputs() to carry rather than + // re-derive it (and to lay the image out as one 4-byte-per-pixel plane). + // Writing vkFormat unconditionally would be pointless for the Y'CbCr families -- + // VerifyInputs() overwrites it there by design, which is how a semi-planar + // session still gets the format its subsampling and bit depth imply. The + // plane count is set here for the same reason: the derivation above speaks + // only for Y'CbCr formats, and the compute filter is built from this field. + // + // The pair is known resolvable by the time it reaches here -- an + // unresolvable one was refused at the top of this function -- so this + // reads as the two-way choice it is written as, and not as a fall to + // Y'CbCr for an answer that meant neither. + cfg->input.colorSpace = + (VkEncResolveColorModel(extConfig.inputFormat, + extConfig.inputColorModel) == + VK_VIDEO_ENCODER_COLOR_MODEL_RGB) + ? VkEncColorSpace::kRGB + : VkEncColorSpace::kYCbCr; + // A caller-supplied image carries its samples where its VkFormat says they + // are, so the alignment is declared here rather than detected. + // DetectInputMsbShift exists to sniff an input FILE's content; with no file + // to read it returns the documented default, which claims an LSB-aligned + // source and makes the preprocess filter scale every sample by 2^msbShift. + // The X6/X4-packed formats this interface accepts hold their samples in the + // high bits, and msbShift == 0 is how that is spelled. + cfg->input.msbShift = 0; - // Frame rate: no CLI arg for this in ParseArguments — set via member directly after config + if (cfg->input.colorSpace == VkEncColorSpace::kRGB) { + cfg->input.numPlanes = VkEncInputFormatPlaneCount(extConfig.inputFormat); + cfg->input.vkFormat = extConfig.inputFormat; + } else if (mpInfo == nullptr) { + // A Y'CbCr input the multi-planar table does not place is a packed + // 4:4:4 alias declared as such -- nothing else resolves to Y'CbCr and + // survives the class gate above. Its geometry has to be written here + // for the same reason the RGBA arm's does: the derivation above + // speaks only for formats that table holds, and left alone the input + // would keep EncoderConfig's 3-plane 4:2:0 default and the session + // would be configured as I420 while the caller declared AYUV. + // + // vkFormat is DELIBERATELY not written, unlike the RGBA arm. + // VerifyInputs() reconstructs it from exactly these three values -- + // CodecGetVkFormat(4:4:4, bitDepth, PLANE_LAYOUT_PACKED_1) spells + // AYUV at 8 bits and Y410 at 10 -- so writing it would be a second + // statement of the same fact, and the round trip is what makes the + // geometry below sufficient rather than merely plausible. + const VkPackedYcbcrFormatDesc* packed = + PackedYcbcrFormatDesc(extConfig.inputFormat); + if (packed != nullptr) { + cfg->input.bpp = packed->bitDepth; + cfg->input.chromaSubsampling = + VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR; + cfg->input.numPlanes = + VkEncInputFormatPlaneCount(extConfig.inputFormat); + } + } - // Tuning mode (VkVideoEncodeTuningModeKHR). LOSSLESS engages transquant - // bypass + QP0 in the codec config. - const bool lossless = (extConfig.tuningMode == 4 /* LOSSLESS */); + // ---- Tuning / rate control ---- switch (extConfig.tuningMode) { - case 1: argStrings.push_back("--tuningMode"); argStrings.push_back("highquality"); break; - case 2: argStrings.push_back("--tuningMode"); argStrings.push_back("lowlatency"); break; - case 3: argStrings.push_back("--tuningMode"); argStrings.push_back("ultralowlatency"); break; - case 4: argStrings.push_back("--tuningMode"); argStrings.push_back("lossless"); break; - default: break; - } - - // Rate control. Constant-QP (a.k.a. DISABLED) when explicitly requested or - // when lossless. Otherwise forward the selected bitrate mode. Previously the - // ext path forwarded ONLY --averageBitrate with no mode, so constant-QP and - // lossless could never be requested and QP0 was dropped. - const bool constantQp = (extConfig.rateControlMode == 1 /* DISABLED */) || lossless; + case VK_VIDEO_ENCODE_TUNING_MODE_HIGH_QUALITY_KHR: + case VK_VIDEO_ENCODE_TUNING_MODE_LOW_LATENCY_KHR: + case VK_VIDEO_ENCODE_TUNING_MODE_ULTRA_LOW_LATENCY_KHR: + case VK_VIDEO_ENCODE_TUNING_MODE_LOSSLESS_KHR: + cfg->tuningMode = extConfig.tuningMode; + break; + case VK_VIDEO_ENCODE_TUNING_MODE_DEFAULT_KHR: + break; // the caller explicitly wants the codec default + default: + // Reject rather than silently substituting the codec default: a + // caller that miscomputed this value would otherwise ship with + // tuning it never selected and no diagnostic anywhere. + VkEncErr() << "[EncoderExt] unrecognized tuningMode " + << (uint32_t)extConfig.tuningMode << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // THE CONSTANT QUANTIZERS, RANGE-CHECKED BEFORE ANYTHING READS THEM. + // + // Checked on every session and not only a constant-QP one. The three + // fields are the caller's statement in the codec's own units whatever + // the rate-control mode; Reconfigure records and applies them whatever + // the mode; and a rule that held only under DISABLED would be a second + // contract for the same three fields, which the header would then have + // to state twice. + result = VkEncValidateConstQpRange(extConfig.constQpI, extConfig.constQpP, + extConfig.constQpB, codecOp, ""); + if (result != VK_SUCCESS) { + return result; + } + + // Constant-QP (DISABLED) when requested explicitly or when lossless. + const bool lossless = + (extConfig.tuningMode == VK_VIDEO_ENCODE_TUNING_MODE_LOSSLESS_KHR); + const bool constantQp = + (extConfig.rateControlMode == + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR) || + lossless; if (constantQp) { - argStrings.push_back("--rateControlMode"); - argStrings.push_back("disabled"); - // Forward QP including 0 (lossless). For lossless, default unset QP to 0. - int32_t qpI = (extConfig.constQpI >= 0) ? extConfig.constQpI : (lossless ? 0 : 26); - int32_t qpP = (extConfig.constQpP >= 0) ? extConfig.constQpP : qpI; - int32_t qpB = (extConfig.constQpB >= 0) ? extConfig.constQpB : qpP; - argStrings.push_back("--qpI"); argStrings.push_back(std::to_string(qpI)); - argStrings.push_back("--qpP"); argStrings.push_back(std::to_string(qpP)); - argStrings.push_back("--qpB"); argStrings.push_back(std::to_string(qpB)); + cfg->rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + // THE DEFAULT IS CODEC-DEPENDENT because the UNIT is. For H.264/H.265 + // this field is a QP on 0..51 and 26 is mid-range. AV1 has no QP: the + // value lands in base_q_idx verbatim -- VkVideoEncoderAV1::EncodeFrame + // writes pictureInfo.constantQIndex from the resolved constQp and + // stdQuantInfo.base_q_idx from that -- on a 0..255 quantizer-index + // scale, where 26 is libaom quantizer 7 of 63: near-lossless, and a + // bitrate to match. + // + // CITED BY FUNCTION AND SYMBOL, not by line. Line references here drift -- + // onto picOrderCntVal arithmetic and srcPictureResource setup, in the two + // cases this paragraph would otherwise carry -- so a reader who follows one + // finds no quantizer at all and concludes the value never reaches the + // bitstream. A symbol survives the edits a line number does not. + // + // So when the caller specified NOTHING for an AV1 session, leave the + // qindices unset rather than inventing one here. + // EncoderConfigAV1::InitDeviceCapabilities then substitutes the + // driver's own preferredConstantQIndex triple -- the only AV1-aware + // default available at this layer. That substitution is guarded on + // constQpSet, which is precisely what this arm must NOT set. + // + // Every other case keeps the established semantics: an explicit + // quantizer is RESOLVED AND CARRIED including 0, lossless resolves to + // 0, P inherits I, B inherits P, and constQpSet marks the result + // fully resolved so the substitution above stays out. + // + // THAT IS A LIBRARY GUARANTEE AND IT STOPS AT THE DRIVER. What this + // arm promises is that the quantizer the caller named is the one the + // library resolves, records and hands down -- NOT that it is the one + // the bitstream comes back carrying. On AV1 with driver 620.18 the + // two part company at exactly one value: an explicit 0 reads back + // base_q_idx 114 on KEY and 131 on INTER, which is the same pair a + // session that named nothing at all receives. The driver is treating + // base_q_idx 0 as unspecified and substituting its own preference; 1 + // is honoured exactly. + // + // 0 IS NEITHER NORMALISED NOR REFUSED HERE. It is a legal AV1 + // quantizer index -- the lossless one -- it is what the LOSSLESS + // tuning mode resolves to a few lines below, and a consumer already + // asserts it survives this binder unchanged. Rewriting it to suit a + // driver that does not honour it would be the silent alteration the + // range check above exists to remove, and would ratify the driver + // behaviour in the library's own contract. The consequence worth + // knowing is that AV1 lossless requested this way does not come back + // lossless on that driver. That is a driver-side deviation, and not + // one this layer can correct by sending a value other than the one it + // was given. + const bool isAv1 = + (extConfig.codec == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR); + const bool av1QIndexUnspecified = + isAv1 && !lossless && (extConfig.constQpI < 0) && + (extConfig.constQpP < 0) && (extConfig.constQpB < 0); + if (av1QIndexUnspecified) { + cfg->constQp.qpIntra = 0; + cfg->constQp.qpInterP = 0; + cfg->constQp.qpInterB = 0; + cfg->constQpSet = 0; + } else { + const int32_t qpI = (extConfig.constQpI >= 0) ? extConfig.constQpI : (lossless ? 0 : 26); + const int32_t qpP = (extConfig.constQpP >= 0) ? extConfig.constQpP : qpI; + const int32_t qpB = (extConfig.constQpB >= 0) ? extConfig.constQpB : qpP; + cfg->constQp.qpIntra = (uint32_t)qpI; + cfg->constQp.qpInterP = (uint32_t)qpP; + cfg->constQp.qpInterB = (uint32_t)qpB; + cfg->constQpSet = 1; + } } else { - // Explicit RC mode selector so behavior is deterministic (2=CBR, 3=VBR). - if (extConfig.rateControlMode == 2) { argStrings.push_back("--rateControlMode"); argStrings.push_back("cbr"); } - else if (extConfig.rateControlMode == 3) { argStrings.push_back("--rateControlMode"); argStrings.push_back("vbr"); } + if ((extConfig.rateControlMode == + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR) || + (extConfig.rateControlMode == + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_BIT_KHR)) { + cfg->rateControlMode = extConfig.rateControlMode; + } else if (extConfig.rateControlMode != + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR) { + // Anything else is a caller error. Value 3 earns its own message: + // it was this API's VBR before the mode field was aligned onto + // the Vulkan bit values, so a consumer still passing its old + // constant would otherwise encode at the codec default rate + // control and look merely mistuned rather than misconfigured. + if (extConfig.rateControlMode == 3u) { + VkEncErr() << "[EncoderExt] rateControlMode 3 is the " + "VBR value this library does not use; VBR is " + << (uint32_t)VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_BIT_KHR + << std::endl; + } else { + VkEncErr() << "[EncoderExt] unrecognized rateControlMode " + << (uint32_t)extConfig.rateControlMode << std::endl; + } + return VK_ERROR_INITIALIZATION_FAILED; + } if (extConfig.averageBitrate > 0) { - argStrings.push_back("--averageBitrate"); - argStrings.push_back(std::to_string(extConfig.averageBitrate)); + cfg->averageBitrate = extConfig.averageBitrate; } if (extConfig.maxBitrate > 0) { - argStrings.push_back("--maxBitrate"); - argStrings.push_back(std::to_string(extConfig.maxBitrate)); + cfg->maxBitrate = extConfig.maxBitrate; } } - if (extConfig.minQp > 0) { argStrings.push_back("--minQp"); argStrings.push_back(std::to_string(extConfig.minQp)); } - if (extConfig.maxQp > 0) { argStrings.push_back("--maxQp"); argStrings.push_back(std::to_string(extConfig.maxQp)); } + if (extConfig.vbvBufferSize > 0) { + // Feeds H.264/H.265 level selection and the HRD/CPB parameters. + cfg->vbvBufferSize = extConfig.vbvBufferSize; + } + if ((extConfig.minQp != 0) || (extConfig.maxQp != 0)) { + if (codecOp == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) { + // A1 forward-or-reject: these fields are H.26x-unit QP clamps + // (0..51). AV1 rate control is quantizer-index based (0..255) + // and the AV1 config consumes no QP-unit clamp, so a non-zero + // value here was accepted-and-ignored on every AV1 session. + // Reject loudly rather than accept and ignore; qIndex clamp + // fields are a versioned addition for when a consumer needs + // them. + VkEncErr() << "[EncoderExt] minQp/maxQp are H.26x-unit QP " + "clamps (0..51); AV1 rate control uses quantizer " + "indices (0..255) and this config version has no " + "qIndex clamp fields. Leave both 0 on AV1 " + "sessions." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // Syntactic H.26x range; the device's supported QP window is + // narrower on some implementations and is checked with the + // capabilities at device init. + if ((extConfig.minQp < 0) || (extConfig.minQp > 51) || + (extConfig.maxQp < 0) || (extConfig.maxQp > 51)) { + VkEncErr() << "[EncoderExt] minQp/maxQp outside the H.26x QP " + "range 0..51 (minQp=" << extConfig.minQp + << ", maxQp=" << extConfig.maxQp << ")" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if ((extConfig.minQp > 0) && (extConfig.maxQp > 0) && + (extConfig.minQp > extConfig.maxQp)) { + VkEncErr() << "[EncoderExt] minQp " << extConfig.minQp + << " > maxQp " << extConfig.maxQp + << " -- inverted clamp window" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + // 0 keeps a field unset: no clamp reaches the driver. An explicit + // minQp=0 collapses onto unset BY DESIGN (QP 0 is the codec floor, so + // the two admit the same QP range); maxQp=0 (force QP 0) is not + // expressible -- see the header note on these fields. + if (extConfig.minQp > 0) { + cfg->minQp = extConfig.minQp; + cfg->minQpSet = 1; + } + if (extConfig.maxQp > 0) { + cfg->maxQp = extConfig.maxQp; + cfg->maxQpSet = 1; + } - // GOP + // ---- GOP ---- if (extConfig.gopLength > 0) { - argStrings.push_back("--gopFrameCount"); - argStrings.push_back(std::to_string(extConfig.gopLength)); - } - // Consecutive-B count: pass through when the caller sets it; otherwise - // EncoderConfig adopts the codec's driver-preferred value from the - // quality-level caps (5 for AV1 when supported). B-frame GOPs reorder the - // encode submissions relative to the input order; the producer's - // input-release timeline stays correct because VkVideoEncoder signals - // the release at queue flush points (end of each ordered batch) with - // the max submitted release value — see SubmitVideoCodingCmds. Note the - // producer's frame pool must be deeper than one mini-GOP (B count + 1), - // since a mini-GOP's inputs are held until its batch flushes. - if (extConfig.consecutiveBFrames > 0) { - argStrings.push_back("--consecutiveBFrameCount"); - argStrings.push_back(std::to_string(extConfig.consecutiveBFrames)); - } - - // Quality + // Pass through at full width. SetGopFrameCount takes uint32_t and + // m_gopFrameCount is uint32_t -- only the constructor's default + // parameter is 8-bit, so an earlier clamp to 255 here rested on a + // false premise and silently rewrote long GOPs: gopLength was + // clamped while idrPeriod was not, and GetPositionInGOP's + // `positionInInputOrder % m_gopFrameCount` then emitted a non-IDR + // intra every 255 frames that no caller had asked for. + cfg->gopStructure.SetGopFrameCount(extConfig.gopLength); + } + // Communicate the B-frame count UNCONDITIONALLY: the gopStructure + // defaults to 2 consecutive B-frames, and omitting the assignment when + // the caller requests 0 would silently encode B-frames against the + // caller's intent (0 == IPPP). When B-frames ARE requested, encode + // submissions reorder relative to input order; the producer's + // input-release timeline stays correct because the encoder signals + // releases at queue flush points. The producer's frame pool must be + // deeper than one mini-GOP (B count + 1). + if (extConfig.consecutiveBFrames == + VK_VIDEO_ENCODER_B_FRAMES_DRIVER_PREFERRED) { + // Leave the library's sentinel in place: InitDeviceCapabilities then + // adopts the driver's preferred count. This is the capability the + // upstream argv path gets by omitting the flag; here it is asked for + // explicitly instead of inferred from an unset field. + } else if (extConfig.consecutiveBFrames >= + (uint32_t)EncoderConfig::CONSECUTIVE_B_FRAME_COUNT_MAX_VALUE) { + // 255 IS the sentinel. The previous clamp turned any larger request + // into it, so asking for 300 B-frames silently selected + // driver-preferred -- a different feature, chosen by accident. + VkEncErr() << "[EncoderExt] consecutiveBFrames " + << extConfig.consecutiveBFrames + << " is out of range (0.." + << (uint32_t)EncoderConfig::CONSECUTIVE_B_FRAME_COUNT_MAX_VALUE - 1 + << ", or VK_VIDEO_ENCODER_B_FRAMES_DRIVER_PREFERRED)" + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } else { + // AV1 CAPTURE CANNOT REORDER IN THIS RELEASE. + // + // Reordering AV1 emits show-existing-frame headers, and the temporal + // unit a capture consumer receives is assembled by the file writer, + // not by the capture path. Capturing a reordered AV1 stream therefore + // hands back temporal units missing those headers, whose frame + // identity no consumer can reconstruct. Refuse the mode instead of + // producing a stream that decodes to the wrong pictures. + // + // File output keeps B-frames, and capture keeps B=0, which is what + // Chromium uses. The driver-preferred sentinel is not decided here -- + // it resolves in InitDeviceCapabilities -- so the definitive refusal + // is repeated in the core once the effective count is known. + if ((codecOp == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) && + (extConfig.disableFileOutput == VK_TRUE) && + (extConfig.consecutiveBFrames > 0)) { + VkEncErr() << "[EncoderExt] AV1 in-memory capture does not support " + "B-frames in this release (consecutiveBFrames=" + << extConfig.consecutiveBFrames + << "); use file output, or request 0" << std::endl; + return VK_ERROR_FEATURE_NOT_PRESENT; + } + // 0 binds as 0: no B-frames, which is what both this header and the + // reference renderer's config document, and what the Chromium frame + // tracking requires. + cfg->gopStructure.SetConsecutiveBFrameCount( + (uint8_t)extConfig.consecutiveBFrames); + } + if (extConfig.idrPeriod > 0) { + // 0 keeps the sentinel default so InitDeviceCapabilities can apply + // the driver's preferredIdrPeriod. + cfg->gopStructure.SetIdrPeriod((int32_t)extConfig.idrPeriod); + } + if (extConfig.closedGop == VK_TRUE) { + cfg->gopStructure.SetClosedGop(); + } + + // ---- Frame rate ---- + // Guarded so zero/unset fields keep the codec-config default (30000/1001) + // instead of poisoning rate control, VUI timing and the AV1 timebase. + if ((extConfig.frameRateNum > 0) && (extConfig.frameRateDen > 0)) { + cfg->frameRateNumerator = extConfig.frameRateNum; + cfg->frameRateDenominator = extConfig.frameRateDen; + } + + // ---- Transfer function ---- + // + // This library converts the colour MODEL and implements no transfer + // function, so the transfer function the input arrives in and the one the + // bitstream advertises have to be the same one. A DECLARED mismatch is + // refused here rather than converted: an unapplied transfer function + // produces pixels that are close enough to look plausible and wrong + // everywhere. + // + // 0 on inputTransferCharacteristics declares nothing and asserts nothing; + // the input is taken to be in transferCharacteristics already. + if ((extConfig.inputTransferCharacteristics != 0) && + (extConfig.inputTransferCharacteristics != + extConfig.transferCharacteristics)) { + VkEncErr() << "[EncoderExt] inputTransferCharacteristics " + << (uint32_t)extConfig.inputTransferCharacteristics + << " differs from transferCharacteristics " + << (uint32_t)extConfig.transferCharacteristics + << ". This encoder applies no transfer function, so the " + "submitted frames and the encoded bitstream must " + "declare the same one. Convert before submitting, or " + "declare the transfer function the input carries." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + // ---- Colour description (VUI) ---- + // + // PER FIELD, NOT ALL-OR-NOTHING. An all-or-nothing gate here would read + // + // if (colourPrimaries || transferCharacteristics || + // matrixCoefficients || videoFullRange) + // + // and then copied ALL FOUR values. So a caller that supplied only + // transferCharacteristics = 16 (PQ) -- which is exactly what an HDR + // caller supplies, and often the only thing it knows -- also emitted + // colour_primaries = 0 (Reserved) and matrix_coefficients = 0 + // (Identity/GBR, i.e. "these samples are RGB"). Two fabricated + // declarations, both wrong, both on the HDR path, from one honest one. + // + // 0 IS "NOT SUPPLIED" ON THIS SURFACE, and that is now a stated property + // rather than an accident of the gate. It costs the ability to REQUEST + // code point 0: Identity/GBR primaries/matrix cannot be asked for + // through these fields. That is deliberate and cheap here -- this + // encoder converts to YCbCr and has no Identity path to offer -- and if + // it ever needs to be requestable it takes a new chained struct, per the + // versioning rules at the top of the public header. + // + // WHAT AN UNSUPPLIED FIELD BECOMES: 2, "Unspecified" in ISO/IEC 23091-4, + // which H.264, H.265 and AV1 all share. It is the code point that MEANS + // "not stated", so a partially-supplied declaration stays truthful in + // every field instead of asserting Reserved/Identity by omission. + // + // video_format 5 = "unspecified" (Rec. ITU-T H.264 Table E-2), the + // correct value when the caller communicates colorimetry only. + const bool anyColourIdcSupplied = (extConfig.colourPrimaries != 0) || + (extConfig.transferCharacteristics != 0) || + (extConfig.matrixCoefficients != 0); + const uint8_t kColourIdcUnspecified = 2; + if (anyColourIdcSupplied || (extConfig.videoFullRange == VK_TRUE)) { + // video_signal_type carries video_format and the range flag; it + // travels alone when only the range was declared. + cfg->video_format = 5; + cfg->video_signal_type_present_flag = 1; + cfg->video_full_range_flag = (extConfig.videoFullRange == VK_TRUE) ? 1 : 0; + } + if (anyColourIdcSupplied) { + // Raise the colour description only when there IS one, and fill each + // of its three fields from the caller or from Unspecified -- + // independently. Taking all three verbatim at 0 emits Reserved + // primaries/transfer and matrix 0 (the samples are RGB), which is why each + // falls back to Unspecified on its own. + cfg->colour_primaries = (extConfig.colourPrimaries != 0) + ? extConfig.colourPrimaries + : kColourIdcUnspecified; + cfg->transfer_characteristics = + (extConfig.transferCharacteristics != 0) + ? extConfig.transferCharacteristics + : kColourIdcUnspecified; + cfg->matrix_coefficients = (extConfig.matrixCoefficients != 0) + ? extConfig.matrixCoefficients + : kColourIdcUnspecified; + cfg->color_description_present_flag = 1; + } + + // ---- Quality / profile ---- if (extConfig.qualityLevel > 0) { - argStrings.push_back("--qualityLevel"); - argStrings.push_back(std::to_string(extConfig.qualityLevel)); + cfg->qualityLevel = extConfig.qualityLevel; + } + // ---- Profile ---- + // + // The value IS the codec standard's own profile number, read against + // |codecOp|. A validating consumer expects the exact requested profile in + // the bitstream and rejects on a mismatch, so a number this library cannot + // bind is refused with the reason rather than ignored: an ignored profile + // request produces a bitstream describing something the caller did not ask + // for. + // + // Which profiles admit which input BIT DEPTHS AND WHICH CHROMA + // SUBSAMPLINGS is the standard's rule and is enforced here, both terms of + // it. Honouring an 8-bit-only profile over deeper input would emit an + // out-of-spec bitstream, and quietly substituting a deeper profile would + // be the same ignored request in the other direction. The subsampling term + // is that same argument on the other axis: H.264 High is 4:2:0 only, so + // binding it over 4:4:4 input would declare 4:2:0 while carrying 4:4:4 -- + // and it would do so by OVERRIDING a derivation that reads the input's own + // subsampling and would have chosen High 4:4:4 Predictive (244). + // + // Both terms are read off cfg->input, which the input-geometry derivation + // above has already written from the caller's inputFormat, so this runs + // after it and not before. + // + // VK_VIDEO_ENCODER_PROFILE_DEFAULT binds nothing and leaves the codec + // config's own derivation in effect, which reads the input depth. On AV1 + // it is also seq_profile 0; see the profile constants in the public + // header for what that overlap does and does not cost. + if (extConfig.profile != VK_VIDEO_ENCODER_PROFILE_DEFAULT) { + // TWO QUESTIONS, IN THIS ORDER. First, is the number one this library + // can bind for this codec? Then, and only for the numbers that + // survive, does the standard let that profile carry the declared + // input? Reversing them answers the second question about a profile + // the caller can never have, which reads as advice to change the + // input format when no input format would help. + const char* unbindable = nullptr; + switch ((uint32_t)codecOp) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: { + EncoderConfigH264* h264 = static_cast(cfg); + switch (extConfig.profile) { + case VK_VIDEO_ENCODER_PROFILE_H264_BASELINE: + case VK_VIDEO_ENCODER_PROFILE_H264_MAIN: + case VK_VIDEO_ENCODER_PROFILE_H264_HIGH: + case STD_VIDEO_H264_PROFILE_IDC_HIGH_10: + case STD_VIDEO_H264_PROFILE_IDC_HIGH_422: + case STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE: { + // ONE ARM, because the work is identical: the + // standard's own limits decide what each number can + // carry, and the table beside them states all six. + // BOTH terms are checked -- the depth check alone let + // an explicit High (100) bind over 4:4:4 input and + // override the derivation that would have chosen High + // 4:4:4 Predictive (244). + // + // 110, 122 and 244 are here because the DEFAULT + // derivation already selects them from the input's own + // depth and subsampling and this library emits those + // streams; refusing the same numbers when a caller + // names them was this library's rule and not the + // standard's. What the DEVICE can encode is a separate + // question, asked where the session is created and + // answerable beforehand through the input-format query. + const VkResult admits = + VkEncRefuseIfProfileCannotCarryInput( + codecOp, extConfig.profile, cfg->input.bpp, + cfg->input.chromaSubsampling); + if (admits != VK_SUCCESS) { + return admits; + } + h264->profileIdc = + (StdVideoH264ProfileIdc)extConfig.profile; + } break; + default: + unbindable = "H.264 profile_idc; this library binds " + "66 (Baseline), 77 (Main), 100 (High), " + "110 (High 10), 122 (High 4:2:2) and " + "244 (High 4:4:4 Predictive)"; + break; + } + } break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: { + EncoderConfigH265* h265 = static_cast(cfg); + switch (extConfig.profile) { + case VK_VIDEO_ENCODER_PROFILE_H265_MAIN: + case VK_VIDEO_ENCODER_PROFILE_H265_MAIN10: + case STD_VIDEO_H265_PROFILE_IDC_MAIN_STILL_PICTURE: + case STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS: + case STD_VIDEO_H265_PROFILE_IDC_SCC_EXTENSIONS: { + // Main is 8-bit 4:2:0 (H.265 A.3.2), Main 10 is 8/10 + // bit 4:2:0 (A.3.3), Main Still Picture is Main's + // single-picture form, and Range Extensions and SCC + // Extensions reach 4:2:2, 4:4:4 and sixteen bits + // (A.3.5, A.3.7). The table states all five and the + // guard reads it, so the arms are one. + // + // 4 is here because the DEFAULT derivation reaches it + // from 4:4:4 or 12-bit input already; naming it was + // refused only because this switch did not list it. + const VkResult admits = + VkEncRefuseIfProfileCannotCarryInput( + codecOp, extConfig.profile, cfg->input.bpp, + cfg->input.chromaSubsampling); + if (admits != VK_SUCCESS) { + return admits; + } + h265->profile = + (StdVideoH265ProfileIdc)extConfig.profile; + } break; + default: + unbindable = "H.265 general_profile_idc; this library " + "binds 1 (Main), 2 (Main 10), 3 (Main " + "Still Picture), 4 (Range Extensions) " + "and 9 (SCC Extensions)"; + break; + } + } break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: { + EncoderConfigAV1* av1 = cfg->GetEncoderConfigAV1(); + switch (extConfig.profile) { + // seq_profile 0 is Main AND is + // VK_VIDEO_ENCODER_PROFILE_DEFAULT, so it never reaches + // this switch: the DEFAULT test above took it, and the + // codec config's own derivation is in effect for it. + case STD_VIDEO_AV1_PROFILE_HIGH: + case STD_VIDEO_AV1_PROFILE_PROFESSIONAL: { + // High is 4:4:4 at 8 or 10 bits and Professional + // reaches 4:2:2 and twelve (AV1 6.4.1, A.2), which is + // exactly what InitProfileLevel derives from the input + // when nothing is named. The library emitted those + // seq_profiles already; only naming one was refused. + const VkResult admits = + VkEncRefuseIfProfileCannotCarryInput( + codecOp, extConfig.profile, cfg->input.bpp, + cfg->input.chromaSubsampling); + if (admits != VK_SUCCESS) { + return admits; + } + av1->profile = (StdVideoAV1Profile)extConfig.profile; + } break; + default: + unbindable = "AV1 seq_profile; this library binds 0 " + "(Main), which is also " + "VK_VIDEO_ENCODER_PROFILE_DEFAULT, 1 " + "(High) and 2 (Professional)"; + break; + } + } break; + default: + unbindable = "the selected codec"; + break; + } + if (unbindable != nullptr) { + VkEncErr() << "[EncoderExt] profile " << extConfig.profile + << " is not a value this library can bind as " + << unbindable + << ". Use VK_VIDEO_ENCODER_PROFILE_DEFAULT to let the " + "library derive the profile from the input." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + + // ---- Session / runtime knobs ---- + cfg->numFrames = 1000000; // streaming mode (not UINT32_MAX -- some + // code paths overflow on it) + cfg->repeatInputFrames = true; // external input: also gates skipping the + // host-visible linear staging pool + cfg->verbose = (extConfig.verbose == VK_TRUE) ? 1 : 0; + cfg->validate = (extConfig.validate == VK_TRUE) ? 1 : 0; + cfg->disableFileOutput = (extConfig.disableFileOutput == VK_TRUE) ? 1 : 0; + // Open the output file ONLY when file output is actually wanted. (The + // argv path opened it during parsing even under --disableFileOutput -- + // a latent 0-byte-artifact bug, fixed in passing here.) + if (!cfg->disableFileOutput && + (extConfig.outputPath != nullptr) && (extConfig.outputPath[0] != '\0')) { + const size_t fileSize = cfg->outputFileHandler.SetFileName(extConfig.outputPath); + if ((int64_t)fileSize <= 0) { + return VK_ERROR_INITIALIZATION_FAILED; + } + } + // The completion edge is raised on the threaded assembly path + // (AssemblyWorkerThread -> WriteBitstreamToFile -> PushCapturedBitstream). + // The synchronous AssembleBitstreamData path predates the edge and does + // not publish it. asyncAssembly already defaults to true and no ext + // config field can clear it; this pin turns that implicit dependency + // into a stated invariant so a future default change cannot silently + // kill every completion edge of every ext consumer. + // + // The pin is necessary and was never sufficient: it fixes the config, + // and m_asyncAssemblyEnabled is RUNTIME state that any drain used to + // clear for good. DrainPendingFrames() -> DrainAndRestartThreads() is + // what keeps the runtime state in agreement with this line, and + // ProcessOrderedFrames now refuses the synchronous fallback outright + // while a completion subscriber is registered, so a third way of + // reaching it would be an error and not another silent stall. + cfg->asyncAssembly = 1; +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // Written unconditionally, both ways -- and that is the point, not a + // style preference. EncoderConfig defaults this member to TRUE + // (VkEncoderConfig.h), so writing it only in the true case would leave a + // directly encodable session with the filter CREATED: InitEncoder swaps + // m_inputCommandBufferPool for the filter's compute-family pool and every + // staged frame moves to the COMPUTE queue -- a queue and a pipeline that + // session has no use for. + // + // WHICH conversion to run is a separate question, and a device-dependent + // one: VkVideoEncoder::InitEncoder derives the filter type from the input + // and encode-source formats. The binder has no device and decides only + // WHETHER. + cfg->enablePreprocessComputeFilter = needsPreprocessFilter ? 1 : 0; +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // ---- HDR10 static metadata (pNext-chained) ---- + // + // THE BINDER WALKS THE CHAIN, not InitializeExt, and that is the whole + // reason this is assertable without a device: VkEncBuildAndProbeConfig + // calls this function and nothing else, so a device-free test can drive + // the chain end to end. InitializeExt's own walk ACCEPTS this sType and + // consumes nothing, with a comment saying so -- the two walks must not + // both try to own it. + // + // An unknown sType anywhere in the chain is still rejected, by that other + // walk. This one only looks for the link it owns and ignores the rest, + // because rejecting here would duplicate a gate whose complete list lives + // there. + for (const void* link = extConfig.pNext; link != nullptr;) { + const auto* base = + reinterpret_cast(link); + if (base->sType == VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO) { + // H.264 HAS NO SUCH SEI. There is no standard H.264 + // mastering-display or content-light message, so an H.264 session + // could only accept this and drop it -- the accepted-and-ignored + // class this API refuses everywhere else. Say no, with the reason. + if (codecOp == VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR) { + VkEncErr() << "[EncoderExt] HDR10 static metadata was chained " + "onto an H.264 session. H.264 defines no " + "mastering-display or content-light-level SEI, " + "so this metadata cannot be carried; select " + "H.265 or AV1, or drop the chain." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + EncoderHdrStaticMetadata& md = cfg->hdrMetadata; + md.masteringDisplayPresent = + (base->masteringDisplayPresent == VK_TRUE) ? 1u : 0u; + md.contentLightLevelPresent = + (base->contentLightLevelPresent == VK_TRUE) ? 1u : 0u; + for (int c = 0; c < 3; c++) { + md.displayPrimaryX[c] = base->displayPrimaryX[c]; + md.displayPrimaryY[c] = base->displayPrimaryY[c]; + } + md.whitePointX = base->whitePointX; + md.whitePointY = base->whitePointY; + md.maxDisplayMasteringLuminance = + base->maxDisplayMasteringLuminance; + md.minDisplayMasteringLuminance = + base->minDisplayMasteringLuminance; + md.maxContentLightLevel = base->maxContentLightLevel; + md.maxFrameAverageLightLevel = base->maxFrameAverageLightLevel; + } else if (base->sType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO) { + const auto* inputColour = + reinterpret_cast(link); + if (inputColour->reserved != 0) { + VkEncErr() << "[EncoderExt] VkVideoEncoderInputColourInfo::" + "reserved must be 0." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // THE CALLER DECLARES WHAT ITS BUFFER IS. Recorded here, and the + // presence flag with it -- 0 is UNDECLARED on every axis, so no + // value field can tell "absent" from "present and zero". + cfg->inputColourChainPresent = 1; + cfg->inputColourPrimaries = inputColour->inputColourPrimaries; + cfg->inputTransferCharacteristics = + inputColour->inputTransferCharacteristics; + cfg->inputMatrixCoefficients = inputColour->inputMatrixCoefficients; + cfg->inputRange = (uint8_t)inputColour->inputRange; + } + link = base->pNext; + } + + // ---- What the input declares against what the bitstream declares ---- + // + // REFUSED ON THE AXES THIS LIBRARY CANNOT CONVERT, which is the rule + // inputTransferCharacteristics already applies to the transfer axis + // (above): the filter performs no primaries conversion and applies no + // transfer function, so an input and an output that name different + // primaries, or different matrices, describe a conversion that does not + // happen. Accepting the pair would produce pixels that are close enough to + // look plausible and wrong everywhere. + // + // ONLY WHEN BOTH SIDES ARE DECLARED, AND BOTH SIDES ARE READ FROM THE + // CALLER. The comparison is against extConfig and not against cfg on + // purpose: the binder above rewrites an unsupplied output field to 2 + // (Unspecified) so a partial declaration stays truthful, and comparing + // against that substitution would read the LIBRARY's fill as a caller + // declaration and refuse a config the caller never contradicted. + // + // 0 IS UNDECLARED AND 2 IS "UNSPECIFIED" -- neither asserts anything a + // declaration on the other side can contradict, so both are skipped. + // + // RANGE IS NOT ON THIS LIST because its rule is not this rule. The three + // axes here are compared on every lane and applied on none. The range is + // APPLIED on the Y'CbCr lane -- where the library converts nothing, so + // the input's range is the stream's -- and compared on the RGB lane, + // where the filter produces it. Both halves are below, after the axes + // that have one rule for both lanes. + if (cfg->inputColourChainPresent != 0) { + struct InputColourAxis { + uint8_t input; + uint8_t output; + const char* name; + const char* why; + }; + const InputColourAxis axes[] = { + { cfg->inputColourPrimaries, extConfig.colourPrimaries, + "colour primaries", + "this library performs no primaries conversion, so the input's " + "primaries and the bitstream's are necessarily the same" }, + { cfg->inputMatrixCoefficients, extConfig.matrixCoefficients, + "matrix coefficients", + "the only matrix this library applies is the RGBA->Y'CbCr " + "filter's, and it produces the matrix the bitstream declares" }, + { cfg->inputTransferCharacteristics, + extConfig.transferCharacteristics, + "transfer characteristics", + "this library applies no transfer function: the code values it " + "writes are the code values it read" }, + }; + for (const InputColourAxis& axis : axes) { + if ((axis.input == 0) || (axis.input == 2) || + (axis.output == 0) || (axis.output == 2) || + (axis.input == axis.output)) { + continue; + } + VkEncErr() << "[EncoderExt] the input declares " + << axis.name << " " << (uint32_t)axis.input + << " and the bitstream declares " + << (uint32_t)axis.output + << ". They must agree: " << axis.why + << ". Declare the same value on both, or leave the " + "input side 0, which asserts nothing." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // The chained transfer value and the flat config field are the same + // declaration spelled twice; supplying both is allowed and they must + // agree, because a caller that contradicts itself has not said what it + // means. + if ((extConfig.inputTransferCharacteristics != 0) && + (cfg->inputTransferCharacteristics != 0) && + (extConfig.inputTransferCharacteristics != + cfg->inputTransferCharacteristics)) { + VkEncErr() << "[EncoderExt] inputTransferCharacteristics was " + "declared as " + << (uint32_t)extConfig.inputTransferCharacteristics + << " on the config and as " + << (uint32_t)cfg->inputTransferCharacteristics + << " on the chained VkVideoEncoderInputColourInfo. " + "They are the same declaration and must agree." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } } - // Verbose / Validate — these flags are not recognized by ParseArguments, - // they would fall to the unknown-arg path. Skip for now. - // if (extConfig.verbose) argStrings.push_back("--verbose"); - // if (extConfig.validate) argStrings.push_back("--validate"); + // ---- The declared input RANGE ---- + // + // ONE VARIABLE, TWO CONSUMERS. EncoderConfig::video_full_range_flag is + // both the VUI bit the bitstream carries and, through + // VkVideoEncoder::InitEncoder, the VkSamplerYcbcrRange the preprocess + // filter is built with. They are the same decision and are deliberately + // not two fields; what a declaration does is decide which value that one + // variable takes on the lane where nothing else can. + // + // THE Y'CbCr LANE APPLIES NO RANGE MAPPING, so the declaration is + // APPLIED. The samples that arrive are the samples that are coded: there + // is no scaler between them, and the Y'CbCr copy filter takes its output + // range from its input range, so a copy stays a copy whatever this flag + // says. The input's range therefore IS the stream's range, and a caller + // that states it is stating a fact about the bitstream. Writing it is the + // only way that fact can survive; leaving it to videoFullRange alone + // means a producer of full-range Y'CbCr has to know that a zero it never + // wrote is an assertion, which is the shape of the bug this declaration + // exists to end. + // + // AND IT IS SIGNALLED, not merely recorded. video_full_range_flag is + // nested inside video_signal_type_present_flag, and an absent + // video_signal_type is inferred as studio swing by H.264 E.2.1 and H.265 + // E.3.1 while decoders report it as unknown at their API boundary. A + // caller that declared a range and got silence would be exactly as badly + // served as one that declared nothing, so a declaration raises the + // presence flag, with video_format 5 (Unspecified) beside it -- the same + // value the colour binder above writes when a caller communicates + // colorimetry only. + // + // THE RGB LANE APPLIES ONE, so the declaration is COMPARED. There the + // filter PRODUCES the Y'CbCr range from this same flag, and the caller's + // declaration is about its RGB buffer: two different quantities, and a + // declaration about the first must not silently retarget the second. The + // filter reads its RGB over the full range and performs no input + // expansion, so an RGB input declared LIMITED describes a conversion that + // does not happen -- refused with the reason, exactly as the primaries, + // matrix and transfer axes are. An RGB input declared FULL agrees with + // what the filter reads and says nothing about the output, which stays + // the bitstream request's to state. + // + // ONLY THE RAISED DIRECTION OF videoFullRange CAN BE CONTRADICTED. It is + // a VkBool32 with no undeclared state, so VK_FALSE is indistinguishable + // from silence and reading it as a positive claim of limited range would + // refuse every caller that simply did not fill it in. VK_TRUE against a + // LIMITED input is a real contradiction and is refused. + if ((cfg->inputColourChainPresent != 0) && (cfg->inputRange != 0)) { + const bool inputIsFullRange = + (cfg->inputRange == (uint8_t)VK_VIDEO_ENCODER_RANGE_FULL); + if (cfg->input.colorSpace == VkEncColorSpace::kRGB) { + if (!inputIsFullRange) { + VkEncErr() + << "[EncoderExt] the input declares limited-range RGB, " + "which this library cannot read: the RGBA->Y'CbCr " + "filter samples its input over the full range and " + "performs no input expansion, so the conversion the " + "declaration describes does not happen. Supply " + "full-range RGB and declare it, or leave inputRange " + "UNDECLARED, which asserts nothing. The range the " + "BITSTREAM carries is stated on " + "VkVideoEncoderConfig::videoFullRange and is a " + "separate question." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } else { + if ((extConfig.videoFullRange == VK_TRUE) && !inputIsFullRange) { + VkEncErr() + << "[EncoderExt] the input declares limited range and the " + "bitstream is requested as full range. This library " + "applies no range scaling to Y'CbCr input, so the two " + "cannot differ: the samples that arrive are the " + "samples that are coded. Declare the same range on " + "both, or leave inputRange UNDECLARED." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + cfg->video_format = 5; + cfg->video_signal_type_present_flag = 1; + cfg->video_full_range_flag = inputIsFullRange ? 1 : 0; + } + } - // Large frame count for streaming mode (not UINT32_MAX — some code paths overflow) - argStrings.push_back("--numFrames"); - argStrings.push_back("1000000"); - argStrings.push_back("--repeatInputFrames"); + // ---- The RGBA->YCbCr matrix contract, settled WITHOUT A DEVICE ---- + // + // An RGBA input necessarily builds the RGBA2YCBCR filter -- every encode + // source this library can select is YCbCr, so VkEncDeriveFilterType has + // exactly one answer for a non-YCbCr input -- which means the matrix + // question can be settled here, before an instance or a device exists, + // instead of only at InitEncoder. That matters twice: the caller hears + // "no" before it allocates a frame pool, and the refusal is reachable + // from a device-free test through VkEncBuildAndProbeConfig. + // + // It is the SAME function InitEncoder calls, and it is idempotent, so the + // later call is a no-op and the two layers cannot answer differently. + if ((cfg->input.colorSpace == VkEncColorSpace::kRGB) && + cfg->IsPreprocessComputeFilterEnabled()) { + VkSamplerYcbcrModelConversion filterModel = + VK_SAMPLER_YCBCR_MODEL_CONVERSION_YCBCR_709; + if (!cfg->ResolveRgbToYcbcrMatrix(&filterModel)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + cfg->ApplyPreprocessFilterChromaSiting(); + } - // Output file path (per-encoder isolation; when set, overrides encoder default) - if (extConfig.outputPath && extConfig.outputPath[0] != '\0') { - argStrings.push_back("--output"); - argStrings.push_back(extConfig.outputPath); + // DEBUG: enable the encoder's built-in input-vs-reconstructed PSNR so the + // encode itself can be judged independent of the decode roundtrip. + if (getenv("VKENC_DEBUG_PSNR")) { + cfg->enablePsnrMetrics = 1; } - // No input file (external frame input) - // Don't pass --input since we'll use SetExternalInputFrame + // ---- Derived tail + codec parameter init (same as the parse path) ---- + if (cfg->FinalizeConfig() != 0) { + return VK_ERROR_INITIALIZATION_FAILED; + } + const VkResult paramsResult = outConfig->InitializeParameters(); + if (paramsResult != VK_SUCCESS) { + return paramsResult; + } - // Build argc/argv - std::vector argv; - for (auto& s : argStrings) { - argv.push_back(s.c_str()); + // THE TWO DERIVATIONS OF THE INPUT'S IDENTITY MUST AGREE, AND THIS IS + // WHERE THAT IS SAID. + // + // This function derives (chroma subsampling, bit depth, plane count) FROM + // the caller's VkFormat, above. EncoderInputImageParameters::VerifyInputs() + // -- which InitializeParameters just ran -- reconstructs a VkFormat FROM + // those same three. Two derivations of one quantity in opposite directions, + // and nothing has ever said they had to agree: whatever the second one + // produced is what the session was configured with, and what the compute + // filter was built from. + // + // THE REVERSE ONE CANNOT SIMPLY BE DELETED, which is why this is an + // assertion. The packed-alias arm above leaves input.vkFormat unwritten ON + // PURPOSE so the reconstruction supplies it: CodecGetVkFormat(4:4:4, depth, + // PLANE_LAYOUT_PACKED_1) spells AYUV at eight bits and Y410 at ten, and + // that is the only route by which either format is nameable. Removing the + // reverse derivation removes two capabilities. + // + // IT COVERS BOTH LANES WITH ONE COMPARISON. On the RGBA lane VerifyInputs + // carries the caller's format through rather than reconstructing it -- + // CodecGetVkFormat spells no RGB layout -- so the equality asserted here is + // a carry-through there and a round trip on the Y'CbCr lanes. The + // proposition is the same either way: the config's idea of the input format + // is the caller's. + // + // A DISAGREEMENT IS A REFUSAL AND NOT A REPAIR. Overwriting one side with + // the other would pick a winner between two derivations without knowing + // which was wrong, and the wrong choice is a session configured for a + // picture the caller is not sending -- which is exactly the failure that + // is invisible until the filter is built from it. + if (cfg->input.vkFormat != extConfig.inputFormat) { + VkEncErr() << "[EncoderExt] internal inconsistency: inputFormat " + << (uint32_t)extConfig.inputFormat + << " was bound as (chroma subsampling " + << (uint32_t)cfg->input.chromaSubsampling << ", " + << (uint32_t)cfg->input.bpp << "-bit, " + << cfg->input.numPlanes + << "-plane), and that geometry describes format " + << (uint32_t)cfg->input.vkFormat + << " instead. The session would be configured for a picture " + "the caller is not sending, so it is refused rather than " + "reconciled." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; } + return VK_SUCCESS; +} - // Debug: dump argv for troubleshooting - std::cout << "[VulkanVideoEncoderExt] CreateCodecConfig argv (" << argv.size() << "):\n"; - for (size_t i = 0; i < argv.size(); i++) { - std::cout << " [" << i << "] " << argv[i] << "\n"; +// Byte-exact projection of the HDR10 payload builders -- see the declaration +// in the internal header for why it exists. A thin translation from the +// public struct to the library's POD and back out as bytes: no config, no +// device, no session. +uint32_t VkEncBuildHdrMetadataPayload(const VkVideoEncoderHdrMetadataInfo* info, + VkVideoCodecOperationFlagBitsKHR codecOp, + uint8_t* out, uint32_t capacity) +{ + if ((info == nullptr) || (out == nullptr)) { + return 0; } + EncoderHdrStaticMetadata md; + md.masteringDisplayPresent = + (info->masteringDisplayPresent == VK_TRUE) ? 1u : 0u; + md.contentLightLevelPresent = + (info->contentLightLevelPresent == VK_TRUE) ? 1u : 0u; + for (int c = 0; c < 3; c++) { + md.displayPrimaryX[c] = info->displayPrimaryX[c]; + md.displayPrimaryY[c] = info->displayPrimaryY[c]; + } + md.whitePointX = info->whitePointX; + md.whitePointY = info->whitePointY; + md.maxDisplayMasteringLuminance = info->maxDisplayMasteringLuminance; + md.minDisplayMasteringLuminance = info->minDisplayMasteringLuminance; + md.maxContentLightLevel = info->maxContentLightLevel; + md.maxFrameAverageLightLevel = info->maxFrameAverageLightLevel; - VkResult ccResult = EncoderConfig::CreateCodecConfig( - static_cast(argv.size()), argv.data(), outConfig); - // DEBUG: enable the encoder's built-in input-vs-reconstructed PSNR so we can - // tell whether the ext encode itself is lossy (input vs recon) independent of - // the decode/reference roundtrip. - if (ccResult == VK_SUCCESS && outConfig && getenv("VKENC_DEBUG_PSNR")) { - outConfig->enablePsnrMetrics = 1; + bool truncated = false; + size_t written = 0; + switch ((uint32_t)codecOp) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + written = VkEncBuildH265HdrSeiNal(md, out, capacity, &truncated); + break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + written = VkEncBuildAv1HdrMetadataObus(md, out, capacity, + &truncated); + break; + default: + // H.264 carries neither payload; the binder refuses the chain. + return 0; } - return ccResult; + return truncated ? 0u : (uint32_t)written; } //============================================================================= @@ -327,6 +2647,22 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( VkVideoCodecOperationFlagBitsKHR codecOp, const VkVideoEncoderConfig& config) { + // Imported-device contract: providing externalDevice without + // externalPhysicalDevice is illegal -- the Vulkan API offers no + // way to recover the physical device from a logical device + // handle, so the library cannot pick the right physical device + // to query format / capability properties on. Both must be + // provided together, or both left VK_NULL_HANDLE. + if (config.externalDevice != VK_NULL_HANDLE && + config.externalPhysicalDevice == VK_NULL_HANDLE) { + VkEncErr() << "[EncoderExt] externalDevice supplied " + << "without externalPhysicalDevice; this " + << "combination is illegal. Provide both, or set " + << "externalDevice = VK_NULL_HANDLE to let the " + << "library pick a device." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + static const char* const requiredInstanceLayers[] = { "VK_LAYER_KHRONOS_validation", nullptr @@ -356,9 +2692,53 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( nullptr }; + // DEPENDENCY EXTENSIONS -- DO NOT DELETE AS UNUSED. + // + // Three of the names below are here purely to satisfy + // VUID-vkCreateDevice-ppEnabledExtensionNames-01387: "All required device + // extensions for each extension in the VkDeviceCreateInfo:: + // ppEnabledExtensionNames list must also be present in that list." Nothing + // in this library calls their entry points directly, so a reader sweeping + // for dead names will be tempted to remove them. Each is marked below with + // the entry that requires it; removing one re-opens 01387 for its parent. + // + // Why this was invisible for so long: all three were core-promoted, at + // Vulkan 1.2 / 1.2 / 1.3 respectively, and validation treats a dependency + // as satisfied by core once the INSTANCE apiVersion reaches the promoting + // version. Every in-tree sample creates its instance at + // VK_HEADER_VERSION_COMPLETE, so the VUID never fires there. Chromium does + // not: gpu/vulkan/vulkan_function_pointers.h pins + // kVulkanRequiredApiVersion = VK_API_VERSION_1_1, and the ADOPT path + // borrows that embedder instance -- below every promotion -- so on Chromium + // all three fire on the encode session's vkCreateDevice. + // + // OPTIONAL is the correct list. AddOptDeviceExtensions() only queues the + // name; HasAllDeviceExtensions() (common/libs/VkCodecUtils/ + // VulkanDeviceContext.cpp) promotes it to required ONLY if the physical + // device enumerates it, and otherwise emits a warning. So a device that + // lacks one of these degrades with a warning rather than failing device + // creation. Caveat for the next editor: that same filter means a device + // exposing a parent but not its dependency would drop the dependency and + // re-open the VUID -- the pairing below is a convention, not an invariant + // the list machinery enforces. static const char* const optionalDeviceExtension[] = { +#if defined(__linux) || defined(__linux__) || defined(linux) + // Used at runtime by the dma-buf import path (VkImageResource DRM + // format-modifier queries, FOREIGN queue-family transfers, dma-buf + // memory import) but previously never requested -- device creation + // relied on drivers tolerating un-enabled extensions. + VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME, + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME, + // 01387 dependency of VK_EXT_image_drm_format_modifier, above. + // Core in 1.2; needed explicitly on a 1.1 instance (Chromium). + VK_KHR_IMAGE_FORMAT_LIST_EXTENSION_NAME, + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME, +#endif VK_EXT_YCBCR_2PLANE_444_FORMATS_EXTENSION_NAME, VK_EXT_DESCRIPTOR_BUFFER_EXTENSION_NAME, + // 01387 dependency of VK_EXT_descriptor_buffer, above. + // Core in 1.2; needed explicitly on a 1.1 instance (Chromium). + VK_EXT_DESCRIPTOR_INDEXING_EXTENSION_NAME, VK_KHR_BUFFER_DEVICE_ADDRESS_EXTENSION_NAME, VK_KHR_PUSH_DESCRIPTOR_EXTENSION_NAME, VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, @@ -366,6 +2746,9 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( VK_KHR_VIDEO_ENCODE_H265_EXTENSION_NAME, VK_KHR_VIDEO_ENCODE_AV1_EXTENSION_NAME, VK_KHR_VIDEO_ENCODE_QUANTIZATION_MAP_EXTENSION_NAME, + // 01387 dependency of VK_KHR_video_encode_quantization_map, above. + // Core in 1.3; needed explicitly on a 1.1 instance (Chromium). + VK_KHR_FORMAT_FEATURE_FLAGS_2_EXTENSION_NAME, VK_KHR_VIDEO_ENCODE_INTRA_REFRESH_EXTENSION_NAME, nullptr }; @@ -378,15 +2761,98 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( m_vkDevCtx.AddReqDeviceExtensions(requiredDeviceExtension); m_vkDevCtx.AddOptDeviceExtensions(optionalDeviceExtension); - VkResult result = m_vkDevCtx.InitVulkanDevice("VulkanVideoEncoderExt", - config.externalInstance, - config.verbose); - if (result != VK_SUCCESS) { - std::cerr << "[EncoderExt] InitVulkanDevice failed: " << result << std::endl; - return result; - } + // Design 3.1 rule 1, and set BEFORE the dlopen InitVulkanDevice performs, + // because a bring-up that fails AFTER LoadVk still owns the loader handle + // and still runs the destructor. + // + // InitVulkanDevice() calls LoadVk() unconditionally at its top -- above + // the imported-instance branch -- so this session dlopen()s libvulkan in + // every mode exactly as it does when it creates its own instance. Without + // the retain, ~VulkanDeviceContext dlclose()s it once per session. + // Measured, not argued: a probe reading /proc/self/maps around one + // create/destroy reports libvulkan.so.1 mapped 0 -> 1 -> 0. Whenever this + // session holds the last reference the shared object is UNMAPPED, and any + // embedder function-pointer table bound to it -- Chromium's + // gpu::VulkanFunctionPointers are bound to exactly this library -- is left + // holding pointers into an unmapped object. + // + // That it has not crashed yet is ordering, not contract: something else + // usually holds a reference first (the capability enumerator building its + // process-floor OWN context, or the embedder initialising its own loader). + // VulkanVideoEncoderContext::Build has retained since it existed; this was + // the one VulkanDeviceContext in an embedded process that did not, which + // is the case VulkanDeviceContext.h's RetainLoaderHandle rule names. + m_vkDevCtx.RetainLoaderHandle(); - result = m_vkDevCtx.InitDebugReport(config.validate, config.verbose && config.validate); + // Where the borrowed instance and physical device come from. A session + // created on a context takes both from the context; one created by the + // plain factory takes them from the config, as it always did. + // + // The two are the SAME borrowing expressed twice, so a session that has a + // context and a config that also names handles is refused rather than + // resolved. Picking a winner is what would hurt: a stale config field + // outranking the context would send this session to a different physical + // device than the one whose capability snapshot the caller read, and + // nothing downstream would report the substitution. + VkInstance borrowedInstance = config.externalInstance; + VkPhysicalDevice borrowedPhysDevice = config.externalPhysicalDevice; + if (m_context) { + if ((config.externalInstance != VK_NULL_HANDLE) || + (config.externalPhysicalDevice != VK_NULL_HANDLE)) { + VkEncErr() << "[EncoderExt] config.externalInstance / " + "externalPhysicalDevice set on a session created " + "with CreateVulkanVideoEncoderExtOnContext. The " + "context already supplies both; remove them from " + "the config." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if (config.externalDevice != VK_NULL_HANDLE) { + // A context creates no VkDevice and a session built on one + // creates its own; a caller-supplied device has nowhere to land. + VkEncErr() << "[EncoderExt] config.externalDevice set on a " + "session created with " + "CreateVulkanVideoEncoderExtOnContext. A context " + "creates no VkDevice, so a caller-supplied one is " + "not accepted on this path." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // deviceId and gpuUUID are device SELECTORS, and the context already + // selected. Left un-refused they still reach InitPhysicalDevice as + // filters over the single-entry candidate list: a mismatch fails with + // a bare "InitPhysicalDevice failed: " the caller cannot tell + // from a hardware problem, and a match is honoured redundantly. Refuse + // both, for the reason CreateVulkanVideoEncoderContext gives when it + // refuses a non-zero gpuUUID in ADOPT -- the device is already chosen, + // so honouring one of the two would be a guess. + static const uint8_t kZeroUUID[VK_UUID_SIZE] = {}; + if (config.deviceId != -1) { + VkEncErr() << "[EncoderExt] config.deviceId (" << config.deviceId + << ") set on a session created with " + "CreateVulkanVideoEncoderExtOnContext. The context's " + "deviceIndex already selects the device; leave " + "deviceId at -1." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if (memcmp(config.gpuUUID, kZeroUUID, VK_UUID_SIZE) != 0) { + VkEncErr() << "[EncoderExt] config.gpuUUID set on a session " + "created with CreateVulkanVideoEncoderExtOnContext. " + "The context's deviceIndex already selects the " + "device." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + borrowedInstance = m_contextInstance; + borrowedPhysDevice = m_contextPhysDevice; + } + + VkResult result = m_vkDevCtx.InitVulkanDevice("VulkanVideoEncoderExt", + borrowedInstance, + config.verbose); + if (result != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] InitVulkanDevice failed: " << result << std::endl; + return result; + } + + result = m_vkDevCtx.InitDebugReport(config.validate, config.verbose && config.validate); if (result != VK_SUCCESS) { return result; } @@ -414,9 +2880,10 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( nullptr, 0, VK_VIDEO_CODEC_OPERATION_NONE_KHR, requestVideoEncodeQueueMask, - codecOp); + codecOp, + borrowedPhysDevice); // borrowed physical device: context or config if (result != VK_SUCCESS) { - std::cerr << "[EncoderExt] InitPhysicalDevice failed: " << result << std::endl; + VkEncErr() << "[EncoderExt] InitPhysicalDevice failed: " << result << std::endl; return result; } @@ -447,7 +2914,7 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( if (strcmp(e.extensionName, requiredExt) == 0) { found = true; break; } } if (!found) { - std::cerr << "[EncoderExt] ERROR: this device does not support " + VkEncErr() << "[EncoderExt] ERROR: this device does not support " << requiredExt << "; cannot encode the requested codec" << std::endl; return VK_ERROR_EXTENSION_NOT_PRESENT; @@ -455,6 +2922,31 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( } } + // For an imported VkDevice, bind the queue families the CALLER + // created queues for instead of the probe's picks. The probe above only + // inspects VkPhysicalDevice properties; it cannot know which families the + // caller's vkCreateDevice actually requested queues on, and + // vkGetDeviceQueue on a family the device was not created with is + // undefined behavior. UINT32_MAX fields keep the probed family + // (default behavior / library-owned-device path). Must run BEFORE the + // needTransferQueue derivation below -- the override refreshes + // GetVideoEncodeQueueFlag() to the real flags of the caller's encode + // family. NOTE: the caller provides no transfer-family index; if its + // encode family lacks TRANSFER, the probed transfer family is still used + // (known limitation -- NVIDIA encode families advertise TRANSFER, + // so an imported-device path never takes that branch). + if (config.externalDevice != VK_NULL_HANDLE) { + result = m_vkDevCtx.OverrideImportedQueueFamilies( + config.externalEncodeQueueFamilyIndex, + codecOp, + config.externalComputeQueueFamilyIndex); + if (result != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] OverrideImportedQueueFamilies " + << "failed: " << result << std::endl; + return result; + } + } + bool needTransferQueue = ((m_vkDevCtx.GetVideoEncodeQueueFlag() & VK_QUEUE_TRANSFER_BIT) == 0); // Always request compute queue — VkVideoEncoder internally creates // VulkanFilter for input format conversion, which asserts m_queue != NULL. @@ -467,9 +2959,10 @@ VkResult VulkanVideoEncoderExtImpl::InitVulkanDevice( needTransferQueue, false, // createGraphicsQueue false, // createDisplayQueue - needComputeQueue); + needComputeQueue, + config.externalDevice); // caller-supplied logical device if (result != VK_SUCCESS) { - std::cerr << "[EncoderExt] CreateVulkanDevice failed: " << result << std::endl; + VkEncErr() << "[EncoderExt] CreateVulkanDevice failed: " << result << std::endl; return result; } @@ -501,7 +2994,17 @@ VkResult VulkanVideoEncoderExtImpl::Initialize( result = VkVideoEncoder::CreateVideoEncoder(&m_vkDevCtx, m_encoderConfig, m_encoder); if (result != VK_SUCCESS) return result; + SnapshotCaps(); m_initialized = true; + // Currency 3 is created up front, on the session-serial thread, so the + // submit path can read the handle without a lock. Creation failure + // degrades the currency (GetCompletionSemaphore stays VK_NULL_HANDLE); + // it does not fail the session: no consumer requires the semaphore. + if (m_encoder->CreateCompletionTimelineSemaphore() != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] completion timeline semaphore creation " + "failed; the GPU-ordering currency is unavailable " + "this session" << std::endl; + } return VK_SUCCESS; } @@ -520,22 +3023,541 @@ VkResult VulkanVideoEncoderExtImpl::EncodeNextFrame(int64_t& frameNumEncoded) return VK_SUCCESS; } +//============================================================================= +// Layout pins for the public structs. (VkVideoEncoderConfig has its own pin +// beside the config binder, where the number carries a second meaning: +// binder completeness.) +// +// WHAT A PIN IS. Each number is the size this build computes for one public +// struct, written down. An in-place field append changes the layout under an +// UNCHANGED sType, so a consumer built against the older header still passes +// the structure-type gate and is then read with a shifted layout, silently -- +// no diagnostic at either end. The pin turns that into a build break in this +// file. +// +// THE RULE THE PINS ENFORCE. The only legal way to extend a struct in this +// API is a new pNext-chained struct with a new sType. Appending a field to an +// existing struct is an ABI break and requires a new sType. sType values are +// numbered from a private base and are never reused or renumbered, and the +// library refuses an sType it does not know rather than guessing -- which is +// what turns version skew into an error at the boundary instead of silent +// misbehaviour. +// +// WHAT TRIPPING ONE MEANS. A pin trips only once a public struct has already +// changed shape, so the assert is not a step in a procedure to be worked +// through: it is where the change stops and its author has to decide what +// this API does about an ABI break. What defeats the pin is answering it +// without that decision -- relaxing the assert, or leaving a number stale. A +// size nobody had to think about records nothing. +//============================================================================= +// PINNED ON 64-BIT ONLY (LP64/LLP64). Every pin is a byte size or a byte +// offset, and both move with pointer width: a 32-bit build puts pNext at 4 +// rather than 8 and shortens every struct carrying a pointer or a handle. +// Re-deriving a second set of numbers for -m32 would double the maintenance +// and buy nothing -- a field appended in place changes the 64-bit layout too, +// so the gate catches it either way, and the embedding target is 64-bit. +// +// THE CONDITION IS ON EACH MACRO, never on a region of this file. A region +// has to be re-closed every time a pin is added past its end, and a pin +// appended after that close is silently unguarded -- which is exactly the +// defect this arrangement removes: a macro that expands to nothing cannot +// drift, wherever a later pin lands. +#if defined(__LP64__) || defined(_LP64) || defined(_WIN64) +#define VK_ENC_PIN_LAYOUT(T, N) \ + static_assert(sizeof(T) == (N), \ + #T " changed size -- extend it via a pNext-chained " \ + "struct with a new sType, never an in-place append") +#else +#define VK_ENC_PIN_LAYOUT(T, N) +#endif +VK_ENC_PIN_LAYOUT(VkVideoEncoderFrameDeadlineInfo, 24); +VK_ENC_PIN_LAYOUT(VkVideoEncoderValidationInfo, 24); +VK_ENC_PIN_LAYOUT(VkVideoEncoderHdrMetadataInfo, 56); +VK_ENC_PIN_LAYOUT(VkVideoEncoderInputColourInfo, 24); +VK_ENC_PIN_LAYOUT(VkVideoEncoderFrameFenceDescriptor, 32); +VK_ENC_PIN_LAYOUT(VkVideoEncoderPlaneLayout, 40); +VK_ENC_PIN_LAYOUT(VkVideoEncoderExternalImageDescriptor, 336); +VK_ENC_PIN_LAYOUT(VkVideoEncoderFrameSubmitInfo, 104); +VK_ENC_PIN_LAYOUT(VkVideoEncoderCompletionInfo, 64); +VK_ENC_PIN_LAYOUT(VkVideoEncoderDiagnosticInfo, 216); +VK_ENC_PIN_LAYOUT(VkVideoEncoderFilterInfo, 40); +VK_ENC_PIN_LAYOUT(VkVideoEncoderInputResidencyInfo, 32); +VK_ENC_PIN_LAYOUT(VkVideoEncoderStagedSubmitInfo, 24); +VK_ENC_PIN_LAYOUT(VkVideoEncodeInputFrame, 128); +VK_ENC_PIN_LAYOUT(VkVideoEncodeResult, 72); +VK_ENC_PIN_LAYOUT(VkVideoEncoderRuntimeInfo, 120); +VK_ENC_PIN_LAYOUT(VkVideoEncoderSemaphoreDescriptor, 32); +VK_ENC_PIN_LAYOUT(VkVideoEncoderFrameSyncDescriptor, 64); +VK_ENC_PIN_LAYOUT(VkVideoEncoderImageSupport, 24); +VK_ENC_PIN_LAYOUT(VkVideoEncoderImageSupportDetails, 280); +VK_ENC_PIN_LAYOUT(VkVideoEncoderStatus, 24); +VK_ENC_PIN_LAYOUT(VkVideoEncoderImportGuardInfo, 40); +VK_ENC_PIN_LAYOUT(VkVideoEncoderImportContentInfo, 56); +VK_ENC_PIN_LAYOUT(VkVideoEncoderCapabilities, 88); +VK_ENC_PIN_LAYOUT(VkVideoEncoderInputFormatProperties, 12); +VK_ENC_PIN_LAYOUT(VkVideoEncoderContextCreateInfo, 64); +VK_ENC_PIN_LAYOUT(VkVideoEncoderDeviceIdentity, 312); +#undef VK_ENC_PIN_LAYOUT + +// The sizeof pins cannot catch a reorder that keeps the size constant -- +// swapping pNext with a payload field, say -- yet every pNext walk in this +// file reads a link's {sType, pNext} through a SIBLING struct type before +// re-casting ("Every public struct opens with {sType, pNext}"). That walk +// is sound only while the prefix sits at the same offsets in every +// chainable struct, so pin the prefix too: a reorder becomes a build break +// here instead of a payload value dereferenced as a pointer. The 8 is as +// 64-bit-specific as the sizeof numbers above. VkVideoEncoderPlaneLayout and +// VkVideoEncoderInputFormatProperties are deliberately absent: both are +// pointer-free PODs with no {sType, pNext} prefix, and neither rides a chain. +// Those two are the only absences -- every other struct the sizeof pins above +// cover is pinned here, which is what makes this list readable as a set. +#if defined(__LP64__) || defined(_LP64) || defined(_WIN64) +#define VK_ENC_PIN_CHAIN_PREFIX(T) \ + static_assert((offsetof(T, sType) == 0) && (offsetof(T, pNext) == 8), \ + #T " moved its {sType, pNext} prefix -- every pNext walk" \ + " reads links through a sibling struct type and" \ + " depends on this layout") +#else +#define VK_ENC_PIN_CHAIN_PREFIX(T) +#endif +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderConfig); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderFrameDeadlineInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderValidationInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderHdrMetadataInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderInputColourInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderFrameFenceDescriptor); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderExternalImageDescriptor); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderFrameSubmitInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderCompletionInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderDiagnosticInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderFilterInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderInputResidencyInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderStagedSubmitInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncodeInputFrame); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncodeResult); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderRuntimeInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderSemaphoreDescriptor); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderFrameSyncDescriptor); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderImageSupport); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderImageSupportDetails); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderStatus); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderImportGuardInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderImportContentInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderCapabilities); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderContextCreateInfo); +VK_ENC_PIN_CHAIN_PREFIX(VkVideoEncoderDeviceIdentity); +#undef VK_ENC_PIN_CHAIN_PREFIX + +// Neither pin above can see the INTERIOR of a struct. A member removed from +// a run of members that ends where a coarser alignment begins -- the next +// member's, or the struct's own round-up at the end -- need not shrink the +// struct at all: the compiler reclaims exactly the bytes that went as +// padding at the close of the run, so sizeof is unchanged, the prefix is +// unchanged, and every member between the removal and that boundary shifts +// down under an UNCHANGED sType. It holds at every width. A byte-wide +// member in a run closing at a 4-aligned one, a 4-byte member in a run +// closing at an 8-aligned one, and a trailing bool in a run that reaches +// only the struct's own tail padding are the same event. A consumer that +// never named the removed field gets no compile error, and there is no +// diagnostic at either end -- the same silence the sizeof pins exist to +// break, one level further in. +// +// Worse than a lost field, where the survivors are the same width as the +// member that went: the consumer does not read a field that is missing, it +// reads a DIFFERENT field's value under the name it asked for, and reports it +// as the answer. A run of same-width members closing at a padding boundary +// is where that happens. +// +// So, in each public struct that HAS such a run, pin the member that CLOSES +// it -- the one immediately before the next member of coarser alignment, and +// the last member of the struct. That member is the one every removal within +// the run displaces, whichever member the removal takes and whether or not +// the run carries a hole today; removing the pinned member itself instead +// deletes a name this file asserts on. One number per run therefore covers +// every REMOVAL inside the run: it either moves that number or fails to +// compile here. +// +// IT DOES NOT COVER EVERY REORDER, which is worth stating because a reorder +// produces the same misread this whole block exists to stop. Transposing two +// members of EQUAL WIDTH changes no size, no chain prefix and no offset but +// those two, so a closing-member pin sees it only when the pinned member is +// one of the pair. In a run of two members it always is. In a longer run of +// equal-width members it need not be, and the two public structs that are +// one such run end to end -- VkVideoEncoderPlaneLayout and +// VkVideoEncoderInputFormatProperties -- are pinned member by member at the +// end of this block for that reason. +// +// INSERTION IS THE SAME HAZARD AND IS WHY THE COVERAGE BELOW IS EXHAUSTIVE. +// A run that ends in padding swallows a field small enough to fit there: the +// size pin above does not move, the members past the padding do not move, and +// the members of the run between the insertion and that padding all slide. +// VkVideoEncoderConfig carries such a run in the shipped layout -- its +// byte-wide colour description ends one byte short of the 4-aligned member +// behind it -- and a consumer built against the older header would go on +// writing a field some bytes from where the library reads it, with nothing +// here objecting. Every run that can absorb a field therefore carries the pin +// that closes it, derived from measured offsets rather than from reading the +// declarations. +// +// THESE PINS ARE THE ONLY LAYOUT ENFORCEMENT THIS API HAS. +// VK_VIDEO_ENCODER_EXT_API_VERSION is 1 and stays 1 -- it is not bumped for +// layout or vtable changes -- so there is no version a consumer can compare +// to discover that a struct moved underneath it. Nothing else in the build +// notices. A pin removed as redundant is enforcement deleted, not tidied. +// +// THE ONE CASE THESE DO NOT COVER, stated so it is not mistaken for one they +// do: a field dropped INTO a padding hole, displacing nothing. The four bytes +// between sType and the 8-aligned pNext are such a hole in every chainable +// struct here, and so are the bytes after the closing member of a trailing +// run. A field placed there moves no member and changes no size, so nothing +// fires -- not these pins, not the size pin, not the {sType, pNext} prefix +// pin. That was measured rather than assumed. Nothing existing is MISREAD in +// that case: every offset a consumer already knows is still correct. What it +// costs is that the library reads bytes an older consumer never wrote, which +// is the field table's business rather than this one's. +// +// Deliberately not every member, and deliberately not every struct. A struct +// whose every removal changes its size is already answered by the size pin +// above and takes nothing here -- a pin that cannot fail independently of +// the two above records nothing. What this one is for is that a change +// landing in padding trips something, not that every field is frozen. +// +// VkVideoEncoderExternalImageDescriptor carries the most of them because it +// is the one structure in this API that IS the IPC payload: a producer in +// another process fills it by field copy, so its member offsets are the wire +// format rather than a property of this build. Its two non-data members -- +// pNext, refused non-NULL, and existingImage, read only on the same-process +// VK_IMAGE arm -- cross no such boundary. It is also long +// enough that a change to one end of it is read nowhere near the other. +#if defined(__LP64__) || defined(_LP64) || defined(_WIN64) +#define VK_ENC_PIN_MEMBER(T, m, N) \ + static_assert(offsetof(T, m) == (N), \ + #T "::" #m " moved -- a member removed or reordered into" \ + " padding leaves sizeof unchanged and shifts the rest" \ + " of its run under an unchanged sType") +#else +#define VK_ENC_PIN_MEMBER(T, m, N) +#endif +VK_ENC_PIN_MEMBER(VkVideoEncoderExternalImageDescriptor, + hasDrmFormatModifier, 64); +VK_ENC_PIN_MEMBER(VkVideoEncoderExternalImageDescriptor, + planeCount, 80); +VK_ENC_PIN_MEMBER(VkVideoEncoderExternalImageDescriptor, + residency, 312); +VK_ENC_PIN_MEMBER(VkVideoEncoderExternalImageDescriptor, + colorModel, 332); +// VkVideoEncoderCapabilities has two such runs: the level, DPB and quality +// scalars closing at the 8-aligned maxBitrate, and the four trailing +// availability flags, which live entirely in the struct's tail padding. +VK_ENC_PIN_MEMBER(VkVideoEncoderCapabilities, + maxQualityLevels, 36); +VK_ENC_PIN_MEMBER(VkVideoEncoderCapabilities, + supportsResizeWithoutIdr, 83); +// VkVideoEncoderImageSupport is the smallest case of the same shape: its two +// payload members are one run, and both of them sit in the round-up to the +// alignment the pNext pointer imposes. +VK_ENC_PIN_MEMBER(VkVideoEncoderImageSupport, + status, 20); +// VkVideoEncoderStagedSubmitInfo has that shape too: the submitted queue +// flag and the family index it resolves to are one run, sitting entirely in +// the round-up the pNext pointer imposes, so dropping either leaves the size +// at 24 and slides the survivor onto the other one's offset. +VK_ENC_PIN_MEMBER(VkVideoEncoderStagedSubmitInfo, + queueFamilyIndex, 20); +// The narrow-integer runs of the HDR10 metadata: the two white-point +// chromaticity coordinates, and the two content light levels that close the +// struct. +VK_ENC_PIN_MEMBER(VkVideoEncoderHdrMetadataInfo, + whitePointY, 34); +VK_ENC_PIN_MEMBER(VkVideoEncoderHdrMetadataInfo, + maxFrameAverageLightLevel, 50); +// VkVideoEncoderConfig has three: the byte-wide colour description, the +// diagnostic switches closing at the 8-aligned external-handle block, and +// the two external queue family indices that close the struct. Its size pin +// lives beside the config binder, where it also serves as the +// binder-completeness check, and cannot see any of these. +VK_ENC_PIN_MEMBER(VkVideoEncoderConfig, + matrixCoefficients, 118); +VK_ENC_PIN_MEMBER(VkVideoEncoderConfig, + silenceStdio, 172); +VK_ENC_PIN_MEMBER(VkVideoEncoderConfig, + externalComputeQueueFamilyIndex, 204); +// The per-frame scalars between the presentation timestamp and the wait +// semaphore array. +VK_ENC_PIN_MEMBER(VkVideoEncodeInputFrame, + waitSemaphoreCount, 84); +// The two PCI identifiers that close the device identity, after the fixed +// UUID and name arrays. +VK_ENC_PIN_MEMBER(VkVideoEncoderDeviceIdentity, + deviceID, 308); +// VkVideoEncoderValidationInfo: the flag word closes the struct on its own, +// in the round-up behind the single pNext pointer. +VK_ENC_PIN_MEMBER(VkVideoEncoderValidationInfo, + flags, 16); +// VkVideoEncoderFrameFenceDescriptor: the acquire fd closes the 4-byte run +// before the 8-aligned release pointer. +VK_ENC_PIN_MEMBER(VkVideoEncoderFrameFenceDescriptor, + acquireFenceFd, 16); +// VkVideoEncoderFrameSubmitInfo: two counts, each closing a run before an +// 8-aligned array pointer. +VK_ENC_PIN_MEMBER(VkVideoEncoderFrameSubmitInfo, + waitSemaphoreCount, 56); +VK_ENC_PIN_MEMBER(VkVideoEncoderFrameSubmitInfo, + signalSemaphoreCount, 80); +// VkVideoEncoderCompletionInfo: the acquired-frame count closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncoderCompletionInfo, + framesAcquired, 56); +// VkVideoEncodeInputFrame: the declared layout closes the run before the +// 8-aligned handle block; the +// signal count closes the run before its array pointer +VK_ENC_PIN_MEMBER(VkVideoEncodeInputFrame, + currentLayout, 40); +VK_ENC_PIN_MEMBER(VkVideoEncodeInputFrame, + signalSemaphoreCount, 104); +// VkVideoEncodeResult: the status closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncodeResult, + status, 64); +// VkVideoEncoderRuntimeInfo: the trailing simulcast flag closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncoderRuntimeInfo, + applyAlignmentToAllSimulcastLayers, 112); +// VkVideoEncoderSemaphoreDescriptor: the ownership mode closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncoderSemaphoreDescriptor, + ownership, 24); +// VkVideoEncoderFrameSyncDescriptor: two counts, each closing a run before +// an 8-aligned array pointer. +VK_ENC_PIN_MEMBER(VkVideoEncoderFrameSyncDescriptor, + waitCount, 16); +VK_ENC_PIN_MEMBER(VkVideoEncoderFrameSyncDescriptor, + signalCount, 40); +// VkVideoEncoderImageSupportDetails: the modifier count closes the run +// before the 8-aligned modifier array. +VK_ENC_PIN_MEMBER(VkVideoEncoderImageSupportDetails, + directModifierCount, 16); +// VkVideoEncoderStatus: the consumed-handle flag closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncoderStatus, + handlesConsumed, 16); +// VkVideoEncoderImportGuardInfo: the errno closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncoderImportGuardInfo, + failureErrno, 32); +// VkVideoEncoderContextCreateInfo: the mode closes the run before the +// 8-aligned Vulkan handles; silenceStdio +// closes the struct +VK_ENC_PIN_MEMBER(VkVideoEncoderContextCreateInfo, + mode, 16); +VK_ENC_PIN_MEMBER(VkVideoEncoderContextCreateInfo, + silenceStdio, 56); +// VkVideoEncoderConfig: the GPU UUID closes the run that carries the colour +// description, the +// input transfer function and the device selector -- the run a 4-byte +// insertion slides whole +VK_ENC_PIN_MEMBER(VkVideoEncoderConfig, + gpuUUID, 132); +#undef VK_ENC_PIN_MEMBER + +// THE TWO STRUCTS A CLOSING-MEMBER PIN CANNOT SPEAK FOR, pinned member by +// member instead. VkVideoEncoderPlaneLayout is five uint64_t and +// VkVideoEncoderInputFormatProperties opens with two VkFormat, so each is one +// run of equal-width members end to end: transposing a pair inside either +// leaves the size, every other member's offset and every other assertion in +// this file untouched, and the consumer reads one member's value under the +// other's name. Neither carries {sType, pNext} and neither rides a chain, so +// the prefix pins have nothing to say about them either. +// +// The FIRST member of each is deliberately left unpinned, so no assertion +// here is offsetof(T, m) == 0. Nothing is lost: a transposition moves two +// members, at most one of which can be the first, so the other one's offset +// has to move and is pinned below. +#if defined(__LP64__) || defined(_LP64) || defined(_WIN64) +#define VK_ENC_PIN_ORDER(T, m, N) \ + static_assert(offsetof(T, m) == (N), \ + #T "::" #m " moved -- every member of this structure is" \ + " the same width, so a transposition changes no size" \ + " and no other offset, and is read as the wrong" \ + " field's value") +#else +#define VK_ENC_PIN_ORDER(T, m, N) +#endif +VK_ENC_PIN_ORDER(VkVideoEncoderPlaneLayout, size, 8); +VK_ENC_PIN_ORDER(VkVideoEncoderPlaneLayout, rowPitch, 16); +VK_ENC_PIN_ORDER(VkVideoEncoderPlaneLayout, arrayPitch, 24); +VK_ENC_PIN_ORDER(VkVideoEncoderPlaneLayout, depthPitch, 32); +VK_ENC_PIN_ORDER(VkVideoEncoderInputFormatProperties, encodeFormat, 4); +VK_ENC_PIN_ORDER(VkVideoEncoderInputFormatProperties, optimality, 8); +#undef VK_ENC_PIN_ORDER + +//============================================================================= +// VkVideoEncoderContentProbe::State is a PRIVATE MIRROR of +// VkVideoEncoderImportContentState. The probe lives in the encoder library, +// which sits below this file and must not include the ext header, so the two +// enums are declared independently -- and independently declared enums drift. +// These pin the mapping to the identity function, which is what +// VkEncMapContentState() below relies on, and pin the one shared numeric +// constant of the predicate. Tripping one means the mirror moved: fix the +// mirror, do not "fix" the assert. +//============================================================================= +static_assert((int)VkVideoEncoderContentProbe::STATE_NOT_EVALUATED == + (int)VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED && + (int)VkVideoEncoderContentProbe::STATE_NOT_APPLICABLE == + (int)VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_APPLICABLE && + (int)VkVideoEncoderContentProbe::STATE_ARMED == + (int)VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED && + (int)VkVideoEncoderContentProbe::STATE_CLEAN == + (int)VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_CLEAN && + (int)VkVideoEncoderContentProbe::STATE_DAMAGED_CHROMA == + (int)VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_CHROMA && + (int)VkVideoEncoderContentProbe::STATE_DAMAGED_ALL == + (int)VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_ALL, + "VkVideoEncoderContentProbe::State drifted from " + "VkVideoEncoderImportContentState"); +static_assert(VkVideoEncoderContentProbe::kDeadPlaneMeanQ8 == + VK_VIDEO_ENCODER_IMPORT_CONTENT_DEAD_PLANE_MEAN_Q8, + "the scorer's dead-plane threshold and the one the public " + "header documents have diverged"); + +static inline VkVideoEncoderImportContentState VkEncMapContentState( + VkVideoEncoderContentProbe::State state) +{ + return (VkVideoEncoderImportContentState)state; +} + //============================================================================= // VulkanVideoEncoderExt interface (external frame input) //============================================================================= +enum VkEncDeviceFormatVerdict { + VK_ENC_DEVICE_FORMAT_ACCEPTED = 0, + // The bit depth is not a VkVideoComponentBitDepthFlagBitsKHR, so no video + // profile can be spelled for it at all. Device-free, and reached before + // the device is asked anything. + VK_ENC_DEVICE_FORMAT_DEPTH_NOT_ENCODABLE, + // The device exposes no encode capability at this (codec, profile, + // subsampling, depth). This is the 4:2:2 and 12-bit answer on both + // measured architectures. + VK_ENC_DEVICE_FORMAT_PROFILE_ABSENT, + // A conversion is needed and no target exists that this device would take. + VK_ENC_DEVICE_FORMAT_NO_CONVERSION_TARGET, + // The device has the profile and does not list the format the encoder + // would be handed as an encode source for it. + VK_ENC_DEVICE_FORMAT_NOT_AN_ENCODE_SOURCE, +}; + +// The reason, in the words a caller can act on. Kept beside the enum so a new +// verdict cannot be added without a sentence. +static const char* VkEncDeviceFormatReason(VkEncDeviceFormatVerdict verdict) +{ + switch (verdict) { + case VK_ENC_DEVICE_FORMAT_DEPTH_NOT_ENCODABLE: + return "that component bit depth is not a video encode bit depth, " + "so no profile can carry it on any device"; + case VK_ENC_DEVICE_FORMAT_PROFILE_ABSENT: + return "this device exposes no encode capability at that chroma " + "subsampling and bit depth"; + case VK_ENC_DEVICE_FORMAT_NO_CONVERSION_TARGET: + return "the conversion this input needs has no output format this " + "device accepts as an encode source"; + case VK_ENC_DEVICE_FORMAT_NOT_AN_ENCODE_SOURCE: + return "this device has that profile and does not list the format " + "the encoder would be handed as an encode source for it"; + case VK_ENC_DEVICE_FORMAT_ACCEPTED: + default: + return "accepted"; + } +} + +// DECLARED HERE, DEFINED BESIDE THE DEVICE QUERY IT CALLS. The vocabulary +// above needs nothing, so it lives where its first reader is; the resolver +// needs VkEncQueryDeviceEncodeSrcFormats and the routable-format helpers, and +// lives with those. +static VkEncDeviceFormatVerdict VkEncResolveDeviceEncodeFormat( + const VulkanDeviceContext& devCtx, + VkPhysicalDevice physDevice, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t codecProfile, + uint32_t chromaSubsampling, + uint32_t bitDepth, + VkFormat inputFormat, + bool viaFilter, + VkFormat& outEncodeFormat); + VkResult VulkanVideoEncoderExtImpl::InitializeExt(const VkVideoEncoderConfig& config) { + // Structure-type gate: reject version skew loudly instead of reading a + // differently-laid-out struct. + if (config.sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG) { + return VK_ERROR_INITIALIZATION_FAILED; + } + // Walk the extension chain. Known links are consumed; anything else is + // rejected rather than ignored -- an extension the library does not + // understand means the caller asked for something it is not getting. + uint64_t requestedFrameTimeoutNs = 0; + VkVideoEncoderValidationFlags validationFlags = 0; + for (const void* link = config.pNext; link != nullptr;) { + // Every public struct opens with {sType, pNext}; read those two + // fields, then re-cast once the type is known. + const auto* base = + reinterpret_cast(link); + switch (base->sType) { + case VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_DEADLINE_INFO: + requestedFrameTimeoutNs = base->frameCompletionTimeoutNs; + break; + case VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO: + case VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO: + // Recognized here so the gate below does not reject it, and + // CONSUMED NOWHERE IN THIS FUNCTION: VkEncBuildEncoderConfig + // walks the same chain and binds it. Two walks, one owner -- + // and the binder is the owner because it is the half a + // device-free test can drive. + break; + case VK_VIDEO_ENCODER_STRUCTURE_TYPE_VALIDATION_INFO: { + const auto* validation = + reinterpret_cast( + link); + if ((validation->flags & + ~VK_VIDEO_ENCODER_VALIDATE_EXTENSIONS_BIT) != 0) { + // A validation this build does not know was asked for + // and will not run; that must be loud, not silent. + return VK_ERROR_INITIALIZATION_FAILED; + } + validationFlags = validation->flags; + break; + } + default: + return VK_ERROR_INITIALIZATION_FAILED; + } + link = base->pNext; + } + // Session-serial: not callable from inside the completion callback. The + // init path reaches WaitForThreadsToComplete, which would join the very + // thread invoking the callback -- a hang where the header promises an + // error code. + if (IsInCompletionCallback()) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + + // Latch the process-wide stdio-silence gate from the config BEFORE any + // output is produced, so no gated VkEncOut() / VkEncErr() line escapes + // ahead of it. Default VK_FALSE preserves behavior. + // + // The ordering requirement stands on the gated streams alone: the latch + // must be installed before anything can write to them. BuildEncoderConfig + // prints nothing at all, and what ParseArguments prints is raw printf / + // fprintf that the latch does not intercept in any case. + // The session's own request, held for as long as the session is. The + // assignment this replaces made the last initializer the only one whose + // choice survived; a token means a second session cannot un-silence the + // first, and this session's silence ends when it does. + m_stdioSilence = VkEncoderStdioSilenceScope(config.silenceStdio == VK_TRUE); + VkVideoCodecOperationFlagBitsKHR codecOp = MapCodecOperation(config.codec); if (codecOp == VK_VIDEO_CODEC_OPERATION_NONE_KHR) { - std::cerr << "[EncoderExt] Unsupported codec: " << (uint32_t)config.codec << std::endl; + VkEncErr() << "[EncoderExt] Unsupported codec: " << (uint32_t)config.codec << std::endl; return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; } // Build EncoderConfig from structured config VkResult result = BuildEncoderConfig(config, codecOp, m_encoderConfig); if (result != VK_SUCCESS) { - std::cerr << "[EncoderExt] BuildEncoderConfig failed: " << result << std::endl; + VkEncErr() << "[EncoderExt] BuildEncoderConfig failed: " << result << std::endl; return result; } @@ -545,17 +3567,184 @@ VkResult VulkanVideoEncoderExtImpl::InitializeExt(const VkVideoEncoderConfig& co return result; } + if ((validationFlags & VK_VIDEO_ENCODER_VALIDATE_EXTENSIONS_BIT) != 0) { + // Everything in this list is an extension the ext import surface + // actually calls into: VkImportMemoryFdInfoKHR + // (external_memory_fd), the DMA_BUF handle type + // (external_memory_dma_buf), DRM-tiled import + // (image_drm_format_modifier), the FOREIGN acquire + // (queue_family_foreign), and vkImportSemaphoreFdKHR + // (external_semaphore_fd). VK_KHR_timeline_semaphore is + // deliberately absent: core-promoted in 1.2, so checking the + // extension STRING would false-fail a driver that stopped + // advertising it. + static const char* const kImportPathExtensions[] = { +#if defined(__linux) || defined(__linux__) || defined(linux) + VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME, + VK_KHR_EXTERNAL_SEMAPHORE_FD_EXTENSION_NAME, + VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME, + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME, + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME, +#elif defined(_WIN32) + VK_KHR_EXTERNAL_MEMORY_WIN32_EXTENSION_NAME, + VK_KHR_EXTERNAL_SEMAPHORE_WIN32_EXTENSION_NAME, +#endif + }; + bool anyMissing = false; + for (const char* name : kImportPathExtensions) { + if (m_vkDevCtx.FindRequiredDeviceExtension(name) == nullptr) { + anyMissing = true; + VkEncErr() << "[EncoderExt] VALIDATE_EXTENSIONS: the " + "session device lacks " << name + << ", which the external-input import path uses" + << std::endl; + } + } + if (anyMissing) { + // Failing loudly HERE, with names, is the point: the + // alternative is passing creation and failing at the first + // import -- or running spec-invalid on a driver that happens + // to tolerate it. + return VK_ERROR_EXTENSION_NOT_PRESENT; + } + } + + // The session-level half of the preprocess-conversion capability check + // (the build-level half is in the binder, which has no device). The + // filter runs on a COMPUTE queue and the staging copy on a TRANSFER + // queue, and they are distinct. Under a caller-supplied VkDevice the + // compute queue is the embedder's, so the queue contract is + // unenforceable and has to be VERIFIED rather than assumed. + // + // Failing here rather than dropping the conversion: with no compute queue + // the filter object cannot be created, the frames that needed converting + // would fall to the copy, and for a 3-plane source that copy is a GPU + // hang, not a slower path. + if ((VkEncClassifyInput(config.inputFormat, config.inputColorModel) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER) && + (m_vkDevCtx.GetComputeQueueFamilyIdx() < 0)) { + VkEncErr() << "[EncoderExt] inputFormat " + << (uint32_t)config.inputFormat + << " is encodable only through the preprocess compute " + "filter, but this device exposes no compute queue " + "family for the session to run it on" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + // ACCEPTANCE IS THE ADVERTISED SET, AND THIS IS WHERE THEY ARE MADE ONE. + // + // The library half of acceptance -- is this a pair this library routes at + // all, and can the named profile carry it -- was settled by + // BuildEncoderConfig above, device-free, and a format that failed there + // never reached this line. What is settled HERE is the half that needs a + // device: whether this device encodes the profile the binder derived, and + // whether it takes the format the encoder would actually be handed. + // + // IT IS THE SAME CALL VkEncQueryInputFormatSupport AND + // VkEncEnumerateInputFormats MAKE, on purpose. Those two advertise what a + // caller may declare; this refuses what they do not advertise. Three + // surfaces, one function, so "consult the list, then find out at init" + // stops being two answers. + // + // WHY BEFORE CreateVideoEncoder AND NOT INSIDE IT. The driver refuses the + // same configuration one call later, and its refusal is the reason this + // gate exists rather than an argument against it: it arrives from + // vkGetPhysicalDeviceVideoCapabilitiesKHR as + // VK_ERROR_VIDEO_PROFILE_FORMAT_NOT_SUPPORTED_KHR with the library + // reporting only "CreateVideoEncoder failed: " -- naming neither + // the format nor its subsampling, which are the two things a caller would + // change. + // + // THE CODE IS THE POINT QUERY'S. VK_ERROR_FORMAT_NOT_SUPPORTED is what + // VkEncQueryInputFormatSupport returns for exactly this verdict, and one + // verdict with two spellings would put the caller back to asking which + // surface it was talking to. + { + // THE ENCODE GEOMETRY, NOT THE INPUT'S. The video profile the session + // is about to create is built from encodeChromaSubsampling and + // encodeBitDepthLuma (EncoderConfig::InitVideoProfile), so a gate that + // asked the device about the INPUT's geometry would be predicting a + // different session than the one it is guarding -- the moment a chroma + // resampler or a depth downgrade makes the two differ. + VkFormat encodeFormat = VK_FORMAT_UNDEFINED; + const VkEncDeviceFormatVerdict verdict = VkEncResolveDeviceEncodeFormat( + m_vkDevCtx, m_vkDevCtx.getPhysicalDevice(), codecOp, + m_encoderConfig->GetCodecProfile(), + (uint32_t)m_encoderConfig->encodeChromaSubsampling, + (uint32_t)m_encoderConfig->encodeBitDepthLuma, + config.inputFormat, + m_encoderConfig->IsPreprocessComputeFilterEnabled(), + encodeFormat); + if (verdict != VK_ENC_DEVICE_FORMAT_ACCEPTED) { + VkEncErr() << "[EncoderExt] inputFormat " + << (uint32_t)config.inputFormat << " (" + << m_encoderConfig->input.numPlanes + << "-plane) cannot be encoded on this device: the " + "stream it derives is " + << VkEncChromaSubsamplingName( + m_encoderConfig->encodeChromaSubsampling) + << " at " << (uint32_t)m_encoderConfig->encodeBitDepthLuma + << " bits, profile " + << m_encoderConfig->GetCodecProfile() << ", and " + << VkEncDeviceFormatReason(verdict) + << ". VkEncEnumerateInputFormats lists what this device " + "does take for this codec and profile, and " + "VkEncQueryInputFormatSupport answers for one pair; " + "this refusal and those two answers are one function." + << std::endl; + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + } + // Create the internal encoder result = VkVideoEncoder::CreateVideoEncoder(&m_vkDevCtx, m_encoderConfig, m_encoder); if (result != VK_SUCCESS) { - std::cerr << "[EncoderExt] CreateVideoEncoder failed: " << result << std::endl; + VkEncErr() << "[EncoderExt] CreateVideoEncoder failed: " << result << std::endl; return result; } + SnapshotCaps(); + m_initConfig = config; + // The session's input declaration, snapshotted for the lock-free readers + // for the same reason as the compute-filter state: SupportsFormat is + // reached from ValidateImageDescriptor without a lock, and m_initConfig + // is written here under one. + m_sessionInputFormat.store(config.inputFormat, std::memory_order_relaxed); + m_sessionInputColorModel.store(config.inputColorModel, + std::memory_order_relaxed); + // The chain was consumed above and points at caller stack storage; + // a retained copy must not carry a pointer about to dangle. + m_initConfig.pNext = nullptr; m_initialized = true; + // Currency 3 is created up front, on the session-serial thread, so the + // submit path can read the handle without a lock. Creation failure + // degrades the currency (GetCompletionSemaphore stays VK_NULL_HANDLE); + // it does not fail the session: no consumer requires the semaphore. + if (m_encoder->CreateCompletionTimelineSemaphore() != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] completion timeline semaphore creation " + "failed; the GPU-ordering currency is unavailable " + "this session" << std::endl; + } + // Remember the rate-control mode for GetRuntimeInfo(). + m_rateControlMode = config.rateControlMode; + // Completion deadline: 0 selects the 8 s default; clamp to stay + // strictly above the library's internal 5 s fence wait so a + // slow-but-successful frame is not converted into a lost one. + m_frameTimeoutNs = (requestedFrameTimeoutNs != 0) + ? requestedFrameTimeoutNs + : 8000000000ull; + if (m_frameTimeoutNs < 6000000000ull) { + VkEncErr() << "[EncoderExt] frameCompletionTimeoutNs clamped to 6s " + "(must exceed the internal 5s fence wait)" << std::endl; + m_frameTimeoutNs = 6000000000ull; + } + // M6: the capture push is the single completion edge; route it to the + // callback machinery (counter + serialized user callback). + m_encoder->SetOnBitstreamCaptured( + [this](uint64_t frameId) { OnBitstreamCaptured(frameId); }); if (config.verbose) { - std::cout << "[EncoderExt] Initialized: " + VkEncOut() << "[EncoderExt] Initialized: " << config.encodeWidth << "x" << config.encodeHeight << " codec=0x" << std::hex << (uint32_t)codecOp << std::dec << " bitrate=" << config.averageBitrate @@ -566,43 +3755,359 @@ VkResult VulkanVideoEncoderExtImpl::InitializeExt(const VkVideoEncoderConfig& co return VK_SUCCESS; } -VkResult VulkanVideoEncoderExtImpl::SubmitExternalFrame( +// Create the PendingFrame entry for a successfully submitted frame. The +// registration reference travels IN the entry from the instant it exists: +// stamping it in afterwards (the previous shape) left a window in which a +// consumer thread could acquire and release the frame before the stamp +// landed, and the stamp's miss was a silent no-op -- an in-flight reference +// nothing would ever drop, which is R-2's pinned-pool-slot stall relocated +// to the success path. |resource| is NULL for the legacy (unregistered) arm. +uint64_t VulkanVideoEncoderExtImpl::ReservePendingFrame( const VkVideoEncodeInputFrame& frame, - VkSemaphore* pStagingCompleteSemaphore) + VkVideoEncoderResource resource, + VkSemaphore releaseFenceSemaphore, + const std::vector* acquireFenceSemaphores) { - if (!m_initialized || !m_encoder) { + std::lock_guard lock(m_pendingMutex); + PendingFrame pending; + pending.admissionToken = m_nextAdmissionToken++; + pending.frameId = frame.frameId; + pending.pts = frame.pts; + pending.resource = resource; + pending.releaseFenceSemaphore = releaseFenceSemaphore; + if (acquireFenceSemaphores != nullptr) { + pending.acquireFenceSemaphores = *acquireFenceSemaphores; + } + // The deadline runs from the reservation, which is the moment the caller + // handed the frame over -- not from the return of a submit that may have + // spent that time inside the driver. + pending.submitTime = std::chrono::steady_clock::now(); + m_pendingFrames.push_back(std::move(pending)); + return m_pendingFrames.back().admissionToken; +} + +void VulkanVideoEncoderExtImpl::CommitPendingFrame( + uint64_t admissionToken, + VkSharedBaseObj& encodeFrameInfo) +{ + std::lock_guard lock(m_pendingMutex); + for (auto& p : m_pendingFrames) { + if (p.admissionToken != admissionToken) { + continue; + } + p.encodeFrameInfo = encodeFrameInfo; + p.admitted = true; + // Counted here and only here, so a rolled-back reservation never + // appears in the submitted total. + m_framesSubmitted++; + return; + } + // Gone already: the frame was released or abandoned while this submit was + // in the driver. Nothing to attach; the submitted count is not raised for + // a frame no longer accounted anywhere. +} + +void VulkanVideoEncoderExtImpl::RollbackPendingFrame( + uint64_t admissionToken, + VkSharedBaseObj& encodeFrameInfo) +{ + std::lock_guard lock(m_pendingMutex); + for (auto it = m_pendingFrames.begin(); it != m_pendingFrames.end(); ++it) { + if (it->admissionToken != admissionToken) { + continue; + } + if (it->hasCapture) { + // Work landed despite the refusal. Destroying this entry would + // drop the registration reference, the release-fence semaphore + // and the imported acquire semaphores while a submission that + // named them completed -- so keep it and let the ordinary + // release path retire them once it is drained. + it->encodeFrameInfo = encodeFrameInfo; + it->admitted = true; + m_framesSubmitted++; + return; + } + // Nothing was accepted, so no completion record can ever pop for + // this entry. It is deliberately NOT disclaimed through + // m_releasedWhilePending: that would leave a record able to swallow + // a genuine capture if the caller reuses the id. + m_pendingFrames.erase(it); + return; + } +} + +void VulkanVideoEncoderExtImpl::EnqueuePendingFrame( + const VkVideoEncodeInputFrame& frame, + VkSharedBaseObj& encodeFrameInfo, + VkVideoEncoderResource resource, + VkSemaphore releaseFenceSemaphore, + const std::vector* acquireFenceSemaphores) +{ + std::lock_guard lock(m_pendingMutex); + PendingFrame pending; + pending.frameId = frame.frameId; + pending.pts = frame.pts; + pending.encodeFrameInfo = encodeFrameInfo; + pending.resource = resource; + // The entry owns the release-fence semaphore for the same reason it owns + // the registration reference: the submission that signals it outlives + // this call, so its destruction has to travel with the frame rather than + // be attempted here. + pending.releaseFenceSemaphore = releaseFenceSemaphore; + // Same argument, same owner: the wait that consumes an acquire fence's + // payload belongs to a submission that outlives this call. + if (acquireFenceSemaphores != nullptr) { + pending.acquireFenceSemaphores = *acquireFenceSemaphores; + } + pending.submitTime = std::chrono::steady_clock::now(); + m_pendingFrames.push_back(std::move(pending)); + m_framesSubmitted++; +} + +VkResult VulkanVideoEncoderExtImpl::SubmitExternalFrameCommon( + const VkVideoEncodeInputFrame& frame, + VkSharedBaseObj* preparedNode, + bool encodeCapable, + bool routeViaFilter, + VkVideoEncoderResource resource, + VkSemaphore* pStagingCompleteSemaphore, + VkSemaphore releaseFenceSemaphore, + int* pReleaseFenceFd, + const std::vector* acquireFenceSemaphores, + bool srcLayoutIsExplicit) +{ + // Class (b) versus Flush()/teardown, which null m_encoder under + // m_pendingMutex and then release it: the ExportCompletionSemaphoreHandle + // pattern. Take a reference under the same lock they clear it under and + // submit through the local reference -- that both closes the TOCTOU on + // the shared pointer and keeps the encoder alive across the inline + // CPU-side encode below if a Flush lands mid-submit (the destructor then + // runs on this thread when the local drops, outside m_pendingMutex, + // exactly as the export path accepts for its driver call). + // Acceptance stops at the start of shutdown, not when the encoder pointer + // is finally cleared. An Unproven shutdown keeps the encoder alive + // deliberately, and without this a caller could keep feeding a pipeline + // whose workers have already been joined. + if (m_shutdownStarted.load(std::memory_order_acquire)) { return VK_ERROR_NOT_PERMITTED_KHR; } + VkSharedBaseObj encoder; + { + std::lock_guard lock(m_pendingMutex); + if (!m_initialized || (!m_encoder && (m_nullBackend == nullptr))) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + encoder = m_encoder; + } + + // Structure-type gate (see the header's versioning rules). + if ((frame.sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME) || + (frame.pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + // Typed rejection replacing a latent null-deref (see SupportsFormat): + // a non-YCbCr frame accepted here would crash in the staging copy. + if (SupportsFormat(frame.format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != VK_TRUE) { + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + // The compute tier is REGISTERED-ONLY, and this is where that is enforced. + // + // SupportsFormat is the SESSION's answer, so it says VK_TRUE for the + // 3-plane family and for the 8-bit RGBA family on any session whose + // filter takes that format. On the REGISTERED arm that VK_TRUE is backed + // by four facts checked at registration: the descriptor's format matched + // the session's filter-input format, the descriptor permitted the views + // ITS arm of the filter reads (per-plane STORAGE views for a multi-planar + // input, a storage-capable combined view for RGBA), the session's filter + // is active, and those views were actually BUILT + // (slot.planeStorageViews / slot.storageReadView). + // The LEGACY arm has none of them -- it has a + // raw VkImage and a format, and WrapExternalImage's LINEAR arm + // deliberately builds a view-less wrapper while its non-linear arm + // FABRICATES MUTABLE_FORMAT | EXTENDED_USAGE and STORAGE on an image whose + // real create flags it cannot see. Routing a filter frame through it binds + // either nothing or views the image never validated. + // + // |routeViaFilter| is the registration's resolved path, false on the + // legacy arm by construction, so this keeps that arm at its documented + // byte-for-byte pre-registration behaviour: the 3-plane family is refused + // outright. + if ((VkEncClassifyInput(frame.format, SessionColorModel(frame.format)) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER) && !routeViaFilter) { + VkEncErr() << "[EncoderExt] submit: inputFormat " + << (uint32_t)frame.format + << " encodes only through the preprocess compute filter, " + "which requires a REGISTERED resource whose descriptor " + "declared MUTABLE_FORMAT | EXTENDED_USAGE and STORAGE; " + "the unregistered submit path cannot supply the " + "per-plane views the filter reads" << std::endl; + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + // The handle gate. It sits BELOW the format and routing refusals on + // purpose: those are statements about what this session can encode and + // must keep answering first (the legacy arm's documented refusal of the + // filter-only formats is VK_ERROR_FORMAT_NOT_SUPPORTED, and a null image + // must not be able to change that answer). It sits ABOVE everything that + // touches the image. + // + // frame.image is required. The gate is unconditional and independent + // of the declared format, so no format classification can decide + // whether it runs; without it a null image reaches the driver on the + // staging path, where the failure is a dereference inside the driver + // rather than a status this call can return. + // + // Unconditional, not legacy-arm-only: SubmitRegisteredFrame builds + // |frame| from the registration slot and always fills image from it, so + // a registered frame cannot reach here null and loses nothing. + if (frame.image == VK_NULL_HANDLE) { + VkEncErr() << "[EncoderExt] submit: frame.image is VK_NULL_HANDLE; " + "the input image is required" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + // Test seam: from here down the submit needs a VkVideoEncoder, so a + // null-backend session (fault injection; see the internal header) + // terminates here -- AFTER every gate a real submit passes. Its success + // arm runs EnqueuePendingFrame, the same bookkeeping the real success + // arm runs below, so what the fault tests exercise is the production + // handoff and not a copy of it. + if (m_nullBackend != nullptr) { + if (m_nullBackend->submitResult != VK_SUCCESS) { + return m_nullBackend->submitResult; + } + if (pStagingCompleteSemaphore != nullptr) { + *pStagingCompleteSemaphore = VK_NULL_HANDLE; + } + // Observation seam (VkEncProbeLastSubmitSync, internal header). This + // point sits below every gate and below the chained-descriptor walk, + // and above the two SetExternalInputFrame* call sites -- which + // forward these same |frame| fields verbatim -- so the record is what + // both of those sites would submit. + RecordSubmitSyncForTest(frame); + VkSharedBaseObj noBackendNode; + // No queue, so nothing ever signals a release fence here. The + // caller's -1 default stands; the fault-injection sessions do not + // create one to begin with. + EnqueuePendingFrame(frame, noBackendNode, resource, + releaseFenceSemaphore, acquireFenceSemaphores); + return VK_SUCCESS; + } + + // Admission control, honoring the header's non-blocking contract: + // refuse with VK_NOT_READY BEFORE any state changes (pool node, GOP + // position, DPB) when the submit path would otherwise block on the + // assembly queue's producer condvar, or when the captured-bitstream + // backlog is unclaimed. A refused frame leaves no residue; the caller + // retries after draining. + if (!encoder->CanAcceptNewInputFrame()) { + return VK_NOT_READY; + } + // Get an available frame info node from the pool VkSharedBaseObj encodeFrameInfo; - bool gotNode = m_encoder->GetAvailablePoolNode(encodeFrameInfo); + bool gotNode = encoder->GetAvailablePoolNode(encodeFrameInfo); if (!gotNode || !encodeFrameInfo) { return VK_NOT_READY; // Pool is full, try again later } - // Build pipeline stage masks (default to TRANSFER for all) - std::vector waitDstStageMasks(frame.waitSemaphoreCount, - VK_PIPELINE_STAGE_2_TRANSFER_BIT_KHR); - - VkResult result = m_encoder->SetExternalInputFrame( - encodeFrameInfo, - frame.image, - VK_NULL_HANDLE, // memory (non-owning, encoder doesn't need it) - frame.format, - frame.width, frame.height, - frame.imageTiling, - frame.currentLayout, // Producer's layout (e.g. GENERAL for compute output) - frame.frameId, - frame.pts, - (frame.isLastFrame == VK_TRUE), - frame.waitSemaphoreCount, - frame.pWaitSemaphores, - frame.pWaitSemaphoreValues, - waitDstStageMasks.data(), - frame.signalSemaphoreCount, - frame.pSignalSemaphores, - frame.pSignalSemaphoreValues); + // No wait-stage masks are fabricated here. VkVideoEncodeInputFrame + // carries none, and the right mask depends on which submission consumes + // the wait, which only the encoder core knows: TRANSFER where the waits + // gate the staging-copy submit (Paths B/C), VIDEO_ENCODE where they gate + // the direct encode submit (Path A). Passing null lets each consumption + // point apply its own default. The TRANSFER-for-everything vector this + // replaces left Path A's vkCmdEncodeVideoKHR outside the waits' scope, + // so the encode could read the input before the producer signaled. + + // Map the public residency declaration onto the internal enum. + VkVideoEncoder::ExternalInputResidency residency = + VkVideoEncoder::EXTERNAL_INPUT_RESIDENCY_AUTO; + switch (frame.inputResidency) { + case VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL: + residency = VkVideoEncoder::EXTERNAL_INPUT_RESIDENCY_LOCAL; break; + case VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN: + residency = VkVideoEncoder::EXTERNAL_INPUT_RESIDENCY_FOREIGN; break; + case VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO: + default: + break; + } + + // Forward the mid-stream keyframe request from the caller. + // + // Registered arm (|preparedNode| non-null): the image, its memory, its + // view and its pool wrapper were created once at registration, so this + // submit performs no wrap and creates no Vulkan object -- the header's + // zero-allocations contract. Routing is the registration-time + // |encodeCapable| predicate, not the per-frame tiling field. + // Legacy arm: byte-for-byte the pre-registration behavior; the + // reference renderer consumes exactly this shape until its M5 + // migration. + // ADMIT BEFORE THE ENCODER CAN PUBLISH. The assembly worker can capture + // this frame's bitstream and call OnBitstreamCaptured before the call + // below returns; without an entry already in the queue that capture is + // unmatched, gets counted as a late capture and is thrown away, and the + // frame that arrives afterwards never completes. The lock is taken and + // released inside ReservePendingFrame -- the workers take it too, so no + // core or queue operation may run while it is held. + const uint64_t admissionToken = ReservePendingFrame( + frame, resource, releaseFenceSemaphore, acquireFenceSemaphores); + + VkResult result; + if (preparedNode != nullptr) { + result = encoder->SetExternalInputFrameWithNode( + encodeFrameInfo, + *preparedNode, + // The registration id, so the staging arm can tell the content + // probe WHICH BUFFER these pixels came out of. Everything else + // the probe needs it already has; this is the only fact that + // lives up here and nowhere down there. + resource, + encodeCapable, + routeViaFilter, + frame.currentLayout, + srcLayoutIsExplicit, + frame.frameId, + frame.pts, + (frame.isLastFrame == VK_TRUE), + (frame.forceIDR == VK_TRUE), + frame.qpOverride, + residency, + frame.waitSemaphoreCount, + frame.pWaitSemaphores, + frame.pWaitSemaphoreValues, + nullptr, // wait-stage masks: each consumption point defaults + frame.signalSemaphoreCount, + frame.pSignalSemaphores, + frame.pSignalSemaphoreValues); + } else { + result = encoder->SetExternalInputFrame( + encodeFrameInfo, + frame.image, + VK_NULL_HANDLE, // memory (non-owning, encoder doesn't need it) + frame.format, + frame.width, frame.height, + frame.imageTiling, + frame.currentLayout, // Producer's layout (e.g. GENERAL for compute output) + frame.frameId, + frame.pts, + (frame.isLastFrame == VK_TRUE), + (frame.forceIDR == VK_TRUE), + frame.qpOverride, + residency, + frame.waitSemaphoreCount, + frame.pWaitSemaphores, + frame.pWaitSemaphoreValues, + nullptr, // wait-stage masks: each consumption point defaults + frame.signalSemaphoreCount, + frame.pSignalSemaphores, + frame.pSignalSemaphoreValues); + } + if (result == VK_SUCCESS) { // Return the semaphore that signals when the encoder is done @@ -624,192 +4129,8961 @@ VkResult VulkanVideoEncoderExtImpl::SubmitExternalFrame( } } - // Track for async retrieval - std::lock_guard lock(m_pendingMutex); - m_pendingFrames.push_back({frame.frameId, frame.pts, encodeFrameInfo}); - m_framesSubmitted++; + // Per-frame release fence (design 3.5): export the SYNC_FD now, and + // only now. + // + // WHY HERE AND NOT EARLIER OR LATER. A SYNC_FD export requires the + // semaphore to be signalled or to have a signal operation PENDING + // EXECUTION, so it cannot precede the submit that carries the + // signal. It also cannot be deferred past this function, because + // pReleaseFenceFd is an out-parameter of the submit call. This point + // -- immediately after the Set*Frame call, still holding the + // frame-info node -- is the one place where the submit has happened + // and the evidence that it happened is still readable. + // + // WHY THE SUBMIT IS NOT ALWAYS ISSUED. The staged paths submit the + // staging copy inline from StageInputFrame, so their release signal + // is always pending on return. The DIRECT path submits from the + // deferred-GOP flush, which under B-frame reordering happens on a + // LATER call -- the frame is queued, not submitted. Asking the + // command-buffer node whether it reached vkQueueSubmit is the only + // answer that cannot drift from that scheduling decision, so that is + // what is asked, rather than predicting it from the GOP config. + // + // WHAT THE fd SIGNALS. The submission it rides is the one that READS + // the input image -- the staging copy on Paths B/C, the encode on + // Path A. It is deliberately NOT the encode's completion: the + // bitstream becoming retrievable is a different event on a different + // resource, and only the read-completion meaning is usable by the + // end-of-read-access seam this fd exists to feed. + if ((releaseFenceSemaphore != VK_NULL_HANDLE) && + (pReleaseFenceFd != nullptr)) { + // ASK BOTH NODES WHETHER THEY WERE SUBMITTED. The staging arm + // used to take the mere existence of inputCmdBuffer as proof, + // which is true of a recorded batch the driver went on to reject: + // the fd was then exported from a semaphore with no signal + // operation queued, and the embedder waited on it forever. + const bool inputConsumingSubmitIssued = + (encodeFrameInfo->inputCmdBuffer && + encodeFrameInfo->inputCmdBuffer->IsCommandBufferSubmitted()) || + (encodeFrameInfo->encodeCmdBuffer && + encodeFrameInfo->encodeCmdBuffer->IsCommandBufferSubmitted()); + if (inputConsumingSubmitIssued) { + *pReleaseFenceFd = ExportReleaseFenceFd(releaseFenceSemaphore); + } + // else: the -1 the walk already wrote stands. The signal will + // still fire on the deferred submit and simply go unconsumed -- + // which is why this semaphore may not be recycled, only + // destroyed once its batch has completed. + } + + // Finish the admission the reservation opened. The entry already + // owns the registration reference and the fence semaphores; what the + // submit adds is the frame info that returns pool resources when the + // frame is released. + CommitPendingFrame(admissionToken, encodeFrameInfo); + } else { + RollbackPendingFrame(admissionToken, encodeFrameInfo); } return result; } -VkResult VulkanVideoEncoderExtImpl::PollEncodeComplete(uint64_t frameId) +VkResult VulkanVideoEncoderExtImpl::SubmitExternalFrame( + const VkVideoEncodeInputFrame& frame, + VkSemaphore* pStagingCompleteSemaphore) { - std::lock_guard lock(m_pendingMutex); - for (auto& pending : m_pendingFrames) { - if (pending.frameId == frameId) { - if (pending.encodeFrameInfo->encodeCmdBuffer) { - VkFence fence = pending.encodeFrameInfo->encodeCmdBuffer->GetFence(); - if (fence != VK_NULL_HANDLE) { - VkResult status = m_vkDevCtx.GetFenceStatus(m_vkDevCtx, fence); - return (status == VK_SUCCESS) ? VK_SUCCESS : VK_NOT_READY; - } + // Legacy arm: no prepared node. Behavior-preserving by construction -- + // the null-node branch of the common path IS the pre-registration code. + // Legacy arm: no registration, so no resolved path. The encoder core + // derives the rung from the frame's own format there. + return NoteDeviceResult( + SubmitExternalFrameCommon(frame, nullptr, false, false, + VK_VIDEO_ENCODER_RESOURCE_NULL, + pStagingCompleteSemaphore)); +} + +// Mark |frame| deliverable and record its place in completion order. This +// is the single point where hasCapture is raised, so no outcome -- capture, +// deadline drop, or cancellation -- can become deliverable without also +// becoming reachable by AcquireNextEncodedFrame. Caller holds m_pendingMutex. +void VulkanVideoEncoderExtImpl::MarkFrameReadyLocked(PendingFrame& frame) +{ + if (frame.hasCapture) { + return; // already deliverable; must not be enqueued twice + } + frame.hasCapture = true; + m_readyOrder.push_back(frame.frameId); +} + +// Pop the oldest deliverable frame off the ready queue and fill |result|. +// Caller holds m_pendingMutex. Stale ids -- a frame since released, or one +// taken first by the keyed AcquireEncodedFrame -- are skipped, which is why +// the queue never has to be kept in exact sync with m_pendingFrames. +bool VulkanVideoEncoderExtImpl::TryPopReadyLocked(VkVideoEncodeResult& result) +{ + while (!m_readyOrder.empty()) { + const uint64_t readyId = m_readyOrder.front(); + m_readyOrder.pop_front(); + for (auto& p : m_pendingFrames) { + if (p.frameId != readyId) { + continue; } - return VK_NOT_READY; + if (!p.acquired && p.hasCapture) { + FillResultLocked(p, result); + return true; + } + break; } } - return VK_ERROR_UNKNOWN; // frameId not found + return false; } -VkResult VulkanVideoEncoderExtImpl::GetEncodedFrame(VkVideoEncodeResult& result) +// Close a POSIX fd the library was handed and STILL HOLDS. The rule the +// caller sees is unconditional -- an fd is consumed on every exit path -- +// so every pre-Vulkan early return in the registration paths funnels +// through here. What must never funnel through here is an exit past the +// vkAllocateMemory handoff in the image import: there the DRIVER owns the +// close (see VkEncImportExternalImage), and a second one would be design +// section 2.3's double close. Win32 handles are never touched. +static void VkEncConsumeOsHandle(VkVideoEncoderExternalHandleType handleType, + uint64_t osHandle) { - std::lock_guard lock(m_pendingMutex); - if (m_pendingFrames.empty()) { - return VK_NOT_READY; +#if defined(__linux__) + if (((handleType == VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD) || + (handleType == VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF)) && + ((int64_t)osHandle >= 0)) { + close((int)osHandle); } +#else + (void)handleType; + (void)osHandle; +#endif +} - // Check the oldest pending frame (FIFO) - auto& front = m_pendingFrames.front(); - if (!front.encodeFrameInfo->encodeCmdBuffer) { - return VK_NOT_READY; +// Writes the ownership echo on every exit path -- a dozen early returns +// and one missed write is a caller asserting against garbage, the same +// failure mode the R-2 release guard exists for, applied to reporting. +// Writes only into a correctly stamped struct: a mis-stamped pStatus was +// answered STRUCTURE_TYPE_UNKNOWN and must not be written through as if +// it had the expected layout. +class VkEncStatusEcho { +public: + VkEncStatusEcho(VkVideoEncoderStatus* pStatus, bool handlesConsumed) + : m_pStatus(pStatus), m_handlesConsumed(handlesConsumed) {} + // The import-ordinal guard's verdict rides the SAME every-exit + // guarantee as the ownership echo, and for the same reason: the guard + // runs before most of the early returns, so reporting it only on the + // success path would leave the failures -- the ones worth reporting -- + // silent all over again. + // + // Installed only AFTER the chain walk has accepted the link. A refused + // chain must not be written through, exactly as a mis-stamped pStatus + // is not. + void SetGuardInfo(VkVideoEncoderImportGuardInfo* pGuardInfo) { + m_pGuardInfo = pGuardInfo; } - - VkFence fence = front.encodeFrameInfo->encodeCmdBuffer->GetFence(); - if (fence == VK_NULL_HANDLE) { - return VK_NOT_READY; + // The content probe's ARMING result rides the same every-exit guarantee, + // and needs it for the same reason: a registration that is refused before + // the probe can be armed must still say so, rather than leaving the + // caller's struct holding whatever it held before the call. + // + // Two setters and not one, because the two facts arrive at different + // times: the POINTER is installed by the chain walk, which runs early, + // and the VERDICT by the armer, which runs after the registration has + // either succeeded or not. Between them the state stays NOT_EVALUATED, + // which is the truthful answer for every exit in between. + void SetContentInfo(VkVideoEncoderImportContentInfo* pContentInfo) { + m_pContentInfo = pContentInfo; } - - VkResult fenceStatus = m_vkDevCtx.GetFenceStatus(m_vkDevCtx, fence); - if (fenceStatus != VK_SUCCESS) { - return VK_NOT_READY; + void SetContentVerdict(VkVideoEncoderImportContentState state, + VkVideoEncoderResource resource) { + m_contentState = state; + m_contentResource = resource; } + ~VkEncStatusEcho() { + if ((m_pStatus != nullptr) && + (m_pStatus->sType == VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS)) { + m_pStatus->handlesConsumed = + m_handlesConsumed ? VK_TRUE : VK_FALSE; + } + if ((m_pGuardInfo != nullptr) && + (m_pGuardInfo->sType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO)) { + VkEncImportOrdinalGuardReport report; + VkEncGetImportOrdinalGuardReport(&report); + m_pGuardInfo->state = report.state; + m_pGuardInfo->requestedCount = report.requestedCount; + m_pGuardInfo->retainedCount = report.retainedCount; + m_pGuardInfo->failureStatus = report.failureStatus; + m_pGuardInfo->failureErrno = report.failureErrno; + } + if ((m_pContentInfo != nullptr) && + (m_pContentInfo->sType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO)) { + // probeGeneration FIRST and unconditionally: it is the writer + // proof, and a writer proof that is only stamped on the paths + // that had something to say proves nothing about the paths that + // did not. This is the field the guard's requestedCount stopped + // being able to be at the guard count of 0. + m_pContentInfo->probeGeneration = + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION; + m_pContentInfo->state = m_contentState; + m_pContentInfo->resource = m_contentResource; + // The registration echo carries no measurement and says so with + // zeros: at RegisterImageResource time the producer has written + // nothing, so there is nothing to have measured. The verdict and + // its arithmetic arrive on GetCompletionInfo. + m_pContentInfo->meanY = 0; + m_pContentInfo->meanU = 0; + m_pContentInfo->meanV = 0; + m_pContentInfo->probedRegistrationCount = 0; + m_pContentInfo->damagedRegistrationCount = 0; + } + } + VkEncStatusEcho(const VkEncStatusEcho&) = delete; + VkEncStatusEcho& operator=(const VkEncStatusEcho&) = delete; +private: + VkVideoEncoderStatus* m_pStatus; + bool m_handlesConsumed; + VkVideoEncoderImportGuardInfo* m_pGuardInfo = nullptr; + VkVideoEncoderImportContentInfo* m_pContentInfo = nullptr; + VkVideoEncoderImportContentState m_contentState = + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED; + VkVideoEncoderResource m_contentResource = VK_VIDEO_ENCODER_RESOURCE_NULL; +}; - // Frame is complete - fill the result - result.frameId = front.frameId; - result.pts = front.pts; - result.dts = 0; // TODO: compute from encode order - result.pBitstreamData = nullptr; // Bitstream is in file for now - result.bitstreamSize = 0; - result.pictureType = 0; // TODO: extract from GOP position - result.isIDR = VK_FALSE; - result.temporalLayerId = 0; - result.status = VK_SUCCESS; - - // Note: the frame stays in m_pendingFrames until ReleaseEncodedFrame() - return VK_SUCCESS; +VulkanVideoEncoderExtImpl::RegisteredImage* +VulkanVideoEncoderExtImpl::LookupResourceLocked(VkVideoEncoderResource resource) +{ + if (resource == VK_VIDEO_ENCODER_RESOURCE_NULL) { + return nullptr; + } + const size_t index = ResourceIndex(resource); + if (index >= m_resources.size()) { + return nullptr; + } + RegisteredImage& slot = m_resources[index]; + // The generation check is the point: a stale id from before an + // unregister names a generation this slot no longer has. + if (!slot.live || (slot.generation != ResourceGeneration(resource))) { + return nullptr; + } + return &slot; } -void VulkanVideoEncoderExtImpl::ReleaseEncodedFrame(uint64_t frameId) +void VulkanVideoEncoderExtImpl::DestroyResourceLocked(RegisteredImage& slot) { - std::lock_guard lock(m_pendingMutex); - for (auto it = m_pendingFrames.begin(); it != m_pendingFrames.end(); ++it) { - if (it->frameId == frameId) { - // Release the encodeFrameInfo ref - returns resources to pools - it->encodeFrameInfo = nullptr; - m_pendingFrames.erase(it); - return; + if (slot.node || slot.imageView) { + // The refcounted view/node own the image and memory now (the import + // arm wraps them owning via CreateFromImport). Dropping the refs IS + // the release; per-frame holders (VkVideoEncodeFrameInfo) keep the + // node alive until the last in-flight frame retires, which is what + // makes this safe to call from deferred retirement -- the wrapper's + // refcount, not this function, is the final arbiter. + slot.node = nullptr; + slot.imageView = nullptr; + } else if (slot.ownsImage) { + // Only reachable when registration failed before the wrapper was + // built; the raw handles are still ours to free. + VkDevice device = m_vkDevCtx; + if (slot.image != VK_NULL_HANDLE) { + m_vkDevCtx.DestroyImage(device, slot.image, nullptr); + } + if (slot.memory != VK_NULL_HANDLE) { + // Also releases the imported fd: Vulkan took ownership of it. + m_vkDevCtx.FreeMemory(device, slot.memory, nullptr); + } + } + // A2 scratch (reserved): nothing creates these + // today; destroying them here means the future population cannot leak + // on retirement. Always library-owned, so never under the ownsImage + // guard above -- view, then image, then memory. + if ((slot.filterScratchView != VK_NULL_HANDLE) || + (slot.filterScratchImage != VK_NULL_HANDLE) || + (slot.filterScratchMemory != VK_NULL_HANDLE)) { + VkDevice scratchDevice = m_vkDevCtx; + if (slot.filterScratchView != VK_NULL_HANDLE) { + m_vkDevCtx.DestroyImageView(scratchDevice, + slot.filterScratchView, nullptr); + } + if (slot.filterScratchImage != VK_NULL_HANDLE) { + m_vkDevCtx.DestroyImage(scratchDevice, + slot.filterScratchImage, nullptr); + } + if (slot.filterScratchMemory != VK_NULL_HANDLE) { + m_vkDevCtx.FreeMemory(scratchDevice, + slot.filterScratchMemory, nullptr); } } + slot.filterScratchView = VK_NULL_HANDLE; + slot.filterScratchImage = VK_NULL_HANDLE; + slot.filterScratchMemory = VK_NULL_HANDLE; + slot.inputPath = VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + slot.image = VK_NULL_HANDLE; + slot.memory = VK_NULL_HANDLE; + slot.live = false; + slot.retired = false; + slot.ownsImage = false; + slot.inFlight = 0; + slot.imageUsage = 0; + slot.encodeCapable = false; + slot.planeStorageViews = false; + slot.storageReadView = false; + slot.importedAllocSize = 0; + slot.generation++; // invalidate every id that named this slot } -VkFence VulkanVideoEncoderExtImpl::GetEncodeFence(uint64_t frameId) +// Cache the capability scalars the lock-free query methods report. Called +// once per successful init, while m_encoderConfig is known live and no +// retrieval thread exists yet. +void VulkanVideoEncoderExtImpl::SnapshotCaps() { - std::lock_guard lock(m_pendingMutex); - for (auto& pending : m_pendingFrames) { - if (pending.frameId == frameId && pending.encodeFrameInfo->encodeCmdBuffer) { - return pending.encodeFrameInfo->encodeCmdBuffer->GetFence(); - } + if (!m_encoderConfig) { + return; } - return VK_NULL_HANDLE; + const auto& caps = m_encoderConfig->videoCapabilities; + m_capsGranularityW.store(caps.pictureAccessGranularity.width, + std::memory_order_relaxed); + m_capsGranularityH.store(caps.pictureAccessGranularity.height, + std::memory_order_relaxed); + // Compute-filter state, for the same lock-free readers. input.vkFormat is + // meaningful here because the binder wrote input.numPlanes from the + // caller's inputFormat (VkEncBuildEncoderConfig), so FinalizeConfig + // reconstructed the format the caller actually declared rather than + // EncoderConfig's 3-plane default. + const bool filterEnabled = + m_encoderConfig->IsPreprocessComputeFilterEnabled(); + m_computeFilterActive.store(filterEnabled, std::memory_order_relaxed); + m_computeFilterInputFormat.store( + filterEnabled ? m_encoderConfig->input.vkFormat : VK_FORMAT_UNDEFINED, + std::memory_order_relaxed); } -VkResult VulkanVideoEncoderExtImpl::Flush() +// Shared drain: move completed captures from the encoder's FIFO onto their +// PendingFrame entries. Caller holds m_pendingMutex. A capture for a frame +// already delivered as a deadline drop or a cancellation -- whether the +// entry is still resident or was released outright -- is LATE: discarded +// and counted, so a deadline/fence-cap collision is observable instead of +// silent. A capture for a frame released while it was still PENDING is not +// late -- nothing was delivered that it could arrive after -- and is +// discarded without comment (m_releasedWhilePending carries the ids). +void VulkanVideoEncoderExtImpl::DrainCapturesLocked() { - if (!m_initialized || !m_encoder) { - return VK_ERROR_NOT_PERMITTED_KHR; + if (!m_encoder) { + return; } - - // Wait for all pending encodes to complete - m_encoder->WaitForThreadsToComplete(); - - // Drain the pending queue - { - std::lock_guard lock(m_pendingMutex); - for (auto& pending : m_pendingFrames) { - if (pending.encodeFrameInfo && pending.encodeFrameInfo->encodeCmdBuffer) { - pending.encodeFrameInfo->encodeCmdBuffer->ResetCommandBuffer( - true, "EncoderExtFlush"); + uint64_t popped_id = 0; + std::vector popped_bytes; + bool popped_idr = false; + uint32_t popped_pic_type = 0; + VkResult popped_status = VK_SUCCESS; + while (m_encoder->TryPopCapturedBitstream( + &popped_id, &popped_bytes, &popped_idr, + &popped_pic_type, &popped_status)) { + bool matched = false; + for (auto& p : m_pendingFrames) { + if (p.frameId == popped_id) { + matched = true; + if (p.timedOut) { + m_lateCaptures++; + VkEncErr() << "[EncoderExt] late capture for timed-out " + "frame " << popped_id << " discarded " + "(lateCaptures=" << m_lateCaptures << ")" + << std::endl; + break; + } + p.bytes = std::move(popped_bytes); + p.isIdr = popped_idr; + p.pictureType = popped_pic_type; + p.status = popped_status; + MarkFrameReadyLocked(p); + break; + } + } + if (!matched) { + auto released = m_releasedWhilePending.find(popped_id); + if (released != m_releasedWhilePending.end()) { + // Disclaimed in advance: ReleaseEncodedFrame (or + // AbandonAllFrames) erased this frame while it was still + // PENDING, which the header documents as legal -- the + // file-output fire-and-forget consumers it names (the + // reference renderer) release on submit success as their + // steady state. Not late: nothing was delivered for this + // capture to arrive after. Consume the record and drop the + // payload silently. + m_releasedWhilePending.erase(released); + } else { + // No resident entry and no released-while-pending record: + // the frame was DELIVERED -- as a deadline drop or a + // cancellation -- then released before this capture, the + // real one, arrived. Same late class as the timed-out arm + // above, counted the same way. (An unmatched pop this arm + // holds no record of is counted late too; it cannot + // classify what it never saw disclaimed.) + m_lateCaptures++; + VkEncErr() << "[EncoderExt] late capture for released frame " + << popped_id << " discarded (lateCaptures=" + << m_lateCaptures << ")" << std::endl; } } } - - // Release the encoder — the destructor calls DeinitEncoder() which - // writes any buffered bitstream (deferred frames from GOP reordering) - // and closes the output file. - m_encoder = nullptr; - - return VK_SUCCESS; } -VkResult VulkanVideoEncoderExtImpl::Reconfigure(const VkVideoEncoderConfig& config) +// Deadline check: true when |frame| (no capture yet) is past its completion +// deadline; synthesizes the 0-byte VK_TIMEOUT drop delivery in place. +bool VulkanVideoEncoderExtImpl::SynthesizeTimeoutLocked(PendingFrame& frame) { - // TODO: Implement dynamic reconfiguration - // For now, rate control changes require session reset - (void)config; - return VK_ERROR_FEATURE_NOT_PRESENT; -} - -VkBool32 VulkanVideoEncoderExtImpl::SupportsFormat(VkFormat inputFormat) const -{ - // Formats the encoder can handle (directly or via filter) - switch (inputFormat) { - // Directly encodable (YCbCr 4:2:0) - case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: - case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16: - case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16: - // Via filter (RGBA) - case VK_FORMAT_R8G8B8A8_UNORM: - case VK_FORMAT_B8G8R8A8_UNORM: - case VK_FORMAT_R8G8B8A8_SRGB: - case VK_FORMAT_B8G8R8A8_SRGB: - case VK_FORMAT_A2B10G10R10_UNORM_PACK32: - case VK_FORMAT_R16G16B16A16_SFLOAT: - return VK_TRUE; - default: - return VK_FALSE; + if (!frame.admitted) { + // Still a reservation: its submit has not returned, so there is no + // outstanding encode to declare late. Dropping it here would also + // make the capture that follows arrive for a timed-out frame and be + // discarded, which is the loss this reservation exists to prevent. + return false; } + const auto waited = std::chrono::steady_clock::now() - frame.submitTime; + const uint64_t waitedNs = (uint64_t)std::chrono::duration_cast< + std::chrono::nanoseconds>(waited).count(); + if (waitedNs < m_frameTimeoutNs) { + return false; + } + frame.bytes.clear(); + frame.isIdr = false; + frame.pictureType = + (uint32_t)VkVideoGopStructure::FRAME_TYPE_I; // translated to I + frame.status = VK_TIMEOUT; + MarkFrameReadyLocked(frame); + frame.timedOut = true; + m_framesTimedOut++; + VkEncErr() << "[EncoderExt] frame " << frame.frameId + << " passed the completion deadline; delivered as a 0-byte " + "VK_TIMEOUT drop (framesTimedOut=" << m_framesTimedOut + << ")" << std::endl; + return true; } -uint32_t VulkanVideoEncoderExtImpl::GetMaxWidth() const -{ - // TODO: query from device capabilities - return 8192; -} - -uint32_t VulkanVideoEncoderExtImpl::GetMaxHeight() const +// Fill |result| from |frame| and mark it delivered. Caller holds the lock. +void VulkanVideoEncoderExtImpl::FillResultLocked(PendingFrame& frame, + VkVideoEncodeResult& result) { - // TODO: query from device capabilities - return 8192; + result.frameId = frame.frameId; + result.pts = frame.pts; + // Always 0. With no B-frames there is no reorder, so decode order equals + // presentation order and a consumer that needs a DTS can use the PTS. + // Deriving a real DTS only becomes meaningful alongside B-frame support, + // and is specified there rather than left as a bare TODO on a public field. + result.dts = 0; + result.pBitstreamData = frame.bytes.empty() ? nullptr : frame.bytes.data(); + result.bitstreamSize = static_cast(frame.bytes.size()); + // Internal captures carry VkVideoGopStructure::FrameType numbering + // (P=0, B=1, I=2, IDR=3, intra-refresh=6); the public field uses + // VkVideoEncoderPictureType (I=0, P=1, B=2). Translate at the boundary. + switch (static_cast(frame.pictureType)) { + case VkVideoGopStructure::FRAME_TYPE_P: + result.pictureType = VK_VIDEO_ENCODER_PICTURE_TYPE_P; + break; + case VkVideoGopStructure::FRAME_TYPE_B: + result.pictureType = VK_VIDEO_ENCODER_PICTURE_TYPE_B; + break; + default: // I / IDR / intra-refresh + result.pictureType = VK_VIDEO_ENCODER_PICTURE_TYPE_I; + break; + } + result.isIDR = frame.isIdr ? VK_TRUE : VK_FALSE; + result.temporalLayerId = 0; + // Per-frame result from the capture (VK_SUCCESS, the assembly/readback + // failure code, or VK_TIMEOUT for a deadline drop). + result.status = frame.status; + // The frame stays in m_pendingFrames until ReleaseEncodedFrame(). + frame.acquired = true; } -void VulkanVideoEncoderExtImpl::Deinitialize() +void VulkanVideoEncoderExtImpl::OnBitstreamCaptured(uint64_t frameId) { - if (m_encoder) { - m_encoder->WaitForThreadsToComplete(); - } + m_completionCounter.fetch_add(1, std::memory_order_release); + // Invoke lock FIRST, snapshot second. A snapshot taken outside + // m_callbackMutex could still be invoked after SetCompletionCallback + // returned (the store only races the read, not the invocation), so a + // detaching caller could never know when its pUserData became safe to + // destroy. Reading the pair inside the invoke lock closes that hole; + // the lock order (callback -> pending) matches SetCompletionCallback + // and any class-(c) method a user callback re-enters. + std::lock_guard invokeLock(m_callbackMutex); + PFN_vkVideoEncoderCompletionCallback callback; + void* userData; { std::lock_guard lock(m_pendingMutex); - m_pendingFrames.clear(); + // Route records onto their PendingFrames ON the edge, not lazily at + // the next retrieval call. Both reasons are load-bearing: + // 1. a consumer that never retrieves (file-output mode with + // fire-and-forget Release -- the reference renderer) would + // otherwise accumulate records nothing drains until the + // unclaimed-captures bound stalls its submits at ~64 frames; + // 2. when the user callback below runs, the frame it names is + // already acquirable, so callback-then-acquire cannot observe + // a pushed-but-unrouted record. + // Lock order callback -> pending -> captured matches the retrieval + // methods; no path anywhere takes captured before pending. + DrainCapturesLocked(); + callback = m_completionCallback; + userData = m_completionUserData; } - - m_encoder = nullptr; - m_encoderConfig = nullptr; - m_initialized = false; + // AFTER the routing above and BEFORE the no-callback early return below. + // Both halves of that placement are load-bearing: + // * after: the load is ordered after this thread's m_pendingMutex + // critical section, which orders it against the creator's critical + // section in GetCompletionEventHandle() (release store, then prime + // scan, under the same mutex). Whichever runs first, the wakeup is + // delivered: if the creator ran first, the acquire load observes the + // published handle; if this routing ran first, DrainCapturesLocked() + // above has already raised hasCapture on the frame, so the creator's + // prime scan signals for it. Loading before the critical section + // (the previous placement) satisfied neither arm: the load could + // miss a handle whose prime scan then ran before this frame's drain, + // and the wakeup was lost. The placement does not change what a + // woken waiter finds -- the record enters the capture FIFO before + // the notify edge, and every retrieval entry point drains under + // m_pendingMutex -- what it orders is this load against the + // handle's creation. + // * before: this function returns early when no callback is + // registered, and a handle-only consumer -- the out-of-process case + // this handle exists for -- would then never be woken at all. + // The counter increment at the top of this function stays ahead of the + // signal, so a waiter that reads the counter immediately on wake sees + // this frame. + vkenc::OsCompletionEventSignal( + m_completionEventHandle.load(std::memory_order_acquire)); + if (callback == nullptr) { + return; + } + m_callbackThreadId.store(std::this_thread::get_id(), + std::memory_order_relaxed); +#if defined(__cpp_exceptions) && (__cpp_exceptions != 0) + // Live only for a consumer building this library WITH exceptions and + // WITHOUT C++17 (so the typedef's noexcept expanded to nothing). Aborting + // is the only safe response -- we hold m_callbackMutex and our caller + // holds the assembly ordering lock -- but an abort inside an anonymous + // library worker is near-impossible to attribute from a core dump, so + // name the frame first. This is the diagnosed abort the design asks for. + try { + callback(frameId, userData); + } catch (...) { + VkEncErr() << "[EncoderExt] completion callback threw for frame " + << frameId << "; the callback contract is noexcept -- " + "aborting" << std::endl; + std::abort(); + } +#else + // Chromium compiles this TU with -fno-exceptions, where `try` is a + // compile error rather than dead code. There is nothing to catch: a + // throwing callback terminates at its own throw site, before any handler + // here could run. The contract is carried by the typedef's noexcept and + // by the header, which is all that is available in this build. + callback(frameId, userData); +#endif + m_callbackThreadId.store(std::thread::id(), std::memory_order_relaxed); } -//============================================================================= +VkResult VulkanVideoEncoderExtImpl::SetCompletionCallback( + PFN_vkVideoEncoderCompletionCallback callback, void* pUserData, + PFN_vkVideoEncoderUserDataRelease releaseUserData) +{ + if (IsInCompletionCallback()) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + PFN_vkVideoEncoderUserDataRelease priorRelease = nullptr; + void* priorUserData = nullptr; + // Both locks, in the OnBitstreamCaptured order (callback -> pending): + // holding m_callbackMutex across the store makes this call a quiesce + // point -- on return, no invocation of the PREVIOUS callback is in + // flight or can start, so a nullptr detach lets the caller safely + // destroy whatever the previous pUserData referenced. + { + std::scoped_lock locks(m_callbackMutex, m_pendingMutex); + priorRelease = m_completionUserDataRelease; + priorUserData = m_completionUserData; + m_completionCallback = callback; + m_completionUserData = pUserData; + m_completionUserDataRelease = releaseUserData; + } + // Released AFTER the locks drop, and only once the quiesce above + // guarantees no invocation can still be holding it. Calling it under the + // locks would deadlock any release that touches this encoder, and a + // release handler is exactly the place a caller tears things down. + if (priorRelease != nullptr) { + priorRelease(priorUserData); + } + return VK_SUCCESS; +} + +uint64_t VulkanVideoEncoderExtImpl::GetCompletionCounter() +{ + return m_completionCounter.load(std::memory_order_acquire); +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::GetCompletionEventHandle( + uint64_t* outHandle) +{ + if (outHandle == nullptr) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + std::lock_guard lock(m_pendingMutex); + uint64_t handle = m_completionEventHandle.load(std::memory_order_acquire); + if (handle == vkenc::kOsCompletionEventNone) { + handle = vkenc::OsCompletionEventCreate(); + if (handle == vkenc::kOsCompletionEventNone) { + // Either this build has no adapter for the platform, or the + // syscall failed. Both mean the same thing to a caller: branch, + // do not wait on a handle nothing will signal. (The 0 written + // here is a benign fill, not a sentinel: the status code is the + // failure signal, and 0 is a legal handle value.) + *outHandle = 0; + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; + } + // Release pairs with the capture path's acquire load: the eventfd's + // creation happens-before any signal through the published value. + m_completionEventHandle.store(handle, std::memory_order_release); + + // PRIME. A consumer is allowed to ask for the handle mid-stream, and + // frames that became ready BEFORE it existed were signalled into + // kOsCompletionEventNone -- a no-op. Without this, such a consumer + // waits on an eventfd whose backlog it can never be woken for, and + // the symptom is indistinguishable from a library stall. One signal + // is enough regardless of how many are already ready: the eventfd + // accumulates, and the header's contract is a drain loop reconciled + // against GetCompletionCounter(), never one-wake-one-frame. Any + // frame routed after this point takes m_pendingMutex, which this + // thread holds, so it observes the store above and signals normally. + for (const auto& p : m_pendingFrames) { + if (!p.acquired && p.hasCapture) { + vkenc::OsCompletionEventSignal(handle); + break; + } + } + } + // The same handle every time: handing out a second one would give the + // caller something the encoder does not signal. + *outHandle = handle; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkSemaphore VulkanVideoEncoderExtImpl::GetCompletionSemaphore() const +{ + // Class (c) versus Flush()/teardown, which clear m_encoder under + // m_pendingMutex and then release the encoder: an unlocked read here + // raced that null-and-destroy (TOCTOU on the shared pointer, then a + // call through an object being torn down). The getter behind the + // lock is a plain member read, so holding m_pendingMutex across it + // cannot deadlock and costs nothing measurable. + std::lock_guard lock(m_pendingMutex); + if (!m_initialized || !m_encoder) { + return VK_NULL_HANDLE; + } + return m_encoder->GetCompletionTimelineSemaphore(); +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::ExportCompletionSemaphoreHandle( + VkVideoEncoderExternalHandleType handleType, + uint64_t* outHandle) +{ + if (outHandle == nullptr) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + *outHandle = 0; + // Type gate first: a property of the request alone, answered the same + // with or without a session -- the RegisterSemaphore ordering. The + // reserved Win32 arm and a device that cannot export share one answer + // and one clean caller branch. + if (handleType != VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD) { + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; + } + // Class (c) versus Flush()/teardown: take a reference to the encoder + // under the same lock they clear it under, then export through the + // local reference. The reference keeps the encoder -- and the + // semaphore it owns -- alive across the driver call below without + // holding m_pendingMutex over a driver entry point (the capture + // path takes this lock). + VkSharedBaseObj encoder; + { + std::lock_guard lock(m_pendingMutex); + if (!m_initialized || !m_encoder) { + return VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED; + } + encoder = m_encoder; + } + VkSemaphore sem = encoder->GetCompletionTimelineSemaphore(); + if ((sem == VK_NULL_HANDLE) || + !encoder->IsCompletionSemaphoreExportable() || + (m_vkDevCtx.GetSemaphoreFdKHR == nullptr)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; + } + // vkGetSemaphoreFdKHR is portable Vulkan: no OS header, so this stays + // in the main TU (the section 3.2 portability constraint). The minted + // fd is the CALLER's -- the reverse direction from the registration + // rule -- to close or to hand to exactly one import, which consumes it. + VkSemaphoreGetFdInfoKHR getInfo{VK_STRUCTURE_TYPE_SEMAPHORE_GET_FD_INFO_KHR}; + getInfo.semaphore = sem; + getInfo.handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT; + int fd = -1; + if (m_vkDevCtx.GetSemaphoreFdKHR(m_vkDevCtx.getDevice(), &getInfo, + &fd) != VK_SUCCESS) { + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + *outHandle = (uint64_t)fd; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkResult VulkanVideoEncoderExtImpl::GetCompletionInfo( + VkVideoEncoderCompletionInfo* pInfo) +{ + if ((pInfo == nullptr) || + (pInfo->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + // Chain walk, the QueryImageSupport shape: the known diagnostic and + // filter structs are consumed; anything else is refused rather than + // ignored -- an extension the library does not understand means the + // caller asked for something it is not getting. A REPEATED known sType + // is refused too: two links of one type mean the caller believes it is + // getting two different things. + VkVideoEncoderDiagnosticInfo* diagnostic = nullptr; + VkVideoEncoderFilterInfo* filter = nullptr; + VkVideoEncoderInputResidencyInfo* residency = nullptr; + VkVideoEncoderStagedSubmitInfo* stagedSubmit = nullptr; + VkVideoEncoderImportGuardInfo* importGuard = nullptr; + VkVideoEncoderImportContentInfo* importContent = nullptr; + for (void* link = const_cast(pInfo->pNext); link != nullptr;) { + // The {sType, pNext} prefix is read through a SIBLING struct type; + // VK_ENC_PIN_CHAIN_PREFIX is what makes that sound, and `next` is + // taken before the link is re-cast to its own type. + auto* prefix = + reinterpret_cast(link); + const VkVideoEncoderStructureType linkType = prefix->sType; + const void* next = prefix->pNext; + if ((linkType == VK_VIDEO_ENCODER_STRUCTURE_TYPE_DIAGNOSTIC_INFO) && + (diagnostic == nullptr)) { + diagnostic = prefix; + } else if ((linkType == VK_VIDEO_ENCODER_STRUCTURE_TYPE_FILTER_INFO) && + (filter == nullptr)) { + filter = reinterpret_cast(link); + } else if ((linkType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_RESIDENCY_INFO) && + (residency == nullptr)) { + residency = + reinterpret_cast(link); + } else if ((linkType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_STAGED_SUBMIT_INFO) && + (stagedSubmit == nullptr)) { + stagedSubmit = + reinterpret_cast(link); + } else if ((linkType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO) && + (importGuard == nullptr)) { + importGuard = + reinterpret_cast(link); + } else if ((linkType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO) && + (importContent == nullptr)) { + importContent = + reinterpret_cast(link); + } else { + return VK_ERROR_INITIALIZATION_FAILED; + } + link = const_cast(next); + } + std::lock_guard lock(m_pendingMutex); + if (diagnostic != nullptr) { + diagnostic->diagnosticCount = m_diagnosticCount; + std::snprintf(diagnostic->lastDiagnostic, + sizeof(diagnostic->lastDiagnostic), "%s", + m_lastDiagnostic); + } + if (importGuard != nullptr) { + // The SESSION snapshot, not this thread's record: this call is + // class (c) -- any thread -- and the registration that produced the + // verdict ran on the submit thread. Answered from the snapshot a + // registration wrote and never recomputed, because "what the guard + // did" and "what the guard would do now" are exactly the two things + // a workaround report must not conflate. + importGuard->state = m_importGuardReport.state; + importGuard->requestedCount = m_importGuardReport.requestedCount; + importGuard->retainedCount = m_importGuardReport.retainedCount; + importGuard->failureStatus = m_importGuardReport.failureStatus; + importGuard->failureErrno = m_importGuardReport.failureErrno; + } + if (importContent != nullptr) { + // THE VERDICT CHANNEL. Answered from the probe object, which is + // reachable under m_pendingMutex -- already held here -- for the + // same reason m_encoder is read under it below: it is the lock the + // teardown path clears these under. + // + // probeGeneration on EVERY path including the no-probe one, which is + // the whole point of it: a caller must be able to tell "this library + // has the observable and has nothing to report" from "this library + // predates the observable", and every other field's empty value is + // 0, which is also what an untouched struct holds. + importContent->probeGeneration = + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION; + if (m_contentProbe) { + VkVideoEncoderContentProbe::Verdict verdict; + uint32_t probed = 0; + uint32_t damaged = 0; + uint32_t armed = 0; + m_contentProbe->GetSnapshot(verdict, probed, damaged, armed); + importContent->state = VkEncMapContentState(verdict.state); + importContent->resource = verdict.registrationId; + importContent->meanY = verdict.meanY; + importContent->meanU = verdict.meanU; + importContent->meanV = verdict.meanV; + importContent->probedRegistrationCount = probed; + importContent->damagedRegistrationCount = damaged; + importContent->armedRegistrationCount = armed; + } else { + // Nobody ever chained the struct onto a registration, so nothing + // was ever armed. NOT_EVALUATED, and the zeros mean it. + importContent->state = + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED; + importContent->resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + importContent->meanY = 0; + importContent->meanU = 0; + importContent->meanV = 0; + importContent->probedRegistrationCount = 0; + importContent->damagedRegistrationCount = 0; + importContent->armedRegistrationCount = 0; + } + } + if (filter != nullptr) { + // Answered from the ENCODER OBJECT, never from m_encoderConfig: + // "requested" and "happened" are precisely the two things this + // channel exists to tell apart, so reading the config here would + // defeat its only purpose. + // + // m_encoder is read under m_pendingMutex because that is the lock + // Flush()/teardown clears it under. A torn-down or null-backend + // session reports an honest zeroed snapshot rather than failing: + // the call is a status query and has no other failure mode. + filter->filterCreated = VK_FALSE; + filter->filterType = VK_VIDEO_ENCODER_FILTER_TYPE_NONE; + filter->filterDispatchCount = 0; + filter->stagedCopyCount = 0; + if (m_encoder) { + filter->filterCreated = + m_encoder->HasInputComputeFilter() ? VK_TRUE : VK_FALSE; + filter->filterType = static_cast( + m_encoder->GetInputFilterKind()); + filter->filterDispatchCount = + m_encoder->GetInputFilterDispatchCount(); + filter->stagedCopyCount = m_encoder->GetStagedCopyCount(); + } + } + if (residency != nullptr) { + // Same rule as the filter block above: answered from the ENCODER + // OBJECT, never from the registration slots. What a caller DECLARED + // is already knowable to it -- it wrote the descriptor -- and the + // one thing it cannot see is what the library then DID with the + // declaration, which is the entire purpose of this channel. Reading + // it back off the slot would answer the question the caller already + // knows the answer to and would stay green through a regression that + // discarded the declaration. + // + // Zeroed first, then filled under the same `if (m_encoder)` and the + // same m_pendingMutex the filter block uses, so a torn-down session + // reports an honest snapshot instead of failing a status query. + residency->foreignAcquireCount = 0; + residency->localAcquireCount = 0; + if (m_encoder) { + residency->foreignAcquireCount = + m_encoder->GetForeignAcquireCount(); + residency->localAcquireCount = + m_encoder->GetLocalAcquireCount(); + } + } + if (stagedSubmit != nullptr) { + // 0 / VK_QUEUE_FAMILY_IGNORED is the honest "no session" answer: a + // torn-down encoder has no staged-input queue, and 0 is not a legal + // VK_QUEUE_* bit so it cannot be mistaken for one. + // + // Read through the SAME two accessors StageInputFrame and + // SubmitStagedInputFrame read, never re-derived here from + // ComputeFilterActive(). Re-deriving it is exactly how the barrier + // site and the submit site once came to be able to disagree, and a + // reporting channel that re-derives its answer cannot witness the one + // class of bug those two accessors exist to prevent. + stagedSubmit->submitTypeQueueFlags = 0; + stagedSubmit->queueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + if (m_encoder) { + stagedSubmit->submitTypeQueueFlags = + m_encoder->GetStagedInputSubmitTypeFlag(); + stagedSubmit->queueFamilyIndex = + m_encoder->GetStagedInputSubmitQueueFamilyIdx(); + } + } + pInfo->completionCounter = m_completionCounter.load(std::memory_order_acquire); + pInfo->framesTimedOut = m_framesTimedOut; + pInfo->lateCaptures = m_lateCaptures; + pInfo->framesCancelled = m_framesCancelled; + uint32_t pending = 0, ready = 0, acquired = 0; + for (const auto& p : m_pendingFrames) { + if (p.acquired) { + acquired++; + } else if (p.hasCapture) { + ready++; + } else { + pending++; + } + } + pInfo->framesPending = pending; + pInfo->framesReady = ready; + pInfo->framesAcquired = acquired; + return VK_SUCCESS; +} + +// Convert an unacquired frame to a 0-byte cancelled drop. Caller holds +// m_pendingMutex. Reuses the timed-out late-capture discard machinery. +void VulkanVideoEncoderExtImpl::CancelFrameLocked(PendingFrame& frame) +{ + frame.bytes.clear(); + frame.isIdr = false; + frame.status = VK_INCOMPLETE; // documented as cancelled + MarkFrameReadyLocked(frame); + frame.timedOut = true; // a late capture is discarded, not delivered + m_framesCancelled++; +} + +VkResult VulkanVideoEncoderExtImpl::CancelFrame(uint64_t frameId) +{ + std::lock_guard lock(m_pendingMutex); + for (auto& p : m_pendingFrames) { + if (p.frameId != frameId) { + continue; + } + if (p.acquired) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + CancelFrameLocked(p); + return VK_SUCCESS; + } + return VK_ERROR_UNKNOWN; +} + +VkResult VulkanVideoEncoderExtImpl::AbandonAllFrames( + uint32_t* pAbandonedCount) +{ + std::vector orphaned; + uint32_t abandoned = 0; + { + std::lock_guard lock(m_pendingMutex); + for (auto it = m_pendingFrames.begin(); it != m_pendingFrames.end();) { + if (it->acquired) { + // Out in the caller's hands; see the header. + ++it; + continue; + } + // Collect rather than release here: ReleaseResourceReference + // takes m_resourceMutex, and releasing after m_pendingMutex + // drops keeps these two locks strictly un-nested. The existing + // release path notes this is the cleaner shape; a new one has no + // reason to inherit the old one's coupling. + if (it->resource != VK_VIDEO_ENCODER_RESOURCE_NULL) { + orphaned.push_back(it->resource); + } + const uint64_t frameId = it->frameId; + if (!it->hasCapture && + (m_encoder || (m_nullBackend != nullptr))) { + // Same rule as ReleaseEncodedFrame: abandoned while still + // PENDING means the completion record is still coming and + // will pop unmatched. Abandonment disclaimed it; it is not + // the late class. + m_releasedWhilePending.insert(frameId); + } + for (auto r = m_readyOrder.begin(); r != m_readyOrder.end();) { + r = (*r == frameId) ? m_readyOrder.erase(r) : r + 1; + } + // An abandoned frame's submission may still be executing, and + // vkDestroySemaphore requires every batch referring to the + // semaphore to have completed. Nothing here proves that, so the + // release-fence semaphore waits for the teardown wait-idle + // rather than being destroyed on a guess. + if (it->releaseFenceSemaphore != VK_NULL_HANDLE) { + m_unprovenSemaphores.push_back(it->releaseFenceSemaphore); + } + // The acquire semaphores this frame's submission waits on have + // the same unproven status for the same reason. + m_unprovenSemaphores.insert(m_unprovenSemaphores.end(), + it->acquireFenceSemaphores.begin(), + it->acquireFenceSemaphores.end()); + it = m_pendingFrames.erase(it); + abandoned++; + } + } + for (VkVideoEncoderResource resource : orphaned) { + ReleaseResourceReference(resource); + } + if (abandoned > 0) { + VkEncErr() << "[EncoderExt] abandoned " << abandoned + << " undelivered frame(s) and released their registration " + "claims" << std::endl; + } + if (pAbandonedCount != nullptr) { + *pAbandonedCount = abandoned; + } + return VK_SUCCESS; +} + +VkResult VulkanVideoEncoderExtImpl::AcquireNextEncodedFrame( + VkVideoEncodeResult& result) +{ + // Structure-type gate: the caller passes a value-initialized (and + // therefore self-stamped) result struct. A chained struct is version + // skew -- nothing chains onto the result today -- and is refused + // rather than skipped, the rule every public pNext position applies. + if ((result.sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_ENCODE_RESULT) || + (result.pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + std::lock_guard lock(m_pendingMutex); + DrainCapturesLocked(); + // Completion order (design 3.4, first half): deliver from the ready + // queue, ordered by when each frame's outcome became deliverable rather + // than by when it was submitted. A frame that never completes therefore + // cannot park at the head and strand the completed frames behind it -- + // head-of-line blocking is structurally impossible here, not merely + // bounded by the deadline. + // + // On a healthy stream this is indistinguishable from the submit order it + // replaces: with no GOP reorder the captures arrive in submit order, so + // delivery stays monotonic frame for frame. The two orders diverge only + // when a frame stalls, which is the case the design is about. + if (TryPopReadyLocked(result)) { + return VK_SUCCESS; + } + // Nothing ready: apply the deadline, 3.4's second half. Undelivered + // frames are checked oldest-first and any past its deadline becomes a + // 0-byte VK_TIMEOUT drop, so a consumer that would otherwise wait + // forever makes progress. Synthesis enqueues onto the ready queue, so + // the retry below delivers it within this same call. + for (auto& p : m_pendingFrames) { + if (p.acquired || p.hasCapture) { + continue; + } + SynthesizeTimeoutLocked(p); + } + if (TryPopReadyLocked(result)) { + return VK_SUCCESS; + } + return VK_NOT_READY; +} + +VkResult VulkanVideoEncoderExtImpl::AcquireEncodedFrame( + uint64_t frameId, VkVideoEncodeResult& result) +{ + if ((result.sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_ENCODE_RESULT) || + (result.pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + std::lock_guard lock(m_pendingMutex); + DrainCapturesLocked(); + for (auto& p : m_pendingFrames) { + if (p.frameId != frameId) { + continue; + } + if (p.acquired) { + return VK_ERROR_NOT_PERMITTED_KHR; // re-acquire before Release + } + if (!p.hasCapture && !SynthesizeTimeoutLocked(p)) { + return VK_NOT_READY; + } + FillResultLocked(p, result); + return VK_SUCCESS; + } + return VK_ERROR_UNKNOWN; // never submitted, or already released +} + +VkVideoEncoderFrameState VulkanVideoEncoderExtImpl::GetFrameStatus( + uint64_t frameId) +{ + std::lock_guard lock(m_pendingMutex); + DrainCapturesLocked(); + for (auto& p : m_pendingFrames) { + if (p.frameId != frameId) { + continue; + } + if (p.acquired) { + return VK_VIDEO_ENCODER_FRAME_STATE_ACQUIRED; + } + if (p.hasCapture) { + return VK_VIDEO_ENCODER_FRAME_STATE_READY; + } + return VK_VIDEO_ENCODER_FRAME_STATE_PENDING; + } + return VK_VIDEO_ENCODER_FRAME_STATE_UNKNOWN; +} + +VkResult VulkanVideoEncoderExtImpl::GetEncodedFrame(VkVideoEncodeResult& result) +{ + // Alias of AcquireNextEncodedFrame(), kept so existing FIFO drain loops + // migrate unchanged. + return AcquireNextEncodedFrame(result); +} + +void VulkanVideoEncoderExtImpl::ReleaseEncodedFrame(uint64_t frameId) +{ + std::lock_guard lock(m_pendingMutex); + for (auto it = m_pendingFrames.begin(); it != m_pendingFrames.end(); ++it) { + if (it->frameId == frameId) { + // Release the encodeFrameInfo ref - returns resources to pools + it->encodeFrameInfo = nullptr; + const VkVideoEncoderResource resource = it->resource; + if (!it->hasCapture && + (m_encoder || (m_nullBackend != nullptr))) { + // Released while still PENDING -- no capture, no deadline + // drop, no cancellation has landed on this entry -- which + // the header documents as legal. The frame's completion + // record is still on its way and will pop with no entry to + // match; record the id so DrainCapturesLocked discards + // that pop silently instead of counting it late. Skipped + // when no pop can ever arrive (after Flush() the capture + // FIFO died with m_encoder); a null-backend session gets + // its capture source lazily from VkEncPushCapture, so its + // records are kept even before the source exists. + m_releasedWhilePending.insert(frameId); + } + // Per-frame fence disposal, and the one path that can PROVE the + // destroy precondition. The proof has THREE terms, not two. + // + // hasCapture a capture is published only after + // ReadbackBitstreamData's fence wait, and the + // encode submit waits on the staging submit, so + // both submissions that could touch these + // semaphores have completed; + // !timedOut a deadline-synthesized drop carries hasCapture + // with no such wait behind it; + // VK_SUCCESS and neither does a FAILED capture. VkVideoEncoder + // publishes a record whose status IS the failure + // -- AssemblyWorkerThread pushes a CapturedBitstream + // carrying the VkResult that + // SyncHostOnCmdBuffComplete returned -- so a fence + // wait that returned VK_TIMEOUT or + // VK_ERROR_DEVICE_LOST arrives here as + // hasCapture=true, timedOut=false, and the batch it + // failed to wait for may still be executing. Reading + // only the first two terms destroyed a semaphore a + // live batch still referenced + // (VUID-vkDestroySemaphore-semaphore-01137) on + // precisely the frames where the GPU was in trouble. + // + // Everything else goes to the unproven graveyard, which + // manufactures the precondition with a DeviceWaitIdle instead of + // asserting it. Residual, named rather than papered over: + // ReadbackBitstreamData's `readbackDone = false; return VK_SUCCESS` + // early exit (VkVideoEncoder.cpp) publishes a VK_SUCCESS record + // with no fence wait at all, on a frame whose node carried no + // output buffer or no encode command buffer. Status cannot see + // that leg; closing it needs a completion signal the encoder does + // not currently expose. + const bool completionProved = + it->hasCapture && !it->timedOut && (it->status == VK_SUCCESS); + if (it->releaseFenceSemaphore != VK_NULL_HANDLE) { + if (completionProved) { + m_releaseFenceRetired.push_back(it->releaseFenceSemaphore); + } else { + m_unprovenSemaphores.push_back( + it->releaseFenceSemaphore); + } + } + // The acquire half rides the same proof: the submission that + // WAITS on an acquire semaphore is the submission that signals + // the release fence. + if (!it->acquireFenceSemaphores.empty()) { + std::vector& sink = + completionProved ? m_releaseFenceRetired + : m_unprovenSemaphores; + sink.insert(sink.end(), it->acquireFenceSemaphores.begin(), + it->acquireFenceSemaphores.end()); + } + m_pendingFrames.erase(it); + // Prune the ready queue too. Lazy skipping alone would be + // correct but not bounded: a consumer that only ever uses the + // keyed AcquireEncodedFrame never pops, so stale ids would + // accumulate for the life of the session. + for (auto r = m_readyOrder.begin(); r != m_readyOrder.end(); ) { + if (*r == frameId) { + r = m_readyOrder.erase(r); + } else { + ++r; + } + } + if (resource != VK_VIDEO_ENCODER_RESOURCE_NULL) { + // Drop this frame's claim on its registration. Taking + // m_resourceMutex after releasing m_pendingMutex would be + // cleaner still, but the two are never held in the opposite + // order anywhere, so this cannot deadlock. + ReleaseResourceReference(resource); + } + return; + } + } + // Documented no-op: already released or never submitted. Logged AND + // recorded: the stderr line dies under silenceStdio (the shipping + // Chromium configuration), so the diagnostic channel carries the + // same text to GetCompletionInfo, where the consumer can put it in + // its own logging. m_pendingMutex is held for the whole function. + m_diagnosticCount++; + std::snprintf(m_lastDiagnostic, sizeof(m_lastDiagnostic), + "ReleaseEncodedFrame(%llu): unknown frame id -- no-op " + "(double release, or an id never submitted)", + static_cast(frameId)); + VkEncErr() << "[EncoderExt] " << m_lastDiagnostic << std::endl; +} + +void VulkanVideoEncoderExtImpl::GetShutdownInfo( + VkVideoEncoderShutdownInfo* pInfo) const +{ + if (pInfo == nullptr) { + return; + } + std::lock_guard lock(m_shutdownMutex); + *pInfo = m_shutdown; +} + +// Latch a device loss wherever it is first seen. A wait that later returns +// VK_SUCCESS does not mean the device recovered -- it means there is nothing +// left to wait for -- and it must not be allowed to erase this. +VkResult VulkanVideoEncoderExtImpl::NoteDeviceResult(VkResult result) +{ + if (result == VK_ERROR_DEVICE_LOST) { + std::lock_guard lock(m_shutdownMutex); + m_shutdown.deviceLostObserved = VK_TRUE; + if (m_shutdown.firstError == VK_SUCCESS) { + m_shutdown.firstError = result; + } + } + return result; +} + +// Caller holds m_shutdownMutex. +VkVideoEncoderShutdownDisposition +VulkanVideoEncoderExtImpl::ClassifyShutdownLocked() const +{ + if (m_shutdown.workersJoined == VK_FALSE) { + // A worker that did not join may still be submitting. + return VK_VIDEO_ENCODER_SHUTDOWN_UNPROVEN; + } + if (m_shutdown.deviceLostObserved != VK_FALSE) { + // On a lost device VK_SUCCESS and VK_ERROR_DEVICE_LOST say the same + // thing about pending use -- there is none -- and neither says + // anything about what the shared contents now hold. + if ((m_shutdown.lastDeviceWait == VK_SUCCESS) || + (m_shutdown.lastDeviceWait == VK_ERROR_DEVICE_LOST)) { + return VK_VIDEO_ENCODER_SHUTDOWN_LOST_DEVICE_RETIRED; + } + return VK_VIDEO_ENCODER_SHUTDOWN_UNPROVEN; + } + if (m_shutdown.lastDeviceWait == VK_SUCCESS) { + return VK_VIDEO_ENCODER_SHUTDOWN_IDLE; + } + return VK_VIDEO_ENCODER_SHUTDOWN_UNPROVEN; +} + +// Caller holds m_shutdownMutex. +VkResult VulkanVideoEncoderExtImpl::WaitWholeDeviceLocked() +{ + // This runs even when the join or an earlier encode failed. It is the only + // step that covers staging work whose dependent encode never submitted -- + // queued work still bound to the producer's image, which no host thread is + // waiting on and which WaitForThreadsToComplete() therefore cannot see. + const VkResult waited = m_vkDevCtx.DeviceWaitIdle(); + m_shutdown.lastDeviceWait = waited; + if (waited == VK_ERROR_DEVICE_LOST) { + m_shutdown.deviceLostObserved = VK_TRUE; + } + if ((waited != VK_SUCCESS) && (m_shutdown.firstError == VK_SUCCESS)) { + m_shutdown.firstError = waited; + } + m_shutdown.disposition = ClassifyShutdownLocked(); + return waited; +} + +VkResult VulkanVideoEncoderExtImpl::Flush() +{ + if (IsInCompletionCallback()) { + return VK_ERROR_NOT_PERMITTED_KHR; // class (a) from the callback + } + + // A repeated Finish is serialized here rather than racing itself. It never + // drains twice, never restarts workers, and returns the sticky result -- + // except that it DOES retry a terminal wait left Unproven, because that is + // the one step whose answer can still change. + std::lock_guard shutdownLock(m_shutdownMutex); + + if (m_shutdown.shutdownComplete != VK_FALSE) { + return m_shutdown.firstError; + } + if (m_shutdown.workersJoined == VK_FALSE && (!m_initialized || !m_encoder)) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + + if (m_shutdown.workersJoined == VK_FALSE) { + // Stop acceptance once. Submissions arriving from here on are refused + // rather than joining a pipeline that is being torn down. + m_shutdownStarted.store(true, std::memory_order_release); + + // Detach the completion callback before the workers go. The detach is + // itself a quiesce point and it hands the caller's cookie back to its + // release, which must not be left owned by an encoder about to be + // retired. No lock is held across it. + const VkResult detached = + SetCompletionCallback(nullptr, nullptr, nullptr); + m_shutdown.callbackDetached = + (detached == VK_SUCCESS) ? VK_TRUE : VK_FALSE; + + // Wait for all pending encodes to complete. The bool is the only + // signal the core offers; a false is recorded as an unclassified + // shutdown error rather than reconstructed into a VkResult that + // claims to know more than it does. + const bool joined = m_encoder->WaitForThreadsToComplete(); + m_shutdown.workersJoined = joined ? VK_TRUE : VK_FALSE; + if (!joined && (m_shutdown.firstError == VK_SUCCESS)) { + m_shutdown.firstError = VK_ERROR_UNKNOWN; + } + + // Drain the pending queue + { + std::lock_guard lock(m_pendingMutex); + for (auto& pending : m_pendingFrames) { + if (pending.encodeFrameInfo && pending.encodeFrameInfo->encodeCmdBuffer) { + pending.encodeFrameInfo->encodeCmdBuffer->ResetCommandBuffer( + true, "EncoderExtFlush"); + } + } + } + } + + const VkResult waited = WaitWholeDeviceLocked(); + + if (m_shutdown.disposition == VK_VIDEO_ENCODER_SHUTDOWN_UNPROVEN) { + // Do NOT release the core here. Nothing has established that the + // device stopped reading the images, imported semaphores and + // registrations it owns, and destroying them on an unproven wait is + // precisely the use-after-free the wait exists to rule out. The + // encoder, its resources and every producer dependency stay alive; a + // later Flush retries the wait. + return (m_shutdown.firstError != VK_SUCCESS) ? m_shutdown.firstError + : waited; + } + + // Release the encoder — the destructor calls DeinitEncoder() which + // writes any buffered bitstream (deferred frames from GOP reordering) + // and closes the output file. + // + // The pointer is cleared under m_pendingMutex because DrainCapturesLocked + // tests it and then calls through it while holding that mutex; clearing + // it outside let the last reference drop after a reader had already + // passed its null check. The reference is handed to |doomed| so the + // destructor still runs OUTSIDE the lock -- it joins library threads, and + // those threads take m_pendingMutex on the capture path. + VkSharedBaseObj doomed; + { + std::lock_guard lock(m_pendingMutex); + doomed = m_encoder; + m_encoder = nullptr; + // The released-while-pending records are consumed by capture pops, + // and the capture FIFO dies with the encoder released below; + // nothing can consume them past this point. Dropped so ids from a + // finished session cannot swallow a genuine late capture after a + // re-init (frame ids are consumer-chosen and may recur). + m_releasedWhilePending.clear(); + } + doomed = nullptr; + + m_shutdown.shutdownComplete = VK_TRUE; + return m_shutdown.firstError; +} + +VkResult VulkanVideoEncoderExtImpl::DrainPendingFrames() +{ + if (IsInCompletionCallback()) { + return VK_ERROR_NOT_PERMITTED_KHR; // class (a) from the callback + } + // Non-terminal drain. Unlike Flush() above, this does NOT null m_encoder: + // the drain pushes the deferred GOP-reorder tail (PushOrderedFrames()) + // and joins the encoder-queue + assembly threads, so every frame + // submitted so far is encoded and its capture is retrievable via + // GetEncodedFrame(). The encoder stays usable for retrieval / Release. + // + // AND FOR SUBMITTING AGAIN, which is the half a plain wait does not cover. + // WaitForThreadsToComplete() alone leaves m_asyncAssemblyEnabled false + // and m_assemblyThreads empty, and nothing outside InitEncoder ever set + // them again; ProcessOrderedFrames then took its synchronous fallback, + // which publishes no CapturedBitstream. Every frame submitted after the + // first drain was still encoded correctly -- the bytes reached the file + // under file output -- and NONE of them ever raised a completion edge or + // became acquirable, so each held its PendingFrame and its resource + // registration for the rest of the session. DrainAndRestartThreads() + // brings the assembly workers back up, which is what makes the word + // "non-terminal" true of the completion surface and not just of the + // encoder object. + if (!m_initialized || !m_encoder) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + if (!m_encoder->DrainAndRestartThreads()) { + // The drain itself happened; only the restart failed, so everything + // submitted before this call is still complete and retrievable. What + // is gone is the completion surface for anything submitted after -- + // so say so here rather than let the caller discover it as frames + // that never come back. Submits from here on are refused by + // ProcessOrderedFrames' synchronous-fallback guard. + VkEncErr() << "[EncoderExt] DrainPendingFrames: the assembly workers " + "could not be restarted; frames already submitted are " + "complete, but this session can no longer report " + "completions" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + return VK_SUCCESS; +} + +VkResult VulkanVideoEncoderExtImpl::Reconfigure(const VkVideoEncoderConfig& config) +{ + if (IsInCompletionCallback()) { + return VK_ERROR_NOT_PERMITTED_KHR; // class (a) from the callback + } + // Structure-type gate (see the header's versioning rules). + if ((config.sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG) || + (config.pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + // MVP scope: MID-STREAM RATE-CONTROL update only -- + // average/max bitrate and frame rate, folded in on the encoder thread + // and carried by the next ENCODE_RATE_CONTROL control command. + // Resolution, codec, profile and rate-control-MODE changes still + // require a session re-init. + if (!m_initialized || !m_encoder) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + // Everything this call cannot carry is REFUSED rather than discarded. + // Returning VK_SUCCESS for a field that was ignored is how a stream ends + // up encoded one way and described another -- for colour, that is an HDR + // frame signalled with the previous VUI, which nothing downstream can + // detect. The sequence header is written once at init; changing any of + // these needs a re-init until Reconfigure can rewrite it. The input + // declaration is refused for a different reason: it is not in the + // sequence header at all, it is what the session's conversion was built + // around, and it is equally unable to change under a live encoder. + // + // inputColorModel is the OTHER half of that declaration, and it is + // compared AS IT RESOLVES against the format rather than as it is + // spelled. FROM_FORMAT and the model the format already carries name the + // same input and route identically, and the session keeps no record of + // which spelling arrived, so refusing a re-spelling would refuse a call + // that changes nothing. What the resolved comparison catches is the + // re-declaration an enumerant can absorb: R8G8B8A8_UNORM resolves to + // R'G'B' undeclared and to Y'CbCr -- AYUV -- under + // VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, and those take opposite arms of the + // preprocess filter. Accepting that would answer VK_SUCCESS while the + // session went on converting as it was built to, which is the wrong + // picture in every pixel ComputeFilterTakesFormat exists to prevent. + const VkVideoEncoderColorModel initColorModel = VkEncResolveColorModel( + m_initConfig.inputFormat, m_initConfig.inputColorModel); + const VkVideoEncoderColorModel newColorModel = + VkEncResolveColorModel(config.inputFormat, config.inputColorModel); + struct Immutable { + const char* name; + bool changed; + }; + const Immutable immutables[] = { + {"codec", config.codec != m_initConfig.codec}, + {"profile", config.profile != m_initConfig.profile}, + {"encodeWidth", config.encodeWidth != m_initConfig.encodeWidth}, + {"encodeHeight", config.encodeHeight != m_initConfig.encodeHeight}, + {"inputFormat", config.inputFormat != m_initConfig.inputFormat}, + {"inputColorModel", newColorModel != initColorModel}, + {"inputWidth", config.inputWidth != m_initConfig.inputWidth}, + {"inputHeight", config.inputHeight != m_initConfig.inputHeight}, + {"rateControlMode", config.rateControlMode != m_initConfig.rateControlMode}, + {"colourPrimaries", + config.colourPrimaries != m_initConfig.colourPrimaries}, + {"transferCharacteristics", + config.transferCharacteristics != m_initConfig.transferCharacteristics}, + {"inputTransferCharacteristics", + config.inputTransferCharacteristics != + m_initConfig.inputTransferCharacteristics}, + {"matrixCoefficients", + config.matrixCoefficients != m_initConfig.matrixCoefficients}, + {"videoFullRange", config.videoFullRange != m_initConfig.videoFullRange}, + }; + for (const auto& field : immutables) { + if (field.changed) { + VkEncErr() << "[EncoderExt] Reconfigure cannot change '" + << field.name << "' mid-stream: it is settled at " + "InitializeExt -- in the sequence header written " + "once there, or in the input routing the session " + "was built around. Re-initialize the session " + "instead." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + + // THE ENCODING LEVERS -- the same contract, applied to the fields that + // reach the BITSTREAM rather than the sequence header. Each one below is + // settled at InitializeExt and can change how the stream is encoded: the + // rate-control levers (vbvBufferSize, minQp, maxQp), + // the GOP structure the session sequences to (gopLength, + // consecutiveBFrames, idrPeriod, closedGop) and the encode-quality + // controls (qualityLevel, tuningMode). NONE of them is carried by the + // ENCODE_RATE_CONTROL command this call emits -- that command carries + // averageBitrate, maxBitrate and the frame rate, and nothing else -- so + // answering VK_SUCCESS to a change would leave the encoder using the + // value it was built with while the caller believed otherwise. That is + // the stream-encoded-one-way-and-described-another shape the comment + // above forbids, so they are REFUSED. + // + // Compared against the init config and refused only on a CHANGE, exactly + // as the group above is. That is what keeps a caller working when it hands + // back a WHOLE config rather than a minimal one -- whether it retains a + // baseline and edits the rate fields in place, or copies the init config + // and edits a couple. Neither shape touches a field below, so neither + // sees a new refusal. + // + // constQpI/P/B ARE NOT HERE, because they are APPLIED rather than + // refused -- see the RequestRateControlUpdate call below. They are the + // one lever a DISABLED (constant-QP) session actually has: the four + // fields this call already carried land in m_rateControlLayersInfo, which + // VkVideoEncoder drops entirely (pLayers null, layerCount 0) whenever the + // mode is DISABLED, and the mode is itself immutable -- so before that + // change Reconfigure answered VK_SUCCESS while changing nothing whatever + // on exactly the sessions with the least other recourse. Applying is + // cheap because the per-frame path already exists: EncodeFrameCommon + // copies m_encoderConfig->constQp into every frame unconditionally, so + // updating the config on the encoder thread is the whole of it. + // + // minQp AND maxQp ARE NOT HERE EITHER, for a reason that corrects the + // note this replaces. They do reach the driver only through the + // CODEC-SPECIFIC rate-control structs; what is not true is that those + // structs are welded to session-parameter creation. The fill that + // produces them, EncoderConfig::GetRateControlParameters, is a pure + // function of config state -- VkEncBuildAndProbeConfig in this same file + // already calls it a second time, on a fresh config, with no device + // anywhere -- and CodecHandleRateControlCmd copies its output into the + // frame and chains it onto EVERY ENCODE_RATE_CONTROL command, not just + // the first. So the fill can be re-invoked on the encoder thread and the + // result rides the very command this call already causes. That + // re-invocation is VkVideoEncoder::RefreshCodecRateControlParameters, + // and the three cases where it would carry nothing are refused below + // rather than answered VK_SUCCESS. + // + // WHY vbvBufferSize IS STILL REFUSED, which is not the same reason. It + // is not an independent input to that fill. What the command carries is + // virtualBufferSizeInMs, computed as vbvBufferSize * 1000 / hrdBitrate + // and paired with initialVirtualBufferSizeInMs, computed the same way + // from vbvInitialDelay -- and vbvInitialDelay is DERIVED FROM THE OLD + // vbvBufferSize, once, inside the codec InitRateControl finalize step + // that this call does not re-run. Both divide by the config hrdBitrate, + // which a bitrate change deliberately does not update: the new bitrate + // lands on the rate-control LAYER, leaving the config holding the + // bitrate the session was built with. Applying vbvBufferSize alone + // would therefore emit a CPB whose initial fullness was computed for a + // different buffer size, against a bitrate the session is no longer + // using -- and on a shrink it can put the initial fullness ABOVE the + // buffer size, which is not a state to hand a driver. Making it correct + // means folding the bitrate into the config and re-running the codec + // finalize, which changes the semantics of the already-shipped bitrate + // path. That is the separate change; refusing is the honest answer + // until it is made. + // + // gopLength, idrPeriod and consecutiveBFrames additionally drive + // gopStructure, which sequences frame types and DPB references -- + // updating the rate-control copy alone would tell the driver one GOP + // while the encoder sequenced another, which is worse than refusing. + // The refresh above would in fact carry them, which is exactly why they + // must not become mutable without the sequencing half of the change. + // qualityLevel is baked into the video session parameters at creation: + // Vulkan does have a coding-control bit for it + // (VK_VIDEO_CODING_CONTROL_ENCODE_QUALITY_LEVEL_BIT_KHR, which + // HandleCtrlCmd already emits at session start), but the session + // parameters are CREATED against a quality level, so moving it + // mid-stream means recreating them as well -- noted, not attempted. + // + // WHAT IS DELIBERATELY NOT IN EITHER GROUP. The session-creation inputs + // (deviceId, gpuUUID, externalInstance, externalPhysicalDevice, + // externalDevice and the two external queue-family indices) and the + // diagnostic ones (outputPath, verbose, validate, disableFileOutput, + // silenceStdio) remain accepted and unread. Not one of them can change + // an encoded bit, so not one can misdescribe the stream -- the rationale + // above does not reach them. Refusing them would gain no correctness and + // would break a caller that builds a fresh minimal config for the + // reconfigure instead of copying its stored one. + const Immutable levers[] = { + {"vbvBufferSize", config.vbvBufferSize != m_initConfig.vbvBufferSize}, + {"gopLength", config.gopLength != m_initConfig.gopLength}, + {"consecutiveBFrames", + config.consecutiveBFrames != m_initConfig.consecutiveBFrames}, + {"idrPeriod", config.idrPeriod != m_initConfig.idrPeriod}, + {"closedGop", config.closedGop != m_initConfig.closedGop}, + {"qualityLevel", config.qualityLevel != m_initConfig.qualityLevel}, + {"tuningMode", config.tuningMode != m_initConfig.tuningMode}, + }; + for (const auto& field : levers) { + if (field.changed) { + VkEncErr() << "[EncoderExt] Reconfigure cannot change '" + << field.name << "' mid-stream: this call carries " + "averageBitrate, maxBitrate, frameRateNum, " + "frameRateDen and the constQp defaults, and " + "nothing else, so the encoder would go on using " + "the value it was initialized with. Re-initialize " + "the session instead." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + + // THE CONSTANT QUANTIZERS, on the same range rule and for the same reason + // InitializeExt applies it -- so a value refused at init cannot arrive + // here instead. A refusal that guarded only one of the two entry points + // would be worse than none, because it would teach the caller that the + // value is validated. + // + // Judged against the codec the SESSION WAS INITIALIZED AS. config.codec + // is refused above on a change, so on every call that reaches this line + // the two agree; the record is the one that stays right if that ever + // stops being true. + // + // Checked on the value as PASSED and not on a change, unlike the clamps + // below: a negative member is "not named" and is skipped by the range + // check exactly as it is skipped by the application below, and every + // non-negative one is applied whether or not it equals the record. + const VkResult constQpRangeResult = VkEncValidateConstQpRange( + config.constQpI, config.constQpP, config.constQpB, m_initConfig.codec, + "Reconfigure "); + if (constQpRangeResult != VK_SUCCESS) { + return constQpRangeResult; + } + + // THE THREE PLACES A QP CLAMP CHANGE STILL HAS TO BE REFUSED, because + // on each of them the codec fill would run and carry nothing -- which + // is the accepted-and-ignored shape the contract at the top of this + // function forbids, not a lesser version of it. + // + // Only checked ON A CHANGE, exactly as every group above is, so a + // caller handing back its stored config is unaffected. The clamps are + // carried LITERALLY: a zero means "no clamp", which is the reading + // InitializeExt already gives an explicit zero, so a clamp set here can + // also be cleared here. Anything else would be a second contract for + // the same two fields. + const bool qpClampChanged = (config.minQp != m_initConfig.minQp) || + (config.maxQp != m_initConfig.maxQp); + if (qpClampChanged) { + if (m_initConfig.codec == + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) { + // EncoderConfigAV1::GetRateControlParameters reads minQIndex + // and maxQIndex -- derived from the DEVICE capability limits -- + // and never reads these QP-unit fields at all. InitializeExt + // rejects a non-zero clamp on an AV1 session rather than + // ignoring it; this is that rule, at the same strength, on the + // mid-stream path. + VkEncErr() << "[EncoderExt] Reconfigure cannot change " + "minQp/maxQp on an AV1 session: these are H.26x " + "QP-unit clamps and AV1 rate control is " + "quantizer-index based, so the value would reach " + "nothing. Leave both as they were." << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if (m_initConfig.rateControlMode == + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR) { + // The DISABLED arm of the H.26x fill sets the codec clamp from + // the quality-level constant QP and ignores the caller request + // entirely, so a clamp change on a constant-QP session is + // ignored BY CONSTRUCTION however the update is delivered. + // constQpI/P/B are that session's lever, and they are applied. + VkEncErr() << "[EncoderExt] Reconfigure cannot change " + "minQp/maxQp on a constant-QP (DISABLED) session: " + "that mode takes its quantizer from constQpI/P/B, " + "which this call does carry, and ignores the " + "clamps. Change the constQp values instead." + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + // The same syntactic range and the same inverted-window rule + // InitializeExt applies, so a value refused at init cannot arrive + // here instead. The DEVICE QP window is checked one layer down, in + // RequestRateControlUpdate, which is where the window this session + // recorded at codec-init lives. + if ((config.minQp < 0) || (config.minQp > 51) || + (config.maxQp < 0) || (config.maxQp > 51)) { + VkEncErr() << "[EncoderExt] Reconfigure minQp/maxQp outside the " + "H.26x QP range 0..51 (minQp=" << config.minQp + << ", maxQp=" << config.maxQp << ")" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + if ((config.minQp > 0) && (config.maxQp > 0) && + (config.minQp > config.maxQp)) { + VkEncErr() << "[EncoderExt] Reconfigure minQp " << config.minQp + << " > maxQp " << config.maxQp + << " -- inverted clamp window" << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + } + + const uint64_t averageBitrate = config.averageBitrate; + const uint64_t maxBitrate = + (config.maxBitrate != 0) ? config.maxBitrate : config.averageBitrate; + // The constant-QP defaults ride the same armed update, so a session + // that changes both changes them together. A NEGATIVE member means + // "not specified" -- the same reading InitializeExt gives it -- and + // leaves that quantizer alone, so a caller copying a config it built + // for a non-CQP session, where all three sit at -1, rewrites nothing. + // + // The clamps ride it too, but they are carried ONLY ON A CHANGE. A + // negative there means the update carries no clamp, which is what + // keeps a bitrate-only reconfigure from re-invoking the codec fill it + // has no reason to run. + const VkResult result = m_encoder->RequestRateControlUpdate( + averageBitrate, maxBitrate, config.frameRateNum, config.frameRateDen, + config.constQpI, config.constQpP, config.constQpB, + qpClampChanged ? config.minQp : -1, + qpClampChanged ? config.maxQp : -1); + if (result == VK_SUCCESS) { + // Keep the record current so a later Reconfigure compares against + // what is actually IN FORCE -- which is not always what was passed. + // Two of these members are coerced on the way in and a third can be + // dropped outright, and recording the raw config made the record + // disagree with the session on exactly those: + // + // * a zero maxBitrate means "track averageBitrate", and the + // session then runs at averageBitrate. The record said 0. + // * a zero frameRateNum leaves the frame rate ALONE -- both + // halves of it keep the values already in force. The record + // said 0, and stored whatever denominator came beside it. + // * a zero frameRateDen beside a non-zero numerator becomes 1. + // The record said 0. + // + // NOTHING COMPARED ABOVE READS THESE MEMBERS. The set written here + // -- the two bitrates, the two frame-rate halves, the constant-QP + // triple and the two clamps -- is disjoint from the set the + // immutables and the levers compare, so recording the applied value + // instead of the raw one cannot alter a single refusal decision. + // What it changes is the record telling the truth about the + // session, which is the only thing the record is for. + m_initConfig.averageBitrate = (uint32_t)averageBitrate; + m_initConfig.maxBitrate = (uint32_t)maxBitrate; + if (config.frameRateNum != 0) { + m_initConfig.frameRateNum = config.frameRateNum; + m_initConfig.frameRateDen = + (config.frameRateDen != 0) ? config.frameRateDen : 1; + } + // Only the quantizers actually named; an unnamed one keeps + // whatever the record already held. + if (config.constQpI >= 0) { m_initConfig.constQpI = config.constQpI; } + if (config.constQpP >= 0) { m_initConfig.constQpP = config.constQpP; } + if (config.constQpB >= 0) { m_initConfig.constQpB = config.constQpB; } + // Carried literally, so applied and passed are the same value; the + // assignment is a no-op when the clamp did not change. + m_initConfig.minQp = config.minQp; + m_initConfig.maxQp = config.maxQp; + } + return result; +} + +bool VulkanVideoEncoderExtImpl::ComputeFilterActive() const +{ + // The session's own answer to "can this encoder convert?", as opposed to + // the format taxonomy's "would conversion help?". Both halves matter: the + // filter has to be compiled in (IsPreprocessComputeFilterEnabled is false + // outright when it is not), and the session's declared input format has + // to be one the binder built a filter for. + // + // Answered from the INIT-TIME SNAPSHOT, not from m_encoderConfig. This is + // reached from SupportsFormat, which ValidateImageDescriptor calls + // without taking any lock -- and Deinitialize nulls and then destroys + // m_encoderConfig under m_pendingMutex, which this path does not take. + // Dereferencing it here is the exact hazard the comment at the + // m_computeFilterActive declaration exists to forbid, and it + // is a data race on the shared_ptr itself independently of the free. + // + // Still VK_FALSE before InitializeExt has bound a session, which is the + // honest answer: with no session there is no filter. + return m_computeFilterActive.load(std::memory_order_relaxed); +} + +// Whether this session's filter takes |inputFormat| under |colorModel| as its +// input. A stricter question than ComputeFilterActive(), and the one the +// registration gate actually asks: ValidateImageDescriptor admits a +// non-encode-format descriptor only when it equals +// m_encoderConfig->input.vkFormat. Answering the loose question on the query +// surface and the strict one at registration is how a producer ends up +// allocating a whole pool on a VK_TRUE it is then refused. +// +// THE COLOUR MODEL IS HALF OF THE MATCH. A filter is built for ONE declared +// pair -- VkEncDeriveFilterType reads the model the config declares and the +// format the device takes as an encode source -- and the packed 4:4:4 layouts +// put two pairs on one enumerant with OPPOSITE arms: R8G8B8A8_UNORM declared +// R'G'B' builds the forward matrix, and the same enumerant declared Y'CbCr is +// AYUV and builds the copy. Matching on the format alone would hand an AYUV +// frame to a filter that applies the R'G'B'-to-Y'CbCr matrix to samples that +// are already luma and chroma -- no error anywhere, and a wrong picture in +// every pixel. +bool VulkanVideoEncoderExtImpl::ComputeFilterTakesFormat( + VkFormat inputFormat, VkVideoEncoderColorModel colorModel) const +{ + return m_computeFilterActive.load(std::memory_order_relaxed) && + (m_computeFilterInputFormat.load(std::memory_order_relaxed) == + inputFormat) && + (VkEncResolveColorModel(inputFormat, colorModel) == + VkEncResolveColorModel(inputFormat, + SessionColorModel(inputFormat))); +} + +VkVideoEncoderColorModel VulkanVideoEncoderExtImpl::SessionColorModel( + VkFormat inputFormat) const +{ + return (inputFormat == m_sessionInputFormat.load(std::memory_order_relaxed)) + ? m_sessionInputColorModel.load(std::memory_order_relaxed) + : VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT; +} + +VkBool32 VulkanVideoEncoderExtImpl::SupportsFormat( + VkFormat inputFormat, VkVideoEncoderColorModel declaredColorModel) const +{ + // THE MODEL THE CLASS QUESTION IS ASKED UNDER. A declaration is a + // statement about the samples that only their producer can make, so it + // outranks the session's model wherever one is given; and it has to, + // because the packed 4:4:4 layouts (AYUV, Y410, Y416) ride RGBA format + // enumerants and are otherwise indistinguishable from an ordinary R'G'B' + // image. Classifying a declared surface under someone else's model + // answers a question the caller did not ask, and answers it about a + // different picture. + // + // FROM_FORMAT declares nothing -- it is what a zero-initialised + // descriptor and the public query both say -- and falls back to the + // session's model, which is the model this session negotiated for the + // format it was initialized with and FROM_FORMAT for every other. + const VkVideoEncoderColorModel colorModel = + (declaredColorModel == VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) + ? SessionColorModel(inputFormat) + : declaredColorModel; + + // The SESSION's answer, which is why this is a member and the taxonomy is + // not. A format that is only encodable after a conversion is supported + // exactly when this session can perform that conversion; answering VK_TRUE + // without one would be the "accepted and ignored" shape, which is the + // worst of the options -- and for the 3-plane family it is worse than + // that, because the copy the frame would fall back to hangs the GPU. + switch (VkEncClassifyInput(inputFormat, colorModel)) { + case VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT: + return VK_TRUE; + case VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER: + // Not merely "a filter exists" -- "the filter takes THIS declared + // pair". A session declared in a different format, or in the same + // format under the other colour model, has a filter built for + // that pair, so it converts nothing for this descriptor; + // ValidateImageDescriptor refuses the descriptor with + // FORMAT_UNSUPPORTED, and a VK_TRUE here would be the over-promise + // that costs the producer its pool allocation. + return ComputeFilterTakesFormat(inputFormat, colorModel) + ? VK_TRUE : VK_FALSE; + case VK_ENC_INPUT_FORMAT_UNSUPPORTED: + default: + return VK_FALSE; + } +} + +VkVideoEncoderColorModel VkEncResolveColorModel( + VkFormat inputFormat, VkVideoEncoderColorModel declared) +{ + // What the FORMAT says. A Y'CbCr VkFormat is one the multi-planar table + // places; everything else names an RGB layout. This answer is what + // EncoderInputImageParameters::colorSpace is written from, and that field + // is what VkEncDeriveFilterType reads to pick a filter arm, so the + // taxonomy and the filter cannot answer differently: one feeds the other. + const VkVideoEncoderColorModel fromFormat = + (YcbcrVkFormatInfo(inputFormat) != nullptr) + ? VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR + : VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + + if (declared == VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) { + return fromFormat; + } + if (declared == fromFormat) { + return declared; + } + // The declaration disagrees with the format. Exactly one disagreement is + // real rather than a caller error: the packed 4:4:4 Y'CbCr layouts have no + // Vulkan enumerant of their own and ride the RGBA ones, so Y'CbCr declared + // over one of those is the ONLY thing that tells AYUV, Y410 and Y416 from + // an ordinary RGBA image. PackedYcbcrFormatDesc is the single table that + // names them, and asking it here is what keeps this from becoming a second + // copy of that one. + if ((declared == VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) && + (PackedYcbcrFormatDesc(inputFormat) != nullptr)) { + return VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + } + // Every other disagreement is unresolvable, and the caller hears so. RGB + // declared over a Y'CbCr format, or Y'CbCr declared over an RGBA layout + // that carries no packed reading, cannot be reconciled: nothing here can + // know which of the two the caller meant, and choosing one produces a + // picture that is plausible and wrong everywhere. + return VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT; +} + +// --------------------------------------------------------------------------- +// WHAT THIS LIBRARY ROUTES, READ OFF THE FORMAT TABLES RATHER THAN LISTED +// --------------------------------------------------------------------------- +// +// The compute filter's Y'CbCr arm is GENERATED, not written per format: +// InitYCBCRCOPY takes its bit depth from GetBitsPerChannel(planesLayout), its +// chroma block from planesLayout.secondaryPlaneSubsampledX/Y, its plane count +// from the image's aspects -- which are numberOfExtraPlanes + 1 -- and, for +// the packed 4:4:4 aliases, the channel each of Y', Cb and Cr sits in from +// PackedYcbcrFormatDesc. Every one of those is a FIELD of a table this file +// can read. +// +// So the routable set is DERIVED from those tables, and what is written here +// is the POLICY: three predicates, each carrying the reason it is one. Which +// formats satisfy them, what each converts into and how many there are is read +// out of the tables. A format the tables gain is routed with no edit here, +// which is the point of deriving it: a list kept beside the tables is a second +// statement of the same set, and what a list holds is what somebody remembered +// to type. +// +// WHAT THE LIST THIS REPLACES HELD. Eleven entries, 4:2:0 throughout on its +// converted arm, while the direct arm beside it already carried 4:4:4. Nothing +// in the filter, in these tables or in the codec layer makes 4:2:0 the +// boundary -- CodecGetVkFormat spells 4:2:2 and 4:4:4 at all three depths, and +// the generator is parameterised over subsampling on both sides -- so the +// restriction was the list's and not the library's. The derivation does not +// reproduce it. +// +// WHAT IT WIDENS IS BOTH WHAT IS ACCEPTED AND WHAT IS ADVERTISED. A converted +// entry still has to name a target the DEVICE reports before +// VkEncAdvertiseInputFormats will write it -- but that question is now put at +// the profile the entry's own binding derives, so a 4:4:4 target is asked +// about at a 4:4:4 profile and reaches the advertised list wherever the device +// takes it. For a caller that DECLARES one of these formats the class question +// is still the library question, "does this library route this input", with +// the device question asked separately where the session is created; the +// difference is that the same pair of questions is now answerable before a +// session exists. + +// The plane layouts the filter's generator MODELS. +// +// YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED and YCBCR_PLANAR_STRIDE_PADDED are the +// two its read and write paths are parameterised over: both take their +// per-plane bindings from numberOfExtraPlanes and their chroma addressing from +// the subsampling bits, so neither has a per-format arm that could be missing. +// +// YCBCR_SINGLE_PLANE_INTERLEAVED -- the packed 4:2:2 family, YUY2 and UYVY and +// their 10-, 12- and 16-bit spellings -- is REFUSED, and this is the one +// exclusion that is about the generator being WRONG rather than about a limit +// elsewhere. The table gives those rows numberOfExtraPlanes = 1 although they +// are one plane, so the generator declares a two-plane read over a +// single-plane image; and the block coordinates emit one luma sample per texel +// for a format that carries two, which no field parameterises. The shader +// COMPILES either way, so admitting them would buy a plausible wrong picture +// instead of an error -- which is why the refusal is here, early, rather than +// left to be discovered. +// +// YCBCR_SINGLE_PLANE_UNNORMALIZED -- the R10X6 / R12X4 rows -- describes a +// PLANE's component format rather than an image a caller hands in, and carries +// neither subsampling nor a plane count to convert. +static bool VkEncFilterModelsLayout(const VkMpFormatInfo& mpInfo) +{ + switch (mpInfo.planesLayout.layout) { + case YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED: + case YCBCR_PLANAR_STRIDE_PADDED: + return true; + default: + return false; + } +} + +// The component depths an encode input can be SPELLED at. +// VkVideoComponentBitDepthFlagBitsKHR names 8, 10 and 12 and nothing else, so +// a 14- or 16-bit surface has no encode input geometry to be described by and +// CodecGetVkFormat spells no target for one. This is the refusal Y416 gets and +// the reason it gets it; stating it once covers the packed half and the +// multi-planar half together. +static bool VkEncEncodableComponentDepth(uint32_t bitsPerChannel) +{ + return (bitsPerChannel == 8u) || (bitsPerChannel == 10u) || + (bitsPerChannel == 12u); +} + +// Rung 1's depth. What vkGetPhysicalDeviceVideoFormatPropertiesKHR reports as +// an encode source is the 8- and 10-bit set: no driver table carries a 12-bit +// encode-input row, so a 12-bit semi-planar input has the encode format's +// plane layout AND its subsampling and is still rung 2, on its depth alone. +static bool VkEncDeviceReadsDepthUnconverted(uint32_t bitsPerChannel) +{ + return (bitsPerChannel == 8u) || (bitsPerChannel == 10u); +} + +// The RGB spellings this library routes. The one part of the set still written +// out, and it has to be: an RGB layout is exactly what the Y'CbCr table does +// not describe, so there is no table to read here. It is READ twice -- by the +// classifier's RGB arm and by the routable enumeration -- rather than stated +// twice, which is what keeps those two from naming different sets. +// +// THE SET IS DELIBERATELY THE 8-BIT UNORM FAMILY AND NOTHING ELSE. A false +// ENCODABLE is worse than a clean refusal, so each exclusion is a positive +// decision, not an oversight: +// +// *_SRGB (R8G8B8A8_SRGB, B8G8R8A8_SRGB, A8B8G8R8_SRGB_PACK32) -- EXCLUDED, +// and the storage read is what decides it. VulkanFilterYuvCompute binds +// the RGBA source as a VK_DESCRIPTOR_TYPE_STORAGE_IMAGE over one combined +// view and reads it with imageLoad, so what an RGBA input must grant is +// VK_IMAGE_USAGE_STORAGE_BIT -- not the sampled usage a texture read would +// need, and no create flags at all. An *_SRGB format does not carry +// VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT -- sRGB is a sampled-image feature +// -- and a view whose usage includes VK_IMAGE_USAGE_STORAGE_BIT must carry +// that feature itself (VUID-VkImageViewCreateInfo-usage-02275). An sRGB +// view can therefore never be the STORAGE_IMAGE this filter binds, so +// admitting the format would only move the refusal to view creation, past +// the point where the producer can still allocate differently. +// Independently of the access form: the filter applies the colour MATRIX +// ONLY, with no transfer function, and that is correct precisely because +// Rec.601/709/2020 Y'CbCr is defined on GAMMA-ENCODED R'G'B'. That is what +// makes the _UNORM spelling of each format the one to take -- it hands the +// stored code values to the matrix untouched. +// A2B10G10R10_UNORM_PACK32, R16G16B16A16_UNORM -- EXCLUDED as RGBA inputs, +// and not merely for bit depth: those two enumerants are also how Y410 and +// Y416 are spelled, and a Y'CbCr declaration over them means exactly that. +// Claiming them as RGBA as well would make one enumerant carry two colour +// models on one path. +// R16G16B16A16_SFLOAT -- EXCLUDED. scRGB is linear and unbounded; both the +// transfer function and the out-of-[0,1] range are unhandled here. +static const VkFormat kVkEncRoutableRgbFormats[] = { + VK_FORMAT_R8G8B8A8_UNORM, // RGBA8 + VK_FORMAT_B8G8R8A8_UNORM, // BGRA8 + VK_FORMAT_A8B8G8R8_UNORM_PACK32, // packed RGBA8 +}; + +// The two-plane semi-planar Y'CbCr format at |bitsPerChannel| and this +// subsampling, or VK_FORMAT_UNDEFINED where the table names none. +// +// This IS the conversion-target derivation. The filter's Y'CbCr arm changes +// plane layout and packing and resamples neither chroma nor bit depth, so the +// target is the semi-planar row that agrees with the input on both -- a +// question for the table, rather than a list of input/target pairs kept beside +// it that could name a row the filter would not produce. +static VkFormat VkEncSemiPlanarFormatAt(uint32_t bitsPerChannel, + uint32_t subsampledX, + uint32_t subsampledY) +{ + for (uint32_t i = 0;; i++) { + const VkMpFormatInfo* const mpInfo = YcbcrVkFormatInfoByIndex(i); + if (mpInfo == nullptr) { + break; + } + if ((mpInfo->planesLayout.layout != + YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED) || + (GetBitsPerChannel(mpInfo->planesLayout) != bitsPerChannel) || + (mpInfo->planesLayout.secondaryPlaneSubsampledX != subsampledX) || + (mpInfo->planesLayout.secondaryPlaneSubsampledY != subsampledY)) { + continue; + } + return mpInfo->vkFormat; + } + return VK_FORMAT_UNDEFINED; +} + +// How a Y'CbCr input reaches the encoder: which rung of the adaptation ladder +// it is on, and what it is converted into if it is converted at all. ONE +// function, because those are one decision -- a class that says a filter runs +// and a target that says no conversion exists describe different libraries. +struct VkEncYcbcrRoute { + VkEncInputFormatClass inputClass; + VkFormat target; // UNDEFINED unless the class is VIA_FILTER +}; + +static VkEncYcbcrRoute VkEncDeriveYcbcrRoute(VkFormat inputFormat) +{ + VkEncYcbcrRoute route = { VK_ENC_INPUT_FORMAT_UNSUPPORTED, + VK_FORMAT_UNDEFINED }; + + // Rung 2, packed half, and the arm the colour-model declaration exists + // for. AYUV and Y410 are Y'CbCr 4:4:4 carried one interleaved texel per + // pixel; they have no Vulkan enumerant of their own and ride the RGBA + // ones, so reaching this arm at all took the Y'CbCr declaration resolved + // by the caller -- undeclared, the same enumerants are an ordinary R'G'B' + // image and are answered by the RGB arm. + // + // What the filter converts for them is the PLANE COUNT, one against the + // encode source's two, which is the mismatch the 3-plane family has and is + // equally beyond a transfer copy. It reads them natively: the packed table + // gives the component depth and the channel each of Y', Cb and Cr sits in, + // and the single texel plane binds as ONE storage image rather than as + // per-plane views. + // + // Y416 (R16G16B16A16_UNORM) is refused by the depth predicate although the + // same packed table names it, and that is a decision rather than an + // oversight: 16 bits per component is not a + // VkVideoComponentBitDepthFlagBitsKHR, so no encode input geometry can + // carry it and CodecGetVkFormat spells no packed 16-bit target. Admitting + // it would replace this early, clear refusal with an opaque one raised + // after the caller had already built a frame pool. + // + // NO TARGET IS NAMED on this arm, and that is not an omission either: + // VkEncConversionTargetFormat takes no colour model and these enumerants + // resolve to R'G'B' without one, so it never reaches here. A target + // written here would be a claim nothing reads. + const VkPackedYcbcrFormatDesc* const packed = + PackedYcbcrFormatDesc(inputFormat); + if (packed != nullptr) { + if (VkEncEncodableComponentDepth(packed->bitDepth)) { + route.inputClass = VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER; + } + return route; + } + + const VkMpFormatInfo* const mpInfo = YcbcrVkFormatInfo(inputFormat); + if ((mpInfo == nullptr) || !VkEncFilterModelsLayout(*mpInfo)) { + return route; + } + const uint32_t bitsPerChannel = GetBitsPerChannel(mpInfo->planesLayout); + if (!VkEncEncodableComponentDepth(bitsPerChannel)) { + return route; + } + + // Rung 1. The input already has the encode source's layout and the device + // is handed it as it lies. + // + // THE SUBSAMPLING IS NOT FIXED AT 4:2:0. A 4:4:4 or 4:2:2 semi-planar + // input is the encoder input of a profile at that subsampling -- H.264 + // High 4:4:4 Predictive, H.265 Range Extensions -- and the codec config + // derives that profile from the input's own chroma subsampling, so nothing + // else has to be asked for. What this answers is whether the LIBRARY + // routes the input unconverted; whether THIS device exposes an encode + // profile at that subsampling is a device question, and it is answered + // where the session is created rather than guessed here. + if ((mpInfo->planesLayout.layout == + YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED) && + VkEncDeviceReadsDepthUnconverted(bitsPerChannel)) { + route.inputClass = VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT; + return route; + } + + // Rung 2. Either the PLANE COUNT is the mismatch -- three against the + // encode source's two -- or, for a semi-planar input that reached here, + // the bit depth is. The ladder's own table assigns subsampling and + // plane-layout conversion to the compute tier and restricts the transfer + // tier to "pure tiling mismatches", and a plane-count mismatch that takes + // the copy anyway hangs the GPU. + // + // A row with no semi-planar sibling at its own depth and subsampling has + // no route and stays UNSUPPORTED: the filter would have nothing to write + // into, and saying otherwise is the accepted-then-refused shape this + // taxonomy exists to prevent. + route.target = VkEncSemiPlanarFormatAt( + bitsPerChannel, mpInfo->planesLayout.secondaryPlaneSubsampledX, + mpInfo->planesLayout.secondaryPlaneSubsampledY); + if (route.target != VK_FORMAT_UNDEFINED) { + route.inputClass = VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER; + } + return route; +} + + +VkEncInputFormatClass VkEncClassifyInput(VkFormat inputFormat, + VkVideoEncoderColorModel colorModel) +{ + const VkVideoEncoderColorModel model = + VkEncResolveColorModel(inputFormat, colorModel); + + if (model == VK_VIDEO_ENCODER_COLOR_MODEL_RGB) { + // Rung 2, RGBA half. The set is kVkEncRoutableRgbFormats, which is + // READ here rather than restated: the routable enumeration reads the + // same array, so the classifier and the advertisement cannot name + // different RGB sets. Every exclusion, and the storage-image reason + // that decides them, is stated where that array is declared. + // + // These are single-plane, so VkEncInputFormatPlaneCount cannot answer + // from the class alone -- see the note there. + for (const VkFormat routed : kVkEncRoutableRgbFormats) { + if (routed == inputFormat) { + return VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER; + } + } + return VK_ENC_INPUT_FORMAT_UNSUPPORTED; + } + + if (model != VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) { + // The declaration and the format cannot both be true -- see + // VkEncResolveColorModel. Refused rather than reconciled. + return VK_ENC_INPUT_FORMAT_UNSUPPORTED; + } + + // Y'CbCr, and the whole of the answer is derived -- see + // VkEncDeriveYcbcrRoute, which is also what names the conversion target, + // so a format's rung and its target are one decision rather than two. + return VkEncDeriveYcbcrRoute(inputFormat).inputClass; +} + +// The routable list, in the order the advertisement emits converted entries: +// the direct rung first, then the converted rung, then RGB. +// +// DERIVED FROM THE SAME TABLES THE CLASSIFIER READS, so the list and the +// classifier are one predicate rather than two statements of one set. The +// eleven-entry literal this replaces had to be held against the classifier by +// a test, because nothing else held them together. +// +// Within each rung the order is the FORMAT TABLE's own, which groups by depth +// and then by subsampling. That keeps the 4:2:0 entries -- the only ones a +// 4:2:0 device list can reach -- in the relative order they have always been +// advertised in. +// +// THE PACKED 4:4:4 ALIASES ARE NOT HERE, and their absence is the decision the +// public header states rather than an omission: this list carries no colour +// model, so an enumerant whose only accepted reading is the Y'CbCr one cannot +// appear as itself. AYUV's enumerant is on the list under its RGB reading; +// Y410's is not on it at all, and absence from the list is not a refusal. +namespace { + +struct VkEncRoutableFormatSet { + VkFormat formats[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS]; + uint32_t count; + + VkEncRoutableFormatSet() : formats{}, count(0) + { + AppendYcbcrRung(VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT); + AppendYcbcrRung(VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER); + for (const VkFormat rgb : kVkEncRoutableRgbFormats) { + Append(rgb); + } + } + + void AppendYcbcrRung(VkEncInputFormatClass rung) + { + for (uint32_t i = 0;; i++) { + const VkMpFormatInfo* const mpInfo = YcbcrVkFormatInfoByIndex(i); + if (mpInfo == nullptr) { + break; + } + if (VkEncDeriveYcbcrRoute(mpInfo->vkFormat).inputClass == rung) { + Append(mpInfo->vkFormat); + } + } + } + + void Append(VkFormat format) + { + for (uint32_t i = 0; i < count; i++) { + if (formats[i] == format) { + return; + } + } + // Dropping an entry would make the advertisement quietly narrower than + // the taxonomy, which is the one failure a derived list could still + // introduce. The bound is static_asserted against the table's own + // length below, so this cannot fire. + assert(count < VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + if (count < VK_ENC_MAX_ROUTABLE_INPUT_FORMATS) { + formats[count++] = format; + } + } +}; + +} // anonymous namespace + +static_assert(VK_ENC_MAX_ROUTABLE_INPUT_FORMATS >= + (YCBCR_VK_FORMAT_INFO_TABLE_SIZE + + (sizeof(kVkEncRoutableRgbFormats) / sizeof(VkFormat))), + "the routable set is at most every row of the multi-planar " + "table plus the RGB spellings, so a bound below that could drop " + "a format the classifier routes"); +const VkFormat* VkEncRoutableInputFormats(uint32_t& outCount) +{ + // Built on the first call and immutable afterwards; the initialisation is + // thread-safe by the language, which is what lets the derivation run once + // rather than on every classification. + static const VkEncRoutableFormatSet kRoutable; + outCount = kRoutable.count; + return kRoutable.formats; +} + +VkFormat VkEncConversionTargetFormat(VkFormat inputFormat, + const VkFormat* deviceFormats, + uint32_t deviceFormatCount) +{ + if (VkEncResolveColorModel(inputFormat, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_VIDEO_ENCODER_COLOR_MODEL_RGB) { + // The encode source an RGB session ends up with. InitEncoder passes + // VK_FORMAT_UNDEFINED as the encode-source request for an RGB + // session and takes the driver's first choice, so naming the same + // entry here makes the advertisement and the session agree by + // construction rather than by luck. + return ((deviceFormats != nullptr) && (deviceFormatCount > 0)) + ? deviceFormats[0] + : VK_FORMAT_UNDEFINED; + } + + // Y'CbCr. Same subsampling, same bit depth, two planes -- which is what + // the filter's Y'CbCr arm produces and the whole of what it changes. + // + // DERIVED, and by the same call that classified the input, so a format's + // rung and the target it converts into are ONE decision, not two switches + // side by side: a second statement of the same set drifts, and the target's + // switch naming four inputs where the classifier's names six is exactly that + // drift. + // + // VK_FORMAT_UNDEFINED for a directly encodable input, which converts into + // nothing, and for one this library does not route at all. + return VkEncDeriveYcbcrRoute(inputFormat).target; +} + +namespace { + +bool VkEncFormatListContains(const VkFormat* formats, uint32_t count, + VkFormat format) +{ + for (uint32_t i = 0; i < count; i++) { + if (formats[i] == format) { + return true; + } + } + return false; +} + +bool VkEncEntryListContains(const VkVideoEncoderInputFormatProperties* entries, + uint32_t count, VkFormat format) +{ + for (uint32_t i = 0; i < count; i++) { + if (entries[i].format == format) { + return true; + } + } + return false; +} + +} // anonymous namespace + +uint32_t VkEncAdvertiseInputFormats( + VkEncInputFormatAdmitFn admit, void* userData, + VkVideoEncoderInputFormatProperties* outEntries, uint32_t outCapacity) +{ + if ((admit == nullptr) || (outEntries == nullptr)) { + return 0; + } + + // ONE CANDIDATE SET, OFFERED ONCE EACH. The routable list is what this + // library can route at all; whether a given device and profile will take + // any particular one of them is the admission's question, asked here and + // answered nowhere else in this function. Walking the routable list rather + // than a device list is what lets the admission be a live per-candidate + // resolve: a candidate is bound at ITS OWN derived profile, and different + // candidates therefore see different device answers. + uint32_t routableCount = 0; + const VkFormat* const routable = VkEncRoutableInputFormats(routableCount); + + VkVideoEncoderInputFormatProperties admitted[ + VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + uint32_t admittedCount = 0; + for (uint32_t i = 0; (i < routableCount) && + (admittedCount < VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + i++) { + const VkFormat format = routable[i]; + // Each format is written once however many times a device reports it, + // because tiling is a property of an image and not of a format. The + // routable list is already unique, so this catches only an admission + // that answered about a format it was not asked about. + if (VkEncEntryListContains(admitted, admittedCount, format)) { + continue; + } + VkVideoEncoderInputFormatProperties entry = {}; + entry.format = format; + if (!admit(userData, format, &entry)) { + continue; + } +#ifndef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // NOT ADVERTISED AT ALL WHEN THE FILTER IS NOT COMPILED IN. Every + // SUBOPTIMAL entry is ENCODABLE_VIA_FILTER, and InitializeExt refuses + // exactly that class with VK_ERROR_INITIALIZATION_FAILED when + // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED is undefined -- the two + // test the same predicate, so the correspondence is exact rather than + // approximate. Advertising them in such a build would name formats the + // library then refuses, which is the one failure this list exists to + // prevent. The condition is the BUILD's, not the device's and not the + // admission's, so it is applied here where no admission can bypass it. + // + // The OPTIMAL entries are deliberately outside this: they are + // ENCODABLE_DIRECT, the encoder reads them as they lie, and no filter + // is involved. A build without the filter advertises exactly the same + // OPTIMAL set as one with it. + if (entry.optimality != VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) { + continue; + } +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + admitted[admittedCount++] = entry; + } + + // OPTIMAL first, then SUBOPTIMAL, each group in the routable list's order. + // A caller reading the list top-down meets the entries the encoder takes as + // they lie before any that cost a conversion. + uint32_t written = 0; + for (uint32_t pass = 0; (pass < 2u) && (written < outCapacity); pass++) { + const VkVideoEncoderInputFormatOptimality wanted = + (pass == 0u) ? VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL + : VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL; + for (uint32_t i = 0; (i < admittedCount) && (written < outCapacity); + i++) { + if (admitted[i].optimality != wanted) { + continue; + } + outEntries[written++] = admitted[i]; + } + } + return written; +} + +VkBool32 VkEncSupportsInput(VkFormat inputFormat, + VkVideoEncoderColorModel colorModel) +{ + return (VkEncClassifyInput(inputFormat, colorModel) != + VK_ENC_INPUT_FORMAT_UNSUPPORTED) ? VK_TRUE : VK_FALSE; +} + +uint32_t VkEncInputFormatPlaneCount(VkFormat inputFormat) +{ + // The plane count is a property of the LAYOUT, and no class implies it. + // RGBA is a single plane and sits in the same class as 3-plane I420; + // 2-plane P012 sits there too, because what puts it on the filter is its + // bit depth and not its layout. This function is what + // EncoderConfig::input.numPlanes is written from, which drives the input + // geometry and the staging pool shape, so answering from the class would + // describe planes that do not exist and size allocations from them. + // + // A layout this library does not route has no plane count to give, and + // that test comes FIRST: an sRGB or scRGB surface is an RGB layout the + // library refuses, so asking "is it RGB" before "is it routed" would + // answer 1 for an input no allocation is ever sized for. + // + // "Routed" is asked under BOTH readings an enumerant can carry, because a + // packed 4:4:4 alias is routed only under a Y'CbCr declaration -- + // undeclared it reads as R'G'B', and as R'G'B' this library routes only + // the 8-bit set. Asking under FROM_FORMAT alone answers 0 for Y410, a + // layout this library does route and does size allocations for. + if ((VkEncClassifyInput(inputFormat, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED) && + (VkEncClassifyInput(inputFormat, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED)) { + return 0; + } + // The packed 4:4:4 layouts are ONE interleaved plane. Asking the packed + // table before the colour model is what makes this answer the same under + // either declaration, which is what the declaration-free signature + // promises: AYUV and the R'G'B' image it shares an enumerant with are + // both one plane, and Y410 is one plane whether or not it was declared. + if (PackedYcbcrFormatDesc(inputFormat) != nullptr) { + return 1; + } + if (VkEncResolveColorModel(inputFormat, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_VIDEO_ENCODER_COLOR_MODEL_RGB) { + return 1; + } + const VkMpFormatInfo* mpInfo = YcbcrVkFormatInfo(inputFormat); + if (mpInfo == nullptr) { + return 0; + } + return mpInfo->planesLayout.numberOfExtraPlanes + 1u; +} + +void VkEncSelectImportMemoryType(const VkEncImportMemoryTypeRequest& request, + const VkPhysicalDeviceMemoryProperties& memProps, + VkEncImportMemoryTypeChoice* outChoice) +{ + outChoice->index = UINT32_MAX; + outChoice->outcome = VK_ENC_IMPORT_MEMTYPE_NONE; + outChoice->exporterIndexOverridden = VK_FALSE; + + uint32_t mask = request.requirementsMask; + if (request.authoritativeMask != 0) { + mask &= request.authoritativeMask; + } + + // An index >= VK_MAX_MEMORY_TYPES cannot be shifted into a mask (and + // the old code would have shifted it anyway -- undefined behaviour). + const bool exporterIndexGiven = (request.exporterIndex != UINT32_MAX); + const bool exporterIndexUsable = + exporterIndexGiven && (request.exporterIndex < VK_MAX_MEMORY_TYPES); + + if (request.exporterIndexExact == VK_TRUE) { + if (exporterIndexGiven) { + // The exporter's parameters or nothing: substituting a + // different type violates the exact-match rule, so there is + // no valid fallback from here. + if (exporterIndexUsable && + ((mask & (1u << request.exporterIndex)) != 0)) { + outChoice->index = request.exporterIndex; + outChoice->outcome = VK_ENC_IMPORT_MEMTYPE_EXPORTER_INDEX; + } + return; + } + // Index unknown: constrain the heuristic by the exporter's own + // mask when it sent one. The type numbering is same-device -- a + // declared cross-device handle already failed DEVICE_MISMATCH at + // validation. + if (request.exporterMask != 0) { + mask &= request.exporterMask; + } + } else if (exporterIndexGiven) { + if (exporterIndexUsable && + ((mask & (1u << request.exporterIndex)) != 0)) { + outChoice->index = request.exporterIndex; + outChoice->outcome = VK_ENC_IMPORT_MEMTYPE_EXPORTER_INDEX; + return; + } + outChoice->exporterIndexOverridden = VK_TRUE; + } + + const uint32_t typeCount = (memProps.memoryTypeCount < VK_MAX_MEMORY_TYPES) + ? memProps.memoryTypeCount + : VK_MAX_MEMORY_TYPES; + for (uint32_t i = 0; i < typeCount; i++) { + if (((mask & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & + VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT) != 0)) { + outChoice->index = i; + outChoice->outcome = VK_ENC_IMPORT_MEMTYPE_MASK_DEVICE_LOCAL; + return; + } + } + // A type the driver declares importable for this handle is spec-valid + // and functional; refusing it over a DEVICE_LOCAL preference would turn + // a workable registration into a failure. + for (uint32_t i = 0; i < typeCount; i++) { + if ((mask & (1u << i)) != 0) { + outChoice->index = i; + outChoice->outcome = VK_ENC_IMPORT_MEMTYPE_MASK_FIRST; + return; + } + } +} + +VkBool32 VkEncDescriptorWithinCreationLimits( + const VkVideoEncoderExternalImageDescriptor& desc, + const VkImageFormatProperties& limits) +{ + // The same 0-means-default rules ImportImageLocked applies when it + // builds VkImageCreateInfo -- anything else and the pre-check and the + // import judge two different images. extent.depth is the create call's + // constant 1, inside any maxExtent a successful query can return. + const uint32_t mipLevels = + (desc.mipLevels == 0) ? 1u : desc.mipLevels; + const uint32_t arrayLayers = + (desc.arrayLayers == 0) ? 1u : desc.arrayLayers; + const VkSampleCountFlags samples = + (desc.samples == 0) ? VK_SAMPLE_COUNT_1_BIT : desc.samples; + if ((desc.width > limits.maxExtent.width) || + (desc.height > limits.maxExtent.height)) { + return VK_FALSE; // VUID-VkImageCreateInfo-extent-02252/-02253 + } + if (mipLevels > limits.maxMipLevels) { + return VK_FALSE; // VUID-VkImageCreateInfo-mipLevels-02255 + } + if (arrayLayers > limits.maxArrayLayers) { + return VK_FALSE; // VUID-VkImageCreateInfo-arrayLayers-02256 + } + if ((samples & limits.sampleCounts) == 0) { + return VK_FALSE; // VUID-VkImageCreateInfo-samples-02258 + } + return VK_TRUE; +} + +VkResult VulkanVideoEncoderExtImpl::GetRuntimeInfo( + VkVideoEncoderRuntimeInfo* outInfo) const +{ + // The struct gate is a property of the argument alone, answered the + // same with or without a session -- malformed is named as malformed, + // never as "not ready yet". A chained struct is refused rather than + // skipped: the snapshot reset below would otherwise silently wipe the + // caller's chain pointer, exactly the accepted-and-ignored failure the + // sType/pNext rules exist to prevent. + if ((outInfo == nullptr) || + (outInfo->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_RUNTIME_INFO) || + (outInfo->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (!m_initialized) { + return VK_NOT_READY; + } + + *outInfo = {}; + + // Static identity. Vulkan-Video encoders are HW-only by definition (no + // SW-fallback path), and the Ext interface accepts external VkImages. + std::snprintf(outInfo->implementationName, + sizeof(outInfo->implementationName), + "VulkanVideoEncoder"); + outInfo->isHardwareAccelerated = VK_TRUE; + outInfo->supportsNativeHandle = VK_TRUE; + + // trustedRateController: VK_TRUE iff the active rate-control mode is a + // tracked/managed-bitrate mode (CBR or VBR). We trust any driver advertising + // a tracked mode -- Vulkan-Video drivers are HW-only, so the historical + // SW-fallback allow-list does not apply; no vendor table here. DEFAULT(0) + // and DISABLED/const-QP(1) are not bitrate-tracked, so report VK_FALSE. + outInfo->trustedRateController = + ((m_rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR) || + (m_rateControlMode == VK_VIDEO_ENCODE_RATE_CONTROL_MODE_VBR_BIT_KHR)) + ? VK_TRUE + : VK_FALSE; + + // MVP: simulcast is handled by N parallel sessions on the Chromium side; + // frame-size-change requires a session re-init (the MVP Reconfigure + // covers mid-stream rate-control only). + outInfo->supportsSimulcast = VK_FALSE; + outInfo->supportsFrameSizeChange = VK_FALSE; + + // We do not yet surface per-frame average QP back through + // VkVideoEncodeResult; report VK_FALSE until that metadata is wired. + outInfo->reportsAverageQp = VK_FALSE; + + // Resolution alignment: report the driver's probed + // pictureAccessGranularity once the session config carries it; 16 stays + // the conservative pre-probe fallback (no HW encoder rejects 16-aligned + // input, and the encoder crops internally). + const uint32_t granularityW = + m_capsGranularityW.load(std::memory_order_relaxed); + const uint32_t granularityH = + m_capsGranularityH.load(std::memory_order_relaxed); + outInfo->requestedResolutionAlignmentWidth = + (granularityW > 0u) ? granularityW : 16u; + outInfo->requestedResolutionAlignmentHeight = + (granularityH > 0u) ? granularityH : 16u; + outInfo->applyAlignmentToAllSimulcastLayers = VK_TRUE; + + return VK_SUCCESS; +} + +void VulkanVideoEncoderExtImpl::Deinitialize() +{ + if (m_encoder) { + m_encoder->WaitForThreadsToComplete(); + } + + // R-5 telemetry summary: the design's removal condition was "zero + // fallbacks", a number nobody could produce. Now the session log can: + // silence here means every import selection ran under its authoritative + // mask. + const uint64_t heuristicSelections = + m_importMemTypeHeuristicSelections.load(std::memory_order_relaxed); + const uint64_t exporterOverrides = + m_importMemTypeExporterOverrides.load(std::memory_order_relaxed); + if ((heuristicSelections != 0) || (exporterOverrides != 0)) { + VkEncErr() << "[EncoderExt] import memory-type telemetry: " + << heuristicSelections << " heuristic selection(s), " + << exporterOverrides << " exporter-index override(s)" + << std::endl; + } + + { + std::lock_guard lock(m_pendingMutex); + if (!m_pendingFrames.empty()) { + VkEncErr() << "[EncoderExt] Deinitialize with " + << m_pendingFrames.size() + << " undelivered/unreleased frames; any retained " + "pBitstreamData pointers are invalid from here" + << std::endl; + } + for (auto& p : m_pendingFrames) { + if (p.releaseFenceSemaphore != VK_NULL_HANDLE) { + m_unprovenSemaphores.push_back(p.releaseFenceSemaphore); + } + m_unprovenSemaphores.insert(m_unprovenSemaphores.end(), + p.acquireFenceSemaphores.begin(), + p.acquireFenceSemaphores.end()); + } + m_pendingFrames.clear(); + m_readyOrder.clear(); + m_releasedWhilePending.clear(); + } + + // Same one-sided-lock hazard as Flush: clear under the mutex the + // retrieval path holds, destruct outside it. + VkSharedBaseObj doomedEncoder; + VkSharedBaseObj doomedConfig; + { + std::lock_guard lock(m_pendingMutex); + doomedEncoder = m_encoder; + doomedConfig = m_encoderConfig; + m_encoder = nullptr; + m_encoderConfig = nullptr; + } + doomedEncoder = nullptr; + doomedConfig = nullptr; + + // Handle-exchange teardown: registrations the consumer never retired + // must not outlive the session -- the images and memory they own were + // created on this session's device. The encoder threads are joined and + // m_encoder is released above, so no frame can still hold a node ref; + // dropping the slot refs here is the final release. + { + std::lock_guard lock(m_resourceMutex); + size_t leaked = 0; + for (auto& slot : m_resources) { + if (slot.live || slot.retired) { + if (slot.inFlight != 0) { + VkEncErr() << "[EncoderExt] Deinitialize: registration " + "still in flight (inFlight=" + << slot.inFlight + << "); destroying anyway -- its frames were " + "torn down with the encoder." << std::endl; + } + DestroyResourceLocked(slot); + leaked++; + } + } + if (leaked != 0) { + VkEncErr() << "[EncoderExt] Deinitialize retired " << leaked + << " registration(s) the consumer never unregistered" + << std::endl; + } + } + + // The semaphore registry's half of the same rule: an imported + // VkSemaphore the consumer never unregistered must not outlive the + // session. The destroy is legal here for the same reason the encoder's + // own completion timeline may be destroyed in DeinitEncoder: by this + // point the encoder has been released (above, or by an earlier Flush), + // which joins the library threads and waits the ENCODE queue idle, so + // no submitted batch can still reference these handles + // (VUID-vkDestroySemaphore-semaphore-01137). Slot hygiene mirrors + // UnregisterSemaphore -- handle nulled, generation bumped -- so an id + // that survives into a re-initialized session resolves to + // RESOURCE_UNKNOWN rather than to a recycled slot. + { + std::lock_guard lock(m_semaphoreMutex); + size_t leakedSemaphores = 0; + for (auto& slot : m_semaphores) { + if (slot.live) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), + slot.semaphore, nullptr); + slot.semaphore = VK_NULL_HANDLE; + slot.live = false; + slot.generation++; + leakedSemaphores++; + } + } + if (leakedSemaphores != 0) { + VkEncErr() << "[EncoderExt] Deinitialize retired " + << leakedSemaphores + << " semaphore registration(s) the consumer never " + "unregistered" << std::endl; + } + } + + // Per-frame release fences. The same legality argument as the semaphore + // registry above -- the encoder has been released, which joins the + // library threads and waits the ENCODE queue idle -- with one gap closed + // explicitly: a staging submit can ride the TRANSFER queue, and an + // ENCODE-queue wait alone does not cover a staging batch whose encode was + // never submitted. A device-wide wait does, and teardown is the one place + // where its cost does not matter. + { + std::vector doomed; + { + std::lock_guard lock(m_pendingMutex); + doomed.swap(m_releaseFenceRetired); + doomed.insert(doomed.end(), m_unprovenSemaphores.begin(), + m_unprovenSemaphores.end()); + m_unprovenSemaphores.clear(); + } + if (!doomed.empty() && (m_vkDevCtx.getDevice() != VK_NULL_HANDLE)) { + m_vkDevCtx.DeviceWaitIdle(); + for (VkSemaphore sem : doomed) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), sem, + nullptr); + } + } + m_releaseFenceExportProbed = false; + m_releaseFenceExportable = false; + } + + // The dma-buf import-ordinal guard, and the last Vulkan object this + // session destroys. At the disabled default (count 0) the guard holds + // nothing, this finds nothing and returns 0 -- but the ORDER stays + // load-bearing for any build that re-arms it, so the call stays where the + // armed guard needs it. LAST is then the requirement and not a + // preference: the guard's whole function is to hold the leading dma-buf + // import live-positions, so it has to outlive every import this device + // will ever make. Everything above + // -- the encoder release that joins the library threads, the registration + // retirement, the semaphore registry, the release fences -- is above a + // line past which VkEncImportExternalImage is no longer reachable on + // m_vkDevCtx; below it, ~VulkanDeviceContext destroys the device itself as + // soon as this object's destructor body returns. + // + // Deinitialize() is private and the destructor is its only caller, so this + // runs exactly once per session, on every exit path including a failed + // init. If Deinitialize ever becomes re-enterable, this call has to move + // with the device's lifetime and not with the session's -- releasing + // while the device lives on un-does the phase an armed guard was built + // for, and reports nothing. + const uint32_t importGuardsReleased = + VkEncReleaseImportOrdinalGuard(m_vkDevCtx); + if (importGuardsReleased != 0) { + VkEncErr() << "[EncoderExt] Deinitialize released " + << importGuardsReleased + << " import-ordinal guard import(s)" << std::endl; + } + + m_initialized = false; + m_rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DEFAULT_KHR; + m_capsGranularityW.store(0, std::memory_order_relaxed); + m_capsGranularityH.store(0, std::memory_order_relaxed); + m_computeFilterActive.store(false, std::memory_order_relaxed); + m_computeFilterInputFormat.store(VK_FORMAT_UNDEFINED, + std::memory_order_relaxed); + m_sessionInputFormat.store(VK_FORMAT_UNDEFINED, std::memory_order_relaxed); + m_sessionInputColorModel.store(VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, + std::memory_order_relaxed); + m_importMemTypeHeuristicSelections.store(0, std::memory_order_relaxed); + m_importMemTypeExporterOverrides.store(0, std::memory_order_relaxed); +} + +//============================================================================= +// Handle exchange +//============================================================================= + +// The view formats a MUTABLE_FORMAT registration must declare, written into +// |outFormats| (capacity kVkEncMaxViewFormats) and returned as a count. +// +// MUTABLE_FORMAT obliges a VkImageFormatListCreateInfo +// (VUID-VkImageCreateInfo-tiling-02353), and the list is not decoration: a +// view whose format is absent from it is invalid +// (VUID-VkImageViewCreateInfo-pNext-01585). So the list has to name every +// format the library will actually build a view over -- the descriptor's own +// format for the combined view, and each PLANE format (R8_UNORM, +// R8G8_UNORM, ...) for the per-plane views the compute filter binds. +// +// Measured on this stack (A4000, GBM dma-buf, modifier 0x0): per-plane +// STORAGE views on an imported multi-planar image are creatable, and +// validation-clean, only with MUTABLE_FORMAT | EXTENDED_USAGE *and* this +// list present. Modifier 0 for the multi-planar format carries no STORAGE in +// its tilingFeatures; modifier 0 for R8_UNORM / R8G8_UNORM does, and +// EXTENDED_USAGE is what makes the usage validate against these VIEW formats +// instead of the image format. The library never adds EXTENDED_USAGE itself +// -- it is the exporter's declaration to make, and the library never +// invents a usage or a flag the exporter did not grant -- but when the +// exporter declared MUTABLE_FORMAT the list is OURS to emit. +enum { kVkEncMaxViewFormats = 4 }; + +// Whether this registration's image can carry the per-plane STORAGE views the +// preprocess compute filter binds. Every clause is the DESCRIPTOR's, never +// ours: the exporter has to have created the image MUTABLE_FORMAT (a plane +// view reinterprets the format, VUID-VkImageViewCreateInfo-image-01762) and +// to have granted STORAGE (a plane view cannot carry a usage the image lacks, +// VUID-VkImageViewCreateInfo-pNext-02662). +// +// EXTENDED_USAGE is part of the requirement, not decoration. Measured on this +// stack: MUTABLE_FORMAT | EXTENDED_USAGE plus the format list is what makes +// per-plane STORAGE views creatable AND validation-clean; MUTABLE_FORMAT +// ALONE makes vkGetPhysicalDeviceImageFormatProperties2 answer +// VK_ERROR_FORMAT_NOT_SUPPORTED, because modifier 0 for the MULTI-PLANAR +// format carries no STORAGE in its tilingFeatures and EXTENDED_USAGE is what +// re-validates the usage against the VIEW formats instead. Without this +// clause a descriptor that has MUTABLE_FORMAT and STORAGE but not +// EXTENDED_USAGE passes the gate whose refusal message already names +// EXTENDED_USAGE in prose, and fails later at ModifierWouldRegister -- which +// then reports "no modifier on this device would work", when the actual fix +// is one create flag the caller was never asked for. +static bool VkEncDescriptorPermitsPlaneStorageViews( + const VkVideoEncoderExternalImageDescriptor& desc, + VkImageUsageFlags resolvedUsage) +{ + return ((desc.imageFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) && + ((desc.imageFlags & VK_IMAGE_CREATE_EXTENDED_USAGE_BIT) != 0) && + ((resolvedUsage & VK_IMAGE_USAGE_STORAGE_BIT) != 0) && + (YcbcrVkFormatInfo(desc.format) != nullptr); +} + +// The SINGLE-PLANE counterpart -- a separate predicate rather than a relaxed +// conjunct in the one above, because the filter's two arms bind different +// VIEWS of a descriptor and collapsing them refuses one arm's shipped +// producer shape. +// +// Both arms bind VK_DESCRIPTOR_TYPE_STORAGE_IMAGE, so both need STORAGE. +// What only the multi-planar arm needs is MUTABLE_FORMAT | EXTENDED_USAGE: a +// per-plane view reinterprets the image as a different format +// (VUID-VkImageViewCreateInfo-image-01762), and that cost belongs to the +// per-plane views, not to the format's colour model. +// +// The SINGLE-PLANE arm binds ONE descriptor over the COMBINED view: +// VulkanFilterYuvCompute sets m_inputImageAspects = +// VK_IMAGE_ASPECT_COLOR_BIT for every input the multi-planar table does not +// place -- an R'G'B' image and, equally, a packed 4:4:4 Y'CbCr surface on an +// RGBA-typed enumerant -- and UpdateImageDescriptorSets binds aspect 0 +// through GetImageView(). So the arm is chosen by PLANE COUNT rather than by +// colour model: AYUV needs this arm and no create flags, exactly as R'G'B' +// does. What this gate requires of a descriptor is exactly two things: +// +// * the image must GRANT VK_IMAGE_USAGE_STORAGE_BIT. A STORAGE_IMAGE +// descriptor may only name a view whose image carries it +// (VUID-VkWriteDescriptorSet-descriptorType-00339), and this library +// never invents a usage the exporter did not grant; and +// * the DEVICE must support a storage read of that format on that tiling +// -- |deviceCanStorageRead|. +// +// It needs NO create flags. Nothing on this arm reinterprets the format, so +// MUTABLE_FORMAT buys nothing here, and requiring it would refuse the shipped +// producer shape (imageFlags = 0) over a constraint that belongs to the other +// arm. +static bool VkEncDescriptorPermitsStorageRead( + const VkVideoEncoderExternalImageDescriptor& desc, + VkImageUsageFlags resolvedUsage, + bool deviceCanStorageRead) +{ + return (VkEncInputFormatPlaneCount(desc.format) == 1u) && + ((resolvedUsage & VK_IMAGE_USAGE_STORAGE_BIT) != 0) && + deviceCanStorageRead; +} + +static uint32_t VkEncCollectViewFormats( + const VkVideoEncoderExternalImageDescriptor& desc, + VkFormat outFormats[kVkEncMaxViewFormats]) +{ + uint32_t count = 0; + outFormats[count++] = desc.format; + // Only when plane views will actually be built. A format list is a + // CONSTRAINT the create call has to satisfy, not a hint: naming R8_UNORM + // on a modifier whose R8 views the device does not support would turn a + // registration that works today into a vkCreateImage failure. So the list + // names exactly the views this registration will ask for. + // + // "Will be built" is VkImageResourceView::Create's condition, NOT the + // narrower storage-view gate. Both Create() overloads build per-plane + // views whenever the image is MUTABLE_FORMAT and the derived plane usage + // is non-zero -- STORAGE is not required by either -- so keying the list + // on the storage gate left a MUTABLE_FORMAT-without-STORAGE registration + // with R8/R8G8 views whose formats are absent from the image's format + // list (VUID-VkImageViewCreateInfo-pNext-01585). Inert on the shipped + // path, where the producer declares imageFlags = 0 and no list is chained + // at all, and wrong for the first producer that declares MUTABLE_FORMAT + // for any other reason. One condition, so the query, the create and the + // views describe one image. + if (((desc.imageFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) == 0) || + (YcbcrVkFormatInfo(desc.format) == nullptr)) { + return count; + } + const VkMpFormatInfo* mpInfo = YcbcrVkFormatInfo(desc.format); + for (uint32_t plane = 0; + (plane < 3u) && (count < kVkEncMaxViewFormats); plane++) { + const VkFormat planeFormat = mpInfo->vkPlaneFormat[plane]; + if (planeFormat == VK_FORMAT_UNDEFINED) { + break; + } + bool duplicate = false; + for (uint32_t i = 0; i < count; i++) { + duplicate = duplicate || (outFormats[i] == planeFormat); + } + if (!duplicate) { + outFormats[count++] = planeFormat; + } + } + return count; +} + +// Import the descriptor into an owned VkImage + VkDeviceMemory. The full +// contract -- including the fd-ownership split on the vkAllocateMemory +// handoff -- is at the declaration in vulkan_video_encoder_ext_internal.h; +// in short, this function is the ONE place that closes the fd, and only +// on the failure exits that never handed it to the driver. + +// ================= THE dma-buf IMPORT-ORDINAL GUARD ======================== +// +// DISABLED BY DEFAULT. kVkEncImportOrdinalGuardCount below is 0, so none of +// this runs in a stock build and every dma-buf import on an NVIDIA device +// reports the verdict DISABLED. The mechanism is kept compiled, reachable +// and tested because it is the only lever this library has ever had on the +// defect described here, and a future driver may make some phase of it worth +// pulling again. It is NOT a fix, and the block below says why so that +// nobody re-arms it expecting one. +// +// WHAT IS WRONG, AND IT IS NOT OURS TO FIX. A dma-buf image import can come +// back bound to memory the buffer's contents never reach. THE DEFECT IS IN +// THE DRIVER'S dma-buf IMPORT PATH. +// Nothing readable at the import boundary distinguishes a damaged import +// from a good one -- descriptor, explicit DRM plane layouts, memReqs, +// vkGetMemoryFdPropertiesKHR mask, chosen memory type and the driver's own +// vkGetImageSubresourceLayout answer for +// VK_IMAGE_ASPECT_MEMORY_PLANE_1_BIT_EXT are identical on both. Only the +// CONTENT the GPU reads back differs, and there are two damage modes: +// +// CHROMA_ZERO plane 1 of the imported VkImage does not alias the buffer's +// chroma pages; plane 0 does. Correct luma over zero chroma. +// A write through this import is SWALLOWED -- the mapping is +// dead, so a consumer cannot repair the buffer through it. +// ALL_ZERO Y = U = V = 0 in the buffer itself. The mapping is fine and +// a write through it lands; what was lost is the EXPORTER's +// write, before this library imported anything at all. +// +// Both render as one saturated green frame in a consumer, on every frame the +// affected buffer serves and on no frame of any other buffer. Legal black +// would be U = V = 128, so a zeroed plane is missing data and not a black +// frame -- which is what makes this visible at all. +// +// WHAT THE GUARD DOES: IT MOVES THE DAMAGE. IT DOES NOT REMOVE IT. +// The guard takes K sacrificial dma-buf imports and RETAINS them, so every +// caller-visible import lands K live-positions later. +// +// The damage is PERIODIC in the import ordinal. A run that registers only +// two or three buffers samples one period at one phase, so a K that moves +// the damaged ordinals off exactly those buffers reads as a fix. Over enough +// periods the PROPORTION of damaged imports does not move with K at all. +// The guard shifts the PHASE of the pattern: it changes WHICH imports are +// damaged and never HOW MANY. Anyone re-arming this is buying a phase, and +// owes the next reader which phase and why. +// +// AND THERE IS NO CONSUMER-SIDE RECOVERY: +// * writing through a damaged import is swallowed (the CHROMA_ZERO mode +// above), so a consumer cannot heal a damaged buffer by rewriting it. +// Where such a write DOES land, the damage was the producer's lost write +// and there was never anything on this side to repair; +// * re-importing the same dma-buf at the next ordinal does not rescue it. +// Once inflicted, the damage is a property of the BUFFER and not of the +// import ordinal that produced it, so retrying the import is not a +// workaround either. +// Those two properties, and not a preference for less code, are why the +// count is 0. +// +// LIFETIME -- BOTH ENDS OF IT, binding again the moment the count is not 0: +// +// * NOT ONE IMPORT EARLY. Freeing a guard hands its live-position straight +// back to the next caller import and un-does the phase the count was +// chosen for -- silently, because the only symptom is green chroma in +// the consumer's output. The guard lives until the device can no longer +// import at all. +// * NOT ONE CALL LATE. Guard VkImage and VkDeviceMemory still alive when +// vkDestroyDevice is called is VUID-vkDestroyDevice-device-05137. +// Vulkan has no rule by which a device destroys its children -- +// vkDestroyDevice destroys nothing but the device, and it is the +// application that must have destroyed the children first. Leaving them +// leaks two images and two allocations per device and violates the spec +// on every teardown, invisibly unless a validation layer is running. +// +// The single point that satisfies both is +// VulkanVideoEncoderExtImpl::Deinitialize(), which the session destructor +// calls after the encoder is released and every registration retired -- so +// no further import is reachable -- and before ~VulkanDeviceContext destroys +// the device. VkEncReleaseImportOrdinalGuard() below is what it calls. +// +// COST when re-armed: K VkImage + K VkDeviceMemory and K extra references on +// the first registered dma-buf, for the lifetime of the VkDevice; no +// per-frame cost. At the default of 0 none of it is taken, and no per-device +// registry entry is created either. +// +// WHAT IS NOT ESTABLISHED, so the next reader does not over-trust any of it: +// * WHY imports are damaged, and why periodically, is not known. The +// pattern is an observation, not a model of the driver. +// * It applies to the GBM dma-buf import path. A composited lane on the +// same driver and the same descriptor shows no damage at all, so some +// lane-level precondition is required and is unidentified. +// * Whether any K helps on a different driver is unknown: K is a phase, +// and the phase-invariance is a property of one import path. +// Kill switch: VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD=1, which reports the +// same DISABLED verdict the disabled default already reports. +namespace { + +// THE KNOB, and the only line that has to change to re-arm the guard. +// 0 = disabled: no sacrificial import is taken and every dma-buf import on an +// NVIDIA device reports DISABLED. Non-zero re-arms it with that many retained +// imports, which buys a PHASE SHIFT of the driver's damage pattern and not a +// repair of it -- read the block above before changing this. Two suites will +// have opinions about a change here, deliberately: +// * test/encoder-ext-import-guard pins this value device-free, in CI. +// * test/encoder-ext-import-ordinal-guard takes --guard-count=N, which +// re-arms its whole positional machinery on hardware for a build that +// sets this to N. +constexpr uint32_t kVkEncImportOrdinalGuardCount = 0; + +// Capacity of the per-device retained-image array, deliberately independent +// of the count above. Two reasons, both concrete: a count of 0 must not +// produce a zero-length array (not standard C++; -Wpedantic rejects it +// outright, and neither this library's nor Chromium's flags happen to pass +// it today, which is not a property to depend on), and the measured phase +// sweep ran K = 0..4, so the whole sweep stays reachable by editing the one +// line above. +constexpr uint32_t kVkEncImportOrdinalGuardCapacity = 4; +static_assert(kVkEncImportOrdinalGuardCount <= kVkEncImportOrdinalGuardCapacity, + "kVkEncImportOrdinalGuardCount must fit the retained-image " + "array; raise kVkEncImportOrdinalGuardCapacity with it"); + +// What one device's guards are. |live| is how many of |images| have actually +// been imported; it can sit below the requested count after a failed dup(2) +// or a failed sacrificial import, and the next import on the device tops it +// up. At the disabled default the guard returns before this is ever reached, +// and no entry is created at all. +struct VkEncImportOrdinalGuardEntry { + VkDevice device = VK_NULL_HANDLE; + uint32_t live = 0; + VkEncImportedImage images[kVkEncImportOrdinalGuardCapacity]{}; +}; + +std::mutex g_importGuardMutex; +// One entry per VkDevice that has taken guards; erased by +// VkEncReleaseImportOrdinalGuard when that device is torn down. Guarded by +// g_importGuardMutex on every access, read and write. +// +// KEYED BY DEVICE, not a single slot, and that is a correctness point rather +// than a generality one. The single slot this replaces carried one VkDevice +// and one counter, and its device-change branch reset the counter and moved +// on. With two sessions alive at once that branch fired on every alternation +// and it (a) ORPHANED the other device's images -- the only handles to them +// were in the array it had just abandoned, so nothing could ever free them -- +// and (b) lost the record that the other device was already guarded, so it +// re-guarded it on the next import and grew the leak by two more objects each +// time. A per-device entry has no such branch: a device's guards are found, +// or created, and are released exactly once by handle. +std::vector g_importGuards; + +// Re-entrancy: the guard builds its imports through the very function it is +// called from, so the nested calls must not try to build guards of their own. +// It is also what keeps the non-recursive mutex above safe: the nested call +// returns at the TOP of VkEncEnsureImportOrdinalGuard, above the lock. +// Dead at the disabled default, which takes no nested import at all. +thread_local bool g_inImportOrdinalGuard = false; + +// What the guard did on THIS thread's most recent import attempt. See +// VkEncImportOrdinalGuardReport (internal header) for why it is +// thread_local: the guard produces it deep inside the import and +// RegisterImageResource -- the only frame that can hand it to a caller -- +// consumes it on the way back out, on the same thread. +thread_local VkEncImportOrdinalGuardReport g_importGuardReport{}; + +void VkEncSetImportGuardVerdict(VkVideoEncoderImportGuardState state, + uint32_t retainedCount, + VkVideoEncoderStatusCode failureStatus, + int32_t failureErrno) +{ + g_importGuardReport.state = state; + g_importGuardReport.retainedCount = retainedCount; + g_importGuardReport.failureStatus = failureStatus; + g_importGuardReport.failureErrno = failureErrno; +} + +bool VkEncImportOrdinalGuardDisabled() +{ + // ONE INITIALIZATION, NOT A SENTINEL. The sentinel form was a + // check-then-store on the registration path, which is reached before the + // guard-count-zero return and well before any mutex is taken, so two + // sessions registering at once both saw -1 and both wrote. A function-local + // static with a dynamic initializer is initialized exactly once, and the + // supported CMake and GN builds both keep the thread-safe guard that makes + // that true -- none of the reviewed Chromium settings disable it. (If a + // supported build ever did, this would need std::call_once; const alone + // would not make dynamic initialization safe.) + // + // The predicate is unchanged: absent, empty and leading-zero all mean + // enabled. The environment must be settled before concurrent use, which + // was already true of the sentinel form. + static const bool disabled = []() { + const char* env = getenv("VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD"); + return (env != nullptr) && (*env != '\0') && (*env != '0'); + }(); + return disabled; +} + +bool VkEncIsNvidiaDevice(const VulkanDeviceContext& vkDevCtx) +{ + // BOTH checks are load-bearing, and neither is defensive habit. + // + // Until the import-ordinal guard existed, VkEncImportExternalImage touched + // no INSTANCE-level dispatch at all -- it worked entirely off the device. + // The guard calls this on the way in to every dma-buf import, so a context + // whose instance table was never populated now reaches a null function + // pointer on a path a caller can reach. That is not + // hypothetical: it segfaults all ten of Chromium's + // VulkanVideoEncoderImportOwnershipTest cases, which construct a bare + // VulkanDeviceContext and stub only the device-level entries the import + // itself uses. ip=0, SEGV_MAPERR at address 0. + // + // The direction of the failure is chosen, not incidental. Returning false + // means "not NVIDIA", so the guard reports NOT_NVIDIA and does not run, and + // the import proceeds exactly as it did before the guard was added. A + // workaround that cannot confirm the vendor it works around MUST NOT run -- + // the alternative is applying an NVIDIA-specific ordering hack to an + // unknown driver. + // + // The physical-device check follows the convention already in + // VulkanDeviceContext.h, which gates GetPhysicalDeviceMemoryProperties on + // `if (m_physDevice)` for the same reason. + if ((vkDevCtx.GetPhysicalDeviceProperties == nullptr) || + (vkDevCtx.getPhysicalDevice() == VK_NULL_HANDLE)) { + return false; + } + VkPhysicalDeviceProperties props{}; + vkDevCtx.GetPhysicalDeviceProperties(vkDevCtx.getPhysicalDevice(), &props); + return props.vendorID == 0x10DE; +} + +// Index of |device|'s entry, appending an empty one on its first guarded +// import. Caller holds g_importGuardMutex. An INDEX and not a reference or a +// pointer: the caller keeps using it across a nested VkEncImportExternalImage, +// and a reference into a std::vector is exactly the thing a push_back would +// invalidate. +size_t VkEncImportOrdinalGuardEntryIndexLocked(VkDevice device) +{ + for (size_t i = 0; i < g_importGuards.size(); i++) { + if (g_importGuards[i].device == device) { + return i; + } + } + VkEncImportOrdinalGuardEntry entry{}; + entry.device = device; + g_importGuards.push_back(entry); + return g_importGuards.size() - 1; +} + +} // namespace + +void VkEncResetImportOrdinalGuardReport() +{ + g_importGuardReport = VkEncImportOrdinalGuardReport{}; +#if defined(__linux__) + // requestedCount is a BUILD constant, not an outcome, so it is stamped + // here rather than on the paths that reach the guard: it must be + // readable even when the guard never runs. Zero off Linux, where the + // guard is not compiled -- and zero on Linux too now that the guard is + // retired, which costs a second property this field can carry. + // While the count was non-zero it also PROVED the library had written + // the caller's struct, every other field's "nothing happened" value + // being 0, which is what an untouched struct already holds. What is left + // of that proof is |state|: it is non-zero on every path the import + // itself reaches (NOT_APPLICABLE, NOT_NVIDIA, DISABLED, COMPLETE, + // INCOMPLETE). A registration that never reaches the import reports + // NOT_EVALUATED and is, at the disabled default, indistinguishable from a + // struct nothing wrote -- so a caller or a test that needs the write + // proved has to poison the struct first and assert the poison is gone. + // test/encoder-ext-import-guard C1 does exactly that. + g_importGuardReport.requestedCount = kVkEncImportOrdinalGuardCount; +#endif +} + +void VkEncGetImportOrdinalGuardReport(VkEncImportOrdinalGuardReport* outReport) +{ + if (outReport != nullptr) { + *outReport = g_importGuardReport; + } +} + +// Runs at the top of every import: a no-op on EVERY import at the retired +// default (count 0), and a no-op after the first one when armed. |osHandle| +// is only ever dup()'d here -- the caller's fd is untouched and its ownership +// rule is unaffected on every path. +static void VkEncEnsureImportOrdinalGuard( + const VulkanDeviceContext& vkDevCtx, + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t osHandle) +{ +#if defined(__linux__) + if (g_inImportOrdinalGuard) { + // A nested (sacrificial) import. It writes NO verdict: the record + // belongs to the OUTER, caller-visible import, and overwriting it + // here would report the guard's own recursion instead of what the + // guard achieved. + return; + } + // EVERY remaining exit names a verdict, and that is the point of this + // block rather than a tidiness preference. The two failure paths below + // return SILENTLY when the report they write to VkEncErr() is + // discarded -- and the shipping Chromium configuration discards it, by + // setting silenceStdio -- so a workaround that had degraded back to the + // defect was indistinguishable from one that was working. The verdict + // is the channel that survives that. See VkVideoEncoderImportGuardInfo. + if (VkEncImportOrdinalGuardDisabled()) { + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_DISABLED, 0, + VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + return; + } + if (desc.handleType != VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF) { + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_APPLICABLE, 0, + VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + return; + } + if (!VkEncIsNvidiaDevice(vkDevCtx)) { + // Measured on NVIDIA only. A workaround must not run where the defect + // it works around has never been observed. + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_NVIDIA, 0, + VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + return; + } + if (kVkEncImportOrdinalGuardCount == 0) { + // THE RETIRED DEFAULT, and DISABLED is the verdict on purpose. + // + // Falling through would reach the already-satisfied early return + // below, whose 0 >= 0 is true, and report COMPLETE -- which + // VkVideoEncoderImportGuardState, in vulkan_video_encoder_ext_internal.h, + // defines as "retainedCount == requestedCount, caller imports land at + // live-position requestedCount+1 or later". The + // arithmetic holds at 0 == 0 and the claim does not: nothing was + // moved anywhere. A verdict that says a workaround ran on a build + // that removed it is the exact failure this reporting channel was + // added to prevent, and it would travel straight into the embedder's + // per-registration log. + // + // DISABLED already means "the workaround is off deliberately", which + // is what a build-time count of 0 is. Reported BEFORE the registry + // lock, so a disabled build also creates no per-device entry for the + // release path to find. + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_DISABLED, 0, + VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + return; + } + std::lock_guard lock(g_importGuardMutex); + VkDevice device = vkDevCtx; + if (device == VK_NULL_HANDLE) { + // No device to hold a position on. Outside the guard's scope in the + // same sense a non-dma-buf handle type is, and reported the same + // way -- the import itself is about to fail on its own terms. + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_APPLICABLE, 0, + VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + return; + } + const size_t entryIdx = VkEncImportOrdinalGuardEntryIndexLocked(device); + if (g_importGuards[entryIdx].live >= kVkEncImportOrdinalGuardCount) { + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_COMPLETE, + g_importGuards[entryIdx].live, + VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + return; + } + while (g_importGuards[entryIdx].live < kVkEncImportOrdinalGuardCount) { + const int guardFd = dup((int)osHandle); + if (guardFd < 0) { + // Captured before anything else can clobber it; the verdict + // carries it, and errno is the whole diagnosis on this path + // (EMFILE under fd pressure is the realistic one). + const int dupErrno = errno; + VkEncErr() << "[EncoderExt] import-ordinal guard: dup failed (" + << dupErrno << "); the guard is INCOMPLETE at " + << g_importGuards[entryIdx].live << " of " + << kVkEncImportOrdinalGuardCount << std::endl; + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE, + g_importGuards[entryIdx].live, + VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED, dupErrno); + return; + } + VkEncImportedImage guard{}; + g_inImportOrdinalGuard = true; + const VkVideoEncoderStatusCode status = + VkEncImportExternalImage(vkDevCtx, desc, (uint64_t)guardFd, &guard); + g_inImportOrdinalGuard = false; + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + // The import consumed or closed guardFd itself on every exit. + VkEncErr() << "[EncoderExt] import-ordinal guard: sacrificial " + "import " + << (g_importGuards[entryIdx].live + 1) << " failed (" + << (int)status << "); the guard is INCOMPLETE -- this " + "build asked to shift the import-damage phase and " + "did not get the shift it asked for" << std::endl; + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE, + g_importGuards[entryIdx].live, status, 0); + return; + } + // Recorded BEFORE live is bumped, so live is never a count of + // handles the registry does not actually hold -- the release below + // walks exactly [0, live). + g_importGuards[entryIdx].images[g_importGuards[entryIdx].live] = guard; + g_importGuards[entryIdx].live++; + } + VkEncSetImportGuardVerdict( + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_COMPLETE, + g_importGuards[entryIdx].live, VK_VIDEO_ENCODER_STATUS_SUCCESS, 0); + VkEncErr() << "[EncoderExt] import-ordinal guard: " + << kVkEncImportOrdinalGuardCount + << " sacrificial dma-buf imports retained; caller imports now " + "start at live-position " + << (kVkEncImportOrdinalGuardCount + 1) + << ". This is a PHASE SHIFT of the driver's import-damage " + "pattern and not a repair of it; the measured damage rate " + "is unchanged. See the block above this function." + << std::endl; +#else + (void)vkDevCtx; + (void)desc; + (void)osHandle; +#endif +} +// The other end of the guard's lifetime. Contract and the reason the count is +// returned rather than only printed: vulkan_video_encoder_ext_internal.h. +uint32_t VkEncReleaseImportOrdinalGuard(const VulkanDeviceContext& vkDevCtx) +{ +#if defined(__linux__) + const VkDevice device = vkDevCtx; + if (device == VK_NULL_HANDLE) { + return 0; + } + // Taken off the registry under the lock, destroyed outside it. Removing + // it first is what makes a second call a no-op instead of a double free, + // and it means no other thread can find these handles while they are + // being destroyed. + VkEncImportOrdinalGuardEntry doomed{}; + { + std::lock_guard lock(g_importGuardMutex); + for (size_t i = 0; i < g_importGuards.size(); i++) { + if (g_importGuards[i].device != device) { + continue; + } + doomed = g_importGuards[i]; + g_importGuards.erase(g_importGuards.begin() + (ptrdiff_t)i); + break; + } + } + // NO QUEUE WAIT IS OWED HERE, and that is a property of what a guard is + // rather than an assumption about the caller. A guard image is imported + // and then never touched again -- it is never written into a command + // buffer, never bound, never transitioned, never submitted -- so + // VUID-vkDestroyImage-image-01000 ("must not be in use by the device") is + // satisfied by construction. The registrations destroyed above it in + // Deinitialize need the encoder release and the DeviceWaitIdle; these do + // not. + // + // Image before memory: freeing a VkDeviceMemory that a live VkImage is + // still bound to is VUID-vkFreeMemory-memory-00677. Same order + // DestroyResourceLocked uses. + for (uint32_t i = 0; i < doomed.live; i++) { + if (doomed.images[i].image != VK_NULL_HANDLE) { + vkDevCtx.DestroyImage(device, doomed.images[i].image, nullptr); + } + if (doomed.images[i].memory != VK_NULL_HANDLE) { + // This is also what closes the dup()'d dma-buf fd the guard took: + // from the vkAllocateMemory import chain onward the fd belongs to + // the VkDeviceMemory, and vkFreeMemory is its release (design + // section 2.3). Nothing close()s it here or anywhere else. + vkDevCtx.FreeMemory(device, doomed.images[i].memory, nullptr); + } + } + return doomed.live; +#else + (void)vkDevCtx; + return 0; +#endif +} +// ============== END dma-buf IMPORT-ORDINAL GUARD =========================== + +VkVideoEncoderStatusCode VkEncImportExternalImage( + const VulkanDeviceContext& vkDevCtx, + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t osHandle, + VkEncImportedImage* outImport) +{ + VkDevice device = vkDevCtx; + + // The import-ordinal guard, RETIRED BY DEFAULT: at the shipped count of 0 + // this returns immediately with the verdict DISABLED and takes no + // imports. It stays on the path so the knob remains reachable, reported + // and tested. When armed it shifts which import ordinals the driver + // damages -- not how many -- and is a no-op after the first call, and on + // every non-NVIDIA / non-dma-buf import. + VkEncEnsureImportOrdinalGuard(vkDevCtx, desc, osHandle); + + const bool isDrmModifier = + (desc.tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT); + + VkExternalMemoryImageCreateInfo extMemCI{ + VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_IMAGE_CREATE_INFO}; + VkExternalMemoryHandleTypeFlagBits vkHandleType = + VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT; + switch (desc.handleType) { + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF: + vkHandleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + break; + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_WIN32: + vkHandleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT; + break; + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_D3D11_TEXTURE: + vkHandleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_D3D11_TEXTURE_BIT; + break; + default: + break; + } + extMemCI.handleTypes = vkHandleType; + + VkImageCreateInfo imageCI{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + imageCI.pNext = &extMemCI; + imageCI.imageType = (desc.imageType == 0) ? VK_IMAGE_TYPE_2D + : desc.imageType; + imageCI.format = desc.format; + imageCI.extent = {desc.width, desc.height, 1}; + imageCI.mipLevels = (desc.mipLevels == 0) ? 1u : desc.mipLevels; + imageCI.arrayLayers = (desc.arrayLayers == 0) ? 1u : desc.arrayLayers; + imageCI.samples = + (desc.samples == 0) ? VK_SAMPLE_COUNT_1_BIT : desc.samples; + imageCI.tiling = desc.tiling; + // The exporter's usage, never a fabricated one. + imageCI.usage = desc.imageUsage; + imageCI.flags = desc.imageFlags; + imageCI.sharingMode = desc.sharingMode; + imageCI.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + // MUTABLE_FORMAT obliges us to declare the view formats + // (VUID-VkImageCreateInfo-tiling-02353) -- and the list must name the + // PLANE formats too, or the per-plane views the filter binds are invalid + // (VUID-VkImageViewCreateInfo-pNext-01585). See VkEncCollectViewFormats. + VkFormat viewFormats[kVkEncMaxViewFormats] = {}; + const uint32_t viewFormatCount = + VkEncCollectViewFormats(desc, viewFormats); + VkImageFormatListCreateInfo formatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + formatListCI.viewFormatCount = viewFormatCount; + formatListCI.pViewFormats = viewFormats; + + VkSubresourceLayout planeLayouts[VK_VIDEO_ENCODER_MAX_PLANES]{}; + VkImageDrmFormatModifierExplicitCreateInfoEXT drmExplicit{ + VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_EXPLICIT_CREATE_INFO_EXT}; + if (isDrmModifier) { + for (uint32_t i = 0; + (i < desc.planeCount) && (i < VK_VIDEO_ENCODER_MAX_PLANES); i++) { + planeLayouts[i].offset = desc.planeLayouts[i].offset; + // Must be 0 on the explicit path (VUID-...-size-02267). + planeLayouts[i].size = 0; + planeLayouts[i].rowPitch = desc.planeLayouts[i].rowPitch; + planeLayouts[i].arrayPitch = desc.planeLayouts[i].arrayPitch; + planeLayouts[i].depthPitch = desc.planeLayouts[i].depthPitch; + } + drmExplicit.drmFormatModifier = desc.drmFormatModifier; + drmExplicit.drmFormatModifierPlaneCount = desc.planeCount; + drmExplicit.pPlaneLayouts = planeLayouts; + drmExplicit.pNext = &extMemCI; + imageCI.pNext = &drmExplicit; + } + if ((desc.imageFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) { + formatListCI.pNext = imageCI.pNext; + imageCI.pNext = &formatListCI; + } + + VkResult result = + vkDevCtx.CreateImage(device, &imageCI, nullptr, &outImport->image); + if (result != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] register: vkCreateImage failed (" + << result << ")" << std::endl; + // Nothing was handed to the driver yet: the fd is the library's + // to close, on this exit like every other pre-handoff one. + VkEncConsumeOsHandle(desc.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + + VkMemoryRequirements memReqs{}; + vkDevCtx.GetImageMemoryRequirements(device, outImport->image, &memReqs); + + // Memory-type selection. The candidate SET is spec-constrained, not + // guessable; the preference ORDER within it is the design's policy + // (exporter's index if representable, else DEVICE_LOCAL). + // + // - DMA_BUF: the importable types for THIS fd are what + // vkGetMemoryFdPropertiesKHR reports, intersected with the image's + // requirements (VUID-VkMemoryAllocateInfo-memoryTypeIndex-00648). + // A type outside that mask imports "successfully" on NVIDIA and + // reads garbage where import-capable heaps are segregated. If the + // query is unavailable, the legacy heuristic runs as a LOGGED + // fallback, never silently (risk R-5). + // - OPAQUE_FD: querying is forbidden + // (VUID-vkGetMemoryFdPropertiesKHR-handleType-00674) and the import + // must reuse the exporter's own allocation parameters + // (VUID-VkMemoryAllocateInfo-allocationSize-01742), which only the + // descriptor can carry. + VkEncImportMemoryTypeRequest typeRequest{}; + typeRequest.requirementsMask = memReqs.memoryTypeBits; + typeRequest.exporterMask = desc.memoryTypeBits; + typeRequest.exporterIndex = desc.memoryTypeIndex; + +#if defined(__linux__) + if (vkHandleType == VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT) { + // Query only -- it does not consume the fd. The handoff is the + // vkAllocateMemory call below: until then a failure leaves the fd + // ours to close, and from that call on it never is again. + VkMemoryFdPropertiesKHR fdProps{ + VK_STRUCTURE_TYPE_MEMORY_FD_PROPERTIES_KHR}; + VkResult propsResult = VK_ERROR_EXTENSION_NOT_PRESENT; + if (vkDevCtx.GetMemoryFdPropertiesKHR != nullptr) { + propsResult = vkDevCtx.GetMemoryFdPropertiesKHR( + device, vkHandleType, (int)osHandle, &fdProps); + } + if (propsResult == VK_SUCCESS) { + if (fdProps.memoryTypeBits == 0) { + // A successful query that names NO importable type is a + // constraint, not an absence of one: the selector reads a + // zero mask as unconstrained and would pick a type this fd + // cannot legally import into (garbage reads where + // import-capable heaps are segregated). + VkEncErr() << "[EncoderExt] register: " + "vkGetMemoryFdPropertiesKHR reports no " + "importable memory type for this fd" + << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + // Still pre-handoff: nothing reached the driver, so the + // close is the library's obligation here. + VkEncConsumeOsHandle(desc.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED; + } + typeRequest.authoritativeMask = fdProps.memoryTypeBits; + } else { + outImport->heuristicSelections += 1; + VkEncErr() << "[EncoderExt] register: " + << "vkGetMemoryFdPropertiesKHR unavailable (" + << propsResult << "); selecting the memory type " + << "heuristically (R-5 fallback)" << std::endl; + } + } else { + // OPAQUE_FD: the exporter's exact allocation parameters or nothing. + // The type numbering is same-device -- a declared cross-device + // handle already failed DEVICE_MISMATCH at validation. + typeRequest.exporterIndexExact = VK_TRUE; + } +#elif defined(_WIN32) + if (vkHandleType == VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_BIT) { + // VUID-VkMemoryAllocateInfo-allocationSize-01743: exact match; the + // analogous query is forbidden for opaque handles + // (VUID-vkGetMemoryWin32HandlePropertiesKHR-handleType-00666). On + // WDDM a guessed type is a hard error, never a fallback: the OS + // validates allocation properties downstream, where nothing names + // the cause. + if (desc.memoryTypeIndex == UINT32_MAX) { + VkEncErr() << "[EncoderExt] register: an OPAQUE_WIN32 import " + << "requires the exporter's memoryTypeIndex" + << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + return VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED; + } + typeRequest.exporterIndexExact = VK_TRUE; + } else { + // D3D11_TEXTURE: an NT handle created outside the Vulkan API, so + // vkGetMemoryWin32HandlePropertiesKHR is the authoritative + // constraint (VUID-VkMemoryAllocateInfo-memoryTypeIndex-00645), + // and running without it is not an option on WDDM. + VkMemoryWin32HandlePropertiesKHR win32Props{ + VK_STRUCTURE_TYPE_MEMORY_WIN32_HANDLE_PROPERTIES_KHR}; + VkResult propsResult = VK_ERROR_EXTENSION_NOT_PRESENT; + if (vkDevCtx.GetMemoryWin32HandlePropertiesKHR != nullptr) { + propsResult = vkDevCtx.GetMemoryWin32HandlePropertiesKHR( + device, vkHandleType, (HANDLE)osHandle, &win32Props); + } + if (propsResult != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] register: " + << "vkGetMemoryWin32HandlePropertiesKHR failed (" + << propsResult << "); refusing to guess a memory " + << "type on WDDM" << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + return VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED; + } + if (win32Props.memoryTypeBits == 0) { + // A successful query that names NO importable type is a + // constraint, not an absence of one: the selector reads a zero + // mask as unconstrained and would pick a type this handle + // cannot legally import into. + VkEncErr() << "[EncoderExt] register: " + "vkGetMemoryWin32HandlePropertiesKHR reports no " + "importable memory type for this handle" + << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + return VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED; + } + typeRequest.authoritativeMask = win32Props.memoryTypeBits; + } +#endif + + VkPhysicalDeviceMemoryProperties memProps{}; + vkDevCtx.GetPhysicalDeviceMemoryProperties( + vkDevCtx.getPhysicalDevice(), &memProps); + + VkEncImportMemoryTypeChoice typeChoice{}; + VkEncSelectImportMemoryType(typeRequest, memProps, &typeChoice); + + if (typeChoice.outcome == VK_ENC_IMPORT_MEMTYPE_NONE) { + VkEncErr() << "[EncoderExt] register: no importable memory type: " + << "requirements mask 0x" << std::hex + << typeRequest.requirementsMask + << ", handle-properties mask 0x" + << typeRequest.authoritativeMask << std::dec + << ", exporter index " + << (int64_t)desc.memoryTypeIndex << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + // Still pre-handoff: nothing reached the driver, so the close is + // the library's obligation here. + VkEncConsumeOsHandle(desc.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_MEMORY_TYPE_UNSUPPORTED; + } + if (typeChoice.exporterIndexOverridden == VK_TRUE) { + outImport->exporterOverrides += 1; + VkEncErr() << "[EncoderExt] register: exporter memoryTypeIndex " + << desc.memoryTypeIndex << " is outside the importable " + << "mask; using type " << typeChoice.index << " instead" + << std::endl; + } + if ((typeRequest.exporterIndexExact == VK_TRUE) && + (typeChoice.outcome != VK_ENC_IMPORT_MEMTYPE_EXPORTER_INDEX)) { + // Opaque import without the exporter's index: the selection is a + // heuristic, and a heuristic here must never be silent. + outImport->heuristicSelections += 1; + VkEncErr() << "[EncoderExt] register: opaque import without the " + << "exporter's memoryTypeIndex; type " << typeChoice.index + << " selected heuristically (R-5 fallback)" << std::endl; + } + const uint32_t memoryTypeIndex = typeChoice.index; + + VkMemoryAllocateInfo allocInfo{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + // The exporter's size when it gave us one: deriving it from + // vkGetImageMemoryRequirements is wrong for dma-buf on NVIDIA. The + // header accepts 0 ("unknown") WITH A LOG, deliberately -- a silent + // fallback leaves no audit trail when that size mismatch bites. + if (desc.allocationSize == 0) { + VkEncErr() << "[EncoderExt] register: descriptor allocationSize is " + "0 (unknown); falling back to " + "vkGetImageMemoryRequirements size " << memReqs.size + << " -- prefer the exporter-reported size for dma-buf" + << std::endl; + } + allocInfo.allocationSize = + (desc.allocationSize != 0) ? desc.allocationSize : memReqs.size; + allocInfo.memoryTypeIndex = memoryTypeIndex; + + // A non-zero exporter size SMALLER than the image's requirement can + // never bind: the dedicated allocation below is bound at offset 0 and + // must cover the image (VUID-vkBindImageMemory-size-01049). Refuse it + // by name rather than flooring to memReqs.size -- the opaque arms + // demand the exporter's exact size + // (VUID-VkMemoryAllocateInfo-allocationSize-01742), and a dma-buf + // allocation larger than the buffer is equally unbindable. + if ((desc.allocationSize != 0) && (desc.allocationSize < memReqs.size)) { + VkEncErr() << "[EncoderExt] register: exporter allocationSize " + << desc.allocationSize << " is below the image's memory " + "requirement " << memReqs.size << "; a dedicated " + "allocation bound at offset 0 cannot cover the image" + << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + // Still pre-handoff: nothing reached the driver, so the close is + // the library's obligation here. + VkEncConsumeOsHandle(desc.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_ALLOCATION_SIZE_INVALID; + } + + VkMemoryDedicatedAllocateInfo dedicated{ + VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO}; + dedicated.image = outImport->image; + +#if defined(__linux__) + VkImportMemoryFdInfoKHR importFd{ + VK_STRUCTURE_TYPE_IMPORT_MEMORY_FD_INFO_KHR}; + importFd.handleType = vkHandleType; + importFd.fd = (int)osHandle; + importFd.pNext = &dedicated; + allocInfo.pNext = &importFd; +#elif defined(_WIN32) + VkImportMemoryWin32HandleInfoKHR importWin32{ + VK_STRUCTURE_TYPE_IMPORT_MEMORY_WIN32_HANDLE_INFO_KHR}; + importWin32.handleType = vkHandleType; + importWin32.handle = (HANDLE)osHandle; + importWin32.pNext = &dedicated; + allocInfo.pNext = &importWin32; +#else + // Neither an fd import nor a Win32 handle import exists on this target, + // and referencing the Win32 structure here is what made this block fail to + // compile on macOS/Fuchsia -- NOT on Windows, which is the platform the + // bare #else looked like it was for. Refuse, releasing what we already + // own: the image created above and the caller's OS handle. + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + VkEncConsumeOsHandle(desc.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; +#endif + + // THE HANDOFF. From this call on -- success or failure -- the fd is + // never the library's to close: NVIDIA consumes it even when the + // import allocation FAILS (Chromium documents the behavior at + // gpu/vulkan/vulkan_memory.cc and splits its own import path on + // exactly this line, gpu/vulkan/vulkan_image.h InitializeResult). + // Closing it again here or in any caller is a double close of an fd + // number the process may already have recycled. + result = vkDevCtx.AllocateMemory(device, &allocInfo, nullptr, + &outImport->memory); + if (result != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] register: import allocation failed (" + << result << ")" << std::endl; + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->image = VK_NULL_HANDLE; + outImport->result = VK_ENC_IMPORT_FAILED_AFTER_ALLOCATE; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + + result = vkDevCtx.BindImageMemory(device, outImport->image, + outImport->memory, 0); + if (result != VK_SUCCESS) { + // The allocation succeeded, so the fd belongs to |memory| and + // this vkFreeMemory is what releases it. Still AFTER_ALLOCATE: + // no close() anywhere. + vkDevCtx.FreeMemory(device, outImport->memory, nullptr); + vkDevCtx.DestroyImage(device, outImport->image, nullptr); + outImport->memory = VK_NULL_HANDLE; + outImport->image = VK_NULL_HANDLE; + outImport->result = VK_ENC_IMPORT_FAILED_AFTER_ALLOCATE; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + + outImport->allocationSize = allocInfo.allocationSize; + outImport->result = VK_ENC_IMPORT_SUCCESS; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +// Thin member wrapper over VkEncImportExternalImage: fold the import's +// memory-type telemetry into the session counters and adopt the imported +// objects into |slot|. Caller holds m_resourceMutex. No fd handling at this +// layer ON ANY PATH -- the import already applied the ownership split (it +// closes the fd itself exactly when the failure preceded the +// vkAllocateMemory handoff), so a close here or in any caller above would be +// a double close. +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::ImportImageLocked( + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t osHandle, RegisteredImage& slot) +{ + VkEncImportedImage imported; + const VkVideoEncoderStatusCode status = + VkEncImportExternalImage(m_vkDevCtx, desc, osHandle, &imported); + if (imported.heuristicSelections != 0) { + m_importMemTypeHeuristicSelections.fetch_add( + imported.heuristicSelections, std::memory_order_relaxed); + } + if (imported.exporterOverrides != 0) { + m_importMemTypeExporterOverrides.fetch_add( + imported.exporterOverrides, std::memory_order_relaxed); + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return status; + } + slot.image = imported.image; + slot.memory = imported.memory; + slot.ownsImage = true; + slot.importedAllocSize = imported.allocationSize; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +// The image usage a registration ACTUALLY carries. +// +// Factored out of BuildRegisteredViewLocked because it now has TWO readers and +// they must not drift: the view builder, and the content probe's arm decision. +// The arm decision cannot read slot.imageUsage instead -- BuildRegisteredViewLocked +// is the only writer of that field and RegisterImageResource SKIPS it entirely on a +// null-backend session (`if (m_nullBackend == nullptr)`), so every device-free +// registration would see 0 and answer NOT_APPLICABLE. Reading the DESCRIPTOR is +// available on every path, which is what a registration-time predicate needs. +static VkImageUsageFlags VkEncResolveRegistrationUsage( + const VkVideoEncoderExternalImageDescriptor& desc) +{ + // Undeclared usage presumes ONLY the access the staging copy needs. See + // BuildRegisteredViewLocked for why it is not VIDEO_ENCODE_SRC. + // Both arms typed as the flags type the function returns. A conditional + // whose arms are an enumerator and its own flags type has no common type + // the language will pick for us, so it promotes to whatever it can and + // the result stops being obviously VkImageUsageFlags. + return (desc.imageUsage == 0) + ? static_cast(VK_IMAGE_USAGE_TRANSFER_SRC_BIT) + : desc.imageUsage; +} + +// DIRECT and only DIRECT -- the registration-time routing predicate, as a pure +// function of the descriptor. +// +// It lived inline in BuildRegisteredViewLocked, which is skipped wholesale on a +// null-backend session, so slot.encodeCapable was left at its default there and +// the device-free suite could not observe the routing decision at ALL. That is +// not a cosmetic gap: encodeCapable is what used to decide whether the content +// probe armed, so the probe's blindness to directly-encodable (block-linear) +// imports was unobservable in the only suite that runs on every host. +static bool VkEncRegistrationIsDirectlyEncodable( + const VkVideoEncoderExternalImageDescriptor& desc) +{ + // The taxonomy's first rung, read from the one classifier so "encodable as + // it stands" cannot drift from "encodable after a conversion". + return (VkEncClassifyInput(desc.format, desc.colorModel) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT) && + (desc.tiling != VK_IMAGE_TILING_LINEAR) && + ((VkEncResolveRegistrationUsage(desc) & + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR) != 0); +} + +// Build the once-per-registration wrapper + view + pool node for |slot| +// from the descriptor's TRUE properties. The import cost -- and the view +// creation -- belong to registration, never to the submit path. This +// replaces the per-frame WrapExternalImage on the registered +// path, and with it the fabricated STORAGE|SAMPLED plane views +// (VUID-VkImageViewCreateInfo-pNext-02662) v1 invented on images it could +// not verify: every view built here carries a usage that is a strict subset +// of what the image actually carries, and per-plane views are built ONLY +// when the descriptor declared MUTABLE_FORMAT and granted STORAGE. Nothing +// on the encode path reads a plane view (the codecs consume +// GetPictureResourceInfo()/GetImageView()); the preprocess compute filter +// does, which is why the plane views exist at all now that the filter is +// reachable for external input. +// +// Caller holds m_resourceMutex. On failure every Vulkan object created for +// the slot -- including, on the import arm, the image and memory +// ImportImageLocked created -- is freed and the slot's handles are zeroed, +// so the caller returns the status without touching them. The fd is NOT +// closed on that path: after a successful import Vulkan owns it, and the +// vkFreeMemory performed by dropping the owning wrapper is what releases it. +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::BuildRegisteredViewLocked( + const VkVideoEncoderExternalImageDescriptor& desc, RegisteredImage& slot) +{ + // Metadata mirrors the true create info -- the same fields + // ImportImageLocked created the image with. For a VK_IMAGE registration + // with no declared usage, only TRANSFER_SRC is presumed -- see the + // usage selection below. + VkImageCreateInfo imageCI{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + imageCI.imageType = (desc.imageType == 0) ? VK_IMAGE_TYPE_2D + : desc.imageType; + imageCI.format = desc.format; + imageCI.extent = {desc.width, desc.height, 1}; + imageCI.mipLevels = (desc.mipLevels == 0) ? 1u : desc.mipLevels; + imageCI.arrayLayers = (desc.arrayLayers == 0) ? 1u : desc.arrayLayers; + imageCI.samples = + (desc.samples == 0) ? VK_SAMPLE_COUNT_1_BIT : desc.samples; + imageCI.tiling = desc.tiling; + imageCI.flags = desc.imageFlags; + imageCI.sharingMode = desc.sharingMode; + + const VkImageUsageFlags usage = VkEncResolveRegistrationUsage(desc); + if (desc.imageUsage == 0) { + // VK_IMAGE arm, usage undeclared (legal; see ValidateImageDescriptor). + // Presume ONLY the access the staging copy needs. The previous shape + // fabricated VIDEO_ENCODE_SRC for OPTIMAL tiling, and encodeCapable + // below derived from that fabricated bit -- an undeclared-usage + // registration could route DIRECT and vkCmdEncodeVideoKHR would read + // an image that may carry no encode usage: v1's fabricated-flags + // defect, narrowed to one arm. An undeclared usage now always routes + // STAGED; DIRECT requires the caller to declare a usage that + // includes VIDEO_ENCODE_SRC. + // + // The substitution itself now lives in VkEncResolveRegistrationUsage + // above, so the content probe's arm decision can apply the SAME rule + // without depending on this function having run -- which on a + // null-backend session it has not. + } + imageCI.usage = usage; + + // Already resolved by RegisterImageResource, from the same descriptor and + // through the same two helpers, BEFORE this function is reached -- so a + // null-backend session (which never reaches it) gets the same answer. + // Asserted rather than recomputed: two copies of a routing predicate is + // how "the path the caller was told about" and "the path its frames take" + // drift apart. + assert(slot.imageUsage == usage); + assert(slot.encodeCapable == VkEncRegistrationIsDirectlyEncodable(desc)); + + VkSharedBaseObj imageResource; + VkResult result; + if (slot.ownsImage) { + // Import arm: hand image+memory lifetime to the refcounted wrapper, + // so slot retirement and per-frame refs share one release path. + result = VkImageResource::CreateFromImport( + &m_vkDevCtx, slot.image, slot.memory, slot.importedAllocSize, + &imageCI, imageResource); + if (result != VK_SUCCESS) { + // The wrapper was never built: the raw handles are still ours. + VkDevice device = m_vkDevCtx; + m_vkDevCtx.DestroyImage(device, slot.image, nullptr); + m_vkDevCtx.FreeMemory(device, slot.memory, nullptr); + slot.image = VK_NULL_HANDLE; + slot.memory = VK_NULL_HANDLE; + slot.ownsImage = false; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + } else { + // VK_IMAGE arm: non-owning wrapper; the caller frees its image + // after Unregister, per the header's precondition. + result = VkImageResource::CreateFromExternal( + &m_vkDevCtx, slot.image, slot.memory, &imageCI, imageResource); + if (result != VK_SUCCESS) { + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + } + + VkImageSubresourceRange subresRange{}; + subresRange.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + subresRange.levelCount = 1; + subresRange.layerCount = 1; + + // One entry point for both arms. Create() derives what this registration + // needs from the image itself, so nothing here has to be pre-computed: + // * staging arm -- a transfer-only image carries no view-compatible + // usage, so no view is created and the slot holds the resource alone. + // The copy path (vkCmdCopyImage + barriers) reads the raw VkImage. + // * encodable arm -- the combined view's usage is the image's own usage + // minus STORAGE|SAMPLED (a sampled YCbCr combined view would demand a + // conversion object, VUID-VkImageViewCreateInfo-usage-06415). + // + // FILTER arm: when the EXPORTER declared MUTABLE_FORMAT and granted + // STORAGE, ask for the per-plane STORAGE views as well. Without them + // GetPlaneImageView() answers VK_NULL_HANDLE for every registered + // external image, which is precisely why the compute filter could not + // bind one and the external-input bypass had nothing to lift. The + // condition is the descriptor's, not a guess: MUTABLE_FORMAT is what + // makes a plane view legal at all + // (VUID-VkImageViewCreateInfo-image-01762) and STORAGE is what the view + // may carry (VUID-VkImageViewCreateInfo-pNext-02662), and EXTENDED_USAGE + // is what makes that usage validate against the PLANE formats rather than + // the image format (measured: MUTABLE_FORMAT alone answers + // VK_ERROR_FORMAT_NOT_SUPPORTED from the capability query). + // + // CAVEAT, stated because it is easy to read the opposite: Create() + // re-tests MUTABLE_FORMAT, but on the create-info this function + // SYNTHESIZED from |desc| and handed to CreateFromExternal -- so it + // re-tests the DESCRIPTOR's claim, not the image's truth. On the IMPORT + // arm the two coincide, because vkCreateImage used the same synthesized + // flags and would have failed otherwise. On the VK_IMAGE arm there is no + // vkCreateImage to catch a false claim: a caller that declares + // MUTABLE_FORMAT for a VkImage created without it gets plane views the + // image never validated, which the NVIDIA driver accepts silently. + // Closing that needs a capability query against the caller's real image; + // it is not closed here. + // + // Combined view for that arm: the image's usage minus STORAGE|SAMPLED, + // the same subset the no-override arm of Create() computes -- a sampled + // or storage COMBINED view of a YCbCr format needs a conversion object + // (VUID-VkImageViewCreateInfo-usage-06415). A registration whose whole + // usage is STORAGE leaves that subset empty and so has no legal combined + // view; it takes the plain arm rather than being handed a zero usage the + // 6-argument overload would silently widen back to the image's own. + const VkImageUsageFlags combinedUsage = + usage & ~(VK_IMAGE_USAGE_STORAGE_BIT | VK_IMAGE_USAGE_SAMPLED_BIT); + if (VkEncDescriptorPermitsPlaneStorageViews(desc, usage) && + (combinedUsage != 0)) { + result = VkImageResourceView::Create( + &m_vkDevCtx, imageResource, subresRange, + VK_IMAGE_USAGE_STORAGE_BIT, VK_NULL_HANDLE, combinedUsage, + slot.imageView); + } else { + result = VkImageResourceView::Create( + &m_vkDevCtx, imageResource, subresRange, slot.imageView); + } + if (result != VK_SUCCESS) { + VkEncErr() << "[EncoderExt] register: view creation failed (" + << result << ")" << std::endl; + // Dropping the (owning, on the import arm) wrapper frees the image + // and memory -- and with them the imported fd. + imageResource = nullptr; + slot.image = VK_NULL_HANDLE; + slot.memory = VK_NULL_HANDLE; + slot.ownsImage = false; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + + // What was actually BUILT, not what was asked for. Create() answers a + // plane count of 0 when it declined the plane views (no MUTABLE_FORMAT, + // or a usage that could not carry them), and routing must key off the + // views that exist -- a slot claiming plane views it does not have would + // route a frame to a filter that then binds VK_NULL_HANDLE. + slot.planeStorageViews = + slot.imageView && (slot.imageView->GetNumberOfPlanes() >= 2); + // The single-plane parallel. GetImageView() is checked EXPLICITLY rather + // than inferred from the wrapper existing, because Create() answers + // VK_NULL_HANDLE for the combined view whenever the image carries no + // view-compatible usage (VUID-VkImageViewCreateInfo-image-04441) and + // still returns a wrapper. A slot claiming a storage read it cannot + // perform would route a frame to a filter that then binds + // VK_NULL_HANDLE -- the same failure the plane-count line above prevents, + // one view over. + // + // The STORAGE bit is re-tested here on the RESOLVED usage rather than + // trusted from the registration gate: on the VK_IMAGE arm the descriptor + // may declare 0 and |usage| is then the legacy set, so the gate's answer + // and the built view's usage are two different facts. + slot.storageReadView = + slot.imageView && + (slot.imageView->GetImageView() != VK_NULL_HANDLE) && + (slot.imageView->GetNumberOfPlanes() == 1) && + (VkEncInputFormatPlaneCount(desc.format) == 1u) && + ((usage & VK_IMAGE_USAGE_STORAGE_BIT) != 0); + + const VkImageLayout initialLayout = + (desc.defaultLayout != VK_IMAGE_LAYOUT_UNDEFINED) + ? desc.defaultLayout + : VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; // the legacy wrap's default + result = VulkanVideoImagePoolNode::CreateExternal( + &m_vkDevCtx, slot.imageView, initialLayout, slot.node); + if (result != VK_SUCCESS) { + slot.imageView = nullptr; + imageResource = nullptr; + slot.image = VK_NULL_HANDLE; + slot.memory = VK_NULL_HANDLE; + slot.ownsImage = false; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +std::vector +VulkanVideoEncoderExtImpl::EnumerateDrmModifiers(VkFormat format) const +{ + std::vector modifiers; + if (!m_initialized || + (m_vkDevCtx.FindRequiredDeviceExtension( + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME) == nullptr)) { + return modifiers; + } + VkDrmFormatModifierPropertiesListEXT modifierList{ + VK_STRUCTURE_TYPE_DRM_FORMAT_MODIFIER_PROPERTIES_LIST_EXT}; + VkFormatProperties2 formatProps{VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_2}; + formatProps.pNext = &modifierList; + m_vkDevCtx.GetPhysicalDeviceFormatProperties2( + m_vkDevCtx.getPhysicalDevice(), format, &formatProps); + if (modifierList.drmFormatModifierCount == 0) { + return modifiers; + } + modifiers.resize(modifierList.drmFormatModifierCount); + modifierList.pDrmFormatModifierProperties = modifiers.data(); + m_vkDevCtx.GetPhysicalDeviceFormatProperties2( + m_vkDevCtx.getPhysicalDevice(), format, &formatProps); + modifiers.resize(modifierList.drmFormatModifierCount); + return modifiers; +} + +bool VulkanVideoEncoderExtImpl::ModifierWouldRegister( + const VkVideoEncoderExternalImageDescriptor& desc, + uint64_t modifier) const +{ + // No usage, no registerable image: the OS-import validation refuses + // imageUsage == 0 (USAGE_INSUFFICIENT -- "0 is INVALID", design + // section 2.2a) before any modifier is judged, so no modifier can make + // this descriptor register -- and asking the device would itself be + // spec-invalid (VUID-VkPhysicalDeviceImageFormatInfo2-usage- + // requiredbitmask demands a non-zero usage). Reachable with 0 only + // through the query surface: QueryImageSupport fills the details + // struct INDEPENDENTLY of the validation verdict, deliberately, so + // this predicate cannot rely on the register path's earlier gate. + if (desc.imageUsage == 0) { + return false; + } + + // CONCURRENT sharing demands a queue-family list the descriptor cannot + // carry, so the validation refuses the descriptor by name + // (SHARING_MODE_UNSUPPORTED) before any image is created. No modifier + // can change that answer, and advertising direct modifiers for a + // descriptor that can never register would be a false promise. + if (desc.sharingMode == VK_SHARING_MODE_CONCURRENT) { + return false; + } + + VkExternalMemoryHandleTypeFlagBits extHandleType = + VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT; + if (desc.handleType == VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF) { + extHandleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + } + VkPhysicalDeviceExternalImageFormatInfo extInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_IMAGE_FORMAT_INFO}; + extInfo.handleType = extHandleType; + + VkPhysicalDeviceImageDrmFormatModifierInfoEXT modifierInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_DRM_FORMAT_MODIFIER_INFO_EXT}; + modifierInfo.drmFormatModifier = modifier; + modifierInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + modifierInfo.pNext = &extInfo; + + VkFormat viewFormats[kVkEncMaxViewFormats] = {}; + VkImageFormatListCreateInfo formatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + formatListCI.viewFormatCount = VkEncCollectViewFormats(desc, viewFormats); + formatListCI.pViewFormats = viewFormats; + + VkPhysicalDeviceImageFormatInfo2 imageInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_FORMAT_INFO_2}; + imageInfo.pNext = &modifierInfo; + if ((desc.imageFlags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) { + // Mirrors ImportImageLocked: MUTABLE_FORMAT obliges the view-format + // declaration (VUID-VkImageCreateInfo-tiling-02353). + formatListCI.pNext = imageInfo.pNext; + imageInfo.pNext = &formatListCI; + } + imageInfo.format = desc.format; + imageInfo.type = (desc.imageType == 0) ? VK_IMAGE_TYPE_2D + : desc.imageType; + imageInfo.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + imageInfo.usage = desc.imageUsage; + imageInfo.flags = desc.imageFlags; + + VkExternalImageFormatProperties extProps{ + VK_STRUCTURE_TYPE_EXTERNAL_IMAGE_FORMAT_PROPERTIES}; + VkImageFormatProperties2 outProps{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_PROPERTIES_2}; + outProps.pNext = &extProps; + + if (m_vkDevCtx.GetPhysicalDeviceImageFormatProperties2( + m_vkDevCtx.getPhysicalDevice(), &imageInfo, &outProps) != + VK_SUCCESS) { + return false; + } + if ((extProps.externalMemoryProperties.externalMemoryFeatures & + VK_EXTERNAL_MEMORY_FEATURE_IMPORTABLE_BIT) == 0) { + return false; + } + return (VkEncDescriptorWithinCreationLimits( + desc, outProps.imageFormatProperties) == VK_TRUE); +} + +// Per-modifier tiling features for DRM tilings -- never +// optimalTilingFeatures, which is simply wrong for DRM-tiled images -- and +// the matching tiling's features otherwise. STORAGE_IMAGE is the same bit in +// the 32- and 64-bit format-feature flag spaces, so the 32-bit list carried +// by VK_EXT_image_drm_format_modifier itself answers without a +// VK_KHR_format_feature_flags2 dependency. +// +// This is the registration predicate for the single-plane arm. One body, so +// the query -- which runs the same gate -- and the gate cannot answer +// differently. +bool VulkanVideoEncoderExtImpl::DeviceCanStorageRead( + const VkVideoEncoderExternalImageDescriptor& desc) const +{ + // Same guard FillImageSupportDetails applied: a null-backend session + // reports initialized with no device, and this dispatches through PFNs + // only a real device load populates. + if (!m_initialized || + (m_vkDevCtx.getPhysicalDevice() == VK_NULL_HANDLE) || + (desc.format == VK_FORMAT_UNDEFINED)) { + return false; + } + if (desc.tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT) { + // Unknowable without a modifier, and "unknowable" must answer FALSE + // on a gate: a true here would admit a registration on a modifier + // that may carry no STORAGE at all. + if (desc.hasDrmFormatModifier != VK_TRUE) { + return false; + } + for (const auto& props : EnumerateDrmModifiers(desc.format)) { + if (props.drmFormatModifier == desc.drmFormatModifier) { + return (props.drmFormatModifierTilingFeatures & + VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0; + } + } + // A modifier this device does not report at all. + return false; + } + VkFormatProperties2 formatProps{VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_2}; + m_vkDevCtx.GetPhysicalDeviceFormatProperties2( + m_vkDevCtx.getPhysicalDevice(), desc.format, &formatProps); + const VkFormatFeatureFlags features = + (desc.tiling == VK_IMAGE_TILING_LINEAR) + ? formatProps.formatProperties.linearTilingFeatures + : formatProps.formatProperties.optimalTilingFeatures; + return (features & VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0; +} + +void VulkanVideoEncoderExtImpl::FillImageSupportDetails( + const VkVideoEncoderExternalImageDescriptor& desc, + VkVideoEncoderImageSupportDetails* details) const +{ + details->directModifierCount = 0; + // getPhysicalDevice() as well as m_initialized: a null-backend session + // (internal header) reports initialized with no device, and both detail + // queries below dispatch through PFNs only a real device load populates. + // The struct was zeroed above, so a device-free session honestly reports + // "no details" instead of crashing. + if (!m_initialized || + (m_vkDevCtx.getPhysicalDevice() == VK_NULL_HANDLE) || + (desc.format == VK_FORMAT_UNDEFINED) || + (desc.width == 0) || (desc.height == 0)) { + return; + } + + // directModifiers: the renegotiation list, meaningful only for an + // OS-handle import (VK_IMAGE never re-imports; the Win32 arms are + // reserved). + if ((desc.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD) || + (desc.handleType == VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF)) { + for (const auto& props : EnumerateDrmModifiers(desc.format)) { + if (details->directModifierCount >= + VK_VIDEO_ENCODER_MAX_DIRECT_MODIFIERS) { + break; + } + if (ModifierWouldRegister(desc, props.drmFormatModifier)) { + details->directModifiers[details->directModifierCount++] = + props.drmFormatModifier; + } + } + } +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::ValidateImageDescriptor( + const VkVideoEncoderExternalImageDescriptor& descriptor) const +{ + if (descriptor.sType != + VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + // Nothing chains onto this descriptor today, so anything chained is an + // extension this build does not understand: refuse it rather than + // import an image the caller believes it described more precisely. + // Living in the shared predicate, the refusal is the same through + // QueryImageSupport and RegisterImageResource by construction. + if (descriptor.pNext != nullptr) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + switch (descriptor.handleType) { + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD: + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF: +#if !defined(__linux__) + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; +#endif + break; + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_WIN32: + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_D3D11_TEXTURE: +#if !defined(_WIN32) + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; +#else + break; +#endif + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE: + break; + default: + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; + } + + // 0 is not a default on an import: the library creates the VkImage there + // and must never grant itself access the exporter did not give -- v1 + // fabricated STORAGE|SAMPLED|... on an image it could not verify. + // + // A VK_IMAGE registration is different in kind: the caller created the + // image, we do not re-create it, and the field is neither used nor + // verifiable. Requiring it would only teach callers to write down a + // plausible constant, which is the habit this field exists to break. + if ((descriptor.handleType != + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE) && + (descriptor.imageUsage == 0)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_USAGE_INSUFFICIENT; + } + + // The import re-creates the image from this descriptor, and CONCURRENT + // sharing demands a queue-family list the descriptor cannot carry + // (VUID-VkImageCreateInfo-sharingMode-00942): the create would run + // spec-invalid. A VK_IMAGE registration is different in kind: the + // caller's own image was created with whatever list it needed, and + // nothing is re-created there. + if ((descriptor.handleType != + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE) && + (descriptor.sharingMode == VK_SHARING_MODE_CONCURRENT)) { + VkEncErr() << "[EncoderExt] register: sharingMode CONCURRENT is not " + "importable -- the descriptor carries no queue-family " + "list, so the re-created image could not be " + "spec-valid; export with EXCLUSIVE sharing" + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_SHARING_MODE_UNSUPPORTED; + } + + // A colour-model DECLARATION must be readable against the format before + // anything is read from it. VkEncResolveColorModel answers + // VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT for a (format, colorModel) + // pair that cannot be reconciled -- RGB declared over a Y'CbCr format, or + // Y'CbCr declared over an RGBA layout that carries no packed 4:4:4 + // reading -- and there is no route to choose from an answer that means + // "the caller stated two things that cannot both be true". + // + // Judged HERE, from the descriptor alone, because that is what it is: the + // question needs no session, no device and no negotiated encode format, + // so QueryImageSupport and RegisterImageResource answer it identically, + // and on every session the same descriptor is offered to. Deferring it to + // the point where the route is picked is what lets a contradictory + // declaration REGISTER and then quietly take the staging copy -- an + // accepted registration that costs a copy the caller never asked for and + // has no way to see. A negotiation interface exists to make exactly that + // answerable before the allocation is committed. + // + // A declaration that AGREES with its format resolves to itself and + // passes. FROM_FORMAT -- what a zero-initialised descriptor says, and the + // ordinary case -- reads the model off the format and passes for every + // format the taxonomy places. + if (VkEncResolveColorModel(descriptor.format, descriptor.colorModel) == + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) { + VkEncErr() << "[EncoderExt] register: declared colorModel " + << (uint32_t)descriptor.colorModel + << " contradicts format " << (uint32_t)descriptor.format + << "; the two cannot both be true and nothing here can " + "know which was meant. Declare the colour model the " + "format carries, or leave the field " + "VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT to read it " + "off the format." << std::endl; + // The DECLARATION is what is refused. The format itself may be + // perfectly encodable -- NV12 declared RGB reaches here -- so the + // code names the field the caller has to change. + return VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED; + } + + // An RGBA input is not "unsupported" -- it is convertible, and the + // distinction is what lets a client fix it rather than give up. + // + // The SESSION's predicate, not the taxonomy's. A 3-plane 4:2:0 input is + // convertible in principle and encodable in practice only on a session + // whose compute filter is active, so on any other session the honest + // answer here is still CONVERSION_REQUIRED -- and it must be, because the + // gates below this one are what keep a 3-plane frame away from the + // transfer copy that hangs the GPU on it. + // + // Asked under the model the DESCRIPTOR declares. The gate above + // established that the declaration can be read against the format; this + // one is what reads it. A descriptor is judged as what it says it is, so + // that the answer a producer negotiates is an answer about the frames it + // intends to hand in -- and the packed 4:4:4 layouts are why that has to + // be said, because they ride RGBA format enumerants and the format alone + // cannot separate them from an R'G'B' image. + if (SupportsFormat(descriptor.format, descriptor.colorModel) != VK_TRUE) { + return VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED; + } + + if ((descriptor.width == 0) || (descriptor.height == 0)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_PLANE_LAYOUT_INVALID; + } + if (descriptor.tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT) { + if ((descriptor.planeCount == 0) || + (descriptor.planeCount > VK_VIDEO_ENCODER_MAX_PLANES)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_PLANE_LAYOUT_INVALID; + } + // Registration binds ONE memory object, so every plane must live in + // the single allocation named by the handle. What is PROVABLE here + // is only the necessary half of that: two planes cannot both begin + // at the same byte of one allocation, so equal offsets describe an + // image that cannot exist in a single memory object. The common + // shape that trips it is exactly the one worth refusing -- a + // disjoint (multi-buffer-object) producer writes offset 0 for every + // plane, because each starts at the base of its own buffer. + // + // This is NOT a disjointness test, and the previous rule here -- + // "offset == 0 for any plane > 0" -- was wrong to be read as one, in + // both directions. It missed a disjoint producer whose chroma plane + // sits at a non-zero offset inside its OWN buffer (accepted, then + // silently wrong chroma: the very failure it claimed to prevent), + // and it refused a legal single-BO image whose planes are packed in + // a different order, where plane 0 follows plane 1. + // + // No arithmetic on ONE fd can do better. Disjointness is a question + // about the planes' PROVENANCE, and this entry point is handed a + // single handle; the descriptor simply does not carry the answer. + // The sufficient test therefore belongs to the caller, which holds + // one fd per plane and can compare their identity directly -- see + // the planeLayouts[] contract in vulkan_video_encoder_ext.h. + // Chromium's VulkanVideoEncodeAccelerator runs it (fstat st_dev/ + // st_ino over gfx::NativePixmapHandle::planes[]) before it builds + // this descriptor at all, and routes a disjoint pixmap to the + // staging copy instead of registering it. + for (uint32_t i = 0; i < descriptor.planeCount; i++) { + for (uint32_t j = i + 1; j < descriptor.planeCount; j++) { + if (descriptor.planeLayouts[i].offset == + descriptor.planeLayouts[j].offset) { + VkEncErr() << "[EncoderExt] register: planes " << i + << " and " << j << " both begin at offset " + << descriptor.planeLayouts[i].offset + << "; one memory object cannot hold two planes " + "at one offset (a disjoint producer " + "describes every plane at offset 0), and " + "this interface binds a single handle." + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_PLANE_LAYOUT_INVALID; + } + } + } + if (descriptor.hasDrmFormatModifier != VK_TRUE) { + // Tiling says modifier, the descriptor supplies none: the caller + // would otherwise get modifier 0 (LINEAR) by accident, which + // NVIDIA refuses for VIDEO_ENCODE_SRC on multiplanar YCbCr. + return VK_VIDEO_ENCODER_STATUS_ERROR_MODIFIER_UNSUPPORTED; + } + } + + // Multi-GPU: a mismatched handle fails HERE with a name, instead of + // failing the import later with a driver error nobody can act on. + // Everything above is a property of the descriptor alone and is + // answered the same with or without a session. Everything below needs a + // physical device, so the initialization gate belongs here rather than at + // the top: a caller with a malformed descriptor learns what is wrong with + // it instead of only that it called too early. + if (!m_initialized) { + return VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED; + } + + // A device that cannot service the import must say so BY NAME here, at + // the negotiation point -- not later at vkAllocateMemory, and never by + // running spec-invalid. What is checkable is stated honestly: on the + // library's own device this set is what vkCreateDevice enabled; on a + // caller's device Vulkan cannot report enablement, so this is + // physical-device support, and enablement is the device creator's to + // audit (Chromium's accelerator does exactly that). + const bool isOsImport = + (descriptor.handleType != + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE); + if (isOsImport) { + const char* missing = nullptr; +#if defined(__linux__) + if (m_vkDevCtx.FindRequiredDeviceExtension( + VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME) == nullptr) { + missing = VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME; + } else if ((descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF) && + (m_vkDevCtx.FindRequiredDeviceExtension( + VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME) == + nullptr)) { + missing = VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME; + } else if (m_vkDevCtx.FindRequiredDeviceExtension( + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME) == + nullptr) { + // An OS-handle import that declares nothing still DERIVES + // FOREIGN, so the staging/encode acquire barrier can need the + // extension. Required unconditionally rather than only for a + // declared/derived FOREIGN: residency is a REGISTRATION-time + // field but the acquire is chosen per frame, and this gate runs + // before the slot exists. Refusing here costs a caller nothing + // real -- every device that can import an OS handle for video + // encode exposes VK_EXT_queue_family_foreign -- whereas + // admitting the registration and discovering the gap at the + // first barrier is a spec-invalid submit. + missing = VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME; + } else if ((descriptor.tiling == + VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT) && + (m_vkDevCtx.FindRequiredDeviceExtension( + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME) == + nullptr)) { + missing = VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME; + } +#elif defined(_WIN32) + // The Win32 arms are reserved through M8; when they open, the + // import needs the external-memory extension on the session device. + if (m_vkDevCtx.FindRequiredDeviceExtension( + VK_KHR_EXTERNAL_MEMORY_WIN32_EXTENSION_NAME) == nullptr) { + missing = VK_KHR_EXTERNAL_MEMORY_WIN32_EXTENSION_NAME; + } +#endif + if (missing != nullptr) { + VkEncErr() << "[EncoderExt] register: the session device lacks " + << missing << ", required for this import" + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_EXTENSION_MISSING; + } + } else if ((descriptor.residency == + VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN) && + (m_vkDevCtx.FindRequiredDeviceExtension( + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME) == + nullptr)) { + // A VK_IMAGE registration declaring FOREIGN residency takes a + // FOREIGN queue-family acquire at submit; without the extension + // that barrier is spec-invalid. + VkEncErr() << "[EncoderExt] register: the session device lacks " + << VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME + << ", required for FOREIGN-residency input" << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_EXTENSION_MISSING; + } + + static const uint8_t kZeroUuid[VK_UUID_SIZE] = {}; + static const uint8_t kZeroLuid[VK_LUID_SIZE] = {}; + const bool uuidMatters = + (descriptor.handleType != + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE); + const bool haveDeviceUuid = + (memcmp(descriptor.deviceUUID, kZeroUuid, VK_UUID_SIZE) != 0); + const bool haveDriverUuid = + (memcmp(descriptor.driverUUID, kZeroUuid, VK_UUID_SIZE) != 0); + const bool haveDeviceLuid = + (descriptor.deviceLUIDValid == VK_TRUE) && + (memcmp(descriptor.deviceLUID, kZeroLuid, VK_LUID_SIZE) != 0); + if (!haveDeviceUuid) { + // Only meaningful for an IMPORT. A VK_IMAGE registration names an + // image the caller created on this very device, so there is nothing + // to propagate and nothing to check -- warning there would fire on + // every in-process run and teach people to ignore the message. + // + // On an import, a zeroed UUID is exactly what a consumer that + // propagates nothing looks like, and on a multi-GPU host that is the + // difference between a named mismatch and a driver error nobody can + // act on. Warn once rather than per registration. + if (uuidMatters) { + // ONCE PER PROCESS, and "once" has to be true rather + // than likely: independent sessions reach this branch + // concurrently, and a plain check-then-store lets two + // of them both read false and both print. The flag + // arbitrates emission and publishes nothing else, so + // relaxed ordering is the entire requirement. It does + // not rely on stdio locking, on per-session + // serialization, or on the output being suppressed. + static std::atomic warnedZeroUuid{false}; + if (!warnedZeroUuid.exchange(true, std::memory_order_relaxed)) { + VkEncErr() << "[EncoderExt] register: imported handle carries " + "a zeroed deviceUUID, so a cross-device handle " + "cannot be detected. Propagate the exporting " + "device's UUID." << std::endl; + } + } + } + if (haveDeviceUuid || haveDriverUuid || haveDeviceLuid) { + VkPhysicalDeviceIDProperties idProps{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_ID_PROPERTIES}; + VkPhysicalDeviceProperties2 props2{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2}; + props2.pNext = &idProps; + m_vkDevCtx.GetPhysicalDeviceProperties2(m_vkDevCtx.getPhysicalDevice(), + &props2); + if (haveDeviceUuid && + (memcmp(descriptor.deviceUUID, idProps.deviceUUID, + VK_UUID_SIZE) != 0)) { + VkEncErr() << "[EncoderExt] register: handle was exported by a " + "different physical device" << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_DEVICE_MISMATCH; + } + if (haveDriverUuid && + (memcmp(descriptor.driverUUID, idProps.driverUUID, + VK_UUID_SIZE) != 0)) { + // The Vulkan external-memory compatibility table binds + // driverUUID identity for the OPAQUE handle types: a + // same-device, different-driver OPAQUE_FD is exactly the + // mismatch these fields exist to catch before the driver's + // unhelpful import error. A dma-buf is a kernel object and + // cross-driver import of one is spec-legal -- suspicious here, + // but not ours to refuse. + const bool opaqueHandle = + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD) || + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_WIN32) || + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_D3D11_TEXTURE); + if (opaqueHandle) { + VkEncErr() << "[EncoderExt] register: opaque handle was " + "exported by a different driver (driverUUID " + "mismatch)" << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_DEVICE_MISMATCH; + } + // Same reasoning as the zeroed-UUID claim above: independent + // sessions reach this through the shared descriptor validator + // and need no common lock to do so. + static std::atomic warnedDriverUuid{false}; + if (!warnedDriverUuid.exchange(true, std::memory_order_relaxed)) { + VkEncErr() << "[EncoderExt] register: the dma-buf exporter's " + "driverUUID differs from the encode device's; " + "legal for a kernel object, but worth knowing" + << std::endl; + } + } + // Windows device identity. The fields exist on every platform, so + // this compiles everywhere and stays inert on Linux (no caller sets + // deviceLUIDValid); on Windows the LUID is what the spec says + // identifies the device for D3D/OPAQUE_WIN32 interop. + if (haveDeviceLuid && (idProps.deviceLUIDValid == VK_TRUE) && + (memcmp(descriptor.deviceLUID, idProps.deviceLUID, + VK_LUID_SIZE) != 0)) { + VkEncErr() << "[EncoderExt] register: handle was exported by a " + "device with a different LUID" << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_DEVICE_MISMATCH; + } + } + + // Session compatibility -- the negotiation point the initialization + // gate above guarantees. A registration that cannot feed THIS session + // must say so by name here, not as a wrong-size staging copy or a + // driver error later. m_encoderConfig is absent only behind the + // null-backend test seam, which negotiates nothing to compare against. + // Checked after the device-identity block for the same precedence + // reason the modifier check runs last: a wrong-GPU handle answers + // DEVICE_MISMATCH before its format or extent is judged against OUR + // session. + if (m_encoderConfig) { + // TWO formats can register against one session, and conflating them + // is what kept the compute tier unreachable: + // + // * the ENCODE-SOURCE format -- semi-planar, derived from the + // negotiated subsampling and bit depth rather than read out of + // input.vkFormat -- which registers on the DIRECT path; and + // * the FILTER-INPUT format, which is what input.vkFormat holds + // once the binder has written numPlanes down, and which registers + // only when this session can actually convert it. A frame that + // arrives in the wrong format is an in-contract case, not an + // error case -- but only where the + // adaptation exists, which is what ComputeFilterActive() answers. + const VkFormat sessionEncodeFormat = + VkVideoCoreProfile::CodecGetVkFormat( + m_encoderConfig->input.chromaSubsampling, + GetComponentBitDepthFlagBits(m_encoderConfig->input.bpp), + VkVideoCoreProfile::PLANE_LAYOUT_SEMIPLANAR_2); + const bool filterActive = ComputeFilterActive(); + const VkFormat sessionFilterFormat = + filterActive ? m_encoderConfig->input.vkFormat + : VK_FORMAT_UNDEFINED; + if ((descriptor.format != sessionEncodeFormat) && + (descriptor.format != sessionFilterFormat)) { + VkEncErr() << "[EncoderExt] register: descriptor format " + << (uint32_t)descriptor.format + << " does not match the session's negotiated encode " + "input format " << (uint32_t)sessionEncodeFormat + << (filterActive + ? " nor its filter input format " + : " (no compute filter on this session, so no " + "filter input format) ") + << (filterActive ? (uint32_t)sessionFilterFormat : 0u) + << "; renegotiate the allocation or reinitialize " + "the session" << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_FORMAT_UNSUPPORTED; + } + // A filter-input registration needs the views the filter reads -- + // and the filter's TWO arms bind different VIEWS of the image, which + // is what this split says: per-plane views for a multi-planar input, + // one combined view for a single-plane one. Both are bound as + // VK_DESCRIPTOR_TYPE_STORAGE_IMAGE, so both require STORAGE; only the + // per-plane arm additionally requires the create flags a plane view + // costs. The split is by PLANE COUNT and not by colour model, + // because a packed 4:4:4 Y'CbCr surface is one interleaved plane and + // the filter binds it exactly as it binds R'G'B'. Refusing here, at + // the negotiation point, is what lets a producer re-export with the + // right declaration instead of discovering at submit that its frames + // convert to nothing. + if (descriptor.format != sessionEncodeFormat) { + if (VkEncInputFormatPlaneCount(descriptor.format) == 1u) { + // Single plane, storage read: usage + device capability, no + // create flags. Computed once -- the DRM arm of this walks + // the modifier list. + const bool deviceCanStorageRead = + DeviceCanStorageRead(descriptor); + if (!VkEncDescriptorPermitsStorageRead( + descriptor, descriptor.imageUsage, + deviceCanStorageRead)) { + VkEncErr() + << "[EncoderExt] register: format " + << (uint32_t)descriptor.format + << " encodes through the compute filter's " + "single-plane arm, which binds ONE combined view " + "as a VK_DESCRIPTOR_TYPE_STORAGE_IMAGE: the " + "descriptor must grant VK_IMAGE_USAGE_STORAGE_BIT " + "(imageUsage declared: " + << (uint32_t)descriptor.imageUsage + << ") and the device must report " + "VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT for this " + "format on this tiling (device storage read: " + << (deviceCanStorageRead ? 1u : 0u) + << "). No create flags are required on this arm" + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED; + } + } else if (!VkEncDescriptorPermitsPlaneStorageViews( + descriptor, descriptor.imageUsage)) { + // Multi-planar, per-plane STORAGE views. Unchanged, and + // still the only arm for which the create flags are real. + VkEncErr() << "[EncoderExt] register: format " + << (uint32_t)descriptor.format + << " encodes only through the compute filter, which " + "reads per-plane views; the descriptor must declare " + "VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT (with " + "EXTENDED_USAGE) and VK_IMAGE_USAGE_STORAGE_BIT" + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED; + } + } + // Strictly-smaller extents read out of bounds at encode; LARGER + // stays legal, because producer pools pad and align (1920x1088 + // feeding a 1080p session is the normal case, not an error). + if ((descriptor.width < m_encoderConfig->encodeWidth) || + (descriptor.height < m_encoderConfig->encodeHeight)) { + VkEncErr() << "[EncoderExt] register: descriptor extent " + << descriptor.width << "x" << descriptor.height + << " does not cover the session's coded extent " + << m_encoderConfig->encodeWidth << "x" + << m_encoderConfig->encodeHeight << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_EXTENT_INVALID; + } + } + + // A modifier the device cannot service is RENEGOTIABLE and must be + // named as such here, not discovered as IMPORT_FAILED at vkCreateImage + // -- "renegotiate the allocation" and "fail the stream" are different + // caller responses, and the code must tell them apart. Checked last: + // a wrong-GPU handle answers DEVICE_MISMATCH before its modifier is + // judged against OUR GPU. + if (isOsImport && + (descriptor.tiling == VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT) && + (descriptor.hasDrmFormatModifier == VK_TRUE) && + !ModifierWouldRegister(descriptor, descriptor.drmFormatModifier)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_MODIFIER_UNSUPPORTED; + } + + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::QueryImageSupport( + const VkVideoEncoderExternalImageDescriptor& descriptor, + VkVideoEncoderImageSupport* outSupport) +{ + if (outSupport == nullptr) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + if (outSupport->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + // The chain is no longer rejected wholesale: the known details struct + // is consumed; anything else is still refused, because an extension the + // library does not understand means the caller asked for something it + // is not getting. + VkVideoEncoderImageSupportDetails* details = nullptr; + for (void* link = const_cast(outSupport->pNext); + link != nullptr;) { + auto* candidate = + reinterpret_cast(link); + if ((candidate->sType != + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT_DETAILS) || + (details != nullptr)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + details = candidate; + link = const_cast(candidate->pNext); + } + + // Same predicate registration runs -- deliberately, so the query cannot + // become a second opinion that disagrees with the answer. + const VkVideoEncoderStatusCode status = ValidateImageDescriptor(descriptor); + outSupport->status = status; + outSupport->supported = + (status == VK_VIDEO_ENCODER_STATUS_SUCCESS) ? VK_TRUE : VK_FALSE; + + // Filled INDEPENDENTLY of the verdict: a MODIFIER_UNSUPPORTED answer + // carries the modifiers that WOULD work, which is renegotiation in one + // round trip, and a caller that never reads the verdict still gets it. + if (details != nullptr) { + FillImageSupportDetails(descriptor, details); + } + + // The call itself succeeded; whether the image is usable is the answer, + // not the return code. A caller must not have to distinguish "the query + // failed" from "the query says no". + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +// Fold the calling thread's import-guard verdict into the session-level +// snapshot GetCompletionInfo reports, and record the one verdict that means +// the guard did not do what its build asked of it -- INCOMPLETE -- in the +// diagnostic channel, which a consumer can read with NO new chained struct +// at all. At the disabled default the guard asks for nothing and INCOMPLETE +// is unreachable; the channel stays for builds that re-arm it. +// +// LOCK ORDER, which is why this is a separate function and not three lines +// inside the registration. It takes m_pendingMutex, and the only order this +// file establishes between the two is pending -> resource +// (ReleaseEncodedFrame holds m_pendingMutex and calls +// ReleaseResourceReference, which takes m_resourceMutex). Taking pending +// while holding resource would invert it. The caller therefore runs this +// from a scope guard declared AHEAD of its m_resourceMutex lock guard, so +// reverse-declaration-order destruction puts it strictly after the release. +void VulkanVideoEncoderExtImpl::PublishImportGuardVerdict() +{ + VkEncImportOrdinalGuardReport report; + VkEncGetImportOrdinalGuardReport(&report); + if (report.state == VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED) { + // The guard did not run for this registration -- a VK_IMAGE slot, + // or a refusal ahead of the import. Leaving the snapshot alone is + // the point: such a registration must not erase the verdict a + // dma-buf registration established -- DISABLED at the shipped + // default of 0, COMPLETE or INCOMPLETE only on a re-armed build. + return; + } + std::lock_guard lock(m_pendingMutex); + m_importGuardReport = report; + if (report.state != VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE) { + return; + } + // No VkEncErr() here: the guard already printed its own line on this + // path. What was missing was a channel that survives silenceStdio. + m_diagnosticCount++; + std::snprintf(m_lastDiagnostic, sizeof(m_lastDiagnostic), + // Kept inside VK_VIDEO_ENCODER_MAX_DIAGNOSTIC_CHARS with + // the widest possible expansion of the four conversions; + // a longer sentence here is silently truncated. + "import-ordinal guard INCOMPLETE: %u of %u sacrificial " + "dma-buf imports retained (status %d, errno %d) -- the " + "requested import-phase shift was NOT applied", + (unsigned)report.retainedCount, + (unsigned)report.requestedCount, + (int)report.failureStatus, (int)report.failureErrno); +} + +// Arm the content probe for one registration, and hand the encoder the probe +// object if it does not have it yet. +// +// LOCK ORDER, the same constraint PublishImportGuardVerdict documents at +// length: this takes m_pendingMutex (to reach m_encoder), so it must not run +// under m_resourceMutex. The caller runs it from a scope guard declared ahead +// of its lock guard. +// +// A NULL-BACKEND SESSION HAS NO ENCODER, and that is not a failure here: the +// probe's registration table is pure host state, so arming, the ARMED / +// NOT_APPLICABLE answer and the GetCompletionInfo snapshot all work with no +// device at all. Only CAPTURE and SCORE need the encoder, and those simply +// never happen -- which is exactly what a device-free session should report. +VkVideoEncoderImportContentState +VulkanVideoEncoderExtImpl::ArmImportContentProbe( + VkVideoEncoderResource resource, + VkVideoEncoderContentProbe::CaptureSite captureSite) +{ + VkSharedBaseObj encoder; + { + std::lock_guard lock(m_pendingMutex); + if (!m_contentProbe) { + if (VkVideoEncoderContentProbe::Create(m_contentProbe) != VK_SUCCESS) { + return VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED; + } + } + encoder = m_encoder; + } + m_contentProbe->ArmRegistration(resource, captureSite); + if (encoder) { + // Idempotent, and deliberately done here rather than at session + // init: the probe object does not exist until the first caller asks + // for it, and by then the encoder is long since built. + encoder->SetContentProbe(m_contentProbe); + } + return VkEncMapContentState(m_contentProbe->GetRegistrationState(resource)); +} + +void VulkanVideoEncoderExtImpl::ForgetImportContentProbe( + VkVideoEncoderResource resource) +{ + VkSharedBaseObj probe; + { + std::lock_guard lock(m_pendingMutex); + probe = m_contentProbe; + } + if (probe) { + // This is also what makes GetCompletionInfo's damaged report DRAIN. + // A consumer reacting to a DAMAGED_* verdict retires the + // registration; that retirement is what stops the same verdict being + // reported forever and lets the next damaged buffer surface. + probe->ForgetRegistration(resource); + } +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::RegisterImageResource( + const VkVideoEncoderExternalImageDescriptor& descriptor, + uint64_t osHandle, + VkVideoEncoderResource* outResource, + VkVideoEncoderStatus* pStatus) +{ + // The import-ordinal guard's verdict is per CALL. Clear it before + // anything can reach the import, so a registration that never gets + // there reports NOT_EVALUATED rather than the previous one's answer. + VkEncResetImportOrdinalGuardReport(); + + // Declared HERE, ahead of the m_resourceMutex lock guard further down, + // because destruction runs in reverse declaration order: this therefore + // fires AFTER that lock is released, which is what keeps + // PublishImportGuardVerdict's m_pendingMutex acquisition on the right + // side of the pending -> resource order. See that function. + struct GuardVerdictPublisher { + VulkanVideoEncoderExtImpl* self; + ~GuardVerdictPublisher() { self->PublishImportGuardVerdict(); } + } guardVerdictPublisher{this}; + + // Ownership is read before any validation, for the same reason + // handleType already is on the failure exits below: the mode must be + // applicable on EVERY exit, including a malformed descriptor. + const bool isPosixFdType = + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD) || + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF); + const bool borrow = + (descriptor.ownership == VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_BORROW); + // The echo is a constant of (platform, type, mode), decided here and + // written on every return -- the caller asserts, never guesses. +#if defined(__linux__) + VkEncStatusEcho statusEcho(pStatus, isPosixFdType && !borrow); +#else + VkEncStatusEcho statusEcho(pStatus, false); +#endif + + // Declared AFTER |statusEcho| and BEFORE the m_resourceMutex lock guard, + // and both halves of that are load-bearing. Reverse-declaration-order + // destruction puts this AFTER the resource lock is released (so + // ArmImportContentProbe's m_pendingMutex acquisition stays on the right + // side of the pending -> resource order) and BEFORE the echo is written + // (so the echo reports what the arming actually achieved, rather than + // what it was about to attempt). + struct ContentProbeArmer { + VulkanVideoEncoderExtImpl* self; + VkEncStatusEcho* echo; + // Set by the chain walk: did the caller ask at all. + bool requested = false; + // Set by the success tail: is there a registration to arm. + bool registered = false; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + VkVideoEncoderContentProbe::CaptureSite captureSite = + VkVideoEncoderContentProbe::CaptureSite::kUnreachable; + ~ContentProbeArmer() { + if (!requested || !registered) { + // Refused before a registration existed. The echo's + // NOT_EVALUATED default is the answer, and probeGeneration + // is still stamped, so the caller can tell that apart from + // a library that never wrote the struct. + return; + } + echo->SetContentVerdict( + self->ArmImportContentProbe(resource, captureSite), + resource); + } + } contentProbeArmer{this, &statusEcho}; + +#if defined(__linux__) + if (borrow && isPosixFdType && ((int64_t)osHandle >= 0)) { + // BORROW: an immediate private duplicate, taken before anything + // that can fail, so no exit below can ever touch the caller's fd. + // Everything from here down runs the unconditional TRANSFER rule + // on our copy. CLOEXEC because this process may spawn. + const int dupFd = fcntl((int)osHandle, F_DUPFD_CLOEXEC, 0); + if (dupFd < 0) { + VkEncErr() << "[EncoderExt] register: BORROW dup failed (errno " + << errno << "); the caller retains its handle" + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + osHandle = (uint64_t)dupFd; + } +#endif + + // A mis-stamped status struct is version skew, refused like any other + // unstamped struct -- with the ownership rule applied on this exit + // like every other (under BORROW, |osHandle| is by now the library's + // private duplicate, so the caller's fd survives regardless). + if ((pStatus != nullptr) && + (pStatus->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS)) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + // The chain is no longer refused wholesale -- the QueryImageSupport + // shape. Exactly one VkVideoEncoderImportGuardInfo is consumed; + // anything else, and a REPEATED known sType, is still refused rather + // than ignored, because an extension the library does not understand + // means the caller asked for something it is not getting. Callers built + // against a header that had nothing to chain here always passed NULL, + // so relaxing this cannot change what any of them see. + if (pStatus != nullptr) { + VkVideoEncoderImportGuardInfo* guardInfo = nullptr; + VkVideoEncoderImportContentInfo* contentInfo = nullptr; + for (void* link = const_cast(pStatus->pNext); link != nullptr;) { + // The {sType, pNext} prefix is read through a SIBLING struct + // type before the link is re-cast; VK_ENC_PIN_CHAIN_PREFIX is + // what makes that sound. + auto* prefix = + reinterpret_cast(link); + const VkVideoEncoderStructureType linkType = prefix->sType; + const void* next = prefix->pNext; + if ((linkType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO) && + (guardInfo == nullptr)) { + guardInfo = prefix; + } else if ((linkType == + VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO) && + (contentInfo == nullptr)) { + contentInfo = + reinterpret_cast(link); + } else { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + link = const_cast(next); + } + // Installed only now: the echo must never write through a chain + // this gate refused. + statusEcho.SetGuardInfo(guardInfo); + statusEcho.SetContentInfo(contentInfo); + // CHAINING THE STRUCT IS THE OPT-IN. There is no second switch: a + // registration whose caller did not ask for a content verdict is + // never armed, records no readback and allocates nothing. + contentProbeArmer.requested = (contentInfo != nullptr); + } + + // Validation runs cheapest-first and BEFORE any Vulkan call, so a + // cross-process caller gets a specific, actionable answer without the + // library having touched the driver. Every early return consumes the fd. + if (outResource == nullptr) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + *outResource = VK_VIDEO_ENCODER_RESOURCE_NULL; + + const VkVideoEncoderStatusCode validation = + ValidateImageDescriptor(descriptor); + if (validation != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + if (validation == + VK_VIDEO_ENCODER_STATUS_ERROR_MODIFIER_UNSUPPORTED) { + // The renegotiation list rides QueryImageSupport; at the + // register boundary it is at least LOGGED, so a refusal is + // never bare. + VkVideoEncoderImageSupportDetails details; + FillImageSupportDetails(descriptor, &details); + std::ostringstream workable; + for (uint32_t i = 0; i < details.directModifierCount; i++) { + workable << (i ? ", " : "") << "0x" << std::hex + << details.directModifiers[i]; + } + VkEncErr() << "[EncoderExt] register: DRM modifier 0x" + << std::hex << descriptor.drmFormatModifier + << std::dec << " is not usable for this descriptor; " + << details.directModifierCount + << " modifier(s) would be: [" << workable.str() + << "]" << std::endl; + } + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return validation; + } + + std::lock_guard lock(m_resourceMutex); + + size_t index = m_resources.size(); + for (size_t i = 0; i < m_resources.size(); i++) { + if (!m_resources[i].live && (m_resources[i].inFlight == 0)) { + index = i; + break; + } + } + if (index == m_resources.size()) { + if (m_resources.size() >= 4096) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT; + } + m_resources.emplace_back(); + } + RegisteredImage& slot = m_resources[index]; + + if (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE) { + if (descriptor.existingImage == VK_NULL_HANDLE) { + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + slot.image = descriptor.existingImage; + slot.memory = VK_NULL_HANDLE; + slot.ownsImage = false; // the caller's image; we never free it + } else { + const VkVideoEncoderStatusCode status = + ImportImageLocked(descriptor, osHandle, slot); + // Ownership rule: the fd is consumed either way, by exactly ONE + // owner. On success the imported VkDeviceMemory holds it and + // vkFreeMemory releases it; on failure the import already applied + // the ownership split -- it closed the fd itself iff the + // failure preceded the vkAllocateMemory handoff, and past that + // call the DRIVER consumed it (NVIDIA does so even when the + // allocation fails; gpu/vulkan/vulkan_memory.cc documents it). + // A close here would be the second close of an fd number this + // multithreaded process may already have recycled. + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + return status; + } + } + + // RESOLVED ON EVERY PATH, from the descriptor alone, and BEFORE the + // null-backend fork below. These two are written here rather than only inside + // BuildRegisteredViewLocked, which that fork skips -- so on a device-free + // session the routing decision and everything derived from it (inputPath, + // and the content probe's arm decision) would silently read a default. + // Neither needs a device to compute. + slot.imageUsage = VkEncResolveRegistrationUsage(descriptor); + slot.encodeCapable = VkEncRegistrationIsDirectlyEncodable(descriptor); + + if (m_nullBackend == nullptr) { + const VkVideoEncoderStatusCode viewStatus = + BuildRegisteredViewLocked(descriptor, slot); + if (viewStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + // BuildRegisteredViewLocked freed everything it and the import + // created (the fd included, via the owning wrapper); the slot + // is clean for reuse and was never live. + return viewStatus; + } + } + // else: null-backend session (test seam). Only the VK_IMAGE arm can get + // here without a device (the import arms already failed validation on + // the missing extensions), its bookkeeping -- slot, generation, id, + // inFlight -- is exactly what the fault tests exercise, and the view is + // the one step that needs a device. Nothing downstream reads it there: + // the null backend terminates the submit before the encoder would. + + slot.live = true; + slot.retired = false; + slot.inFlight = 0; + slot.handleType = descriptor.handleType; + slot.residency = descriptor.residency; + slot.format = descriptor.format; + slot.width = descriptor.width; + slot.height = descriptor.height; + slot.defaultLayout = descriptor.defaultLayout; + slot.tiling = descriptor.tiling; + // Section 6 routing, resolved ONCE here and read back by the submit -- + // which is what makes the path the caller is told about and the path its + // frames take the same decision rather than two that happen to agree. + // + // FILTER is now produced, and every clause of it is checkable: + // * the format needs converting rather than re-tiling (rung 2, not + // rung 3: a format that MATCHES the session's encode format and is + // merely wrongly tiled takes the copy, per the ladder's + // "transfer only for pure tiling mismatches"); + // * this session can convert (the filter compiled in and requested); + // * and the views the filter reads were actually BUILT on this image, + // not merely asked for -- per-plane STORAGE views for a multi-planar + // input, ONE storage-capable combined view for RGBA. EITHER satisfies + // this clause and neither substitutes for the other: the filter binds + // whichever its arm was built for, and asking only the plane-count + // question is what routed every RGBA registration to STAGED. + // Any clause failing leaves STAGED, which for a format that genuinely + // needs converting is unreachable -- ValidateImageDescriptor refused the + // registration before this line. Unreachable, not harmless: were the two + // ever to disagree, the submit's own filter-only-format fence answers + // VK_ERROR_FORMAT_NOT_SUPPORTED rather than letting the frame fall into + // the staging copy, which for a format that copy cannot handle is a + // measured device loss. + const bool routesViaFilter = + !slot.encodeCapable && + (VkEncClassifyInput(descriptor.format, descriptor.colorModel) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER) && + ComputeFilterActive() && + (slot.planeStorageViews || slot.storageReadView); + slot.inputPath = slot.encodeCapable + ? VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT + : (routesViaFilter + ? VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER + : VK_VIDEO_EXTERNAL_INPUT_PATH_STAGED); + + *outResource = MakeResourceId(index, slot.generation); + // ===== WHAT THE PROBE CAN ACTUALLY RIDE -- AND IT IS NO LONGER + // ===== |encodeCapable| + // + // This must NOT read `contentProbeArmer.directlyEncodable = + // slot.encodeCapable`, because encodeCapable is + // + // encodableFormat && (tiling != VK_IMAGE_TILING_LINEAR) && + // (usage & VIDEO_ENCODE_SRC) + // + // Two of those three clauses have nothing to say about whether a readback + // is possible, and the TILING clause inverted the answer for the one + // damage class the defect appears on. A BLOCK-LINEAR import + // (tiling != LINEAR) that also declares VIDEO_ENCODE_SRC came out + // encodeCapable, and the probe latched NOT_APPLICABLE for it -- while + // block-linear buffers are precisely the class that gets poisoned. + // Tiling is gone from this decision. + // + // What is left is the only thing that decides it: does a + // transfer-readable copy of the producer's pixels pass through this + // library for this registration. + // + // * TRANSFER_SRC on the imported image. The probe's capture is a + // vkCmdCopyImage OUT of that image; without the usage bit the copy is + // a VUID violation, and arming would be a promise the capture site + // cannot keep. + // * NOT the FILTER path. A filter-routed frame is read in place by + // the filter, never copied, so it never reaches the capture site at + // all -- arming it + // would leave the registration ARMED forever, reporting + // NOT_EVALUATED, which is the "cannot fail" shape rather than an + // answer, and this is the line that decides it.) + // + // DIRECT registrations are now INCLUDED. They reach the capture site + // through a one-frame staged detour taken only while a capture is still + // owed -- see VkVideoEncoder::SetExternalInputFrameWithNode. + // + // READ FROM THE DESCRIPTOR, NOT FROM |slot|. slot.imageUsage and + // slot.encodeCapable are written only by BuildRegisteredViewLocked, which + // RegisterImageResource skips on a null-backend session -- so a predicate + // reading them answers NOT_APPLICABLE for every device-free registration + // and the whole device-free carrier suite goes dark. slot.inputPath is + // safe: it is assigned above on every path. + // + // THE THIRD CLAUSE IS THE FORMAT, and it was missing. The first two ask + // whether a copy out of this image can be RECORDED; this one asks whether + // the bytes that copy produces can be READ. They are different questions + // and only RecordCapture used to ask the second, on the first frame, + // which made a 10-bit registration echo ARMED at import and get silently + // downgraded to NOT_APPLICABLE afterwards -- leaving the session + // reporting probed=0 damaged=0 armed=0, which is exactly what a session + // that probed everything and found it clean reports. Measured, before + // this clause existed, on RTX A4000 by test/encoder-ext-format-encode + // --content-probe row [2/13] P010. + const bool probeCanRideThisRegistration = + (slot.inputPath != VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER) && + ((VkEncResolveRegistrationUsage(descriptor) & + VK_IMAGE_USAGE_TRANSFER_SRC_BIT) != 0) && + VkVideoEncoderContentProbe::IsProbeableFormat(descriptor.format); + contentProbeArmer.registered = true; + contentProbeArmer.resource = *outResource; + contentProbeArmer.captureSite = + probeCanRideThisRegistration + ? VkVideoEncoderContentProbe::CaptureSite::kReachable + : VkVideoEncoderContentProbe::CaptureSite::kUnreachable; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::RegisterSemaphore( + const VkVideoEncoderSemaphoreDescriptor& descriptor, + uint64_t osHandle, + VkVideoEncoderResource* outResource, + VkVideoEncoderStatus* pStatus) +{ + // Same ownership contract as image registration, applied in the same + // order: mode read first so it holds on every exit, dup before + // anything that can fail, echo written on every return. + const bool isPosixFdType = + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD) || + (descriptor.handleType == + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF); + const bool borrow = + (descriptor.ownership == VK_VIDEO_ENCODER_HANDLE_OWNERSHIP_BORROW); +#if defined(__linux__) + VkEncStatusEcho statusEcho(pStatus, isPosixFdType && !borrow); +#else + VkEncStatusEcho statusEcho(pStatus, false); +#endif + +#if defined(__linux__) + if (borrow && isPosixFdType && ((int64_t)osHandle >= 0)) { + // BORROW: dup-first, so no exit below -- including the sType gates + // -- can ever touch the caller's fd. + const int dupFd = fcntl((int)osHandle, F_DUPFD_CLOEXEC, 0); + if (dupFd < 0) { + VkEncErr() << "[EncoderExt] register: BORROW dup failed (errno " + << errno << "); the caller retains its handle" + << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + osHandle = (uint64_t)dupFd; + } +#endif + + if ((pStatus != nullptr) && + ((pStatus->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS) || + (pStatus->pNext != nullptr))) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + + // A chained descriptor is refused like a mis-stamped one -- nothing + // chains onto it today -- with the ownership rule applied on this + // exit like every other. + if ((descriptor.sType != + VK_VIDEO_ENCODER_STRUCTURE_TYPE_SEMAPHORE_DESCRIPTOR) || + (descriptor.pNext != nullptr)) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + if (outResource == nullptr) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + *outResource = VK_VIDEO_ENCODER_RESOURCE_NULL; + + // A binary semaphore cannot express "wait for frame N" and is single-use, + // so it cannot be registered once and named repeatedly -- which is the + // only reason to register it at all. + if (descriptor.semaphoreType != VK_SEMAPHORE_TYPE_TIMELINE) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; + } + VkExternalSemaphoreHandleTypeFlagBits vkHandleType; + switch (descriptor.handleType) { + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD: +#if defined(__linux__) + vkHandleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD_BIT; + break; +#else + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; +#endif + case VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_WIN32: +#if defined(_WIN32) + vkHandleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32_BIT; + break; +#else + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; +#endif + default: + // DMA_BUF and VK_IMAGE are image handle types; naming one here is + // a caller error worth saying out loud rather than coercing. + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; + } + + // Everything above is a property of the descriptor alone, answered the + // same with or without a session -- same ordering as image registration, + // so a malformed descriptor is named as malformed rather than merely + // early. Everything below needs a device, and m_initialized alone does + // not prove one: a null-backend session (internal header) reports + // initialized with no device behind it, and the create/import calls + // below would then go through never-loaded PFNs. + if (!m_initialized || (m_vkDevCtx.getDevice() == VK_NULL_HANDLE)) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED; + } + + // A device that cannot service the import must say so BY NAME here, at + // the negotiation point -- the image arm's always-on rule, applied + // before the first device call rather than inside it: without the + // extension the import PFN below was never populated, so the miss was a + // call through null, a crash rather than a status. Same honesty caveat + // as the image arm: on the library's own device this set is what + // vkCreateDevice enabled; on a caller's device Vulkan cannot report + // enablement, so this is physical-device support, and enablement is the + // device creator's to audit. + const char* missing = nullptr; +#if defined(__linux__) + if (m_vkDevCtx.FindRequiredDeviceExtension( + VK_KHR_EXTERNAL_SEMAPHORE_FD_EXTENSION_NAME) == nullptr) { + missing = VK_KHR_EXTERNAL_SEMAPHORE_FD_EXTENSION_NAME; + } +#elif defined(_WIN32) + if (m_vkDevCtx.FindRequiredDeviceExtension( + VK_KHR_EXTERNAL_SEMAPHORE_WIN32_EXTENSION_NAME) == nullptr) { + missing = VK_KHR_EXTERNAL_SEMAPHORE_WIN32_EXTENSION_NAME; + } +#endif + if (missing != nullptr) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + VkEncErr() << "[EncoderExt] register: the session device lacks " + << missing << ", required for this import" << std::endl; + return VK_VIDEO_ENCODER_STATUS_ERROR_EXTENSION_MISSING; + } + + VkSemaphoreTypeCreateInfo typeInfo{ + VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO}; + typeInfo.semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + typeInfo.initialValue = 0; + VkSemaphoreCreateInfo createInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + createInfo.pNext = &typeInfo; + + VkSemaphore semaphore = VK_NULL_HANDLE; + if (m_vkDevCtx.CreateSemaphore(m_vkDevCtx.getDevice(), &createInfo, + nullptr, &semaphore) != VK_SUCCESS) { + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + +#if defined(__linux__) + VkImportSemaphoreFdInfoKHR importInfo{ + VK_STRUCTURE_TYPE_IMPORT_SEMAPHORE_FD_INFO_KHR}; + importInfo.semaphore = semaphore; + importInfo.handleType = vkHandleType; + importInfo.fd = (int)osHandle; + // No TEMPORARY bit: a temporary import is consumed by the first wait, + // which would silently turn a registration into a one-shot. + importInfo.flags = 0; + if (m_vkDevCtx.ImportSemaphoreFdKHR(m_vkDevCtx.getDevice(), + &importInfo) != VK_SUCCESS) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), semaphore, + nullptr); + // The import consumed the fd on success only; on failure it is still + // ours to close, and the ownership rule says every exit path. + VkEncConsumeOsHandle(descriptor.handleType, osHandle); + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } +#elif defined(_WIN32) + // |osHandle| must already be valid in THIS process -- see the header's + // note on RegisterSemaphore. A Win32 handle is process-local, and this + // interface carries no source PID with which to OpenProcess and + // duplicate, so the caller does that. + // + // Not TEMPORARY, for the same reason as the fd path: a temporary import + // is consumed by the first wait, turning a registration into a one-shot. + // + // The handle is NOT closed here on any path, success or failure. That is + // this header's published rule for Win32 and it is the right one: the + // value belongs to the caller's handle table, and closing a number we do + // not own closes whatever else happens to hold it. + VkImportSemaphoreWin32HandleInfoKHR importInfo{ + VK_STRUCTURE_TYPE_IMPORT_SEMAPHORE_WIN32_HANDLE_INFO_KHR}; + importInfo.semaphore = semaphore; + importInfo.handleType = vkHandleType; + importInfo.handle = (HANDLE)osHandle; + importInfo.name = nullptr; + importInfo.flags = 0; + if (m_vkDevCtx.ImportSemaphoreWin32HandleKHR(m_vkDevCtx.getDevice(), + &importInfo) != VK_SUCCESS) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), semaphore, + nullptr); + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } +#else + (void)vkHandleType; + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), semaphore, nullptr); + return VK_VIDEO_ENCODER_STATUS_ERROR_HANDLE_TYPE_UNSUPPORTED; +#endif + + std::lock_guard lock(m_semaphoreMutex); + size_t index = m_semaphores.size(); + for (size_t i = 0; i < m_semaphores.size(); i++) { + if (!m_semaphores[i].live) { + index = i; + break; + } + } + if (index == m_semaphores.size()) { + if (m_semaphores.size() >= 4096) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), semaphore, + nullptr); + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT; + } + m_semaphores.emplace_back(); + } + RegisteredSemaphore& slot = m_semaphores[index]; + slot.semaphore = semaphore; + slot.live = true; + *outResource = MakeResourceId(index, slot.generation) | kSemaphoreTag; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::UnregisterSemaphore( + VkVideoEncoderResource resource) +{ + if ((resource & kSemaphoreTag) == 0) { + // An image id, or nothing. Rejected rather than resolved against the + // wrong table. + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + const uint64_t untagged = resource & ~kSemaphoreTag; + const size_t index = (size_t)((untagged & 0xFFFFFFFFull) - 1); + const uint32_t generation = (uint32_t)(untagged >> 32); + + VkSemaphore doomed = VK_NULL_HANDLE; + { + std::lock_guard lock(m_semaphoreMutex); + if (index >= m_semaphores.size()) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + RegisteredSemaphore& slot = m_semaphores[index]; + if (!slot.live || (slot.generation != generation)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + doomed = slot.semaphore; + slot.semaphore = VK_NULL_HANDLE; + slot.live = false; + slot.generation++; // every id naming this slot is now stale + } + if (doomed != VK_NULL_HANDLE) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), doomed, nullptr); + } + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkSemaphore VulkanVideoEncoderExtImpl::ResolveSemaphore( + VkVideoEncoderResource resource) +{ + if ((resource & kSemaphoreTag) == 0) { + return VK_NULL_HANDLE; + } + const uint64_t untagged = resource & ~kSemaphoreTag; + const size_t index = (size_t)((untagged & 0xFFFFFFFFull) - 1); + const uint32_t generation = (uint32_t)(untagged >> 32); + std::lock_guard lock(m_semaphoreMutex); + if (index >= m_semaphores.size()) { + return VK_NULL_HANDLE; + } + const RegisteredSemaphore& slot = m_semaphores[index]; + if (!slot.live || (slot.generation != generation)) { + return VK_NULL_HANDLE; + } + return slot.semaphore; +} + +// Import a per-frame acquire fence (sync_fd) as a BINARY semaphore the encode +// waits on. TEMPORARY, unlike RegisterSemaphore's permanent import: SYNC_FD is +// copy-transference, so the payload is consumed by the first wait -- the +// defining property of a per-frame fence. Takes ownership of |fd| on every +// path on POSIX; on Win32 the field is unused and nothing is closed. +VkSemaphore VulkanVideoEncoderExtImpl::ImportAcquireFenceLocked(int fd) +{ + if (fd < 0) { + return VK_NULL_HANDLE; + } +#if defined(__linux__) + VkSemaphoreCreateInfo createInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + VkSemaphore semaphore = VK_NULL_HANDLE; + if (m_vkDevCtx.CreateSemaphore(m_vkDevCtx.getDevice(), &createInfo, + nullptr, &semaphore) != VK_SUCCESS) { + close(fd); + return VK_NULL_HANDLE; + } + VkImportSemaphoreFdInfoKHR importInfo{ + VK_STRUCTURE_TYPE_IMPORT_SEMAPHORE_FD_INFO_KHR}; + importInfo.semaphore = semaphore; + importInfo.handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT_KHR; + importInfo.fd = fd; + importInfo.flags = VK_SEMAPHORE_IMPORT_TEMPORARY_BIT; + if (m_vkDevCtx.ImportSemaphoreFdKHR(m_vkDevCtx.getDevice(), + &importInfo) != VK_SUCCESS) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), semaphore, nullptr); + close(fd); + return VK_NULL_HANDLE; + } + return semaphore; +#elif defined(_WIN32) + // NOT closed, and this is not a build-order point: on Win32 the value in + // acquireFenceFd is not a file descriptor at all, and this file's own + // Win32 rule is that a handle we do not own is never closed. There is no + // SYNC_FD import on this arm, so refuse. + (void)fd; + return VK_NULL_HANDLE; +#else + // Any other POSIX target -- macOS, the BSDs. The value IS an fd here + // (gpu_fence_handle.h: ScopedPlatformFence = base::ScopedFD under + // IS_POSIX), and Chromium hands this entry point a dup() it then forgets, + // its own comment saying "the library's close hits only its own copy". + // There is no VK_KHR_external_semaphore_fd import wired up on this arm, so + // consume the fd the only way left. Dropping this close is one leaked fd + // per submitted frame, not a no-op -- which is why the include above is + // guarded on !_WIN32 rather than on __linux__. + close(fd); + return VK_NULL_HANDLE; +#endif +} + +// The export half of the same per-frame entry point as the acquire import +// above (design 3.5). A BINARY semaphore, exportable as SYNC_FD, created per +// fenced frame: SYNC_FD is copy-transference, so a release fence is consumed +// by whoever waits on it and cannot be registered once and named repeatedly +// -- the same property that keeps this off RegisterSemaphore's timeline-only +// path in the acquire direction. +VkSemaphore VulkanVideoEncoderExtImpl::CreateReleaseFenceSemaphore() +{ +#if defined(__linux__) + if (m_vkDevCtx.GetSemaphoreFdKHR == nullptr) { + return VK_NULL_HANDLE; + } + // Ask once, never assume: naming a handle type in VkExportSemaphoreCreateInfo + // that the physical device does not report EXPORTABLE is invalid usage, so + // the query has to run before the first create. Same shape, and same + // reason, as CreateCompletionTimelineSemaphore's OPAQUE_FD probe. + if (!m_releaseFenceExportProbed) { + m_releaseFenceExportProbed = true; + if (m_vkDevCtx.GetPhysicalDeviceExternalSemaphoreProperties != nullptr) { + VkSemaphoreTypeCreateInfo queryTypeInfo{ + VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO}; + queryTypeInfo.semaphoreType = VK_SEMAPHORE_TYPE_BINARY; + VkPhysicalDeviceExternalSemaphoreInfo extInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_SEMAPHORE_INFO}; + extInfo.pNext = &queryTypeInfo; + extInfo.handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT; + VkExternalSemaphoreProperties extProps{ + VK_STRUCTURE_TYPE_EXTERNAL_SEMAPHORE_PROPERTIES}; + m_vkDevCtx.GetPhysicalDeviceExternalSemaphoreProperties( + m_vkDevCtx.getPhysicalDevice(), &extInfo, &extProps); + m_releaseFenceExportable = + ((extProps.externalSemaphoreFeatures & + VK_EXTERNAL_SEMAPHORE_FEATURE_EXPORTABLE_BIT) != 0); + } + if (!m_releaseFenceExportable) { + VkEncErr() << "[EncoderExt] this device cannot export SYNC_FD " + "binary semaphores; per-frame release fences will " + "answer -1" << std::endl; + } + } + if (!m_releaseFenceExportable) { + return VK_NULL_HANDLE; + } + VkExportSemaphoreCreateInfo exportInfo{ + VK_STRUCTURE_TYPE_EXPORT_SEMAPHORE_CREATE_INFO}; + exportInfo.handleTypes = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT; + VkSemaphoreCreateInfo createInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + createInfo.pNext = &exportInfo; + VkSemaphore semaphore = VK_NULL_HANDLE; + if (m_vkDevCtx.CreateSemaphore(m_vkDevCtx.getDevice(), &createInfo, + nullptr, &semaphore) != VK_SUCCESS) { + return VK_NULL_HANDLE; + } + return semaphore; +#else + // SYNC_FD is a POSIX-only currency; the Win32 arm of the release fence + // is reserved, exactly as the Win32 semaphore-export arm is. + return VK_NULL_HANDLE; +#endif +} + +// Export the pending signal on |semaphore| as a SYNC_FD. +// +// MUST be called only after the batch carrying that signal has been handed +// to vkQueueSubmit: a SYNC_FD export requires the semaphore to be signalled +// or to have a signal operation pending execution. Exporting earlier is +// invalid usage, and the ordering is the caller's obligation, not something +// this function can check. +// +// The export is a copy-transference operation and therefore has the side +// effects of a WAIT on the source semaphore: the payload is moved into the +// returned fd and the semaphore is left unsignalled. That is what makes the +// returned fd independent of the VkSemaphore's remaining lifetime. +// +// -1 is a legal answer, not an error: vkGetSemaphoreFdKHR is permitted to +// return -1 for an already-signalled SYNC_FD export, which means "nothing to +// wait for". It coincides with this API's "-1 = no fence, not failure". +int VulkanVideoEncoderExtImpl::ExportReleaseFenceFd(VkSemaphore semaphore) +{ +#if defined(__linux__) + if ((semaphore == VK_NULL_HANDLE) || + (m_vkDevCtx.GetSemaphoreFdKHR == nullptr)) { + return -1; + } + VkSemaphoreGetFdInfoKHR getInfo{ + VK_STRUCTURE_TYPE_SEMAPHORE_GET_FD_INFO_KHR}; + getInfo.semaphore = semaphore; + getInfo.handleType = VK_EXTERNAL_SEMAPHORE_HANDLE_TYPE_SYNC_FD_BIT; + int fd = -1; + if (m_vkDevCtx.GetSemaphoreFdKHR(m_vkDevCtx.getDevice(), &getInfo, + &fd) != VK_SUCCESS) { + return -1; + } + return fd; +#else + (void)semaphore; + return -1; +#endif +} + +// Destroy the release-fence semaphores whose frames retired with a real +// capture. A capture is published only after the encode command buffer's +// fence wait, and the encode submit waits on the staging submit, so both +// submissions that can signal a release fence have completed by then -- +// which is the vkDestroySemaphore precondition, stated rather than assumed. +void VulkanVideoEncoderExtImpl::DrainRetiredReleaseFences() +{ + std::vector doomed; + { + std::lock_guard lock(m_pendingMutex); + doomed.swap(m_releaseFenceRetired); + } + for (VkSemaphore s : doomed) { + if (s != VK_NULL_HANDLE) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), s, nullptr); + } + } +} + +// The unproven graveyard's in-session drain. Every route into it -- a deadline +// drop, a CancelFrame, an AbandonAllFrames, a failed capture, a non-NOT_READY +// submit failure -- is a repeatable operation a long session performs many +// times, so leaving Deinitialize as its only exit made it grow without bound +// for the life of the session. Nothing in it carries a completion proof, which +// is why the sweep costs a DeviceWaitIdle; the bound is what keeps that cost +// off the steady-state path, and the log line is what keeps a session that +// pays it from doing so silently. +void VulkanVideoEncoderExtImpl::DrainUnprovenSemaphores() +{ + std::vector doomed; + { + std::lock_guard lock(m_pendingMutex); + if (m_unprovenSemaphores.size() < kUnprovenSemaphoreBound) { + return; + } + doomed.swap(m_unprovenSemaphores); + } + if (m_vkDevCtx.getDevice() == VK_NULL_HANDLE) { + // No device to destroy them on and none coming: dropping the handles + // is all that is left, and it is what Deinitialize would have done. + return; + } + m_vkDevCtx.DeviceWaitIdle(); + for (VkSemaphore s : doomed) { + if (s != VK_NULL_HANDLE) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), s, nullptr); + } + } + VkEncErr() << "[EncoderExt] swept " << doomed.size() + << " per-frame fence semaphore(s) whose completion no " + "retirement could prove, behind a device wait-idle" + << std::endl; +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::SubmitRegisteredFrame( + const VkVideoEncoderFrameSubmitInfo& info, + VkSemaphore* pStagingCompleteSemaphore) +{ + // PRE-PASS over the chained descriptors, ahead of EVERY refusal in this + // function -- including the three at the top of it, which is why it now + // stands in FRONT of them rather than after the registration lookup. + // + // It discharges the two handle promises this entry point makes. Both are + // unconditional and neither can be kept by the resolving walk further + // down: + // + // pReleaseFenceFd [out] ext.h promises the library always writes it, + // which is what makes a bare `int fd;` plus + // close(fd) a legal caller shape. + // acquireFenceFd [in] ext.h promises "the library takes ownership + // and closes it on every exit path", and design + // section 2.3 states the rule the whole API is + // built on -- the library consumes POSIX fds it + // is given, always, on every exit path, + // "including argument-validation failures that + // never reach Vulkan". The SYNC_FD row of that + // section's ownership table names the acquire + // fence explicitly, so it is in scope. + // + // The resolving walk cannot keep either promise, for one shared reason: it + // refuses on an unknown sType or an unresolvable id, and a chain may + // legally put either of those AHEAD of the fence descriptor + // (`info.pNext = &sync; sync.pNext = &fence;` is one of the three shapes + // this header blesses), in which case the fence node is never reached at + // all. The three top-of-function refusals are worse still -- a mis-stamped + // info.sType, a session that is not initialized, and a resource id that + // does not resolve all return before the walk begins. Chromium's VEA arms + // acquireFenceFd on essentially every frame + // (vulkan_video_encode_accelerator.cc: `fence_desc.acquireFenceFd = + // fd.release();`, under a comment saying ownership transfers), so each of + // those exits was one leaked sync_fd on the production path, not a + // theoretical one. + // + // AN UNRESOLVED TENSION, LEFT UNRESOLVED ON PURPOSE. Read this before + // moving either half of it. This pre-pass dereferences info.pNext BEFORE + // info.sType has been validated, and the two rules behind that ordering + // genuinely conflict: + // + // FOR reading pNext first: section 2.3 says the library consumes a + // handle it was given on EVERY exit path, "including argument-validation + // failures that never reach Vulkan". A refusal ON info.sType is one of + // those exits. To consume the fd there it has to be FOUND there, which + // means walking pNext before the sType is known to be good. There is no + // ordering that validates sType first and still honours 2.3 for a handle + // supplied alongside a bad sType. + // + // AGAINST: VkVideoEncoderFrameSubmitInfo has NO default member + // initialisers -- unlike VkVideoEncoderFrameFenceDescriptor and + // VkVideoEncoderFrameSyncDescriptor, which both default pNext to + // nullptr. A stack-local `VkVideoEncoderFrameSubmitInfo info;` is + // therefore indeterminate in every field, and the sType check is + // the gate that caught exactly that. Reading pNext off such a struct is + // undefined behaviour, and if the garbage reads as a fence node this + // pre-pass stores -1 through a garbage int* and records a garbage int + // for the guard to close. + // + // WHAT WAS CHANGED WHEN THIS WAS RAISED: only the boundedness of the walk + // below -- a node cap and a cycle check, which removes the + // spin-at-100%-CPU half of the hazard and protects the pReleaseFenceFd + // half of this same pre-pass as a side effect. The ORDERING was not + // touched. + // + // WHAT WAS NOT, AND WHY IT IS A HUMAN'S CALL: the obvious middle is to + // validate info.sType first and treat a wrong-sType submit as "no handle + // was legibly given" -- nothing is walked, so nothing is consumed. That is + // defensible on its own terms (a struct whose type tag is wrong is not a + // struct whose pNext can be trusted to name anything), and it costs + // exactly one row of 2.3's promise: a caller who mis-stamps sType while + // arming a real fd leaks it, and ext.h's "closes it on every exit path" + // would need that exception written into it in so many words. The other + // option, giving VkVideoEncoderFrameSubmitInfo default member + // initialisers, removes the indeterminate-struct case for C++ callers + // only, and changes the initialisation rules of a pinned public layout. + // Both are API decisions rather than implementation ones, so neither was + // taken here. + // + // It traverses UNKNOWN nodes rather than stopping at them, and that is not + // a liberty: this header states that its chain is "Vulkan's own + // convention", and the whole point of that convention is that every + // chained struct opens with {sType, pNext} so a consumer can traverse a + // chain containing structs it does not know -- which is what + // VkBaseInStructure exists for. A caller whose struct does not open that + // way has already broken the ABI in a way the resolving walk's very first + // sType read would fault on too. Traversing is not HONOURING: the walk + // below still refuses on an unknown sType rather than skipping it, because + // refuse-don't-skip is about never silently ignoring a struct the caller + // believed was acted on, which is a different question from layout. + struct ChainNode { + VkVideoEncoderStructureType sType; + const void* pNext; + }; + // The IN half. Ownership of every armed acquire fd is taken HERE and held + // until either the import takes it or this function returns; exactly one + // of those two closes it. Consume() hands ownership to + // ImportAcquireFenceLocked, which owns the fd on every one of its own + // paths (POSIX; the Win32 arm refuses without an fd to own) + // -- both failure legs close it, and a successful + // vkImportSemaphoreFdKHR of a SYNC_FD consumes it -- and anything this + // guard still holds at scope exit provably never reached an import. + // + // The handoff is a REMOVAL from the list, not a flag, because a double + // close is strictly worse than the leak this replaces: the fd number is + // free the instant the first close returns, so a second one can close an + // unrelated descriptor the process has since opened at that number. + struct AcquireFdOwnership { + std::vector pending; + void Record(int fd) { + // Deduplicated by VALUE, so this list holds each number at most + // once and the destructor below can never close one twice. + // + // That is the whole of what the dedup does, and the whole of what + // it claims. NOTHING HERE DIAGNOSES THE DUPLICATE -- an earlier + // revision of this comment said the walk would report it when "the + // second import fails on an fd the first one consumed", and that + // was wrong twice over: nothing compared the two nodes, and a + // second vkImportSemaphoreFdKHR on a number the first import + // consumed is not guaranteed to fail, so the caller could come + // away with two semaphores whose payload came from one fence. The + // duplicate is caught at the CONSUME site in the walk below + // instead, which is the first point at which it is knowable. + for (size_t i = 0; i < pending.size(); i++) { + if (pending[i] == fd) { + return; + } + } + pending.push_back(fd); + } + // Hands ownership of |fd| to the caller. Returns false when this guard + // does not hold it -- which, since the pre-pass recorded every armed + // fd in the chain before anything could consume one, can only mean an + // EARLIER fence node named the same number and its import already took + // it. The walk turns that into a typed refusal rather than importing + // one descriptor twice. + bool Consume(int fd) { + for (std::vector::iterator it = pending.begin(); + it != pending.end(); ++it) { + if (*it == fd) { + pending.erase(it); + return true; + } + } + return false; + } + ~AcquireFdOwnership() { +#if defined(__linux__) + for (size_t i = 0; i < pending.size(); i++) { + close(pending[i]); + } +#endif + } + } acquireFds; + // BOUNDED, because the chain is caller memory and nothing in the ABI stops + // it from pointing at itself. `fence.pNext = &fence` is one assignment + // away, and an unbounded walk over it does not refuse and does not return + // -- it spins inside the library at 100% CPU with the caller's fd still + // open. A hang is a worse answer than any refusal, so the traversal is + // capped and a cycle is named. + // + // ONE bound covers both walks in this function. The resolving walk further + // down traverses the same chain and is equally unbounded on paper, but it + // is unreachable with a cyclic or absurd chain because this pre-pass runs + // first and refuses; bounding it twice would be two things to keep in step + // for no extra coverage. The same bound is what now protects the + // pReleaseFenceFd half of this pre-pass, which had the identical exposure. + // + // The cap is a sanity bound, not a design limit: the header blesses three + // chain shapes and none is longer than two nodes. + static const size_t kMaxChainNodes = 64; + size_t chainNodes = 0; + const ChainNode* slow = (const ChainNode*)info.pNext; + for (const ChainNode* pre = (const ChainNode*)info.pNext; pre != nullptr; + pre = (const ChainNode*)pre->pNext) { + if (++chainNodes > kMaxChainNodes) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + if ((chainNodes % 2) == 0) { + // Floyd at half speed: |slow| sits at node chainNodes/2 while + // |pre| sits at node chainNodes-1, so the gap between them grows + // through every non-negative integer and therefore reaches a + // multiple of the cycle length once both are inside one -- at + // which point they are the same node. |slow| always trails |pre| + // through nodes |pre| has already dereferenced, so its pNext is + // never a fresh read. + // + // The chainNodes > 2 guard is load-bearing, not decoration: at + // chainNodes 1 and 2 those two indices coincide legitimately on a + // straight chain, and comparing there would call every two-node + // chain -- one of the three shapes the header blesses -- a cycle. + slow = (const ChainNode*)slow->pNext; + if ((chainNodes > 2) && (slow == pre)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + } + if (pre->sType != + VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR) { + continue; + } + const VkVideoEncoderFrameFenceDescriptor* f = + (const VkVideoEncoderFrameFenceDescriptor*)pre; + // The OUT half, carried across unchanged from where this pre-pass used + // to sit: answering -1 for every fence descriptor in the chain first + // is the only order in which the promise holds for all of them. A + // successful export overwrites it after the submit. + if (f->pReleaseFenceFd != nullptr) { + *f->pReleaseFenceFd = -1; + } + if (f->acquireFenceFd >= 0) { + acquireFds.Record(f->acquireFenceFd); + } + } + + if (info.sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + // A session is a real encoder OR the test seam's null backend standing + // in for one -- never both (see VkEncInstallNullBackend). + if (!m_initialized || (!m_encoder && (m_nullBackend == nullptr))) { + return VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED; + } + + // Reap release-fence semaphores whose frames retired with a real capture. + // Done on the submit path rather than at retirement because the + // retirement sites hold m_pendingMutex, and a driver call under that lock + // is the one-sided-lock hazard the rest of this file avoids by + // construction. Bounded: a steady-state consumer releases every frame it + // acquires, so this stays at the in-flight depth. + DrainRetiredReleaseFences(); + // And the graveyard that cannot prove itself, once it has grown past its + // bound. A no-op below the bound, which is every steady-state submit. + DrainUnprovenSemaphores(); + + // Resolve the registration and take a reference in one locked step, so + // an Unregister cannot land between the lookup and the reference. + VkImage image = VK_NULL_HANDLE; + VkFormat format = VK_FORMAT_UNDEFINED; + uint32_t width = 0; + uint32_t height = 0; + VkImageLayout registeredLayout = VK_IMAGE_LAYOUT_UNDEFINED; + VkVideoEncoderInputResidency residency = + VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN; + VkImageTiling tiling = VK_IMAGE_TILING_OPTIMAL; + VkSharedBaseObj node; + bool encodeCapable = false; + bool routeViaFilter = false; + + // Release-obligation guard. Once the reference below is taken, EVERY exit + // from this function owes a release, and a single missed site does not + // fail loudly -- it pins a producer's pool slot forever, which surfaces + // much later as a stall with no cause attached. Arming a scope-exit here + // and disarming it only on successful enqueue makes the obligation + // structural instead of a rule each new early return has to remember. + struct ReleaseObligation { + VulkanVideoEncoderExtImpl* owner = nullptr; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + bool armed = false; + void Arm(VulkanVideoEncoderExtImpl* o, VkVideoEncoderResource r) { + owner = o; resource = r; armed = true; + } + void Disarm() { armed = false; } + ~ReleaseObligation() { + if (armed) { + owner->ReleaseResourceReference(resource); + } + } + } obligation; + + { + std::lock_guard lock(m_resourceMutex); + RegisteredImage* slot = LookupResourceLocked(info.resource); + if ((slot == nullptr) || slot->retired) { + // Stale, retired, or never registered. The generation counter is + // what makes this a rejection rather than a wrong-image encode. + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + slot->inFlight++; + obligation.Arm(this, info.resource); + image = slot->image; + format = slot->format; + width = slot->width; + height = slot->height; + registeredLayout = slot->defaultLayout; + tiling = slot->tiling; + // The once-per-registration node: a VkSharedBaseObj copy (refcount + // bump, no allocation). Shared by every frame of this registration + // and read-only on the encode path. encodeCapable is the + // registration-time routing predicate; the per-frame tiling field + // no longer routes registered frames. + node = slot->node; + encodeCapable = slot->encodeCapable; + // Read from the resolved path, not recomputed: the registration + // already answered which rung this image takes. That answer is + // INTERNAL: no public accessor returns it. The route is the + // library's choice, not a contract, and the type that names it is + // declared in the internal header for that reason. + routeViaFilter = + (slot->inputPath == VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER); + // RESIDENCY: HONOURED WHEREVER IT WAS DECLARED, DERIVED ONLY WHERE + // IT WAS NOT. + // + // This must NOT read `handleType == VK_IMAGE ? slot->residency : + // FOREIGN`, which DISCARDS an explicit declaration on every + // OS-handle type. The justification given for that -- "the library + // performed the import, so it knows the memory is foreign to the + // encode device and the caller cannot get it wrong" -- is false of + // the one shipping consumer of this path. Chromium's CPU staging + // tier is a SELF-IMPORT: the staging VkImage and its VkDeviceMemory + // are created on the LIBRARY's own VkDevice (it asks for it with + // GetVkDevice()), exported OPAQUE_FD from that device, and + // re-imported into that same device. There is no second device and + // no second queue family, so nothing about that memory is foreign + // and the caller is the only party that knows it. + // + // WHAT ACTUALLY DISCRIMINATES is ownership by an external allocator + // or queue family -- NOT who wrote the pixels, and NOT the handle + // type. The distinction has a live counterexample in each direction: + // * host-written yet correctly FOREIGN -- the Wayland zero-copy + // lane writes its pixels from the CPU, but the EXPORTER is + // GBM/DRM and the buffer really is owned outside this device; + // * OS-handle yet correctly LOCAL -- the self-import above. + // A rule keyed on handleType cannot express either. + // + // SO THE DERIVATION IS KEPT, AND ONLY WHERE IT IS THE ONLY ANSWER + // AVAILABLE: an OS-handle registration that declared AUTO (which is + // 0, so it is also what a zero-initialised descriptor says) told the + // library nothing, and FOREIGN remains the safe reading -- a + // cross-process producer that propagates no residency is far more + // likely to be a genuine import than a self-import. Both live + // AUTO consumers depend on that and are covered by name in + // test/encoder-ext-input-residency: the Wayland/GBM lane and the + // out-of-tree Vulkan renderer app, neither of which declares + // residency at all. + // + // VK_IMAGE IS UNTOUCHED IN BOTH DIRECTIONS, including VK_IMAGE + + // AUTO, which still reaches VkVideoEncoder's legacy layout + // heuristic rather than being forced to FOREIGN. That matters + // concretely: the sibling suites register VK_IMAGE with a + // zero-initialised descriptor, and forcing those to FOREIGN would + // hand their images to VK_QUEUE_FAMILY_FOREIGN_EXT. + // + // THE HAZARD IS THE RELEASE, NOT A MISSING ONE. This comment used to + // say such a registration would take "an acquire with no matching + // release". That has been false since VkVideoEncoder gained + // ReleaseImageToForeignQueue, which every acquire arm now pairs with + // under a byte-identical gate -- StageInputFrame's copy and filter + // arms, and RecordVideoCodingCmd's direct arm. The real cost of the + // hypothetical is the opposite, and worse: the forced registration + // would get a CORRECTLY paired acquire AND release, and the release + // is what does the damage. It gives a library-local image away to a + // foreign owner that does not exist, so nothing transitions it back, + // and the paired SetStagedInputResidualLayout(VK_IMAGE_LAYOUT_MAX_ENUM) + // discards the library's record of the layout it left behind. The + // next frame's acquire then names a layout the image is not in. + const bool deriveForeign = + (slot->handleType != + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE) && + (slot->residency == VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO); + residency = deriveForeign + ? VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN + : slot->residency; + } + + VkVideoEncodeInputFrame frame = {}; + frame.image = image; + frame.format = format; + frame.width = width; + frame.height = height; + frame.imageTiling = tiling; + // THE SENTINEL, AND THE FACT IT ERASES. VkVideoEncoderFrameSubmitInfo:: + // currentLayout documents UNDEFINED as "as declared at registration", + // so this line is where a per-frame statement by the producer and a + // standing registration default become one indistinguishable value. + // The encoder core needs them distinguished -- an explicit per-frame + // declaration means "I moved it" and must beat the library's own record + // of where its last handback left the image -- so the bit is captured + // here, at the only place that still has it, and passed down. + const bool srcLayoutIsExplicit = + (info.currentLayout != VK_IMAGE_LAYOUT_UNDEFINED); + frame.currentLayout = srcLayoutIsExplicit ? info.currentLayout + : registeredLayout; + frame.frameId = info.frameId; + frame.pts = info.pts; + frame.forceIDR = info.forceIDR; + frame.isLastFrame = info.isLastFrame; + frame.qpOverride = info.qpOverride; + frame.inputResidency = residency; + // No caller tag on this path: VkVideoEncoderFrameSubmitInfo carries + // none, because the registration already names the image. + frame.uniqueImageIndex = -1; + frame.waitSemaphoreCount = info.waitSemaphoreCount; + frame.pWaitSemaphores = info.pWaitSemaphores; + frame.pWaitSemaphoreValues = info.pWaitSemaphoreValues; + frame.signalSemaphoreCount = info.signalSemaphoreCount; + frame.pSignalSemaphores = info.pSignalSemaphores; + frame.pSignalSemaphoreValues = info.pSignalSemaphoreValues; + + // A chained FrameSyncDescriptor names REGISTERED semaphores by id, which + // is what a cross-process producer can actually send: its handles crossed + // once, at registration, and per frame it has only integers. Resolved + // here into the same arrays the in-process path fills, so the encode path + // below is identical either way. + // + // ONE FLAT CHAIN, POSITION-INDEPENDENT. Both descriptor types hang off + // VkVideoEncoderFrameSubmitInfo::pNext and both are honoured wherever in + // the list they appear -- fence alone, fence ahead of sync, or fence + // behind it. That is Vulkan's own pNext convention and it is now what the + // header says. The fence descriptor does not have to chain off the SYNC + // descriptor's pNext specifically: this loop does not require it and no + // caller does. + // + // These live until the submit below returns, which is the only span the + // encoder reads them for. + // Where the walk found a caller-supplied release-fence out-parameter, if + // any. Recorded rather than acted on inside the loop: the loop still has + // failing exits after the fence descriptor, and creating a Vulkan object + // in front of them would leave it stranded on every one of them. + // Both halves of the chain's fd contract were already discharged by the + // pre-pass at the top of this function: the -1 stores are done, and every + // armed acquire fd is held by a guard that closes anything no import takes. + int* pReleaseFenceFd = nullptr; + // Imported acquire-fence semaphores, in a scope guard rather than a bare + // vector: every refusal exit BELOW an import -- an unknown sType or an + // unresolvable id in a node the fence descriptor precedes -- would + // otherwise strand one with no submit to retire it. Nothing has been + // queued at any of those exits, so destroying there is legal; on the + // success path the guard is disarmed and the PendingFrame takes over. + struct AcquireFenceScope { + VulkanVideoEncoderExtImpl* owner = nullptr; + std::vector semaphores; + bool armed = true; + ~AcquireFenceScope() { + if (!armed) { + return; + } + for (VkSemaphore s : semaphores) { + if (s != VK_NULL_HANDLE) { + owner->m_vkDevCtx.DestroySemaphore( + owner->m_vkDevCtx.getDevice(), s, nullptr); + } + } + } + } acquireFences; + acquireFences.owner = this; + std::vector resolvedWait; + std::vector resolvedWaitValues; + std::vector resolvedSignal; + std::vector resolvedSignalValues; + for (const void* next = info.pNext; next != nullptr;) { + const VkVideoEncoderStructureType* stype = + (const VkVideoEncoderStructureType*)next; + if ((*stype != VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_SYNC_DESCRIPTOR) && + (*stype != VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR)) { + // Unknown chained struct: refuse rather than skip. Silently + // ignoring a struct a caller believed was honoured is the + // accepted-and-ignored failure this ABI exists to prevent. + // The armed guard drops the reference on the way out -- the + // a per-site release/disarm pair at this exit would be the memory-based + // pattern the guard exists to abolish (V2 design, risk R-2). + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + if (*stype == VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_FENCE_DESCRIPTOR) { + const VkVideoEncoderFrameFenceDescriptor* fence = + (const VkVideoEncoderFrameFenceDescriptor*)next; + // Recorded BEFORE the import below, which has its own refusal + // exit. The -1 itself was already written by the pre-pass above; + // this only remembers where to put the export. + if (fence->pReleaseFenceFd != nullptr) { + pReleaseFenceFd = fence->pReleaseFenceFd; + } + if (fence->acquireFenceFd >= 0) { + // Ownership passes from the pre-pass guard to the importer + // HERE, before the call rather than after it: the importer + // owns the fd on every one of its own paths (POSIX; the + // Win32 arm refuses without an fd to own), both failure legs + // included, so leaving it in the guard's list across this call + // would be the double close the guard's comment forbids. + if (!acquireFds.Consume(fence->acquireFenceFd)) { + // The pre-pass recorded every armed fd in this chain, so + // the only way the guard no longer holds this one is that + // an EARLIER fence node named the same number and its + // import has already taken it. Importing it again is not a + // smaller error than refusing: SYNC_FD is consumed by the + // first import, so the second call runs against a number + // that is either closed or has since been reused, and if + // it happens to succeed the caller is handed two + // semaphores whose payload came from one fence. + // + // Refuse, and close NOTHING: the descriptor belongs to the + // first import now, and a close here is the double close + // the guard exists to prevent. IMPORT_FAILED rather than a + // new code -- this is an acquire fence that could not be + // turned into a semaphore, which is what that status + // already says at the site just below. + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + VkSemaphore acquired = + ImportAcquireFenceLocked(fence->acquireFenceFd); + if (acquired == VK_NULL_HANDLE) { + // fd already closed by the helper on POSIX; on Win32 + // there was no fd. Refuse rather than + // encode without the wait -- that reads the surface before + // the producer finished writing it. + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + // NOT resolvedWait. resolvedWait is the REGISTERED-id answer + // to "whose wait list does this frame use", and an acquire + // fence is not an answer to that question -- it is one + // additional producer fence, and the caller's raw + // pWaitSemaphores may name others that no sync_fd covers (a + // sync_fd is a single fence object; N producers need N fds or + // a merge). Putting it here made `!resolvedWait.empty()` fire + // and threw those raw handles away, encoding from a surface + // a producer was still writing -- the exact mirror, on the + // wait side, of the signal-side defect this walk was fixed + // for. It is APPENDED below instead, like the release fence + // on the signal side. + acquireFences.semaphores.push_back(acquired); + } + next = fence->pNext; + continue; + } + + const VkVideoEncoderFrameSyncDescriptor* sync = + (const VkVideoEncoderFrameSyncDescriptor*)next; + + for (uint32_t i = 0; i < sync->waitCount; i++) { + VkSemaphore s = ResolveSemaphore(sync->pWaitSemaphores[i]); + if (s == VK_NULL_HANDLE) { + // A wait on an unresolvable id would either be skipped -- + // encoding from a surface the producer has not finished + // writing -- or hang. Both are worse than a named refusal. + // The guard releases on the way out. + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + resolvedWait.push_back(s); + resolvedWaitValues.push_back(sync->pWaitValues[i]); + } + for (uint32_t i = 0; i < sync->signalCount; i++) { + VkSemaphore s = ResolveSemaphore(sync->pSignalSemaphores[i]); + if (s == VK_NULL_HANDLE) { + // Same refusal as the wait side; the guard releases. + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + resolvedSignal.push_back(s); + resolvedSignalValues.push_back(sync->pSignalValues[i]); + } + next = sync->pNext; + } + // Registered ids win outright rather than being merged with the raw + // arrays: naming a frame's sync twice is two answers to one question, + // and merging would make which one took effect unknowable. + // + // The override is PER DIRECTION, and the direction the chain did NOT name + // keeps the caller's own array. One combined test that then overwrote both + // is what this replaces, and it was live rather than latent: a chain + // carrying only an acquire fence -- exactly what Chromium's VEA sends, + // chaining a FrameFenceDescriptor whose only armed field is acquireFenceFd + // while it fills the submit info's pSignalSemaphores from its own + // end_semaphores -- resolves a non-empty wait list and an EMPTY signal + // list. That passed the combined test and then set signalSemaphoreCount to + // zero, discarding the caller's end-of-access semaphores on every fenced + // frame without a word. Whoever waits on those semaphores waits forever, + // which is the accepted-and-ignored failure this ABI exists to prevent. + if (!resolvedWait.empty()) { + frame.waitSemaphoreCount = (uint32_t)resolvedWait.size(); + frame.pWaitSemaphores = resolvedWait.data(); + frame.pWaitSemaphoreValues = resolvedWaitValues.data(); + } + if (!resolvedSignal.empty()) { + frame.signalSemaphoreCount = (uint32_t)resolvedSignal.size(); + frame.pSignalSemaphores = resolvedSignal.data(); + frame.pSignalSemaphoreValues = resolvedSignalValues.data(); + } + + // Per-frame ACQUIRE fences, appended to whichever wait array won above -- + // the exact mirror of the release fence's append on the signal side, and + // for the same reason. An acquire fence is additive: it names one more + // producer, not a different way of naming the ones the caller already + // named. These vectors have to outlive the submit call below, which is + // why they are declared here rather than inside the branch. + std::vector waitWithAcquire; + std::vector waitWithAcquireValues; + if (!acquireFences.semaphores.empty()) { + for (uint32_t i = 0; i < frame.waitSemaphoreCount; i++) { + waitWithAcquire.push_back(frame.pWaitSemaphores[i]); + waitWithAcquireValues.push_back( + frame.pWaitSemaphoreValues ? frame.pWaitSemaphoreValues[i] : 0); + } + for (VkSemaphore s : acquireFences.semaphores) { + // Value 0: binary semantics, as for the release fence. + waitWithAcquire.push_back(s); + waitWithAcquireValues.push_back(0); + } + frame.waitSemaphoreCount = (uint32_t)waitWithAcquire.size(); + frame.pWaitSemaphores = waitWithAcquire.data(); + frame.pWaitSemaphoreValues = waitWithAcquireValues.data(); + } + + // THE WAIT SIDE OF kMaxCallerSignalsForReleaseFence -- and the reason it + // is here as well as in the submit. + // + // The direct-encode submit assembles its waits into a fixed 8-slot array. + // It used to stop quietly at the boundary, discarding the surplus; since + // the acquire semaphore is appended LAST, just above, the surplus was + // exactly the producer fence this entry point exists to honour. + // SubmitVideoCodingCmds now refuses that shape outright instead of + // encoding without a wait -- but it runs on the assembly path, LONG AFTER + // this function has returned a status to the caller. Measured on an + // A4000: the frame is correctly not submitted and no release fd is + // produced, and the caller is still handed SUCCESS. Whoever believed that + // status waits forever for a completion that is never coming, which is + // the same accepted-and-ignored failure in a new place. + // + // So the count is checked HERE too, where a status can still be returned, + // and RESOURCE_LIMIT is what it is: a fixed resource, named, exceeded. + // + // Gated on |encodeCapable| because the limit is a property of the DIRECT + // submit alone. A staged registration assembles its waits into a + // std::vector that never truncates -- nine waits work there today, and + // refusing them because a different path has an array would be inventing + // a limit the library does not have. + // + // The bound is the array, not a smaller number of "caller" waits: seven + // caller waits plus an acquire fence is eight entries, it fits, and it + // must keep working. + // + // Not exact in one direction, deliberately: an enableQpMap session also + // spends a slot on its qpMap command buffer, which this layer cannot see, + // so such a session can still be refused by the submit rather than here. + // That path keeps its own check; this one removes the common case from + // the silent-failure class without pretending to knowledge it lacks. + // + // Exiting here leaves |acquireFences| ARMED, so the semaphore imported + // above is destroyed on the way out, and the caller's fd was already + // consumed by the import -- the ownership contract this entry point + // publishes is unchanged by the refusal. + constexpr uint32_t kMaxDirectSubmitWaits = 8; + if (encodeCapable && (frame.waitSemaphoreCount > kMaxDirectSubmitWaits)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_LIMIT; + } + + // Per-frame release fence (design 3.5), the export half of this entry + // point: a library-owned BINARY semaphore riding the frame's signal list, + // so that whichever submission consumes the input image signals it. + // SubmitExternalFrameCommon exports the SYNC_FD from it, once that + // submission has been issued -- see the long note there. + // + // APPENDED to the winner of the per-direction override above, never + // merged into the override decision itself. Two reasons, both load- + // bearing. First, the override answers "whose sync list does this frame + // use", and a library semaphore joining the list must not change that + // answer -- pushing into resolvedSignal would make an acquire-only chain + // look like a chain that named signals, and throw the caller's signal + // array away, which is the defect the per-direction rule exists to + // prevent. Second, appending is what keeps this OUT of index 0: the + // direct-encode submit gives signal index 0 the batched input-release + // timeline treatment (max-value-so-far at a queue flush point), which is + // correct for a timeline and wrong for a binary semaphore. + // + // The count gate is not decoration. The direct-encode submit assembles + // its signal list into a fixed 8-slot array shared with the encoder's own + // internal signals and silently stops appending when it fills. A release + // fence that got dropped there would never be signalled while its fd had + // already been handed out -- a caller waiting forever on a fence nobody + // signals, i.e. the hang this fence exists to prevent, caused by the + // fence. Refusing with -1 is the honest answer. + constexpr uint32_t kMaxCallerSignalsForReleaseFence = 4; + VkSemaphore releaseFenceSemaphore = VK_NULL_HANDLE; + std::vector signalWithRelease; + std::vector signalWithReleaseValues; + if ((pReleaseFenceFd != nullptr) && (m_nullBackend == nullptr) && + (frame.signalSemaphoreCount < kMaxCallerSignalsForReleaseFence)) { + releaseFenceSemaphore = CreateReleaseFenceSemaphore(); + } + if (releaseFenceSemaphore != VK_NULL_HANDLE) { + for (uint32_t i = 0; i < frame.signalSemaphoreCount; i++) { + signalWithRelease.push_back(frame.pSignalSemaphores[i]); + signalWithReleaseValues.push_back( + frame.pSignalSemaphoreValues ? frame.pSignalSemaphoreValues[i] + : 0); + } + // Value 0: binary semantics, which is also what tells the direct + // encode submit's release-timeline block to pass this entry through + // per frame instead of batching it. + signalWithRelease.push_back(releaseFenceSemaphore); + signalWithReleaseValues.push_back(0); + frame.signalSemaphoreCount = (uint32_t)signalWithRelease.size(); + frame.pSignalSemaphores = signalWithRelease.data(); + frame.pSignalSemaphoreValues = signalWithReleaseValues.data(); + } + + const VkResult result = NoteDeviceResult(SubmitExternalFrameCommon( + frame, node ? &node : nullptr, encodeCapable, routeViaFilter, + info.resource, + pStagingCompleteSemaphore, releaseFenceSemaphore, pReleaseFenceFd, + acquireFences.semaphores.empty() ? nullptr + : &acquireFences.semaphores, + srcLayoutIsExplicit)); + if (result != VK_SUCCESS) { + // The release-fence semaphore never reached a PendingFrame, so its + // disposal is this scope's. VK_NOT_READY is admission control, which + // refuses BEFORE the encoder is touched -- nothing was queued and the + // destroy precondition is trivially met. Any other failure can in + // principle have got as far as a staging submit before failing, and + // vkDestroySemaphore requires every batch referring to the semaphore + // to have completed, so those go to the teardown graveyard instead of + // being destroyed on a guess. + if (releaseFenceSemaphore != VK_NULL_HANDLE) { + if (result == VK_NOT_READY) { + m_vkDevCtx.DestroySemaphore(m_vkDevCtx.getDevice(), + releaseFenceSemaphore, nullptr); + } else { + std::lock_guard lock(m_pendingMutex); + m_unprovenSemaphores.push_back(releaseFenceSemaphore); + } + } + // The imported acquire semaphores follow the same rule, and for the + // same reason. On VK_NOT_READY the guard's destructor destroys them + // (nothing was queued); on anything else a staging submit may already + // be waiting on them, so they go to the graveyard and the guard is + // told to keep its hands off. + if ((result != VK_NOT_READY) && !acquireFences.semaphores.empty()) { + std::lock_guard lock(m_pendingMutex); + m_unprovenSemaphores.insert(m_unprovenSemaphores.end(), + acquireFences.semaphores.begin(), + acquireFences.semaphores.end()); + acquireFences.armed = false; + } + // Nothing was queued, so the reference must not survive the failure. + // The guard does this on the way out; no explicit release here. + if (result == VK_NOT_READY) { + // Admission-control backpressure: transient by construction -- + // the caller drains completions and retries this same call. + // Distinct from ERROR_RESOURCE_LIMIT, which the registries + // return for a FULL 4096-slot table: a condition only an + // Unregister can clear, where retry alone never succeeds. + return VK_VIDEO_ENCODER_STATUS_NOT_READY; + } + return VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + } + // Queued: EnqueuePendingFrame copied the acquire semaphores into the + // PendingFrame, which now owns them, so the scope guard must not. + acquireFences.armed = false; + + // Queued, and the entry took ownership of the reference at its own + // creation, under the lock that created it (EnqueuePendingFrame) -- so + // the frame drops it when it retires and this scope must not. Stamping + // the resource in AFTER the fact was R-2's window inside this + // function's own mitigation: a consumer thread could retire the entry + // between the enqueue and the stamp, the stamp's miss was silent, and + // the reference became un-droppable -- with the guard already + // disarmed. Nothing can return between the enqueue and this disarm, so + // the reference has exactly one owner on every path. + obligation.Disarm(); + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +void VulkanVideoEncoderExtImpl::ReleaseResourceReference( + VkVideoEncoderResource resource) +{ + if (resource == VK_VIDEO_ENCODER_RESOURCE_NULL) { + return; + } + std::lock_guard lock(m_resourceMutex); + const size_t index = ResourceIndex(resource); + if (index >= m_resources.size()) { + return; + } + RegisteredImage& slot = m_resources[index]; + // Deliberately NOT generation-checked: this drops a reference taken + // while the id was valid, and the slot may already be retired. + if (slot.inFlight > 0) { + slot.inFlight--; + } + // The deferred half of retirement: the last frame out frees the image. + if (slot.retired && (slot.inFlight == 0)) { + DestroyResourceLocked(slot); + } +} + +VkVideoEncoderStatusCode VulkanVideoEncoderExtImpl::UnregisterImageResource( + VkVideoEncoderResource resource) +{ + // Declared ahead of the lock guard so it fires after the lock is + // released -- ForgetImportContentProbe takes m_pendingMutex, and the + // order is pending -> resource. Unconditional: forgetting an id the + // probe never knew is a no-op, and running it on the RESOURCE_UNKNOWN + // exit too is what keeps a double-unregister from leaving a verdict + // behind that GetCompletionInfo would keep reporting forever. + struct ContentProbeForgetter { + VulkanVideoEncoderExtImpl* self; + VkVideoEncoderResource resource; + ~ContentProbeForgetter() { self->ForgetImportContentProbe(resource); } + } contentProbeForgetter{this, resource}; + + std::lock_guard lock(m_resourceMutex); + RegisteredImage* slot = LookupResourceLocked(resource); + if (slot == nullptr) { + // Either never registered or already retired. Naming it rather than + // returning a bare failure is the difference between a client that + // can debug its own map and one that cannot. + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + if (slot->inFlight > 0) { + // Deferred retirement: frames still reference this image. Marking it + // and freeing on the last release is what keeps a teardown from + // freeing memory the GPU is still reading. + slot->retired = true; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; + } + DestroyResourceLocked(*slot); + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkResult VkEncInjectImportContentMeasurement(VulkanVideoEncoderExt* encoder, + VkVideoEncoderResource resource, + uint32_t meanYQ8, + uint32_t meanUQ8, + uint32_t meanVQ8) +{ + if (encoder == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + auto* impl = static_cast(encoder); + VkSharedBaseObj probe; + { + std::lock_guard lock(impl->m_pendingMutex); + probe = impl->m_contentProbe; + } + if (!probe) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + probe->ApplyVerdict(resource, meanYQ8, meanUQ8, meanVQ8); + return VK_SUCCESS; +} + +//============================================================================= // Factory function //============================================================================= VK_VIDEO_ENCODER_EXPORT -VkResult CreateVulkanVideoEncoderExt( - VkSharedBaseObj& vulkanVideoEncoder) +VkResult CreateVulkanVideoEncoderExt( + VkSharedBaseObj& vulkanVideoEncoder) +{ + VkSharedBaseObj impl(new VulkanVideoEncoderExtImpl()); + if (!impl) { + return VK_ERROR_OUT_OF_HOST_MEMORY; + } + vulkanVideoEncoder = impl; + return VK_SUCCESS; +} + +//============================================================================= +// Fault-injection seam (vulkan_video_encoder_ext_internal.h). Only objects +// from CreateVulkanVideoEncoderExt may be passed: the static_cast below is +// sound because that factory mints every real VulkanVideoEncoderExt in +// existence. (The Chromium test fake implements the interface but never +// travels through these functions -- they exist for exactly what the fake +// cannot test.) +//============================================================================= + +VkResult VkEncInstallNullBackend(VulkanVideoEncoderExt* encoder, + const VkEncNullBackendState* state) +{ + if ((encoder == nullptr) || (state == nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_initialized) { + // Backends stand in for a real session; they never replace one. + return VK_ERROR_NOT_PERMITTED_KHR; + } + impl->m_nullBackend = state; + impl->m_initialized = true; + return VK_SUCCESS; +} + +// Copy out what the submit was handed. Counts are reported in full even when +// they exceed the probe's array capacity, so a test can never mistake a +// truncated record for a short list. +void VulkanVideoEncoderExtImpl::RecordSubmitSyncForTest( + const VkVideoEncodeInputFrame& frame) +{ + const uint32_t cap = (uint32_t)kVkEncSubmitSyncProbeCapacity; + std::lock_guard lock(m_lastSubmitSyncMutex); + m_lastSubmitSync = VkEncSubmitSyncProbe{}; + m_lastSubmitSync.recorded = VK_TRUE; + m_lastSubmitSync.waitCount = frame.waitSemaphoreCount; + m_lastSubmitSync.signalCount = frame.signalSemaphoreCount; + for (uint32_t i = 0; (i < frame.waitSemaphoreCount) && (i < cap); i++) { + if (frame.pWaitSemaphores != nullptr) { + m_lastSubmitSync.waitSemaphores[i] = frame.pWaitSemaphores[i]; + } + if (frame.pWaitSemaphoreValues != nullptr) { + m_lastSubmitSync.waitValues[i] = frame.pWaitSemaphoreValues[i]; + } + } + for (uint32_t i = 0; (i < frame.signalSemaphoreCount) && (i < cap); i++) { + if (frame.pSignalSemaphores != nullptr) { + m_lastSubmitSync.signalSemaphores[i] = frame.pSignalSemaphores[i]; + } + if (frame.pSignalSemaphoreValues != nullptr) { + m_lastSubmitSync.signalValues[i] = frame.pSignalSemaphoreValues[i]; + } + } +} + +VkResult VkEncInstallTestSemaphore(VulkanVideoEncoderExt* encoder, + VkSemaphore semaphore, + VkVideoEncoderResource* outResource) +{ + if ((encoder == nullptr) || (outResource == nullptr) || + (semaphore == VK_NULL_HANDLE)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + // A real session registers through RegisterSemaphore, which creates a + // timeline semaphore and imports an OS handle into it. This + // device-free shortcut stands in for that; it never stands beside it. + return VK_ERROR_NOT_PERMITTED_KHR; + } + std::lock_guard lock(impl->m_semaphoreMutex); + size_t index = impl->m_semaphores.size(); + for (size_t i = 0; i < impl->m_semaphores.size(); i++) { + if (!impl->m_semaphores[i].live) { + index = i; + break; + } + } + if (index == impl->m_semaphores.size()) { + impl->m_semaphores.emplace_back(); + } + VulkanVideoEncoderExtImpl::RegisteredSemaphore& slot = + impl->m_semaphores[index]; + slot.semaphore = semaphore; + slot.live = true; + *outResource = + VulkanVideoEncoderExtImpl::MakeResourceId(index, slot.generation) | + VulkanVideoEncoderExtImpl::kSemaphoreTag; + return VK_SUCCESS; +} + +VkVideoEncoderStatusCode VkEncUninstallTestSemaphore( + VulkanVideoEncoderExt* encoder, VkVideoEncoderResource resource) +{ + if (encoder == nullptr) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + if ((resource & VulkanVideoEncoderExtImpl::kSemaphoreTag) == 0) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + const uint64_t untagged = + resource & ~VulkanVideoEncoderExtImpl::kSemaphoreTag; + const size_t index = (size_t)((untagged & 0xFFFFFFFFull) - 1); + const uint32_t generation = (uint32_t)(untagged >> 32); + std::lock_guard lock(impl->m_semaphoreMutex); + if (index >= impl->m_semaphores.size()) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + VulkanVideoEncoderExtImpl::RegisteredSemaphore& slot = + impl->m_semaphores[index]; + if (!slot.live || (slot.generation != generation)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + // UnregisterSemaphore's slot hygiene without its driver destroy: the + // handle was never a real VkSemaphore and the session has no device to + // destroy one with. + slot.semaphore = VK_NULL_HANDLE; + slot.live = false; + slot.generation++; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkVideoEncoderStatusCode VkEncProbeLastSubmitSync( + VulkanVideoEncoderExt* encoder, VkEncSubmitSyncProbe* outProbe) +{ + if ((encoder == nullptr) || (outProbe == nullptr)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + std::lock_guard lock(impl->m_lastSubmitSyncMutex); + *outProbe = impl->m_lastSubmitSync; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkVideoEncoderStatusCode VkEncProbeResource(VulkanVideoEncoderExt* encoder, + VkVideoEncoderResource resource, + VkEncResourceProbe* outProbe) +{ + if ((encoder == nullptr) || (outProbe == nullptr)) { + return VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + std::lock_guard lock(impl->m_resourceMutex); + const size_t index = VulkanVideoEncoderExtImpl::ResourceIndex(resource); + if (index >= impl->m_resources.size()) { + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + const VulkanVideoEncoderExtImpl::RegisteredImage& slot = + impl->m_resources[index]; + if (slot.generation != + VulkanVideoEncoderExtImpl::ResourceGeneration(resource)) { + // Retirement completed: the generation advanced past this id. + return VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN; + } + outProbe->inFlight = slot.inFlight; + outProbe->live = slot.live ? VK_TRUE : VK_FALSE; + outProbe->retired = slot.retired ? VK_TRUE : VK_FALSE; + outProbe->inputPath = slot.inputPath; + outProbe->planeStorageViews = + slot.planeStorageViews ? VK_TRUE : VK_FALSE; + outProbe->storageReadView = + slot.storageReadView ? VK_TRUE : VK_FALSE; + return VK_VIDEO_ENCODER_STATUS_SUCCESS; +} + +VkBool32 VkEncSessionInitialized(VulkanVideoEncoderExt* encoder) +{ + if (encoder == nullptr) { + return VK_FALSE; + } + return static_cast(encoder) + ->m_initialized.load() + ? VK_TRUE + : VK_FALSE; +} + +void VkEncFireCompletionEdge(VulkanVideoEncoderExt* encoder, uint64_t frameId) +{ + if (encoder == nullptr) { + return; + } + // The production edge, verbatim: OnBitstreamCaptured is what the + // assembly path calls on every capture push, so the stress exercises + // the real counter/signal/invoke sequence and the real lock order. + static_cast(encoder) + ->OnBitstreamCaptured(frameId); +} + +namespace { + +// Device-free capture source for VkEncPushCapture: the codec pure virtuals +// are stubbed (never called -- the null backend terminates every submit +// before the encoder would run) and the protected funnel entry point is +// re-exposed. Constructed with a null device context, which the base +// class's constructor (pure member-init) and destructor path (DeinitEncoder +// guards the missing device; no threads were started) both tolerate -- the +// same shape the capture-funnel unit tests rely on. +// AN H.264 ENCODER, NOT AN IMITATION OF ONE. The device-free capture +// backend derives from VkVideoEncoderH264 and holds a real +// EncoderConfigH264 so that the codec arm a test drives is the SHIPPED +// one: VkVideoEncoderH264::RefreshCodecRateControlParameters runs the +// real GetRateControlParameters into the real +// m_h264.m_rateControlLayersInfoH264, which is the struct +// CodecHandleRateControlCmd chains onto a control command. A stand-in +// that reimplemented that one line would prove only that the stand-in +// worked. +// +// NONE OF IT NEEDS A DEVICE. VkVideoEncoderH264 constructs from a null +// device context by pure member-init; its destructor joins threads that +// were never started and releases handles that were never created; and +// EncoderConfigH264 default-constructs and its rate-control fill reads +// config state only. The one thing InitEncoderCodec would have done that +// matters here is record the device QP window, so this declares one -- +// [0, 51], the window a real H.264 device reports. +class VkEncNullBackendCaptureSource : public VkVideoEncoderH264 { +public: + VkEncNullBackendCaptureSource() : VkVideoEncoderH264(nullptr) { + // Qualified: VkVideoEncoderH264 declares a private member of the + // same name that aliases this one in a device-initialized session. + // The base pointer is the one ApplyPendingRateControlUpdate writes + // and the one the refresh reads, so it is the one to stand up. + VkVideoEncoder::m_encoderConfig = + VkSharedBaseObj(new EncoderConfigH264()); + SetDeviceQpWindowForTest(0, 51); + } + + void PushCaptureRecord(uint64_t frameId, VkResult status) { + CapturedBitstream cap; + cap.frameId = frameId; + cap.isIdr = false; + cap.pictureType = 0; + cap.status = status; + PushCapturedBitstream(std::move(cap)); + } + + VkResult CreateFrameInfoBuffersQueue(uint32_t) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + bool GetAvailablePoolNode( + VkSharedBaseObj&) override { + return false; + } + VkResult InitEncoderCodec(VkSharedBaseObj&) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult EncodeFrame(VkSharedBaseObj&) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult CodecHandleRateControlCmd( + VkSharedBaseObj&) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult InitRateControl(VkCommandBuffer, uint32_t) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult ProcessDpb(VkSharedBaseObj&, + uint32_t, uint32_t) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } +}; + +} // namespace + +VkResult VkEncPushCapture(VulkanVideoEncoderExt* encoder, + uint64_t frameId, + VkResult status) +{ + if (encoder == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + // Injection stands in for a real session's completion path; it + // never runs beside one. + return VK_ERROR_NOT_PERMITTED_KHR; + } + VkSharedBaseObj source; + { + std::lock_guard lock(impl->m_pendingMutex); + if (!impl->m_encoder) { + VkSharedBaseObj created( + new VkEncNullBackendCaptureSource()); + // The production wiring, verbatim (InitializeExt): the push + // below then raises the real completion edge, which routes + // the record through DrainCapturesLocked under the real lock + // order. + created->SetOnBitstreamCaptured( + [impl](uint64_t id) { impl->OnBitstreamCaptured(id); }); + impl->m_encoder = created; + } + source = impl->m_encoder; + } + // Pushed OUTSIDE m_pendingMutex, like every production push: the + // completion edge takes m_callbackMutex then m_pendingMutex. The cast + // is sound because only this function installs an encoder on a + // null-backend session, and only null-backend sessions get here. + static_cast(source.get()) + ->PushCaptureRecord(frameId, status); + return VK_SUCCESS; +} + +VkResult VkEncApplyAndGetSessionConstQp(VulkanVideoEncoderExt* encoder, + int32_t* pQpIntra, + int32_t* pQpInterP, + int32_t* pQpInterB) +{ + if ((encoder == nullptr) || (pQpIntra == nullptr) || + (pQpInterP == nullptr) || (pQpInterB == nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + VkSharedBaseObj source; + { + std::lock_guard lock(impl->m_pendingMutex); + source = impl->m_encoder; + } + if (!source) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + return source->ApplyAndGetConstQpForTest(pQpIntra, pQpInterP, pQpInterB); +} + +VkResult VkEncApplyAndGetRateControl(VulkanVideoEncoderExt* encoder, + VkEncRateControlObservation* pOut) +{ + if ((encoder == nullptr) || (pOut == nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + VkSharedBaseObj source; + { + std::lock_guard lock(impl->m_pendingMutex); + source = impl->m_encoder; + } + if (!source) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + VkVideoEncoder::RateControlObservation observed{}; + const VkResult result = source->ApplyAndGetRateControlForTest(&observed); + if (result != VK_SUCCESS) { + return result; + } + *pOut = {}; + pOut->layerAverageBitrate = observed.layerAverageBitrate; + pOut->layerMaxBitrate = observed.layerMaxBitrate; + pOut->layerFrameRateNumerator = observed.layerFrameRateNumerator; + pOut->layerFrameRateDenominator = observed.layerFrameRateDenominator; + pOut->constQpIntra = observed.constQpIntra; + pOut->constQpInterP = observed.constQpInterP; + pOut->constQpInterB = observed.constQpInterB; + pOut->configMinQp = observed.configMinQp; + pOut->configMaxQp = observed.configMaxQp; + pOut->configMinQpSet = observed.configMinQpSet; + pOut->configMaxQpSet = observed.configMaxQpSet; + pOut->resolvedUseMinQp = observed.resolvedUseMinQp; + pOut->resolvedUseMaxQp = observed.resolvedUseMaxQp; + pOut->resolvedMinQpI = observed.resolvedMinQpI; + pOut->resolvedMaxQpI = observed.resolvedMaxQpI; + pOut->codecRefreshCount = observed.codecRefreshCount; + return VK_SUCCESS; +} + +VkResult VkEncGetRecordedConfig(VulkanVideoEncoderExt* encoder, + VkVideoEncoderConfig* pOut) +{ + if ((encoder == nullptr) || (pOut == nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + *pOut = impl->m_initConfig; + return VK_SUCCESS; +} + +VkResult VkEncSeedRecordedConfig(VulkanVideoEncoderExt* encoder, + const VkVideoEncoderConfig* pConfig) +{ + if ((encoder == nullptr) || (pConfig == nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + impl->m_initConfig = *pConfig; + impl->m_initConfig.pNext = nullptr; + return VK_SUCCESS; +} + +VkResult VkEncSetDeviceQpWindow(VulkanVideoEncoderExt* encoder, + int32_t minQp, int32_t maxQp) +{ + if (encoder == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VulkanVideoEncoderExtImpl* impl = + static_cast(encoder); + if (impl->m_nullBackend == nullptr) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + VkSharedBaseObj source; + { + std::lock_guard lock(impl->m_pendingMutex); + source = impl->m_encoder; + } + if (!source) { + return VK_ERROR_NOT_PERMITTED_KHR; + } + source->SetDeviceQpWindowForTest(minQp, maxQp); + return VK_SUCCESS; +} + +//============================================================================= +// Capability enumeration before InitializeExt() +// +// These free functions query the driver for a codec's encode capabilities +// WITHOUT creating a VkDevice or an encode session. They reuse the library's +// existing VkVideoCoreProfile + VulkanVideoCapabilities::GetVideoEncodeCap- +// abilities<>() machinery (the same code EncoderConfig*::InitDeviceCapabilities +// runs at session build time) and the VulkanDeviceContext instance/physical- +// device bring-up (which already tracks imported-handle ownership on this +// path via m_importedInstanceHandle, so caller handles are never destroyed). +//============================================================================= + +namespace { + +// (codec, profile, bit depth) -> probe parameters. The probe is +// per-VkVideoProfileInfoKHR, so each advertised (profile-idc, bit-depth) +// combination must be queried literally. VK_VIDEO_ENCODER_PROFILE_DEFAULT +// probes the representative profile per codec (H.264 High, H.265 Main, AV1 +// Main -- all 8-bit); a named profile probes exactly itself. +// +// THE BIT DEPTH IS PART OF THE KEY AND IS NOT DERIVED FROM THE PROFILE +// NUMBER. For H.264 and H.265 the number does decide the depth -- Baseline, +// Main and High are 8-bit, Main 10 is the 10-bit one -- so those arms accept +// exactly the depth their profile carries and refuse every other, which is +// the answer they already gave. AV1 IS NOT LIKE THAT: seq_profile 0 (Main) +// carries 8 OR 10 bits at 4:2:0 (AV1 A.2), one profile at two depths. A key +// that was only a profile number could not name the 10-bit half at all, so it +// was the one combination this probe could not put to a driver -- while a +// 10-bit AV1 session is built and encoded on hardware today. +// +// H.264 High 10 (profile_idc 110) is deliberately NOT a row here. It has +// never been probed on any device in this tree, and a row is an +// advertisement: adding one would claim a capability with no evidence behind +// it. The same reasoning keeps H.265 Main 10 at 10 bits only, although the +// standard admits 8-bit input under it (H.265 A.3.3) -- that pairing has +// never been probed either. Widening either is a row with a visible diff, +// which is the point of putting the depth in the key. +struct ProbeProfile { + uint32_t profileIdc; + VkVideoComponentBitDepthFlagBitsKHR lumaBits; + VkVideoComponentBitDepthFlagBitsKHR chromaBits; + VkVideoChromaSubsamplingFlagBitsKHR chromaSubsampling; +}; + +// Returns false when `profile` is not a value this library probes for `codec` +// (the caller maps that onto VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR). +// Profile numbers are the codec standard's own, so they repeat across codecs: +// 1 is H.265 Main and is not an H.264 profile_idc at all. The codec arm is +// what disambiguates, and a number that belongs to another codec falls through +// to the refusal rather than answering with that codec's capabilities. +// The Vulkan component-bit-depth flag for |bitDepth|, or false for a depth +// this probe cannot spell. NOT a default: silently probing 8 bits for a +// caller that asked about 12 would answer a question nobody put. +static bool MapProbeBitDepth(uint32_t bitDepth, + VkVideoComponentBitDepthFlagBitsKHR& out) +{ + switch (bitDepth) { + case 8: + out = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; return true; + case 10: + out = VK_VIDEO_COMPONENT_BIT_DEPTH_10_BIT_KHR; return true; + default: + return false; + } +} + +static bool MapProbeProfile(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t bitDepth, + ProbeProfile& out) +{ + VkVideoComponentBitDepthFlagBitsKHR depthFlag = + VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + if (!MapProbeBitDepth(bitDepth, depthFlag)) { + return false; + } + out.lumaBits = depthFlag; + out.chromaBits = depthFlag; + // THE PROBE ENVELOPE, and it is narrower than the taxonomy it serves. + // + // Every profile below is 4:2:0, so every capability query THIS PROBE + // issues asks the device about a 4:2:0 profile. + // + // THAT IS THE RIGHT ENVELOPE FOR WHAT THE PROBE STILL ANSWERS -- the + // capability SCALARS: coded extent, bitrate ceiling, DPB slots, rate + // control modes, quality levels, std syntax flags. It was the wrong one + // for a format list, and a format list is no longer built from it: + // VkEncEnumerateInputFormats resolves each candidate live, at the profile + // that candidate's own binding derives, so a 4:4:4 input asks the device + // about a 4:4:4 profile whichever entry point put the question. + // + // Widening the envelope is still not a matter of changing the value below. + // A 4:4:4 encode profile IS a different profile -- H.264 High 4:4:4 + // Predictive, H.265 Range Extensions -- so it needs a row of its own here + // to be probed at all, and what a driver then answers is a device fact to + // be measured rather than assumed. The field exists so that widening is a + // row in this table with a visible diff, rather than a literal buried in + // the call below. + out.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + const bool eightBit = (bitDepth == 8); + const bool tenBit = (bitDepth == 10); + switch ((uint32_t)codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + // Depth is gated per profile, not once for the codec: the + // admissible depth is a property of the profile, not of H.264. + // Baseline, Main and High admit 8-bit input only (H.264 A.2); + // High 10 admits 10-bit. Gating the codec as a whole would + // refuse a pairing before its profile is read. + switch (profile) { + case VK_VIDEO_ENCODER_PROFILE_DEFAULT: + case VK_VIDEO_ENCODER_PROFILE_H264_HIGH: + if (!eightBit) { + return false; + } + out.profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH; return true; + case VK_VIDEO_ENCODER_PROFILE_H264_BASELINE: + if (!eightBit) { + return false; + } + out.profileIdc = STD_VIDEO_H264_PROFILE_IDC_BASELINE; return true; + case VK_VIDEO_ENCODER_PROFILE_H264_MAIN: + if (!eightBit) { + return false; + } + out.profileIdc = STD_VIDEO_H264_PROFILE_IDC_MAIN; return true; + case VK_VIDEO_ENCODER_PROFILE_H264_HIGH_10: + if (!tenBit) { + return false; + } + out.profileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH_10; return true; + default: + return false; + } + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + switch (profile) { + case VK_VIDEO_ENCODER_PROFILE_DEFAULT: + case VK_VIDEO_ENCODER_PROFILE_H265_MAIN: + // Main is 8-bit 4:2:0 (H.265 A.3.2). + if (!eightBit) { + return false; + } + out.profileIdc = STD_VIDEO_H265_PROFILE_IDC_MAIN; return true; + case VK_VIDEO_ENCODER_PROFILE_H265_MAIN10: + // 10 bits only, which is NARROWER than H.265 A.3.3 + // allows. See the note on the struct above for why the + // 8-bit Main 10 pairing is not a row. + if (!tenBit) { + return false; + } + out.profileIdc = STD_VIDEO_H265_PROFILE_IDC_MAIN_10; return true; + default: + return false; + } + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + switch (profile) { + // seq_profile 0 IS Main and IS DEFAULT: one case, not two. + // BOTH DEPTHS, because Main is one profile at two of them and + // the depth is a separate term of this key. This is the row + // the table could not previously express. + case VK_VIDEO_ENCODER_PROFILE_AV1_MAIN: + if (!eightBit && !tenBit) { + return false; + } + out.profileIdc = STD_VIDEO_AV1_PROFILE_MAIN; return true; + default: + return false; + } + default: + return false; + } +} + +static bool DeviceHasExtension(const VulkanDeviceContext& ctx, + const char* extName) +{ + // Read the context's populated list rather than re-asking the driver. + // + // This used to enumerate device extensions from scratch on every call -- + // two vkEnumerateDeviceExtensionProperties round-trips each, three calls + // per capability probe, so six for a list the context already holds. That + // was a workaround, not a preference: PopulateDeviceExtensions() lives + // inside InitPhysicalDevice's selection block, and the caller-handle + // capability path could never reach that block (it passed + // requestQueueTypes = 0), so m_deviceExtensions was empty and a lookup + // would have answered false for everything. + // + // Both bring-up paths now populate it -- AdoptPhysicalDevice() for the + // adopted probe, InitPhysicalDevice()'s selection for the ephemeral one + // -- so the list is authoritative and the re-query is dead weight. + return ctx.FindDeviceExtension(extName) != nullptr; +} + +// Query caps for (`codec`, `profile`, `bitDepth`) against the (already +// instance+physical-device initialized) device context. No VkDevice / +// session is created. +static VkResult QueryEncoderCapsInternal(const VulkanDeviceContext& ctx, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t bitDepth, + VkEncProfileCapabilitySnapshot* outSnapshot) +{ + if (outSnapshot == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VkVideoEncoderCapabilities* const outCaps = &outSnapshot->caps; + switch ((uint32_t)codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + break; + default: + return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + } + + ProbeProfile pp; + if (!MapProbeProfile(codec, profile, bitDepth, pp)) { + // Valid codec, but (`profile`, `bitDepth`) is not a pairing this + // library probes for it -- either the profile belongs to another + // codec, or the depth is not one this profile is probed at. + return VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR; + } + + VkVideoCoreProfile coreProfile(codec, + pp.chromaSubsampling, + pp.lumaBits, pp.chromaBits, + pp.profileIdc); + + VkVideoCapabilitiesKHR videoCaps{}; + VkVideoEncodeCapabilitiesKHR encodeCaps{}; + VkVideoEncodeQuantizationMapCapabilitiesKHR qpMapCaps{}; + VkVideoEncodeIntraRefreshCapabilitiesKHR intraRefreshCaps{}; + + VkResult result = VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + if ((outCaps->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES) || + (outCaps->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + *outSnapshot = {}; + outCaps->codec = codec; + + // Per-codec caps query (reuses the library's chaining template). + switch ((uint32_t)codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: { + VkVideoEncodeH264CapabilitiesKHR h264Caps{}; + VkVideoEncodeH264QuantizationMapCapabilitiesKHR h264QpMap{}; + result = VulkanVideoCapabilities::GetVideoEncodeCapabilities< + VkVideoEncodeH264CapabilitiesKHR, + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_CAPABILITIES_KHR, + VkVideoEncodeH264QuantizationMapCapabilitiesKHR, + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_QUANTIZATION_MAP_CAPABILITIES_KHR>( + &ctx, coreProfile, videoCaps, encodeCaps, h264Caps, + qpMapCaps, h264QpMap, intraRefreshCaps); + if (result == VK_SUCCESS) { + outCaps->maxLevelIdc = (uint32_t)h264Caps.maxLevelIdc; + outSnapshot->stdFlags[outSnapshot->stdFlagCount++] = + h264Caps.stdSyntaxFlags; + } + break; + } + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: { + VkVideoEncodeH265CapabilitiesKHR h265Caps{}; + VkVideoEncodeH265QuantizationMapCapabilitiesKHR h265QpMap{}; + result = VulkanVideoCapabilities::GetVideoEncodeCapabilities< + VkVideoEncodeH265CapabilitiesKHR, + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_CAPABILITIES_KHR, + VkVideoEncodeH265QuantizationMapCapabilitiesKHR, + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_QUANTIZATION_MAP_CAPABILITIES_KHR>( + &ctx, coreProfile, videoCaps, encodeCaps, h265Caps, + qpMapCaps, h265QpMap, intraRefreshCaps); + if (result == VK_SUCCESS) { + outCaps->maxLevelIdc = (uint32_t)h265Caps.maxLevelIdc; + outSnapshot->stdFlags[outSnapshot->stdFlagCount++] = + h265Caps.stdSyntaxFlags; + } + break; + } + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: { + VkVideoEncodeAV1CapabilitiesKHR av1Caps{}; + VkVideoEncodeAV1QuantizationMapCapabilitiesKHR av1QpMap{}; + result = VulkanVideoCapabilities::GetVideoEncodeCapabilities< + VkVideoEncodeAV1CapabilitiesKHR, + VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_CAPABILITIES_KHR, + VkVideoEncodeAV1QuantizationMapCapabilitiesKHR, + VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_QUANTIZATION_MAP_CAPABILITIES_KHR>( + &ctx, coreProfile, videoCaps, encodeCaps, av1Caps, + qpMapCaps, av1QpMap, intraRefreshCaps); + if (result == VK_SUCCESS) { + outCaps->maxLevelIdc = (uint32_t)av1Caps.maxLevel; + outSnapshot->stdFlags[outSnapshot->stdFlagCount++] = + av1Caps.stdSyntaxFlags; + } + break; + } + default: + return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + } + + if (result != VK_SUCCESS) { + // Driver does not support encode for this codec on this device. Leave + // *outCaps zeroed (codec field already set) and propagate the result; + // callers (e.g. the Chromium enumerator) treat this as "no profile". + return result; + } + + // --- Map the common (codec-agnostic) capability fields. --- + outCaps->minLevelIdc = 0; // no min-level cap in Vulkan today + outCaps->maxDpbSlots = videoCaps.maxDpbSlots; + outCaps->maxActiveReferencePictures= videoCaps.maxActiveReferencePictures; + outCaps->maxQualityLevels = encodeCaps.maxQualityLevels; + outCaps->maxBitrate = encodeCaps.maxBitrate; + outCaps->pictureAccessGranularityWidth = videoCaps.pictureAccessGranularity.width; + outCaps->pictureAccessGranularityHeight = videoCaps.pictureAccessGranularity.height; + outCaps->supportedRateControlModes = encodeCaps.rateControlModes; + outCaps->flags = encodeCaps.flags; + outCaps->minCodedExtent = videoCaps.minCodedExtent; + outCaps->maxCodedExtent = videoCaps.maxCodedExtent; + + // Optional-feature availability: extension presence + non-empty caps. + outCaps->supportsMaintenance1 = + DeviceHasExtension(ctx, VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME); + outCaps->supportsQuantizationMap = + DeviceHasExtension(ctx, VK_KHR_VIDEO_ENCODE_QUANTIZATION_MAP_EXTENSION_NAME); + outCaps->supportsIntraRefresh = + DeviceHasExtension(ctx, VK_KHR_VIDEO_ENCODE_INTRA_REFRESH_EXTENSION_NAME) && + (intraRefreshCaps.intraRefreshModes != 0); + // Resize-without-IDR is not advertised by a dedicated Vulkan cap today; the + // safe/default answer is false (resize forces an IDR unless a driver + // later exposes an explicit capability we can map here). + outCaps->supportsResizeWithoutIdr = false; + + // NO INPUT-FORMAT LIST IS BUILT HERE. Building one from a device format + // query issued at this probe's own 4:2:0 envelope is the right + // envelope for the scalars above and the wrong one for a format list. A + // caller asking which formats it may feed the encoder is asking about the + // profile ITS OWN input derives, and no fixed envelope answers that for + // every input. VkEncEnumerateInputFormats now resolves each candidate live, + // through the same function the point query answers from. + return VK_SUCCESS; +} + +// Common device-extension requests for a capability-only bring-up. We request +// the video-queue + encode-queue extensions (required to load the physical- +// device video PFNs) and the optional feature extensions so DeviceHasExtension +// reflects true device support. +static void AddCapsProbeExtensions(VulkanDeviceContext& ctx) +{ + static const char* const requiredDeviceExtension[] = { + VK_KHR_VIDEO_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME, + nullptr + }; + static const char* const optionalDeviceExtension[] = { + VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_H264_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_H265_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_AV1_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_QUANTIZATION_MAP_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_INTRA_REFRESH_EXTENSION_NAME, + nullptr + }; + ctx.AddReqDeviceExtensions(requiredDeviceExtension); + ctx.AddOptDeviceExtensions(optionalDeviceExtension); +} + +} // anonymous namespace + +//============================================================================= +// VulkanVideoEncoderContext -- the encoder context (context design 3.1, 3.2) +// +// Phases 1+2 of the library's Vulkan bring-up held once, so the capability +// path stops paying a loader load, an instance creation and a teardown per +// query (D1). Phases 3+4 -- the VkDevice and its queues -- stay with the +// encode session; a context never creates one. +//============================================================================= + +namespace { + +// Snapshot table geometry. +// +// Profile numbers are the codec standard's own: they are not contiguous, they +// do not start at the same place per codec, and they repeat across codecs. So +// the profile axis is a SLOT INDEX into a per-codec probe list rather than the +// number itself, and the list below is the single place that says which pairs +// the snapshot covers. +enum { + VK_ENC_CTX_CODEC_H264 = 0, + VK_ENC_CTX_CODEC_H265 = 1, + VK_ENC_CTX_CODEC_AV1 = 2, + VK_ENC_CTX_CODEC_COUNT = 3, +}; + +// One row of the snapshot: the profile number AND the bit depth it is probed +// at, because the probe is per-VkVideoProfileInfoKHR and the depth is part of +// that structure. Two rows may carry the SAME profile number at different +// depths -- see the AV1 list -- which is exactly what a profile number alone +// could not express. +struct VkEncCtxProbeEntry { + uint32_t profile; + uint32_t bitDepth; +}; + +const VkEncCtxProbeEntry kVkEncCtxProfilesH264[] = { + { VK_VIDEO_ENCODER_PROFILE_DEFAULT, 8 }, + { VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, 8 }, + { VK_VIDEO_ENCODER_PROFILE_H264_MAIN, 8 }, + { VK_VIDEO_ENCODER_PROFILE_H264_HIGH, 8 }, + { VK_VIDEO_ENCODER_PROFILE_H264_HIGH_10, 10 }, +}; +const VkEncCtxProbeEntry kVkEncCtxProfilesH265[] = { + { VK_VIDEO_ENCODER_PROFILE_DEFAULT, 8 }, + { VK_VIDEO_ENCODER_PROFILE_H265_MAIN, 8 }, + { VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, 10 }, +}; +// AV1 seq_profile 0 IS Main and IS DEFAULT, so the codec has one PROFILE +// NUMBER and not two: a second row holding a DIFFERENT number would report +// the same profile under two names. It has TWO ROWS all the same, because +// Main carries 8 or 10 bits (AV1 A.2) and the row key is (profile, depth). +// Those are two different VkVideoProfileInfoKHR values and a driver answers +// them separately, so one row could only ever describe half the profile. +// +// THE 8-BIT ROW IS FIRST, AND THAT ORDER IS LOAD-BEARING. The public lookup +// below resolves a profile number to the FIRST row carrying it, so every +// published answer for AV1 Main is the 8-bit one, byte for byte what it was. +// +// THE 10-BIT ROW IS PROBED AND STORED BUT NOT PUBLISHED, deliberately and not +// by oversight. There is no public key for it: every capability entry point +// names a profile by the codec standard number alone, and AV1 has one number +// for both depths. Publishing it needs either a depth argument on those entry +// points -- a public-header change this library does not make on its own -- or +// a rule for folding two rows into one answer. Folding is not free: the +// scalars (coded extent, bitrate ceiling, rate-control modes, quality levels) +// may differ between the depths, and an advertisement assembled from the wider +// of two rows would outrun what a session at the other depth accepts. Neither +// is decided here. What IS decided here is that the question now reaches the +// driver, so whichever route is chosen has an answer to publish. +const VkEncCtxProbeEntry kVkEncCtxProfilesAV1[] = { + { VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, 8 }, + { VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, 10 }, +}; + +// The widest per-codec list. Sized from the lists so a list that grows past +// the array fails the build here instead of truncating the snapshot. +enum { VK_ENC_CTX_PROFILE_SLOTS = 5 }; +static_assert(sizeof(kVkEncCtxProfilesH264) / sizeof(VkEncCtxProbeEntry) <= + VK_ENC_CTX_PROFILE_SLOTS && + sizeof(kVkEncCtxProfilesH265) / sizeof(VkEncCtxProbeEntry) <= + VK_ENC_CTX_PROFILE_SLOTS && + sizeof(kVkEncCtxProfilesAV1) / sizeof(VkEncCtxProbeEntry) <= + VK_ENC_CTX_PROFILE_SLOTS, + "a per-codec probe list outgrew the snapshot's profile axis"); + +// The probe list for |codec|, or nullptr with a zero count for a codec the +// snapshot does not cover. +const VkEncCtxProbeEntry* VkEncCtxProfileList( + VkVideoCodecOperationFlagBitsKHR codec, uint32_t& outCount) +{ + switch ((uint32_t)codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + outCount = sizeof(kVkEncCtxProfilesH264) / sizeof(VkEncCtxProbeEntry); + return kVkEncCtxProfilesH264; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + outCount = sizeof(kVkEncCtxProfilesH265) / sizeof(VkEncCtxProbeEntry); + return kVkEncCtxProfilesH265; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + outCount = sizeof(kVkEncCtxProfilesAV1) / sizeof(VkEncCtxProbeEntry); + return kVkEncCtxProfilesAV1; + default: + outCount = 0; + return nullptr; + } +} + +// The probe row at |slot| for |codec|. False when the codec has no such slot. +bool VkEncCtxProbeAtSlot(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t slot, VkEncCtxProbeEntry& outEntry) +{ + uint32_t count = 0; + const VkEncCtxProbeEntry* list = VkEncCtxProfileList(codec, count); + if ((list == nullptr) || (slot >= count)) { + return false; + } + outEntry = list[slot]; + return true; +} + +// Slot holding |profile| for |codec|, or -1 when this library does not probe +// that pair. A number belonging to a different codec lands here, which is what +// keeps a cross-codec request from reading another codec's row. +// +// FIRST MATCH ON THE PROFILE NUMBER, and the depth is not part of the lookup +// because it is not part of the public key -- an entry point names a profile +// by the standard number alone. Where a codec has two rows under one number +// (AV1 Main, at 8 and 10 bits) this therefore resolves to the first, which the +// list orders as the 8-bit one so that every published answer is unchanged. +int32_t VkEncCtxProfileSlotIndex(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile) +{ + uint32_t count = 0; + const VkEncCtxProbeEntry* list = VkEncCtxProfileList(codec, count); + for (uint32_t slot = 0; slot < count; slot++) { + if (list[slot].profile == profile) { + return (int32_t)slot; + } + } + return -1; +} + +VkVideoCodecOperationFlagBitsKHR VkEncCtxCodecAtIndex(uint32_t index) +{ + switch (index) { + case VK_ENC_CTX_CODEC_H264: + return VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + case VK_ENC_CTX_CODEC_H265: + return VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + case VK_ENC_CTX_CODEC_AV1: + return VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + default: + return VK_VIDEO_CODEC_OPERATION_NONE_KHR; + } +} + +int32_t VkEncCtxCodecIndex(VkVideoCodecOperationFlagBitsKHR codec) +{ + switch ((uint32_t)codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + return VK_ENC_CTX_CODEC_H264; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + return VK_ENC_CTX_CODEC_H265; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + return VK_ENC_CTX_CODEC_AV1; + default: + return -1; + } +} + +bool VkEncCtxUuidIsZero(const uint8_t* uuid) +{ + for (uint32_t i = 0; i < VK_UUID_SIZE; i++) { + if (uuid[i] != 0) { + return false; + } + } + return true; +} + +// The 64-bit format features a DRM modifier must carry for an image with +// |usage| to be creatable on it. Returns false -- refusing to answer -- for a +// usage bit this mapping does not cover, because a filter that silently drops +// a term it did not understand is worse than one that says it cannot answer. +bool VkEncCtxUsageToFormatFeatures(VkImageUsageFlags usage, + VkFormatFeatureFlags2& outFeatures) +{ + struct UsageFeature { + VkImageUsageFlags usageBit; + VkFormatFeatureFlags2 featureBit; + }; + static const UsageFeature kMap[] = { + { VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR, + VK_FORMAT_FEATURE_2_VIDEO_ENCODE_INPUT_BIT_KHR }, + { VK_IMAGE_USAGE_VIDEO_ENCODE_DPB_BIT_KHR, + VK_FORMAT_FEATURE_2_VIDEO_ENCODE_DPB_BIT_KHR }, + { VK_IMAGE_USAGE_TRANSFER_SRC_BIT, + VK_FORMAT_FEATURE_2_TRANSFER_SRC_BIT }, + { VK_IMAGE_USAGE_TRANSFER_DST_BIT, + VK_FORMAT_FEATURE_2_TRANSFER_DST_BIT }, + { VK_IMAGE_USAGE_SAMPLED_BIT, + VK_FORMAT_FEATURE_2_SAMPLED_IMAGE_BIT }, + { VK_IMAGE_USAGE_STORAGE_BIT, + VK_FORMAT_FEATURE_2_STORAGE_IMAGE_BIT }, + { VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT, + VK_FORMAT_FEATURE_2_COLOR_ATTACHMENT_BIT }, + }; + + VkFormatFeatureFlags2 features = 0; + VkImageUsageFlags unmapped = usage; + for (size_t i = 0; i < sizeof(kMap) / sizeof(kMap[0]); i++) { + if ((usage & kMap[i].usageBit) != 0) { + features |= kMap[i].featureBit; + unmapped &= ~kMap[i].usageBit; + } + } + if (unmapped != 0) { + VkEncErr() << "[EncoderContext] no format-feature mapping for image " + << "usage bits 0x" << std::hex << unmapped << std::dec + << "; refusing to filter modifiers on a term this build " + << "does not understand" << std::endl; + return false; + } + outFeatures = features; + return true; +} + +} // anonymous namespace + +//============================================================================= +// CAPABILITY-PROBE KEY OBSERVATION SEAM (internal; device-free). +// +// What the probe CAN ask a driver is a build-time fact of the two tables +// above, and it is invisible from the public surface on the only host where +// it matters: a machine with no encode-capable device enumerates no device, +// so every capability entry point answers "not present" whatever the tables +// say. These three read the tables directly, so the shape is regression-tested +// on a GPU-less runner and a row that disappears fails a test rather than an +// advertisement. +// +// What they CANNOT say is whether a driver answers yes to any of it. That is a +// device fact and is measured on hardware or not at all. +//============================================================================= + +// Is (codec, profile, bitDepth) a combination this library probes? +bool VkEncProbeNamesProfileBitDepth(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t bitDepth) +{ + ProbeProfile pp; + return MapProbeProfile(codec, profile, bitDepth, pp); +} + +// How many probe rows the context snapshot carries for |codec|. +uint32_t VkEncProbeSnapshotRowCount(VkVideoCodecOperationFlagBitsKHR codec) +{ + uint32_t count = 0; + (void)VkEncCtxProfileList(codec, count); + return count; +} + +// The (profile, bit depth) of snapshot row |slot| for |codec|. +bool VkEncProbeSnapshotRowAt(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t slot, + uint32_t* outProfile, + uint32_t* outBitDepth) +{ + VkEncCtxProbeEntry entry = {}; + if (!VkEncCtxProbeAtSlot(codec, slot, entry)) { + return false; + } + if (outProfile != nullptr) { + *outProfile = entry.profile; + } + if (outBitDepth != nullptr) { + *outBitDepth = entry.bitDepth; + } + return true; +} + +class VulkanVideoEncoderContext : public VkVideoRefCountBase { +public: + // One enumerated physical device and everything the context knows about + // it. Written ONLY by Build(), which runs inside Create() under the + // construction lock; const for the object's whole visible life. + struct DeviceEntry { + VkPhysicalDevice physDevice = VK_NULL_HANDLE; + VkVideoEncoderDeviceIdentity identity = {}; + + // Union of videoCodecOperations over the queue families that carry + // VK_QUEUE_VIDEO_ENCODE_BIT_KHR. This is the exact predicate + // InitPhysicalDevice() applies when it picks a "first capable" + // device for a codec, kept here so the ephemeral capability entry + // points can apply the same one without a second bring-up. + VkVideoCodecOperationFlagsKHR encodeCodecOps = 0; + + bool hasRequiredVideoExtensions = false; + bool hasDrmFormatModifierExtension = false; + // 64-bit format features, needed to answer a modifier query about + // VIDEO_ENCODE_INPUT at all. + bool hasFormatFeatureFlags2 = false; + + // probed[c][p] is false when no driver query was issued for that + // pair, in which case capsResult[c][p] says why and caps[c][p] must + // not be handed out. + bool probed[VK_ENC_CTX_CODEC_COUNT] + [VK_ENC_CTX_PROFILE_SLOTS]; + VkResult capsResult[VK_ENC_CTX_CODEC_COUNT] + [VK_ENC_CTX_PROFILE_SLOTS]; + VkEncProfileCapabilitySnapshot snapshot[VK_ENC_CTX_CODEC_COUNT] + [VK_ENC_CTX_PROFILE_SLOTS]; + }; + + static VkResult Create(const VkVideoEncoderContextCreateInfo& createInfo, + VkSharedBaseObj& outContext); + + uint32_t GetPhysicalDeviceCount() const { + return (uint32_t)m_devices.size(); + } + const DeviceEntry* GetDeviceEntry(uint32_t index) const { + return (index < m_devices.size()) ? &m_devices[index] : nullptr; + } + const VulkanDeviceContext& GetDeviceContext() const { return *m_devCtx; } + VkVideoEncoderContextMode GetMode() const { return m_mode; } + + virtual ~VulkanVideoEncoderContext() { + // Nothing Vulkan is destroyed by this body. What the members do on + // the way out differs by mode. + // + // ADOPT (rule 3): the instance is flagged imported inside + // VulkanDeviceContext, so its destructor leaves it alone, and the + // physical device was never ours to destroy. Rule 1: the loader + // handle was retained in Build() before anything could fail, so the + // destructor below does not unload it either. + // + // OWN (rule 2): reached only from VkEncRetireOwnContexts(), which is + // the sole release of the floor reference that otherwise holds an + // OWN-mode context above zero. ~VulkanDeviceContext then destroys the + // instance -- so the destruction this comment says does not happen + // here happens one member below, at a moment the caller picked. A + // later create cannot re-issue vkCreateInstance because retirement is + // latched, not because nothing was ever destroyed. + } + +private: + VulkanVideoEncoderContext() = default; + VkResult Build(const VkVideoEncoderContextCreateInfo& createInfo); + + VkVideoEncoderContextMode m_mode = + VK_VIDEO_ENCODER_CONTEXT_MODE_OWN; + std::unique_ptr m_devCtx; + std::vector m_devices; + // This context's silence request, held for its whole life. Released by + // the destructor, which is what makes silence end with the last owner + // rather than with whoever assigned the flag most recently. + VkEncoderStdioSilenceScope m_stdioSilence; +}; + +namespace { + +struct VkEncOwnContextEntry { + uint8_t gpuUUID[VK_UUID_SIZE]; + VkSharedBaseObj context; +}; + +// The floor reference (design 3.1 rule 2). +// +// An OWN-mode context is cached here for the process lifetime so that repeated +// create/destroy cycles reuse one VkInstance. The rule this enforces is NOT +// "never destroyed" but "never re-created": a second vkCreateInstance must not +// be issued after the first, because by then a sandbox may have locked the +// process down and the create would fail where a reuse would have worked. +// +// Those two are separable, and VkEncRetireOwnContexts() separates them. The +// registry is released only through that call, which latches +// VkEncOwnContextsRetired() on the way so no later create can stand a second +// instance up. Release at an arbitrary static-destruction time is what must +// not happen -- the driver is called from an exit path with no ordering +// guarantee against the loader -- so nothing here has a destructor: the +// pointer is a function-local static, which also leaves no static-init order +// to get wrong. +std::vector& VkEncOwnContextRegistry() +{ + static std::vector* const registry = + new std::vector(); + return *registry; +} + +// Latched by VkEncRetireOwnContexts() and never cleared. Read under +// VkEncContextConstructionMutex(), which is also what makes retire-vs-create +// a decided order rather than a race. +bool& VkEncOwnContextsRetired() +{ + static bool* const retired = new bool(false); + return *retired; +} + +// The construction lock. Guards the floor registry AND the snapshot build, so +// two sequences racing to create the same OWN context get one instance and +// one snapshot rather than two. +std::mutex& VkEncContextConstructionMutex() +{ + static std::mutex* const constructionMutex = new std::mutex(); + return *constructionMutex; +} + +} // anonymous namespace + +VkResult VulkanVideoEncoderContext::Build( + const VkVideoEncoderContextCreateInfo& createInfo) +{ + m_mode = createInfo.mode; + m_devCtx.reset(new VulkanDeviceContext()); + + // Rule 1, and set BEFORE anything can fail: even a context whose bring-up + // fails must leave the loader mapped, because the failure path still ran + // dlopen and an embedder may already be bound to the same object. + m_devCtx->RetainLoaderHandle(); + + AddCapsProbeExtensions(*m_devCtx); + + const VkInstance adoptInstance = + (m_mode == VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT) + ? createInfo.adoptInstance : VK_NULL_HANDLE; + + // Phase 1. With a non-null instance this imports rather than creates: + // VulkanDeviceContext records m_importedInstanceHandle and its destructor + // will not destroy it. + VkResult result = m_devCtx->InitVulkanDevice("VulkanVideoEncoderContext", + adoptInstance, + /*verbose*/ false); + if (result != VK_SUCCESS) { + return result; + } + + // Phase 2. + std::vector candidates; + if (m_mode == VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT) { + // ADOPT never re-selects (D8): the caller already chose the device, + // and a context only issues physical-device-level queries, so there + // is no queue family to pick and no VkDevice to create. + candidates.push_back(createInfo.adoptPhysicalDevice); + } else { + result = vk::enumerate(m_devCtx.get(), m_devCtx->getInstance(), + candidates); + if (result != VK_SUCCESS) { + return result; + } + } + + const bool pinToUuid = (m_mode == VK_VIDEO_ENCODER_CONTEXT_MODE_OWN) && + !VkEncCtxUuidIsZero(createInfo.gpuUUID); + + for (size_t candidate = 0; candidate < candidates.size(); candidate++) { + VkPhysicalDevice physDevice = candidates[candidate]; + if (physDevice == VK_NULL_HANDLE) { + continue; + } + + VkPhysicalDeviceVulkan11Properties vulkan11Props = {}; + vulkan11Props.sType = + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_1_PROPERTIES; + VkPhysicalDeviceProperties2 props2 = {}; + props2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2; + props2.pNext = &vulkan11Props; + m_devCtx->GetPhysicalDeviceProperties2(physDevice, &props2); + + if (pinToUuid && (memcmp(vulkan11Props.deviceUUID, createInfo.gpuUUID, + VK_UUID_SIZE) != 0)) { + continue; + } + + // Populates the device-extension list this device's probes read + // through DeviceHasExtension(). PopulateDeviceExtensions() RESIZES + // that list, so it is per candidate rather than accumulated -- unlike + // m_reqDeviceExtensions, which HasAllDeviceExtensions() appends to and + // which this loop deliberately never touches. + result = m_devCtx->AdoptPhysicalDevice(physDevice); + if (result != VK_SUCCESS) { + if (m_mode == VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT) { + return result; + } + continue; + } + + DeviceEntry entry; + memcpy(entry.identity.deviceUUID, vulkan11Props.deviceUUID, + VK_UUID_SIZE); + memcpy(entry.identity.driverUUID, vulkan11Props.driverUUID, + VK_UUID_SIZE); + memcpy(entry.identity.deviceName, props2.properties.deviceName, + sizeof(entry.identity.deviceName)); + entry.identity.deviceName[sizeof(entry.identity.deviceName) - 1] = 0; + entry.identity.vendorID = props2.properties.vendorID; + entry.identity.deviceID = props2.properties.deviceID; + entry.physDevice = physDevice; + + entry.hasRequiredVideoExtensions = + (m_devCtx->FindDeviceExtension( + VK_KHR_VIDEO_QUEUE_EXTENSION_NAME) != nullptr) && + (m_devCtx->FindDeviceExtension( + VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME) != nullptr); + entry.hasDrmFormatModifierExtension = + (m_devCtx->FindDeviceExtension( + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME) != nullptr); + entry.hasFormatFeatureFlags2 = + (props2.properties.apiVersion >= VK_API_VERSION_1_3) || + (m_devCtx->FindDeviceExtension( + VK_KHR_FORMAT_FEATURE_FLAGS_2_EXTENSION_NAME) != nullptr); + + // Which codecs the device's encode-capable queue families carry. + std::vector queues; + std::vector videoQueues; + std::vector queryStatus; + vk::get(m_devCtx.get(), physDevice, queues, videoQueues, queryStatus); + // The three arrays are filled in parallel, one entry per family; + // indexing one against another's length would be an out-of-bounds + // read past a check that looked like it covered it. + if (videoQueues.size() == queues.size()) { + for (size_t family = 0; family < queues.size(); family++) { + if ((queues[family].queueFamilyProperties.queueFlags & + VK_QUEUE_VIDEO_ENCODE_BIT_KHR) != 0) { + entry.encodeCodecOps |= + videoQueues[family].videoCodecOperations; + } + } + } + + // The capability snapshot itself -- the whole reason the context + // exists. Every (codec, profile) pair the library can probe is + // resolved here, once, so a later query is a table read. + for (uint32_t c = 0; c < VK_ENC_CTX_CODEC_COUNT; c++) { + const VkVideoCodecOperationFlagBitsKHR codec = + VkEncCtxCodecAtIndex(c); + for (uint32_t p = 0; p < VK_ENC_CTX_PROFILE_SLOTS; p++) { + entry.snapshot[c][p] = {}; + entry.probed[c][p] = false; + entry.capsResult[c][p] = VK_ERROR_EXTENSION_NOT_PRESENT; + + // A slot this codec does not have. The array is rectangular + // and the lists are not, so the surplus rows exist and must + // say why they hold nothing. + VkEncCtxProbeEntry slotProbe = {}; + if (!VkEncCtxProbeAtSlot(codec, p, slotProbe)) { + entry.capsResult[c][p] = + VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR; + continue; + } + + // vkGetPhysicalDeviceVideoCapabilitiesKHR is an entry point + // of VK_KHR_video_queue; asking a device that does not + // advertise it is invalid usage, not a cheap "no". ADOPT is + // exempt because the caller named the device explicitly and + // the pre-context entry point probed it unconditionally + // -- changing that would be a behaviour change, not a fix. + if (!entry.hasRequiredVideoExtensions && + (m_mode != VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT)) { + continue; + } + + VkEncProfileCapabilitySnapshot probeSnapshot = {}; + const VkResult probeResult = QueryEncoderCapsInternal( + *m_devCtx, codec, slotProbe.profile, slotProbe.bitDepth, + &probeSnapshot); + entry.capsResult[c][p] = probeResult; + // QueryEncoderCapsInternal leaves *out untouched when it + // rejects the (codec, profile) pair before probing; only a + // pair it actually took to the driver has an out-struct + // worth handing back. + if ((probeResult != VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR) && + (probeResult != VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR)) { + entry.probed[c][p] = true; + entry.snapshot[c][p] = probeSnapshot; + } else if (probeSnapshot.caps.codec == codec) { + // The driver, not the pair check, returned that code: + // QueryEncoderCapsInternal had already stamped the codec + // into the out-struct, so there IS a result to hand back. + entry.probed[c][p] = true; + entry.snapshot[c][p] = probeSnapshot; + } + } + } + + m_devices.push_back(entry); + } + + if (m_devices.empty()) { + // Same code the pre-context ephemeral path returned when no candidate + // matched (VulkanDeviceContext::InitPhysicalDevice). + return VK_ERROR_FEATURE_NOT_PRESENT; + } + return VK_SUCCESS; +} + +VkResult VulkanVideoEncoderContext::Create( + const VkVideoEncoderContextCreateInfo& createInfo, + VkSharedBaseObj& outContext) +{ + std::lock_guard lock(VkEncContextConstructionMutex()); + + // Latched before any output can be produced, including the device- + // selection chatter inside VulkanDeviceContext. Only VK_TRUE acts: see + // the header for why VK_FALSE must not un-silence. + // Taken before any output can be produced, including the device-selection + // chatter inside VulkanDeviceContext, and held for the life of the + // context. VK_FALSE takes no token rather than clearing one: a context + // that wants output does not get to silence-off another owner. + VkEncoderStdioSilenceScope contextSilence( + createInfo.silenceStdio == VK_TRUE); + + if (createInfo.mode == VK_VIDEO_ENCODER_CONTEXT_MODE_OWN) { + if (VkEncOwnContextsRetired()) { + // The instance this process was entitled to has already been + // destroyed. Standing up a second one is the thing the floor + // reference exists to prevent, so refuse rather than try: after a + // sandbox has locked down the create would fail anyway, and here + // it fails with a reason instead of a driver error. + return VK_ERROR_INITIALIZATION_FAILED; + } + std::vector& registry = VkEncOwnContextRegistry(); + for (size_t i = 0; i < registry.size(); i++) { + if (memcmp(registry[i].gpuUUID, createInfo.gpuUUID, + VK_UUID_SIZE) == 0) { + // Rule 2: hand back the existing context rather than issuing + // a second vkCreateInstance. + outContext = registry[i].context; + return VK_SUCCESS; + } + } + } + + VkSharedBaseObj context( + new VulkanVideoEncoderContext()); + // Move the request into the object: from here the context owns it, and it + // ends when the context does rather than when this function returns. A + // failed Build below leaves the token on the local, which releases it. + context->m_stdioSilence = std::move(contextSilence); + const VkResult result = context->Build(createInfo); + if (result != VK_SUCCESS) { + return result; + } + + if (createInfo.mode == VK_VIDEO_ENCODER_CONTEXT_MODE_OWN) { + VkEncOwnContextEntry entry; + memcpy(entry.gpuUUID, createInfo.gpuUUID, VK_UUID_SIZE); + entry.context = context; + // The floor reference. Never removed. + VkEncOwnContextRegistry().push_back(entry); + } + + outContext = context; + return VK_SUCCESS; +} + +// Release the floor reference, destroying every OWN-mode context and with it +// the VkInstance each holds. +// +// CALLED AT A POINT THE CALLER CHOOSES, which is the whole reason this is a +// function and not a destructor. It runs vkDestroyInstance, so it has to +// happen while the process can still call the driver -- before a sandbox +// tightens, and well before static destruction, where the ordering against +// the loader is not defined. +// +// Retirement is permanent and is latched BEFORE the release, so a create that +// races this call is refused rather than served a context that is about to +// die, and a create that follows it cannot re-issue vkCreateInstance. +// +// Idempotent: a second call has nothing to release and says so by returning +// zero. +uint32_t VkEncRetireOwnContexts() +{ + std::lock_guard lock(VkEncContextConstructionMutex()); + + VkEncOwnContextsRetired() = true; + + std::vector& registry = VkEncOwnContextRegistry(); + const uint32_t retired = (uint32_t)registry.size(); + // clear() drops each VkSharedBaseObj reference; the context whose refcount + // reaches zero destroys its VulkanDeviceContext and, in OWN mode, its + // instance with it. + registry.clear(); + return retired; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoderContext( + const VkVideoEncoderContextCreateInfo* pCreateInfo, + VkSharedBaseObj& outContext) { - VkSharedBaseObj impl(new VulkanVideoEncoderExtImpl()); - if (!impl) { + outContext = nullptr; + if (pCreateInfo == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + // Structure-type gate, and the chain rule with it: this struct defines no + // extension structs, so it refuses ANY chain rather than walking past + // something it does not understand. + if ((pCreateInfo->sType != + VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONTEXT_CREATE_INFO) || + (pCreateInfo->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + switch (pCreateInfo->mode) { + case VK_VIDEO_ENCODER_CONTEXT_MODE_OWN: + if ((pCreateInfo->adoptInstance != VK_NULL_HANDLE) || + (pCreateInfo->adoptPhysicalDevice != VK_NULL_HANDLE)) { + // Handles supplied to a mode that will not borrow them: the + // caller asked for one thing and meant another, and silently + // creating a second instance beside handles it already had is + // the expensive half of that mistake. + return VK_ERROR_INITIALIZATION_FAILED; + } + break; + case VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT: + // Both required together; either alone is an error. + if ((pCreateInfo->adoptInstance == VK_NULL_HANDLE) || + (pCreateInfo->adoptPhysicalDevice == VK_NULL_HANDLE)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (!VkEncCtxUuidIsZero(pCreateInfo->gpuUUID)) { + // gpuUUID is an OWN-mode selector. In ADOPT the device is + // already chosen, so a non-zero one is a contradiction, and + // honouring one of the two would be a guess. + return VK_ERROR_INITIALIZATION_FAILED; + } + break; + default: + return VK_ERROR_INITIALIZATION_FAILED; + } + + return VulkanVideoEncoderContext::Create(*pCreateInfo, outContext); +} + +// Defined HERE, below VulkanVideoEncoderContext, and not next to +// CreateVulkanVideoEncoderExt: it dereferences the context, which is an +// incomplete type at that point in this file. +VK_VIDEO_ENCODER_EXPORT +VkResult CreateVulkanVideoEncoderExtOnContext( + const VkSharedBaseObj& context, + uint32_t deviceIndex, + VkSharedBaseObj& vulkanVideoEncoder) +{ + // |vulkanVideoEncoder| is deliberately NOT cleared here. Clearing it would + // make a FAILING create destroy whatever session the caller already held in + // that variable, and it would differ from CreateVulkanVideoEncoderExt, + // which leaves its out-param untouched on failure. Assigned once, on + // success, at the bottom. + if (!context) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + // Resolve both handles now, while the type is complete, and hand them to + // the session as plain values -- the session cannot call these accessors + // from where it needs them. + const VulkanVideoEncoderContext::DeviceEntry* entry = + context->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + VkEncErr() << "[EncoderExt] deviceIndex " << deviceIndex + << " is out of range for this context (" + << context->GetPhysicalDeviceCount() << " device(s))" + << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + const VkInstance instance = context->GetDeviceContext().getInstance(); + if ((instance == VK_NULL_HANDLE) || (entry->physDevice == VK_NULL_HANDLE)) { + // A built context always has both. Checked anyway because everything + // downstream treats them as valid without re-testing, and a null + // instance reaching InitVulkanDevice reads as "create your own". + VkEncErr() << "[EncoderExt] context holds no usable instance or " + "physical device at index " << deviceIndex << std::endl; + return VK_ERROR_INITIALIZATION_FAILED; + } + + VulkanVideoEncoderExtImpl* impl = new VulkanVideoEncoderExtImpl(); + VkSharedBaseObj obj(impl); + if (!obj) { return VK_ERROR_OUT_OF_HOST_MEMORY; } - vulkanVideoEncoder = impl; + + // Bind before publishing, so the session is never observable without its + // context reference. + impl->SetContext(context, instance, entry->physDevice); + + vulkanVideoEncoder = obj; + return VK_SUCCESS; +} + +VK_VIDEO_ENCODER_EXPORT +uint32_t VkEncGetPhysicalDeviceCount(VulkanVideoEncoderContext* ctx) +{ + return (ctx != nullptr) ? ctx->GetPhysicalDeviceCount() : 0u; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult VkEncGetPhysicalDeviceIdentity(VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoEncoderDeviceIdentity* pOut) +{ + // Struct gate first, and before the context is even dereferenced: an + // unstamped or chained out-struct is version skew. + if ((pOut == nullptr) || + (pOut->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_DEVICE_IDENTITY) || + (pOut->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (ctx == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + const VulkanVideoEncoderContext::DeviceEntry* entry = + ctx->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + *pOut = entry->identity; + return VK_SUCCESS; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult VkEncGetEncodeCapabilities(VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkVideoEncoderCapabilities* pOut) +{ + // Same gate, same order, same reason as the free-function enumerators. + if ((pOut == nullptr) || + (pOut->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES) || + (pOut->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (ctx == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + const VulkanVideoEncoderContext::DeviceEntry* entry = + ctx->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + const int32_t codecIndex = VkEncCtxCodecIndex(codec); + if (codecIndex < 0) { + return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + } + // Profile numbers are the standard's own, so the row is found by lookup + // and not by using the number as an index. A number this library does not + // probe for this codec -- including one that names a profile of a + // DIFFERENT codec -- has no row, and says so. + const int32_t profileSlot = VkEncCtxProfileSlotIndex(codec, profile); + if (profileSlot < 0) { + return VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR; + } + if (!entry->probed[codecIndex][profileSlot]) { + // No driver query was issued for this pair, so there is no out-struct + // to hand back. Leave the caller's alone -- exactly what the + // pre-context entry point did when it rejected a pair before probing. + return entry->capsResult[codecIndex][profileSlot]; + } + *pOut = entry->snapshot[codecIndex][profileSlot].caps; + return entry->capsResult[codecIndex][profileSlot]; +} + +// The one place the snapshot row for a (codec, profile) pair is resolved. +// Returns nullptr and leaves |outResult| holding the code the caller must +// propagate when there is no row to read. +static const VkEncProfileCapabilitySnapshot* VkEncCtxSnapshotRow( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkResult& outResult) +{ + outResult = VK_ERROR_INITIALIZATION_FAILED; + if (ctx == nullptr) { + return nullptr; + } + const VulkanVideoEncoderContext::DeviceEntry* entry = + ctx->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + return nullptr; + } + const int32_t codecIndex = VkEncCtxCodecIndex(codec); + if (codecIndex < 0) { + outResult = VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + return nullptr; + } + const int32_t profileSlot = VkEncCtxProfileSlotIndex(codec, profile); + if (profileSlot < 0) { + outResult = VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR; + return nullptr; + } + outResult = entry->capsResult[codecIndex][profileSlot]; + if (!entry->probed[codecIndex][profileSlot]) { + return nullptr; + } + return &entry->snapshot[codecIndex][profileSlot]; +} + +// The two-call write shared by every list this context answers: copy at most +// |capacity| entries, report how many were written, and say VK_INCOMPLETE +// when the caller's buffer could not hold the answer. +// +// A null |pArray| is the counting call and is not an error: *pCount receives +// the full number and nothing is written. +template +static VkResult VkEncCtxWriteList(const T* entries, uint32_t entryCount, + uint32_t* pCount, T* pArray) +{ + if (pArray == nullptr) { + *pCount = entryCount; + return VK_SUCCESS; + } + const uint32_t capacity = *pCount; + const uint32_t written = (capacity < entryCount) ? capacity : entryCount; + for (uint32_t i = 0; i < written; i++) { + pArray[i] = entries[i]; + } + *pCount = written; + return (written < entryCount) ? VK_INCOMPLETE : VK_SUCCESS; +} + +// The device's encode-source format list for |coreProfile| on |physDevice|. +// +// vkGetPhysicalDeviceVideoFormatPropertiesKHR is reached DIRECTLY rather than +// through VulkanVideoCapabilities::GetVideoFormats, and that is not a +// shortcut. GetVideoFormats reads the device context's CURRENT physical +// device; a context holds one entry per device and this query names WHICH, so +// routing through the context's current handle would answer about whichever +// device happened to be adopted last during construction. Re-adopting to fix +// that would mutate the very state the capability snapshot was built against, +// from a query documented as not moving anything. +static VkResult VkEncQueryDeviceEncodeSrcFormats( + const VulkanDeviceContext& devCtx, + VkPhysicalDevice physDevice, + const VkVideoCoreProfile& coreProfile, + VkFormat* outFormats, + uint32_t& ioCount) +{ + const VkVideoProfileListInfoKHR profileList = { + VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR, nullptr, 1, + coreProfile.GetProfile() }; + const VkPhysicalDeviceVideoFormatInfoKHR formatInfo = { + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VIDEO_FORMAT_INFO_KHR, + const_cast(&profileList), + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR }; + + const uint32_t capacity = ioCount; + ioCount = 0; + + uint32_t count = 0; + VkResult result = devCtx.GetPhysicalDeviceVideoFormatPropertiesKHR( + physDevice, &formatInfo, &count, nullptr); + if (result != VK_SUCCESS) { + // A profile this device has no encode support for answers here, with + // the driver's own reason. That is the answer, not an error to hide. + return result; + } + if (count == 0) { + return VK_SUCCESS; + } + if (count > capacity) { + count = capacity; + } + VkVideoFormatPropertiesKHR props[VK_ENC_MAX_DEVICE_INPUT_FORMATS] = {}; + for (uint32_t i = 0; i < count; i++) { + props[i].sType = VK_STRUCTURE_TYPE_VIDEO_FORMAT_PROPERTIES_KHR; + } + result = devCtx.GetPhysicalDeviceVideoFormatPropertiesKHR( + physDevice, &formatInfo, &count, props); + // VK_INCOMPLETE means the clamp above dropped entries, which is a short + // buffer and not a failed query: what was written is still true. + if ((result != VK_SUCCESS) && (result != VK_INCOMPLETE)) { + return result; + } + for (uint32_t i = 0; i < count; i++) { + outFormats[i] = props[i].format; + } + ioCount = count; + return VK_SUCCESS; +} + +// WHAT THIS DEVICE WILL TAKE, ASKED ONCE AND ANSWERED FOR THREE SURFACES. +// +// The enumerator, the point query and InitializeExt all have to give the same +// answer to "will this device encode this input for this profile", and until +// this function existed only the first two shared one. InitializeExt asked +// nothing: it built the config, created the session, and let the DRIVER refuse +// at vkGetPhysicalDeviceVideoCapabilitiesKHR -- past the point where a caller +// could still choose differently, with a message naming neither the format nor +// its subsampling. The advertised set was therefore narrower than the accepted +// set, and a caller had to consult two surfaces to learn what it could hand in. +// +// EXTRACTED RATHER THAN RESTATED, for the reason the resolver below states +// about its own halves: a second statement of this rule is what produced two +// surfaces that disagreed for identical arguments in the first place. +// +// THE VERDICT IS AN ENUM AND NOT A VkResult, and that is the whole point of +// the extraction. Every one of these is VK_ERROR_FORMAT_NOT_SUPPORTED to the +// caller of a query -- which is the right answer for a query, since the +// question was "yes or no". At an INITIALISATION boundary the same yes-or-no +// is a refusal a caller has to act on, and "no" without WHICH of these is what +// the driver already said. +static VkEncDeviceFormatVerdict VkEncResolveDeviceEncodeFormat( + const VulkanDeviceContext& devCtx, + VkPhysicalDevice physDevice, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t codecProfile, + uint32_t chromaSubsampling, + uint32_t bitDepth, + VkFormat inputFormat, + bool viaFilter, + VkFormat& outEncodeFormat) +{ + // Written only on ACCEPTED, and never partially, for the same reason the + // resolver states about its own out-parameter: a half-answer a caller + // cannot tell from a whole one. + outEncodeFormat = VK_FORMAT_UNDEFINED; + + const VkVideoComponentBitDepthFlagBitsKHR depthFlag = + GetComponentBitDepthFlagBits(bitDepth); + if (depthFlag == VK_VIDEO_COMPONENT_BIT_DEPTH_INVALID_KHR) { + return VK_ENC_DEVICE_FORMAT_DEPTH_NOT_ENCODABLE; + } + + // A LIVE QUERY, AT THE INPUT'S OWN GEOMETRY. Not the context's capability + // snapshot: that probes every profile at 4:2:0 (MapProbeProfile), so + // reading it here would answer "no" for every 4:4:4 input on every device. + // The profile asked about is the one the binder ACTUALLY derived or bound, + // and the subsampling and depth are the input's own. + VkVideoCoreProfile coreProfile( + codec, (VkVideoChromaSubsamplingFlagBitsKHR)chromaSubsampling, + depthFlag, depthFlag, codecProfile); + + VkFormat deviceFormats[VK_ENC_MAX_DEVICE_INPUT_FORMATS] = {}; + uint32_t deviceFormatCount = VK_ENC_MAX_DEVICE_INPUT_FORMATS; + if (VkEncQueryDeviceEncodeSrcFormats(devCtx, physDevice, coreProfile, + deviceFormats, + deviceFormatCount) != VK_SUCCESS) { + return VK_ENC_DEVICE_FORMAT_PROFILE_ABSENT; + } + if (deviceFormatCount == 0) { + // The query succeeded and named nothing, which is the same fact as a + // failed query and is worth reporting as the same reason: this device + // has no encode source for that profile. + return VK_ENC_DEVICE_FORMAT_PROFILE_ABSENT; + } + + // WHAT THE ENCODER IS HANDED is what the device has to accept, and for a + // converted input that is the conversion's OUTPUT and not the caller's + // format. Asking the device about the caller's format would refuse every + // RGBA and packed-Y'CbCr input on a device that encodes them perfectly + // well through the filter. + const VkFormat encodeFormat = + viaFilter ? VkEncConversionTargetFormat(inputFormat, deviceFormats, + deviceFormatCount) + : inputFormat; + if (encodeFormat == VK_FORMAT_UNDEFINED) { + return VK_ENC_DEVICE_FORMAT_NO_CONVERSION_TARGET; + } + if (!VkEncFormatListContains(deviceFormats, deviceFormatCount, + encodeFormat)) { + return VK_ENC_DEVICE_FORMAT_NOT_AN_ENCODE_SOURCE; + } + + outEncodeFormat = encodeFormat; + return VK_ENC_DEVICE_FORMAT_ACCEPTED; +} + +// THE ONE ANSWER BOTH THE POINT QUERY AND THE ENUMERATOR GIVE. +// +// Extracted rather than restated. Two statements of this rule is exactly what +// made an advertised encodeFormat and a queried one disagree for identical +// arguments on the same device in the same process: the enumerator read a +// capability snapshot probed at a fixed 4:2:0 envelope, the point query bound +// the caller's own configuration and asked the device at the profile that +// binding derived. Only one of those describes the session a caller would get. +// +// Writes |outProps| only on VK_SUCCESS, and never partially: a caller reading +// the struct after a refusal would be reading a half-answer it cannot tell from +// a whole one. +// +// Defined ABOVE the enumerator on purpose -- the enumerator is now one of its +// two callers. +static VkResult VkEncResolveInputFormatSupport( + VulkanVideoEncoderContext* ctx, + const VulkanVideoEncoderContext::DeviceEntry* entry, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkFormat format, + VkVideoEncoderColorModel colorModel, + VkVideoEncoderInputFormatProperties* outProps) +{ + // ---- The library half, answered BY the binder rather than beside it. ---- + // + // Running VkEncBuildAndProbeConfig is what makes this query and + // InitializeExt one answer instead of two. Every rule that decides + // acceptance -- the input taxonomy, the colour-model declaration, and the + // profile's own bit-depth and chroma-subsampling limits -- is applied + // there, once. A reimplementation here would be a second statement of the + // same rules, and what that eventually produces is a query promising what + // init refuses. + VkVideoEncoderConfig probeConfig = {}; + probeConfig.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + probeConfig.codec = codec; + probeConfig.profile = profile; + probeConfig.inputFormat = format; + probeConfig.inputColorModel = colorModel; + // Geometry the binder needs to produce a config at all, and that no part + // of this answer depends on: a format's routing and its encode profile are + // both independent of the frame size. It is stated here rather than taken + // from the caller for exactly that reason -- an extent parameter on this + // entry point would be a knob with no effect on what it returns. + probeConfig.encodeWidth = 1920; + probeConfig.encodeHeight = 1080; + probeConfig.inputWidth = 1920; + probeConfig.inputHeight = 1080; + probeConfig.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + probeConfig.averageBitrate = 4000000; + probeConfig.frameRateNum = 30; + probeConfig.frameRateDen = 1; + probeConfig.gopLength = 30; + // No file is opened for a query: the binder opens one only when an + // outputPath is set AND file output is wanted, and neither is, here. + probeConfig.disableFileOutput = VK_TRUE; + + VkEncBoundConfigProbe probe = {}; + if (VkEncBuildAndProbeConfig(probeConfig, codec, &probe) != VK_SUCCESS) { + // The binder refused: either the pair is not an input this library + // routes, or the named profile cannot carry it, or it is a profile + // number this library does not bind. All three are "you cannot feed me + // this on this profile", which is the question that was asked. + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + // ---- The device half, and it is the SAME CALL InitializeExt makes. ---- + // + // VkEncResolveDeviceEncodeFormat above holds it, so a query that says yes + // and a session that refuses cannot be written without changing one + // function. The profile handed to it is the one the binder ACTUALLY + // derived, and the subsampling and depth are the ENCODE side's -- the + // geometry the video profile is built from at session creation, which is + // what makes this query predict that session rather than a different one. + // Both are read back off the probe rather than assumed. + // + // THE REASON IS DISCARDED HERE, deliberately. A point query answers yes or + // no; the reason is what an initialisation boundary owes its caller, and + // that is where it is spent. + const bool viaFilter = (probe.preprocessComputeFilter != 0); + VkFormat encodeFormat = VK_FORMAT_UNDEFINED; + if (VkEncResolveDeviceEncodeFormat( + ctx->GetDeviceContext(), entry->physDevice, codec, + probe.codecProfile, probe.encodeChromaSubsampling, + probe.encodeBitDepthLuma, + format, viaFilter, encodeFormat) != + VK_ENC_DEVICE_FORMAT_ACCEPTED) { + return VK_ERROR_FORMAT_NOT_SUPPORTED; + } + + if (outProps != nullptr) { + outProps->format = format; + outProps->encodeFormat = encodeFormat; + outProps->optimality = + viaFilter ? VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL + : VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL; + } return VK_SUCCESS; } + +// The enumerator's admission: one live resolve per candidate. |userData| is a +// VkEncLiveAdmitContext naming the device and the key being enumerated. +struct VkEncLiveAdmitContext { + VulkanVideoEncoderContext* ctx; + const VulkanVideoEncoderContext::DeviceEntry* entry; + VkVideoCodecOperationFlagBitsKHR codec; + uint32_t profile; +}; + +static bool VkEncAdmitByLiveResolve( + void* userData, VkFormat candidate, + VkVideoEncoderInputFormatProperties* outEntry) +{ + VkEncLiveAdmitContext* const live = + static_cast(userData); + return VkEncResolveInputFormatSupport( + live->ctx, live->entry, live->codec, live->profile, candidate, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, outEntry) == + VK_SUCCESS; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult VkEncEnumerateStdFlags(VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t* pCount, + VkVideoEncoderStdFlags* pFlags) +{ + if (pCount == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + VkResult rowResult = VK_ERROR_INITIALIZATION_FAILED; + const VkEncProfileCapabilitySnapshot* row = + VkEncCtxSnapshotRow(ctx, deviceIndex, codec, profile, rowResult); + if (row == nullptr) { + // No row to read. The count is still answered, because a caller that + // treats a refused profile as "advertise nothing" needs a zero rather + // than a stale number. + *pCount = 0; + return rowResult; + } + const VkResult writeResult = + VkEncCtxWriteList(row->stdFlags, row->stdFlagCount, pCount, pFlags); + // A short buffer outranks the probe's own result: the caller must know it + // did not receive everything before it acts on what it did receive. + const VkResult result = + (writeResult == VK_SUCCESS) ? rowResult : writeResult; + if ((result != VK_SUCCESS) && (result != VK_INCOMPLETE)) { + // An entry can be probed and still carry a refusal. A list read out of + // one is not an answer, so it is reported as the count every other + // error reports. + *pCount = 0; + } + return result; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult VkEncEnumerateInputFormats( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + uint32_t* pCount, + VkVideoEncoderInputFormatProperties* pFormats) +{ + if (pCount == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + // THE SNAPSHOT IS THE KEY GATE AND NOTHING ELSE. It is what returns + // VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR for a profile number + // this context does not carry, and what carries the probe's own capsResult + // for a pair the driver refused. It is NOT where the list comes from: the + // probe keys every profile at a fixed 4:2:0 envelope, and a list built from + // that describes a session no caller asked for. + VkResult rowResult = VK_ERROR_INITIALIZATION_FAILED; + const VkEncProfileCapabilitySnapshot* row = + VkEncCtxSnapshotRow(ctx, deviceIndex, codec, profile, rowResult); + if (row == nullptr) { + *pCount = 0; + return rowResult; + } + const VulkanVideoEncoderContext::DeviceEntry* const entry = + ctx->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + // Unreachable while the snapshot row exists -- the row was resolved + // through the same entry -- and stated rather than assumed, because + // what follows dereferences it. + *pCount = 0; + return VK_ERROR_INITIALIZATION_FAILED; + } + + // BUILT LIVE, per candidate, through the SAME resolver the point query + // answers from. The cost is one binder run and one device format query per + // routable candidate, which is what the point query already pays per call; + // what it buys is that the two surfaces cannot disagree, because there is + // only one of them. + VkEncLiveAdmitContext live = { ctx, entry, codec, profile }; + VkVideoEncoderInputFormatProperties entries[ + VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + + // A REFUSED CANDIDATE IS THE ANSWER HERE, NOT AN ERROR. The binder explains + // every refusal on the gated error stream, which is right for a caller that + // declared one configuration and wrong for a sweep that deliberately offers + // every routable format to a profile most of them cannot reach. An + // enumeration that printed twenty refusals per successful call would be + // unusable in the sandboxed process the latch exists for, and would say + // nothing a caller could act on. + // + // Saved and restored rather than set: a caller that had already silenced + // the streams stays silenced, and one that had not is unaffected the moment + // this returns. + // A scoped request rather than a read/save/restore. The restore could + // write back a value another thread had changed while the enumeration + // ran, un-silencing an owner that still needed silence. A token adds this + // query's request and removes exactly that one. + uint32_t entryCount = 0; + { + const VkEncoderStdioSilenceScope quietQuery(true); + entryCount = VkEncAdvertiseInputFormats( + &VkEncAdmitByLiveResolve, &live, entries, + VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + } + + const VkResult writeResult = + VkEncCtxWriteList(entries, entryCount, pCount, pFormats); + const VkResult result = + (writeResult == VK_SUCCESS) ? rowResult : writeResult; + if ((result != VK_SUCCESS) && (result != VK_INCOMPLETE)) { + *pCount = 0; + } + return result; +} + +VK_VIDEO_ENCODER_EXPORT +VkResult VkEncEnumerateDrmModifiers(VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkFormat format, + VkImageUsageFlags usage, + uint32_t* pCount, + uint64_t* pModifiers) +{ + // A null pCount is the one failure with nowhere to put the count. Every + // other return below leaves one behind. + if (pCount == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (ctx == nullptr) { + *pCount = 0; + return VK_ERROR_INITIALIZATION_FAILED; + } + const VulkanVideoEncoderContext::DeviceEntry* entry = + ctx->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + *pCount = 0; + return VK_ERROR_INITIALIZATION_FAILED; + } + + VkFormatFeatureFlags2 requiredFeatures = 0; + if (!VkEncCtxUsageToFormatFeatures(usage, requiredFeatures)) { + *pCount = 0; + return VK_ERROR_INITIALIZATION_FAILED; + } + + if (!entry->hasDrmFormatModifierExtension) { + // No modifiers is a true answer on a device without the extension + // (every Windows device, for one), not a failure. + *pCount = 0; + return VK_SUCCESS; + } + if (!entry->hasFormatFeatureFlags2) { + // The 32-bit modifier list cannot express VIDEO_ENCODE_INPUT, the one + // feature this entry point exists to filter on, so an answer built + // from it would be wrong rather than partial. + *pCount = 0; + return VK_ERROR_EXTENSION_NOT_PRESENT; + } + + const VulkanDeviceContext& devCtx = ctx->GetDeviceContext(); + + VkDrmFormatModifierPropertiesList2EXT modifierList = {}; + modifierList.sType = + VK_STRUCTURE_TYPE_DRM_FORMAT_MODIFIER_PROPERTIES_LIST_2_EXT; + VkFormatProperties2 formatProps = {}; + formatProps.sType = VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_2; + formatProps.pNext = &modifierList; + + devCtx.GetPhysicalDeviceFormatProperties2(entry->physDevice, format, + &formatProps); + if (modifierList.drmFormatModifierCount == 0) { + *pCount = 0; + return VK_SUCCESS; + } + std::vector modifierProps( + modifierList.drmFormatModifierCount); + modifierList.pDrmFormatModifierProperties = modifierProps.data(); + devCtx.GetPhysicalDeviceFormatProperties2(entry->physDevice, format, + &formatProps); + if (modifierList.drmFormatModifierCount < modifierProps.size()) { + modifierProps.resize(modifierList.drmFormatModifierCount); + } + + const uint32_t capacity = (pModifiers != nullptr) ? *pCount : 0u; + uint32_t matched = 0; + uint32_t written = 0; + for (size_t i = 0; i < modifierProps.size(); i++) { + if ((modifierProps[i].drmFormatModifierTilingFeatures & + requiredFeatures) != requiredFeatures) { + continue; + } + matched++; + if ((pModifiers != nullptr) && (written < capacity)) { + pModifiers[written++] = modifierProps[i].drmFormatModifier; + } + } + + if (pModifiers == nullptr) { + *pCount = matched; + return VK_SUCCESS; + } + *pCount = written; + return (written < matched) ? VK_INCOMPLETE : VK_SUCCESS; +} + + +VK_VIDEO_ENCODER_EXPORT +VkResult VkEncQueryInputFormatSupport( + VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkFormat format, + VkVideoEncoderColorModel colorModel, + VkVideoEncoderInputFormatProperties* pProperties) +{ + // Same gate, same order, same reason as the enumerators above -- except + // that pProperties is OPTIONAL here, because a caller that wants only the + // verdict should not have to supply somewhere to put an answer it will not + // read. + if (ctx == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + const VulkanVideoEncoderContext::DeviceEntry* entry = + ctx->GetDeviceEntry(deviceIndex); + if (entry == nullptr) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (VkEncCtxCodecIndex(codec) < 0) { + return VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR; + } + + // Everything below the argument gate is the shared resolver, which the + // enumerator next door runs over every routable candidate. This entry point + // is that resolver applied to one. + return VkEncResolveInputFormatSupport(ctx, entry, codec, profile, format, + colorModel, pProperties); +} + +//============================================================================= +// The four pre-context capability entry points are thin wrappers over a +// context they build and throw away, so that a caller holding no Vulkan +// handles reaches the same code as one that supplies its own. +// +// They are the reason the context is on the executed path from day one: the +// Chromium enumerator calls the ephemeral per-profile variant once per +// candidate profile, and every one of those calls now goes through +// CreateVulkanVideoEncoderContext. +//============================================================================= + +VK_VIDEO_ENCODER_EXPORT +VkResult EnumerateVulkanVideoEncoderProfileCapabilities( + VkInstance instance, + VkPhysicalDevice physicalDevice, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkVideoEncoderCapabilities* outCaps) +{ + // Struct gate first, before any Vulkan work: an unstamped or chained + // outCaps is version skew, and must be refused before the caller's + // handles are touched. + if ((outCaps == nullptr) || + (outCaps->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES) || + (outCaps->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + if (instance == VK_NULL_HANDLE || physicalDevice == VK_NULL_HANDLE) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + // An ADOPT-mode context over the caller's instance and physical + // device. It creates nothing and destroys nothing of the caller's, and it + // adopts rather than re-selects -- D8's repair, now living in the context + // instead of beside it. + VkVideoEncoderContextCreateInfo createInfo = {}; + createInfo.mode = VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT; + createInfo.adoptInstance = instance; + createInfo.adoptPhysicalDevice = physicalDevice; + + VkSharedBaseObj context; + const VkResult result = CreateVulkanVideoEncoderContext(&createInfo, + context); + if (result != VK_SUCCESS) { + return result; + } + // ADOPT enumerates exactly the one device it was handed. + return VkEncGetEncodeCapabilities(context.get(), /*deviceIndex*/ 0, + codec, profile, outCaps); +} + +VK_VIDEO_ENCODER_EXPORT +VkResult EnumerateVulkanVideoEncoderProfileCapabilitiesEphemeral( + int32_t deviceId, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkVideoEncoderCapabilities* outCaps) +{ + // Struct gate first -- same rule and same reason as the caller-handles + // variant above, and it still runs before any context is built. + if ((outCaps == nullptr) || + (outCaps->sType != VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES) || + (outCaps->pNext != nullptr)) { + return VK_ERROR_INITIALIZATION_FAILED; + } + + // "Ephemeral" now names the API contract, not the implementation: the + // caller still supplies no handles and still gets none back. What is no + // longer ephemeral is the VkInstance -- an OWN-mode context is built once + // and floor-referenced, so the five calls the Chromium enumerator makes + // per cache computation cost one loader load and one vkCreateInstance in + // total instead of five of each. That is D1, and it is the whole reason + // the context exists. + VkVideoEncoderContextCreateInfo createInfo = {}; + createInfo.mode = VK_VIDEO_ENCODER_CONTEXT_MODE_OWN; + // All-zero gpuUUID: enumerate everything, and select below. + + VkSharedBaseObj context; + const VkResult result = CreateVulkanVideoEncoderContext(&createInfo, + context); + if (result != VK_SUCCESS) { + return result; + } + + // The selection predicate InitPhysicalDevice() applied when this entry + // point drove it directly, term for term: the deviceID filter, the + // required video extensions, and an encode-capable queue family carrying + // the requested codec. It is reproduced here rather than pushed into the + // context because it is THIS entry point's contract -- a context + // enumerates, it does not choose. + const uint32_t deviceCount = VkEncGetPhysicalDeviceCount(context.get()); + for (uint32_t index = 0; index < deviceCount; index++) { + const VulkanVideoEncoderContext::DeviceEntry* entry = + context->GetDeviceEntry(index); + if ((deviceId != -1) && + (entry->identity.deviceID != (uint32_t)deviceId)) { + continue; + } + if (!entry->hasRequiredVideoExtensions) { + continue; + } + if ((entry->encodeCodecOps & codec) == 0) { + continue; + } + return VkEncGetEncodeCapabilities(context.get(), index, codec, profile, + outCaps); + } + + // The code InitPhysicalDevice() returned when no candidate matched. + return VK_ERROR_FEATURE_NOT_PRESENT; +} diff --git a/vk_video_encoder/src/vulkan_video_encoder_os_event_linux.cpp b/vk_video_encoder/src/vulkan_video_encoder_os_event_linux.cpp new file mode 100644 index 00000000..e66e4324 --- /dev/null +++ b/vk_video_encoder/src/vulkan_video_encoder_os_event_linux.cpp @@ -0,0 +1,79 @@ +/* + * Linux implementation of the completion-handle seam declared in + * vulkan_video_encoder_os_event_linux.h. + * + * This translation unit is platform-specific by construction. Another + * platform is served by another translation unit implementing the same + * declarations, selected by the build; it is never served by an #else arm in + * this one. + */ +#include "vulkan_video_encoder_os_event_linux.h" + +#if !defined(__linux__) +#error "vulkan_video_encoder_os_event_linux.cpp implements the completion-handle seam for Linux only. Port the seam in a sibling translation unit and select it in the build." +#endif + +#include +#include +#include + +namespace vkenc { + +uint64_t OsCompletionEventCreate() +{ + // NONBLOCK so a caller polling on a sequence that forbids blocking cannot + // block there; CLOEXEC so the handle does not leak into a child process, + // which for a browser is a sandbox concern rather than hygiene. + const int fd = ::eventfd(0, EFD_CLOEXEC | EFD_NONBLOCK); + if (fd < 0) { + return kOsCompletionEventNone; + } + // Any legal fd -- INCLUDING 0 -- stays distinct from the sentinel: + // eventfd() never returns a negative fd on success. + return (uint64_t)(int64_t)fd; +} + +void OsCompletionEventSignal(uint64_t handle) +{ + if (handle == kOsCompletionEventNone) { + return; + } + const uint64_t one = 1; + // eventfd ACCUMULATES rather than latching, so an edge raised while the + // reader is busy is added rather than lost -- which is what makes + // coalescing safe here instead of lossy. EAGAIN is only reachable at the + // 64-bit ceiling, where the reader is so far behind that the completion + // counter reconciliation is the thing that will recover it. + const ssize_t written = ::write((int)handle, &one, sizeof(one)); + (void)written; +} + +uint64_t OsCompletionEventDuplicate(uint64_t handle) +{ + if (handle == kOsCompletionEventNone) { + return kOsCompletionEventNone; + } + // F_DUPFD_CLOEXEC rather than dup(), so the duplicate carries close-on-exec + // from the moment it exists. dup() followed by an fcntl() to set the flag + // leaves a window in which a concurrent fork/exec inherits the descriptor, + // which for a browser is the sandbox concern Create() already guards. + // + // The lowest free descriptor is fine: 0 is a legal result and stays + // distinct from the sentinel, which is all-ones. + const int duplicate = ::fcntl((int)handle, F_DUPFD_CLOEXEC, 0); + if (duplicate < 0) { + // The original is untouched. A caller that cannot export still has a + // working encoder. + return kOsCompletionEventNone; + } + return (uint64_t)(int64_t)duplicate; +} + +void OsCompletionEventDestroy(uint64_t handle) +{ + if (handle != kOsCompletionEventNone) { + ::close((int)handle); + } +} + +} // namespace vkenc diff --git a/vk_video_encoder/src/vulkan_video_encoder_os_event_linux.h b/vk_video_encoder/src/vulkan_video_encoder_os_event_linux.h new file mode 100644 index 00000000..8f6104c8 --- /dev/null +++ b/vk_video_encoder/src/vulkan_video_encoder_os_event_linux.h @@ -0,0 +1,56 @@ +/* + * Optional OS-handle completion currency. + * + * The ext layer's default path carries no OS-specific code -- no + * , no , no poll/epoll. The currency is declared + * here in platform-neutral terms and the ext layer talks to it through these + * three functions and nothing else. + * + * These declarations are portable; the implementations are not. Exactly one + * implementation translation unit is compiled per platform and the build + * selects it; the Linux one is vulkan_video_encoder_os_event_linux.cpp. A + * platform with no implementation stops the build at configure time. + * + * Create can also fail on a platform that has an implementation -- the + * process can be out of handles -- and returns kOsCompletionEventNone when it + * does, which the caller reports as ERROR_HANDLE_TYPE_UNSUPPORTED. + */ +#ifndef VULKAN_VIDEO_ENCODER_OS_EVENT_LINUX_H_ +#define VULKAN_VIDEO_ENCODER_OS_EVENT_LINUX_H_ + +#include + +namespace vkenc { + +// The "no handle" value. All-ones rather than 0, because 0 is a legal +// handle: a process that closed stdin can be handed exactly that. A +// sentinel that collided with a valid handle would make a live completion +// event indistinguishable from a refusal. +constexpr uint64_t kOsCompletionEventNone = ~0ull; + +// A waitable OS handle signalled once per completion, or +// kOsCompletionEventNone if one could not be created. Non-blocking and +// close-on-exec where the platform expresses those. +uint64_t OsCompletionEventCreate(); + +// Raise the edge. Called from the capture path, so it must not allocate, must +// not lock, and must be safe to call after Destroy on another thread has been +// ordered against it by the caller. +void OsCompletionEventSignal(uint64_t handle); + +// A NEW handle onto the same completion object, carrying its own close +// obligation. The library keeps its original; a caller that exports one owns +// the duplicate and closes it. Returns kOsCompletionEventNone if the platform +// refuses, and MUST leave the original open when it does -- a failed export is +// not a reason to take the library's own event away from it. +// +// Duplicates share the underlying object. They are independent close +// obligations, not independent broadcast queues: a completion drained through +// one is not redelivered to the other. +uint64_t OsCompletionEventDuplicate(uint64_t handle); + +void OsCompletionEventDestroy(uint64_t handle); + +} // namespace vkenc + +#endif // VULKAN_VIDEO_ENCODER_OS_EVENT_LINUX_H_ diff --git a/vk_video_encoder/src/vulkan_video_encoder_os_event_windows.cpp b/vk_video_encoder/src/vulkan_video_encoder_os_event_windows.cpp new file mode 100644 index 00000000..8788d05d --- /dev/null +++ b/vk_video_encoder/src/vulkan_video_encoder_os_event_windows.cpp @@ -0,0 +1,83 @@ +/* + * Copyright 2025 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "vulkan_video_encoder_os_event_linux.h" + +#if !defined(_WIN32) +#error "vulkan_video_encoder_os_event_windows.cpp implements the completion-handle seam for Windows only." +#endif + +#include + +namespace vkenc { + +uint64_t OsCompletionEventCreate() +{ + // AUTO-RESET, matching the eventfd the Linux seam creates with an initial + // count of zero: a wait that arrives first blocks, and a manual-reset event + // would leave the handle signalled after the first completion and report + // every later wait as already-complete. + // + // The two platforms do NOT agree on what an unread signal accumulates to. + // An auto-reset event is a latch: signalling one that is already signalled + // coalesces, so N completions raised before any wait release ONE waiter. + // A Linux eventfd accumulates a counter and a single read drains all of it. + // Neither is one-signal-per-one-wait. The supported use on both is the + // same: wake, drain the completions that are available, reconcile against + // the monotonic completion count, then wait again. + const HANDLE handle = ::CreateEventW(nullptr, /*bManualReset=*/FALSE, + /*bInitialState=*/FALSE, nullptr); + if (handle == nullptr) { + return kOsCompletionEventNone; + } + return static_cast(reinterpret_cast(handle)); +} + +void OsCompletionEventSignal(uint64_t handle) +{ + if (handle == kOsCompletionEventNone) { + return; + } + ::SetEvent(reinterpret_cast(static_cast(handle))); +} + +uint64_t OsCompletionEventDuplicate(uint64_t handle) +{ + if (handle == kOsCompletionEventNone) { + return kOsCompletionEventNone; + } + HANDLE duplicate = nullptr; + // Non-inheritable, for the reason CreateEventW's unnamed handle is: the + // duplicate must not cross into a child process. DUPLICATE_SAME_ACCESS + // keeps the wait/signal rights the original has. + if (!::DuplicateHandle(::GetCurrentProcess(), + reinterpret_cast(static_cast(handle)), + ::GetCurrentProcess(), &duplicate, 0, + /*bInheritHandle=*/FALSE, DUPLICATE_SAME_ACCESS)) { + // The original is untouched. + return kOsCompletionEventNone; + } + return static_cast(reinterpret_cast(duplicate)); +} + +void OsCompletionEventDestroy(uint64_t handle) +{ + if (handle != kOsCompletionEventNone) { + ::CloseHandle(reinterpret_cast(static_cast(handle))); + } +} + +} // namespace vkenc diff --git a/vk_video_encoder/test/av1_encoder_quality_test.py b/vk_video_encoder/test/av1_encoder_quality_test.py index 7509d169..bbe180c5 100644 --- a/vk_video_encoder/test/av1_encoder_quality_test.py +++ b/vk_video_encoder/test/av1_encoder_quality_test.py @@ -41,6 +41,27 @@ ENCODER_REL_PATH = "build/vk_video_encoder/test/vulkan-video-enc-test" +# The gate. A GOP whose per-frame PSNR spans more than this has collapsed +# somewhere inside it, which is the defect this script was written to find. +PSNR_SPREAD_LIMIT_DB = 10.0 + +# What the encoder prints when it cannot stand up a device at all. Matched +# as a substring of its stderr, so the VkResult that follows may vary. +# The encoder's own exit status. 69 (VVS_EXIT_UNSUPPORTED) is reserved for a +# VkResult that says the device cannot do this; every other nonzero status is +# an ordinary failure. Classify on this, never on the stderr text -- the +# encoder prints "Error creating the encoder instance" before all of them. +ENCODER_EXIT_UNSUPPORTED = 69 + +# Lines the encoder prints only for a failure it has already judged fatal. +# Used to catch a zero exit that contradicts the process's own output. +ENCODER_FATAL_MARKERS = ( + "Error creating the encoder instance:", + "Error encoding frame:", + "Error obtaining the encoded bitstream file:", + "Error: encoder creation reported success but produced no", +) + CODECS = { "h264": {"flag": "h264", "ext": "264", "label": "H.264"}, "h265": {"flag": "h265", "ext": "265", "label": "H.265"}, @@ -72,12 +93,44 @@ class EncodeResult: encode_ok: bool = True decode_ok: bool = True error_msg: str = "" + # The encoder process's exit status, and how many frames actually came + # back out of the decoder. Both are needed to tell a skip from a failure + # and a complete stream from a readable prefix. + exit_code: Optional[int] = None + frames_decoded: Optional[int] = None + + +def is_unsupported_run(results): + """True when EVERY row failed with the encoder's unsupported status. + + This is the whole skip condition. One row that got far enough to be judged + means the host has a device, so the run is a result and not a skip -- a + device that encodes BADLY must still be able to fail this script. + """ + return bool(results) and all( + not r.encode_ok and r.exit_code == ENCODER_EXIT_UNSUPPORTED + for r in results) + + +def short_rows(results, requested_frames): + """Rows that encoded and decoded but returned fewer frames than asked. + + A row whose frame count is unknown is short: the count is the evidence, + and its absence is not evidence of completeness. + """ + short = [] + for r in results: + if not r.encode_ok or not r.decode_ok: + continue + if r.frames_decoded is None or r.frames_decoded < requested_frames: + short.append(r) + return short def run_cmd(cmd, description="", timeout=120, remote_host=None): """Run a command, return (returncode, stdout, stderr). - If remote_host is set (e.g. "user@192.168.122.216" or "192.168.122.216"), + If remote_host is set (e.g. "user@host" or a bare host name), the command is wrapped with ssh and executed on the remote host. Paths are assumed to be valid on the remote (typically via NFS-shared mounts). """ @@ -145,10 +198,17 @@ def encode(yuv_path, out_path, codec, width, height, num_frames, gop, qp, rc, out, err = run_cmd(cmd, f"Encoding {CODECS[codec]['label']} GOP={gop}", remote_host=remote_host) if rc != 0: - return False, f"Encoder returned {rc}: {err[-500:]}" + return False, f"Encoder returned {rc}: {err[-500:]}", rc + # A ZERO EXIT THAT PRINTED A FATAL DIAGNOSTIC IS STILL A FAILURE. The + # encoder is expected to propagate these into its status; trusting the + # status alone means a regression in that propagation arrives here as a + # quality pass over a truncated stream, which is what this guards. + for marker in ENCODER_FATAL_MARKERS: + if marker in err: + return False, f"Encoder exited 0 after reporting: {marker}", rc if not os.path.exists(out_path) or os.path.getsize(out_path) == 0: - return False, "Output file missing or empty" - return True, "" + return False, "Output file missing or empty", rc + return True, "", rc def decode_to_yuv(encoded_path, decoded_yuv_path, codec, bpp=8): @@ -572,8 +632,8 @@ def main(): %(prog)s --samples-root /path/to/vulkan-video-samples # Remote run on GPU VM (NFS-shared paths) - %(prog)s --samples-root /data/.../vulkan-video-samples \\ - --remote-host tzlatinski@192.168.122.216 + %(prog)s --samples-root /path/to/vulkan-video-samples \\ + --remote-host user@host.example """) parser.add_argument("--width", type=int, default=DEFAULT_WIDTH) parser.add_argument("--height", type=int, default=DEFAULT_HEIGHT) @@ -597,7 +657,7 @@ def main(): "(overrides --samples-root derivation).") parser.add_argument("--remote-host", type=str, default=None, help="Run encoder on remote host via ssh " - "(e.g. user@192.168.122.216 or 192.168.122.216). " + "(e.g. user@host.example). " "Default: run locally. Encoder paths must be " "reachable on the remote (NFS-shared).") args = parser.parse_args() @@ -608,15 +668,20 @@ def main(): parser.error("either --encoder or --samples-root must be specified") encoder_bin = os.path.join(args.samples_root, ENCODER_REL_PATH) + # A missing prerequisite is a SKIP, not a usage error and not a failure. + # Exit 77 is what this tree's gpu-labelled tests use for "this host cannot + # run me", declared as SKIP_RETURN_CODE where the test is registered. A + # host without ffmpeg, or without the encoder built, has not disproved + # anything, so reporting red here would be a false negative. if not args.remote_host and not os.path.isfile(encoder_bin): - parser.error(f"encoder binary not found: {encoder_bin}\n" - f" Pass --encoder PATH or --samples-root DIR (must contain " - f"{ENCODER_REL_PATH}). Use --remote-host to skip the local " - f"existence check.") + print(f"SKIP: encoder binary not found: {encoder_bin}", file=sys.stderr) + return 77 for tool in ("ffmpeg", "ffprobe"): if shutil.which(tool) is None: - parser.error(f"{tool} not found on PATH (required for encode/decode/PSNR)") + print(f"SKIP: {tool} not found on PATH " + f"(required for encode/decode/PSNR)", file=sys.stderr) + return 77 os.makedirs(args.output_dir, exist_ok=True) if args.report is None: @@ -667,10 +732,11 @@ def main(): result = EncodeResult(codec=codec_key, gop=gop, file_size=0) # Encode - ok, err = encode(yuv_path, encoded_path, codec_key, + ok, err, enc_rc = encode(yuv_path, encoded_path, codec_key, args.width, args.height, args.num_frames, gop, args.qp, encoder_bin=encoder_bin, remote_host=args.remote_host, bpp=args.bpp) + result.exit_code = enc_rc if not ok: result.encode_ok = False result.error_msg = err @@ -701,6 +767,7 @@ def main(): args.width, args.height, psnr_log, bpp=args.bpp) per_frame_data[(codec_key, gop)] = frames_psnr + result.frames_decoded = len(frames_psnr) # Per-frame sizes if codec_key == "av1": @@ -737,23 +804,69 @@ def main(): print() # Flag AV1 issues + # WHAT THIS GATES. A detected quality collapse, and an encode that did + # not produce a comparable stream. Both must turn this script RED: + # printing them and returning 0 with everything else would mean the + # collapse this script exists to find cannot fail it. + failed = False + + # NO ENCODE-CAPABLE DEVICE IS A SKIP, NOT A FAILURE, and this is the only + # test in the gpu suite written in Python -- every C++ sibling reports the + # same condition by returning 77 (ctest SKIP_RETURN_CODE). Returning 1 + # instead makes a GPU-less runner, which is what CI uses, look like a + # quality regression. + # + # THE SIGNAL IS THE EXIT STATUS, NOT THE STDERR TEXT. The encoder exits 69 + # only when the VkResult says the device cannot do this, and EXIT_FAILURE + # for out of memory, initialization, device loss and unknown alike. This + # used to match "Error creating the encoder instance" instead, which the + # encoder prints before every one of those -- so each of them skipped. + if is_unsupported_run(results): + print(f"SKIP: no encode-capable device " + f"({results[0].error_msg.strip()})") + return 77 + + if not results: + print("FAIL: no encode produced a result") + failed = True + for r in results: + if not r.encode_ok: + print(f"FAIL: encode did not complete: codec={r.codec} GOP={r.gop}" + f" (exit {r.exit_code}): {r.error_msg}") + failed = True + + # A SHORT STREAM IS NOT A PASS. An encode that stops early still leaves a + # file the decoder reads happily, and PSNR over the frames that survived + # says nothing about the ones that did not. The requested count is the + # only number that makes the comparison below mean what it claims. + for r in short_rows(results, args.num_frames): + got = "no" if r.frames_decoded is None else str(r.frames_decoded) + print(f"FAIL: codec={r.codec} GOP={r.gop} decoded {got} frame(s) of " + f"{args.num_frames} requested") + failed = True + av1_results = [r for r in results if r.codec == "av1" and r.encode_ok] if av1_results: print("AV1 Quality Analysis:") for r in av1_results: frames = per_frame_data.get(("av1", r.gop), []) if not frames: + print(f"FAIL: GOP={r.gop}: no per-frame PSNR to judge") + failed = True continue min_psnr = min(f["psnr_avg"] for f in frames) max_psnr = max(f["psnr_avg"] for f in frames) spread = max_psnr - min_psnr - if spread > 10: - print(f" ⚠ GOP={r.gop}: PSNR spread {spread:.1f} dB " - f"(min={min_psnr:.1f}, max={max_psnr:.1f}) — quality collapse detected") + if spread > PSNR_SPREAD_LIMIT_DB: + print(f" FAIL GOP={r.gop}: PSNR spread {spread:.1f} dB " + f"(min={min_psnr:.1f}, max={max_psnr:.1f}) " + f"— quality collapse detected") + failed = True else: - print(f" ✓ GOP={r.gop}: PSNR spread {spread:.1f} dB — acceptable") + print(f" ok GOP={r.gop}: PSNR spread {spread:.1f} dB " + f"— within {PSNR_SPREAD_LIMIT_DB:g} dB") - return 0 + return 1 if failed else 0 if __name__ == "__main__": diff --git a/vk_video_encoder/test/common/encoder_test_support.h b/vk_video_encoder/test/common/encoder_test_support.h new file mode 100644 index 00000000..9b0452a3 --- /dev/null +++ b/vk_video_encoder/test/common/encoder_test_support.h @@ -0,0 +1,399 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef _ENCODER_TEST_SUPPORT_H_ +#define _ENCODER_TEST_SUPPORT_H_ + +// What every test driving the encoder interface needs, and nothing about any +// one test. +// +// THREE VERDICTS, NOT TWO. A check that could not be made is not a check that +// passed. Skip() says the machine could not answer -- no device, no importable +// format -- and is counted separately, so a run on a machine that can measure +// nothing reports that rather than a clean sheet. + +#include "vulkan_video_encoder.h" + +#include +#include +#include +#include + +namespace enctest { + +//============================================================================= +// REPORTING +//============================================================================= + +class Report { +public: + void Check(bool ok, const char* what) + { + ++m_checks; + if (ok) { + fprintf(stderr, " ok %s\n", what); + } else { + ++m_failures; + fprintf(stderr, " FAIL %s\n", what); + } + } + + void Skip(const char* what, const char* why) + { + ++m_skipped; + fprintf(stderr, " SKIP %s (%s)\n", what, why); + } + + int Summarise(const char* name) const + { + fprintf(stderr, "\n%s: %u checks, %u failures, %u skipped\n", + name, m_checks, m_failures, m_skipped); + if (m_failures != 0) { + return 1; + } + // A RUN THAT MEASURED NOTHING IS NOT A PASS, and the exit code has to + // carry that or the suite goes green without asserting anything. + // + // Zero checks with a skip recorded is a SKIP and reports itself as one + // (ctest SKIP_RETURN_CODE 77, set on the test). Zero checks with no + // skip recorded is a harness that ran nothing and did not say why, + // which is a failure: it is indistinguishable at the exit code from a + // suite whose every assertion was deleted. + if (m_checks == 0) { + return (m_skipped != 0) ? 77 : 1; + } + return 0; + } + + unsigned failures() const { return m_failures; } + unsigned skipped() const { return m_skipped; } + +private: + unsigned m_checks = 0; + unsigned m_failures = 0; + unsigned m_skipped = 0; +}; + +//============================================================================= +// DEVICE ENTRY POINTS +// +// Loaded through the loader the session was built with, so a test reaches the +// same driver the encoder does rather than whichever one is first on the path. +//============================================================================= + +struct DeviceFns { + PFN_vkGetDeviceProcAddr GetDeviceProcAddr = nullptr; + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetImageSubresourceLayout GetImageSubresourceLayout = nullptr; + PFN_vkDeviceWaitIdle DeviceWaitIdle = nullptr; + + bool Load(const vk::video::enc::Ref& binding) + { + if (!binding) { + return false; + } + PFN_vkGetInstanceProcAddr gipa = binding->GetInstanceProcAddr(); + const VkInstance instance = binding->Instance(); + const VkDevice device = binding->Device(); + if (gipa == nullptr || instance == VK_NULL_HANDLE || device == VK_NULL_HANDLE) { + return false; + } + GetDeviceProcAddr = reinterpret_cast( + gipa(instance, "vkGetDeviceProcAddr")); + GetPhysicalDeviceMemoryProperties = + reinterpret_cast( + gipa(instance, "vkGetPhysicalDeviceMemoryProperties")); + if (GetDeviceProcAddr == nullptr || GetPhysicalDeviceMemoryProperties == nullptr) { + return false; + } +#define V1_LOAD(name) \ + name = reinterpret_cast(GetDeviceProcAddr(device, "vk" #name)) + V1_LOAD(CreateImage); + V1_LOAD(DestroyImage); + V1_LOAD(GetImageMemoryRequirements); + V1_LOAD(AllocateMemory); + V1_LOAD(FreeMemory); + V1_LOAD(BindImageMemory); + V1_LOAD(MapMemory); + V1_LOAD(UnmapMemory); + V1_LOAD(GetImageSubresourceLayout); + V1_LOAD(DeviceWaitIdle); +#undef V1_LOAD + return CreateImage && DestroyImage && GetImageMemoryRequirements && + AllocateMemory && FreeMemory && BindImageMemory && MapMemory && + UnmapMemory && GetImageSubresourceLayout && DeviceWaitIdle; + } +}; + +//============================================================================= +// AN INPUT IMAGE THE HOST CAN WRITE +// +// LINEAR and host-visible, so the pattern is written by mapping it. That is +// not a shortcut: a session created without an adopted device exposes no queue +// family, so there is no queue on which a staging copy could be submitted. +// Mapping needs none. +//============================================================================= + +class HostImage { +public: + HostImage(const DeviceFns& fns, VkPhysicalDevice phys, VkDevice device, + uint32_t width, uint32_t height) + : m_fns(fns), m_phys(phys), m_device(device), + m_width(width), m_height(height) { } + + ~HostImage() { Destroy(); } + + HostImage(const HostImage&) = delete; + HostImage& operator=(const HostImage&) = delete; + + bool Create(const char** why) + { + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {m_width, m_height, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = VK_IMAGE_TILING_LINEAR; + ci.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + // Written from the host before anything else touches it, so its + // contents must survive the first transition. + ci.initialLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + + if (m_fns.CreateImage(m_device, &ci, nullptr, &m_image) != VK_SUCCESS) { + *why = "vkCreateImage refused a linear NV12 image"; + return false; + } + + VkMemoryRequirements req{}; + m_fns.GetImageMemoryRequirements(m_device, m_image, &req); + VkPhysicalDeviceMemoryProperties props{}; + m_fns.GetPhysicalDeviceMemoryProperties(m_phys, &props); + + const VkMemoryPropertyFlags want = + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; + uint32_t typeIndex = UINT32_MAX; + for (uint32_t i = 0; i < props.memoryTypeCount; ++i) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((props.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + *why = "no host-visible memory type accepts a linear NV12 image"; + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (m_fns.AllocateMemory(m_device, &ai, nullptr, &m_memory) != VK_SUCCESS) { + *why = "vkAllocateMemory failed"; + return false; + } + if (m_fns.BindImageMemory(m_device, m_image, m_memory, 0) != VK_SUCCESS) { + *why = "vkBindImageMemory failed"; + return false; + } + m_size = req.size; + return true; + } + + // A moving pattern, so successive frames differ and the encoder has + // something to predict. A flat image lets a broken submission path still + // produce plausible output. + bool WritePattern(uint32_t frameIndex, const char** why) + { + VkImageSubresource luma{VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0}; + VkImageSubresource chroma{VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0}; + VkSubresourceLayout lumaLayout{}, chromaLayout{}; + m_fns.GetImageSubresourceLayout(m_device, m_image, &luma, &lumaLayout); + m_fns.GetImageSubresourceLayout(m_device, m_image, &chroma, &chromaLayout); + + void* mapped = nullptr; + if (m_fns.MapMemory(m_device, m_memory, 0, VK_WHOLE_SIZE, 0, &mapped) != VK_SUCCESS) { + *why = "vkMapMemory failed"; + return false; + } + uint8_t* base = static_cast(mapped); + const uint32_t bar = (frameIndex * 24) % m_width; + for (uint32_t y = 0; y < m_height; ++y) { + uint8_t* row = base + lumaLayout.offset + y * lumaLayout.rowPitch; + for (uint32_t x = 0; x < m_width; ++x) { + const bool inBar = (x >= bar) && (x < bar + 24); + row[x] = inBar ? 235 : static_cast(16 + (x * 200) / m_width); + } + } + for (uint32_t y = 0; y < m_height / 2; ++y) { + uint8_t* row = base + chromaLayout.offset + y * chromaLayout.rowPitch; + for (uint32_t x = 0; x < m_width / 2; ++x) { + row[2 * x] = 128; + row[2 * x + 1] = 128; + } + } + m_fns.UnmapMemory(m_device, m_memory); + return true; + } + + void Destroy() + { + if (m_image != VK_NULL_HANDLE) { + m_fns.DestroyImage(m_device, m_image, nullptr); + m_image = VK_NULL_HANDLE; + } + if (m_memory != VK_NULL_HANDLE) { + m_fns.FreeMemory(m_device, m_memory, nullptr); + m_memory = VK_NULL_HANDLE; + } + } + + // How this image is described to the interface. The usage and creation + // flags must match vkCreateImage exactly: the library cannot read them + // back from a VkImage and uses them to decide how the frame reaches the + // encoder. + vk::video::enc::ExternalImage Describe() const + { + vk::video::enc::ExternalImage image; + image.handleType = vk::video::enc::ExternalHandleType::VkImageHandle; + image.existingImage = m_image; + image.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + image.width = m_width; + image.height = m_height; + image.tiling = VK_IMAGE_TILING_LINEAR; + image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + image.colorModel = vk::video::enc::ColorModel::YCbCr; + image.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + image.createFlags = 0; + image.residency = vk::video::enc::Residency::Local; + image.allocationSize = m_size; + return image; + } + + VkImage Image() const { return m_image; } + +private: + const DeviceFns& m_fns; + VkPhysicalDevice m_phys; + VkDevice m_device; + uint32_t m_width; + uint32_t m_height; + VkImage m_image = VK_NULL_HANDLE; + VkDeviceMemory m_memory = VK_NULL_HANDLE; + uint64_t m_size = 0; +}; + +//============================================================================= +// A CONFIGURED SESSION +// +// The shape every test here starts from: a platform on the default device, a +// rate-controlled H.264 configuration at the requested size, and the roles. +// Returns false when the machine cannot answer, with |why| set -- which the +// caller reports as a skip rather than a failure. +//============================================================================= + +struct Session { + vk::video::enc::Ref platform; + vk::video::enc::Ref config; + vk::video::enc::Ref session; + vk::video::enc::Ref submitter; + vk::video::enc::Ref bitstream; + vk::video::enc::Ref completion; + vk::video::enc::Ref registry; + vk::video::enc::Ref binding; + vk::video::enc::Ref diagnostics; + + bool Open(uint32_t width, uint32_t height, const char** why) + { + using namespace vk::video::enc; + + PlatformCreateInfo info; + info.silenceStdio = true; + if (VkEncCreatePlatform(info, platform) != VK_SUCCESS || !platform) { + *why = "no Vulkan encode platform"; + return false; + } + if (platform->Caps()->Codecs().empty()) { + *why = "no encode-capable device"; + return false; + } + + Expected> made = + platform->CreateConfig(Codec::H264, Profile::H264Main); + if (!made) { + *why = made.status().detail(); + return false; + } + config = *made; + + RateControl rate; + rate.mode = RateControlMode::Cbr; + rate.averageBitrate = 2 * 1000 * 1000; + rate.maxBitrate = 2 * 1000 * 1000; + config->SetCodedExtent(width, height) + .SetFrameRate(30, 1) + .SetInputFormat(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ColorModel::YCbCr) + .SetRateControl(rate); + + Expected> opened = platform->CreateSession(config); + if (!opened) { + *why = opened.status().detail(); + return false; + } + session = *opened; + submitter = Query(session); + bitstream = Query(session); + completion = Query(session); + registry = Query(session); + binding = Query(session); + diagnostics = Query(session); + if (!submitter || !bitstream || !registry || !binding) { + *why = "the session does not offer the roles a test needs"; + return false; + } + return true; + } + + // Take everything ready, returning how many frames were retired. + uint32_t DrainRetrievable() + { + uint32_t retired = 0; + for (;;) { + vk::video::enc::Expected got = + bitstream->AcquireNext(); + if (!got) { + break; + } + bitstream->Release(got->frameId); + ++retired; + } + return retired; + } +}; + +} // namespace enctest + +#endif /* _ENCODER_TEST_SUPPORT_H_ */ diff --git a/vk_video_encoder/test/encoder-drain/CMakeLists.txt b/vk_video_encoder/test/encoder-drain/CMakeLists.txt new file mode 100644 index 00000000..1c5954c9 --- /dev/null +++ b/vk_video_encoder/test/encoder-drain/CMakeLists.txt @@ -0,0 +1,28 @@ +# Drain() leaves a session usable; Finish() ends it. +# +# Drives the encoder interface only -- no internal header, no probe seam -- +# because the property under test is one the interface promises and a caller +# can observe. Skips (exit 0 with a skip count) without an encode-capable GPU. + +set(TEST_NAME vk-video-enc-drain-test) + +add_executable(${TEST_NAME} src/main.cpp) + +target_include_directories(${TEST_NAME} PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/../common + ${CMAKE_CURRENT_SOURCE_DIR}/../../include + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} +) + +target_link_libraries(${TEST_NAME} PRIVATE vkvideo-encoder-static ${CMAKE_DL_LIBS}) + +if (TARGET GenVulkanDispatchTable) + add_dependencies(${TEST_NAME} GenVulkanDispatchTable) +endif() + +add_test(NAME ${TEST_NAME} COMMAND ${TEST_NAME}) +# A run that skips every check reports SKIP, not PASS: see Report::Summarise. +set_tests_properties(${TEST_NAME} PROPERTIES + LABELS "gpu" + SKIP_RETURN_CODE 77) diff --git a/vk_video_encoder/test/encoder-drain/src/main.cpp b/vk_video_encoder/test/encoder-drain/src/main.cpp new file mode 100644 index 00000000..d167b1b7 --- /dev/null +++ b/vk_video_encoder/test/encoder-drain/src/main.cpp @@ -0,0 +1,158 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// Drain() is not the end of the session, and Finish() is. +// +// THE INVARIANT. After a drain, everything submitted has been encoded and the +// session is exactly as usable as it was before: further frames are accepted, +// each becomes retrievable once, and the completion surface still reports +// them. A drain may be taken as often as a caller likes. +// +// WHY IT IS WORTH A TEST OF ITS OWN. The failure this guards is silent. A +// drain that quietly ends the completion surface leaves a session that still +// ACCEPTS frames and never retires any: the submissions succeed, the encoder +// reports no error, and the caller waits forever for output that will not +// come. Nothing crashes and nothing returns a failure -- so only a second +// batch, submitted after a drain and required to retire, catches it. +// +// The second batch is what this test is. The first batch exists to prove the +// pipeline retires anything at all, because a second batch that retires +// nothing proves nothing if the first did not either. + +#include "encoder_test_support.h" + +#include + +using namespace vk::video::enc; + +namespace { + +const uint32_t kWidth = 320; +const uint32_t kHeight = 240; +const uint32_t kBatch = 6; + +// Submit |count| frames from |id|, returning how many the session accepted. +uint32_t SubmitBatch(enctest::Session& s, + enctest::HostImage& image, + ResourceId resource, + uint64_t firstId, + uint32_t count) +{ + uint32_t accepted = 0; + for (uint32_t i = 0; i < count; ++i) { + const char* why = ""; + if (!image.WritePattern(i, &why)) { + break; + } + FrameSubmit frame; + frame.frameId = firstId + i; + frame.pts = (firstId + i) * 3000; + frame.registeredImage = resource; + frame.image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + if (!s.submitter->SubmitFrame(frame)) { + break; + } + ++accepted; + } + return accepted; +} + +} // namespace + +int main(int argc, const char** argv) +{ + (void)argc; + (void)argv; + + enctest::Report report; + + enctest::Session s; + const char* why = ""; + if (!s.Open(kWidth, kHeight, &why)) { + report.Skip("every check", why); + return report.Summarise("encoder-drain"); + } + + enctest::DeviceFns fns; + if (!fns.Load(s.binding)) { + report.Skip("every check", "device entry points unavailable"); + return report.Summarise("encoder-drain"); + } + + enctest::HostImage image(fns, s.binding->PhysicalDevice(), s.binding->Device(), + kWidth, kHeight); + if (!image.Create(&why)) { + report.Skip("every check", why); + return report.Summarise("encoder-drain"); + } + + Expected registered = s.registry->RegisterImage(image.Describe()); + if (!registered) { + report.Skip("every check", registered.status().detail()); + return report.Summarise("encoder-drain"); + } + + // ---- batch one: prove the pipeline retires at all ---- + const uint32_t submitted1 = SubmitBatch(s, image, *registered, 1000, kBatch); + report.Check(submitted1 == kBatch, "every frame in the first batch is accepted"); + + report.Check(static_cast(s.session->Drain()), "the first drain completes"); + const uint32_t retired1 = s.DrainRetrievable(); + fprintf(stderr, " batch 1: %u submitted, %u retired\n", submitted1, retired1); + + // Asserted before the second batch, because a second batch that retires + // nothing proves nothing if the first retired nothing either. + report.Check(retired1 > 0, "the first batch retires frames"); + + // ---- batch two: the session survived the drain ---- + const uint32_t submitted2 = SubmitBatch(s, image, *registered, 2000, kBatch); + report.Check(submitted2 == kBatch, + "a frame is still accepted after a drain"); + + report.Check(static_cast(s.session->Drain()), "a second drain completes"); + const uint32_t retired2 = s.DrainRetrievable(); + fprintf(stderr, " batch 2: %u submitted, %u retired\n", submitted2, retired2); + + report.Check(retired2 > 0, + "a frame submitted after a drain still becomes retrievable"); + + // ---- and the completion surface survived it too ---- + if (s.completion) { + report.Check(s.completion->CompletedCount() >= submitted1 + submitted2, + "the completion counter covers both batches"); + report.Check(s.completion->CompletionSemaphore() != VK_NULL_HANDLE, + "the completion semaphore outlives a drain"); + } + + // ---- Finish, by contrast, does end it ---- + report.Check(static_cast(s.session->Finish()), "Finish ends the stream"); + s.DrainRetrievable(); + + FrameSubmit after; + after.frameId = 3000; + after.registeredImage = *registered; + after.image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + report.Check(!s.submitter->SubmitFrame(after), + "a frame submitted after Finish is refused"); + report.Check(s.completion && s.completion->CompletionSemaphore() == VK_NULL_HANDLE, + "the completion semaphore is withdrawn once the stream ends"); + + s.registry->UnregisterImage(*registered); + fns.DeviceWaitIdle(s.binding->Device()); + image.Destroy(); + + return report.Summarise("encoder-drain"); +} diff --git a/vk_video_encoder/test/encoder-ext-acquire-fd/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-acquire-fd/CMakeLists.txt new file mode 100644 index 00000000..4a173606 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-acquire-fd/CMakeLists.txt @@ -0,0 +1,156 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_acquire_fd_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The PUBLIC API only, like the sibling release-fence test -- a real device +# needs no internal seams. The STATIC archive is what it links, which is also +# what puts the code under test on the near side of this executable's own +# close(2) definition. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # The descriptor API is an internal header: the public surface of this + # library is the encoder interface, and a test that drives the layer + # beneath it names the internal directory to say so. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +# This test defines close(2) in the executable and counts the calls the +# library makes on one watched descriptor, forwarding every one of them to +# libc through RTLD_NEXT. +# +# The binding that matters happens at STATIC LINK TIME, not at load time: +# vulkan_video_encoder_ext.cpp.o carries `U close`, the archive is linked into +# this executable, and the executable defines close -- so the code under test +# calls this counter and nothing else. +# +# WHAT THE SHIPPED BINARY ACTUALLY SHOWS: +# +# nm vulkan_video_encoder_ext.cpp.o | grep -w 'U close' -> present +# nm | grep -w close -> `t close`: LOCAL, defined, in .symtab +# nm -D | grep -w close -> nothing +# readelf --dyn-syms | grep -w close -> nothing +# +# The third and fourth lines are the same fact stated twice, and the earlier +# claim that `readelf --dyn-syms ... $8=="close"` showed "FUNC, defined" was +# simply false: there is NO `close` in .dynsym at all, defined or undefined. +# The definition is LOCAL because the project builds with hidden visibility, +# and a local symbol is never placed in the dynamic table. +# +# `nm -D | grep -w close -> nothing` is still worth recording and is still +# true: a residual UNDEFINED `close` in .dynsym would mean some call site in +# this executable still reaches libc through the PLT. It is a NEGATIVE check +# and it is all the symbol table can offer here -- it cannot show that the +# encoder archive binds to the local definition, only that nothing in the +# executable binds past it. +# +# WHAT ACTUALLY ESTABLISHES THE BINDING is the RED/GREEN differential the +# suite produces: the same test source, linked against the unfixed library, +# reports closes=0 on all six refusals; linked against the fixed one it +# reports closes=1, cross-checked by an independent fcntl -> EBADF and a +# /proc/self/fd delta of -1. A counter the library never reached could not +# move between those two builds. Symbol-table inspection is corroboration, +# not the proof. +# +# --export-dynamic is asked for below so the same definition would be offered +# to shared libraries too. It is INERT as things stand, and measurably so: the +# symbol is LOCAL under hidden visibility, so --export-dynamic has nothing to +# export and the Vulkan driver's own fd traffic is NOT counted. That is the +# scope this test wants anyway, since every assertion here is about what the +# ENCODER LIBRARY did to one descriptor. The flag is kept because it costs +# nothing and would matter if the visibility settings ever changed; it is +# recorded as inert so nobody reads it as evidence of anything. +if(UNIX AND NOT APPLE) + target_link_options(${PROJECT_NAME} PRIVATE -Wl,--export-dynamic) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# CTest semantics, matching the sibling release-fence test: 0 every assertion +# held, 1 an assertion failed, 77 no encode-capable GPU and therefore nothing +# proved either way -- reported as SKIPPED, never as a pass. +enable_testing() +add_test(NAME EncoderExtAcquireFenceFdOwnership + COMMAND ${PROJECT_NAME}) +set_tests_properties(EncoderExtAcquireFenceFdOwnership PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu" + TIMEOUT 900) + +# --------------------------------------------------------------------------- +# A SECOND ENTRY: THE ORDERING CLAIM, ON ITS OWN, AS A GATE. +# --------------------------------------------------------------------------- +# The header promises "Supplying an acquireFenceFd and a pWaitSemaphores array +# together is legal and loses neither". It holds, and this entry is what says +# so by name. --caller-wait-order runs ONLY that case, so a CTest failure line +# points at the claim rather than at the whole fd-ownership suite. +# +# It carried WILL_FAIL for one day, on the strength of a measurement that was +# an artifact. The mutation proof for the case forces the carry-across loop at +# vulkan_video_encoder_ext.cpp:6268 to zero iterations; that mutant was built, +# run red, reverted -- and the revert restored an mtime OLDER than the object +# the mutant had produced, so make rebuilt nothing and every later run re-ran +# the mutant. Rebuilding the mutant reproduces the reported red exactly, +# including its second fingerprint: a red run emits no "asyncAssemblyFence is +# not done after N mSec" warning because the frame is never parked, a green +# run emits two. Clean source is green 63 runs out of 63. +# +# So WILL_FAIL is gone, the library was not changed, and this is a gating +# assertion in both directions: the case also holds a caller TIMELINE SIGNAL +# open, which is the first coverage in this tree of the release-fence +# append's own carry-across loop. +add_test(NAME EncoderExtAcquireFdKeepsCallerWaitArray + COMMAND ${PROJECT_NAME} --caller-wait-order) +set_tests_properties(EncoderExtAcquireFdKeepsCallerWaitArray PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu" + TIMEOUT 900) + +message(STATUS "encoder_ext_acquire_fd_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-acquire-fd/src/main.cpp b/vk_video_encoder/test/encoder-ext-acquire-fd/src/main.cpp new file mode 100644 index 00000000..8d7eaee8 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-acquire-fd/src/main.cpp @@ -0,0 +1,1761 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Per-frame ACQUIRE fence fd-ownership coverage, on a real device. + * + * WHAT IS UNDER TEST. VkVideoEncoderFrameFenceDescriptor::acquireFenceFd -- + * the IMPORT half of the per-frame fence entry point, and specifically the + * ownership promise attached to it. The public header says of that field "the + * library takes ownership and closes it on every exit path", and design + * section 2.3 states the rule the whole handle API is built on: the library + * consumes POSIX fds it is given, always, on every exit path, "including + * argument-validation failures that never reach Vulkan". The SYNC_FD row of + * that section's ownership table names the acquire fence explicitly. + * + * THE DEFECT THIS PINS. acquireFenceFd was read at exactly one site -- inside + * the resolving walk over the chained descriptors -- and closed only inside + * ImportAcquireFenceLocked. Every refusal that returns before that site + * therefore leaked the caller's fd: + * + * - a mis-stamped info.sType (top of the function) + * - a session that is not initialized (top of the function) + * - a resource id that does not resolve (top of the function) + * - an unknown chained sType AHEAD of the fence node (in the walk) + * - an unresolvable registered id AHEAD of the fence node (in the walk) + * + * The last two are reachable only because the header blesses the nested chain + * shape `info.pNext = &sync; sync.pNext = &fence;` -- a refusal raised while + * walking the leading node returns before the fence node is ever visited. + * + * This is a live leak, not a theoretical one: Chromium's + * media/gpu/vulkan/vulkan_video_encode_accelerator.cc arms the field on + * essentially every frame with `fence_desc.acquireFenceFd = fd.release();`, + * under a comment saying ownership transfers. + * + * HOW IT IS PROVED, and why each instrument is here. + * + * fcntl(fd, F_GETFD) == -1 with errno EBADF -- the fd is gone. This is the + * direct statement of the promise, and it is what goes red on the + * unfixed library. + * an interposed close(2) -- counts the closes the + * library performed on that EXACT fd number during the call. EBADF + * alone cannot tell one close from two, and a double close is worse + * than the leak: the number is free the instant the first close + * returns, so the second one can land on an unrelated descriptor the + * process has since opened. "Exactly once" needs a counter, so there is + * one. The wrapper is pure passthrough unless a watch is armed. + * /proc/self/fd -- an independent second + * opinion that does not depend on the interposer being wired at all. + * + * WHY THE FDS ARE REAL SYNC FDS. A pipe fd would exercise close(2) just as + * well on the refusal paths, but not on the success path: there the fd must + * survive vkImportSemaphoreFdKHR, which only a genuine sync_file will. So + * every fd here is minted the way a producer mints one -- exported from a + * semaphore carrying a real queue signal -- by asking the library itself for + * a per-frame RELEASE fence and then handing that fd back as the next frame's + * ACQUIRE fence. That is not a trick: the header documents exactly that + * lifetime for an exported fence ("close(2) it, or hand it to exactly one + * import, which consumes it"), and it is the true producer/consumer shape. + * + * WHY IT NEEDS A GPU. Minting a real sync_fd needs a real queue signal, and + * the success case needs a real import. With no encode-capable device this + * exits 77, which CTest reads as SKIP. A session that DID come up and then + * could not mint an fd is a FAILURE, not a skip -- otherwise a regressed + * export would turn this whole file green by making it vanish. + */ + +#include "vulkan_video_encoder_ext.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; +const char* g_case = "setup"; + +// Selected by argv. --caller-wait-order runs ONLY case 10, the ordering +// case, so that the claim it pins has a CTest entry of its own that names it. +// It is not a different assertion level: case 10 asserts the same things in +// both modes. The claim holds (see the note on case 10), so the split buys +// focus and nothing else: it gives the ordering claim a name of its own in +// the CTest output. +bool g_assertCallerWaitOrder = false; + +// The per-case check floor for case 10. Named because two places assert it: +// the ledger table in main(), and the --caller-wait-order mode, which does +// not run the table and would otherwise be a mode that can pass by doing +// nothing. +const int kCallerWaitOrderFloor = 10; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + return; + } + g_failures++; + std::printf(" FAIL [%s] %s : %s\n", g_case, what, detail.c_str()); +} + +std::string I64(long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%lld", v); + return buf; +} + +//============================================================================= +// The close(2) interposer. +// +// A strong definition of close() in the EXECUTABLE takes precedence over +// libc's for every call site linked into it, which includes the whole static +// encoder archive under test. Everything is forwarded to the real close via +// RTLD_NEXT, so this changes no behaviour; it only counts. +// +// It counts closes of ONE fd, and only while a watch is armed, deliberately. +// A global tally would be dominated by the driver's own fd traffic and could +// prove nothing about a particular handle. +//============================================================================= +std::atomic g_watchedFd(-1); +std::atomic g_closeCount(0); +std::atomic g_realClose(nullptr); + +} // namespace + +extern "C" int close(int fd) +{ + void* real = g_realClose.load(std::memory_order_acquire); + if (real == nullptr) { + // Racing resolvers all compute the same address, so the last writer + // wins harmlessly. dlsym does not itself close descriptors. + real = dlsym(RTLD_NEXT, "close"); + g_realClose.store(real, std::memory_order_release); + } + if (fd == g_watchedFd.load(std::memory_order_relaxed)) { + g_closeCount.fetch_add(1, std::memory_order_relaxed); + } + if (real == nullptr) { + errno = EBADF; + return -1; + } + return ((int (*)(int))real)(fd); +} + +namespace { + +// Live entries in /proc/self/fd, minus the handle the scan itself holds open. +int OpenFdCount() +{ + DIR* d = opendir("/proc/self/fd"); + if (d == nullptr) { + return -1; + } + int n = 0; + while (struct dirent* e = readdir(d)) { + if ((std::strcmp(e->d_name, ".") == 0) || + (std::strcmp(e->d_name, "..") == 0)) { + continue; + } + n++; + } + closedir(d); + return n - 1; +} + +// poll() for readability. A sync_fd becomes readable when its fence signals. +// Returns 1 signalled, 0 not yet, <0 error. +int PollSignalled(int fd, int timeoutMs) +{ + struct pollfd p = {}; + p.fd = fd; + p.events = POLLIN; + const int r = poll(&p, 1, timeoutMs); + if (r < 0) { + return -1; + } + if (r == 0) { + return 0; + } + return ((p.revents & POLLIN) != 0) ? 1 : -1; +} + +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; +// Enough repetitions for the fd table to show drift if one fd per refusal is +// retained, and short enough to stay well inside the CTest timeout. +const uint32_t kRefusalLoopIterations = 120; + +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = + nullptr; + PFN_vkGetPhysicalDeviceProperties GetPhysicalDeviceProperties = + nullptr; + // For case 10: a caller wait semaphore this test can hold unsignalled and + // then signal from the host, which is what turns "both waits were + // honoured" into something observable rather than merely consistent. + PFN_vkCreateSemaphore CreateSemaphore = nullptr; + PFN_vkDestroySemaphore DestroySemaphore = nullptr; + PFN_vkSignalSemaphore SignalSemaphore = nullptr; + // The SIGNAL half of case 10: reading the caller timeline back is what + // makes "the caller signal array survived the release-fence append" an + // observation rather than an inference. + PFN_vkGetSemaphoreCounterValue GetSemaphoreCounterValue = nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) { + lib = dlopen("libvulkan.so", RTLD_NOW); + } + if (lib == nullptr) { + std::printf(" ERROR: dlopen(libvulkan) failed: %s\n", dlerror()); + return false; + } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) { + std::printf(" ERROR: no vkGetInstanceProcAddr\n"); + return false; + } + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + std::printf(" ERROR: no vkGetDeviceProcAddr\n"); + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) + LOAD_DEV(CreateSemaphore) + LOAD_DEV(DestroySemaphore) + LOAD_DEV(SignalSemaphore) + LOAD_DEV(GetSemaphoreCounterValue) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + fns->GetPhysicalDeviceProperties = + (PFN_vkGetPhysicalDeviceProperties)gipa( + instance, "vkGetPhysicalDeviceProperties"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr) && + (fns->GetPhysicalDeviceProperties != nullptr); +} + +struct InputImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; +}; + +// A host-written LINEAR NV12 image, transfer-source usage only, so the +// registration routes STAGED: the staging copy is then the submission that +// reads the input and therefore the one that signals the release fence. That +// is the arm the sibling release-fence test measured at 300/300 fds, which is +// what makes it a dependable source of real sync_fds here. +bool CreateInputImage(const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, InputImage* out) +{ + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = VK_IMAGE_TILING_LINEAR; + ci.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + std::printf(" ERROR: vkCreateImage(LINEAR NV12) failed\n"); + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + const VkMemoryPropertyFlags want = + (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + std::printf(" ERROR: no host-visible memory type\n"); + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + std::printf(" ERROR: vkAllocateMemory failed\n"); + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != VK_SUCCESS) { + std::printf(" ERROR: vkBindImageMemory failed\n"); + return false; + } + void* mapped = nullptr; + if (fns.MapMemory(device, out->memory, 0, req.size, 0, &mapped) == + VK_SUCCESS) { + uint8_t* bytes = (uint8_t*)mapped; + for (VkDeviceSize i = 0; i < req.size; i++) { + bytes[i] = (uint8_t)((i * 7u) ^ (i >> 9)); + } + fns.UnmapMemory(device, out->memory); + } + return true; +} + +//============================================================================= +// The session under test, plus the fd mint. +//============================================================================= +VkSharedBaseObj g_encoder; +DeviceFns g_fns; +VkDevice g_device = VK_NULL_HANDLE; +VkVideoEncoderResource g_resource = VK_VIDEO_ENCODER_RESOURCE_NULL; +uint64_t g_frameId = 0; +uint32_t g_captured = 0; +uint64_t g_capturedBytes = 0; + +void DrainCaptures() +{ + VkVideoEncodeResult r; + while (g_encoder->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + g_captured++; + g_capturedBytes += r.bitstreamSize; + g_encoder->ReleaseEncodedFrame(r.frameId); + } +} + +VkVideoEncoderFrameSubmitInfo BaseInfo() +{ + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = g_resource; + info.frameId = g_frameId; + info.pts = g_frameId; + info.qpOverride = -1; + info.currentLayout = VK_IMAGE_LAYOUT_UNDEFINED; + return info; +} + +// One real sync_fd, exported by the library from a semaphore its own queue +// signals. Carries NO acquire fence itself, so a NOT_READY retry here can +// never lose a handle: the only thing at stake is the out-parameter, which +// the next attempt rewrites. +int MintSyncFd() +{ + for (int attempt = 0; attempt < 64; attempt++) { + int fd = -1; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &fd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + + VkVideoEncoderStatusCode status = + g_encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 2000); + retry++) { + DrainCaptures(); + status = g_encoder->SubmitRegisteredFrame(info, nullptr); + if (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + struct timespec ts = {0, 1000000}; // 1 ms + nanosleep(&ts, nullptr); + } + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + std::printf(" MINT: submit failed, status %d\n", (int)status); + return -1; + } + g_frameId++; + DrainCaptures(); + if (fd >= 0) { + return fd; + } + // -1 is a legal answer ("no fence"), not a failure. Try the next + // frame rather than giving up -- but never silently, so a run that + // needed many attempts is visible. + std::printf(" MINT: frame %llu answered -1, retrying\n", + (unsigned long long)(g_frameId - 1)); + } + return -1; +} + +//============================================================================= +// Everything observable about one watched call. +//============================================================================= +struct CallResult { + VkVideoEncoderStatusCode status = VK_VIDEO_ENCODER_STATUS_SUCCESS; + int closes = 0; + bool stillOpen = true; + int errnoAfter = 0; + int fdCountDelta = 0; +}; + +CallResult RunWatched(VulkanVideoEncoderExt* enc, + VkVideoEncoderFrameSubmitInfo& info, int fd) +{ + CallResult r; + const int before = OpenFdCount(); + g_closeCount.store(0, std::memory_order_relaxed); + g_watchedFd.store(fd, std::memory_order_relaxed); + + r.status = enc->SubmitRegisteredFrame(info, nullptr); + + g_watchedFd.store(-1, std::memory_order_relaxed); + r.closes = g_closeCount.load(std::memory_order_relaxed); + // Immediately, before anything in this process can open a descriptor and + // be handed the same number back. + errno = 0; + r.stillOpen = (fcntl(fd, F_GETFD) != -1); + r.errnoAfter = errno; + r.fdCountDelta = OpenFdCount() - before; + return r; +} + +// The three assertions every refusal owes, plus the status it must answer. +void ExpectRefusalConsumedTheFd(const CallResult& r, + VkVideoEncoderStatusCode want) +{ + Check(r.status == want, "status", + "got " + I64((long long)r.status) + ", want " + + I64((long long)want)); + Check(!r.stillOpen, "the acquire fd was closed", + r.stillOpen ? "fcntl(F_GETFD) still succeeds -- THE FD LEAKED" + : "closed"); + Check(r.stillOpen || (r.errnoAfter == EBADF), "fcntl reports EBADF", + "errno " + I64(r.errnoAfter)); + Check(r.closes == 1, "closed exactly once", + I64(r.closes) + " close(2) calls on that fd -- 0 is a leak, " + "2 can close an unrelated descriptor"); + Check(r.fdCountDelta == -1, "/proc/self/fd fell by exactly one", + "delta " + I64(r.fdCountDelta)); + std::printf(" [%s] status=%d closes=%d stillOpen=%d fdDelta=%d\n", + g_case, (int)r.status, r.closes, (int)r.stillOpen, + r.fdCountDelta); +} + +//============================================================================= +// Case 1 -- a mis-stamped info.sType. Refuses at the very top of the +// function, before the descriptor walk exists. +//============================================================================= +void CaseWrongTopLevelSType() +{ + g_case = "WrongTopLevelSType"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + // A real sType from this same header, just not the one this entry point + // takes -- a likelier caller error than a random integer. + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_SYNC_DESCRIPTOR; + info.pNext = &fence; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN); + // The OUT half of the same promise, on the same exit. ext.h says the + // library writes pReleaseFenceFd before anything in the call can refuse, + // naming ERROR_STRUCTURE_TYPE_UNKNOWN specifically. + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); +} + +//============================================================================= +// Case 2 -- a session that is not initialized. Same top-of-function region, +// a different refusal, and the one a caller hits by mis-sequencing its own +// startup. +//============================================================================= +void CaseNotInitialized() +{ + g_case = "NotInitialized"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + VkSharedBaseObj fresh; + if ((CreateVulkanVideoEncoderExt(fresh) != VK_SUCCESS) || !fresh) { + Check(false, "CreateVulkanVideoEncoderExt", "second session"); + ::close(fd); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.pNext = &fence; + info.resource = g_resource; // meaningless to a session with none + info.qpOverride = -1; + info.currentLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + const CallResult r = RunWatched(fresh.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_NOT_INITIALIZED); + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); + fresh = nullptr; +} + +//============================================================================= +// Case 3 -- a resource id that does not resolve. Refuses under the resource +// lock, still ahead of the descriptor walk. This is the one a producer hits +// by racing an Unregister against an in-flight submit, which is exactly when +// it is holding a fence fd. +//============================================================================= +void CaseResourceUnknown() +{ + g_case = "ResourceUnknown"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + info.resource = (VkVideoEncoderResource)0xDEADBEEFull; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN); + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); +} + +// A chained struct this library does not know, laid out the way Vulkan's +// pNext convention requires: {sType, pNext} first. +struct UnknownChainNode { + VkVideoEncoderStructureType sType; + const void* pNext; +}; + +//============================================================================= +// Case 4 -- an unknown chained sType AHEAD of the fence descriptor. The walk +// refuses on the leading node and never reaches the fence node at all. Only +// reachable because the header blesses a flat, order-independent chain. +//============================================================================= +void CaseUnknownSTypeAheadOfFence() +{ + g_case = "UnknownSTypeAheadOfFence"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + UnknownChainNode unknown; + unknown.sType = (VkVideoEncoderStructureType)0x7F00DEAD; + unknown.pNext = &fence; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &unknown; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN); + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); +} + +//============================================================================= +// Case 5 -- an unresolvable registered WAIT id ahead of the fence descriptor. +// `info.pNext = &sync; sync.pNext = &fence;` is one of the three chain shapes +// the header blesses by name. +//============================================================================= +void CaseUnresolvableWaitIdAheadOfFence() +{ + g_case = "UnresolvableWaitIdAheadOfFence"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + const VkVideoEncoderResource bogus = (VkVideoEncoderResource)0xBADC0FFEEull; + const uint64_t value = 7; + VkVideoEncoderFrameSyncDescriptor sync; + sync.pNext = &fence; + sync.waitCount = 1; + sync.pWaitSemaphores = &bogus; + sync.pWaitValues = &value; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &sync; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN); + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); +} + +//============================================================================= +// Case 6 -- the same, on the SIGNAL side of the leading sync node. A separate +// refusal site in the walk, and the fence node is equally unreached. +//============================================================================= +void CaseUnresolvableSignalIdAheadOfFence() +{ + g_case = "UnresolvableSignalIdAheadOfFence"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + const VkVideoEncoderResource bogus = (VkVideoEncoderResource)0xBADC0FFEEull; + const uint64_t value = 7; + VkVideoEncoderFrameSyncDescriptor sync; + sync.pNext = &fence; + sync.signalCount = 1; + sync.pSignalSemaphores = &bogus; + sync.pSignalValues = &value; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &sync; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN); + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); +} + +//============================================================================= +// Case 6b -- A CYCLIC pNext CHAIN. `fence.pNext = &fence` is one assignment +// away in caller memory and nothing in the ABI forbids it. Before the chain +// walk was bounded this call did not refuse and did not return: it spun inside +// the library at 100% CPU, with the caller's fd still open, forever. A hang is +// a worse answer than any refusal, which is why this is asserted. +// +// That the case terminates at all is half the assertion, and CTest's TIMEOUT +// is the instrument for that half. The other half is that the bound did not +// cost the consumption promise: the fd here is real and armed, so a chain the +// library refuses to WALK is still a chain it was handed an fd on and still +// owes a close for. +// +// It also pins the pReleaseFenceFd half of the same pre-pass, which had the +// identical unbounded exposure. +//============================================================================= +void CaseCyclicChainIsRefused() +{ + g_case = "CyclicChainRefused"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + fence.pNext = &fence; // the cycle + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + ExpectRefusalConsumedTheFd( + r, VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN); + Check(releaseFd == -1, "pReleaseFenceFd was written", + "got " + I64(releaseFd) + ", want -1"); +} + +//============================================================================= +// Case 6c -- ONE fd NAMED IN TWO FENCE DESCRIPTORS. +// +// The guard's Record() deduplicates by value so it can never close one number +// twice, and for a while a comment claimed that dedup also made the walk +// REPORT the duplicate. It did not: nothing compared the two nodes, the second +// Consume() was a silent no-op, and ImportAcquireFenceLocked ran a second time +// on a descriptor the first import had already consumed -- which is not +// guaranteed to fail, so the caller could come away with two semaphores whose +// payload came from one fence. +// +// The consume site now refuses. What is asserted here: +// +// status == ERROR_IMPORT_FAILED -- a TYPED refusal, not silence. +// libraryCloses == 0 -- the load-bearing one. The first import +// owns the descriptor; a close on this +// path is exactly the double close the +// guard exists to prevent, and it is the +// failure this whole file calls worse than +// the leak it fixes. +// +// stillOpen is REPORTED, not asserted, for the same reason as on the success +// path: the fd went to the driver at the first import and the driver is under +// no obligation to close it. +//============================================================================= +void CaseDuplicateAcquireFdIsRefused() +{ + g_case = "DuplicateAcquireFd"; + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFdA = 0x5EED; + int releaseFdB = 0x5EED; + VkVideoEncoderFrameFenceDescriptor second; + second.acquireFenceFd = fd; // the SAME number + second.pReleaseFenceFd = &releaseFdB; + + VkVideoEncoderFrameFenceDescriptor first; + first.acquireFenceFd = fd; + first.pReleaseFenceFd = &releaseFdA; + first.pNext = &second; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &first; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + Check(r.status == VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED, + "a duplicated acquire fd is a typed refusal", + "got " + I64((long long)r.status) + ", want IMPORT_FAILED (" + + I64((long long)VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED) + + ")"); + Check(r.closes == 0, + "the library did not close what the first import had taken", + I64(r.closes) + " close(2) calls -- anything above 0 here is the " + "double close"); + Check(releaseFdA == -1, "the first pReleaseFenceFd was written", + "got " + I64(releaseFdA) + ", want -1"); + Check(releaseFdB == -1, "the second pReleaseFenceFd was written", + "got " + I64(releaseFdB) + ", want -1"); + std::printf(" [%s] status=%d libraryCloses=%d " + "stillOpenAfterImport=%d (reported, not asserted) " + "fdDelta=%d\n", + g_case, (int)r.status, r.closes, (int)r.stillOpen, + r.fdCountDelta); +} + +//============================================================================= +// Case 7 -- THE SUCCESS PATH IS UNCHANGED, AND IS NOT DOUBLE-CONSUMED. +// +// WHAT THE LIBRARY MUST DO HERE IS THE OPPOSITE OF THE CASES ABOVE, and that +// is why it is worth running. On a refusal the library owes a close. On +// success it owes NO close at all: vkImportSemaphoreFdKHR takes ownership of +// a SYNC_FD, so from that call onward the fd belongs to the driver, and a +// library close would be a double close -- the worse of the two failures, +// ahead of the leak. +// +// So the assertion is `the library closed it ZERO times`, measured on the +// same counter the refusal cases use. That is exactly what goes red if the +// pre-pass guard is left holding the fd across the import, which is the one +// way this fix could have broken the working path. It was checked by deleting +// that handoff and rebuilding: the mutant reports 1 close here and stays +// green on all six refusals. +// +// AND THE FRAME REALLY ENCODED WITH THE FENCE IN ITS WAIT LIST. The proof is +// the RELEASE fence of the same frame. That fd is exported from a semaphore +// signalled by the submission which CONSUMES the input image -- so if it +// becomes signalled, that submission ran to completion, which it could not +// have done unless the imported acquire semaphore it was told to wait on was +// satisfied. A dropped import, a semaphore closed out from under the driver, +// or a wait on a payload that never arrives all show up here as a fence that +// never signals. +// +// WHY fcntl(F_GETFD) IS NOT ASSERTED HERE. It is reported. The driver takes +// ownership but is under no obligation to close the descriptor at import +// time, and NVIDIA's does not -- measured identically on both builds, fixed +// and unfixed. Asserting EBADF here would be asserting a driver implementation +// detail. Where the fds must provably go away is the refusal path, and that +// is asserted, per call and again in bulk in case 9. +//============================================================================= +void CaseSuccessPathConsumesOnce() +{ + g_case = "SuccessPathConsumesOnce"; + int notReadyRetries = 0; + long long notReadyCloses = 0; + for (int attempt = 0; attempt < 64; attempt++) { + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + return; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + if (r.status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + // Admission-control backpressure. That refusal sits BELOW the + // import, so the fd went to the driver exactly as on the success + // path and the library owes no close here either; the retry must + // still carry a fresh fd rather than the same number. + // + // ACCUMULATED, NOT ASSERTED PER ITERATION. How many times this + // arm is taken is timing dependent -- it moved between 0 and 1 + // across three consecutive runs -- so a Check() here made the + // suite's total check count run-variable, which is exactly the + // kind of number that gets quoted as evidence and cannot bear it. + // One assertion below covers every retry instead, which is also + // strictly stronger than asserting each one alone. + notReadyRetries++; + notReadyCloses += r.closes; + DrainCaptures(); + struct timespec ts = {0, 2000000}; + nanosleep(&ts, nullptr); + continue; + } + g_frameId++; + Check(r.status == VK_VIDEO_ENCODER_STATUS_SUCCESS, "status", + "got " + I64((long long)r.status) + ", want SUCCESS"); + // SUCCESS means the import ran and returned a semaphore -- a failed + // import is reported as ERROR_IMPORT_FAILED, never as SUCCESS -- so + // the driver has the fd. The library must not also have closed it. + Check(r.closes == 0, + "the library did not close what it handed to the driver", + I64(r.closes) + " close(2) calls on a descriptor " + "vkImportSemaphoreFdKHR already owns -- that is a DOUBLE CLOSE, " + "which can shut an unrelated fd opened at the same number"); + std::printf(" [%s] status=%d libraryCloses=%d " + "stillOpenAfterImport=%d (the driver's to close; reported, " + "not asserted) releaseFd=%d\n", + g_case, (int)r.status, r.closes, (int)r.stillOpen, + releaseFd); + + // The frame really ran WITH the acquire fence in its wait list. + Check(releaseFd >= 0, + "the input-consuming submit was issued for the fenced frame", + "release fd " + I64(releaseFd)); + if (releaseFd >= 0) { + const int signalled = PollSignalled(releaseFd, 10000); + Check(signalled == 1, + "the input-consuming submit COMPLETED, so the imported " + "acquire wait was satisfied", + "poll returned " + I64(signalled) + + " -- a frame whose acquire semaphore never resolves " + "never gets here"); + ::close(releaseFd); + } + DrainCaptures(); + std::printf(" [%s] NOT_READY retries before the accepted submit: " + "%d (timing dependent; reported so a moving check count " + "can never be mistaken for a moving result)\n", + g_case, notReadyRetries); + Check(notReadyCloses == 0, + "no NOT_READY retry closed an fd the driver had taken", + I64(notReadyCloses) + " library closes across " + + I64(notReadyRetries) + " NOT_READY retries"); + DrainCaptures(); + return; + } + std::printf(" [%s] NOT_READY retries: %d (loop exhausted)\n", + g_case, notReadyRetries); + Check(notReadyCloses == 0, + "no NOT_READY retry closed an fd the driver had taken", + I64(notReadyCloses) + " library closes across " + + I64(notReadyRetries) + " NOT_READY retries"); + Check(false, "success path submitted", "never got past NOT_READY"); +} + +//============================================================================= +// Case 8 -- AN ARMED ACQUIRE FENCE, FRAME AFTER FRAME, against an unarmed +// control run through the identical code path. +// +// ASSERTED: every armed submit is accepted, closes the fd zero times, gets a +// release fence that signals, AND RETIRES like the control. That is the +// success-path contract repeated, so a fix that works once and leaks on the +// second frame is caught, and a fence that stalled assembly would be caught +// too. +// +// NO DrainPendingFrames() IN THIS FUNCTION, and that is a requirement of the +// measurement rather than a preference. The call is terminal for the +// completion surface: it reaches VkVideoEncoder::WaitForThreadsToComplete, +// which sets m_asyncAssemblyEnabled = false (VkVideoEncoder.cpp:4303) and +// joins the only threads that ever call PushCapturedBitstream; nothing turns +// it back on outside InitEncoder (:3054). A drain placed between the two +// loops therefore decides the outcome by POSITION: whichever loop runs after +// it can retire nothing, fence or no fence, and the fence stops being the +// variable under test. +// +// Swapping the loops with the drains left in moves every retirement to +// whichever loop ran first; removing the drains and keeping the order lets +// both loops retire. +// The variable is position relative to the first DrainPendingFrames(), not +// the fence. The drains are gone from this function accordingly, and the +// retirement claim is now asserted rather than excused. +// +// The defect itself is owned by the sibling test +// vk_video_encoder/test/encoder-ext-drain-assembly, which reproduces it with +// NO fence of any kind and is registered WILL_FAIL until it is fixed. +//============================================================================= +void CaseArmedFramesRepeatEdly() +{ + g_case = "ArmedFramesRepeated"; + const uint32_t kArmed = 8; + + // CONTROL FIRST: the identical frame with NO fence, so "did it retire?" + // has a same-session baseline rather than a remembered one. + const uint32_t capturedBeforeControl = g_captured; + uint32_t controlSubmitted = 0; + for (uint32_t f = 0; f < kArmed; f++) { + int releaseFd = -1; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFd; + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + VkVideoEncoderStatusCode status = + g_encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 500); + retry++) { + DrainCaptures(); + struct timespec ts = {0, 1000000}; + nanosleep(&ts, nullptr); + status = g_encoder->SubmitRegisteredFrame(info, nullptr); + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + break; + } + g_frameId++; + controlSubmitted++; + if (releaseFd >= 0) { + ::close(releaseFd); + } + DrainCaptures(); + } + // NO DrainPendingFrames() here. It is terminal for the completion surface + // (see the note above), so calling it between the two loops would decide + // this case's result before the fence had any say in it. Poll instead. + for (int i = 0; i < 500; i++) { + DrainCaptures(); + if ((g_captured - capturedBeforeControl) >= controlSubmitted) { + break; + } + struct timespec ts = {0, 2000000}; + nanosleep(&ts, nullptr); + } + const uint32_t controlRetired = g_captured - capturedBeforeControl; + + // NOW THE ARMED RUN. + const uint32_t capturedBeforeArmed = g_captured; + const int fdBefore = OpenFdCount(); + uint32_t armedSubmitted = 0; + uint32_t armedFencesSignalled = 0; + uint32_t armedLibraryCloses = 0; + for (uint32_t f = 0; f < kArmed; f++) { + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + break; + } + int releaseFd = -1; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + armedLibraryCloses += (uint32_t)r.closes; + if (r.status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + Check(false, "armed submit accepted", + "frame " + I64(f) + " status " + I64((long long)r.status)); + break; + } + g_frameId++; + armedSubmitted++; + if (releaseFd >= 0) { + if (PollSignalled(releaseFd, 10000) == 1) { + armedFencesSignalled++; + } + ::close(releaseFd); + } + DrainCaptures(); + } + // Same reason as the control loop: poll, never DrainPendingFrames(). + for (int i = 0; i < 500; i++) { + DrainCaptures(); + if ((g_captured - capturedBeforeArmed) >= armedSubmitted) { + break; + } + struct timespec ts = {0, 2000000}; + nanosleep(&ts, nullptr); + } + const uint32_t armedRetired = g_captured - capturedBeforeArmed; + const int fdAfter = OpenFdCount(); + + std::printf(" [%s] control(no fence): submitted=%u retired=%u\n", + g_case, controlSubmitted, controlRetired); + std::printf(" [%s] armed: submitted=%u retired=%u " + "fencesSignalled=%u libraryCloses=%u /proc/self/fd %d -> %d\n", + g_case, armedSubmitted, armedRetired, armedFencesSignalled, + armedLibraryCloses, fdBefore, fdAfter); + Check(armedSubmitted == kArmed, "every armed frame was accepted", + I64(armedSubmitted) + " of " + I64(kArmed)); + Check(armedLibraryCloses == 0, + "the library never closed an fd the driver had taken", + I64(armedLibraryCloses) + " closes across " + I64(armedSubmitted) + + " armed frames"); + Check(armedFencesSignalled == armedSubmitted, + "every armed frame's input-consuming submit completed", + I64(armedFencesSignalled) + " of " + I64(armedSubmitted) + + " release fences signalled"); + // THE CLAIM THIS CASE USED TO DUCK. An armed frame must come back through + // AcquireNextEncodedFrame like any other. Measured 16 retired for 8 armed + // submits on an A4000 (the extra 8 are the unarmed frames MintSyncFd + // submits to source each sync_fd), against 8 for the 8-frame control. + // The bar is therefore "at least as many as were submitted": a fence that + // stalled assembly would show 0 here, which is what the old note wrongly + // attributed to the fence when the cause was this function's own + // DrainPendingFrames() call. + Check(armedRetired >= armedSubmitted, + "armed frames retire, exactly like the unarmed control", + I64(armedRetired) + " retired for " + I64(armedSubmitted) + + " armed submits; control retired " + I64(controlRetired) + + " for " + I64(controlSubmitted)); + // ASSERTED, not merely printed. libraryCloses == 0 above is a + // ONE-SIDED test: it catches a success path that closes too much and is + // silent about one that closes too little. A future change that handed + // the driver an fd AND kept a copy -- or that recorded an fd in the guard + // and never consumed it on a path that then succeeded -- would leak eight + // descriptors here and pass every other assertion in this file. The fd + // table is the only instrument that can see that, so it has to be an + // assertion rather than a number in a log line. + // + // Equality is the right bar, not a bound: this loop mints kArmed fds and + // hands every one to the library, and closes every release fd it is given, + // so a correct run returns the table to exactly where it started. It has + // measured 88 -> 88 on both the fixed and the unfixed build. + Check(fdAfter == fdBefore, + "the success path left the fd table where it found it", + "/proc/self/fd " + I64(fdBefore) + " -> " + I64(fdAfter) + + ", delta " + I64(fdAfter - fdBefore) + + " -- libraryCloses==0 catches over-closing only; this is the " + "side that catches a leak"); +} + +//============================================================================= +// Case 9 -- THE LEAK, AT SCALE. One refusal is one fd; a session is millions. +// +// 120 refusals, each carrying a freshly minted real sync_fd, with the fd +// table read before and after. A single-call assertion cannot distinguish +// "closed" from "closed most of the time", and the fd table is the only +// instrument that sees an accumulating handle. The refusal chosen is +// RESOURCE_UNKNOWN because it is the one a producer actually hits in +// production -- an Unregister racing an in-flight submit, which is exactly +// when a caller is holding a fence fd. +// +// This is a REFUSAL loop rather than a success loop on purpose: an armed +// SUCCESS holds its fd inside a driver semaphore that lives until the frame +// retires, so a success loop's fd table would be measuring retirement timing +// rather than the leak. (An earlier version of this comment said armed frames +// "do not retire on this library". They do -- see the correction in case 8. +// The reason for preferring a refusal loop stands on its own.) +//============================================================================= +void CaseRefusalsAtScaleDoNotGrowTheFdTable() +{ + g_case = "RefusalsAtScale"; + DrainCaptures(); + const int fdBefore = OpenFdCount(); + + uint32_t iterations = 0; + uint32_t leaked = 0; + uint32_t wrongCloses = 0; + for (uint32_t f = 0; f < kRefusalLoopIterations; f++) { + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", + "MintSyncFd returned -1 at iteration " + I64(f)); + break; + } + int releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + info.resource = (VkVideoEncoderResource)0xDEADBEEFull; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + iterations++; + if (r.status != VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN) { + Check(false, "status", "iteration " + I64(f) + " status " + + I64((long long)r.status)); + break; + } + if (r.stillOpen) { + leaked++; + ::close(fd); // keep the loop from exhausting the table + } + if (r.closes != 1) { + wrongCloses++; + } + DrainCaptures(); + } + const int fdAfter = OpenFdCount(); + std::printf(" [%s] iterations=%u leaked=%u wrongCloseCount=%u " + "/proc/self/fd %d -> %d\n", + g_case, iterations, leaked, wrongCloses, fdBefore, fdAfter); + + Check(iterations == kRefusalLoopIterations, "the whole loop ran", + I64(iterations) + " of " + I64(kRefusalLoopIterations)); + Check(leaked == 0, "not one refusal leaked its fd", + I64(leaked) + " of " + I64(iterations) + " left the fd open"); + Check(wrongCloses == 0, "every refusal closed exactly once", + I64(wrongCloses) + " of " + I64(iterations) + + " closed a number of times other than one"); + Check((fdBefore >= 0) && (fdAfter >= 0) && (fdAfter <= fdBefore), + "no fd growth across the refusal loop", + "before " + I64(fdBefore) + ", after " + I64(fdAfter)); +} + +} // namespace + +// Both of case 10 timeline semaphores, on every exit. Two handles and five +// exits is how one of them comes to be leaked on the path nobody reran. +void DestroyCase10Semaphores(VkSemaphore callerWait, VkSemaphore callerSignal) +{ + if (callerWait != VK_NULL_HANDLE) { + g_fns.DestroySemaphore(g_device, callerWait, nullptr); + } + if (callerSignal != VK_NULL_HANDLE) { + g_fns.DestroySemaphore(g_device, callerSignal, nullptr); + } +} + +//============================================================================= +// Case 10 -- AN ARMED ACQUIRE FD *ALONGSIDE* A CALLER WAIT ARRAY. +// +// WHY THIS CASE EXISTS, and why it is the only one added rather than one per +// -1 site. Every other acquireFenceFd in this tree is -1: six sites in +// encoder-ext-sync and one in encoder-ext-release-fence. For all seven, -1 is +// the RIGHT input and was left alone -- the sync suite runs on a null-backend +// session with no device, so it cannot mint a real sync_fd at all and what it +// pins is the chained-descriptor WALK, not the import; and the release-fence +// loop is measuring the export half, which an armed acquire fence would only +// add noise to (case 8 above already covers armed frames exporting release +// fences, 8 for 8). +// +// One genuine gap survived that audit. The library APPENDS the imported +// acquire semaphore to whatever wait array the frame ended up with +// (vulkan_video_encoder_ext.cpp:6265-6281), and the header promises +// "Supplying an acquireFenceFd and a pWaitSemaphores array together is legal +// and loses neither". The copy loop that carries the caller's existing waits +// across that append (:6269-6272) runs ONLY when a fence is armed, and every +// armed submit in this file passes BaseInfo(), whose waitSemaphoreCount is 0. +// So the loop body has never executed once, anywhere in this tree: the append +// has only ever been tested appending to nothing. +// +// HOW IT IS MADE OBSERVABLE. Two signalled waits prove nothing -- the frame +// completes whether or not the caller's wait was carried across. So the +// caller's wait is a TIMELINE semaphore held at 0 and required at 1, and the +// acquire fence is a real, already-signalled sync_fd. The release fence then +// tells us which of two worlds we are in: +// +// not signalled while the timeline is at 0 => the caller's wait survived +// the append. ASSERTED. +// signalled after vkSignalSemaphore(1) => the acquire wait did not +// deadlock it. ASSERTED. +// +// WHAT THIS MEASURES: +// +// armed (acquireFenceFd = a real sync_fd) : earlyPoll=0 correct +// control(acquireFenceFd = -1, same frame, +// same caller timeline at 0) : earlyPoll=0 correct +// +// The claim HOLDS. This case is a gating assertion, not a known failure. +// +// A HAZARD FOR ANYONE RE-RUNNING THE MUTATION PROOF FOR THIS CASE. The proof +// forces the copy loop at vulkan_video_encoder_ext.cpp:6268 to zero +// iterations, which IS "the caller wait array is dropped when an acquire +// fence is armed", so the mutant runs red as it should. +// +// Reverting that mutant can restore a source file whose mtime is OLDER than +// the object already built from it. make then rebuilds nothing, and every run +// afterwards -- including the ones believed to be on clean source -- re-runs +// the mutant binary. Confirm the object is newer than the source before +// reading any result here as a property of the library, or a stale binary +// will be reported as one. +// +// Settled by measurement, not by argument: +// +// * The wait array reaches vkQueueSubmit2 intact. Traced with a temporary +// print at three points -- the ext layer immediately above +// SubmitExternalFrameCommon, StampExternalFrameInfo, and the +// VkSubmitInfo2 handed to MultiThreadedQueueSubmit in +// SubmitStagedInputFrame -- an armed frame carrying one caller TIMELINE +// wait submits waitSemaphoreInfoCount=2: {callerTimeline, value 1} and +// {importedAcquire, value 0}, both at TRANSFER. Nothing is dropped, +// truncated or overwritten anywhere at or below :6333. +// * Clean source: 63 consecutive runs of this case, earlyPoll=0 every +// time, 24 of them pinned to a single core to skew the timing. +// * Rebuilding the mutant reproduces the reported failure exactly, and +// reproduces a second fingerprint the reported failure also carried: a +// red run emits NO "asyncAssemblyFence ... is not done after N mSec" +// warning, because the frame is never parked; a green run emits exactly +// two, at 100 ms and 200 ms, because it is. Every archived red log has +// zero of them and every archived green log has two. +// +// So there is no read-before-write race here, and nothing in the library was +// changed to make this green. The case is KEPT, and promoted from WILL_FAIL +// to gating, because the mutation proof shows it bites: it is still the only +// test in this tree that executes that copy loop with anything to copy. +// +// THE SIGNAL DIRECTION IS COVERED HERE TOO, and for the same reason the wait +// direction was uncovered. The release-fence append on the signal side +// (:6316-6331) has the same shape as the acquire append on the wait side -- +// an append plus a loop that carries the caller array across it -- and every +// armed submit in this tree passed signalSemaphoreCount == 0, so ITS +// carry-across loop had never executed either. The frame therefore also +// carries a caller TIMELINE SIGNAL, and this case asserts that it is still at +// 0 while the frame is parked and reaches its requested value once the frame +// goes through. A library that discarded the caller signal array when it +// appended the release fence would leave that timeline at 0 forever -- +// whoever waits on it waits forever -- and would pass every other assertion +// in this file. Measured: it survives. +// +// LAST, on purpose: it parks a frame on an unsignalled wait for 200 ms, so it +// must not sit ahead of any loop that needs admission slots. +//============================================================================= +void CaseArmedFdAlongsideCallerWaitArray() +{ + g_case = "ArmedFdWithCallerWaitArray"; + + VkSemaphoreTypeCreateInfo typeInfo{VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO}; + typeInfo.semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + typeInfo.initialValue = 0; + VkSemaphoreCreateInfo semInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + semInfo.pNext = &typeInfo; + + VkSemaphore callerWait = VK_NULL_HANDLE; + const VkResult semRes = + g_fns.CreateSemaphore(g_device, &semInfo, nullptr, &callerWait); + Check(semRes == VK_SUCCESS, + "a host-signallable TIMELINE semaphore could be created", + "vkCreateSemaphore returned " + I64((long long)semRes) + + " -- without one this case cannot hold a caller wait open and " + "the append below would be untestable"); + if ((semRes != VK_SUCCESS) || (callerWait == VK_NULL_HANDLE)) { + return; + } + + // The mirror of |callerWait| on the other side of the frame: a TIMELINE + // the library must SIGNAL, held at 0 by construction, so that the + // release-fence append has a caller signal array to append TO. + VkSemaphore callerSignal = VK_NULL_HANDLE; + const VkResult sigSemRes = + g_fns.CreateSemaphore(g_device, &semInfo, nullptr, &callerSignal); + Check(sigSemRes == VK_SUCCESS, + "a readable TIMELINE semaphore could be created for the signal side", + "vkCreateSemaphore returned " + I64((long long)sigSemRes) + + " -- without one the SIGNAL half of this case cannot be observed"); + if ((sigSemRes != VK_SUCCESS) || (callerSignal == VK_NULL_HANDLE)) { + DestroyCase10Semaphores(callerWait, callerSignal); + return; + } + + const uint64_t kCallerWaitValue = 1; + const uint64_t kCallerSignalValue = 7; + int releaseFd = 0x5EED; + bool submitted = false; + int libCloses = -1; + + for (int attempt = 0; attempt < 64; attempt++) { + const int fd = MintSyncFd(); + if (fd < 0) { + Check(false, "mint a real sync_fd", "MintSyncFd returned -1"); + DestroyCase10Semaphores(callerWait, callerSignal); + return; + } + releaseFd = 0x5EED; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = fd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + // THE POINT OF THE CASE: a NON-EMPTY caller wait array, so the append + // has something to append TO. + info.waitSemaphoreCount = 1; + info.pWaitSemaphores = &callerWait; + info.pWaitSemaphoreValues = &kCallerWaitValue; + // The other half of the point: a NON-EMPTY caller SIGNAL array, so + // the release-fence append has something to append to as well. + info.signalSemaphoreCount = 1; + info.pSignalSemaphores = &callerSignal; + info.pSignalSemaphoreValues = &kCallerSignalValue; + + const CallResult r = RunWatched(g_encoder.get(), info, fd); + if (r.status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + DrainCaptures(); + struct timespec ts = {0, 2000000}; + nanosleep(&ts, nullptr); + continue; + } + libCloses = r.closes; + Check(r.status == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "an armed fd together with a caller wait array is accepted", + "got " + I64((long long)r.status) + ", want SUCCESS"); + submitted = (r.status == VK_VIDEO_ENCODER_STATUS_SUCCESS); + break; + } + if (!submitted) { + Check(false, "the frame was submitted", + "never got past NOT_READY, or the submit was refused"); + DestroyCase10Semaphores(callerWait, callerSignal); + return; + } + + Check(libCloses == 0, + "the library did not close what it handed to the driver", + I64(libCloses) + " close(2) calls on a descriptor " + "vkImportSemaphoreFdKHR already owns"); + Check(releaseFd >= 0, + "the input-consuming submit was issued", + "release fd " + I64(releaseFd)); + if (releaseFd < 0) { + DestroyCase10Semaphores(callerWait, callerSignal); + return; + } + + // (1) The caller's wait SURVIVED the append. The acquire fence is already + // signalled, so if the caller's timeline had been dropped there would be + // nothing left to hold this submit back and the release fence would be + // signalled by now. + const int earlyPoll = PollSignalled(releaseFd, 200); + Check(earlyPoll == 0, + "the caller wait array survived the acquire-fence append", + "release fence poll returned " + I64(earlyPoll) + + " while the caller timeline is still at 0 -- 1 means the " + "input-consuming submit ran anyway, i.e. the caller wait was " + "lost once an acquire fence was armed. The same frame with " + "acquireFenceFd = -1 polls 0 here, so the wait itself works; " + "arming the fence would be what lost it"); + + // (1b) The SIGNAL direction, read at the same instant and for the mirror + // reason. The frame is parked, so nothing in its batch has run and the + // caller timeline must still be at 0. A non-zero here is the same event + // the poll above names, seen through the other array. + uint64_t earlySignalValue = ~(uint64_t)0; + const VkResult earlyGet = g_fns.GetSemaphoreCounterValue( + g_device, callerSignal, &earlySignalValue); + Check((earlyGet == VK_SUCCESS) && (earlySignalValue == 0), + "the caller signal timeline has not moved while the frame is parked", + "vkGetSemaphoreCounterValue returned " + I64((long long)earlyGet) + + " value " + I64((long long)earlySignalValue) + ", want 0"); + + // (2) And nothing deadlocked: once the caller's wait is satisfied the + // frame goes through, which it could not do if the appended acquire + // semaphore were unsignalled or waited on twice. + VkSemaphoreSignalInfo signalInfo{VK_STRUCTURE_TYPE_SEMAPHORE_SIGNAL_INFO}; + signalInfo.semaphore = callerWait; + signalInfo.value = kCallerWaitValue; + const VkResult sigRes = g_fns.SignalSemaphore(g_device, &signalInfo); + Check(sigRes == VK_SUCCESS, "the caller's timeline could be signalled", + "vkSignalSemaphore returned " + I64((long long)sigRes)); + + const int latePoll = PollSignalled(releaseFd, 10000); + Check(latePoll == 1, + "with both waits satisfied the input-consuming submit completes", + "release fence poll returned " + I64(latePoll) + + " after the caller's timeline reached " + + I64((long long)kCallerWaitValue)); + + // (3) The caller SIGNAL array survived the release-fence append. That + // fence and this timeline ride the SAME VkSubmitInfo2, so once the fd is + // signalled the batch has retired and every signal in it has happened; + // the bounded retry below is for counter visibility only, never for + // ordering, and it cannot turn a dropped array into a pass because a + // dropped array leaves this at 0 for the whole second. + uint64_t lateSignalValue = 0; + for (int i = 0; i < 200; i++) { + if (g_fns.GetSemaphoreCounterValue(g_device, callerSignal, + &lateSignalValue) != VK_SUCCESS) { + break; + } + if (lateSignalValue >= kCallerSignalValue) { + break; + } + struct timespec ts = {0, 5000000}; + nanosleep(&ts, nullptr); + } + Check(lateSignalValue == kCallerSignalValue, + "the caller signal array survived the release-fence append", + "caller signal timeline reached " + I64((long long)lateSignalValue) + + ", want " + I64((long long)kCallerSignalValue) + + " -- a library that discarded the caller signal array when it " + "appended the release fence leaves this at 0 forever, and " + "whoever waits on it waits forever"); + + std::printf(" [%s] earlyPoll=%d (want 0) latePoll=%d (want 1) " + "callerSignal early=%llu late=%llu (want 0 then %llu) " + "libraryCloses=%d\n", + g_case, earlyPoll, latePoll, + (unsigned long long)earlySignalValue, + (unsigned long long)lateSignalValue, + (unsigned long long)kCallerSignalValue, libCloses); + + ::close(releaseFd); + DrainCaptures(); + // The frame must be released before the semaphore it waited on is + // destroyed; the drain above does that via ReleaseEncodedFrame. + g_encoder->DrainPendingFrames(); + DrainCaptures(); + DestroyCase10Semaphores(callerWait, callerSignal); +} + +int main(int argc, char** argv) +{ + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--caller-wait-order") == 0) { + g_assertCallerWaitOrder = true; + } + } + std::printf("Encoder-ext per-frame ACQUIRE fence fd ownership " + "(real device)%s\n", + g_assertCallerWaitOrder + ? " -- CALLER-WAIT-ORDER MODE (ordering gate only)" + : ""); + std::printf("------------------------------------------------\n"); + + if ((CreateVulkanVideoEncoderExt(g_encoder) != VK_SUCCESS) || !g_encoder) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + return 77; + } + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + config.encodeWidth = kWidth; + config.encodeHeight = kHeight; + config.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + config.inputWidth = kWidth; + config.inputHeight = kHeight; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 5000000; + config.maxBitrate = 5000000; + config.gopLength = 30; + // No reordering: the input-consuming submit is then issued inline with + // the frame that produced it, which is what makes the release fd this + // test mints from available on the same call. + config.consecutiveBFrames = 0; + config.idrPeriod = 30; + config.frameRateNum = 30; + config.frameRateDen = 1; + config.deviceId = -1; // -1 is auto-select; 0 names index 0 + config.disableFileOutput = VK_TRUE; + + if (g_encoder->InitializeExt(config) != VK_SUCCESS) { + std::printf("SKIP: InitializeExt failed -- no encode-capable Vulkan " + "device on this host\n"); + return 77; + } + + VkInstance instance = g_encoder->GetVkInstance(); + VkDevice device = g_encoder->GetVkDevice(); + VkPhysicalDevice phys = g_encoder->GetVkPhysicalDevice(); + DeviceFns& fns = g_fns; + g_device = device; + if (!LoadDeviceFns(instance, device, &fns)) { + std::printf("SKIP: could not load the Vulkan entry points needed\n"); + return 77; + } + + // Printed, not assumed. Nine ICDs are installed on the test host and a + // silent fall-through to a software driver would make every number below + // meaningless. + VkPhysicalDeviceProperties props{}; + fns.GetPhysicalDeviceProperties(phys, &props); + std::printf(" device: %s (vendor 0x%04X, driver 0x%08X)\n", + props.deviceName, props.vendorID, props.driverVersion); + + InputImage input; + if (!CreateInputImage(fns, phys, device, &input)) { + std::printf("SKIP: could not create the input image\n"); + return 77; + } + + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = kWidth; + desc.height = kHeight; + desc.tiling = VK_IMAGE_TILING_LINEAR; + desc.imageUsage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + desc.planeCount = 0; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.defaultLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + desc.existingImage = input.image; + + const VkVideoEncoderStatusCode regStatus = + g_encoder->RegisterImageResource(desc, 0, &g_resource, nullptr); + if ((regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (g_resource == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + std::printf("SKIP: RegisterImageResource failed, status %d\n", + (int)regStatus); + return 77; + } + + // The counter has to be live before the first assertion depends on it. + // + // WHAT THIS PROVES, EXACTLY, AND NO MORE: that the counting wrapper + // increments and that its RTLD_NEXT forward resolved. The ::close(probe) + // below is compiled in the SAME translation unit as the definition of + // close() above it, so the compiler binds it straight to that definition. + // It therefore CANNOT fail, and it says nothing whatever about whether + // the encoder archive's call sites bind here. It is a smoke test against + // a broken dlsym or a miscompiled counter -- which would report 0 closes + // for everything and read exactly like a total leak. + // + // What establishes that the ARCHIVE binds to this definition is the + // RED/GREEN differential of the suite as a whole: the same source, linked + // against the unfixed library, reports closes=0 on all six refusals; + // against the fixed one, closes=1, corroborated by fcntl -> EBADF and a + // /proc/self/fd delta of -1. A counter the library never reached could + // not move between those two builds. + { + const int probe = ::dup(2); + if (probe < 0) { + std::printf("SKIP: dup(2) failed\n"); + return 77; + } + g_closeCount.store(0, std::memory_order_relaxed); + g_watchedFd.store(probe, std::memory_order_relaxed); + ::close(probe); + g_watchedFd.store(-1, std::memory_order_relaxed); + const int seen = g_closeCount.load(std::memory_order_relaxed); + g_case = "InterposerSelfTest"; + Check(seen == 1, "the close(2) interposer is wired", + I64(seen) + " closes counted for one close -- every " + "closed-exactly-once assertion below depends on this"); + if (seen != 1) { + std::printf("RESULT: FAIL (interposer not wired; the rest would " + "be meaningless)\n"); + return 1; + } + } + + const int fdAtStart = OpenFdCount(); + std::printf(" /proc/self/fd at start: %d\n", fdAtStart); + + // WHY THE CASES RUN FROM A TABLE NOW: the per-case check LEDGER. + // + // "checks: N" was the only summary this file offered, and it is a weak + // signal in two directions. It moved run to run (fixed above), and -- + // worse -- a case that returned early on a setup failure, or that got + // dropped from this list, subtracted from it silently while the suite + // still reported PASS, because returning early fires no Check(). The + // floor below is per case, so "this case stopped exercising anything" + // now fails loudly and NAMES the case instead of showing up as a smaller + // number nobody was tracking. + // + // The floors are the assertion counts each case reaches today. They are + // FLOORS, not equalities: a case may add assertions without anybody + // having to update a magic number, and only a case that stops asserting + // what it already asserted trips the audit. + // + // ORDER MATTERS, and not for style. A frame with an armed acquire fence + // DOES retire on this library, which case 8 pins, so the ordering rests on + // admission slots and not on retirement. What has to hold is: + // - the 120-mint bulk loop needs a working submit path for every + // iteration, so it runs before anything that could tie up admission + // slots; + // - the two chain refusals refuse before the frame is admitted, so + // neither holds a slot and both belong up with the other refusals; + // - nothing in this file may call DrainPendingFrames() before teardown. + // That call is terminal for the completion surface, so a drain + // anywhere above would starve every later loop on NOT_READY and get + // reported as a leak. + struct CaseEntry { + const char* name; + void (*fn)(); + int minChecks; + }; + static const CaseEntry kCaseLedger[] = { + { "WrongTopLevelSType", CaseWrongTopLevelSType, 6 }, + { "NotInitialized", CaseNotInitialized, 6 }, + { "ResourceUnknown", CaseResourceUnknown, 6 }, + { "UnknownSTypeAheadOfFence", + CaseUnknownSTypeAheadOfFence, 6 }, + { "UnresolvableWaitIdAheadOfFence", + CaseUnresolvableWaitIdAheadOfFence, 6 }, + { "UnresolvableSignalIdAheadOfFence", + CaseUnresolvableSignalIdAheadOfFence, 6 }, + { "CyclicChainRefused", CaseCyclicChainIsRefused, 6 }, + { "DuplicateAcquireFd", CaseDuplicateAcquireFdIsRefused, 4 }, + { "RefusalsAtScale", CaseRefusalsAtScaleDoNotGrowTheFdTable, 4 }, + { "SuccessPathConsumesOnce", CaseSuccessPathConsumesOnce, 5 }, + { "ArmedFramesRepeated", CaseArmedFramesRepeatEdly, 5 }, + // LAST. It parks a frame on an unsignalled caller wait for 200 ms and + // ends with a DrainPendingFrames() of its own, so nothing that needs + // admission slots or the completion surface may follow it. + { "ArmedFdWithCallerWaitArray", + CaseArmedFdAlongsideCallerWaitArray, + kCallerWaitOrderFloor }, + }; + + // --caller-wait-order runs ONLY case 10, so that the ordering claim has + // a CTest entry that names it. The ledger table above is not walked in + // this mode, so its floor for that case is asserted directly here: a mode + // whose one case returned early would otherwise fire no Check() at all + // and exit 0, which is a gate that cannot fail. + if (g_assertCallerWaitOrder) { + const int beforeOrdering = g_checks; + CaseArmedFdAlongsideCallerWaitArray(); + const int orderingChecks = g_checks - beforeOrdering; + g_case = "CaseLedger"; + std::printf(" [ledger] %-34s %3d checks (floor %d)\n", + "ArmedFdWithCallerWaitArray", orderingChecks, + kCallerWaitOrderFloor); + Check(orderingChecks >= kCallerWaitOrderFloor, + "the ordering case contributed at least its floor of checks", + "contributed " + I64(orderingChecks) + ", floor " + + I64(kCallerWaitOrderFloor) + + " -- a case that returns early fires no Check() and would " + "otherwise leave this mode green"); + g_case = "teardown"; + g_encoder->UnregisterImageResource(g_resource); + fns.DestroyImage(device, input.image, nullptr); + fns.FreeMemory(device, input.memory, nullptr); + g_encoder = nullptr; + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; + } + + int subjectChecksBeforeLedger = 0; + for (const CaseEntry& entry : kCaseLedger) { + const int before = g_checks; + entry.fn(); + const int delta = g_checks - before; + subjectChecksBeforeLedger += delta; + std::printf(" [ledger] %-34s %3d checks (floor %d)\n", + entry.name, delta, entry.minChecks); + g_case = "CaseLedger"; + Check(delta >= entry.minChecks, + "the case contributed at least its floor of checks", + std::string(entry.name) + " contributed " + I64(delta) + + ", floor " + I64(entry.minChecks) + + " -- a case that returns early fires no Check() and would " + "otherwise leave the suite green"); + } + std::printf(" [ledger] %d subject checks across %d cases, " + "plus one ledger check per case\n", + subjectChecksBeforeLedger, + (int)(sizeof(kCaseLedger) / sizeof(kCaseLedger[0]))); + + g_case = "teardown"; + g_encoder->DrainPendingFrames(); + DrainCaptures(); + const int fdAtEnd = OpenFdCount(); + std::printf(" /proc/self/fd at end: %d\n", fdAtEnd); + std::printf(" captured frames=%u bitstream bytes=%llu\n", + g_captured, (unsigned long long)g_capturedBytes); + Check(g_captured > 0, "the encode produced frames", + I64(g_captured) + " captured"); + Check(g_capturedBytes > 0, "the encode produced a bitstream", + I64((long long)g_capturedBytes) + " bytes"); + + // Order matters and the header states it: unregister BEFORE destroying + // the image, and destroy the image BEFORE releasing the encoder. + g_encoder->UnregisterImageResource(g_resource); + fns.DestroyImage(device, input.image, nullptr); + fns.FreeMemory(device, input.memory, nullptr); + g_encoder = nullptr; + + std::printf("------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-adopt-device/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-adopt-device/CMakeLists.txt new file mode 100644 index 00000000..e973998a --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-adopt-device/CMakeLists.txt @@ -0,0 +1,248 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_adopt_device_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The PUBLIC API only, like the sibling real-device tests. The whole point of +# this one is that an EMBEDDER can drive the library through the shipped +# header; reaching an internal seam would weaken exactly the claim it makes. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # The descriptor API is an internal header: the public surface of this + # library is the encoder interface, and a test that drives the layer + # beneath it names the internal directory to say so. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# CTest semantics, matching the siblings: 0 every assertion held, 1 an +# assertion failed, 77 no encode-capable GPU and therefore nothing proved +# either way -- reported as SKIPPED, never as a pass. +enable_testing() + +configure_file(vk_layer_settings.txt + ${CMAKE_CURRENT_BINARY_DIR}/vk_layer_settings.txt COPYONLY) + +# -------------------------------------------------------------------------- +# VALIDATION GATING. +# +# Every entry below runs through validation_gate.cmake rather than calling the +# binary directly, because the binary CANNOT REPORT A VALIDATION FAILURE ITSELF. +# On the --own-validate arm the LIBRARY creates the VkInstance, so the harness +# has nothing to attach a VkDebugUtilsMessenger to, and the ext API exposes no +# message callback -- vulkan_video_encoder_ext.h's `VkBool32 validate` is a bare +# bool. Measured consequence: the layer reports 46 errors while the binary +# prints "PASSED : 5 checks, 0 failures" and exits 0. +# +# WHAT PINS WHAT. VK_LAYER_SETTINGS_PATH controls HOW MANY messages print; +# VK_LAYER_PATH controls WHETHER ANY DO. Only the second decides whether this +# suite can fail at all. +# +# READ THE DEFAULT HONESTLY: VVS_VALIDATION_LAYER_PATH is EMPTY unless set, and +# while it is empty VK_LAYER_PATH is NOT placed in the tests' ENVIRONMENT -- the +# ambient shell decides, and the validation arms skip on any machine that has +# not exported it. That is the configuration this ships in, and it is why these +# arms can still be green over a live defect. Set VVS_VALIDATION_LAYER_PATH to +# make a build self-contained, and VVS_REQUIRE_VALIDATION_LAYER=ON so a fleet +# cannot sit green purely because the layer was missing everywhere. +# +# SCOPE: this gates the 7 tests in THIS file. Other GPU tests in the tree are +# ungated and VVS_REQUIRE_VALIDATION_LAYER does not reach them. +# -------------------------------------------------------------------------- +# Matched against the MESSAGE BODY, not the VUID name -- the name cannot tell +# layer/header skew ("unknown VkStructureType") from a struct that is genuinely +# not permitted in that chain ("which is not allowed here"), and both arrive +# under VUID-Vk-pNext-pNext. +set(VVS_VALIDATION_ALLOW_VUIDS "unknown VkStructureType" CACHE STRING + "Regex matched against a validation message BODY; matches are tolerated") +option(VVS_REQUIRE_VALIDATION_LAYER + "Fail, rather than skip, the validation arms when no layer is installed" + OFF) +set(VVS_VALIDATION_LAYER_PATH "" CACHE PATH + "Directory holding a validation-layer manifest; becomes the tests' VK_LAYER_PATH") +set(VVS_VALIDATION_LAYER_LIBDIR "" CACHE PATH + "Directory added to the tests' LD_LIBRARY_PATH. The Vulkan SDK's layer manifest names a bare soname, so without this it loads only if the library is already on the default search path -- the reason an earlier --own-validate run skipped despite VK_LAYER_PATH being set.") + +set(_vvs_test_env + "VK_LAYER_SETTINGS_PATH=${CMAKE_CURRENT_BINARY_DIR}/vk_layer_settings.txt") +if(VVS_VALIDATION_LAYER_PATH) + list(APPEND _vvs_test_env "VK_LAYER_PATH=${VVS_VALIDATION_LAYER_PATH}") +endif() +if(VVS_VALIDATION_LAYER_LIBDIR) + # Replaces rather than prepends. Deliberate: it is opt-in, and a test whose + # loader path half-comes-from-the-ambient-shell is not reproducible. + list(APPEND _vvs_test_env "LD_LIBRARY_PATH=${VVS_VALIDATION_LAYER_LIBDIR}") +endif() + +# SKIP_REGULAR_EXPRESSION, not SKIP_RETURN_CODE. Measured on ctest 3.28.3: +# SKIP_RETURN_CODE beats FAIL_REGULAR_EXPRESSION, so a run that emitted +# validation errors on its way to a 77 exit was recorded as Skipped rather than +# Failed -- and main.cpp has two 77 returns that are reached after the layer is +# live. The gate script emits VALIDATION_GATE_RESULT=SKIP only on a path that +# has already established there were no validation errors, so the precedence +# cannot bite: the token does not exist on any failing path. +function(vvs_add_gated_test _name) + add_test(NAME ${_name} + COMMAND ${CMAKE_COMMAND} + -DTEST_EXE=$ + "-DTEST_ARGS=${ARGN}" + -DREQUIRE_LAYER=${VVS_REQUIRE_VALIDATION_LAYER} + "-DALLOW_VUIDS=${VVS_VALIDATION_ALLOW_VUIDS}" + "-DGATE_SUMMARY=${CMAKE_BINARY_DIR}/validation_gate_summary.txt" + -P ${CMAKE_CURRENT_SOURCE_DIR}/validation_gate.cmake) + set_tests_properties(${_name} PROPERTIES + LABELS "gpu" + TIMEOUT 900 + SKIP_REGULAR_EXPRESSION "VALIDATION_GATE_RESULT=SKIP" + ENVIRONMENT "${_vvs_test_env}") +endfunction() + +# --------------------------------------------------------------------------- +# THE SUBJECT. An ADOPT-mode session: the embedder owns the VkInstance and +# names the VkPhysicalDevice, the library creates its own VkDevice on them and +# encodes. This is the recommended embedding shape, and this file is the only +# test in the tree that touches externalInstance / externalPhysicalDevice / +# externalDevice. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtAdoptInstanceAndPhysicalDevice) + +# --------------------------------------------------------------------------- +# THE CONTROL, and it is what makes a red subject attributable. Identical +# config with NO external handles, i.e. the OWN path that ships today. If the +# subject is red and this is green, the host encodes fine and the difference +# is the adopted instance / physical device. If both are red, the machine is +# the story. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtAdoptOwnControl --own) + +# --------------------------------------------------------------------------- +# THE PIN. An adopted physical device PLUS a deviceId that names a different +# GPU. The library must refuse rather than quietly select something else -- +# PLAN:427 makes physical-device pinning REQUIRED once a device is named, +# and on a single-GPU host a refusal is the only observable that separates a +# real pin from an advisory one. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtAdoptPhysicalDevicePinIsEnforced --conflict) + +# --------------------------------------------------------------------------- +# THE CONTEXT PATH -- the same adoption expressed the way the context design +# requires. The context design splits the library's bring-up as "context = +# phases 1+2, session = phases 3+4", and the header's context rule 4 says a +# context creates no VkDevice. This case proves the split works end to end: an +# ADOPT context over the embedder's instance and physical device, a session +# created ON it with a config that names NO external handles at all, and a real +# encode. The instance/physical-device assertions further down the file are +# shared with the config-level ADOPT case, so a pass here means the context +# path borrowed exactly what the config path borrows. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtAdoptViaContext --context) + +# --------------------------------------------------------------------------- +# THE COLLISION. A session on a context AND a config that still names +# externalInstance/externalPhysicalDevice. The two express the same borrowing, +# so honouring either one silently is the failure: a stale config field +# outranking the context would send the session to a different physical device +# than the one whose capability snapshot the caller read, with nothing +# downstream reporting the substitution. It must be a typed refusal. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtContextRefusesConflictingConfigHandles --context-conflict) + + +# --------------------------------------------------------------------------- +# VALIDATION OVER A BORROWED INSTANCE. config.validate on an ADOPTED instance. +# The library never created that instance, so it cannot have enabled the +# validation layer or the debug-callback extension on it; it must check what +# the instance ACTUALLY has rather than what the loader offers. The embedder +# here enables the validation layer but deliberately NOT VK_EXT_debug_report, +# which is the realistic compositor shape. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtAdoptValidateOverBorrowedInstance --validate) + +# --------------------------------------------------------------------------- +# THE NEGATIVE CONTROL FOR THE ARM ABOVE. config.validate with the library +# owning the instance. The refusal to attach a debug callback is supposed to +# be scoped to IMPORTED instances; a change that stopped attaching callbacks +# altogether would turn the --validate arm green and could not be told apart +# from the real fix without this entry. +# +# THIS IS THE ONE ENTRY IN THE SUITE THAT NEEDS AN INSTALLED VALIDATION LAYER, +# because here the LIBRARY creates the instance and asks for +# VK_LAYER_KHRONOS_validation by name. Without it on the loader's search path +# the session cannot come up at all and the entry SKIPS (77) rather than +# pretending to prove something. Export VK_LAYER_PATH to a directory holding +# the layer manifest to make it run. The SUBJECT arm above deliberately does +# NOT need this -- it was rewritten to use a bare embedder instance precisely +# so it could not skip on the machine it guards. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtAdoptValidateOwnInstanceControl --own-validate) + +# --------------------------------------------------------------------------- +# THE TWO-CALL LIST QUERIES. VkVideoEncoderCapabilities carries scalars; every +# list is answered by its own pCount/pArray entry point, so the capacity +# belongs to the caller instead of to a constant in the header. This arm +# drives every branch of that idiom against a real snapshot: the counting +# call, the fetching call, the sentinel tail, the count's stability across the +# pair, VK_INCOMPLETE on a short buffer, and the refusal codes with a zero +# count. +# +# It also drives the input-format list against what the driver reports for two +# profiles of different bit depth, asserted as invariants of any answer rather +# than as this host's format list: each format once, direct entries first, at +# least one direct entry, the filtered entries present with the format each +# becomes, and the 12-bit pair advertised together or not at all. +# +# It builds a context and nothing else -- no session, no encode -- because the +# answers come from the construction-time snapshot. That is also what makes it +# cheap enough to run on every commit. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtContextEnumeratesLists --enumerate) + +message(STATUS "encoder_ext_adopt_device_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-adopt-device/src/main.cpp b/vk_video_encoder/test/encoder-ext-adopt-device/src/main.cpp new file mode 100644 index 00000000..d23aba1d --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-adopt-device/src/main.cpp @@ -0,0 +1,1505 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * ADOPT-MODE SESSION: the library creates its OWN VkDevice on an instance and + * a physical device the EMBEDDER already owns. + * + * WHY THIS TEST EXISTS. Every external-handle field this API has -- + * externalInstance / externalPhysicalDevice / externalDevice and the two queue + * family indices -- had ZERO coverage anywhere in the tree. A grep for + * externalDevice under vk_video_encoder/test returned nothing, so the only + * thing exercising them was the Chromium embedder, out-of-tree, in the + * configuration where all three handles are supplied. + * + * WHAT ADOPT IS, and how it differs from the configuration next to it: + * + * IMPORT externalInstance + externalPhysicalDevice + externalDevice. + * The library encodes on the EMBEDDER's logical device and binds the + * embedder's queue families. + * + * ADOPT externalInstance + externalPhysicalDevice, externalDevice NULL. + * The library BORROWS the instance and is PINNED to the physical + * device, then runs vkCreateDevice itself and probes its own queue + * families. This is the recommended shape: the embedder hands over + * identity, never a logical device. + * + * OWN no external handles at all. Library creates everything. Unchanged, + * and run here as the control. + * + * THE FOUR THINGS THIS PROVES, none of which a device-free test can: + * + * 1. IDENTITY. GetVkInstance() is the embedder's instance, + * GetVkPhysicalDevice() is the embedder's physical device, and + * GetVkDevice() is NEITHER null nor anything the embedder handed over -- + * there is no embedder device to confuse it with, because this test + * deliberately never creates one. + * + * 2. IT ENCODES. Registration, submit and bitstream capture over a real + * queue on the library-created device. An ADOPT path that stands a device + * up and then cannot encode on it is not an embedding path at all. + * + * 3. THE PIN IS LOAD-BEARING. --conflict supplies the physical device AND a + * deviceId that names a different GPU. The library must REFUSE, not + * silently select some other enumerated device. PLAN:427 makes + * physical-device pinning REQUIRED once a device is named; a pin that + * falls back is not a pin, and on a single-GPU host refusal is the only + * observable that distinguishes the two. + * + * 4. VALIDATION OVER A BORROWED INSTANCE. --validate asks for + * config.validate on an ADOPTED instance. The library must not assume the + * borrowed instance enabled the layers and instance extensions that the + * LOADER merely offers -- it never created that instance and cannot have + * enabled anything on it. See the note in InitVulkanDevice. + * + * WHY THE EMBEDDER HERE NEVER CREATES A VkDevice. If it did, this test could + * pass on a library that silently ignored externalDevice==NULL and reused some + * device of the caller's -- and it would also stop being a test of the shape + * the user actually asked for ("the library always creates its OWN VkDevice"). + * Every device-level call this file makes goes through GetVkDevice(), i.e. + * through the device the LIBRARY created. That is deliberate: it is what makes + * "the library stood up a working device on a borrowed physical device" + * observable at all. + * + * Exits 77 (CTest SKIP) with no encode-capable Vulkan device, like its + * siblings. 0 all assertions held, 1 an assertion failed. + */ + +#include "vulkan_video_encoder_ext.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + std::printf(" ok %s\n", what); + return; + } + g_failures++; + std::printf(" FAIL %s : %s\n", what, detail.c_str()); +} + +std::string U64Hex(unsigned long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "0x%llx", v); + return buf; +} + +std::string I64(long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%lld", v); + return buf; +} + +const uint32_t kWidth = 1280; +const uint32_t kHeight = 720; +const uint32_t kFrames = 24; + +// --------------------------------------------------------------------------- +// THE EMBEDDER. A VkInstance and a VkPhysicalDevice, and deliberately NOTHING +// else -- this stands in for a compositor that has a Vulkan implementation up +// but is not going to lend its logical device to anyone. +// --------------------------------------------------------------------------- +struct Embedder { + void* lib = nullptr; + PFN_vkGetInstanceProcAddr gipa = nullptr; + VkInstance instance = VK_NULL_HANDLE; + VkPhysicalDevice phys = VK_NULL_HANDLE; + uint32_t deviceID = 0; + char name[VK_MAX_PHYSICAL_DEVICE_NAME_SIZE] = {}; + + PFN_vkDestroyInstance DestroyInstance = nullptr; +}; + +bool BuildEmbedder(Embedder* e, bool withValidationLayer) +{ + e->lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (e->lib == nullptr) { + e->lib = dlopen("libvulkan.so", RTLD_NOW); + } + if (e->lib == nullptr) { + std::printf(" SKIP-CAUSE: dlopen(libvulkan) failed: %s\n", dlerror()); + return false; + } + e->gipa = (PFN_vkGetInstanceProcAddr)dlsym(e->lib, + "vkGetInstanceProcAddr"); + if (e->gipa == nullptr) { + std::printf(" SKIP-CAUSE: no vkGetInstanceProcAddr\n"); + return false; + } + + auto createInstance = + (PFN_vkCreateInstance)e->gipa(VK_NULL_HANDLE, "vkCreateInstance"); + if (createInstance == nullptr) { + std::printf(" SKIP-CAUSE: no vkCreateInstance\n"); + return false; + } + + // A BARE instance by default: no layers, no instance extensions. + // Everything the library needs off an adopted instance -- + // GetPhysicalDeviceProperties2, GetPhysicalDeviceQueueFamilyProperties2, + // GetPhysicalDeviceFeatures2, the external-memory capability queries -- is + // Vulkan 1.1 core, so a bare 1.3 instance is a legitimate embedder. It is + // also the STRICTEST embedder: anything the library assumes an adopted + // instance enabled for it fails here rather than being masked by a richer + // instance. That is exactly what the --validate arm is for. + VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO}; + app.pApplicationName = "EncoderExtAdoptEmbedder"; + app.apiVersion = VK_API_VERSION_1_3; + + VkInstanceCreateInfo ici{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO}; + ici.pApplicationInfo = &app; + + // THE EMBEDDER INSTANCE IS BARE ON EVERY ARM, INCLUDING --validate, and + // that is deliberate rather than lazy. + // + // An earlier draft enabled VK_LAYER_KHRONOS_validation here so the + // --validate arm would have a layer to report through. The cost was that + // the arm SKIPPED on any host where the layer is not on the loader's + // search path -- which is every CI runner and, as it turned out, this + // project's own GPU host unless VK_LAYER_PATH is exported by hand. A + // red-then-green test that skips on the machine it guards proves nothing. + // + // It is also unnecessary. What --validate exercises is the LIBRARY's + // reaction to `config.validate` over an instance IT DID NOT CREATE, and a + // bare instance is the sharpest version of that: it has no layer and no + // debug extension, so every assumption the library might make about what + // the loader offers is false of this instance. Both failure modes the fix + // addresses reproduce on it, and which one fires depends only on whether + // the host happens to have the layer installed: + // * layer present in the LOADER -- the pre-fix loader-level checks pass, + // and InitDebugReport walks into a null dispatch entry: SIGSEGV. + // * layer absent from the loader -- the pre-fix CheckAllInstanceLayers + // fails the whole session with VK_ERROR_LAYER_NOT_PRESENT because the + // LOADER lacks a layer that has nothing to do with the borrowed + // instance. + // Either way the arm is red before the fix and green after, on any host. + // + // The one arm that genuinely needs an installed layer is the + // --own-validate control, where the LIBRARY creates the instance and asks + // for the layer by name; that one skips (77) without it, which is correct + // -- it is a control, and a missing layer really does make it unable to + // say anything. + (void)withValidationLayer; + + if (createInstance(&ici, nullptr, &e->instance) != VK_SUCCESS) { + std::printf(" SKIP-CAUSE: vkCreateInstance failed\n"); + return false; + } + e->DestroyInstance = + (PFN_vkDestroyInstance)e->gipa(e->instance, "vkDestroyInstance"); + + auto enumPhys = (PFN_vkEnumeratePhysicalDevices)e->gipa( + e->instance, "vkEnumeratePhysicalDevices"); + auto getProps = (PFN_vkGetPhysicalDeviceProperties)e->gipa( + e->instance, "vkGetPhysicalDeviceProperties"); + auto getQueues = (PFN_vkGetPhysicalDeviceQueueFamilyProperties)e->gipa( + e->instance, "vkGetPhysicalDeviceQueueFamilyProperties"); + if ((enumPhys == nullptr) || (getProps == nullptr) || + (getQueues == nullptr)) { + std::printf(" SKIP-CAUSE: missing instance-level entry points\n"); + return false; + } + + uint32_t count = 0; + enumPhys(e->instance, &count, nullptr); + if (count == 0) { + std::printf(" SKIP-CAUSE: no physical devices\n"); + return false; + } + std::vector devices(count); + enumPhys(e->instance, &count, devices.data()); + + // The embedder picks the first ENCODE-CAPABLE device, which is the choice + // a compositor would make on the caller's behalf. + for (uint32_t i = 0; i < count; i++) { + uint32_t qcount = 0; + getQueues(devices[i], &qcount, nullptr); + if (qcount == 0) { + continue; + } + std::vector qprops(qcount); + getQueues(devices[i], &qcount, qprops.data()); + bool encodes = false; + for (uint32_t q = 0; q < qcount; q++) { + if ((qprops[q].queueFlags & VK_QUEUE_VIDEO_ENCODE_BIT_KHR) != 0) { + encodes = true; + break; + } + } + if (!encodes) { + continue; + } + VkPhysicalDeviceProperties props{}; + getProps(devices[i], &props); + e->phys = devices[i]; + e->deviceID = props.deviceID; + std::snprintf(e->name, sizeof(e->name), "%s", props.deviceName); + return true; + } + std::printf(" SKIP-CAUSE: no encode-capable physical device\n"); + return false; +} + +void TearDownEmbedder(Embedder* e) +{ + if ((e->instance != VK_NULL_HANDLE) && (e->DestroyInstance != nullptr)) { + e->DestroyInstance(e->instance, nullptr); + e->instance = VK_NULL_HANDLE; + } +} + +// --------------------------------------------------------------------------- +// Device-level entry points, resolved off the LIBRARY's device. +// --------------------------------------------------------------------------- +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = + nullptr; +}; + +bool LoadDeviceFns(PFN_vkGetInstanceProcAddr gipa, VkInstance instance, + VkDevice device, DeviceFns* fns) +{ + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + std::printf(" ERROR: no vkGetDeviceProcAddr\n"); + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr); +} + +struct InputImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; +}; + +// A host-written LINEAR NV12 image, on the LIBRARY's device, registered as +// TRANSFER_SRC so the registration routes STAGED. Staged is the right arm +// here: it is the one that copies through a real queue on the device under +// test, so a device that came up but cannot actually submit work shows up. +bool CreateInputImage(const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, InputImage* out) +{ + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = VK_IMAGE_TILING_LINEAR; + ci.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + std::printf(" ERROR: vkCreateImage(LINEAR NV12) failed\n"); + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + const VkMemoryPropertyFlags want = + (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + std::printf(" ERROR: no host-visible memory type\n"); + // Same lifetime rule as the registration exit in EncodeAndDrain: the + // image already exists on the library's device, and main() answers + // this false by resetting the encoder, which destroys that device. A + // live VkImage at that point is a vkDestroyDevice VUID violation -- + // and unlike the registration exit, THIS one legitimately stays a + // skip, so the violation would ride out on a green run. + fns.DestroyImage(device, out->image, nullptr); + out->image = VK_NULL_HANDLE; + out->memory = VK_NULL_HANDLE; + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + std::printf(" ERROR: vkAllocateMemory failed\n"); + fns.DestroyImage(device, out->image, nullptr); + out->image = VK_NULL_HANDLE; + out->memory = VK_NULL_HANDLE; + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != VK_SUCCESS) { + std::printf(" ERROR: vkBindImageMemory failed\n"); + // Image then memory, matching the success path below. The reverse is + // equally legal -- vkFreeMemory has no ordering constraint against + // vkDestroyImage even for bound memory, only against submitted work -- + // so there is nothing to justify, and uniformity is worth more. + fns.DestroyImage(device, out->image, nullptr); + fns.FreeMemory(device, out->memory, nullptr); + out->image = VK_NULL_HANDLE; + out->memory = VK_NULL_HANDLE; + return false; + } + + // Real content, not zeros: a flat surface encodes to a degenerate + // bitstream and would make "the encode actually ran" hard to assert. + void* mapped = nullptr; + if (fns.MapMemory(device, out->memory, 0, req.size, 0, &mapped) == + VK_SUCCESS) { + uint8_t* bytes = (uint8_t*)mapped; + for (VkDeviceSize i = 0; i < req.size; i++) { + bytes[i] = (uint8_t)((i * 7u) ^ (i >> 9)); + } + fns.UnmapMemory(device, out->memory); + } + return true; +} + +void FillConfig(VkVideoEncoderConfig* config) +{ + config->sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config->codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + config->encodeWidth = kWidth; + config->encodeHeight = kHeight; + config->inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + config->inputWidth = kWidth; + config->inputHeight = kHeight; + config->rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config->averageBitrate = 5000000; + config->maxBitrate = 5000000; + config->gopLength = 30; + config->consecutiveBFrames = 0; + config->idrPeriod = 30; + config->frameRateNum = 30; + config->frameRateDen = 1; + config->deviceId = -1; + config->disableFileOutput = VK_TRUE; +} + +// Encode kFrames and report how many bitstream payloads came back. +// Returns false only on an infrastructure failure (which is a SKIP cause); +// assertion failures go through Check(). +bool EncodeAndDrain(VkSharedBaseObj& encoder, + const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, uint32_t* framesOut, uint64_t* bytesOut) +{ + InputImage input; + if (!CreateInputImage(fns, phys, device, &input)) { + return false; + } + + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = kWidth; + desc.height = kHeight; + desc.tiling = VK_IMAGE_TILING_LINEAR; + desc.imageUsage = (VkImageUsageFlags)VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + desc.planeCount = 0; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.defaultLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + desc.existingImage = input.image; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode regStatus = + encoder->RegisterImageResource(desc, 0, &resource, nullptr); + if ((regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (resource == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + // AN ASSERTION, NOT A SKIP. + // + // A refusal reaching this line is a LIBRARY answer, not a HOST + // condition, so it must not take `return false`: this function's + // contract maps that to "infrastructure failure" and main() maps it to + // `return 77` -- ctest Skipped, which does not fail a run, and a + // library refusal that does not fail a run is not asserted at all. + // FIVE of the seven registered + // arms reach this line (the two --*conflict arms refuse earlier), so + // a registration regression could turn this whole suite green by + // skipping it, and an ADOPT-only regression would skip the three + // adoption arms while the two OWN controls stayed green -- destroying + // the exact comparison this test exists to make. + // + // The host-capability question was already settled ABOVE, and that is + // what makes the reclassification honest rather than merely stricter: + // CreateInputImage builds this very LINEAR NV12 image on this very + // device, and ITS failure is still the skip. By the time control + // reaches this line the device has demonstrably created the image we + // are handing straight back to the library. + // + // ON THIS DESCRIPTOR THE REGISTRATION MAKES NO DEVICE CALL AT ALL, + // which is what makes the reclassification safe rather than merely + // defensible. BuildRegisteredViewLocked looks like the device-facing + // step and is not, for this input: CreateFromExternal and + // VulkanVideoImagePoolNode::CreateExternal are plain allocations with + // no Vulkan entry point, and the combined image view is SKIPPED -- + // VkImageResource declines to create a view over an image whose usage + // holds no view-compatible bit (VUID-VkImageViewCreateInfo-image-04441) + // and this descriptor declares VK_IMAGE_USAGE_TRANSFER_SRC_BIT alone, + // while the per-plane views additionally need + // VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT, which it does not set. So every + // refusal reachable from here -- descriptor validation, the 4096-slot + // resource limit, a null existingImage -- is decided in library code + // before the driver is touched. + // + // ONE GENUINE DEVICE-DEPENDENT REFUSAL EXISTS, and it is named rather + // than hidden: the descriptor extent must cover the session's coded + // extent, and InitializeExt clamps encodeWidth/encodeHeight UP to + // videoCapabilities.minCodedExtent. A device reporting a minCodedExtent + // larger than this test's image would raise the session above it and + // refuse EXTENT_INVALID. No real encode hardware does; if one ever + // does, this line is where it will announce itself, which is better + // than a silent skip. + // + // The same file already makes this argument three times, for context + // creation, for CreateVulkanVideoEncoderExtOnContext, and for + // InitializeExt(ADOPT). This site APPLIES IT MORE BROADLY than any of + // them, which is worth saying rather than leaving to be noticed: the + // first two exist only on the --context* arms, and the third + // explicitly carves the OWN arms back out into a skip. This line is + // reached by --own and --own-validate too. That widening is + // deliberate -- by here InitializeExt has already SUCCEEDED on every + // arm, so "can this host encode?" is settled and a refusal can no + // longer be answering it. + // Status only: the resource half would be a constant. |resource| is + // initialised to VK_VIDEO_ENCODER_RESOURCE_NULL right here in the + // test, and a SUCCESS return can never overwrite it with 0 + // (MakeResourceId ORs index + 1), so inside this branch it is always + // 0 and reporting it says nothing. (The library also zeroes + // *outResource itself, but only after three earlier returns, so that + // is not what makes this safe -- the initialiser above is.) + Check(false, "RegisterImageResource(VK_IMAGE)", + "status " + I64((long long)regStatus)); + + // Unwind the input image on this exit too. InputImage is a plain POD + // with no destructor and only the success path at the bottom of this + // function frees it, so the old `return false` leaked both handles -- + // and then main() reset the encoder, destroying the library-created + // VkDevice while one of its images was still live. That is a + // vkDestroyDevice VUID violation (every child object must be + // destroyed first). Only --own-validate could ever have reported it: + // BuildEmbedder is called with withValidationLayer=false on every arm + // including --validate, deliberately, so the borrowed instance carries + // no layer and only the arm where the LIBRARY creates the instance has + // one live. + if (input.image != VK_NULL_HANDLE) { + fns.DestroyImage(device, input.image, nullptr); + } + if (input.memory != VK_NULL_HANDLE) { + fns.FreeMemory(device, input.memory, nullptr); + } + *framesOut = 0; + *bytesOut = 0; + // TRUE: the contract this function documents is "false means SKIP", + // and this is not one. main() goes on to fail `frames > 0` and + // `bytes > 0` as well, which is accurate -- nothing encoded. + return true; + } + + uint32_t captured = 0; + uint64_t bytes = 0; + uint32_t submitted = 0; + for (uint32_t f = 0; f < kFrames; f++) { + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = resource; + info.frameId = f; + info.pts = f; + info.qpOverride = -1; + info.currentLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + VkVideoEncoderStatusCode status = + encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 2000); + retry++) { + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + } + status = encoder->SubmitRegisteredFrame(info, nullptr); + if (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + struct timespec ts = {0, 1000000}; // 1 ms + nanosleep(&ts, nullptr); + } + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + Check(false, "SubmitRegisteredFrame", + "frame " + I64(f) + " status " + I64((long long)status)); + break; + } + submitted++; + + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + } + } + + // Flush whatever is still in flight. + for (int spin = 0; (spin < 4000) && (captured < submitted); spin++) { + VkVideoEncodeResult drained; + bool got = false; + while (encoder->AcquireNextEncodedFrame(drained) == VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + got = true; + } + if (!got) { + struct timespec ts = {0, 1000000}; + nanosleep(&ts, nullptr); + } + } + + // THEN THE REAL DRAIN, because the poll above is only a poll. It is + // bounded at 4000 spins and can exit with captured < submitted, and the + // handles freed below are read by submitted work: destroying the image + // while a frame still references it violates + // VUID-vkDestroyImage-image-01000, and freeing its memory violates + // VUID-vkFreeMemory-memory-00677 -- both, not just the first. + // + // DrainPendingFrames is the library's own answer to exactly this and the + // test simply never called it. It flushes the deferred GOP tail and waits + // for in-flight encodes WITHOUT releasing the encoder, by joining the + // encoder-queue and assembly threads, so it needs no vkDeviceWaitIdle -- + // which DeviceFns does not load -- and no external synchronisation care + // around the library's own submitting thread. It is documented threading + // class (a), which this call site satisfies: single-threaded, not inside + // a completion callback. + // + // It can release frames the poll never saw, so collect once more + // afterwards -- otherwise this change would silently lower the frame + // count it is meant to make trustworthy. + Check(encoder->DrainPendingFrames() == VK_SUCCESS, "DrainPendingFrames", + "the encoder could not flush its in-flight work"); + { + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + } + } + + // UNREGISTER BEFORE DESTROYING, which is the order the public header + // requires in as many words: "the caller must Unregister BEFORE + // destroying it -- a driver may recycle the handle value, and a surviving + // registration would then name freed memory". This call was simply + // missing; the test destroyed the image with the registration still live. + // + // NOT a bare call: this whole change exists because a library status was + // being thrown away, so throwing this one away would be the same mistake + // one screen further down. Unregister answers RESOURCE_UNKNOWN if the id + // does not resolve, which is precisely the generation/slot regression + // worth catching. + // + // Its detection power is bounded and the bound is worth knowing: the + // deferred arm sets slot.retired without clearing slot.live, so a lookup + // still succeeds and even a repeat Unregister answers SUCCESS -- which + // does not match the header's "The id is invalid immediately on return". + // That is a pre-existing library/header mismatch, not this test's to fix, + // but it is why this assertion proves the id RESOLVED, not that the slot + // retired. + // + // No null guard: |resource| is initialised non-null above and the only + // branch that could leave it null already returned, so a guard here would + // just tell a reader that null is reachable when it is not. + Check(encoder->UnregisterImageResource(resource) == + VK_VIDEO_ENCODER_STATUS_SUCCESS, + "UnregisterImageResource", "the registration id did not resolve"); + if (input.image != VK_NULL_HANDLE) { + fns.DestroyImage(device, input.image, nullptr); + } + if (input.memory != VK_NULL_HANDLE) { + fns.FreeMemory(device, input.memory, nullptr); + } + + *framesOut = captured; + *bytesOut = bytes; + return true; +} + +} // namespace + + +// --------------------------------------------------------------------------- +// THE TWO-CALL LIST QUERIES ON A CONTEXT. +// +// A list capability is answered by its own entry point rather than by a +// fixed-capacity member of VkVideoEncoderCapabilities, so the capacity is the +// caller's. What has to hold for that to be usable is asserted here: the +// counting call answers without writing, the fetching call writes exactly what +// it was told it could, a caller that asks for less is TOLD it got less, and a +// pair the library does not probe answers a code with a count of zero rather +// than a stale number. +// +// THE SNAPSHOT IS WHAT MAKES THE IDIOM SAFE HERE, and it is asserted rather +// than assumed: a context is immutable after construction, so the counting +// call and the fetching call cannot disagree. The re-count below is that +// claim; without it the pair of calls would carry the usual Vulkan +// retry-until-complete obligation. +// +// THE ADVERTISED INPUT FORMATS, against a real driver answer. +// +// The device-free suite drives the advertisement's membership rule with +// synthetic device lists. What only a device can supply is that the rule is +// applied to what the driver actually reports for a profile, and that the +// result is self-consistent -- so the assertions here are stated as +// invariants of ANY answer rather than as a format list this one host +// produces. +void CheckInputFormatEnumerator(VulkanVideoEncoderContext* ctx, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, const char* label) +{ + // THE COUNTING CALL. A null array is not an error; it is the question. + uint32_t counted = 0xFFFFFFFFu; + VkResult r = VkEncEnumerateInputFormats(ctx, 0u, codec, profile, &counted, + nullptr); + if (r != VK_SUCCESS) { + // A profile this device does not expose is not a failure of the + // enumerator; it is the device's answer, and the count still has to + // be written. + Check(counted == 0u, + "a profile the device refuses reports a count of zero", + std::string(label) + " counted " + I64((long long)counted)); + std::printf(" %s: not advertised (%lld)\n", label, (long long)r); + return; + } + Check(counted > 0u, + "an advertised profile advertises at least one input format", + std::string(label) + " counted 0"); + + // THE FETCHING CALL, over a buffer of the caller's choosing. The sentinel + // tail is what proves the library wrote what it said and not one entry + // more. + const uint32_t kCapacity = 32u; + VkVideoEncoderInputFormatProperties entries[kCapacity]; + for (uint32_t i = 0; i < kCapacity; i++) { + entries[i].format = VK_FORMAT_UNDEFINED; + entries[i].encodeFormat = VK_FORMAT_UNDEFINED; + entries[i].optimality = VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL; + } + uint32_t fetched = kCapacity; + r = VkEncEnumerateInputFormats(ctx, 0u, codec, profile, &fetched, entries); + Check((r == VK_SUCCESS) && (fetched == counted), + "the fetching call writes what the counting call promised", + std::string(label) + ": counted " + I64((long long)counted) + + ", wrote " + I64((long long)fetched)); + if (fetched > kCapacity) { + return; + } + if (fetched < kCapacity) { + Check(entries[fetched].format == VK_FORMAT_UNDEFINED, + "nothing is written past the reported count", + std::string(label) + ": the entry after the last one was " + "written"); + } + + // THE SNAPSHOT CLAIM. The count cannot move between the two calls, + // because the answer was computed before either of them. + uint32_t recounted = 0xFFFFFFFFu; + r = VkEncEnumerateInputFormats(ctx, 0u, codec, profile, &recounted, + nullptr); + Check((r == VK_SUCCESS) && (recounted == counted), + "the count is stable across calls -- the snapshot cannot move", + std::string(label) + ": first " + I64((long long)counted) + + ", then " + I64((long long)recounted)); + + std::printf(" %s advertises %u input format(s):\n", label, fetched); + uint32_t direct = 0; + uint32_t converted = 0; + bool orderHeld = true; + for (uint32_t i = 0; i < fetched; i++) { + const VkVideoEncoderInputFormatProperties& e = entries[i]; + const bool isDirect = + (e.optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL); + std::printf(" %d -> %d%s\n", (int)e.format, (int)e.encodeFormat, + isDirect ? " (direct)" : " (filtered)"); + Check(!isDirect || (e.format == e.encodeFormat), + "a direct entry names itself as what it is encoded as", + std::string(label) + ": format " + I64((long long)e.format) + + " names " + I64((long long)e.encodeFormat)); + if (isDirect) { + direct++; + // Ordering contract: every entry the device takes unconverted + // precedes every filtered one, so a caller reading top-down sees + // the free tier first. + orderHeld = orderHeld && (converted == 0); + } else { + converted++; + } + for (uint32_t j = 0; j < i; j++) { + Check(entries[j].format != e.format, + "each format is advertised once, however many tilings the " + "device reports it at", + std::string(label) + ": format " + + I64((long long)e.format) + " twice"); + } + } + Check(orderHeld, "the direct entries come before the filtered ones", + std::string(label) + ": a direct entry follows a filtered one"); + Check(direct > 0u, + "at least one advertised format is taken by the device unconverted", + std::string(label) + ": no direct entry"); + + // THE SUBSTANCE OF THE ADVERTISEMENT. A list that were the device's own + // VIDEO_ENCODE_SRC set reduced to what the library routes would carry + // direct entries only -- no driver reports an RGBA encode source. An RGB + // producer learns from this list that it may hand RGBA over, and what + // that becomes. + bool rgbaAdvertised = false; + for (uint32_t i = 0; i < fetched; i++) { + const VkVideoEncoderInputFormatProperties& e = entries[i]; + if (e.format == VK_FORMAT_R8G8B8A8_UNORM) { + rgbaAdvertised = true; + Check((e.optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL) && + (e.encodeFormat != VK_FORMAT_R8G8B8A8_UNORM), + "RGBA8 is advertised as FILTERED, and names what it becomes", + std::string(label) + ": it claims to be direct"); + } + } + Check(rgbaAdvertised && (converted > 0u), + "the list carries the formats the library CONVERTS, not only the " + "ones the device takes", + std::string(label) + ": " + I64((long long)converted) + + " filtered entries, RGBA8 " + + (rgbaAdvertised ? "present" : "absent")); + + // THE 12-BIT PAIR, stated as an implication rather than as a fact about + // this host. I420-12 and P012 both convert into P012, so either both are + // advertised -- where the device takes P012 as an encode source -- or + // neither is. + bool i420_12 = false; + bool p012 = false; + for (uint32_t i = 0; i < fetched; i++) { + const VkFormat fmt = entries[i].format; + i420_12 = i420_12 || + (fmt == VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16); + p012 = p012 || + (fmt == VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16); + } + Check(i420_12 == p012, + "I420-12 and P012 are advertised together: they share a conversion " + "target, so one device answer decides both", + std::string(label) + ": I420-12 " + (i420_12 ? "in" : "out") + + ", P012 " + (p012 ? "in" : "out")); + + // A SHORT BUFFER IS REPORTED, NOT SILENTLY TRUNCATED. This is the whole + // difference between a two-call query and a fixed-capacity member: the + // caller has to be able to tell a complete answer from a clipped one. + uint32_t shortCount = 1u; + r = VkEncEnumerateInputFormats(ctx, 0u, codec, profile, &shortCount, + entries); + Check((counted <= 1u) || (r == VK_INCOMPLETE), + "a buffer too small answers VK_INCOMPLETE", + std::string(label) + ": returned " + I64((long long)r)); + Check(shortCount == ((counted < 1u) ? counted : 1u), + "a truncated answer reports how many entries were written", + std::string(label) + ": reported " + I64((long long)shortCount)); + + // THE COUNT IS THE ONE OUTPUT THAT IS NEVER OPTIONAL. + r = VkEncEnumerateInputFormats(ctx, 0u, codec, profile, nullptr, entries); + Check(r == VK_ERROR_INITIALIZATION_FAILED, "a null count is refused", + std::string(label) + ": returned " + I64((long long)r)); + + r = VkEncEnumerateInputFormats(nullptr, 0u, codec, profile, &counted, + nullptr); + Check(r == VK_ERROR_INITIALIZATION_FAILED, "a null context is refused", + std::string(label) + ": returned " + I64((long long)r)); +} + +// THE Std SYNTAX FLAGS, against a real driver answer. +// +// One entry point with a codec selector answers all three codecs, so the +// codec the query named is what an entry is read as. A pair the library does +// not probe answers a code and a count of zero. +void CheckStdFlagEnumerator(VulkanVideoEncoderContext* ctx) +{ + const VkVideoCodecOperationFlagBitsKHR kCodec = + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + const uint32_t kProfile = VK_VIDEO_ENCODER_PROFILE_DEFAULT; + + // THE COUNTING CALL. A null array is not an error; it is the question. + uint32_t counted = 0xFFFFFFFFu; + VkResult r = VkEncEnumerateStdFlags(ctx, 0u, kCodec, kProfile, &counted, + nullptr); + Check(r == VK_SUCCESS, "the counting call succeeds", + "returned " + I64((long long)r)); + Check(counted == 1u, + "one Std syntax-flag entry is reported for the probed profile", + "counted " + I64((long long)counted)); + + // THE FETCHING CALL, at exactly the size the counting call asked for. + // The sentinel tail is what proves the library wrote what it said and not + // one entry more. + VkVideoEncoderStdFlags entries[4]; + for (uint32_t i = 0; i < 4u; i++) { + entries[i] = 0xDEADBEEFu; + } + uint32_t fetched = 4u; + r = VkEncEnumerateStdFlags(ctx, 0u, kCodec, kProfile, &fetched, entries); + Check(r == VK_SUCCESS, "the fetching call succeeds", + "returned " + I64((long long)r)); + Check(fetched == counted, + "the fetching call writes as many entries as the counting call " + "promised", + "counted " + I64((long long)counted) + ", wrote " + + I64((long long)fetched)); + Check(entries[counted] == 0xDEADBEEFu, + "nothing is written past the reported count", + "the entry after the last one was overwritten"); + + // THE SNAPSHOT CLAIM. The count cannot move between the two calls, + // because the answer was computed before either of them. + uint32_t recounted = 0xFFFFFFFFu; + r = VkEncEnumerateStdFlags(ctx, 0u, kCodec, kProfile, &recounted, nullptr); + Check((r == VK_SUCCESS) && (recounted == counted), + "the count is stable across calls -- the snapshot cannot move", + "first " + I64((long long)counted) + ", then " + + I64((long long)recounted)); + + // THE ANSWER IS READ AGAINST THE CODEC THE QUERY NAMED. A second codec + // this device also encodes answers its own list through the same entry + // point, which is what the codec selector is for. + uint32_t h265Count = 0xFFFFFFFFu; + r = VkEncEnumerateStdFlags(ctx, 0u, + VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, &h265Count, + nullptr); + Check((r != VK_SUCCESS) || (h265Count == 1u), + "the codec selector picks the list, so a second codec answers its " + "own", + "returned " + I64((long long)r) + " with " + + I64((long long)h265Count)); + + // A SHORT BUFFER IS REPORTED, NOT SILENTLY TRUNCATED. This is the whole + // difference between a two-call query and a fixed-capacity member: the + // caller has to be able to tell a complete answer from a clipped one. + uint32_t shortCount = 0u; + r = VkEncEnumerateStdFlags(ctx, 0u, kCodec, kProfile, &shortCount, + entries); + Check(r == VK_INCOMPLETE, "a buffer too small answers VK_INCOMPLETE", + "returned " + I64((long long)r)); + Check(shortCount == 0u, + "a truncated answer reports how many entries were written", + "reported " + I64((long long)shortCount)); + + // THE COUNT IS THE ONE OUTPUT THAT IS NEVER OPTIONAL. + r = VkEncEnumerateStdFlags(ctx, 0u, kCodec, kProfile, nullptr, entries); + Check(r == VK_ERROR_INITIALIZATION_FAILED, "a null count is refused", + "returned " + I64((long long)r)); + + r = VkEncEnumerateStdFlags(nullptr, 0u, kCodec, kProfile, &counted, + nullptr); + Check(r == VK_ERROR_INITIALIZATION_FAILED, "a null context is refused", + "returned " + I64((long long)r)); + + // A PAIR THE LIBRARY DOES NOT PROBE ANSWERS ZERO, NOT A STALE NUMBER. + // A caller that treats a refused profile as "advertise nothing" reads the + // count, so it has to be written even on the refusal path. + uint32_t decodeCount = 0xFFFFFFFFu; + r = VkEncEnumerateStdFlags(ctx, 0u, + VK_VIDEO_CODEC_OPERATION_DECODE_H264_BIT_KHR, + kProfile, &decodeCount, nullptr); + Check(r == VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR, + "a codec this library does not encode is refused by code", + "returned " + I64((long long)r)); + Check(decodeCount == 0u, "a refused codec reports a count of zero", + "reported " + I64((long long)decodeCount)); + + uint32_t strayCount = 0xFFFFFFFFu; + r = VkEncEnumerateStdFlags(ctx, 0u, kCodec, /*profile*/ 199u, &strayCount, + nullptr); + Check(r == VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR, + "a profile number this codec does not have is refused by code", + "returned " + I64((long long)r)); + Check(strayCount == 0u, "a refused profile reports a count of zero", + "reported " + I64((long long)strayCount)); + + // THE SCALAR ANSWER STILL NAMES ITS CODEC, and a refused pair still + // leaves the caller's structure alone. + VkVideoEncoderCapabilities caps = {}; + r = VkEncGetEncodeCapabilities(ctx, 0u, kCodec, kProfile, &caps); + Check(r == VK_SUCCESS, "the H.264 default profile is probed", + "returned " + I64((long long)r)); + Check(caps.codec == kCodec, + "the answer names the codec it was asked about", + "names " + I64((long long)caps.codec)); + + VkVideoEncoderCapabilities untouched = {}; + r = VkEncGetEncodeCapabilities(ctx, 0u, + VK_VIDEO_CODEC_OPERATION_DECODE_H264_BIT_KHR, + kProfile, &untouched); + Check(r == VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR, + "a codec this library does not encode is refused by code", + "returned " + I64((long long)r)); + Check(untouched.codec == VK_VIDEO_CODEC_OPERATION_NONE_KHR, + "a refused codec leaves the caller's structure unwritten", + "names " + I64((long long)untouched.codec)); + + r = VkEncGetEncodeCapabilities(nullptr, 0u, kCodec, kProfile, &caps); + Check(r == VK_ERROR_INITIALIZATION_FAILED, "a null context is refused", + "returned " + I64((long long)r)); +} + +int main(int argc, const char** argv) +{ + // --own-validate is the NEGATIVE CONTROL for the --validate arm. The + // library's refusal to attach a debug callback must be scoped to an + // IMPORTED instance; a fix that simply stopped attaching callbacks + // altogether would make --validate green and would be indistinguishable + // without this. Here the library owns the instance, so it created it with + // whatever it needs and must still bring the callback up and encode. + const bool ownValidate = + (argc > 1) && (std::strcmp(argv[1], "--own-validate") == 0); + const bool own = + ownValidate || ((argc > 1) && (std::strcmp(argv[1], "--own") == 0)); + const bool conflict = + (argc > 1) && (std::strcmp(argv[1], "--conflict") == 0); + const bool validate = + ownValidate || + ((argc > 1) && (std::strcmp(argv[1], "--validate") == 0)); + + // THE CONTEXT PATH. Same adoption, expressed the way the context design + // says it must be: a VulkanVideoEncoderContext in ADOPT mode, and a + // session created ON it, instead of two handles copied onto the config. + // --context-conflict is the refusal arm -- it builds the session on a + // context AND leaves the config's external handles set, which must be a + // typed error rather than one of the two silently winning. + const bool contextConflict = + (argc > 1) && (std::strcmp(argv[1], "--context-conflict") == 0); + // THE LIST QUERIES. A context, and nothing built on it: the two-call + // enumerators answer from the construction-time snapshot, so this arm + // needs no session and no encode to drive every branch of the idiom. + const bool enumerate = + (argc > 1) && (std::strcmp(argv[1], "--enumerate") == 0); + const bool useContext = + contextConflict || enumerate || + ((argc > 1) && (std::strcmp(argv[1], "--context") == 0)); + + std::printf("Encoder-ext ADOPT-mode session (%s)\n", + ownValidate + ? "OWN + config.validate control -- library owns the " + "instance and must still attach its callback" + : (own ? "OWN control -- no external handles" + : (conflict + ? "PIN -- adopted physical device vs a " + "conflicting deviceId" + : (validate + ? "ADOPT + config.validate over a " + "borrowed instance" + : "ADOPT -- borrowed instance + " + "physical device, library-created " + "device")))); + if (useContext) { + std::printf(" session built by CreateVulkanVideoEncoderExtOnContext " + "on an ADOPT-mode context%s\n", + contextConflict + ? ", with conflicting config.external* still set" + : ""); + } + // LAYER PROVENANCE. A validation claim without the layer configuration + // that produced it is not a measurement, and this suite has already + // produced two uncomparable numbers for the same arm (10 messages once, 46 + // another time) because neither run recorded what was in effect. + // + // VK_LAYER_PATH decides whether a layer runs at all -- unset, the + // --own-validate arm SKIPs with 77 rather than pretending to prove + // something. VK_LAYER_SETTINGS_PATH decides whether repeated VUIDs can be + // capped; it is deliberately NOT described as "capped at 10" when unset, + // because measurement says otherwise: on the project GPU host this binary + // reports the same 46 messages set and unset, with no cap notice either + // way. Whether the documented default applies is a property of the layer + // build, which is exactly why the path is echoed rather than assumed. + // Read once each. The wording deliberately avoids the token "SKIP": + // rc/t23run.sh greps this output for SKIP and keeps only the first + // three matches, so an unconditional line carrying that word would + // crowd out the real cause of a skip. + const char* layerPath = std::getenv("VK_LAYER_PATH"); + const char* layerSettings = std::getenv("VK_LAYER_SETTINGS_PATH"); + std::printf(" VK_LAYER_PATH=%s\n", + layerPath ? layerPath + : "(unset -- validation arms cannot run)"); + std::printf(" VK_LAYER_SETTINGS_PATH=%s\n", + layerSettings + ? layerSettings + : "(unset -- layer defaults, message cap unknown)"); + std::printf("------------------------------------------------\n"); + + Embedder emb; + if (!own) { + if (!BuildEmbedder(&emb, /*withValidationLayer=*/false)) { + std::printf("SKIP: could not stand up an embedder instance / " + "encode-capable physical device\n"); + TearDownEmbedder(&emb); + return 77; + } + std::printf(" embedder instance %s physicalDevice %s (%s, deviceID " + "0x%x)\n", + U64Hex((unsigned long long)(uintptr_t)emb.instance).c_str(), + U64Hex((unsigned long long)(uintptr_t)emb.phys).c_str(), + emb.name, emb.deviceID); + } + + int rc = 0; + { + // Declared BEFORE |encoder| so it is released AFTER it. The session + // takes its own reference, so this is belt-and-braces -- but it is the + // same ordering the library's own member layout uses, and stating it + // the same way here is free. + VkSharedBaseObj context; + + VkSharedBaseObj encoder; + if (useContext) { + VkVideoEncoderContextCreateInfo ci = {}; + ci.mode = VK_VIDEO_ENCODER_CONTEXT_MODE_ADOPT; + ci.adoptInstance = emb.instance; + ci.adoptPhysicalDevice = emb.phys; + + // NOT A SKIP. An ADOPT context over an instance and physical + // device this process just built successfully is a LIBRARY + // failure, not a host condition -- the embedder stood up fine a + // few lines above. Skipping here would let a regression that + // breaks context creation outright report as ctest "Skipped", + // which does not fail a run. Same rule the InitializeExt arm + // further down already states for itself. + const VkResult cr = CreateVulkanVideoEncoderContext(&ci, context); + Check((cr == VK_SUCCESS) && (bool)context, + "CreateVulkanVideoEncoderContext(ADOPT) succeeded", + "returned " + I64((long long)cr)); + if ((cr != VK_SUCCESS) || !context) { + std::printf("----------------------------------------------" + "--\n"); + std::printf("FAILED : %d checks, %d failures\n", g_checks, + g_failures); + TearDownEmbedder(&emb); + return 1; + } + Check(VkEncGetPhysicalDeviceCount(context.get()) == 1u, + "an ADOPT context enumerates exactly the adopted device", + "count is " + I64((long long)VkEncGetPhysicalDeviceCount( + context.get()))); + + if (enumerate) { + CheckStdFlagEnumerator(context.get()); + // Two profiles, because the list is derived per profile: an + // 8-bit one and a 10-bit one take different encode sources, + // so a list that were profile-independent would show up here. + CheckInputFormatEnumerator( + context.get(), + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, "H.264 default"); + CheckInputFormatEnumerator( + context.get(), + VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, "H.265 Main 10"); + const int listRc = (g_failures == 0) ? 0 : 1; + // The context is released before the borrowed instance is + // destroyed, for the ordering reason stated at the end of + // main(). + context.reset(); + TearDownEmbedder(&emb); + std::printf("--------------------------------------------" + "----\n"); + std::printf("%s : %d checks, %d failures\n", + (listRc == 0) ? "PASSED" : "FAILED", g_checks, + g_failures); + return listRc; + } + + // Out of range must be REFUSED, not clamped. A clamp would make + // every wrong index quietly encode on device 0, which on a + // multi-GPU host is the exact substitution the context exists to + // prevent -- and is unobservable on a one-GPU test machine. + { + VkSharedBaseObj outOfRange; + const VkResult r = + CreateVulkanVideoEncoderExtOnContext(context, 99u, + outOfRange); + Check((r != VK_SUCCESS) && !outOfRange, + "deviceIndex out of range is refused", + "returned " + I64((long long)r)); + } + + // THE OUT-PARAM CONTRACT. Asserting !outOfRange above proves + // only that nothing was written into a null; it cannot tell + // 'left alone' from 'cleared', and clearing is exactly the + // behaviour that changed. Pre-load a real session and prove it + // SURVIVES a failing create. + { + VkSharedBaseObj survivor; + const VkResult okRes = + CreateVulkanVideoEncoderExtOnContext(context, 0u, + survivor); + Check((okRes == VK_SUCCESS) && (bool)survivor, + "a valid session for the out-param survival check", + "returned " + I64((long long)okRes)); + if (survivor) { + VulkanVideoEncoderExt* before = survivor.get(); + const VkResult badRes = + CreateVulkanVideoEncoderExtOnContext(context, 99u, + survivor); + Check((badRes != VK_SUCCESS) && + (survivor.get() == before), + "a failing create does NOT destroy the caller's " + "existing session", + "the out-param was cleared or overwritten"); + survivor.reset(); + } + } + + // NOT A SKIP, for the same reason: this is the entry point under + // test. + const VkResult er = + CreateVulkanVideoEncoderExtOnContext(context, 0u, encoder); + Check((er == VK_SUCCESS) && (bool)encoder, + "CreateVulkanVideoEncoderExtOnContext succeeded", + "returned " + I64((long long)er)); + if ((er != VK_SUCCESS) || !encoder) { + std::printf("----------------------------------------------" + "--\n"); + std::printf("FAILED : %d checks, %d failures\n", g_checks, + g_failures); + encoder.reset(); + context.reset(); + TearDownEmbedder(&emb); + return 1; + } + } else if ((CreateVulkanVideoEncoderExt(encoder) != VK_SUCCESS) || + !encoder) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + TearDownEmbedder(&emb); + return 77; + } + + VkVideoEncoderConfig config = {}; + FillConfig(&config); + + // On the context path the CONTEXT supplies the instance and physical + // device, so the config must not -- except in --context-conflict, + // which sets them precisely to prove the collision is refused. + if (!own && (!useContext || contextConflict)) { + config.externalInstance = emb.instance; + config.externalPhysicalDevice = emb.phys; + // THE WHOLE POINT: no logical device crosses the boundary. + config.externalDevice = VK_NULL_HANDLE; + } + if (conflict) { + // A deviceId that cannot be the adopted device. If the pin is + // real the library refuses; if it is advisory it quietly selects + // something else and we would never know on a one-GPU host. + config.deviceId = (int32_t)(emb.deviceID ^ 0x5A5A); + } + if (validate) { + config.validate = VK_TRUE; + } + + const VkResult init = encoder->InitializeExt(config); + + if (contextConflict) { + // TYPED, not merely non-success. Checking only != VK_SUCCESS would + // pass for the wrong reason on a host that fails InitializeExt for + // an unrelated cause -- a missing codec, no encode queue, + // VK_ERROR_LAYER_NOT_PRESENT -- and the CMake comment and commit + // record both claim a typed refusal. + Check(init == VK_ERROR_INITIALIZATION_FAILED, + "a config that ALSO names externalInstance / " + "externalPhysicalDevice is refused with the TYPED error", + "InitializeExt returned " + I64((long long)init) + + ", expected VK_ERROR_INITIALIZATION_FAILED (" + + I64((long long)VK_ERROR_INITIALIZATION_FAILED) + ")"); + // THE OTHER TWO DEVICE SELECTORS. deviceId and gpuUUID name a + // device just as externalPhysicalDevice does, and the context has + // already chosen one. Untested, these refusals would be new code + // no input reaches -- and their absence is not observable from the + // arm above, which never sets either field. Each gets its OWN + // session: InitializeExt is not re-entrant after a refusal in any + // documented sense, and reusing the refused one would test + // recovery rather than the gate. + struct SelectorCase { + const char* what; + bool setDeviceId; + bool setUuid; + }; + static const SelectorCase kSelectorCases[] = { + {"config.deviceId", true, false}, + {"config.gpuUUID", false, true}, + }; + for (const SelectorCase& sc : kSelectorCases) { + VkSharedBaseObj selEncoder; + const VkResult sr = CreateVulkanVideoEncoderExtOnContext( + context, 0u, selEncoder); + if ((sr != VK_SUCCESS) || !selEncoder) { + Check(false, "second session on the same context", + "CreateVulkanVideoEncoderExtOnContext returned " + + I64((long long)sr)); + continue; + } + VkVideoEncoderConfig selConfig = {}; + FillConfig(&selConfig); + if (sc.setDeviceId) { + // A deviceId that cannot be the adopted device, so a + // pass-through would be caught even if it were honoured + // rather than refused. + selConfig.deviceId = (int32_t)(emb.deviceID ^ 0x5A5A); + } + if (sc.setUuid) { + selConfig.gpuUUID[0] = 0xA5; + } + const VkResult si = selEncoder->InitializeExt(selConfig); + const std::string selLabel = + std::string(sc.what) + + " is refused with the TYPED error on the context path"; + Check(si == VK_ERROR_INITIALIZATION_FAILED, selLabel.c_str(), + "InitializeExt returned " + I64((long long)si) + + ", expected VK_ERROR_INITIALIZATION_FAILED"); + selEncoder.reset(); + } + + rc = (g_failures == 0) ? 0 : 1; + std::printf("------------------------------------------------\n"); + std::printf("%s : %d checks, %d failures\n", + (rc == 0) ? "PASSED" : "FAILED", g_checks, g_failures); + encoder.reset(); + context.reset(); + TearDownEmbedder(&emb); + return rc; + } + + if (conflict) { + Check(init != VK_SUCCESS, + "conflicting deviceId is REFUSED, not silently re-selected", + "InitializeExt returned VK_SUCCESS (" + I64((long long)init) + + ")"); + if (init == VK_SUCCESS) { + Check(encoder->GetVkPhysicalDevice() == emb.phys, + "if it did succeed, at least the pin held", + "library selected a DIFFERENT physical device"); + } + rc = (g_failures == 0) ? 0 : 1; + std::printf("------------------------------------------------\n"); + std::printf("%s : %d checks, %d failures\n", + (rc == 0) ? "PASSED" : "FAILED", g_checks, g_failures); + // The init == VK_SUCCESS sub-case just above is explicitly + // contemplated, and in it the session owns a VkDevice created + // on the borrowed instance. Release it before the instance. + encoder.reset(); + context.reset(); + TearDownEmbedder(&emb); + return rc; + } + + if (init != VK_SUCCESS) { + // In ADOPT mode this is the failure the whole test is about, so it + // is an ASSERTION, not a skip: the OWN control proves the host can + // encode, so a red here is the library and not the machine. + if (own) { + // VK_ERROR_LAYER_NOT_PRESENT (-6) has one cause on the OWN + // path and it is not the GPU: the library asks for + // VK_LAYER_KHRONOS_validation by name when config.validate is + // set, and the loader does not have it. Saying "no + // encode-capable device" there sent a reader looking at the + // hardware for a missing file. + if ((int)init == (int)VK_ERROR_LAYER_NOT_PRESENT) { + std::printf("SKIP: VK_LAYER_KHRONOS_validation is not on " + "this host's Vulkan loader search path, so the " + "OWN+validate control cannot run. Export " + "VK_LAYER_PATH to a directory containing its " + "manifest to enable it.\n"); + } else { + std::printf("SKIP: InitializeExt failed (%d) -- no " + "encode-capable Vulkan device on this host\n", + (int)init); + } + TearDownEmbedder(&emb); + return 77; + } + Check(false, "InitializeExt(ADOPT)", + "returned " + I64((long long)init) + + " on a borrowed instance + physical device"); + std::printf("------------------------------------------------\n"); + std::printf("FAILED : %d checks, %d failures\n", g_checks, + g_failures); + // InitializeExt has failure returns AFTER InitVulkanDevice + // succeeded, so the session may hold a live VkDevice on the + // borrowed instance. Release it before the instance. + encoder.reset(); + context.reset(); + TearDownEmbedder(&emb); + return 1; + } + + VkInstance instance = encoder->GetVkInstance(); + VkDevice device = encoder->GetVkDevice(); + VkPhysicalDevice phys = encoder->GetVkPhysicalDevice(); + + std::printf(" session instance %s physicalDevice %s device %s\n", + U64Hex((unsigned long long)(uintptr_t)instance).c_str(), + U64Hex((unsigned long long)(uintptr_t)phys).c_str(), + U64Hex((unsigned long long)(uintptr_t)device).c_str()); + + if (!own) { + Check(instance == emb.instance, + "the session BORROWED the embedder's VkInstance", + "session instance " + + U64Hex((unsigned long long)(uintptr_t)instance) + + " != embedder " + + U64Hex((unsigned long long)(uintptr_t)emb.instance)); + // WHAT THIS CANNOT CATCH ON A SINGLE-GPU HOST, stated so the + // green is not read as more than it is. If the borrowed physical + // device were lost on the way in, InitPhysicalDevice would fall + // back to enumerating the (correct, borrowed) instance with + // deviceId -1 and no UUID filter, pick the first encode-capable + // device, and on a one-GPU box that is the SAME HANDLE VALUE. + // This check passes either way here. The instance assertion above + // does discriminate -- losing the borrowed instance makes the + // library create its own, which is a different handle -- and the + // deviceIndex 99 case discriminates on the factory's validation. + // A falsifiable pin test needs two GPUs. + Check(phys == emb.phys, + "the session is PINNED to the embedder's VkPhysicalDevice", + "session physicalDevice " + + U64Hex((unsigned long long)(uintptr_t)phys) + + " != embedder " + + U64Hex((unsigned long long)(uintptr_t)emb.phys)); + } + Check(device != VK_NULL_HANDLE, + "the library created its OWN VkDevice", "GetVkDevice() is null"); + + DeviceFns fns; + PFN_vkGetInstanceProcAddr gipa = emb.gipa; + void* localLib = nullptr; + if (gipa == nullptr) { + localLib = dlopen("libvulkan.so.1", RTLD_NOW); + if (localLib == nullptr) { + localLib = dlopen("libvulkan.so", RTLD_NOW); + } + if (localLib != nullptr) { + gipa = (PFN_vkGetInstanceProcAddr)dlsym( + localLib, "vkGetInstanceProcAddr"); + } + } + if ((gipa == nullptr) || + !LoadDeviceFns(gipa, instance, device, &fns)) { + std::printf("SKIP: could not load the Vulkan entry points this " + "test needs\n"); + // Release the session BEFORE the embedder's instance: by here the + // library has created a VkDevice on that instance, and + // ~VulkanDeviceContext runs DeviceWaitIdle + DestroyDevice on it. + // Letting TearDownEmbedder go first destroys the parent instance + // out from under a live device. + encoder.reset(); + context.reset(); + TearDownEmbedder(&emb); + return 77; + } + + uint32_t frames = 0; + uint64_t bytes = 0; + if (!EncodeAndDrain(encoder, fns, phys, device, &frames, &bytes)) { + std::printf("SKIP: the encode harness could not be set up\n"); + // Same ordering hazard as above: a live VkDevice on the borrowed + // instance must go first. + encoder.reset(); + context.reset(); + TearDownEmbedder(&emb); + return 77; + } + std::printf(" encoded %u frames, %llu bitstream bytes\n", frames, + (unsigned long long)bytes); + Check(frames > 0, "the library-created device ENCODED", + "no frames came back"); + Check(bytes > 0, "the bitstream is non-empty", + "captured " + I64((long long)frames) + " frames, 0 bytes"); + + rc = (g_failures == 0) ? 0 : 1; + } + // The encoder is released before the embedder's instance is destroyed: + // the library-created VkDevice lives on the borrowed instance, so the + // reverse order is a use-after-free of the instance and would make a + // teardown ordering bug look like an encode bug. + TearDownEmbedder(&emb); + + std::printf("------------------------------------------------\n"); + std::printf("%s : %d checks, %d failures\n", (rc == 0) ? "PASSED" : "FAILED", + g_checks, g_failures); + return rc; +} diff --git a/vk_video_encoder/test/encoder-ext-adopt-device/validation_gate.cmake b/vk_video_encoder/test/encoder-ext-adopt-device/validation_gate.cmake new file mode 100644 index 00000000..aaed7bfb --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-adopt-device/validation_gate.cmake @@ -0,0 +1,232 @@ +# Run a test binary and decide pass / fail / skip on the exit code AND the +# validation-layer output together. +# +# WHY A WRAPPER RATHER THAN set_tests_properties(FAIL_REGULAR_EXPRESSION): +# SKIP_RETURN_CODE outranks FAIL_REGULAR_EXPRESSION in CTest, so a run that +# emitted validation errors on its way to a 77 exit was recorded as Skipped. +# main.cpp has two 77 returns reached AFTER device creation, i.e. after the +# layer can already have spoken. A property-only gate cannot see both signals. +# +# ------------------------------------------------------------------------ +# THE ECHO IS SANITISED, AND THAT IS LOAD-BEARING, NOT TIDINESS. +# ------------------------------------------------------------------------ +# CTest decides SKIP by regex-matching the WHOLE captured output, and this +# script replays the child's stdout+stderr into that same space. An earlier +# version claimed the skip token "does not exist on any failing path" because +# only this script emits it. That guarantee was void: the child's text is in the +# match space too. Demonstrated end-to-end -- a directory named +# .../VALIDATION_GATE_RESULT=SKIP passed to VK_LAYER_PATH is echoed verbatim by +# main.cpp's provenance line, and a real run with 46 genuine validation errors +# was recorded "100% tests passed ... ***Skipped". +# +# So every occurrence of the token is defanged in the child's text before it is +# printed. The token then appears in the output only when THIS script writes it, +# on a path that has already established there were no validation errors. +# +# ------------------------------------------------------------------------ +# WHAT IS COUNTED: MESSAGES, NOT VUID TOKENS. +# ------------------------------------------------------------------------ +# The same defect yields a different token count depending on which reporter is +# active, with the library untouched: +# * arms where the harness installs its own debug-utils messenger print a +# "Validation Error: [ VUID-x ]" header AND a "The Vulkan spec states: +# ...(VUID-x)" trailer -- TWO tokens per message, on stdout; +# * --own-validate has no harness callback (the library owns the instance), so +# the layer's default reporter prints ONE token per message, on stderr. +# Counting tokens made five arms look like "92 occurrences" against +# --own-validate's 46 when all five in fact have exactly 46 errors. A count that +# doubles under a reporter swap cannot distinguish "removed a defect" from +# "changed a message format". What both reporters emit exactly once per message +# is the "The Vulkan spec states:" trailer, so that closes a message block and +# the block carries the body the allowlist needs. Verified against both: +# default reporter tokens=46 headers=0 -> messages=46; messenger reporter +# tokens=92 headers=46 -> messages=46. +# +# CONTRACT +# -DTEST_EXE= required +# -DTEST_ARGS= optional, ONE argument or a space-separated string +# -DREQUIRE_LAYER= optional. ON: "no layer" is a failure, not a skip, +# so a fleet cannot sit green purely because the layer +# was missing everywhere. +# -DALLOW_VUIDS= optional, matched against the MESSAGE BODY. +# -DRUN_TIMEOUT= optional, default 600. + +if(NOT DEFINED TEST_EXE) + message(FATAL_ERROR "validation_gate: TEST_EXE is required") +endif() +if(NOT DEFINED RUN_TIMEOUT OR RUN_TIMEOUT STREQUAL "") + set(RUN_TIMEOUT 600) +endif() +if(NOT DEFINED ALLOW_VUIDS OR ALLOW_VUIDS STREQUAL "") + # Matched against the BODY, deliberately, because the VUID NAME cannot carry + # this distinction. VUID-Vk-pNext-pNext fires for two unrelated things: + # a struct type the layer does not recognise (version skew, benign), and a + # struct the layer knows perfectly well but which is NOT PERMITTED in that + # chain -- a genuine defect, and exactly the kind a video-encode library that + # chains many extension structs is at risk of. Only the body separates them. + # An earlier version allowlisted on the name and would have passed a run whose + # only message was "...which is not allowed here". + # + # SCOPE, measured, because it is easy to overstate: with the Chrome-bundled + # layer there are ZERO skew messages on all seven arms. With the Vulkan SDK + # layer they appear on six, but are the SOLE cause of redness on exactly ONE + # (--context-conflict); the others are red from the genuine defect anyway. So + # this allowlist keeps one arm honest, not six. "Older layer" is not the + # explanation either -- both manifests declare api_version 1.4.304, and it is + # the SDK build that reports skew while the newer Chrome-bundled one does not. + set(ALLOW_VUIDS "unknown VkStructureType") +endif() + +set(_args "") +if(DEFINED TEST_ARGS AND NOT TEST_ARGS STREQUAL "") + separate_arguments(_args NATIVE_COMMAND "${TEST_ARGS}") +endif() + +# TIMEOUT rather than streaming. execute_process buffers, so a hang would +# otherwise reach CTest's own TIMEOUT and take every captured byte with it -- +# measured: "", zero diagnostics, on a suite whose most plausible +# hang is a GPU encode. Timing out INSIDE the script keeps what was captured. +# Streaming (ECHO_*_VARIABLE) is not the answer: it would put the child's raw +# text back into CTest's match space and re-open the token collision above. +execute_process( + COMMAND "${TEST_EXE}" ${_args} + TIMEOUT ${RUN_TIMEOUT} + RESULT_VARIABLE _rc + OUTPUT_VARIABLE _out + ERROR_VARIABLE _err) + +set(_all "${_out}${_err}") +set(_safe "${_all}") +string(REPLACE "VALIDATION_GATE_RESULT" "VALIDATION_GATE_RESULT_FROM_CHILD" _safe "${_safe}") +message("${_safe}") + +# Message count. Prefer the header, which exists exactly once per message when a +# debug-utils messenger is installed; fall back to raw tokens for the default +# reporter, which emits one per message. +string(REGEX MATCHALL "Validation Error" _headers "${_all}") +list(LENGTH _headers _n_headers) +string(REGEX MATCHALL "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+" _tokens "${_all}") +list(LENGTH _tokens _n_tokens) + +# BLOCK-BASED, and the reason matters -- a line-based count is wrong for one of +# the two reporters and an earlier attempt silently passed a 46-error run +# because of it. +# +# messenger reporter (arms where the harness installs its own callback): +# "Validation Error: [ VUID-x ] ... " <- token here +# "The Vulkan spec states: ... (...#VUID-x)" <- token here too +# default reporter (--own-validate; the library owns the instance so there is +# no harness callback): +# "vkQueueSubmit2KHR(): ... " <- NO token +# "The Vulkan spec states: ... (...#VUID-x)" <- token ONLY here +# +# So "skip the trailer" discards every message on the default reporter, and +# "count every token" double-counts on the messenger one. What both emit exactly +# once per message is the TRAILER, so the trailer closes a block and the block +# carries the body the allowlist needs. +string(REPLACE ";" "\\;" _lines "${_all}") +string(REPLACE "\n" ";" _lines "${_lines}") +set(_real 0) +set(_allowed 0) +set(_names "") +set(_block "") +foreach(_line IN LISTS _lines) + set(_block "${_block}\n${_line}") + if(_line MATCHES "The Vulkan spec states") + if(_block MATCHES "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+") + set(_vuid "${CMAKE_MATCH_0}") + if(_block MATCHES "${ALLOW_VUIDS}") + math(EXPR _allowed "${_allowed}+1") + else() + math(EXPR _real "${_real}+1") + list(APPEND _names "${_vuid}") + endif() + endif() + set(_block "") + endif() +endforeach() +# A layer that emits a VUID with no trailer at all would leave a dangling block; +# count it rather than lose it. +if(_block MATCHES "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+") + set(_vuid "${CMAKE_MATCH_0}") + if(_block MATCHES "${ALLOW_VUIDS}") + math(EXPR _allowed "${_allowed}+1") + else() + math(EXPR _real "${_real}+1") + list(APPEND _names "${_vuid}") + endif() +endif() +if(_names) + list(REMOVE_DUPLICATES _names) + string(REPLACE ";" " " _names "${_names}") +endif() + +message("VALIDATION_GATE: exit=${_rc} messages=${_real} allowlisted=${_allowed} " + "(raw tokens=${_n_tokens}, headers=${_n_headers})") + +# Allowlisted messages are tolerated but never silent. On a green run CTest +# shows no output at all, so an accumulation of them would otherwise be +# invisible forever; this line is appended to a file that survives the run. +if(DEFINED GATE_SUMMARY AND NOT GATE_SUMMARY STREQUAL "") + file(APPEND "${GATE_SUMMARY}" + "${TEST_EXE} ${TEST_ARGS}: exit=${_rc} messages=${_real} allowlisted=${_allowed}\n") +endif() + +# ------------------------------------------------------------------------ +# THE ENCODE-SOURCE LAYOUT REGRESSION, WHICH NO VALIDATION MESSAGE CAN CARRY. +# ------------------------------------------------------------------------ +# VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 is checked against the image-layout +# map of the command buffer the encode is recorded into. The staged input's +# barriers are recorded into a DIFFERENT command buffer, so that map has no +# entry for the encode-source image and the check returns true without +# comparing anything -- and the submit-time sweep reads the same registry, so it +# is blind in the same way. CF-02a and CF-02b lived in that blind spot through +# a run reported as "144 -> 0 validation messages"; zero was the correct count +# for a check that never ran. +# +# So the counter above CANNOT see this defect class, and a gate built only on it +# is a gate that passes an encode reading its source in TRANSFER_DST_OPTIMAL. +# VkVideoEncoder::RecordVideoCodingCmd emits its own unconditional diagnostic +# when the staging arm left the image in anything other than +# VIDEO_ENCODE_SRC_KHR; this promotes that from loud to gating. +# +# DEMONSTRATED IN BOTH DIRECTIONS, because a gate that has never been red is not +# known to be a gate: with the hand-off barriers present the two staged arms +# emit zero of these, and with them deleted the copy arm emits 8 and the filter +# arm 60. +string(REGEX MATCHALL "staged encode-source image is in layout" _layout_hits "${_all}") +list(LENGTH _layout_hits _n_layout) +if(_n_layout GREATER 0) + message(FATAL_ERROR + "validation_gate: ${_n_layout} frame(s) reached vkCmdEncodeVideoKHR" + " with the staged encode-source image in the wrong layout" + " (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811). A StageInputFrame arm" + " is missing its hand-off barrier to VIDEO_ENCODE_SRC_KHR. This is" + " invisible to the validation layer -- see the note above -- so this" + " check is the only thing that reports it.") +endif() + +if(_real GREATER 0) + message(FATAL_ERROR + "validation_gate: ${_real} validation message(s): ${_names}" + " (exit=${_rc}, ${_allowed} allowlisted by body /${ALLOW_VUIDS}/)") +endif() + +if(_rc EQUAL 77) + if(REQUIRE_LAYER) + message(FATAL_ERROR + "validation_gate: the run skipped (77) but REQUIRE_LAYER is ON." + " A skipped validation arm proves nothing; set" + " VVS_VALIDATION_LAYER_PATH, or configure with" + " -DVVS_REQUIRE_VALIDATION_LAYER=OFF to allow skipping.") + endif() + message("VALIDATION_GATE_RESULT=SKIP") + return() +endif() + +# _rc is a STRING on abnormal exit ("Segmentation fault", "Process terminated +# due to timeout", "no such file or directory"). EQUAL comparisons are correctly +# false for those, so both branches above fall through to here and fail. +if(NOT _rc EQUAL 0) + message(FATAL_ERROR "validation_gate: test did not exit cleanly: ${_rc}") +endif() diff --git a/vk_video_encoder/test/encoder-ext-adopt-device/vk_layer_settings.txt b/vk_video_encoder/test/encoder-ext-adopt-device/vk_layer_settings.txt new file mode 100644 index 00000000..21623b8b --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-adopt-device/vk_layer_settings.txt @@ -0,0 +1,27 @@ +# Layer settings for the encoder-ext-adopt-device suite. +# +# duplicate_message_limit = 0 means "print forever". +# +# WHY IT IS PINNED, and this is a real cap, not a theoretical one. With no +# settings file reachable the layer applies its documented default of 10 +# messages; with this file in effect the run reports every message; with a +# small explicit limit it reports that many plus the layer's own "reported N +# times ... last time" notice. +# +# The cap is keyed on the VUID string alone, with no per-object component, so a +# single repeated VUID truncates and the run silently understates itself. +# +# METHOD WARNING. The layer also searches the CURRENT WORKING DIRECTORY for +# vk_layer_settings.txt. A control run that merely unsets +# VK_LAYER_SETTINGS_PATH while sitting in a directory that holds this file is +# NOT uncapped, and will report the full count while appearing to prove the cap +# does not apply. Run controls from a clean directory. +# +# WHAT THIS FILE DOES NOT BUY: portability of the count across LAYER BUILDS. +# Different validation-layer builds report different totals and different sets +# of distinct VUIDs for the same binary on the same host. One of them reports +# VUID-vkCmdEncodeVideoKHR-pEncodeInfo-08206 -- an ENCODE-specific VUID, the +# class this suite exists to catch -- which another never emits. So no single +# count is canonical, and the lower one must not be read as the answer. Which +# layer runs is controlled by VVS_VALIDATION_LAYER_PATH in CMakeLists.txt. +khronos_validation.duplicate_message_limit = 0 diff --git a/vk_video_encoder/test/encoder-ext-direct-wait-capacity/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-direct-wait-capacity/CMakeLists.txt new file mode 100644 index 00000000..8a30ee0d --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-direct-wait-capacity/CMakeLists.txt @@ -0,0 +1,184 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_direct_wait_capacity_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The PUBLIC API only, like the sibling release-fence and acquire-fd tests: a +# real device needs no internal seams, and in this case the seam that exists +# would actively mislead. VkEncProbeLastSubmitSync records at the NULL BACKEND, +# which sits above the two SetExternalInputFrame* call sites and therefore +# above the fixed-array assembly this test is about -- it would report the +# full merged count and pass on a library that truncates. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # The descriptor API is an internal header: the public surface of this + # library is the encoder interface, and a test that drives the layer + # beneath it names the internal directory to say so. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# CTest semantics, matching the siblings: 0 every assertion held, 1 an +# assertion failed, 77 no encode-capable GPU and therefore nothing proved +# either way -- reported as SKIPPED, never as a pass. +enable_testing() + +# --------------------------------------------------------------------------- +# THE SUBJECT. Runs on a DIRECT-routed registration (non-LINEAR NV12 declaring +# VIDEO_ENCODE_SRC), which is the only routing whose submit assembles waits +# into the fixed 8-slot array. +# +# Three cases, and the third is why the first two cannot be satisfied by a fix +# that simply refuses more work than it likes: +# A capacity+1 caller waits, last one unsignalled -- a surplus wait must not +# be silently discarded. +# B capacity caller waits plus an armed acquire fence -- 9 into 8, where the +# 9th is the imported acquire semaphore because the ext layer appends it +# last. The header's "loses neither" promise, at the count nothing tested. +# C capacity-1 caller waits plus an acquire fence -- exactly 8, fits, and +# must still be accepted and complete. +# --------------------------------------------------------------------------- +add_test(NAME EncoderExtDirectWaitArrayCapacity + COMMAND ${PROJECT_NAME}) +set_tests_properties(EncoderExtDirectWaitArrayCapacity PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu" + TIMEOUT 900) + +# --------------------------------------------------------------------------- +# THE CONTROL, AND IT IS EVIDENCE RATHER THAN DECORATION. +# +# The identical capacity+1 case against a LINEAR/TRANSFER_SRC registration, +# which routes STAGED onto the std::vector assembly that never truncates. It +# must report the release fence UNSIGNALLED while the surplus wait is still at +# 0 -- i.e. the same apparatus, reading a wait that WAS honoured. +# +# This is what makes a green subject arm mean something. There is no public or +# seam accessor for a slot's resolved inputPath, so "did this registration +# really route DIRECT" cannot be asserted directly; the differential between +# these two entries is the substitute, and it is also the standing answer to +# "is this a test that can fail at all". +# --------------------------------------------------------------------------- +add_test(NAME EncoderExtDirectWaitArrayCapacityStagedControl + COMMAND ${PROJECT_NAME} --staged) +set_tests_properties(EncoderExtDirectWaitArrayCapacityStagedControl PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu" + TIMEOUT 900) + +# --------------------------------------------------------------------------- +# THE LEGACY ENTRY POINT, WHICH REACHES THE SAME FIXED ARRAY BY A DIFFERENT +# DOOR. +# +# SubmitExternalFrame carries a raw VkImage and no registration, so the +# encoder core routes it from the frame's own format and tiling: an OPTIMAL +# NV12 frame is directly encodable and its waits land in the same 8-slot +# array. The registered arms above cannot cover it -- they enter through +# SubmitRegisteredFrame, which is a different entry point with a different +# return type and its own bound. +# +# The readout here is COMPLETION rather than a release fence, because this +# entry point refuses any pNext chain and so can carry no fence descriptor. +# Every wait is pre-signalled, so an accepted frame that never becomes +# retrievable is a frame that was dropped after its caller was told it was +# taken -- a success that can only be waited on forever. +# +# Three cases, and the outer two are what stop a fix that merely refuses: +# 3 waits -- inside capacity, must be accepted AND complete. +# 9 waits -- over capacity, must be reported rather than accepted-and-lost, +# and reported with the code the public header names. +# 8 waits -- exactly full, legal input, must still be accepted AND +# complete. +# --------------------------------------------------------------------------- +# +# GATED ON VALIDATION, because this arm and its H.265 sibling below are the +# only arms in the tree that code a B frame. consecutiveBFrames is 2 on the +# legacy arms and 0 everywhere else, so the encoder builds a bidirectional +# reference list here and nowhere else, and a spec violation on that path is +# reported by nothing unless these entries read the layer. +# +# THE CEILING BELOW NAMES A DEFECT, IT DOES NOT EXCUSE IT: +# +# VUID-VkImageCreateInfo-pNext-06811 +# the direct input image is created with MUTABLE_FORMAT|EXTENDED_USAGE +# and the full usage set the direct path may view it under, and the +# driver reports no video format properties for that combination under +# this profile. It is raised once, at the single vkCreateImage this +# arm performs. +vvs_add_validation_gated_test(EncoderExtLegacySubmitWaitArrayCapacity + TARGET ${PROJECT_NAME} + ARGS --legacy + EXPECT VUID-VkImageCreateInfo-pNext-06811=1) +set_tests_properties(EncoderExtLegacySubmitWaitArrayCapacity PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +# --------------------------------------------------------------------------- +# THE SAME ARM UNDER H.265, BECAUSE THE REFERENCE LISTS ARE ASSEMBLED PER +# CODEC. +# +# VkVideoEncoderH264 and VkVideoEncoderH265 each fill referenceSlotsInfo[] +# from their own L0 and L1 walks, in separate files. A B frame exercised +# through one of them leaves the other's walk unread, so the codec is a +# parameter of this arm rather than a constant: --h265 selects an H.265 +# session and chains VkVideoEncodeH265ProfileInfoKHR on the input image's +# profile, and everything else about the arm is unchanged. +# +# The ceiling is the same one, for the same reason, and names the same defect. +vvs_add_validation_gated_test(EncoderExtLegacySubmitWaitArrayCapacityH265 + TARGET ${PROJECT_NAME} + ARGS --legacy --h265 + EXPECT VUID-VkImageCreateInfo-pNext-06811=1) +set_tests_properties(EncoderExtLegacySubmitWaitArrayCapacityH265 PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +message(STATUS "encoder_ext_direct_wait_capacity_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-direct-wait-capacity/src/main.cpp b/vk_video_encoder/test/encoder-ext-direct-wait-capacity/src/main.cpp new file mode 100644 index 00000000..a205a03e --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-direct-wait-capacity/src/main.cpp @@ -0,0 +1,1219 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * DIRECT-path WAIT-ARRAY CAPACITY, on a real device. + * + * WHAT IS UNDER TEST. VkVideoEncoder::SubmitVideoCodingCmds assembles the + * direct-encode submit's waits into a FIXED 8-slot stack array + * (VkVideoEncoder.cpp: `const uint32_t waitSemaphoreMaxCount = 8;`), and the + * loop that injects the caller's external waits is bounded by + * `waitSemaphoreCount < waitSemaphoreMaxCount`. There is no overflow -- and + * there was no diagnostic either: surplus waits were DISCARDED, silently, and + * the submit went to the queue as though they had never been named. + * + * WHY THAT IS A CORRECTNESS DEFECT AND NOT A CAPACITY LIMIT. A wait names a + * producer that has not finished writing the input image. Dropping one does + * not degrade the encode, it makes vkCmdEncodeVideoKHR read a surface while + * the producer is still writing it -- a read-before-write race, intermittent + * by nature, with a SUCCESS status and a valid release fence handed back to a + * caller who has been given no way to find out. The library already says this + * about the other direction, in its own words, at + * vulkan_video_encoder_ext.cpp: the release fence is gated by + * `kMaxCallerSignalsForReleaseFence = 4` precisely because "the direct-encode + * submit ... silently stops appending when it fills", and a dropped signal + * would be "the hang this fence exists to prevent, caused by the fence". The + * wait side carried the same 8-slot hazard with no gate at all. + * + * WHY THE ACQUIRE FENCE IS THE FIRST CASUALTY. The per-frame acquire fence is + * imported and APPENDED to whichever wait array the frame ended up using + * (vulkan_video_encoder_ext.cpp, the `waitWithAcquire` block), unconditionally + * on count. It is therefore the LAST entry of the merged array, so it is the + * first thing truncation reaches. A caller that supplied 8 waits and an armed + * acquireFenceFd got a 9-entry array whose 9th entry -- the producer fence the + * whole handle API exists to honour -- was dropped, while the public header + * promises "Supplying an acquireFenceFd and a pWaitSemaphores array together + * is legal and loses neither". + * + * WHY THE EXISTING SUITE COULD NOT SEE IT. EncoderExtAcquireFdKeepsCallerWaitArray + * asserts that same header promise, and holds -- but it registers a + * VK_IMAGE_TILING_LINEAR, TRANSFER_SRC-only image, which sets encodeCapable + * false and routes STAGED. The staged submit assembles its waits into a + * std::vector (VkVideoEncoder.cpp, SubmitStagedInputFrame) that never + * truncates, and it exercises ONE caller wait. Neither the path nor the count + * that the defect needs is reachable from it. That is why this file exists. + * + * HOW EACH CASE IS MADE TO FAIL WHEN THE LIBRARY IS BROKEN. Every wait this + * test supplies is a TIMELINE semaphore. All but one are host-signalled to + * their target value BEFORE the submit, so they cannot hold the frame back. + * Exactly one is left at 0. The frame therefore has precisely one reason not + * to run, and the release fence fd is the readout: poll() it. + * + * - honoured -> the frame is parked, the fence has not signalled, poll 0. + * - discarded -> nothing is holding the frame, it encodes, poll 1. + * + * A poll of 1 while the semaphore it was told to wait on is still at 0 IS the + * race, observed. It is not a proxy for it. + * + * THE CONTROL ARM IS PART OF THE EVIDENCE. --staged runs the identical + * 9-wait case against a LINEAR/TRANSFER_SRC registration, which routes to the + * non-truncating vector. It must report poll 0 -- the same apparatus, the same + * counts, the same semaphores, reading a wait that WAS honoured. Without it a + * poll of 0 on the direct arm would be indistinguishable from a test that + * cannot fail, and there is no way to assert the routing directly: no public + * or seam accessor reports a slot's resolved inputPath. + */ + +#include "vulkan_video_encoder_ext.h" + +#include "vk_video/vulkan_video_codec_h264std.h" +#include "vk_video/vulkan_video_codec_h265std.h" + +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace { + +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; + +// The library's own capacity, restated here so the arms below read as +// intentions rather than as magic numbers. Kept as a literal on purpose: if +// VkVideoEncoder.cpp ever raises its array, this test must be re-derived +// deliberately, not silently follow along and stop testing the boundary. +const uint32_t kDirectWaitCapacity = 8; + +int g_failures = 0; +int g_checks = 0; +const char* g_case = ""; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (!ok) { + g_failures++; + std::printf(" FAIL [%s] %s\n %s\n", g_case, what, + detail.c_str()); + } else { + std::printf(" ok [%s] %s\n", g_case, what); + } +} + +std::string I64(long long v) { return std::to_string(v); } + +// poll() for readability. A sync_fd becomes readable when its fence signals. +// Returns 1 signalled, 0 not yet, <0 error. +int PollSignalled(int fd, int timeoutMs) +{ + struct pollfd p = {}; + p.fd = fd; + p.events = POLLIN; + const int r = poll(&p, 1, timeoutMs); + if (r < 0) { + return -1; + } + if (r == 0) { + return 0; + } + return ((p.revents & POLLIN) != 0) ? 1 : -1; +} + +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = + nullptr; + PFN_vkGetPhysicalDeviceProperties GetPhysicalDeviceProperties = nullptr; + PFN_vkCreateSemaphore CreateSemaphore = nullptr; + PFN_vkDestroySemaphore DestroySemaphore = nullptr; + PFN_vkSignalSemaphore SignalSemaphore = nullptr; + PFN_vkGetSemaphoreCounterValue GetSemaphoreCounterValue = nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) { + lib = dlopen("libvulkan.so", RTLD_NOW); + } + if (lib == nullptr) { + std::printf(" ERROR: dlopen(libvulkan) failed: %s\n", dlerror()); + return false; + } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) { + std::printf(" ERROR: no vkGetInstanceProcAddr\n"); + return false; + } + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + std::printf(" ERROR: no vkGetDeviceProcAddr\n"); + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) + LOAD_DEV(CreateSemaphore) + LOAD_DEV(DestroySemaphore) + LOAD_DEV(SignalSemaphore) + LOAD_DEV(GetSemaphoreCounterValue) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + fns->GetPhysicalDeviceProperties = + (PFN_vkGetPhysicalDeviceProperties)gipa( + instance, "vkGetPhysicalDeviceProperties"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr) && + (fns->GetPhysicalDeviceProperties != nullptr); +} + +struct InputImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; +}; + +// |direct| selects the arm. OPTIMAL + VIDEO_ENCODE_SRC + a profile list at +// create time is what the encoder can read without a staging copy, and is +// therefore what makes the registration below encodeCapable; LINEAR + +// TRANSFER_SRC is the control that routes STAGED. Copied in shape from the +// sibling release-fence test, which measures the same two routings. +bool CreateInputImage(const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, InputImage* out, bool direct, + bool h265) +{ + // THE CODEC-SPECIFIC PROFILE STRUCT IS PART OF THE PROFILE, not an + // optional decoration on it. A VkVideoProfileInfoKHR naming an H.264 + // encode operation is only a complete profile once a + // VkVideoEncodeH264ProfileInfoKHR is chained onto it + // (VUID-VkVideoProfileInfoKHR-videoCodecOperation-07181). Without it the + // profile list below describes no profile the session can be matched + // against, so vkCreateImage is asked about an image no session can read + // (VUID-VkImageCreateInfo-pNext-06811) and every encode that names the + // resulting view is incompatible with the bound session + // (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-08206). + // + // HIGH because that is the profile the SESSION will use, not because it + // is the richest one available. The config below leaves |profile| at + // VK_VIDEO_ENCODER_PROFILE_DEFAULT, and the library derives profile_idc + // 100 for 8-bit 4:2:0 input under the default adaptive-transform mode. + // Naming a profile here that the session does not use is the same + // mismatch as naming none. + VkVideoEncodeH264ProfileInfoKHR h264Profile{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_PROFILE_INFO_KHR}; + h264Profile.stdProfileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH; + // MAIN for the reason HIGH is named above: it is the profile the SESSION + // uses. The config leaves |profile| at VK_VIDEO_ENCODER_PROFILE_DEFAULT + // and the library derives general_profile_idc Main for 8-bit 4:2:0 input. + VkVideoEncodeH265ProfileInfoKHR h265Profile{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_PROFILE_INFO_KHR}; + h265Profile.stdProfileIdc = STD_VIDEO_H265_PROFILE_IDC_MAIN; + VkVideoProfileInfoKHR profile{VK_STRUCTURE_TYPE_VIDEO_PROFILE_INFO_KHR}; + profile.pNext = h265 ? (const void*)&h265Profile + : (const void*)&h264Profile; + profile.videoCodecOperation = + h265 ? VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR + : VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + profile.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + profile.lumaBitDepth = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + profile.chromaBitDepth = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + VkVideoProfileListInfoKHR profileList{ + VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR}; + profileList.profileCount = 1; + profileList.pProfiles = &profile; + + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.pNext = direct ? (const void*)&profileList : nullptr; + // WHAT THE LIBRARY WILL DO TO THIS IMAGE, DECLARED AT CREATION. + // + // A frame that routes DIRECT is viewed PER PLANE -- R8_UNORM over the + // luma, R8G8_UNORM over the chroma of an NV12 image -- and a per-plane + // view of a multi-planar image requires MUTABLE_FORMAT + // (VUID-VkImageViewCreateInfo-image-01762). + // + // Without it this image is one the encoder cannot legally view, and the + // arms below would be measuring that rather than wait capacity. The + // staged control needs none of it: it is copied, not viewed per plane. + ci.flags = direct + ? (VkImageCreateFlags)( + VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT | + VK_IMAGE_CREATE_EXTENDED_USAGE_BIT) + : (VkImageCreateFlags)0; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = direct ? VK_IMAGE_TILING_OPTIMAL + : VK_IMAGE_TILING_LINEAR; + // THE USAGE THE VIEWS WILL NAME, not the narrowest one an encode source + // could get away with. A view may not name a usage bit the image was not + // created with (VUID-VkImageViewCreateInfo-pNext-02662), and the wrap the + // direct path performs builds its views over the full set the encoder can + // put an input image to -- the transfer pair for a staged copy and the + // sampled/storage pair for the preprocess filter -- whichever rung this + // particular frame ends up on. So a producer handing an image to this + // path declares all of them, and this test declares what a producer + // would. The staged control declares only what its copy reads. + ci.usage = direct + ? (VkImageUsageFlags)( + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | + VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_SAMPLED_BIT | + VK_IMAGE_USAGE_STORAGE_BIT) + : (VkImageUsageFlags)VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = direct ? VK_IMAGE_LAYOUT_UNDEFINED + : VK_IMAGE_LAYOUT_PREINITIALIZED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + std::printf(" ERROR: vkCreateImage(NV12, direct=%d) failed\n", + (int)direct); + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + const VkMemoryPropertyFlags want = + direct ? (VkMemoryPropertyFlags)VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT + : (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + std::printf(" ERROR: no suitable memory type\n"); + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + std::printf(" ERROR: vkAllocateMemory failed\n"); + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != VK_SUCCESS) { + std::printf(" ERROR: vkBindImageMemory failed\n"); + return false; + } + if (!direct) { + void* mapped = nullptr; + if (fns.MapMemory(device, out->memory, 0, req.size, 0, &mapped) == + VK_SUCCESS) { + uint8_t* bytes = (uint8_t*)mapped; + for (VkDeviceSize i = 0; i < req.size; i++) { + bytes[i] = (uint8_t)((i * 7u) ^ (i >> 9)); + } + fns.UnmapMemory(device, out->memory); + } + } + return true; +} + +//============================================================================= +// The session under test. +//============================================================================= +VkSharedBaseObj g_encoder; +DeviceFns g_fns; +VkDevice g_device = VK_NULL_HANDLE; +VkVideoEncoderResource g_resource = VK_VIDEO_ENCODER_RESOURCE_NULL; +uint64_t g_frameId = 0; + +// WHAT THIS SESSION ACTUALLY CODED, indexed by VkVideoEncoderPictureType. +// Counted at EVERY acquisition site, because a frame is delivered once and +// the sites are not interchangeable: RunLegacyWaitCase drains inside its own +// bounded wait and DrainCaptures() takes whatever is left. +uint32_t g_pictureTypeCount[3] = {0, 0, 0}; + +void CountPictureType(const VkVideoEncodeResult& r) +{ + if ((uint32_t)r.pictureType < 3) { + g_pictureTypeCount[(uint32_t)r.pictureType]++; + } +} + +void DrainCaptures() +{ + VkVideoEncodeResult r; + while (g_encoder->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + CountPictureType(r); + g_encoder->ReleaseEncodedFrame(r.frameId); + } +} + +VkVideoEncoderFrameSubmitInfo BaseInfo() +{ + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = g_resource; + info.frameId = g_frameId; + info.pts = g_frameId; + info.qpOverride = -1; + info.currentLayout = VK_IMAGE_LAYOUT_UNDEFINED; + return info; +} + +// A bank of TIMELINE semaphores. |signalledCount| of them are host-signalled +// to kTargetValue before the caller uses them, so they are already satisfied; +// the remainder stay at 0 and are the only thing that can hold a frame back. +const uint64_t kTargetValue = 1; + +struct WaitBank { + std::vector semaphores; + std::vector values; + + bool Create(uint32_t count, uint32_t signalledCount) + { + VkSemaphoreTypeCreateInfo typeInfo{ + VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO}; + typeInfo.semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + typeInfo.initialValue = 0; + VkSemaphoreCreateInfo semInfo{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + semInfo.pNext = &typeInfo; + + for (uint32_t i = 0; i < count; i++) { + VkSemaphore s = VK_NULL_HANDLE; + if (g_fns.CreateSemaphore(g_device, &semInfo, nullptr, &s) != + VK_SUCCESS) { + return false; + } + semaphores.push_back(s); + values.push_back(kTargetValue); + } + for (uint32_t i = 0; i < signalledCount; i++) { + VkSemaphoreSignalInfo si{VK_STRUCTURE_TYPE_SEMAPHORE_SIGNAL_INFO}; + si.semaphore = semaphores[i]; + si.value = kTargetValue; + if (g_fns.SignalSemaphore(g_device, &si) != VK_SUCCESS) { + return false; + } + } + return true; + } + + // Release whatever is still holding a frame back, so the parked submit can + // retire and the semaphores become destroyable. + void SignalRest(uint32_t from) + { + for (uint32_t i = from; i < semaphores.size(); i++) { + VkSemaphoreSignalInfo si{VK_STRUCTURE_TYPE_SEMAPHORE_SIGNAL_INFO}; + si.semaphore = semaphores[i]; + si.value = kTargetValue; + g_fns.SignalSemaphore(g_device, &si); + } + } + + void Destroy() + { + for (VkSemaphore s : semaphores) { + g_fns.DestroySemaphore(g_device, s, nullptr); + } + semaphores.clear(); + values.clear(); + } +}; + +// One already-signalled real sync_fd, exported by the library from a frame of +// its own. Carries no acquire fence itself, so a retry here can never lose a +// handle. +int MintSyncFd() +{ + for (int attempt = 0; attempt < 64; attempt++) { + int fd = -1; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &fd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + + VkVideoEncoderStatusCode status = + g_encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 2000); + retry++) { + DrainCaptures(); + status = g_encoder->SubmitRegisteredFrame(info, nullptr); + if (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + struct timespec ts = {0, 1000000}; + nanosleep(&ts, nullptr); + } + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + std::printf(" MINT: submit failed, status %d\n", (int)status); + return -1; + } + g_frameId++; + DrainCaptures(); + if (fd >= 0) { + // Wait for it, so the arm that uses it as an acquire fence is + // supplying a fence that is already satisfied and therefore adds + // no reason of its own for the frame not to run. + PollSignalled(fd, 10000); + return fd; + } + std::printf(" MINT: frame %llu answered -1, retrying\n", + (unsigned long long)(g_frameId - 1)); + } + return -1; +} + +//============================================================================= +// The measurement. +// +// |callerWaits| timeline waits, of which all but the LAST are already +// satisfied. Optionally an already-signalled acquire fd on top. Returns +// through its out-params what the library did. +//============================================================================= +struct WaitCaseResult { + VkVideoEncoderStatusCode status = VK_VIDEO_ENCODER_STATUS_SUCCESS; + int releaseFd = -1; + int earlyPoll = -1; + int latePoll = -1; + bool submitted = false; +}; + +WaitCaseResult RunWaitCase(uint32_t callerWaits, bool withAcquireFd) +{ + WaitCaseResult out; + + WaitBank bank; + if (!bank.Create(callerWaits, callerWaits - 1)) { + Check(false, "the timeline semaphore bank could be created", + "vkCreateSemaphore/vkSignalSemaphore failed"); + bank.Destroy(); + return out; + } + + int acquireFd = -1; + if (withAcquireFd) { + acquireFd = MintSyncFd(); + if (acquireFd < 0) { + Check(false, "an acquire sync_fd could be minted", + "MintSyncFd returned -1"); + bank.Destroy(); + return out; + } + } + + int releaseFd = -1; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = acquireFd; + fence.pReleaseFenceFd = &releaseFd; + + VkVideoEncoderFrameSubmitInfo info = BaseInfo(); + info.pNext = &fence; + info.waitSemaphoreCount = (uint32_t)bank.semaphores.size(); + info.pWaitSemaphores = bank.semaphores.data(); + info.pWaitSemaphoreValues = bank.values.data(); + + VkVideoEncoderStatusCode status = + g_encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 2000); + retry++) { + DrainCaptures(); + status = g_encoder->SubmitRegisteredFrame(info, nullptr); + if (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + struct timespec ts = {0, 1000000}; + nanosleep(&ts, nullptr); + } + } + out.status = status; + out.submitted = (status == VK_VIDEO_ENCODER_STATUS_SUCCESS); + + if (out.submitted) { + g_frameId++; + out.releaseFd = releaseFd; + if (releaseFd >= 0) { + // THE READOUT. One wait is still at 0. A fence that has already + // signalled means the submit did not wait for it. + out.earlyPoll = PollSignalled(releaseFd, 300); + } + // Let the frame go, whatever the verdict, so teardown is clean. + bank.SignalRest(0); + if (releaseFd >= 0) { + out.latePoll = PollSignalled(releaseFd, 10000); + ::close(releaseFd); + } + } else if (releaseFd >= 0) { + ::close(releaseFd); + } + + DrainCaptures(); + g_encoder->DrainPendingFrames(); + DrainCaptures(); + bank.Destroy(); + return out; +} + +//============================================================================= +// BASELINE -- A WAIT ARRAY WELL WITHIN CAPACITY, ON THE DIRECT PATH ITSELF. +// +// Three caller waits, the last unsignalled, with and without an armed acquire +// fence. Both must park the frame. +// +// This is not decoration and it is not a smoke test. Without it, the surplus +// cases below are uninterpretable: "the 9th wait was discarded because it did +// not fit" and "this path never honours a caller wait at all" produce the +// IDENTICAL reading (submit accepted, release fence already signalled), and +// only this arm tells them apart. The staged control answers a different +// question -- whether the apparatus works -- and cannot substitute, because it +// exercises the other assembly entirely. +//============================================================================= +void CaseDirectSmallWaitArrayIsHonoured(bool withAcquire) +{ + g_case = withAcquire ? "DirectBaselineFewWaitsPlusAcquire" + : "DirectBaselineFewWaits"; + + const uint32_t waits = 3; // +1 acquire at most = 4, half the capacity + const WaitCaseResult r = RunWaitCase(waits, withAcquire); + + std::printf(" [%s] waits=%u acquire=%d status=%d releaseFd=%d " + "earlyPoll=%d latePoll=%d\n", + g_case, waits, (int)withAcquire, (int)r.status, r.releaseFd, + r.earlyPoll, r.latePoll); + + Check(r.submitted, "the submit was accepted", + "status " + I64((long long)r.status)); + + // The early poll GATES only on the arm without an acquire fence, and that + // is a limitation of this harness rather than a statement about the + // library. The acquire arm has to mint its sync_fd from a real frame of + // its own (MintSyncFd), and that frame consumes a release fence + // immediately before the frame under test exports one; the fd numbers are + // reused and the early poll on the acquire arm was observed to report + // signalled on a frame that instrumentation showed was correctly parked + // with all four waits present. So the reading is not trustworthy there and + // is printed rather than asserted -- an unreliable check that sometimes + // goes red is worse than no check, because it teaches people to re-run. + // + // Nothing is lost by that: the no-acquire arm establishes the fact the + // surplus cases need ("this path does honour caller waits"), and the + // acquire path's own merge is pinned by Case C, which fits exactly and + // must still complete. + if (!withAcquire) { + Check(r.earlyPoll == 0, + "BASELINE: a wait array well within capacity is honoured on DIRECT", + "release fence polled " + I64(r.earlyPoll) + " while wait #" + + I64(waits) + " was still at 0. If THIS fails, the direct path " + "is not honouring caller waits at any count and the surplus " + "cases below say nothing about capacity"); + } + Check(r.latePoll == 1, "and completes once every wait is signalled", + "release fence polled " + I64(r.latePoll)); +} + +//============================================================================= +// Case A -- MORE CALLER WAITS THAN THE DIRECT ARRAY HOLDS. +// +// capacity+1 caller waits, the last one unsignalled. Nothing else is in the +// array on this path, so the surplus entry IS the unsignalled one. +//============================================================================= +void CaseSurplusCallerWaitIsNotDiscarded(bool direct) +{ + g_case = direct ? "DirectSurplusCallerWait" : "StagedSurplusCallerWait"; + + const uint32_t waits = kDirectWaitCapacity + 1; // 9 + const WaitCaseResult r = RunWaitCase(waits, false); + + std::printf(" [%s] waits=%u status=%d releaseFd=%d earlyPoll=%d " + "latePoll=%d\n", + g_case, waits, (int)r.status, r.releaseFd, r.earlyPoll, + r.latePoll); + + if (!direct) { + // CONTROL. The staged path assembles into a std::vector, so all nine + // waits are honoured and the frame must be parked on the ninth. + Check(r.submitted, "the staged submit was accepted", + "status " + I64((long long)r.status)); + Check(r.earlyPoll == 0, + "CONTROL: the staged path honoured the 9th wait", + "release fence polled " + I64(r.earlyPoll) + + " while wait #9 was still at 0. This arm is the proof that " + "the apparatus can SEE an honoured wait; if it reports 1 " + "the readout is broken and the direct arm proves nothing"); + Check(r.latePoll == 1, + "CONTROL: and it completed once every wait was signalled", + "release fence polled " + I64(r.latePoll)); + return; + } + + // SUBJECT. Nine waits cannot fit an eight-slot array. The only two honest + // outcomes are to refuse the submit, or to accept it and still honour + // every wait. Accepting it and dropping one is the defect. + const bool silentlyDiscarded = r.submitted && (r.earlyPoll == 1); + Check(!silentlyDiscarded, + "a wait that does not fit is not silently discarded", + "the submit returned SUCCESS and the release fence had ALREADY " + "signalled (poll " + I64(r.earlyPoll) + ") while the semaphore it " + "was told to wait on was still at 0. vkCmdEncodeVideoKHR read the " + "input image without waiting for the producer that wait named -- " + "the read-before-write race, observed, with a SUCCESS status and a " + "valid release fd already handed back to the caller"); + + if (!r.submitted) { + std::printf(" [%s] submit refused with status %d -- the surplus " + "wait was reported, not dropped\n", + g_case, (int)r.status); + } +} + +//============================================================================= +// Case B -- A FULL CALLER ARRAY PLUS AN ARMED ACQUIRE FENCE. +// +// Exactly |capacity| caller waits, all satisfied, and an armed acquireFenceFd +// on top. The acquire semaphore is appended LAST, so it is entry 9 of 9 and +// the first thing truncation reaches. This is the header's "loses neither" +// promise at the one count where it was not tested. +//============================================================================= +void CaseAcquireFenceOnAFullWaitArray() +{ + g_case = "DirectAcquireOnFullWaitArray"; + + // All |capacity| caller waits satisfied, so the ONLY entry that could park + // the frame is gone either way; what is under test here is whether the + // library will quietly build a 9-entry array it cannot submit. + const uint32_t waits = kDirectWaitCapacity; // 8, +1 acquire = 9 + const WaitCaseResult r = RunWaitCase(waits, true); + + std::printf(" [%s] waits=%u +acquire status=%d releaseFd=%d " + "earlyPoll=%d latePoll=%d\n", + g_case, waits, (int)r.status, r.releaseFd, r.earlyPoll, + r.latePoll); + + Check(!r.submitted, + "a merged wait array larger than the submit array is refused, " + "not truncated", + "the submit returned SUCCESS with " + I64(waits) + + " caller waits and an armed acquireFenceFd -- 9 entries into an " + "8-slot array. The 9th is the imported ACQUIRE semaphore, " + "because the ext layer appends it last, so the producer fence " + "the handle API exists to honour is exactly what got dropped, " + "while the header promises the pair 'loses neither'"); +} + +//============================================================================= +// Case C -- THE BOUNDARY THAT MUST STILL WORK. +// +// capacity-1 caller waits plus an acquire fence is exactly |capacity| entries. +// It fits, so it must be accepted and must complete. Without this, "refuse +// everything" would pass Cases A and B. +//============================================================================= +void CaseExactlyFullIsStillAccepted() +{ + g_case = "DirectExactlyFullStillWorks"; + + const uint32_t waits = kDirectWaitCapacity - 1; // 7, +1 acquire = 8 + const WaitCaseResult r = RunWaitCase(waits, true); + + std::printf(" [%s] waits=%u +acquire status=%d releaseFd=%d " + "earlyPoll=%d latePoll=%d\n", + g_case, waits, (int)r.status, r.releaseFd, r.earlyPoll, + r.latePoll); + + Check(r.submitted, + "a merged wait array that exactly fills the submit array is accepted", + "status " + I64((long long)r.status) + " -- a fix that refuses this " + "has traded a silent drop for a refusal of legal input"); + Check(r.latePoll == 1, + "and that frame completes", + "release fence polled " + I64(r.latePoll)); +} + +//============================================================================= +// THE LEGACY ENTRY POINT, WHICH REACHES THE SAME FIXED ARRAY BY A DIFFERENT +// DOOR -- AND CAN ANSWER ITS CALLER BEFORE THE ARRAY IS EVER BUILT. +// +// SubmitExternalFrame() carries a raw VkImage and no registration, so the +// encoder core routes it from the frame's own format and tiling: an OPTIMAL +// NV12 frame is directly encodable, and its waits are assembled into the same +// 8-slot array the registered direct path uses. +// +// WHY THE ASSEMBLY'S OWN REFUSAL IS NOT A SUBSTITUTE FOR A GATE AT THE ENTRY +// POINT, and why this arm runs with B-frames while the registered arms above +// do not. A frame is recorded and submitted from the deferred-GOP flush. With +// no reordering that flush runs per frame, inline, so an over-capacity list +// is refused on the caller's own thread and the caller learns of it. WITH +// reordering the queue is drained on a later call -- so the entry point has +// already returned, and whatever the assembly then decides has no route back +// to whoever holds that answer. +// +// So this arm configures consecutiveBFrames > 0 deliberately. It is the shape +// in which "the submit refuses it" and "the caller is told" come apart, and +// therefore the only shape in which a bound at the entry point is doing any +// work at all. +// +// THE READOUT IS COMPLETION, not a release fence. This entry point refuses +// any pNext chain, so it can carry no fence descriptor; what every accepted +// frame does have is the completion path. Every wait here is a timeline +// semaphore ALREADY SIGNALLED to its target before the submit, so nothing +// legitimately holds a frame back: +// +// accepted and completes -> the frame was really submitted. +// accepted and NEVER completes-> the caller holds a success for a frame +// that was dropped: no bitstream, no completion edge, nothing that will +// ever arrive. That is the silent hang, observed rather than inferred. +// refused -> the over-capacity list was reported. +// +// CASE ORDER IS LOAD-BEARING. A flush that refuses a frame discards the whole +// deferred chain around it, so the over-capacity case is run LAST: the two +// arms that must stay green are measured on a session nothing has poisoned. +//============================================================================= + +struct LegacyResult { + VkResult result = VK_SUCCESS; // what the entry point told the caller + VkResult flushResult = VK_SUCCESS; // what the later, flushing call said + bool completed = false; +}; + +// Fill a legacy input frame naming |input|, with |waits| already-satisfied +// timeline waits taken from |bank|. +VkVideoEncodeInputFrame LegacyFrame(const InputImage& input, + const WaitBank& bank, + uint64_t frameId, + bool forceIdr) +{ + VkVideoEncodeInputFrame frame = {}; + frame.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME; + frame.pNext = nullptr; + frame.image = input.image; + frame.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + frame.width = kWidth; + frame.height = kHeight; + // OPTIMAL NV12 is what makes the core route this frame DIRECT, onto the + // fixed-array assembly. LINEAR here would route staged and measure + // nothing. + frame.imageTiling = VK_IMAGE_TILING_OPTIMAL; + frame.currentLayout = VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR; + frame.frameId = frameId; + frame.pts = frameId; + frame.forceIDR = forceIdr ? VK_TRUE : VK_FALSE; + frame.isLastFrame = VK_FALSE; + frame.qpOverride = -1; + // Declared LOCAL, so the direct path records no queue-family transfer and + // the declared layout above stays inert -- the frame under test differs + // from the ones beside it in wait COUNT and nothing else. + frame.inputResidency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + frame.waitSemaphoreCount = (uint32_t)bank.semaphores.size(); + frame.pWaitSemaphores = bank.semaphores.empty() + ? nullptr + : const_cast(bank.semaphores.data()); + frame.pWaitSemaphoreValues = bank.values.empty() + ? nullptr + : const_cast(bank.values.data()); + return frame; +} + +VkResult LegacySubmit(const VkVideoEncodeInputFrame& frame) +{ + VkVideoEncodeInputFrame f = frame; + VkResult status = g_encoder->SubmitExternalFrame(f, nullptr); + for (int retry = 0; (status == VK_NOT_READY) && (retry < 2000); retry++) { + DrainCaptures(); + f = frame; + status = g_encoder->SubmitExternalFrame(f, nullptr); + if (status == VK_NOT_READY) { + struct timespec ts = {0, 1000000}; + nanosleep(&ts, nullptr); + } + } + return status; +} + +LegacyResult RunLegacyWaitCase(const InputImage& input, uint32_t callerWaits) +{ + LegacyResult out; + + WaitBank bank; + // EVERY wait satisfied. This case is not about whether a wait is honoured + // -- the registered arms above measure that -- it is about whether a + // frame its caller was told was accepted ever runs. + if (!bank.Create(callerWaits, callerWaits)) { + Check(false, "the timeline semaphore bank could be created", + "vkCreateSemaphore/vkSignalSemaphore failed"); + bank.Destroy(); + return out; + } + + const uint64_t frameId = g_frameId; + out.result = LegacySubmit(LegacyFrame(input, bank, frameId, false)); + + if (out.result == VK_SUCCESS) { + g_frameId++; + + // FLUSH THE DEFERRED GOP. A forced-IDR frame drains the queue before + // it is itself inserted, so this later call is where the frame above + // is finally recorded and submitted -- and where a refusal it can no + // longer be told about would land. + WaitBank empty; + const uint64_t flushId = g_frameId; + out.flushResult = LegacySubmit(LegacyFrame(input, empty, flushId, true)); + if (out.flushResult == VK_SUCCESS) { + g_frameId++; + } + + // Bounded, because the failure under test is a frame that never + // arrives: an unbounded wait would hang the suite rather than report. + for (int i = 0; (i < 400) && !out.completed; i++) { + const VkVideoEncoderFrameState state = + g_encoder->GetFrameStatus(frameId); + if ((state == VK_VIDEO_ENCODER_FRAME_STATE_READY) || + (state == VK_VIDEO_ENCODER_FRAME_STATE_ACQUIRED)) { + out.completed = true; + break; + } + VkVideoEncodeResult r; + while (g_encoder->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + CountPictureType(r); + if (r.frameId == frameId) { + out.completed = true; + } + g_encoder->ReleaseEncodedFrame(r.frameId); + } + if (out.completed) { + break; + } + struct timespec ts = {0, 5000000}; // 5 ms + nanosleep(&ts, nullptr); + } + } + + DrainCaptures(); + bank.Destroy(); + return out; +} + +//============================================================================= +// Legacy baseline -- a wait list well inside capacity. +// +// Not decoration. Without it, "the 9-wait frame never completed" and "under +// reordering this harness never sees a completion at all" read identically, +// and the case below would prove nothing either way. +//============================================================================= +void CaseLegacySmallWaitListCompletes(const InputImage& input) +{ + g_case = "LegacyBaselineFewWaits"; + + const uint32_t waits = 3; + const LegacyResult r = RunLegacyWaitCase(input, waits); + + std::printf(" [%s] waits=%u result=%d flush=%d completed=%d\n", + g_case, waits, (int)r.result, (int)r.flushResult, + (int)r.completed); + + Check(r.result == VK_SUCCESS, "the legacy submit was accepted", + "VkResult " + I64((long long)r.result)); + Check(r.completed, + "BASELINE: a legacy frame inside capacity completes under reordering", + "the frame was accepted and never became retrievable. If THIS " + "fails, this harness cannot see a completion on this arm at all and " + "the surplus case says nothing"); +} + +//============================================================================= +// Legacy Case A -- THE BOUNDARY THAT MUST STILL WORK. +// +// Exactly |capacity| waits fits the array. This entry point takes no pNext +// chain and therefore no acquire fence, so nothing else is competing for a +// slot: 8 is legal input, and a refusal here would be a regression rather +// than a fix. +//============================================================================= +void CaseLegacyExactlyFullIsStillAccepted(const InputImage& input) +{ + g_case = "LegacyExactlyFullStillWorks"; + + const uint32_t waits = kDirectWaitCapacity; // 8 + const LegacyResult r = RunLegacyWaitCase(input, waits); + + std::printf(" [%s] waits=%u result=%d flush=%d completed=%d\n", + g_case, waits, (int)r.result, (int)r.flushResult, + (int)r.completed); + + Check(r.result == VK_SUCCESS, + "a wait list that exactly fills the direct array is accepted", + "VkResult " + I64((long long)r.result) + " -- a gate that refuses " + "this has traded a silent drop for a refusal of legal input"); + Check(r.completed, "and that frame completes", + "the frame was accepted and never became retrievable"); +} + +//============================================================================= +// Legacy Case B -- MORE CALLER WAITS THAN THE DIRECT ARRAY HOLDS. +// +// capacity+1 waits, every one already satisfied, on a session that reorders. +// The frame has no legitimate reason not to run, and the call that would +// discover it cannot fit has not happened yet when this entry point answers. +// So an accepted frame that never completes is a frame dropped after its +// caller was told it was taken -- and told nothing since. +// +// Run LAST: the flush that refuses it discards the deferred chain around it. +//============================================================================= +void CaseLegacySurplusWaitIsReported(const InputImage& input) +{ + g_case = "LegacySurplusCallerWait"; + + const uint32_t waits = kDirectWaitCapacity + 1; // 9 + const LegacyResult r = RunLegacyWaitCase(input, waits); + + std::printf(" [%s] waits=%u result=%d flush=%d completed=%d\n", + g_case, waits, (int)r.result, (int)r.flushResult, + (int)r.completed); + + const bool acceptedAndLost = (r.result == VK_SUCCESS) && !r.completed; + Check(!acceptedAndLost, + "a wait list the direct array cannot hold is reported to the caller, " + "not accepted and dropped", + "SubmitExternalFrame returned VK_SUCCESS for " + I64(waits) + + " waits into an 8-slot array and the frame never completed " + "(the later flushing call said " + + I64((long long)r.flushResult) + ", which no holder of the " + "success can see). Every wait was already signalled, so nothing " + "was holding it: the caller holds a success for a frame that " + "was never submitted, raises no completion edge, and can only " + "be waited on forever"); + + if (r.result != VK_SUCCESS) { + // The public header names this code on this entry point. A different + // refusal would still avoid the hang but would not be the documented + // contract, so it is asserted rather than merely tolerated. + Check(r.result == VK_ERROR_TOO_MANY_OBJECTS, + "and the refusal is the code the header names", + "VkResult " + I64((long long)r.result) + + ", expected VK_ERROR_TOO_MANY_OBJECTS"); + } +} + +} // namespace + +int main(int argc, char** argv) +{ + bool direct = true; + bool legacy = false; + // THE CODEC IS A PARAMETER OF THE LEGACY ARM ONLY. That arm is the one + // that codes B frames, and the bidirectional reference list a B frame + // needs is assembled by per-codec code: H.264 and H.265 fill + // referenceSlotsInfo[] in separate files, so an arm that exercises one of + // them leaves the other unread. + bool h265 = false; + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--staged") == 0) { + direct = false; + } else if (std::strcmp(argv[i], "--legacy") == 0) { + legacy = true; + } else if (std::strcmp(argv[i], "--h265") == 0) { + h265 = true; + } + } + + std::printf("Encoder-ext DIRECT wait-array capacity (real device) -- %s " + "arm, %s\n", + legacy ? "LEGACY SubmitExternalFrame" + : (direct ? "DIRECT" : "STAGED CONTROL"), + h265 ? "H.265" : "H.264"); + std::printf("------------------------------------------------\n"); + + if ((CreateVulkanVideoEncoderExt(g_encoder) != VK_SUCCESS) || !g_encoder) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + return 77; + } + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = + h265 ? VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR + : VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + config.encodeWidth = kWidth; + config.encodeHeight = kHeight; + config.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + config.inputWidth = kWidth; + config.inputHeight = kHeight; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 5000000; + config.maxBitrate = 5000000; + config.gopLength = 30; + // No reordering on the registered arms: the input-consuming submit is + // issued inline with the frame that produced it, which is what makes the + // release fd readable on the same call and the early poll meaningful. + // + // The LEGACY arm needs the opposite, and needs it to mean anything at + // all. With no reordering the deferred-GOP queue is flushed per frame, so + // the assembly runs on the caller's own thread and an over-capacity wait + // list is refused back to the caller by the assembly itself. Reordering + // moves that flush to a LATER call -- after the entry point has answered + // -- which is the one shape where the entry point's own bound is what + // stands between the caller and a success it can never collect on. + config.consecutiveBFrames = legacy ? 2 : 0; + config.idrPeriod = 30; + config.frameRateNum = 30; + config.frameRateDen = 1; + config.deviceId = -1; + config.disableFileOutput = VK_TRUE; + + if (g_encoder->InitializeExt(config) != VK_SUCCESS) { + std::printf("SKIP: InitializeExt failed -- no encode-capable Vulkan " + "device on this host\n"); + return 77; + } + + VkInstance instance = g_encoder->GetVkInstance(); + VkDevice device = g_encoder->GetVkDevice(); + VkPhysicalDevice phys = g_encoder->GetVkPhysicalDevice(); + g_device = device; + if (!LoadDeviceFns(instance, device, &g_fns)) { + std::printf("SKIP: could not load the Vulkan entry points needed\n"); + return 77; + } + + // Printed, not assumed: a silent fall-through to a software ICD would make + // every number below meaningless. + VkPhysicalDeviceProperties props{}; + g_fns.GetPhysicalDeviceProperties(phys, &props); + std::printf(" device: %s (vendor 0x%04X, driver 0x%08X)\n", + props.deviceName, props.vendorID, props.driverVersion); + + InputImage input; + if (!CreateInputImage(g_fns, phys, device, &input, direct, h265)) { + std::printf("SKIP: could not create the input image\n"); + return 77; + } + + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = kWidth; + desc.height = kHeight; + desc.tiling = direct ? VK_IMAGE_TILING_OPTIMAL + : VK_IMAGE_TILING_LINEAR; + // Declaring VIDEO_ENCODE_SRC on a non-LINEAR image is exactly the + // encodeCapable predicate, and therefore what routes this registration + // DIRECT -- onto the fixed-array submit the defect lives in. + desc.imageUsage = direct + ? (VkImageUsageFlags)( + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT) + : (VkImageUsageFlags) + VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + desc.planeCount = 0; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.defaultLayout = direct ? VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR + : VK_IMAGE_LAYOUT_PREINITIALIZED; + desc.existingImage = input.image; + + // The legacy arm registers NOTHING. SubmitExternalFrame carries the raw + // VkImage and consults no registration, so leaving the descriptor unused + // is what keeps that arm on the lane it is measuring. + if (!legacy) { + const VkVideoEncoderStatusCode regStatus = + g_encoder->RegisterImageResource(desc, 0, &g_resource, nullptr); + if ((regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (g_resource == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + std::printf("SKIP: RegisterImageResource failed, status %d\n", + (int)regStatus); + return 77; + } + } + + if (legacy) { + // Baseline first: it is what makes the surplus case interpretable. + // The surplus case runs LAST because the flush that refuses it + // discards the deferred chain around it. + CaseLegacySmallWaitListCompletes(input); + CaseLegacyExactlyFullIsStillAccepted(input); + CaseLegacySurplusWaitIsReported(input); + } else if (direct) { + // Baselines first: they are what make the rest interpretable. + CaseDirectSmallWaitArrayIsHonoured(false); + CaseDirectSmallWaitArrayIsHonoured(true); + CaseSurplusCallerWaitIsNotDiscarded(true); + CaseAcquireFenceOnAFullWaitArray(); + CaseExactlyFullIsStillAccepted(); + } else { + CaseSurplusCallerWaitIsNotDiscarded(false); + } + + g_encoder->DrainPendingFrames(); + DrainCaptures(); + + if (legacy) { + // THE POSITIVE WITNESS FOR THIS ARM, and the only one there is. + // Everything above asserts a VkResult and a completion, and every one + // of those assertions is equally true of a session that coded no B + // picture -- which is the one shape in which the deferred-GOP flush + // this arm exists to measure never happens, and in which the codec's + // L0/L1 walk and the slot de-duplication over it are never reached. + // consecutiveBFrames is a REQUEST; what the GOP machine and the + // device then type is what decides whether the path under test ran, + // and the delivered picture type is the only reading of that. + std::printf(" picture types coded: I=%u P=%u B=%u\n", + g_pictureTypeCount[0], g_pictureTypeCount[1], + g_pictureTypeCount[2]); + Check(g_pictureTypeCount[2] > 0, + "a B picture was coded on this arm, so the bidirectional " + "reference walk under test ran", + "I=" + I64(g_pictureTypeCount[0]) + " P=" + + I64(g_pictureTypeCount[1]) + " B=" + + I64(g_pictureTypeCount[2]) + + "; with no B picture this arm reorders nothing and every " + "result above is green having exercised no reordering"); + } + + std::printf("------------------------------------------------\n"); + std::printf("%d checks, %d failures\n", g_checks, g_failures); + + g_fns.DestroyImage(device, input.image, nullptr); + g_fns.FreeMemory(device, input.memory, nullptr); + g_encoder = nullptr; + + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-drain-assembly/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-drain-assembly/CMakeLists.txt new file mode 100644 index 00000000..bbea017e --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-drain-assembly/CMakeLists.txt @@ -0,0 +1,114 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_drain_assembly_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The PUBLIC API only, like the sibling release-fence and acquire-fd tests -- +# a real device needs no internal seams. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # The descriptor API is an internal header: the public surface of this + # library is the encoder interface, and a test that drives the layer + # beneath it names the internal directory to say so. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +enable_testing() + +# --------------------------------------------------------------------------- +# WAS A KNOWN FAILURE. FIXED -- WILL_FAIL REMOVED. +# --------------------------------------------------------------------------- +# This test used to exit 1 and carried WILL_FAIL TRUE. It pinned the defect +# described at the top of src/main.cpp: DrainPendingFrames() -> +# WaitForThreadsToComplete() set m_asyncAssemblyEnabled = false and joined the +# only threads that ever call PushCapturedBitstream, and nothing outside +# InitEncoder ever turned it back on -- so no frame submitted after that call +# could produce a completion record, and every one of them held its +# PendingFrame and its resource registration forever. +# +# The fix is VkVideoEncoder::DrainAndRestartThreads(), which DrainPendingFrames +# now calls: it performs the same drain and then restarts the assembly +# workers, so the completion surface survives a drain. The binary exits 0, and +# WILL_FAIL is gone -- it is a plain gating test again. Leaving WILL_FAIL in +# place would have reported the working fix as a FAILURE, which is precisely +# what it was there to force. +# +# Exit 77 on a host with no encode-capable GPU is still reported as SKIPPED: +# the gate recognises that code and emits the skip token CTest matches on. +# +# GATED ON VALIDATION. The drain restarts the assembly workers mid-session and +# the frames submitted after it re-enter the same registration and submission +# path, which is where a lifetime or synchronisation violation would appear. +# vvs_add_validation_gated_test() forces the validation layer on and fails +# the arm that produced a message; no ceiling is declared, so any message at +# all is a failure. +vvs_add_validation_gated_test(EncoderExtDrainPendingFramesIsNotTerminal + TARGET ${PROJECT_NAME}) +set_tests_properties(EncoderExtDrainPendingFramesIsNotTerminal PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +# The SECOND arm -- and it is not optional. +# +# The defect above is a contract violation in BOTH output modes: the single +# completion publisher lives inside WriteBitstreamToFile, which serves the +# capture arm and the file-output arm alike, and DrainPendingFrames() used to +# stop the workers that call it either way. src/main.cpp says as much, and +# says it "therefore runs in file-output mode too, under --file-output, to +# show the divergence" -- but only the capture arm was ever registered, so +# half of the contract the test exists to pin was gated by nothing and had to +# be re-checked by hand every time. +# +# Same label, same gate, same timeout as its sibling, so this Skips rather +# than fails on a host with no encode-capable GPU. The two arms share no state +# -- only this one writes encoder_ext_drain_assembly_out.264 -- so they are +# safe to run in parallel. +vvs_add_validation_gated_test(EncoderExtDrainPendingFramesIsNotTerminalFileOutput + TARGET ${PROJECT_NAME} + ARGS --file-output) +set_tests_properties(EncoderExtDrainPendingFramesIsNotTerminalFileOutput PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +message(STATUS "encoder_ext_drain_assembly_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-drain-assembly/src/main.cpp b/vk_video_encoder/test/encoder-ext-drain-assembly/src/main.cpp new file mode 100644 index 00000000..5a1d4306 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-drain-assembly/src/main.cpp @@ -0,0 +1,518 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * DrainPendingFrames() must NOT be terminal for the COMPLETION SURFACE. + * + * --------------------------------------------------------------------------- + * FIXED. This test passes, and is a plain gating test (no WILL_FAIL). + * --------------------------------------------------------------------------- + * It was registered WILL_FAIL while the defect diagnosed below was open. The + * fix is VkVideoEncoder::DrainAndRestartThreads(): DrainPendingFrames() + * performs the same drain as before and then RESTARTS the assembly workers, + * so the completion surface survives a drain instead of ending at it. The + * WILL_FAIL line is gone from this directory's CMakeLists.txt. + * + * The diagnosis is kept in full below, because it is what this test guards. + * + * THE SHAPE THIS EXISTS FOR. A frame carrying an armed acquireFenceFd is + * accepted, its input IS read (its release fence signals), and it is then + * never assembled into a capture; an unarmed control on the identical path + * retires normally. It presents as an armed loop that submits every frame and + * retires none, against a control that retires all of them, ending with every + * armed frame undelivered and unreleased. + * + * THE ACQUIRE FENCE IS NOT THE VARIABLE. That reading came from running the + * unarmed CONTROL loop first and the ARMED loop second, and + * the control loop ends with DrainPendingFrames(). Everything submitted after + * that call fails to retire, fence or no fence. This file carries NO fence of + * any kind -- every acquireFenceFd here is -1 and no release fd is requested + * -- so if the second batch still fails to retire, the fence is exonerated by + * construction and cannot be the cause. + * + * THE DEFECT, mechanically, three links: + * + * 1. VulkanVideoEncoderExtImpl::DrainPendingFrames() + * (vulkan_video_encoder_ext.cpp:3268) calls + * VkVideoEncoder::WaitForThreadsToComplete(). + * + * 2. WaitForThreadsToComplete() (VkVideoEncoder.cpp:5896-5920) joins the + * assembly workers, clears m_assemblyThreads, and sets + * m_asyncAssemblyEnabled = false (:5913). m_asyncAssemblyEnabled is set + * true at exactly one site, VkVideoEncoder.cpp:3702 inside InitEncoder, + * and the workers are emplaced at exactly one site, :3704, in the same + * block. Neither runs again. Async assembly is off for the rest of the + * session. + * + * 3. With it off, ProcessOrderedFrames (:5719; the branch at :5728, the + * async arm at :5774) skips QueueFramesForAssembly and appends + * AssembleBitstreamData instead. AssembleBitstreamData (:2126) never + * calls PushCapturedBitstream -- and PushCapturedBitstream is the ONLY + * producer of a CapturedBitstream in production code (VkVideoEncoder.cpp + * :2305 inside WriteBitstreamToFile, which only the assembly worker + * calls, and :2422 a direct call in that worker's readback-failure arm; + * VkVideoEncoderAV1.cpp:1050 inside the AV1 override; the ext.cpp:7297 + * site is a device-free unit-test seam, not a live path). + * + * NOTE: the contract, not the line numbers, is what this test rests + * on; encoder-sync-assembly is the bar that pins it directly. + * + * So after DrainPendingFrames() every subsequent frame is encoded and then + * has no completion record made for it. No capture => AcquireNextEncodedFrame + * never yields it => ReleaseEncodedFrame is never called => the PendingFrame + * and the resource registration are held forever and inFlight only grows. + * That is the "16 undelivered/unreleased frames" at teardown, exactly. + * + * THE FIX, against those same three links. Link 2 is the one that was wrong: + * WaitForThreadsToComplete() is the TEARDOWN drain and Flush()/Deinitialize() + * still call it directly. DrainPendingFrames() now calls + * DrainAndRestartThreads(), which runs that same drain -- so link 1 is + * unchanged and everything already submitted still completes -- and then + * calls StartAssemblyThreads() to bring the workers back. Link 3 therefore + * never happens: m_asyncAssemblyEnabled is true again by the time the next + * frame is submitted, ProcessOrderedFrames takes QueueFramesForAssembly, and + * the single publisher inside WriteBitstreamToFile runs for every frame in + * every batch. + * + * WHY NOT "make the synchronous fallback publish too", which looks like the + * smaller change: it cannot work. The + * synchronous assembly runs INLINE inside SetExternalInputFrame(), and + * SubmitExternalFrameCommon only calls EnqueuePendingFrame() -- which creates + * the PendingFrame a record has to land on -- AFTER that call returns. So a + * record published from AssembleBitstreamData arrives before its own + * PendingFrame exists, DrainCapturesLocked() finds nothing to match it to, + * and discards it: batch 2 stayed at retired=0 and gained eight + * "late capture for released frame N discarded" lines. It converts a silent + * drop into a counted late capture, which is worse -- m_lateCaptures exists + * to detect deadline/fence-cap collisions and would now be reporting noise. + * + * THIS IS A CONTRACT VIOLATION IN BOTH OUTPUT MODES, not a capture-mode + * quirk. vulkan_video_encoder_ext.h says of disableFileOutput: "The COMPLETION + * surface does not depend on this flag. In both modes every submitted frame + * raises the completion edge ... and becomes acquirable exactly once." The + * single publisher lives inside WriteBitstreamToFile, which handles BOTH arms + * (VkVideoEncoder.cpp:1851-1856) and which only the assembly worker calls. In + * file-output mode the bytes still reach the file through the synchronous + * fallback, so the encode itself is fine; it is only the completion record + * that is dropped. This test therefore runs in file-output mode too, under + * --file-output, to show the divergence. + * + * And the header says of DrainPendingFrames that it is a "Non-terminal drain + * ... The encoder stays usable for retrieval / Release" -- true for frames + * already captured, false for anything submitted afterwards. + * + * WHY IT NEEDS A GPU. It submits real frames to a real encode queue. With no + * encode-capable device it exits 77, which CTest reads as SKIP. + */ + +#include "vulkan_video_encoder_ext.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; +const char* g_case = "setup"; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + std::printf(" ok [%s] %s\n", g_case, what); + return; + } + g_failures++; + std::printf(" FAIL [%s] %s : %s\n", g_case, what, detail.c_str()); +} + +std::string I64(long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%lld", v); + return buf; +} + +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; +// Small enough to stay far inside admission control (so a NOT_READY retry is +// not the thing under test) and large enough that "none of them retired" is +// not a one-frame coincidence. +const uint32_t kBatch = 8; + +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = nullptr; + PFN_vkGetPhysicalDeviceProperties GetPhysicalDeviceProperties = nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) { + lib = dlopen("libvulkan.so", RTLD_NOW); + } + if (lib == nullptr) { + std::printf(" ERROR: dlopen(libvulkan) failed: %s\n", dlerror()); + return false; + } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) { + return false; + } + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + fns->GetPhysicalDeviceProperties = + (PFN_vkGetPhysicalDeviceProperties)gipa( + instance, "vkGetPhysicalDeviceProperties"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr) && + (fns->GetPhysicalDeviceProperties != nullptr); +} + +struct InputImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; +}; + +// A host-written LINEAR NV12 image, transfer-source usage only, so the +// registration routes STAGED -- the same routing as the acquire-fd and +// release-fence siblings, so the numbers here are comparable to theirs. +bool CreateInputImage(const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, InputImage* out) +{ + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = VK_IMAGE_TILING_LINEAR; + ci.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + std::printf(" ERROR: vkCreateImage(LINEAR NV12) failed\n"); + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + const VkMemoryPropertyFlags want = + (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + std::printf(" ERROR: no host-visible memory type\n"); + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + std::printf(" ERROR: vkAllocateMemory failed\n"); + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != VK_SUCCESS) { + std::printf(" ERROR: vkBindImageMemory failed\n"); + return false; + } + void* mapped = nullptr; + if (fns.MapMemory(device, out->memory, 0, req.size, 0, &mapped) == + VK_SUCCESS) { + uint8_t* bytes = (uint8_t*)mapped; + for (VkDeviceSize i = 0; i < req.size; i++) { + bytes[i] = (uint8_t)((i * 7u) ^ (i >> 9)); + } + fns.UnmapMemory(device, out->memory); + } + return true; +} + +VkSharedBaseObj g_encoder; +VkVideoEncoderResource g_resource = VK_VIDEO_ENCODER_RESOURCE_NULL; +uint64_t g_frameId = 0; +uint32_t g_captured = 0; +uint64_t g_capturedBytes = 0; + +void DrainCaptures() +{ + VkVideoEncodeResult r; + while (g_encoder->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + g_captured++; + g_capturedBytes += r.bitstreamSize; + g_encoder->ReleaseEncodedFrame(r.frameId); + } +} + +// Submit `count` frames carrying NO fence descriptor at all, and return how +// many of them came back as captures. Every submit is plain: no pNext chain, +// no acquireFenceFd, no pReleaseFenceFd, no caller semaphores. There is +// nothing here for a fence to be the cause of. +uint32_t SubmitBatchAndCount(uint32_t count, uint32_t* outSubmitted) +{ + const uint32_t capturedBefore = g_captured; + uint32_t submitted = 0; + for (uint32_t f = 0; f < count; f++) { + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = g_resource; + info.frameId = g_frameId; + info.pts = g_frameId; + info.qpOverride = -1; + info.currentLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + VkVideoEncoderStatusCode status = + g_encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 500); + retry++) { + DrainCaptures(); + struct timespec ts = {0, 1000000}; + nanosleep(&ts, nullptr); + status = g_encoder->SubmitRegisteredFrame(info, nullptr); + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + std::printf(" submit of frame %llu returned %d\n", + (unsigned long long)g_frameId, (int)status); + break; + } + g_frameId++; + submitted++; + DrainCaptures(); + } + g_encoder->DrainPendingFrames(); + DrainCaptures(); + *outSubmitted = submitted; + return g_captured - capturedBefore; +} + +} // namespace + +int main(int argc, char** argv) +{ + bool fileOutput = false; + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--file-output") == 0) { + fileOutput = true; + } + } + + std::printf("DrainPendingFrames() and the completion surface " + "(real device, NO fences anywhere)\n"); + std::printf("mode: %s\n", + fileOutput ? "FILE OUTPUT (disableFileOutput=VK_FALSE)" + : "CAPTURE (disableFileOutput=VK_TRUE)"); + std::printf("--------------------------------------------------------\n"); + + if ((CreateVulkanVideoEncoderExt(g_encoder) != VK_SUCCESS) || !g_encoder) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + return 77; + } + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + config.encodeWidth = kWidth; + config.encodeHeight = kHeight; + config.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + config.inputWidth = kWidth; + config.inputHeight = kHeight; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 5000000; + config.maxBitrate = 5000000; + config.gopLength = 30; + // No reordering, so a frame is submitted on the call that offers it and + // "did it retire" is not confounded by a deferred B-frame batch. + config.consecutiveBFrames = 0; + config.idrPeriod = 30; + config.frameRateNum = 30; + config.frameRateDen = 1; + config.deviceId = -1; + config.disableFileOutput = fileOutput ? VK_FALSE : VK_TRUE; + if (fileOutput) { + config.outputPath = "encoder_ext_drain_assembly_out.264"; + } + + if (g_encoder->InitializeExt(config) != VK_SUCCESS) { + std::printf("SKIP: InitializeExt failed -- no encode-capable Vulkan " + "device on this host\n"); + return 77; + } + + VkInstance instance = g_encoder->GetVkInstance(); + VkDevice device = g_encoder->GetVkDevice(); + VkPhysicalDevice phys = g_encoder->GetVkPhysicalDevice(); + DeviceFns fns; + if (!LoadDeviceFns(instance, device, &fns)) { + std::printf("SKIP: could not load the Vulkan entry points needed\n"); + return 77; + } + + // Printed, not assumed. Nine ICDs are installed on the test host and a + // silent fall-through to a software driver would make every number below + // meaningless. + VkPhysicalDeviceProperties props{}; + fns.GetPhysicalDeviceProperties(phys, &props); + std::printf(" device: %s (vendor 0x%04X, driver 0x%08X)\n", + props.deviceName, props.vendorID, props.driverVersion); + + InputImage input; + if (!CreateInputImage(fns, phys, device, &input)) { + std::printf("SKIP: could not create the input image\n"); + return 77; + } + + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = kWidth; + desc.height = kHeight; + desc.tiling = VK_IMAGE_TILING_LINEAR; + desc.imageUsage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + desc.planeCount = 0; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.defaultLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + desc.existingImage = input.image; + + const VkVideoEncoderStatusCode regStatus = + g_encoder->RegisterImageResource(desc, 0, &g_resource, nullptr); + if ((regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (g_resource == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + std::printf("SKIP: RegisterImageResource failed, status %d\n", + (int)regStatus); + return 77; + } + + //========================================================================= + // BATCH 1 -- the control. Ends with DrainPendingFrames(). + //========================================================================= + g_case = "Batch1BeforeAnyDrain"; + uint32_t submitted1 = 0; + const uint32_t retired1 = SubmitBatchAndCount(kBatch, &submitted1); + std::printf(" batch 1: submitted=%u retired=%u\n", + submitted1, retired1); + + Check(submitted1 == kBatch, "every frame in batch 1 was accepted", + I64(submitted1) + " of " + I64(kBatch)); + // This one MUST hold today. If it does not, the harness is broken and the + // batch-2 result below proves nothing -- so it is asserted first and + // separately, rather than being folded into the interesting claim. + Check(retired1 > 0, + "batch 1 retires, so the completion surface works at all", + I64(retired1) + " captures for " + I64(submitted1) + + " submits -- if this is 0 the rest of this file is not " + "measuring what it claims"); + + //========================================================================= + // BATCH 2 -- IDENTICAL frames, IDENTICAL code path, submitted after + // DrainPendingFrames() has been called once. THIS IS THE DEFECT. + //========================================================================= + g_case = "Batch2AfterDrainPendingFrames"; + uint32_t submitted2 = 0; + const uint32_t retired2 = SubmitBatchAndCount(kBatch, &submitted2); + std::printf(" batch 2: submitted=%u retired=%u\n", + submitted2, retired2); + + Check(submitted2 == kBatch, + "every frame in batch 2 was accepted -- the submit path is fine", + I64(submitted2) + " of " + I64(kBatch)); + + // THE ASSERTION THAT FAILS TODAY. + Check(retired2 > 0, + "batch 2 retires too: DrainPendingFrames() is a drain, not an " + "end-of-stream for the completion surface", + I64(retired2) + " captures for " + I64(submitted2) + + " submits after a DrainPendingFrames(). " + "WaitForThreadsToComplete (VkVideoEncoder.cpp:4303) set " + "m_asyncAssemblyEnabled=false and joined the only threads that " + "ever call PushCapturedBitstream; nothing sets it true again " + "outside InitEncoder (:3054), so no frame submitted after that " + "call can ever produce a completion record. Every one of these " + "frames now holds its PendingFrame and its resource " + "registration forever. NO FENCE WAS INVOLVED IN THIS RUN."); + + g_case = "teardown"; + std::printf(" captured frames=%u bitstream bytes=%llu\n", + g_captured, (unsigned long long)g_capturedBytes); + + // Order matters and the header states it: unregister BEFORE destroying + // the image, and destroy the image BEFORE releasing the encoder. + g_encoder->UnregisterImageResource(g_resource); + fns.DestroyImage(device, input.image, nullptr); + fns.FreeMemory(device, input.memory, nullptr); + g_encoder = nullptr; + + std::printf("--------------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-filter/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-filter/CMakeLists.txt new file mode 100644 index 00000000..9cea2a3c --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-filter/CMakeLists.txt @@ -0,0 +1,113 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_filter_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, not the shared one: the internal header's +# free functions (the format taxonomy, the config binder probe) are +# deliberately not exported from libvkvideo-encoder.so, so only the archive +# can satisfy them. This is also what makes the test a real one -- it links +# and calls the SAME object files the library ships, not a copy. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + # EncoderInputImageParameters::VerifyInputs() is header-only and is the + # subject of the single-plane-gate case. + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT} + # The public encoder header reaches VkCodecUtils/VkVideoRefCountBase.h, + # which lives under the shared common-libs root. + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +# The test branches on whether the preprocess compute filter is compiled in, +# and it must branch the SAME way the library did. Propagating the option +# here rather than re-deriving it is what keeps the two from disagreeing -- +# a test that assumed the filter present against a library built without it +# would assert the wrong half and fail for the wrong reason. +if(BUILD_ENCODER_COMPUTE_FILTER) + target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED) +endif() + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# NOTE ON CTest SEMANTICS, matching the sibling library tests: 0 means every +# assertion held, 1 means an assertion failed, and 2 -- reachable only from +# the registration-gate entry below -- means the session it needs could not be +# stood up at all, deliberately a FAILURE and not a skip. There is no GPU, +# driver or display dependence in either entry, so there is no skip arm and a +# failure to run is a failure. +enable_testing() + +# The taxonomy, the preprocess decision, the transfer-function declaration and +# the colour description. Every function these reach reads only its arguments. +add_test(NAME EncoderExtInputFormatTaxonomy + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtInputFormatTaxonomy PROPERTIES + LABELS "device-free") + +# The colour-model declaration where a producer meets it: the registration +# gate. A separate entry because it is the one group in this binary that +# stands up an encoder session -- a null-backend one, so still device-free -- +# and a session that fails to open must be reported as itself rather than as a +# taxonomy failure. The library answers the same on a session that has +# negotiated nothing, which is what makes the gate testable here at all. +add_test(NAME EncoderExtColorModelRegistrationGate + COMMAND ${PROJECT_NAME} --registration) +set_tests_properties(EncoderExtColorModelRegistrationGate PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_ext_filter_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-filter/src/main.cpp b/vk_video_encoder/test/encoder-ext-filter/src/main.cpp new file mode 100644 index 00000000..588021f8 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-filter/src/main.cpp @@ -0,0 +1,5413 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Device-free coverage for the input-format taxonomy, the preprocess- + * conversion decision the config binder derives from it, the transfer-function + * declaration, and the COLOUR DESCRIPTION the binder and the preprocess filter + * agree on. + * + * The colour half (section 3) was added with the four colour-description + * defects it pins: a partially-supplied declaration that fabricated the + * fields it was not given, an AV1 colour config that dropped full range on + * the floor, an RGBA matrix contract that converted as BT.709 under someone + * else's label, and chroma siting that was declared in one place and + * signalled nowhere. It lives HERE rather than in a new harness because + * three of the four are decided by this same binder, and the fourth is the + * preprocess filter's own contract -- the subject of this file. + * + * WHAT THIS PINS. The adaptation ladder has three + * rungs -- hardware conversion, compute filter, transfer copy -- in that + * order, and states that falling to a copy where the device could have + * converted is a defect. Two library-side decisions implement the second + * rung, and both are pure functions of their arguments: + * + * 1. the FORMAT taxonomy: is a given input directly encodable, encodable + * only after a conversion, or neither; + * 2. the CONFIG binder: does an input that NEEDS a conversion get one + * without asking, does an input that does not need one stay clear of it, + * and is a declared transfer-function mismatch refused rather than + * quietly encoded. + * + * Neither needs a device, an instance or a queue, so both are asserted here + * from a plain process. What this CANNOT assert is that the filter then + * produces correct pixels, or that the queue-family acquire it now records + * is accepted by a real driver -- both need the GPU and are deferred to + * vk_filter_test and vk-video-enc-test. + * + * WHY THIS FILE EXISTS AT ALL. The binder conformance suite that would + * normally carry these assertions (VulkanVideoEncoderConfigBinderTest) lives + * in Chromium, not in this tree, so a change to the field table's + * dispositions had no in-tree test that could fail. It does now. + * + * CTest semantics, matching the sibling library tests: 0 means every + * assertion held, 1 means an assertion failed, 2 means the harness could not + * run at all -- deliberately a FAILURE and not a skip. There is no GPU, + * driver or display dependence here. + */ + +#include "vulkan_video_encoder_ext_internal.h" +#include "VkVideoEncoder/VkEncoderConfig.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Scrub them before anything else sees them -- the +// same block, for the same reason, as the sibling library test TUs. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; +const char* g_currentCase = ""; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + return; + } + g_failures++; + std::printf(" FAIL [%s] %s : %s\n", g_currentCase, what, detail.c_str()); +} + +std::string U32(uint32_t v) +{ + char buf[24]; + std::snprintf(buf, sizeof(buf), "%u", v); + return buf; +} + +// Is the preprocess compute filter compiled into the library this test links? +// The library's own answer, not a guess: the binder refuses the flag when the +// filter is absent, and every expectation below that depends on the flag +// being honourable has to follow the same build gate the library was built +// with. The test and the library are one build here (the test links the +// static archive), so the macro is the same macro. +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED +constexpr bool kFilterCompiledIn = true; +#else +constexpr bool kFilterCompiledIn = false; +#endif + +VkVideoEncoderConfig BaseConfig() +{ + VkVideoEncoderConfig cfg{}; + cfg.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + cfg.pNext = nullptr; + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + cfg.encodeWidth = 1920; + cfg.encodeHeight = 1080; + cfg.inputWidth = 1920; + cfg.inputHeight = 1080; + cfg.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + cfg.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + cfg.averageBitrate = 4000000; + cfg.frameRateNum = 30; + cfg.frameRateDen = 1; + cfg.gopLength = 30; + return cfg; +} + +//============================================================================= +// 1. The input taxonomy (VkEncClassifyInput / VkEncSupportsInput) +//============================================================================= + +void CaseSemiPlanarIsDirect() +{ + g_currentCase = "8- and 10-bit semi-planar is encodable DIRECTLY"; + // Both subsamplings, because the rung is the LAYOUT and the DEPTH and not + // the subsampling: a driver reports semi-planar 4:4:4 as an encode source + // for its 4:4:4 profiles exactly as it reports 4:2:0 for its 4:2:0 ones, + // and the encode profile follows the input rather than the other way + // round. A 4:4:4 input classified anything but DIRECT would be converted + // by the filter into a format the encoder already took. + const VkFormat direct[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, // NV12 + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, // P010 + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, // NV24 + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, // S410 + }; + for (VkFormat f : direct) { + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "classified DIRECT", "format " + U32((uint32_t)f)); + Check(VkEncSupportsInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == VK_TRUE, + "supported", "format " + U32((uint32_t)f)); + Check(VkEncInputFormatPlaneCount(f) == 2, + "2 planes", "format " + U32((uint32_t)f) + " -> " + + U32(VkEncInputFormatPlaneCount(f))); + } +} +void CaseTwelveBitSemiPlanarIsViaFilter() +{ + g_currentCase = "12-bit semi-planar 4:2:0 is encodable only VIA THE FILTER"; + // P012 already has the encode format's plane layout and subsampling, so + // the plane-count argument that puts the 3-plane family on the filter does + // not apply to it. What does apply is bit depth: no encode-source table + // carries a 12-bit row, so the depth has to be converted before the + // encoder sees the frame, and conversion is the filter's work. + const VkFormat f = VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16; + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, + "classified VIA_FILTER", "P012"); + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "NOT classified DIRECT", "P012"); + Check(VkEncInputFormatPlaneCount(f) == 2, + "2 planes -- the layout was never the mismatch", "P012"); +} + + + +void CaseThreePlaneIsViaFilter() +{ + g_currentCase = "3-plane 4:2:0 is encodable only VIA THE FILTER"; + // The distinction the old single VkBool32 could not carry. A copy cannot + // stand in for the conversion: a 3-plane input through a filter-off + // library is VK_ERROR_DEVICE_LOST with a 0-byte bitstream. + const VkFormat viaFilter[] = { + VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, // I420 + VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, + }; + for (VkFormat f : viaFilter) { + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, + "classified VIA_FILTER", "format " + U32((uint32_t)f)); + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "NOT classified DIRECT -- a direct encode of this hangs the GPU", + "format " + U32((uint32_t)f)); + Check(VkEncInputFormatPlaneCount(f) == 3, + "3 planes", "format " + U32((uint32_t)f) + " -> " + + U32(VkEncInputFormatPlaneCount(f))); + } +} + +void CaseRgbaIsViaFilterAndSinglePlane() +{ + g_currentCase = "8-bit RGBA is encodable VIA THE FILTER, and is 1 plane"; + // The filter binds the RGBA source as ONE combined view, declared + // VK_DESCRIPTOR_TYPE_STORAGE_IMAGE and read with imageLoad. The component + // order comes from the view's VkFormat rather than from a GLSL format + // qualifier, which is what lets ONE shader take all three spellings below + // -- notably B8G8R8A8_UNORM, which has no `bgra8` qualifier to declare. + const VkFormat viaFilter[] = { + VK_FORMAT_R8G8B8A8_UNORM, + VK_FORMAT_B8G8R8A8_UNORM, + VK_FORMAT_A8B8G8R8_UNORM_PACK32, + }; + for (VkFormat f : viaFilter) { + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, + "classified VIA_FILTER", "format " + U32((uint32_t)f)); + Check(VkEncResolveColorModel(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, + "identified as RGBA", "format " + U32((uint32_t)f)); + // PLANE COUNT COMES FROM THE FORMAT, NOT FROM ITS CLASS. Answering it from + // the class gives every VIA_FILTER format 3, because the 3-plane family + // dominates it. RGBA + // is ONE plane, and this number is what EncoderConfig::input.numPlanes + // -- and therefore the input geometry -- is written from. + Check(VkEncInputFormatPlaneCount(f) == 1, + "1 plane", "format " + U32((uint32_t)f) + " -> " + + U32(VkEncInputFormatPlaneCount(f))); + } +} + +void CaseDeclaredColorModelDecidesTheClass() +{ + g_currentCase = "the declared colour model decides the class"; + // A VkFormat names a component layout. For every format but the packed + // 4:4:4 Y'CbCr layouts the layout implies exactly one colour model, and + // FROM_FORMAT -- what a zero-initialised structure says -- asks for it. + // Declaring the model the format already implies changes nothing. + const VkFormat nv12 = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + Check(VkEncClassifyInput(nv12, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "NV12 undeclared is DIRECT", "NV12"); + Check(VkEncClassifyInput(nv12, VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "NV12 declared Y'CbCr is DIRECT -- the declaration agrees", "NV12"); + const VkFormat rgba = VK_FORMAT_R8G8B8A8_UNORM; + Check(VkEncClassifyInput(rgba, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, + "RGBA8 undeclared is VIA_FILTER", "R8G8B8A8_UNORM"); + Check(VkEncClassifyInput(rgba, VK_VIDEO_ENCODER_COLOR_MODEL_RGB) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, + "RGBA8 declared RGB is VIA_FILTER -- the declaration agrees", + "R8G8B8A8_UNORM"); + + // A DECLARATION THE FORMAT CANNOT CARRY IS REFUSED, NOT RECONCILED. This + // is the whole of what the declaration buys: without it the library reads + // the format and cannot be told it is wrong, so a caller holding luma and + // chroma in an RGBA-spelled image has no way to say so and no way to be + // refused when it says something impossible. + Check(VkEncClassifyInput(nv12, VK_VIDEO_ENCODER_COLOR_MODEL_RGB) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "RGB declared over a Y'CbCr format is refused", "NV12 as RGB"); + Check(VkEncClassifyInput(VK_FORMAT_B8G8R8A8_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "Y'CbCr declared over an RGBA layout with no packed reading is " + "refused", "BGRA8 as Y'CbCr"); + Check(VkEncResolveColorModel(VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_RGB) == + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, + "an unresolvable pair resolves to no model at all", "I420 as RGB"); + + // THE PACKED 4:4:4 LAYOUTS ARE THE ONE PLACE A DISAGREEMENT IS REAL. AYUV, + // Y410 and Y416 have no Vulkan enumerant of their own and ride the RGBA + // ones, so a Y'CbCr declaration over them is a statement of fact and is + // resolved as such. Whether the library can then ROUTE that input is a + // separate question, asked of VkEncClassifyInput. + Check(VkEncResolveColorModel(rgba, VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "AYUV declared Y'CbCr resolves to Y'CbCr", "R8G8B8A8_UNORM"); + Check(VkEncResolveColorModel(rgba, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, + "the same enumerant undeclared is still RGB", "R8G8B8A8_UNORM"); + Check(VkEncResolveColorModel(VK_FORMAT_A2B10G10R10_UNORM_PACK32, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "Y410 declared Y'CbCr resolves to Y'CbCr", + "A2B10G10R10_UNORM_PACK32"); + Check(VkEncResolveColorModel(VK_FORMAT_R16G16B16A16_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "Y416 declared Y'CbCr resolves to Y'CbCr", "R16G16B16A16_UNORM"); +} + +void CaseSrgbAndJunkAreUnsupported() +{ + g_currentCase = "sRGB, wide/deep RGB and unrelated formats are UNSUPPORTED"; + // Each of these is a POSITIVE refusal, not a gap. + // + // *_SRGB: the filter binds its RGBA source as a storage image and + // reads it with imageLoad, and no *_SRGB format carries the + // storage-image format feature -- sRGB is a sampled-image feature -- + // so an sRGB view can never be the descriptor the filter binds. + // Answering here puts the refusal where a producer can still act on + // it. The _UNORM spellings are the ones taken because the filter + // applies the colour matrix only, with no transfer function, which is + // correct exactly because Y'CbCr is defined on gamma-encoded R'G'B'. + // A2B10G10R10 / R16G16B16A16_UNORM: these same enumerants are how Y410 + // and Y416 are spelled. Undeclared they read as RGB, and as RGB the + // library does not route them; a Y'CbCr declaration over them says + // something else entirely and is judged separately. + // R16G16B16A16_SFLOAT: scRGB -- linear, and not confined to [0,1]. + // + // |planes| is carried per row rather than asserted as a constant zero, + // because plane count is NOT the same question. It is a fact about the + // LAYOUT and carries no colour model: Y410 is one interleaved plane + // whichever model is declared over it, and this library does route it + // under a Y'CbCr declaration, so 0 there would describe an input no + // allocation could be sized from. Everything else here is routed under no + // declaration at all and has no plane count to give. + struct Row { VkFormat format; uint32_t planes; }; + const Row unsupported[] = { + { VK_FORMAT_R8G8B8A8_SRGB, 0 }, + { VK_FORMAT_B8G8R8A8_SRGB, 0 }, + { VK_FORMAT_A8B8G8R8_SRGB_PACK32, 0 }, + { VK_FORMAT_A2B10G10R10_UNORM_PACK32, 1 }, // Y410's enumerant + { VK_FORMAT_R16G16B16A16_UNORM, 0 }, // Y416's, and unrouted + { VK_FORMAT_R16G16B16A16_SFLOAT, 0 }, + { VK_FORMAT_R8G8B8_UNORM, 0 }, // 3-component, no alpha + // YUY2. Refused on its LAYOUT, not its subsampling: it is one + // interleaved plane the generator would read as two. Semi-planar + // 4:2:2 at the same subsampling IS routed -- + // CaseSinglePlaneInterleavedIsRefused drives both halves. + { VK_FORMAT_G8B8G8R8_422_UNORM, 0 }, + { VK_FORMAT_UNDEFINED, 0 }, + }; + for (const Row& row : unsupported) { + const VkFormat f = row.format; + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "classified UNSUPPORTED", "format " + U32((uint32_t)f)); + Check(VkEncSupportsInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == VK_FALSE, + "not supported", "format " + U32((uint32_t)f)); + Check(VkEncInputFormatPlaneCount(f) == row.planes, + "plane count as the layout has it", + "format " + U32((uint32_t)f) + " gave " + + U32(VkEncInputFormatPlaneCount(f)) + ", want " + + U32(row.planes)); + // Declaring the model does not widen the set. Naming RGB over an sRGB + // or scRGB layout states what the layout already says and is still + // refused; naming it over a Y'CbCr format is a contradiction and is + // refused for that reason instead. Either way the answer is the same, + // which is what makes the refusal a decision rather than a gap. + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_RGB) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "still UNSUPPORTED when RGB is declared over it", + "format " + U32((uint32_t)f)); + } +} + +// The advertised input-format list: every format this library can route to an +// encoder input the DEVICE accepts for the profile, each naming what it is +// encoded as. Driven with synthetic device lists because the advertisement is +// a pure function of one, and the interesting lists -- a format the library +// refuses, one format reported twice, a device with no reachable target for a +// routable input -- are not what any one device reports. +// +// NV12 is spelled out per case rather than hoisted, so a case that changes +// the device list changes it visibly. + +// A named format, for assertion text. Not a public mapping: the advertisement +// carries enumerants and this only makes a failure readable. +const char* AdvName(VkFormat f) +{ + switch (f) { + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: return "NV12"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16: return "P010"; + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16: return "P012"; + case VK_FORMAT_G8_B8R8_2PLANE_444_UNORM: return "NV24"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16: return "S410"; + case VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM: return "I420"; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16: return "I420-10"; + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16: return "I420-12"; + case VK_FORMAT_G8_B8R8_2PLANE_422_UNORM: return "NV16"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_422_UNORM_3PACK16: return "P210"; + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_422_UNORM_3PACK16: return "P212"; + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_444_UNORM_3PACK16: return "S412"; + case VK_FORMAT_G8_B8_R8_3PLANE_422_UNORM: return "I422"; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_422_UNORM_3PACK16: return "I422-10"; + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_422_UNORM_3PACK16: return "I422-12"; + case VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM: return "I444"; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_444_UNORM_3PACK16: return "I444-10"; + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_444_UNORM_3PACK16: return "I444-12"; + case VK_FORMAT_R8G8B8A8_UNORM: return "RGBA8"; + case VK_FORMAT_B8G8R8A8_UNORM: return "BGRA8"; + case VK_FORMAT_A8B8G8R8_UNORM_PACK32: return "ABGR8"; + default: return "?"; + } +} + +// Check() takes a C string and the labels below are built per entry, so the +// built label has to outlive the argument list. +const char* Lbl(const std::string& text) +{ + static std::string held; + held = text; + return held.c_str(); +} + +// THE SYNTHETIC ADMISSION, and what it is for. +// +// VkEncAdvertiseInputFormats takes an admission callback rather than a device +// list, because the production caller is a LIVE per-candidate resolve against a +// real driver -- the same one the point query answers from, which is what stops +// the two surfaces drifting. The cases below are device-free, so they supply +// this instead: the rule the function used to contain, stated once here rather +// than twelve times in twelve lambdas. +// +// It reproduces exactly what the old device-list form did: +// - ENCODABLE_DIRECT and on the device list -> OPTIMAL, naming itself; +// - ENCODABLE_VIA_FILTER with a conversion target the device list carries +// -> SUBOPTIMAL, naming the target; +// - anything else -> not advertised. +// +// WHAT IS UNDER TEST HERE IS EVERYTHING BUT THIS RULE: the ordering, the +// uniqueness, the capacity stop and the compute-filter build gate all live +// inside VkEncAdvertiseInputFormats and are exercised through this admission, +// so the assertions in the cases below are unchanged from when the rule was +// inside the function. +struct SyntheticDeviceList { + const VkFormat* formats; + uint32_t count; +}; + +bool DeviceListHas(const VkFormat* formats, uint32_t count, VkFormat format) +{ + for (uint32_t i = 0; i < count; i++) { + if (formats[i] == format) { + return true; + } + } + return false; +} + +bool AdmitFromSyntheticDeviceList( + void* userData, VkFormat candidate, + VkVideoEncoderInputFormatProperties* outEntry) +{ + const SyntheticDeviceList* const dev = + static_cast(userData); + const VkEncInputFormatClass cls = + VkEncClassifyInput(candidate, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT); + if (cls == VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT) { + if (!DeviceListHas(dev->formats, dev->count, candidate)) { + return false; + } + outEntry->format = candidate; + outEntry->encodeFormat = candidate; + outEntry->optimality = VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL; + return true; + } + if (cls == VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER) { + const VkFormat target = + VkEncConversionTargetFormat(candidate, dev->formats, dev->count); + if (target == VK_FORMAT_UNDEFINED) { + return false; + } + if (!DeviceListHas(dev->formats, dev->count, target)) { + return false; + } + outEntry->format = candidate; + outEntry->encodeFormat = target; + outEntry->optimality = VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL; + return true; + } + return false; +} + +uint32_t AdvertiseFromDeviceList( + const VkFormat* deviceFormats, uint32_t deviceFormatCount, + VkVideoEncoderInputFormatProperties* outEntries, uint32_t outCapacity) +{ + SyntheticDeviceList dev = { deviceFormats, deviceFormatCount }; + return VkEncAdvertiseInputFormats(&AdmitFromSyntheticDeviceList, &dev, + outEntries, outCapacity); +} + +// Index of |format| in the advertised list, or -1. +int32_t AdvIndexOf(const VkVideoEncoderInputFormatProperties* entries, + uint32_t count, VkFormat format) +{ + for (uint32_t i = 0; i < count; i++) { + if (entries[i].format == format) { + return (int32_t)i; + } + } + return -1; +} + +void CaseAdvertisedListDropsWhatTheLibraryRefuses() +{ + g_currentCase = "the advertised list drops what the library refuses"; + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, // routed, direct + VK_FORMAT_G8B8G8R8_422_UNORM, // packed 4:2:2, refused + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, // routed, direct + VK_FORMAT_R16G16B16A16_SFLOAT, // refused + }; + VkVideoEncoderInputFormatProperties out[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + deviceList, 4, out, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + + // The refusals are the point: neither may appear anywhere in the answer, + // as an input format or as a conversion target. + for (uint32_t i = 0; i < n; i++) { + Check((out[i].format != VK_FORMAT_G8B8G8R8_422_UNORM) && + (out[i].format != VK_FORMAT_R16G16B16A16_SFLOAT) && + (out[i].encodeFormat != VK_FORMAT_G8B8G8R8_422_UNORM) && + (out[i].encodeFormat != VK_FORMAT_R16G16B16A16_SFLOAT), + "no refused format is advertised, on either side of an entry", + "entry " + U32(i) + " = " + U32((uint32_t)out[i].format) + " -> " + + U32((uint32_t)out[i].encodeFormat)); + } + // And the two the device DOES offer that the library routes unconverted + // are still there, in the device's order -- without which a function that + // answered nothing would pass the loop above. + Check((n >= 2) && (out[0].format == VK_FORMAT_G8_B8R8_2PLANE_420_UNORM) && + (out[0].encodeFormat == out[0].format), + "the first device format the library routes is direct and first", + "got " + U32(n) + " entries"); + Check((n >= 2) && (out[1].format == VK_FORMAT_G8_B8R8_2PLANE_444_UNORM) && + (out[1].encodeFormat == out[1].format), + "the second is direct and second -- the device order is kept", + "got " + U32(n) + " entries"); +} + +void CaseAdvertisedListPassesWhatTheLibraryRoutes() +{ + g_currentCase = "the advertised list keeps every format the library routes"; + // The positive control for the case above: an advertisement that answered + // nothing would pass it just as well. Four direct formats, and every + // converted entry whose target is among them. + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, + }; + VkVideoEncoderInputFormatProperties out[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + deviceList, 4, out, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + // Four direct entries always; the seven converted ones only where the + // filter is compiled in, because the advertisement is gated on it. + Check(n == (kFilterCompiledIn ? 11u : 4u), + "four direct entries, and seven filtered ones where the build has " + "the filter to honour them", + "got " + U32(n)); + for (uint32_t i = 0; (i < n) && (i < 4); i++) { + Check((out[i].format == deviceList[i]) && + (out[i].encodeFormat == deviceList[i]) && + (out[i].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL), + "the direct entries come first, in the device's order", + "entry " + U32(i)); + } + // Each 3-plane input converts into the semi-planar sibling at its own + // subsampling and depth, and this device list carries all four of those + // siblings; the three RGBA spellings convert into the device's first + // choice. THE 4:4:4 PAIR IS HERE BECAUSE THE DEVICE LIST CARRIES NV24 AND + // S410 -- take those two out of the list and both entries go with them, + // which is what the 4:2:0 case below asserts. + struct Expect { VkFormat in; VkFormat out; }; + const Expect converted[] = { + { VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM }, + { VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM }, + { VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16 }, + { VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_444_UNORM_3PACK16, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16 }, + { VK_FORMAT_R8G8B8A8_UNORM, VK_FORMAT_G8_B8R8_2PLANE_420_UNORM }, + { VK_FORMAT_B8G8R8A8_UNORM, VK_FORMAT_G8_B8R8_2PLANE_420_UNORM }, + { VK_FORMAT_A8B8G8R8_UNORM_PACK32, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM }, + }; + for (const Expect& e : converted) { + const int32_t at = AdvIndexOf(out, n, e.in); + if (!kFilterCompiledIn) { + // The build cannot convert, so the session would refuse each of + // these. Advertising them would be the advertise-then-refuse + // case the gate exists to prevent, so absence IS the assertion. + Check(at < 0, + Lbl(std::string(AdvName(e.in)) + + " is NOT advertised without the filter"), + "index " + U32((uint32_t)(at + 1))); + continue; + } + Check(at >= 0, Lbl(std::string(AdvName(e.in)) + " is advertised"), + "not in the list"); + if (at >= 0) { + Check(out[at].encodeFormat == e.out, + Lbl(std::string(AdvName(e.in)) + " names " + + AdvName(e.out) + " as what it is encoded as"), + "names " + U32((uint32_t)out[at].encodeFormat)); + Check(out[at].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL, + Lbl(std::string(AdvName(e.in)) + + " is advertised as filtered"), + "optimality " + U32((uint32_t)out[at].optimality)); + } + } +} + +void CaseAdvertisedListReportsOneFormatOnce() +{ + g_currentCase = "one format reported at two tilings is advertised once"; + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + }; + VkVideoEncoderInputFormatProperties out[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + deviceList, 4, out, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + uint32_t nv12Entries = 0; + for (uint32_t i = 0; i < n; i++) { + if (out[i].format == VK_FORMAT_G8_B8R8_2PLANE_420_UNORM) { + nv12Entries++; + } + } + Check(nv12Entries == 1, + "a format the device reports three times is advertised once", + "appears " + U32(nv12Entries) + " times"); + Check((n >= 2) && (out[0].format == VK_FORMAT_G8_B8R8_2PLANE_420_UNORM) && + (out[1].format == + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16), + "and the distinct direct entries keep the device's order", ""); +} + +void CaseAdvertisedListStopsAtCapacity() +{ + g_currentCase = "the advertised list never writes past its capacity"; + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + }; + VkVideoEncoderInputFormatProperties out[4] = {}; + const uint32_t n = AdvertiseFromDeviceList(deviceList, 3, out, 2); + Check(n == 2, "clamped to the capacity given", "got " + U32(n)); + Check((out[2].format == VK_FORMAT_UNDEFINED) && + (out[2].encodeFormat == VK_FORMAT_UNDEFINED) && + (out[2].optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL), + "and wrote nothing past it", + "slot 2 = " + U32((uint32_t)out[2].format)); +} + +// THE MEMBERSHIP RULE THAT KEEPS A PRODUCER OUT OF A REFUSED SESSION. +// +// Two of the formats the taxonomy routes -- 12-bit I420 and P012 -- convert +// into P012. Where P012 is an encode source on no profile the device exposes, +// advertising either would put an entry a producer can size a pool from in +// front of a caller the session then refuses, so neither is advertised. Where +// the device DOES take P012, both are, and the difference is read from the +// device list rather than from a table in this library. +void CaseAdvertisedListDropsAnUnreachableConversionTarget() +{ + g_currentCase = "an entry whose conversion target the device will not " + "take is not advertised"; + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + }; + VkVideoEncoderInputFormatProperties out[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + deviceList, 2, out, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + Check(AdvIndexOf(out, n, + VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16) < 0, + "I420-12 is not advertised: it converts into P012, which this " + "device list does not carry", + "it is in the list"); + Check(AdvIndexOf(out, n, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16) < 0, + "P012 is not advertised: it converts into P012, which this device " + "list does not carry either", + "it is in the list"); + + // THE POSITIVE CONTROL, on the same function and the same shape: add P012 + // to what the device takes and BOTH entries that route into it appear. + // P012 is one of them -- its route is the filter and its target is + // itself, which the entry states in its optimality rather than leaving to + // be inferred from two equal formats. + const VkFormat twelveBitDevice[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + }; + VkVideoEncoderInputFormatProperties wide[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t wn = AdvertiseFromDeviceList( + twelveBitDevice, 3, wide, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + const int32_t at = AdvIndexOf( + wide, wn, VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16); + // Only meaningful where converted entries are advertised at all; without + // the filter the answer is the build's, not the device's, and the case + // above already asserts that. + Check(kFilterCompiledIn ? (at >= 0) : (at < 0), + "on a device that DOES take P012, I420-12 is advertised -- the " + "exclusion is the device's answer, not a hardcoded one (and it is " + "absent altogether in a build without the filter)", + "index " + U32((uint32_t)(at + 1))); + if (at >= 0) { + Check(wide[at].encodeFormat == + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + "and it names P012 as what it is encoded as", + "names " + U32((uint32_t)wide[at].encodeFormat)); + Check(wide[at].optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL, + "and it is filtered", "optimality " + + U32((uint32_t)wide[at].optimality)); + } + const int32_t p012 = AdvIndexOf( + wide, wn, VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16); + // Same shape as the row above: P012 is a converted entry, so whether the + // device takes its target only decides the answer in a build that can + // convert at all. + Check(kFilterCompiledIn ? (p012 >= 0) : (p012 < 0), + "P012 is advertised there too: the device takes its conversion " + "target, so it is a route -- and it is absent altogether where the " + "build has no filter", + "index " + U32((uint32_t)(p012 + 1))); + if (p012 >= 0) { + Check(wide[p012].encodeFormat == wide[p012].format, + "P012 names itself as what it is encoded as", + "names " + U32((uint32_t)wide[p012].encodeFormat)); + Check(wide[p012].optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_SUBOPTIMAL, + "and it is FILTERED, which is what the two equal formats " + "cannot say", + "optimality " + U32((uint32_t)wide[p012].optimality)); + } +} + +// WHY THIS EXISTS. NV24 and S410 are the two strongest claims the taxonomy +// makes -- ENCODABLE_DIRECT, and on the routable list -- and nothing in this +// tree has ever encoded either. This case records exactly what stands between +// them and a caller, so the answer is measured rather than re-derived, and so +// a change to any of the three gates shows up here. +// +// The three gates, and which one actually blocks them: +// 1. the converted arm requires ENCODABLE_VIA_FILTER and they are DIRECT, +// so that arm skips them and their routable-list entries are inert; +// 2. VkEncConversionTargetFormat names no target for them, so even a +// VIA_FILTER classification would not hand them a route; +// 3. the OPTIMAL arm is the device list -- which is the ONLY way either can +// be advertised, and so is where the block actually lives. +// +// CALIBRATION. The same call is driven with two device lists differing only +// in whether they carry 4:4:4, and it must answer differently: absent on the +// first, present on the second. A check that passed both would be asserting +// nothing about reachability at all. +void CaseFourFourFourReachesTheListOnlyFromTheDevice() +{ + g_currentCase = "4:4:4 is advertised only where the device reports it, " + "and never off the routable list"; + + const VkFormat nv24 = VK_FORMAT_G8_B8R8_2PLANE_444_UNORM; + const VkFormat s410 = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16; + + // Gate 1: DIRECT, so the converted arm's VIA_FILTER test skips them. + Check(VkEncClassifyInput(nv24, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "NV24 classifies DIRECT, so the converted arm cannot emit it", + "class " + U32((uint32_t)VkEncClassifyInput( + nv24, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT))); + Check(VkEncClassifyInput(s410, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "S410 classifies DIRECT, likewise", + "class " + U32((uint32_t)VkEncClassifyInput( + s410, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT))); + + // Gate 2: and no conversion produces them either. + const VkFormat narrow[] = { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM }; + Check(VkEncConversionTargetFormat(nv24, narrow, 1) == VK_FORMAT_UNDEFINED, + "NV24 names no conversion target, so it has no filter route", + "target " + U32((uint32_t)VkEncConversionTargetFormat(nv24, narrow, 1))); + Check(VkEncConversionTargetFormat(s410, narrow, 1) == VK_FORMAT_UNDEFINED, + "S410 names no conversion target either", + "target " + U32((uint32_t)VkEncConversionTargetFormat(s410, narrow, 1))); + + // Gate 3, ABSENT HALF. A device reporting no 4:4:4 gets no 4:4:4 entry, + // although both formats are on the routable list this walk covers. + VkVideoEncoderInputFormatProperties without[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t wn = AdvertiseFromDeviceList( + narrow, 1, without, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + Check(wn > 0, "the 4:2:0-only device still advertises something, so the " + "absence below is a filter and not an empty list", + "got " + U32(wn)); + uint32_t found444 = 0; + for (uint32_t i = 0; i < wn; i++) { + if ((without[i].format == nv24) || (without[i].format == s410)) { + found444++; + } + } + Check(found444 == 0, + "no 4:4:4 entry is advertised when the device reports none", + "found " + U32(found444)); + + // Gate 3, PRESENT HALF. The same call on a device that does report them + // advertises both, so the advertisement is not what withholds them. + const VkFormat wide444[] = { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, nv24, s410 }; + VkVideoEncoderInputFormatProperties withList[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t gn = AdvertiseFromDeviceList( + wide444, 3, withList, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + int nv24At = -1; + int s410At = -1; + for (uint32_t i = 0; i < gn; i++) { + if (withList[i].format == nv24) { nv24At = (int)i; } + if (withList[i].format == s410) { s410At = (int)i; } + } + Check(nv24At >= 0, "NV24 IS advertised once the device reports it", + "index " + U32((uint32_t)(nv24At + 1))); + Check(s410At >= 0, "S410 IS advertised once the device reports it", + "index " + U32((uint32_t)(s410At + 1))); + if (nv24At >= 0) { + Check(withList[nv24At].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "and it is OPTIMAL, read off the device rather than the list", + "optimality " + U32((uint32_t)withList[nv24At].optimality)); + Check(withList[nv24At].encodeFormat == nv24, + "naming itself, because nothing converted it", + "names " + U32((uint32_t)withList[nv24At].encodeFormat)); + } + if (s410At >= 0) { + Check(withList[s410At].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "S410 likewise OPTIMAL", + "optimality " + U32((uint32_t)withList[s410At].optimality)); + } +} + +// WHY THIS EXISTS. Every SUBOPTIMAL entry is ENCODABLE_VIA_FILTER, and +// InitializeExt refuses exactly that class when the preprocess filter is not +// compiled in. Advertising them in such a build would hand a caller a format +// and then refuse the session declaring it -- the single failure this list is +// supposed to make impossible. The advertisement is gated on the same build +// condition, and this is what holds the two together. +// +// CALIBRATION, AND IT IS A TWO-BUILD ONE. This case is deliberately NOT +// compiled out with the filter: it runs in both configurations and branches +// on kFilterCompiledIn, so the configuration that matters -- the one without +// the filter -- is the one where it still asserts. A check that vanished +// alongside the feature would prove nothing about the build it was written +// for. +// +// The OPTIMAL expectation is stated as an absolute and is NOT conditioned on +// kFilterCompiledIn. That is the point of it: those entries are +// ENCODABLE_DIRECT and involve no filter, so the same four must appear in +// both builds. Conditioning it would let the gate quietly take the direct arm +// with it and still pass. +void CaseFilterlessBuildAdvertisesNoConvertedEntry() +{ + g_currentCase = "a build without the preprocess filter advertises no " + "converted entry, and the same direct ones"; + + // All four DIRECT formats, so the OPTIMAL arm is fully exercised, and + // NV12/P010 present so several converted entries resolve. + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, // NV12 + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, // P010 + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, // NV24 + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, // S410 + }; + VkVideoEncoderInputFormatProperties out[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + deviceList, 4, out, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + + uint32_t optimalCount = 0; + uint32_t suboptimalCount = 0; + for (uint32_t i = 0; i < n; i++) { + if (out[i].optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) { + optimalCount++; + } else { + suboptimalCount++; + } + } + + // UNCONDITIONAL. Four DIRECT formats on the device list, four OPTIMAL + // entries, in both builds. + Check(optimalCount == 4, + "the direct arm advertises all four device formats, whether or not " + "the filter is compiled in", + "optimal " + U32(optimalCount)); + for (uint32_t j = 0; j < 4; j++) { + bool present = false; + for (uint32_t i = 0; i < n; i++) { + present = present || + ((out[i].format == deviceList[j]) && + (out[i].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL)); + } + Check(present, + Lbl(std::string(AdvName(deviceList[j])) + + ": advertised OPTIMAL in either build"), + "device entry " + U32(j)); + } + + // CONDITIONAL, and this is the gate itself. + if (kFilterCompiledIn) { + Check(suboptimalCount == 7, + "with the filter, the converted entries this device can reach " + "are advertised -- the four 3-plane inputs whose semi-planar " + "sibling is on this list, and the three RGBA spellings", + "suboptimal " + U32(suboptimalCount)); + } else { + Check(suboptimalCount == 0, + "without the filter, no converted entry is advertised -- the " + "session would refuse every one of them", + "suboptimal " + U32(suboptimalCount)); + } + + // And the totals follow from the two above, stated so a drift in either + // shows up as a count rather than only as a membership failure. + Check(n == (kFilterCompiledIn ? 11u : 4u), + "the advertised total is the direct set plus the converted set the " + "build can actually honour", + "total " + U32(n)); +} +// The route is stated per entry, so a caller never has to infer it -- and the +// inference it would otherwise make is wrong on exactly the entry whose +// conversion target is its own format. +void CaseOptimalityNamesTheEncodersOwnFormat() +{ + g_currentCase = "optimality names the encoder-read set, on " + "every entry of every list"; + const VkFormat deviceList[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + }; + VkVideoEncoderInputFormatProperties out[VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + deviceList, 4, out, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + Check(n > 0, "the list is non-empty, so the loop below measures something", + "got " + U32(n)); + uint32_t equalButFiltered = 0; + for (uint32_t i = 0; i < n; i++) { + bool inDeviceList = false; + for (uint32_t j = 0; j < 4; j++) { + inDeviceList = inDeviceList || (deviceList[j] == out[i].format); + } + const bool routedUnconverted = + inDeviceList && + (VkEncClassifyInput(out[i].format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT); + Check((out[i].optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) == + routedUnconverted, + Lbl(std::string(AdvName(out[i].format)) + + ": optimality is OPTIMAL exactly where the device takes it " + "unconverted"), + "entry " + U32(i) + " optimality " + + U32((uint32_t)out[i].optimality)); + // An OPTIMAL entry always names itself; a SUBOPTIMAL one may too. + if (out[i].optimality == VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) { + Check(out[i].format == out[i].encodeFormat, + Lbl(std::string(AdvName(out[i].format)) + + ": a direct entry names itself"), + "names " + U32((uint32_t)out[i].encodeFormat)); + } else if (out[i].format == out[i].encodeFormat) { + equalButFiltered++; + } + // Every advertised target must be something the device takes, + // whichever route the entry takes to it. + bool targetInDeviceList = false; + for (uint32_t j = 0; j < 4; j++) { + targetInDeviceList = + targetInDeviceList || (deviceList[j] == out[i].encodeFormat); + } + Check(targetInDeviceList, + Lbl(std::string(AdvName(out[i].format)) + + ": what it is encoded as is a format the device accepts"), + "target " + U32((uint32_t)out[i].encodeFormat)); + } + // THE COLLISION, DRIVEN. P012 is on this device list and is advertised + // with encodeFormat == format, so a caller reading the equality alone + // would call it direct. Exactly one entry is in that position, and it is + // FILTERED. + // The collision needs a converted entry to exist, so it is a claim about + // a build that has the filter. Without one there is no filtered entry at + // all, which the count below states rather than skips. + Check(equalButFiltered == (kFilterCompiledIn ? 1u : 0u), + "one advertised entry names its own format and is still filtered, " + "which is the case the equality cannot answer -- and none at all " + "where no converted entry is advertised", + "found " + U32(equalButFiltered)); +} + +// The conversion target is DERIVED, and this is the derivation stated as a +// property rather than as a table: the filter's Y'CbCr arm changes plane +// layout and packing and resamples neither chroma nor bit depth, so the +// target must agree with its input on both and must be semi-planar. +void CaseConversionTargetPreservesSubsamplingAndDepth() +{ + g_currentCase = "a Y'CbCr conversion target keeps the subsampling and " + "the bit depth of its input"; + uint32_t routableCount = 0; + const VkFormat* routable = VkEncRoutableInputFormats(routableCount); + uint32_t checked = 0; + for (uint32_t i = 0; i < routableCount; i++) { + const VkFormat in = routable[i]; + const VkMpFormatInfo* inInfo = YcbcrVkFormatInfo(in); + if (inInfo == nullptr) { + continue; // the RGB half; its target is the device's choice + } + // Asked with a device list that carries every semi-planar target, so + // the derivation is exercised rather than the membership rule. + const VkFormat targets[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, + }; + const VkFormat target = VkEncConversionTargetFormat(in, targets, 5); + if (target == VK_FORMAT_UNDEFINED) { + // Direct inputs convert into nothing, which is the honest answer. + Check(VkEncClassifyInput( + in, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + Lbl(std::string(AdvName(in)) + + ": only a directly encodable input has no target"), + "no target for a converted input"); + continue; + } + const VkMpFormatInfo* outInfo = YcbcrVkFormatInfo(target); + Check(outInfo != nullptr, + Lbl(std::string(AdvName(in)) + ": its target is a Y'CbCr format"), + "target " + U32((uint32_t)target)); + if (outInfo == nullptr) { + continue; + } + Check(outInfo->planesLayout.bpp == inInfo->planesLayout.bpp, + Lbl(std::string(AdvName(in)) + ": the target keeps the bit depth"), + "in " + U32(inInfo->planesLayout.bpp) + ", out " + + U32(outInfo->planesLayout.bpp)); + Check((outInfo->planesLayout.secondaryPlaneSubsampledX == + inInfo->planesLayout.secondaryPlaneSubsampledX) && + (outInfo->planesLayout.secondaryPlaneSubsampledY == + inInfo->planesLayout.secondaryPlaneSubsampledY), + Lbl(std::string(AdvName(in)) + ": the target keeps the subsampling"), + ""); + Check(outInfo->planesLayout.numberOfExtraPlanes == 1u, + Lbl(std::string(AdvName(in)) + ": the target is semi-planar"), + "extra planes " + + U32(outInfo->planesLayout.numberOfExtraPlanes)); + checked++; + } + // Twelve: the 3-plane family at three subsamplings and three depths, + // plus the three semi-planar 12-bit rows whose target is themselves. The + // number is stated so that a derivation that quietly stopped naming + // targets shows up as a count rather than only as a silent pass over an + // empty loop. + Check(checked == 12, + "twelve Y'CbCr inputs have a conversion target", + "checked " + U32(checked)); +} + +// The routable list and the classifier are two statements of one set, and a +// change to either that does not change the other is what this catches. +void CaseRoutableListAgreesWithTheClassifier() +{ + g_currentCase = "the routable list and the classifier name the same set"; + uint32_t routableCount = 0; + const VkFormat* routable = VkEncRoutableInputFormats(routableCount); + for (uint32_t i = 0; i < routableCount; i++) { + Check(VkEncClassifyInput(routable[i], + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != + VK_ENC_INPUT_FORMAT_UNSUPPORTED, + Lbl(std::string(AdvName(routable[i])) + + " is on the routable list and the classifier routes it"), + "the classifier says UNSUPPORTED"); + } + // The other direction, over the formats this suite reaches: nothing the + // classifier routes may be missing from the list, or the advertisement + // would silently never offer it. Population: 22 candidates. This is the + // SPOT check; CaseRoutableSetIsDerivedFromTheFormatTables walks the whole + // table in both directions and is what actually holds the predicate. + const VkFormat candidates[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, + VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_R8G8B8A8_UNORM, + VK_FORMAT_B8G8R8A8_UNORM, + VK_FORMAT_A8B8G8R8_UNORM_PACK32, + // The neighbours. Some are refused for a stated reason elsewhere in + // this file -- the two sRGB spellings, the two packed 4:4:4 aliases, + // scRGB, packed 4:2:2, and the 16-bit rows -- and some are ROUTED, + // which is the half of this loop that matters: a routed format that + // is off the list fails here. Semi-planar and 3-plane 4:2:2 and + // 3-plane 4:4:4 are in the second group since the set was derived. + VK_FORMAT_R8G8B8A8_SRGB, + VK_FORMAT_B8G8R8A8_SRGB, + VK_FORMAT_A2B10G10R10_UNORM_PACK32, + VK_FORMAT_R16G16B16A16_UNORM, + VK_FORMAT_R16G16B16A16_SFLOAT, + VK_FORMAT_G8B8G8R8_422_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, + VK_FORMAT_G8_B8_R8_3PLANE_422_UNORM, + VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM, + VK_FORMAT_G16_B16R16_2PLANE_420_UNORM, + VK_FORMAT_G16_B16_R16_3PLANE_420_UNORM, + }; + uint32_t routedOffList = 0; + for (VkFormat f : candidates) { + if (VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + continue; + } + bool onList = false; + for (uint32_t i = 0; i < routableCount; i++) { + onList = onList || (routable[i] == f); + } + if (!onList) { + routedOffList++; + Check(false, + "a format the classifier routes is on the routable list", + "format " + U32((uint32_t)f) + " is routed but not listed"); + } + } + Check(routedOffList == 0, + "0 of 22 candidate formats are routed without being on the list", + U32(routedOffList) + " were"); +} + +// THE DERIVATION, PINNED. The routable set is COMPUTED from the multi-planar +// Y'CbCr format table rather than listed, so what a test can hold it to is the +// PREDICATE. Asserting a copy of the answer would only move the literal this +// replaced into the test file. +// +// Walked over the whole table in BOTH directions -- every row the predicate +// admits is on the list, every row it refuses is off it -- which is what makes +// this a test of the rule rather than a spot check of six formats. +// +// The predicate is restated here from the two table fields it reads, and +// deliberately NOT by calling the library's own helpers: a test that asked the +// implementation what the implementation does would agree with any answer. +void CaseRoutableSetIsDerivedFromTheFormatTables() +{ + g_currentCase = "the routable set is the format table filtered by the " + "predicate, in both directions"; + uint32_t routableCount = 0; + const VkFormat* routable = VkEncRoutableInputFormats(routableCount); + + uint32_t rows = 0; + uint32_t admitted = 0; + uint32_t refused = 0; + for (uint32_t i = 0;; i++) { + const VkMpFormatInfo* mpInfo = YcbcrVkFormatInfoByIndex(i); + if (mpInfo == nullptr) { + break; + } + rows++; + const uint32_t layout = mpInfo->planesLayout.layout; + const uint32_t bits = GetBitsPerChannel(mpInfo->planesLayout); + const bool modelledLayout = + (layout == YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED) || + (layout == YCBCR_PLANAR_STRIDE_PADDED); + const bool encodableDepth = + (bits == 8u) || (bits == 10u) || (bits == 12u); + const bool direct = (layout == YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED) && + ((bits == 8u) || (bits == 10u)); + // A converted row is routed only where the table names a semi-planar + // sibling at its own depth and subsampling to convert into. Asked of + // the table here too, so the target rule is pinned alongside the + // membership rule rather than assumed. + bool hasSibling = false; + for (uint32_t j = 0;; j++) { + const VkMpFormatInfo* other = YcbcrVkFormatInfoByIndex(j); + if (other == nullptr) { + break; + } + hasSibling = hasSibling || + ((other->planesLayout.layout == + YCBCR_SEMI_PLANAR_CBCR_INTERLEAVED) && + (GetBitsPerChannel(other->planesLayout) == bits) && + (other->planesLayout.secondaryPlaneSubsampledX == + mpInfo->planesLayout.secondaryPlaneSubsampledX) && + (other->planesLayout.secondaryPlaneSubsampledY == + mpInfo->planesLayout.secondaryPlaneSubsampledY)); + } + const bool expectRouted = + modelledLayout && encodableDepth && (direct || hasSibling); + + bool onList = false; + for (uint32_t k = 0; k < routableCount; k++) { + onList = onList || (routable[k] == mpInfo->vkFormat); + } + Check(onList == expectRouted, + Lbl("table row " + U32(i) + " (format " + + U32((uint32_t)mpInfo->vkFormat) + ", " + + AdvName(mpInfo->vkFormat) + + ") is on the routable list exactly when the predicate " + "admits it"), + std::string(onList ? "listed" : "absent") + ", predicate says " + + (expectRouted ? "route" : "refuse")); + // And the classifier has to agree with the list, since both are the + // same derivation read twice. + Check((VkEncClassifyInput(mpInfo->vkFormat, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != + VK_ENC_INPUT_FORMAT_UNSUPPORTED) == expectRouted, + Lbl(std::string("row ") + U32(i) + + ": the classifier answers the same predicate"), + "class " + U32((uint32_t)VkEncClassifyInput( + mpInfo->vkFormat, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT))); + if (expectRouted) { + admitted++; + } else { + refused++; + } + } + + // CALIBRATION. The walk has to see BOTH answers, or the loop above would + // be asserting one of them over an empty population and would pass for a + // derivation that routed everything, or nothing. + Check(rows == (uint32_t)YCBCR_VK_FORMAT_INFO_TABLE_SIZE, + "the walk visits every row of the multi-planar table", + U32(rows) + " rows"); + Check(admitted > 0, "the predicate admits some rows", U32(admitted)); + Check(refused > 0, "and refuses others", U32(refused)); + + // The RGB spellings are the rest of the list, and they are not in this + // table at all: an RGB layout is precisely what it does not describe. + Check(routableCount == (admitted + 3u), + "the list is the admitted table rows plus the three RGB spellings", + U32(routableCount) + " listed, " + U32(admitted) + " admitted"); + + for (uint32_t i = 0; i < routableCount; i++) { + uint32_t seen = 0; + for (uint32_t j = 0; j < routableCount; j++) { + if (routable[j] == routable[i]) { + seen++; + } + } + Check(seen == 1, "each routable format appears exactly once", + "format " + U32((uint32_t)routable[i]) + " appears " + + U32(seen)); + } +} + +// THE PACKED 4:2:2 FAMILY IS REFUSED, AND ON ITS LAYOUT. Singled out because +// it is the one exclusion that is about the shader generator being WRONG +// rather than about a Vulkan or codec limit: those rows are ONE plane and the +// table gives them numberOfExtraPlanes = 1, so the generator declares a +// two-plane read over a single-plane image and its luma addressing assumes one +// sample per texel on a format carrying two. The shader compiles either way, +// so nothing downstream would catch it -- a widening that dropped the layout +// predicate would admit them silently and produce a plausible wrong picture. +void CaseSinglePlaneInterleavedIsRefused() +{ + g_currentCase = "the packed 4:2:2 family is refused, at every depth"; + const VkFormat packed422[] = { + VK_FORMAT_G8B8G8R8_422_UNORM, // YUY2 + VK_FORMAT_B8G8R8G8_422_UNORM, // UYVY + VK_FORMAT_G10X6B10X6G10X6R10X6_422_UNORM_4PACK16, // Y210 + VK_FORMAT_G12X4B12X4G12X4R12X4_422_UNORM_4PACK16, // Y212 + VK_FORMAT_G16B16G16R16_422_UNORM, // Y216 + }; + for (VkFormat f : packed422) { + // The row EXISTS in the table, so the refusal is the predicate's and + // not a lookup miss. That distinction is the whole point: a format the + // table does not describe is refused by accident. + Check(YcbcrVkFormatInfo(f) != nullptr, + "the format table describes it, so the refusal is a decision", + "format " + U32((uint32_t)f)); + Check(VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "classified UNSUPPORTED", "format " + U32((uint32_t)f)); + Check(VkEncConversionTargetFormat(f, nullptr, 0) == + VK_FORMAT_UNDEFINED, + "and names no conversion target", "format " + + U32((uint32_t)f)); + } + // CALIBRATION, and it is the one that matters: the SEMI-PLANAR row at the + // same 4:2:2 subsampling IS routed. Without it this case would pass just + // as well for a library that refused 4:2:2 outright, which is a different + // and weaker claim. + Check(VkEncClassifyInput(VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != + VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "semi-planar 4:2:2 at the same subsampling IS routed, so the " + "refusals above are the layout's and not the subsampling's", + "NV16 was refused too"); +} + +// WHAT A 4:2:0 DEVICE SEES, AND THAT NEITHER DERIVING THE ROUTABLE SET NOR +// PARAMETERISING THE ADMISSION MOVED IT. +// +// The device lists below are 4:2:0 and are supplied by this file, not read off +// a driver: what they record is the advertised answer as it stood before the +// routable set was derived -- not re-derived here, which would let the record +// and the code move together. They are still the right record after the +// admission became a callback, because a 4:2:0 candidate resolves at a 4:2:0 +// profile and this synthetic admission answers exactly what the device-list +// form used to answer. +// +// The rows are the whole answer, in order, so a change that ADDED an entry +// fails as loudly as one that dropped it. +void CaseFourTwoZeroDeviceAdvertisesTheHistoricalSet() +{ + g_currentCase = "a 4:2:0 device list advertises exactly what it did " + "before the routable set was derived"; + struct Row { VkFormat in; VkFormat out; bool optimal; }; + struct Scenario { + const char* name; + const VkFormat* device; + uint32_t deviceCount; + const Row* expected; + uint32_t expectedCount; + }; + + static const VkFormat kNv12[] = { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM }; + static const Row kNv12Expected[] = { + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, true }, + { VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + { VK_FORMAT_R8G8B8A8_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + { VK_FORMAT_B8G8R8A8_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + { VK_FORMAT_A8B8G8R8_UNORM_PACK32, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + }; + + static const VkFormat k420Full[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + }; + static const Row k420FullExpected[] = { + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, true }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, true }, + { VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + { VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, false }, + { VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, false }, + { VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, false }, + { VK_FORMAT_R8G8B8A8_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + { VK_FORMAT_B8G8R8A8_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + { VK_FORMAT_A8B8G8R8_UNORM_PACK32, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, false }, + }; + + // P010 alone, which is not a hypothetical: it is what an NVIDIA RTX A4000 + // (0x10DE:0x24B0, driver 620.72.0) reports as the encode source for + // H.265 Main 10 at 4:2:0 / 10 bits, measured with + // vkGetPhysicalDeviceVideoFormatPropertiesKHR. Every RGB entry names P010 + // here because an RGB session takes the device's FIRST choice and that is + // the only choice. + static const VkFormat kP010Only[] = { + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + }; + static const Row kP010OnlyExpected[] = { + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, true }, + { VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, false }, + { VK_FORMAT_R8G8B8A8_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, false }, + { VK_FORMAT_B8G8R8A8_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, false }, + { VK_FORMAT_A8B8G8R8_UNORM_PACK32, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, false }, + }; + + const Scenario scenarios[] = { + // NV12 alone is what the same A4000 reports for H.264 Baseline, Main + // and High and for H.265 Main, all at 8 bits -- one format per + // profile, which is also the only figure the corpus had from any + // other device. + { "NV12 only", kNv12, 1, kNv12Expected, 5 }, + { "P010 only (A4000, H.265 Main 10)", kP010Only, 1, + kP010OnlyExpected, 5 }, + { "every 4:2:0 semi-planar depth", k420Full, 3, k420FullExpected, 9 }, + }; + + for (const Scenario& sc : scenarios) { + VkVideoEncoderInputFormatProperties out[ + VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t n = AdvertiseFromDeviceList( + sc.device, sc.deviceCount, out, + VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + // Without the filter the converted half is not advertised at all, so + // the record is the OPTIMAL prefix of it. Stated rather than skipped, + // because that build is the one this list's gate exists for. + uint32_t wanted = 0; + for (uint32_t i = 0; i < sc.expectedCount; i++) { + if (kFilterCompiledIn || sc.expected[i].optimal) { + wanted++; + } + } + Check(n == wanted, + Lbl(std::string(sc.name) + ": the advertised count is what it " + "was before the set was derived"), + "got " + U32(n) + ", want " + U32(wanted)); + uint32_t at = 0; + for (uint32_t i = 0; (i < sc.expectedCount) && (at < n); i++) { + if (!kFilterCompiledIn && !sc.expected[i].optimal) { + continue; + } + Check((out[at].format == sc.expected[i].in) && + (out[at].encodeFormat == sc.expected[i].out) && + ((out[at].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) == + sc.expected[i].optimal), + Lbl(std::string(sc.name) + " entry " + U32(at) + ": " + + AdvName(sc.expected[i].in) + " -> " + + AdvName(sc.expected[i].out)), + "got " + U32((uint32_t)out[at].format) + " -> " + + U32((uint32_t)out[at].encodeFormat) + " opt " + + U32((uint32_t)out[at].optimality)); + at++; + } + } + + // CALIBRATION: the same function on a device list that DOES carry a 4:4:4 + // encode source answers differently, so the two records above are a + // measurement and not a function that ignores its argument. + const VkFormat with444[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + }; + VkVideoEncoderInputFormatProperties wide[ + VK_ENC_MAX_ROUTABLE_INPUT_FORMATS] = {}; + const uint32_t wn = AdvertiseFromDeviceList( + with444, 2, wide, VK_ENC_MAX_ROUTABLE_INPUT_FORMATS); + Check(AdvIndexOf(wide, wn, VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM) >= + (kFilterCompiledIn ? 0 : -1) && + ((AdvIndexOf(wide, wn, VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM) >= 0) + == kFilterCompiledIn), + "3-plane 4:4:4 IS advertised once the device reports NV24, and is " + "absent from every 4:2:0 list above -- the device is what decides " + "it, not this library", + "index " + U32((uint32_t)(AdvIndexOf( + wide, wn, VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM) + 1))); +} + +void CaseYcbcrIsNeverClaimedAsRgba() +{ + g_currentCase = "no YCbCr input is mistaken for RGBA"; + // The two halves of ENCODABLE_VIA_FILTER must stay disjoint: the RGBA + // predicate drives both the plane count and whether VerifyInputs() + // re-derives input.vkFormat, so a YCbCr format leaking into it would be + // described as a single 4-byte-per-pixel plane. + const VkFormat ycbcr[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, + }; + for (VkFormat f : ycbcr) { + Check(VkEncResolveColorModel(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) != + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, + "not claimed as RGBA", "format " + U32((uint32_t)f)); + } +} + +void CaseRgbaSessionSurvivesTheSinglePlaneGate() +{ + g_currentCase = "an RGBA session is accepted by VerifyInputs"; + // An RGBA input is ONE plane of four bytes per pixel and carries no chroma + // subsampling of its own, so |chromaSubsampling| keeps the constructor + // default of 4:2:0 on this path. VerifyInputs() also refuses a single-plane + // input that is not 4:4:4, because a single-plane Y'CbCr image is packed + // and packed is only defined for 4:4:4 here. The two are compatible only + // while the RGBA arm is reached first: an RGBA image is not a packed + // Y'CbCr image and the subsampling field does not describe it. + // + // This case pins that order. With the arms transposed the code still + // compiles and links, and every RGBA input to the preprocess filter is + // refused at configuration time. + // The list carries A2B10G10R10_UNORM_PACK32 on purpose. It is an RGB + // layout the library does not route, so the guard below skips it -- and + // the guard is what this list is here to exercise: an RGB layout is not + // by itself an RGBA input, and sizing a session from one that is refused + // would describe a plane no allocation is made for. + for (VkFormat f : { VK_FORMAT_R8G8B8A8_UNORM, + VK_FORMAT_B8G8R8A8_UNORM, + VK_FORMAT_A2B10G10R10_UNORM_PACK32 }) { + if (VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_RGB) == + VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + continue; + } + EncoderInputImageParameters input; + input.width = 256; + input.height = 256; + input.numPlanes = VkEncInputFormatPlaneCount(f); + input.vkFormat = f; + input.colorSpace = VkEncColorSpace::kRGB; + + Check(input.numPlanes == 1, + "RGBA input is a single plane", "format " + U32((uint32_t)f)); + Check(input.chromaSubsampling == VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR, + "no chroma subsampling is written on the RGBA path", + "format " + U32((uint32_t)f)); + Check(input.VerifyInputs(), + "VerifyInputs accepts the RGBA session", + "format " + U32((uint32_t)f)); + Check(input.vkFormat == f, + "the caller format is carried, not re-derived", + "format " + U32((uint32_t)f)); + Check(input.planeLayouts[0].rowPitch >= (4u * input.width), + "the row pitch is four bytes per pixel", + "format " + U32((uint32_t)f)); + } + + // The control: a genuinely packed Y'CbCr single plane that is NOT 4:4:4 is + // still refused, so the case above is not passing because the gate is inert. + EncoderInputImageParameters packed422; + packed422.width = 256; + packed422.height = 256; + packed422.numPlanes = 1; + packed422.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR; + packed422.colorSpace = VkEncColorSpace::kYCbCr; + Check(!packed422.VerifyInputs(), + "packed single-plane 4:2:2 is still refused", "control"); +} + + +//============================================================================= +// 2. The binder's preprocess-conversion decision, derived from inputFormat +//============================================================================= + +// What the library wrote to the gated error stream while |fn| ran. Two +// separate gates in the binder refuse the same config, and only the wording +// says which one did -- so the wording is the contract under test. +template +std::string CaptureEncErr(Fn fn) +{ + std::ostringstream sink; + std::streambuf* saved = std::cerr.rdbuf(sink.rdbuf()); + fn(); + std::cerr.rdbuf(saved); + return sink.str(); +} + +void CaseContradictoryColorModelIsRefusedByTheBinder() +{ + g_currentCase = "a contradictory inputColorModel is refused by the binder"; + // The registration gate refuses a (format, colorModel) pair that cannot + // be reconciled. Configuring a SESSION from one is refused for the same + // reason and says the same thing, or the library answers one + // contradiction in two voices -- and the voice is what the caller acts + // on, because the FORMAT in these pairs is the half that is not wrong. + struct Row { + VkFormat format; + VkVideoEncoderColorModel declared; + const char* why; + }; + static const Row rows[] = { + { VK_FORMAT_B8G8R8A8_UNORM, VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "YCbCr declared over BGRA8, which carries no packed reading" }, + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, + "RGB declared over NV12, which is directly encodable" }, + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + (VkVideoEncoderColorModel)7, + "a colorModel value the enumeration does not define" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = row.format; + cfg.inputColorModel = row.declared; + VkEncBoundConfigProbe probe{}; + VkResult r = VK_SUCCESS; + const std::string said = CaptureEncErr([&] { + r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + }); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + (std::string("the binder refuses it: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + Check(said.find("inputColorModel") != std::string::npos, + (std::string("and names the declaration as the reason: ") + + row.why).c_str(), + "said: " + said); + // The control on the wording. The unencodable-format message lists + // B8G8R8A8_UNORM among the formats it accepts, so answering the + // first row with it names the submitted format as both refused and + // accepted, and sends the caller to change the one field that is + // correct. + Check(said.find("is not encodable") == std::string::npos, + (std::string("and not as an unencodable format: ") + + row.why).c_str(), + "said: " + said); + } +} + +void CaseAgreeingColorModelStillBinds() +{ + g_currentCase = "an unstated or agreeing inputColorModel still binds"; + // VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT is 0, so this is what every + // caller that never set the field declares, and it must configure + // exactly as it always did -- a gate that caught it would refuse the + // entire installed base. Naming the model the format already implies + // must bind too: what is refused is a CONTRADICTION, not a declaration. + struct Row { + VkFormat format; + VkVideoEncoderColorModel declared; + uint32_t wantPlanes; + uint32_t wantFilter; + bool needsFilterBuild; + const char* why; + }; + static const Row rows[] = { + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, 2, 0, false, + "NV12 declaring nothing" }, + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, 2, 0, false, + "NV12 declared YCbCr" }, + { VK_FORMAT_B8G8R8A8_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, 1, 1, true, + "BGRA8 declaring nothing" }, + { VK_FORMAT_B8G8R8A8_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, 1, 1, true, + "BGRA8 declared RGB" }, + }; + for (const Row& row : rows) { + if (row.needsFilterBuild && !kFilterCompiledIn) { + continue; + } + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = row.format; + cfg.inputColorModel = row.declared; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, + (std::string("the binder accepts it: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + Check(probe.inputNumPlanes == row.wantPlanes, + (std::string("and lays out the input it named: ") + + row.why).c_str(), + "planes " + U32(probe.inputNumPlanes) + ", want " + + U32(row.wantPlanes)); + Check(probe.preprocessComputeFilter == row.wantFilter, + (std::string("and routes it the same way: ") + row.why).c_str(), + "filter " + U32(probe.preprocessComputeFilter) + ", want " + + U32(row.wantFilter)); + } +} + +// THE TWO DERIVATIONS OF THE INPUT'S IDENTITY, AND THE ASSERTION THAT THEY +// AGREE. +// +// The binder derives (chroma subsampling, bit depth, plane count) FROM the +// caller's VkFormat. EncoderInputImageParameters::VerifyInputs() reconstructs a +// VkFormat FROM those same three. Two derivations of one quantity in opposite +// directions, and until now nothing said they had to agree -- the config that +// reached the encoder simply carried whatever the second one produced. +// +// THE REVERSE ONE CANNOT BE DELETED, which is why this is an assertion and not +// a removal. The packed-alias arm of the binder leaves input.vkFormat unwritten +// ON PURPOSE so the reconstruction supplies it: CodecGetVkFormat(4:4:4, depth, +// PACKED_1) spells AYUV at eight bits and Y410 at ten, and that is the only +// route by which either is nameable. Deleting the reverse derivation deletes +// two capabilities. +// +// WHAT THIS CASE READS. probe.inputVkFormat is the reconstruction's OUTPUT -- +// input.vkFormat as it stands after InitializeParameters -- and the assertion +// is that it is the format the caller declared. Every routable candidate is +// swept, plus the two packed aliases, which the routable list cannot carry +// because it has no colour-model column. +// +// AND WHY THE PLANE COUNT IS STILL READ BESIDE IT. inputVkFormat is one value +// standing for three, so a disagreement says THAT the round trip broke and not +// WHICH term broke it. The triple is printed with every row for that reason. +// +// H.265 IS THE CODEC FOR EVERY ROW ON PURPOSE. Range Extensions is the one +// profile in this tree whose limits admit 4:2:0, 4:2:2 and 4:4:4 at 8, 10 and +// 12 bits together, so a refusal anywhere in this sweep is about the round trip +// and not about a profile that could not carry the input. +void CaseInputFormatSurvivesTheRoundTripThroughGeometry() +{ + g_currentCase = "the input format the binder derives geometry from is the " + "format that geometry reconstructs"; + + struct Row { + VkFormat format; + VkVideoEncoderColorModel declared; + const char* what; + }; + // The packed 4:4:4 aliases are named by hand: they ride RGBA enumerants and + // are reached only through a Y'CbCr declaration, so the routable list -- + // which carries no colour model -- cannot name them. + static const Row kPacked[] = { + { VK_FORMAT_R8G8B8A8_UNORM, VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "AYUV (R8G8B8A8_UNORM declared Y'CbCr)" }, + { VK_FORMAT_A2B10G10R10_UNORM_PACK32, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "Y410 (A2B10G10R10_UNORM_PACK32 declared Y'CbCr)" }, + }; + const uint32_t kPackedCount = (uint32_t)(sizeof(kPacked) / sizeof(kPacked[0])); + + uint32_t routableCount = 0; + const VkFormat* const routable = VkEncRoutableInputFormats(routableCount); + Check(routableCount >= 10u, "the routable set is worth sweeping", + "count " + U32(routableCount)); + + uint32_t swept = 0; + for (uint32_t i = 0; i < routableCount + kPackedCount; i++) { + const bool packed = (i >= routableCount); + const VkFormat format = + packed ? kPacked[i - routableCount].format : routable[i]; + const VkVideoEncoderColorModel declared = + packed ? kPacked[i - routableCount].declared + : VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT; + const std::string what = + packed ? std::string(kPacked[i - routableCount].what) + : ("routable enumerant " + U32((uint32_t)format)); + + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + cfg.inputFormat = format; + cfg.inputColorModel = declared; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probe); + Check(r == VK_SUCCESS, + ("the binder accepts it: " + what).c_str(), + "VkResult " + U32((uint32_t)r)); + if (r != VK_SUCCESS) { + continue; + } + swept++; + Check(probe.inputVkFormat == (uint32_t)format, + ("and the geometry it derived reconstructs the same format: " + + what).c_str(), + "declared " + U32((uint32_t)format) + ", reconstructed " + + U32(probe.inputVkFormat) + " from (subsampling " + + U32(probe.inputChromaSubsampling) + ", " + + U32(probe.inputBpp) + "-bit, " + U32(probe.inputNumPlanes) + + " planes)"); + } + Check(swept == routableCount + kPackedCount, + "every candidate bound, so no row was skipped into agreement", + "bound " + U32(swept) + " of " + + U32(routableCount + kPackedCount)); +} + +void CaseSemiPlanarBindsTwoPlanes() +{ + g_currentCase = "an NV12 session describes its input as 2-plane"; + // EncoderConfig does not store the input format; it reconstructs + // input.vkFormat from subsampling, bit depth and numPlanes, so numPlanes + // has to carry the caller's actual count. A default of 3 makes every ext + // session describe its own input as 3-plane I420 whatever the caller + // passed -- invisible until the compute filter is built from that + // description. + VkVideoEncoderConfig cfg = BaseConfig(); + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted an NV12 config", + "VkResult " + U32((uint32_t)r)); + Check(probe.inputNumPlanes == 2, + "input.numPlanes follows inputFormat", + "got " + U32(probe.inputNumPlanes) + ", want 2"); + Check(probe.inputBpp == 8, "8-bit", "got " + U32(probe.inputBpp)); +} + +void CaseTenBitBindsBitDepthAndPlanes() +{ + g_currentCase = "a P010 session binds both bit depth and plane count"; + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted a P010 config", + "VkResult " + U32((uint32_t)r)); + Check(probe.inputBpp == 10, "10-bit", "got " + U32(probe.inputBpp)); + Check(probe.inputNumPlanes == 2, "2 planes", + "got " + U32(probe.inputNumPlanes)); +} + +void CaseFourFourFourBindsItsOwnSubsampling() +{ + g_currentCase = "a 4:4:4 session binds 4:4:4, 2 planes and no filter"; + // input.chromaSubsampling is what the codec arm derives the encode + // profile from, and it is what an encoder writes into the bitstream's + // chroma_format_idc. Left at the 4:2:0 default, a 4:4:4 request encodes + // as 4:2:0 and reports success -- a wrong bitstream rather than a + // refusal, which is why this is asserted rather than assumed from the + // plane count. The layout is unchanged from 4:2:0 semi-planar, so the + // plane count alone cannot tell the two apart. + const struct { VkFormat format; uint32_t bpp; const char* name; } rows[] = { + { VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, 8, "NV24" }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, 10, "S410" }, + }; + for (const auto& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = row.format; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted the config", + std::string(row.name) + " VkResult " + U32((uint32_t)r)); + Check(probe.inputChromaSubsampling == + (uint32_t)VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR, + "input.chromaSubsampling follows inputFormat", + std::string(row.name) + " got " + + U32(probe.inputChromaSubsampling) + ", want " + + U32((uint32_t)VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR)); + Check(probe.inputNumPlanes == 2, "2 planes", + std::string(row.name) + " got " + U32(probe.inputNumPlanes)); + Check(probe.inputBpp == row.bpp, "bit depth follows inputFormat", + std::string(row.name) + " got " + U32(probe.inputBpp)); + Check(probe.preprocessComputeFilter == 0, + "no preprocess filter is built for a directly encodable input", + std::string(row.name) + " got " + + U32(probe.preprocessComputeFilter)); + } +} + +// The control for the case above: the 4:2:0 semi-planar set must still bind +// 4:2:0, so a pass there is not a projection that answers 4:4:4 for +// everything. +void CaseFourTwoZeroStillBindsFourTwoZero() +{ + g_currentCase = "a 4:2:0 session still binds 4:2:0"; + const VkFormat rows[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + }; + for (VkFormat f : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = f; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + // I420 is the one row here that needs converting, so a build without + // the filter refuses it -- and refusing is correct, since the staging + // copy cannot change plane count. The other two are read directly and + // are accepted in either build. This row asserted acceptance + // unconditionally and so could only ever have been run with the + // filter present. + const bool needsFilter = + (VkEncClassifyInput(f, VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER); + if (needsFilter && !kFilterCompiledIn) { + Check(r == VK_ERROR_INITIALIZATION_FAILED, + "a converted 4:2:0 input is refused, not bound, when the " + "build has no filter to convert it", + "format " + U32((uint32_t)f) + " VkResult " + + U32((uint32_t)r)); + continue; + } + Check(r == VK_SUCCESS, "binder accepted the config", + "format " + U32((uint32_t)f) + " VkResult " + U32((uint32_t)r)); + Check(probe.inputChromaSubsampling == + (uint32_t)VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR, + "input.chromaSubsampling is 4:2:0", + "format " + U32((uint32_t)f) + " got " + + U32(probe.inputChromaSubsampling)); + } +} + +void CaseDirectFormatGetsNoFilter() +{ + g_currentCase = "a directly encodable input gets NO filter"; + // EncoderConfig's own default for enablePreprocessComputeFilter is TRUE, + // so this is a positive write-down and not an inherited value. Without + // it, InitEncoder would swap the input command-buffer pool for the + // filter's compute-family pool on a session with nothing to convert. + VkVideoEncoderConfig cfg = BaseConfig(); // NV12 + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted the config", + "VkResult " + U32((uint32_t)r)); + Check(probe.preprocessComputeFilter == 0, + "enablePreprocessComputeFilter written DOWN, not inherited", + "got " + U32(probe.preprocessComputeFilter)); +} + +void CaseThreePlaneGetsAFilterWithoutAsking() +{ + g_currentCase = "a 3-plane input gets a filter without asking for one"; + // The case the derivation exists for: an I420 dma-buf producer the + // library adapts, rather than an embedder that converts before it submits. + // Nothing on the config requests the conversion -- the format is the whole + // input to the decision. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + if (kFilterCompiledIn) { + Check(r == VK_SUCCESS, "accepted", "VkResult " + U32((uint32_t)r)); + Check(probe.preprocessComputeFilter == 1, "the filter is enabled", + "got " + U32(probe.preprocessComputeFilter)); + Check(probe.inputNumPlanes == 3, + "the session describes its input as 3-plane, which is what " + "input.vkFormat -- and therefore the filter's input format -- " + "is derived from", + "got " + U32(probe.inputNumPlanes)); + } else { + // A build with no filter must refuse rather than encode the frame + // unconverted: with no conversion the frame falls to the staging + // copy, whose two-region copy from a three-plane source is a GPU + // hang and not a slower path. + Check(r != VK_SUCCESS, + "a build without the filter refuses a format only it could take", + "VkResult " + U32((uint32_t)r)); + } +} + +void CaseRgbaGetsAFilterWithoutAsking() +{ + g_currentCase = "a BGRA8 input gets a filter without asking, as 1 plane"; + // A BGRA8 compositor buffer the library adapts itself. The stronger of the + // two conversion cases: a transfer copy cannot perform a colour-model + // conversion at all, so encoding this without a filter would encode raw + // RGB bytes as if they were luma and chroma. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_B8G8R8A8_UNORM; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + if (kFilterCompiledIn) { + Check(r == VK_SUCCESS, "accepted", "VkResult " + U32((uint32_t)r)); + Check(probe.preprocessComputeFilter == 1, "the filter is enabled", + "got " + U32(probe.preprocessComputeFilter)); + // The session must describe its input as ONE plane. input.vkFormat -- + // which the compute filter's input format is taken from -- is derived + // alongside this, and a 3 here would have built the filter to read + // planes a BGRA import does not have. + Check(probe.inputNumPlanes == 1, + "the session describes its input as single-plane", + "got " + U32(probe.inputNumPlanes)); + Check(probe.inputBpp == 8, "8-bit", "got " + U32(probe.inputBpp)); + } else { + Check(r != VK_SUCCESS, + "a build without the filter refuses a format only it could take", + "VkResult " + U32((uint32_t)r)); + } +} + +void CasePackedYcbcrIsRoutedWhereItIsDeclared() +{ + g_currentCase = "the packed 4:4:4 aliases are routed where they are declared"; + // WHAT THE COLOUR-MODEL DECLARATION BUYS. AYUV and Y410 are Y'CbCr 4:4:4 + // carried one interleaved texel per pixel. They have no Vulkan enumerant + // of their own and ride R8G8B8A8_UNORM and A2B10G10R10_UNORM_PACK32, + // which are also how an ordinary R'G'B' frame is spelled, so the + // declaration is the only thing that can say which of the two a surface + // is -- and the taxonomy has to ROUTE what the declaration names, or + // stating the truth about the pixels is what refuses them. + struct Row { + const char* name; + VkFormat format; + VkVideoEncoderColorModel declared; + VkEncInputFormatClass wantClass; + uint32_t wantPlanes; + uint32_t wantBpp; // 0 when the class is UNSUPPORTED + const char* why; + }; + static const Row rows[] = { + { "AYUV", VK_FORMAT_R8G8B8A8_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, 1, 8, + "AYUV declared Y'CbCr is routed through the filter" }, + { "Y410", VK_FORMAT_A2B10G10R10_UNORM_PACK32, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, 1, 10, + "Y410 declared Y'CbCr is routed through the filter" }, + // Y416 is the one the packed table names and the taxonomy does not + // take. 16 bits per component is not a + // VkVideoComponentBitDepthFlagBitsKHR, so no input geometry can carry + // it; the refusal is early and clear rather than an opaque one raised + // after the caller has built a frame pool. + { "Y416", VK_FORMAT_R16G16B16A16_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + VK_ENC_INPUT_FORMAT_UNSUPPORTED, 0, 0, + "Y416 declared Y'CbCr is refused for its bit depth" }, + // THE CONTROL, and it is the whole point of the case: the SAME + // enumerant as the first row, declaring nothing. It must still be an + // ordinary 8-bit R'G'B' image bound at 4:2:0, so a change that made + // the packed rows pass by widening the R'G'B' arm would fail here. + { "RGBA8", VK_FORMAT_R8G8B8A8_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, + VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, 1, 8, + "the same enumerant undeclared is still R'G'B'" }, + }; + for (const Row& row : rows) { + Check(VkEncClassifyInput(row.format, row.declared) == row.wantClass, + row.why, + std::string(row.name) + " classified " + + U32((uint32_t)VkEncClassifyInput(row.format, row.declared))); + Check(VkEncInputFormatPlaneCount(row.format) == row.wantPlanes, + "and its layout is one plane, or none if it is not routed", + std::string(row.name) + " planes " + + U32(VkEncInputFormatPlaneCount(row.format))); + } + + // WHAT THE BINDER THEN WRITES. A class answer that no session geometry + // follows would be an acceptance in name only: EncoderConfig does not + // store the input format, it RECONSTRUCTS it from subsampling, bit depth + // and plane count, so a packed input left at the 3-plane 4:2:0 default + // would configure the session as I420 while the caller declared AYUV. + // The probe reads exactly those three values back. + if (!kFilterCompiledIn) { + Check(true, "no preprocess filter is compiled into this build", ""); + return; + } + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = row.format; + cfg.inputColorModel = row.declared; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + if (row.wantClass == VK_ENC_INPUT_FORMAT_UNSUPPORTED) { + Check(r != VK_SUCCESS, "the binder refuses it too", + std::string(row.name) + " VkResult " + U32((uint32_t)r)); + continue; + } + Check(r == VK_SUCCESS, "the binder accepts it", + std::string(row.name) + " VkResult " + U32((uint32_t)r)); + Check(probe.preprocessComputeFilter == 1, + "and builds the filter without being asked", + std::string(row.name) + " got " + + U32(probe.preprocessComputeFilter)); + Check(probe.inputNumPlanes == row.wantPlanes, + "and describes the input as single-plane", + std::string(row.name) + " got " + U32(probe.inputNumPlanes)); + Check(probe.inputBpp == row.wantBpp, + "and at the component depth the layout carries", + std::string(row.name) + " got " + U32(probe.inputBpp)); + // The subsampling is where the control separates from the packed + // rows: a packed 4:4:4 input must move it off the 4:2:0 default, + // and the same enumerant read as R'G'B' must not -- an R'G'B' + // session's subsampling is the encode profile's, not the input's. + const uint32_t wantSubsampling = + (row.declared == VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) + ? (uint32_t)VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR + : (uint32_t)VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + Check(probe.inputChromaSubsampling == wantSubsampling, + "and at the subsampling the declaration implies", + std::string(row.name) + " got " + + U32(probe.inputChromaSubsampling) + ", want " + + U32(wantSubsampling)); + } +} + +void CaseUnsupportedFormatStillRefused() +{ + g_currentCase = "an sRGB RGBA input is refused"; + // The 8-bit UNORM RGBA family is convertible; the _SRGB spellings of the + // same formats are not. The filter reads its RGBA source as a storage + // image, and no *_SRGB format carries the storage-image format feature, + // so an sRGB view can never be the descriptor the filter binds. The + // refusal lands at init, where the caller can still allocate a _UNORM + // view instead. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_B8G8R8A8_SRGB; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r != VK_SUCCESS, "refused", "VkResult " + U32((uint32_t)r)); +} + +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED +const char* FilterArmName(VulkanFilterYuvCompute::FilterType arm) +{ + switch (arm) { + case VulkanFilterYuvCompute::RGBA2YCBCR: return "RGBA2YCBCR"; + case VulkanFilterYuvCompute::YCBCR2RGBA: return "YCBCR2RGBA"; + case VulkanFilterYuvCompute::YCBCRCOPY: return "YCBCRCOPY"; + case VulkanFilterYuvCompute::YCBCRCLEAR: return "YCBCRCLEAR"; + default: break; + } + return "unknown"; +} +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + +void CaseFilterArmReadsTheDeclaredColourModel() +{ + g_currentCase = "the filter arm reads the declared colour model"; +#ifdef VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED + // THE CONTRACT. VkEncDeriveFilterType decides which conversion the + // preprocess filter performs from the colour model the input is DECLARED + // to carry and the format the device accepts as an encode source, and it + // reads each side by what the surface MEANS rather than by which format + // table places its enumerant. + // + // Both sides matter, and the packed 4:4:4 layouts are why. AYUV, Y410 and + // Y416 have no Vulkan format of their own and ride R8G8B8A8_UNORM, + // A2B10G10R10_UNORM_PACK32 and R16G16B16A16_UNORM, which are also the + // formats an ordinary R'G'B' frame is spelled with. A derivation that + // read the enumerant alone would call a packed encode source R'G'B' and + // convert a Y'CbCr input through the inverse matrix, and would call a + // packed INPUT R'G'B' and convert it through the forward one -- writing + // R, G and B into the channels the encoder reads as Cr, Cb and Y. Neither + // raises an error, because the inverse conversion emits the very + // enumerant the packed encode source is spelled with. + // + // WHAT THIS PINS AND WHAT IT DOES NOT. This is the decision, not the + // pixels: the arm this asserts is the arm the encoder builds, but whether + // the built shader then writes the right samples is a hardware question, + // and no device that advertises a packed 4:4:4 encode source is required + // to exist for this case to run. It reads only its arguments. + struct Row { + VkEncColorSpace declared; + VkFormat encodeSource; + VulkanFilterYuvCompute::FilterType expected; + const char* why; + }; + static const Row rows[] = { + { VkEncColorSpace::kYCbCr, VK_FORMAT_R8G8B8A8_UNORM, + VulkanFilterYuvCompute::YCBCRCOPY, + "Y'CbCr in, AYUV encode source: both sides are Y'CbCr, so it is a " + "copy and no matrix is applied" }, + { VkEncColorSpace::kYCbCr, VK_FORMAT_A2B10G10R10_UNORM_PACK32, + VulkanFilterYuvCompute::YCBCRCOPY, + "Y'CbCr in, Y410 encode source" }, + { VkEncColorSpace::kYCbCr, VK_FORMAT_R16G16B16A16_UNORM, + VulkanFilterYuvCompute::YCBCRCOPY, + "Y'CbCr in, Y416 encode source" }, + // The planar controls. They differ from the three rows above in the + // encode-source format ONLY, so a packed row that passed because the + // whole derivation had collapsed onto one answer would be caught by + // the R'G'B' rows below rather than hidden by these. + { VkEncColorSpace::kYCbCr, VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VulkanFilterYuvCompute::YCBCRCOPY, + "Y'CbCr in, semi-planar 4:2:0 encode source" }, + { VkEncColorSpace::kYCbCr, VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + VulkanFilterYuvCompute::YCBCRCOPY, + "Y'CbCr in, semi-planar 4:4:4 encode source" }, + // An R'G'B' input takes the matrix, and takes it whether the encode + // source is planar or packed. + { VkEncColorSpace::kRGB, VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VulkanFilterYuvCompute::RGBA2YCBCR, + "R'G'B' in, semi-planar 4:2:0 encode source: the matrix conversion" }, + { VkEncColorSpace::kRGB, VK_FORMAT_R8G8B8A8_UNORM, + VulkanFilterYuvCompute::RGBA2YCBCR, + "R'G'B' in, AYUV encode source: still the matrix conversion" }, + }; + for (const Row& row : rows) { + const VulkanFilterYuvCompute::FilterType arm = + VkEncDeriveFilterType(row.declared, row.encodeSource); + Check(arm == row.expected, row.why, + std::string("derived ") + FilterArmName(arm) + ", expected " + + FilterArmName(row.expected)); + } + + // THE CALIBRATION. Every row above is a positive expectation, and a + // derivation that had collapsed onto a single answer would satisfy half + // of them. This pair does not: it holds the encode-source format fixed + // and changes ONLY the declared colour model, so it can be satisfied only + // by a derivation that actually reads the declaration. + Check(VkEncDeriveFilterType(VkEncColorSpace::kRGB, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM) != + VkEncDeriveFilterType(VkEncColorSpace::kYCbCr, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM), + "the declared colour model, and nothing else, decides the arm", + "both declarations derived the same arm"); + + // No encode source the library can select is R'G'B', so the inverse arm + // is not an answer the encoder ever wants: it forces the filter output to + // an RGBA format, which for AYUV is the encode source's own enumerant. + for (const Row& row : rows) { + Check(VkEncDeriveFilterType(row.declared, row.encodeSource) != + VulkanFilterYuvCompute::YCBCR2RGBA, + "no encode source derives the inverse arm", row.why); + } +#else + // With no filter compiled in there is no arm to derive; the binder + // refuses every format that would need one, which the cases above assert. + Check(true, "no preprocess filter is compiled into this build", ""); +#endif // VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED +} + +//============================================================================= +// 2b. The transfer-function declaration +//============================================================================= + +void CaseUndeclaredInputOtfAssertsNothing() +{ + g_currentCase = "inputTransferCharacteristics 0 declares nothing"; + // 0 is "not declared". It must not be read as a code point and compared, + // or every caller that never heard of the field would be refused the + // moment it declared a bitstream transfer function. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.colourPrimaries = 1; + cfg.transferCharacteristics = 16; // PQ on the bitstream + cfg.matrixCoefficients = 1; + cfg.inputTransferCharacteristics = 0; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "accepted", "VkResult " + U32((uint32_t)r)); + Check(probe.transferCharacteristics == 16, + "and the bitstream still declares what it was given", + "got " + U32(probe.transferCharacteristics)); +} + +void CaseAgreeingOtfDeclarationsAreAccepted() +{ + g_currentCase = "input and bitstream declaring the same OTF is accepted"; + // The shape a caller uses to state the requirement explicitly, which is + // the whole point of the field: both ends named, and they agree. + for (uint8_t tc : { (uint8_t)1, (uint8_t)16, (uint8_t)18 }) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.colourPrimaries = 1; + cfg.transferCharacteristics = tc; + cfg.matrixCoefficients = 1; + cfg.inputTransferCharacteristics = tc; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, + ("transfer function " + U32(tc) + " on both ends is accepted") + .c_str(), + "VkResult " + U32((uint32_t)r)); + } +} + +void CaseMismatchedOtfDeclarationIsRefused() +{ + g_currentCase = "a declared OTF mismatch is refused, not converted"; + // THE CONTRACT THIS PINS. The library applies a colour matrix and no + // transfer function, so it cannot serve an input in one transfer function + // and a bitstream in another. Converting anyway would emit pixels that are + // close enough to look plausible and wrong everywhere; the refusal is the + // only honest answer. + struct Row { uint8_t in; uint8_t out; const char* why; }; + static const Row rows[] = { + { 16, 1, "PQ input, BT.709 bitstream" }, + { 1, 16, "BT.709 input, PQ bitstream" }, + { 18, 16, "HLG input, PQ bitstream" }, + { 1, 0, "declared input OTF, undeclared bitstream OTF" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.colourPrimaries = 1; + cfg.transferCharacteristics = row.out; + cfg.matrixCoefficients = 1; + cfg.inputTransferCharacteristics = row.in; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + (std::string("refused: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + } +} + +//============================================================================= +// 2c. The profile number +//============================================================================= + +struct ProfileRow { + VkVideoCodecOperationFlagBitsKHR codec; + const char* codecName; + uint32_t profile; + const char* name; +}; + +void CaseProfileNumbersReachTheCodecConfigUnchanged() +{ + g_currentCase = "a profile number reaches the codec config unchanged"; + // THE CONTRACT. VkVideoEncoderConfig::profile carries the codec + // standard's own number -- H.264 profile_idc, H.265 general_profile_idc, + // AV1 seq_profile -- and the codec-typed config is what session creation + // reads. Asserting equality end to end is what makes "the values come from + // the standard" a property of the code rather than of the comment: a + // translation table reintroduced between the two would show up here as a + // number that changed on the way through. + static const ProfileRow rows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "H.264", + VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, "Baseline (66)" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "H.264", + VK_VIDEO_ENCODER_PROFILE_H264_MAIN, "Main (77)" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "H.264", + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, "High (100)" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, "H.265", + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, "Main (1)" }, + }; + for (const ProfileRow& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = row.profile; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + const std::string what = + std::string(row.codecName) + " " + row.name; + Check(r == VK_SUCCESS, (what + " is accepted").c_str(), + "VkResult " + U32((uint32_t)r)); + Check(probe.codecProfile == row.profile, + (what + " reaches the codec config as its own number").c_str(), + "got " + U32(probe.codecProfile) + ", asked for " + + U32(row.profile)); + } +} + +void CaseProfileDefaultIsDerivedPerCodec() +{ + g_currentCase = "DEFAULT derives a profile rather than binding one"; + // DEFAULT binds nothing; the codec config derives from the input depth. + // On AV1 the number is also seq_profile 0, which is Main -- the overlap + // the public header states -- so the derivation and the named profile + // land on the same value for 8-bit input. + static const ProfileRow rows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "H.264", + VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, "H.265", + VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, "AV1", + VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + }; + for (const ProfileRow& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = row.profile; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + Check(r == VK_SUCCESS, + (std::string(row.codecName) + " accepts DEFAULT").c_str(), + "VkResult " + U32((uint32_t)r)); + } + // And AV1's overlap, asserted rather than described: 0 selects seq_profile + // 0 on 8-bit input, which is Main. + VkVideoEncoderConfig av1 = BaseConfig(); + av1.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + av1.profile = VK_VIDEO_ENCODER_PROFILE_DEFAULT; + VkEncBoundConfigProbe aprobe{}; + Check(VkEncBuildAndProbeConfig( + av1, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, &aprobe) == + VK_SUCCESS, + "AV1 DEFAULT session builds", "init failed"); + Check(aprobe.codecProfile == VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, + "AV1 0 selects seq_profile 0 (Main) on 8-bit input", + "got " + U32(aprobe.codecProfile)); +} + +void CaseProfileNumbersAreReadAgainstTheCodec() +{ + g_currentCase = "a profile number is read against the codec, not alone"; + // Standard profile numbers repeat across codecs: 1 is H.265 Main and is + // not an H.264 profile_idc; 100 is H.264 High and is not an assigned H.265 + // general_profile_idc. A number that belongs to another codec must be + // REFUSED, not bound and not ignored -- binding it would emit a bitstream + // declaring a profile the caller never asked for, and ignoring it would + // emit the library's default under the caller's label. + // + // THE UNBINDABLE ROWS NAME NUMBERS THE STANDARD'S LIMITS TABLE DOES NOT + // STATE, so they must not name 122, H.265 4 or AV1 1 -- the library binds + // those. Worse, AV1 High over this case's 4:2:0 input is refused by the + // SUBSAMPLING guard rather than by unbindability, so such a row keeps + // passing while asserting something that has stopped + // being true. A refusal row is only evidence when nothing else produces the + // same code. + struct Row { VkVideoCodecOperationFlagBitsKHR codec; uint32_t profile; + const char* why; }; + static const Row rows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, + "H.265 Main (1) on an H.264 session" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, + "H.264 High (100) on an H.265 session" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, 44, + "H.264 CAVLC 4:4:4 Intra (44), an assigned profile_idc the limits " + "table does not state and this library does not bind" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, 5, + "H.265 High Throughput (5), likewise unstated and unbound" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, 3, + "AV1 seq_profile 3, which the format does not define at all" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = row.profile; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + (std::string("refused: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + } +} + +void CaseProfileMustAdmitTheInputDepth() +{ + g_currentCase = "a profile the input depth does not admit is refused"; + // The standard's rule, enforced instead of worked around. Honouring an + // 8-bit-only profile over 10-bit input would emit an out-of-spec + // bitstream; overriding the request with a deeper profile would be the + // same ignored request in the other direction. + struct Row { VkVideoCodecOperationFlagBitsKHR codec; uint32_t profile; + VkFormat fmt; const char* why; }; + static const Row rows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + "H.264 High (100) over 10-bit input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, + "H.265 Main (1) over 10-bit input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, + "H.265 Main 10 (2) over 12-bit input" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = row.profile; + cfg.inputFormat = row.fmt; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + (std::string("refused: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + } + // The control: the same depth WITH a profile that admits it is accepted, + // so the rows above measure the profile rule and not the bit depth. + VkVideoEncoderConfig ok = BaseConfig(); + ok.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + ok.profile = VK_VIDEO_ENCODER_PROFILE_H265_MAIN10; + ok.inputFormat = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16; + VkEncBoundConfigProbe okProbe{}; + const VkResult okR = VkEncBuildAndProbeConfig( + ok, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &okProbe); + Check(okR == VK_SUCCESS, + "H.265 Main 10 (2) over 10-bit input is accepted", + "VkResult " + U32((uint32_t)okR)); + Check(okProbe.codecProfile == VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, + "and binds general_profile_idc 2", + "got " + U32(okProbe.codecProfile)); +} + + +void CaseProfileMustAdmitTheInputSubsampling() +{ + g_currentCase = "a profile the input subsampling does not admit is refused"; + // THE OTHER HALF OF THE STANDARD'S RULE. A guard reading the input bit depth + // and nothing else lets an + // explicitly named H.264 High (100) over 4:4:4 input passed the 8-bit + // check, bound profile_idc 100, and OVERRODE the derivation that reads + // input.chromaSubsampling and would have chosen High 4:4:4 Predictive + // (244). The result declared 4:2:0 in the SPS while the session carried + // 4:4:4 content -- a bitstream describing something the caller never + // asked for, which is the exact failure the profile guard exists to + // prevent on the depth axis. + // + // These rows are DEVICE-FREE: VkEncBuildAndProbeConfig binds a config and + // reads it back without an encode-capable device, so what they measure is + // the library's rule and not a driver's answer. The device half of the + // same fact is measured separately, where a device exists. + struct Row { VkVideoCodecOperationFlagBitsKHR codec; uint32_t profile; + VkFormat fmt; const char* why; }; + static const Row rows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + "H.264 High (100) over 4:4:4 (NV24) input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_MAIN, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + "H.264 Main (77) over 4:4:4 (NV24) input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, + VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, + "H.264 Baseline (66) over 4:2:2 input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + "H.265 Main (1) over 4:4:4 (NV24) input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, + "H.265 Main 10 (2) over 10-bit 4:4:4 (S410) input" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = row.profile; + cfg.inputFormat = row.fmt; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + (std::string("refused: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + } + + // CONTROL 1 -- THE SAME PROFILE AT ITS OWN SUBSAMPLING IS STILL ACCEPTED. + // A guard that refused everything would pass every row above. These say + // the rows measure the profile/subsampling PAIR and not the presence of an + // explicit profile, and not the format. + struct OkRow { VkVideoCodecOperationFlagBitsKHR codec; uint32_t profile; + VkFormat fmt; uint32_t expectBound; const char* what; }; + static const OkRow okRows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, + "H.264 High (100) over 4:2:0 (NV12) input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, + "H.264 Baseline (66) over 4:2:0 (NV12) input" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, + "H.265 Main (1) over 4:2:0 (NV12) input" }, + }; + for (const OkRow& row : okRows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = row.profile; + cfg.inputFormat = row.fmt; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + Check(r == VK_SUCCESS, + (std::string("accepted: ") + row.what).c_str(), + "VkResult " + U32((uint32_t)r)); + Check(probe.codecProfile == row.expectBound, + (std::string("and binds it unchanged: ") + row.what).c_str(), + "got " + U32(probe.codecProfile)); + } + + // CONTROL 2 -- AND THE DERIVATION THE REFUSAL POINTS AT ACTUALLY EXISTS. + // Every refusal above tells the caller to use DEFAULT instead. That advice + // is only worth giving if DEFAULT reaches a profile that CAN carry the + // input, so the same 4:4:4 formats are put through DEFAULT here and the + // derived number is read back. Without this the refusals would be a dead + // end dressed as a remedy. + struct DerRow { VkVideoCodecOperationFlagBitsKHR codec; VkFormat fmt; + uint32_t expect; const char* what; }; + static const DerRow derRows[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE, + "H.264 DEFAULT over 4:4:4 derives High 4:4:4 Predictive (244)" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, + STD_VIDEO_H264_PROFILE_IDC_HIGH_422, + "H.264 DEFAULT over 4:2:2 derives High 4:2:2 (122)" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS, + "H.265 DEFAULT over 4:4:4 derives Range Extensions (4)" }, + }; + for (const DerRow& row : derRows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = row.codec; + cfg.profile = VK_VIDEO_ENCODER_PROFILE_DEFAULT; + cfg.inputFormat = row.fmt; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, row.codec, &probe); + Check(r == VK_SUCCESS, + (std::string("accepted: ") + row.what).c_str(), + "VkResult " + U32((uint32_t)r)); + Check(probe.codecProfile == row.expect, + row.what, "got " + U32(probe.codecProfile)); + } +} + +//============================================================================= +// VkVideoEncoderInputColourInfo -- what the caller's OWN samples carry. +// +// The config's colour fields say what the BITSTREAM should advertise. This +// chain says what the INPUT is. Before it, the RGBA->Y'CbCr filter derived its +// matrix from colourPrimaries -- an output field -- which is sound only because +// this library performs no primaries conversion, and nothing said so. +// +// Every case here is device-free: VkEncBuildAndProbeConfig walks pNext and the +// derivation runs inside the binder. +//============================================================================= + +VkVideoEncoderInputColourInfo InputColour(uint8_t primaries, uint8_t transfer, + uint8_t matrix, + VkVideoEncoderRangeDeclaration range) +{ + VkVideoEncoderInputColourInfo ic{}; + ic.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO; + ic.pNext = nullptr; + ic.inputColourPrimaries = primaries; + ic.inputTransferCharacteristics = transfer; + ic.inputMatrixCoefficients = matrix; + ic.reserved = 0; + ic.inputRange = range; + return ic; +} + +void CaseInputColourChainBindsEachAxis() +{ + g_currentCase = "the chained input-colour struct reaches the encoder " + "config on every axis"; + // An RGBA session, because that is the lane the declaration steers, and + // BT.2020 on both sides so the axes AGREE and the refusal below is not + // what is being measured here. + VkVideoEncoderInputColourInfo ic = + InputColour(9, 14, 9, VK_VIDEO_ENCODER_RANGE_FULL); + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + cfg.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + cfg.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + cfg.colourPrimaries = 9; + cfg.transferCharacteristics = 14; + cfg.matrixCoefficients = 9; + cfg.pNext = ⁣ + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "the binder accepted the chained input-colour " + "struct", + "VkResult " + U32((uint32_t)r)); + Check(probe.inputColourChainPresent == 1u, + "the chain is recorded as PRESENT, which no value field could say", + "got " + U32(probe.inputColourChainPresent)); + Check(probe.inputColourPrimaries == 9u, + "inputColourPrimaries bound", "got " + U32(probe.inputColourPrimaries)); + Check(probe.inputTransferCharacteristics == 14u, + "inputTransferCharacteristics bound", + "got " + U32(probe.inputTransferCharacteristics)); + Check(probe.inputMatrixCoefficients == 9u, + "inputMatrixCoefficients bound", + "got " + U32(probe.inputMatrixCoefficients)); + Check(probe.inputRange == (uint32_t)VK_VIDEO_ENCODER_RANGE_FULL, + "inputRange bound", "got " + U32(probe.inputRange)); +} + +void CaseAbsentInputColourChainChangesNothing() +{ + g_currentCase = "an absent input-colour chain leaves every projected " + "field where it was"; + // THE DEVICE-FREE HALF OF THE REGRESSION CONTROL, and the one assertion + // that protects every existing caller: a config with no chain must build + // exactly what it built before. An ALL-ZERO chain is compared alongside + // it, because "absent" and "present and all zero" must differ in exactly + // one projected field -- the presence flag -- and in nothing else. + VkVideoEncoderConfig plain = BaseConfig(); + plain.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + plain.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + plain.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + VkEncBoundConfigProbe noChain{}; + Check(VkEncBuildAndProbeConfig( + plain, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &noChain) == + VK_SUCCESS, + "an unchained RGBA config builds", "init failed"); + Check((noChain.inputColourChainPresent == 0u) && + (noChain.inputColourPrimaries == 0u) && + (noChain.inputTransferCharacteristics == 0u) && + (noChain.inputMatrixCoefficients == 0u) && + (noChain.inputRange == 0u), + "an unchained config declares nothing about its input's colour", + "present " + U32(noChain.inputColourChainPresent)); + // THE BT.709 FALLBACK SURVIVES for a genuinely undeclared input. This is + // the regression the derivation change could have caused and did not. + Check(noChain.matrixCoefficients == 0u, + "and the binder writes no matrix for it, exactly as before", + "got " + U32(noChain.matrixCoefficients)); + + VkVideoEncoderInputColourInfo zero = + InputColour(0, 0, 0, VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig chained = plain; + chained.pNext = &zero; + VkEncBoundConfigProbe zeroChain{}; + Check(VkEncBuildAndProbeConfig( + chained, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + &zeroChain) == VK_SUCCESS, + "an all-zero chain is accepted", "init failed"); + Check(zeroChain.inputColourChainPresent == 1u, + "an all-zero chain is PRESENT, which is the distinction no value " + "field carries", + "got " + U32(zeroChain.inputColourChainPresent)); + Check((zeroChain.inputColourPrimaries == 0u) && + (zeroChain.matrixCoefficients == noChain.matrixCoefficients) && + (zeroChain.colourPrimaries == noChain.colourPrimaries) && + (zeroChain.transferCharacteristics == + noChain.transferCharacteristics) && + (zeroChain.videoFullRangeFlag == noChain.videoFullRangeFlag) && + (zeroChain.colorDescriptionPresent == + noChain.colorDescriptionPresent) && + (zeroChain.videoSignalTypePresent == + noChain.videoSignalTypePresent), + "and it changes nothing else -- undeclared is undeclared however it " + "is spelled", "a projected colour field moved"); +} + +void CaseInputColourDisagreementIsRefused() +{ + g_currentCase = "an input colour that contradicts the bitstream's is " + "refused"; + // A MATCHED PAIR, AND THE CONTROL RUNS FIRST. There is no distinct error + // code available -- every binder refusal is + // VK_ERROR_INITIALIZATION_FAILED, and the sibling transfer-axis refusal + // returns exactly that -- so an expectation on the code alone would also + // match a refusal from somewhere else entirely and would assert nothing. + // Two configs differing on ONE FIELD, one of which must succeed, is what + // attributes the refusal to the axis under test. + struct Pair { + uint8_t outputPrimaries; + uint8_t inputPrimaries; + const char* what; + }; + VkVideoEncoderConfig base = BaseConfig(); + base.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + base.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + base.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + base.colourPrimaries = 9; + + VkVideoEncoderInputColourInfo agree = + InputColour(9, 0, 0, VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig cfgA = base; + cfgA.pNext = &agree; + VkEncBoundConfigProbe probeA{}; + const VkResult rA = VkEncBuildAndProbeConfig( + cfgA, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probeA); + Check(rA == VK_SUCCESS, + "CONTROL: input primaries 9 with bitstream primaries 9 is accepted", + "VkResult " + U32((uint32_t)rA)); + if (rA != VK_SUCCESS) { + // The pair is measuring a broken fixture; B's refusal would mean + // nothing, so do not read it. + return; + } + + VkVideoEncoderInputColourInfo disagree = + InputColour(1, 0, 0, VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig cfgB = base; + cfgB.pNext = &disagree; + VkEncBoundConfigProbe probeB{}; + const VkResult rB = VkEncBuildAndProbeConfig( + cfgB, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probeB); + Check(rB == VK_ERROR_INITIALIZATION_FAILED, + "and input primaries 1 with the SAME bitstream primaries 9 is " + "refused -- one field apart, so the refusal is that field's", + "VkResult " + U32((uint32_t)rB)); + + // THE SAME SHAPE ON THE MATRIX AXIS, so "refused" is a property of the + // rule and not of the primaries field alone. + VkVideoEncoderConfig mBase = base; + mBase.matrixCoefficients = 9; + VkVideoEncoderInputColourInfo mAgree = + InputColour(0, 0, 9, VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig mA = mBase; + mA.pNext = &mAgree; + VkEncBoundConfigProbe mProbeA{}; + Check(VkEncBuildAndProbeConfig( + mA, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &mProbeA) == + VK_SUCCESS, + "CONTROL: input matrix 9 with bitstream matrix 9 is accepted", + "init failed"); + VkVideoEncoderInputColourInfo mDisagree = + InputColour(0, 0, 1, VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig mB = mBase; + mB.pNext = &mDisagree; + VkEncBoundConfigProbe mProbeB{}; + Check(VkEncBuildAndProbeConfig( + mB, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &mProbeB) == + VK_ERROR_INITIALIZATION_FAILED, + "and input matrix 1 against bitstream matrix 9 is refused", + "it was accepted"); + + // AND A HALF-DECLARED PAIR IS NOT A DISAGREEMENT. 0 is UNDECLARED, so a + // caller that states one side asserts nothing about the other. Without + // this the refusal could be "any chain with a value in it". + VkVideoEncoderInputColourInfo halfA = + InputColour(1, 0, 0, VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig half = BaseConfig(); + half.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + half.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + half.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + half.pNext = &halfA; + VkEncBoundConfigProbe halfProbe{}; + Check(VkEncBuildAndProbeConfig( + half, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + &halfProbe) == VK_SUCCESS, + "an input primaries declaration against an UNDECLARED bitstream is " + "accepted -- 0 asserts nothing to contradict", + "it was refused"); +} + +// A DECLARED INPUT RANGE DECIDES THE STREAM'S RANGE ON THE LANE THAT APPLIES +// NOTHING, AND IS COMPARED ON THE LANE THAT APPLIES SOMETHING. +// +// EncoderConfig::video_full_range_flag is one variable with two consumers: +// the VUI bit the bitstream carries, and the VkSamplerYcbcrRange the +// preprocess filter is built with. They are one decision and the cases below +// read the projection of that one variable, so a change that moved only the +// VUI or only the filter would show up as a disagreement rather than as a +// pass. +// +// WHY THE TWO LANES DIFFER. On the Y'CbCr lane the library applies no range +// mapping -- the copy filter's own shader takes its output range from its +// input range, so a copy stays a copy -- and therefore the samples that +// arrive are the samples that are coded. The input's range IS the stream's +// range and a declaration of it is a fact about the stream. On the RGB lane +// the filter PRODUCES the Y'CbCr range, from this same flag, and the caller's +// declaration is about its RGB buffer; the two are different quantities, so +// the declaration is compared against what the filter can read and never +// silently retargets the output. +// +// THE REFUSAL ROWS COME IN PAIRS, one field apart. VK_ERROR_INITIALIZATION_ +// FAILED is the binder's only refusal code, so a row that merely fails could +// be failing for any reason; each refusal below sits beside an otherwise +// identical config that succeeds, which is what makes the refusal that +// field's. +void CaseDeclaredInputRangeDecidesTheStreamsRange() +{ + g_currentCase = "a declared input range decides what the stream says"; + + // CALIBRATION, BEFORE ANY DECLARATION IS READ. The two projected fields + // this case turns on have to be shown to move at all, or a run in which + // they were stuck at zero would read as "the declaration did nothing" and + // as "the declaration is undeclared" identically. + VkVideoEncoderConfig calFull = BaseConfig(); + calFull.videoFullRange = VK_TRUE; + VkEncBoundConfigProbe calFullProbe{}; + Check(VkEncBuildAndProbeConfig( + calFull, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &calFullProbe) == VK_SUCCESS, + "calibration: videoFullRange = VK_TRUE builds", "init failed"); + Check((calFullProbe.videoFullRangeFlag == 1u) && + (calFullProbe.videoSignalTypePresent == 1u), + "calibration: the two fields this case reads DO move -- " + "videoFullRange raises both", + "flag " + U32(calFullProbe.videoFullRangeFlag) + ", present " + + U32(calFullProbe.videoSignalTypePresent)); + VkVideoEncoderConfig calBare = BaseConfig(); + VkEncBoundConfigProbe calBareProbe{}; + Check(VkEncBuildAndProbeConfig( + calBare, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &calBareProbe) == VK_SUCCESS, + "calibration: an undeclared config builds", "init failed"); + Check((calBareProbe.videoFullRangeFlag == 0u) && + (calBareProbe.videoSignalTypePresent == 0u), + "calibration: and they are BOTH ZERO when nothing is declared, so " + "the instrument reads two states and not one", + "flag " + U32(calBareProbe.videoFullRangeFlag) + ", present " + + U32(calBareProbe.videoSignalTypePresent)); + + // ---- The Y'CbCr lane: the declaration is applied ---- + struct DirectRow { + VkVideoEncoderRangeDeclaration declared; + uint32_t wantFlag; + const char* what; + }; + static const DirectRow directRows[] = { + { VK_VIDEO_ENCODER_RANGE_FULL, 1u, + "a Y'CbCr input declared FULL is coded as full range" }, + { VK_VIDEO_ENCODER_RANGE_LIMITED, 0u, + "a Y'CbCr input declared LIMITED is coded as limited range" }, + }; + for (const DirectRow& row : directRows) { + VkVideoEncoderInputColourInfo ic = InputColour(0, 0, 0, row.declared); + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + cfg.pNext = ⁣ + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, Lbl(std::string("accepted: ") + row.what), + "VkResult " + U32((uint32_t)r)); + Check(probe.videoFullRangeFlag == row.wantFlag, Lbl(row.what), + "video_full_range_flag " + U32(probe.videoFullRangeFlag) + + ", wanted " + U32(row.wantFlag)); + // AND IT IS SIGNALLED. An unsignalled range is inferred by the + // standards and reported as unknown by decoders at their API + // boundary, so a caller that declared one and got silence is no + // better off than one that declared nothing. + Check(probe.videoSignalTypePresent == 1u, + Lbl(std::string("and it is SIGNALLED: ") + row.what), + "video_signal_type_present_flag " + + U32(probe.videoSignalTypePresent)); + } + + // A DECLARED LIMITED IS NOT THE SAME AS SILENCE, and this is the pair + // that says so: the same config without the chain signals nothing. + VkVideoEncoderConfig unchained = BaseConfig(); + unchained.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + VkEncBoundConfigProbe unchainedProbe{}; + Check(VkEncBuildAndProbeConfig( + unchained, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &unchainedProbe) == VK_SUCCESS, + "CONTROL: the same Y'CbCr config with no chain builds", + "init failed"); + Check((unchainedProbe.videoSignalTypePresent == 0u) && + (unchainedProbe.videoFullRangeFlag == 0u), + "CONTROL: and it declares nothing -- an absent chain is not a " + "declaration of limited range", + "present " + U32(unchainedProbe.videoSignalTypePresent) + ", flag " + + U32(unchainedProbe.videoFullRangeFlag)); + + // ---- The Y'CbCr lane: a contradiction is refused ---- + // + // videoFullRange has no undeclared state -- it is a VkBool32 whose zero + // is indistinguishable from silence -- so only the raised direction can + // be contradicted, and only that direction is refused. + VkVideoEncoderInputColourInfo limited = + InputColour(0, 0, 0, VK_VIDEO_ENCODER_RANGE_LIMITED); + VkVideoEncoderConfig clash = BaseConfig(); + clash.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + clash.videoFullRange = VK_TRUE; + clash.pNext = &limited; + VkEncBoundConfigProbe clashProbe{}; + Check(VkEncBuildAndProbeConfig( + clash, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &clashProbe) == VK_ERROR_INITIALIZATION_FAILED, + "a LIMITED input under a full-range bitstream request is refused -- " + "the library scales nothing", + "it was accepted"); + + VkVideoEncoderInputColourInfo full = + InputColour(0, 0, 0, VK_VIDEO_ENCODER_RANGE_FULL); + VkVideoEncoderConfig agree = BaseConfig(); + agree.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + agree.videoFullRange = VK_TRUE; + agree.pNext = &full; + VkEncBoundConfigProbe agreeProbe{}; + Check(VkEncBuildAndProbeConfig( + agree, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &agreeProbe) == VK_SUCCESS, + "PAIR: the same config with the input declared FULL is accepted -- " + "one field apart, so the refusal above is that field's", + "it was refused"); + Check(agreeProbe.videoFullRangeFlag == 1u, + "and both sides agreeing on full range still codes full range", + "flag " + U32(agreeProbe.videoFullRangeFlag)); + + // ---- The RGB lane: the declaration is compared, not applied ---- + VkVideoEncoderInputColourInfo rgbLimited = + InputColour(0, 0, 0, VK_VIDEO_ENCODER_RANGE_LIMITED); + VkVideoEncoderConfig rgbBad = BaseConfig(); + rgbBad.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + rgbBad.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + rgbBad.pNext = &rgbLimited; + VkEncBoundConfigProbe rgbBadProbe{}; + Check(VkEncBuildAndProbeConfig( + rgbBad, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &rgbBadProbe) == VK_ERROR_INITIALIZATION_FAILED, + "an RGB input declared LIMITED is refused -- the filter reads its " + "RGB over the full range and performs no input expansion", + "it was accepted"); + + VkVideoEncoderInputColourInfo rgbFull = + InputColour(0, 0, 0, VK_VIDEO_ENCODER_RANGE_FULL); + VkVideoEncoderConfig rgbOk = BaseConfig(); + rgbOk.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + rgbOk.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + rgbOk.pNext = &rgbFull; + VkEncBoundConfigProbe rgbOkProbe{}; + Check(VkEncBuildAndProbeConfig( + rgbOk, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &rgbOkProbe) == VK_SUCCESS, + "PAIR: the same RGB config declared FULL is accepted -- one field " + "apart", + "it was refused"); + // AND IT DOES NOT RETARGET THE OUTPUT. The RGB buffer's range and the + // Y'CbCr the filter emits are different quantities; the second is the + // bitstream request's to state. + VkVideoEncoderConfig rgbBare = BaseConfig(); + rgbBare.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + rgbBare.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + VkEncBoundConfigProbe rgbBareProbe{}; + Check(VkEncBuildAndProbeConfig( + rgbBare, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + &rgbBareProbe) == VK_SUCCESS, + "CONTROL: the same RGB config with no chain builds", "init failed"); + Check((rgbOkProbe.videoFullRangeFlag == + rgbBareProbe.videoFullRangeFlag) && + (rgbOkProbe.videoSignalTypePresent == + rgbBareProbe.videoSignalTypePresent), + "and a FULL declaration on the RGB lane leaves the bitstream's " + "range where the caller put it", + "flag " + U32(rgbOkProbe.videoFullRangeFlag) + " against " + + U32(rgbBareProbe.videoFullRangeFlag)); + + // ---- AV1, whose range syntax is unconditional ---- + // + // color_range is a mandatory bit in every AV1 sequence header, so the + // question there is never "is it signalled" but "which value" -- and + // before a declaration existed the answer was always 0. + VkVideoEncoderInputColourInfo av1Full = + InputColour(0, 0, 0, VK_VIDEO_ENCODER_RANGE_FULL); + VkVideoEncoderConfig av1 = BaseConfig(); + av1.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + av1.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + av1.pNext = &av1Full; + VkEncBoundConfigProbe av1Probe{}; + Check(VkEncBuildAndProbeConfig( + av1, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, &av1Probe) == + VK_SUCCESS, + "an AV1 session accepts a declared input range", "init failed"); + Check(av1Probe.videoFullRangeFlag == 1u, + "and an AV1 input declared FULL writes color_range 1 rather than " + "the unconditional 0 it wrote before", + "flag " + U32(av1Probe.videoFullRangeFlag)); +} + +void CaseInputColourPrimariesDriveTheDerivedMatrix() +{ + g_currentCase = "the filter's matrix is derived from the INPUT's " + "primaries when they are declared"; + // matrixCoefficients 2 (Unspecified) is the arm that derives, and the + // derived value is SIGNALLED back, so the probe reads what the filter will + // actually apply. Declaring the primaries on the INPUT side and leaving + // the bitstream's undeclared is the configuration that could only ever + // have come out BT.709 before. + struct Row { + uint8_t inputPrimaries; + uint32_t wantMatrix; + const char* what; + }; + static const Row rows[] = { + { 9u, 9u, "input primaries 9 (BT.2020) derive matrix 9" }, + { 6u, 6u, "input primaries 6 (SMPTE 170M) derive matrix 6" }, + { 1u, 1u, "input primaries 1 (BT.709) derive matrix 1" }, + }; + for (const Row& row : rows) { + VkVideoEncoderInputColourInfo ic = + InputColour(row.inputPrimaries, 0, 0, + VK_VIDEO_ENCODER_RANGE_UNDECLARED); + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + cfg.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + cfg.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + cfg.matrixCoefficients = 2u; // Unspecified: derive and signal + cfg.pNext = ⁣ + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probe); + Check(r == VK_SUCCESS, Lbl(std::string("bound: ") + row.what), + "VkResult " + U32((uint32_t)r)); + if (r != VK_SUCCESS) { + continue; + } + Check(probe.matrixCoefficients == row.wantMatrix, Lbl(row.what), + "got " + U32(probe.matrixCoefficients) + ", want " + + U32(row.wantMatrix)); + } + // THE CONTROL. With no chain the SAME config derives from the bitstream's + // primaries, which is what it did before -- so the rows above measure + // where the derivation READS and not that it derives at all. + VkVideoEncoderConfig ctl = BaseConfig(); + ctl.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + ctl.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + ctl.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + ctl.matrixCoefficients = 2u; + ctl.colourPrimaries = 9u; + VkEncBoundConfigProbe ctlProbe{}; + Check(VkEncBuildAndProbeConfig( + ctl, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &ctlProbe) == + VK_SUCCESS, + "CONTROL: the unchained config still builds", "init failed"); + Check(ctlProbe.matrixCoefficients == 9u, + "and with no chain the derivation still reads the bitstream's " + "primaries", + "got " + U32(ctlProbe.matrixCoefficients)); +} + +// THE AV1 SEQUENCE HEADER'S SUBSAMPLING AGAINST THE seq_profile IT DERIVED. +// +// InitSequenceHeader hardcoded subsampling_x = subsampling_y = 1 under a +// comment calling 4:2:0 the only chroma format this encoder admits, while +// InitProfileLevel a hundred lines below picks seq_profile 1 from 4:4:4 input +// and 2 from 4:2:2. AV1 6.4.1 gives seq_profile 1 subsampling_x == +// subsampling_y == 0, so the pair was an invalid sequence header: the two +// halves of one structure decided by two functions that disagreed. +// +// DEVICE-FREE IS THE ONLY PLACE THIS IS ASSERTABLE. AV1 High and Professional +// are absent from every driver this project can reach, so a 4:4:4 or 4:2:2 AV1 +// session dies at the capability query before a sequence header is built. +// InitSequenceHeader needs no device, and the probe calls it and reads the +// colour config back -- the same mechanism the HDR payload projection uses. +// Passing this asserts that the header the library BUILDS is self-consistent. +// It asserts nothing about any driver accepting it. +// +// THE 4:2:0 ROWS ARE THE CONTROL AND THEY RUN FIRST: they read (1, 1) before +// this change and after it. A change that wrote (0, 0) unconditionally would +// satisfy every 4:4:4 row and fail these. +// +// (internal.h warns that STD_VIDEO_AV1_PROFILE_MAIN == 0, so a codecProfile of +// 0 on the AV1 arm is a real value rather than "arm not exercised". The 4:4:4 +// and 4:2:2 rows read a non-zero profile, which removes that ambiguity where it +// would matter.) +// WHICH SIDE OF THE INPUT/ENCODE BOUNDARY EACH CODEC ARM READS. +// +// EncoderConfig carries the input's geometry and the ENCODE's geometry in +// separate fields, and they exist separately so that the encode value can +// differ from the input value -- a chroma resampler or a device-driven depth +// downgrade is what would make them differ. Today one writer sets the encode +// side from the input side and nothing else touches either, so the two are +// always equal and a read of the wrong one costs nothing. +// +// WHAT THE ARMS MUST READ, AND IT IS NOT A PREFERENCE. A codec profile is +// defined by the standard over the values CARRIED IN THE BITSTREAM: H.264 +// Annex A Table A-1 constrains profile_idc against the SPS's chroma_format_idc +// and bit_depth_*_minus8; H.265 Annex A does the same for +// general_profile_idc; AV1 6.4.1 defines seq_profile over the sequence +// header's BitDepth, mono_chrome and subsampling_x/y. Every one of those +// syntax elements is written in this tree from the ENCODE fields. So a +// derivation that picks the profile from the INPUT side selects a profile for +// a picture that is not the one the syntax describes. +// +// WHAT THIS CASE ASSERTS is exactly that rule and nothing weaker: the profile +// the arm derived is the profile the ENCODE geometry implies, with the encode +// geometry read off the probe rather than assumed to equal the input's. The +// three tables below restate the standards' rule; they are a second statement +// of the derivation, deliberately, because a test that recomputed it by +// calling the derivation would agree with itself whichever side it read. +// +// IT IS SWEPT OVER EVERY ROUTABLE FORMAT AND ALL THREE CODECS, at +// VK_VIDEO_ENCODER_PROFILE_DEFAULT, which is the only profile value that +// reaches the derivations at all. +// IS THE ORACLE'S OWN ANSWER LEGAL AT THAT GEOMETRY? +// +// An expectation table can be wrong in a way no comparison against the code +// catches: if the code and the table make the same mistake the row passes and +// ratifies it. This is the second reading, from the standards' limits and not +// from the derivations -- H.264 Table A-1, H.265 A.3, AV1 6.4.1 and A.2 -- +// and every row the sweep asserts is put through it first. +// +// IT IS NOT THE LIBRARY'S OWN LIMITS TABLE. VkEncGetProfileInputLimits states +// the same rule inside the library; calling it here would make the check +// agree with whatever that table says, which is exactly the shape of +// self-agreement this exists to break. +bool ProfileAdmitsGeometry(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, uint32_t subsampling, + uint32_t bpp) +{ + const bool is420 = (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR); + const bool is422 = (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR); + const bool is444 = (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR); + switch ((uint32_t)codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + switch (profile) { + case STD_VIDEO_H264_PROFILE_IDC_BASELINE: + case STD_VIDEO_H264_PROFILE_IDC_MAIN: + case STD_VIDEO_H264_PROFILE_IDC_HIGH: + return is420 && (bpp == 8); + case STD_VIDEO_H264_PROFILE_IDC_HIGH_10: + return is420 && (bpp <= 10); + case STD_VIDEO_H264_PROFILE_IDC_HIGH_422: + return (is420 || is422) && (bpp <= 10); + case STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE: + return (is420 || is422 || is444) && (bpp <= 14); + default: return false; + } + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + switch (profile) { + case STD_VIDEO_H265_PROFILE_IDC_MAIN: + case STD_VIDEO_H265_PROFILE_IDC_MAIN_STILL_PICTURE: + return is420 && (bpp == 8); + case STD_VIDEO_H265_PROFILE_IDC_MAIN_10: + return is420 && (bpp <= 10); + case STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS: + case STD_VIDEO_H265_PROFILE_IDC_SCC_EXTENSIONS: + return (is420 || is422 || is444) && (bpp <= 16); + default: return false; + } + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + switch (profile) { + case STD_VIDEO_AV1_PROFILE_MAIN: return is420 && (bpp <= 10); + case STD_VIDEO_AV1_PROFILE_HIGH: return is444 && (bpp <= 10); + case STD_VIDEO_AV1_PROFILE_PROFESSIONAL: + return (is420 || is422 || is444) && (bpp <= 12); + default: return false; + } + default: return false; + } +} + +// "THE STANDARD ADMITS NO ANSWER THIS LIBRARY DERIVES." Returned instead of a +// profile for a geometry whose only in-spec H.264 answer the derivation does +// not produce; the sweep excludes those rows loudly rather than writing a +// forbidden profile into an expectation. +const uint32_t kNoAdmissibleProfile = 0xFFFFFFFFu; + +uint32_t WantH264Profile(uint32_t subsampling, uint32_t bpp) +{ + // adaptiveTransformMode has no setter, so use8x8Transform is always true + // and the Baseline/Main seeds are unreachable -- the derivation starts at + // High and is only widened from there. + if (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR) { + // 244 admits 8 to 14 bits at every chroma format, so this arm is + // in-spec at every depth the library can reach. + return STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE; + } + // TODO: H.264 above ten bits at 4:2:0 or 4:2:2 HAS NO IN-SPEC ANSWER FROM + // THIS DERIVATION, and the defect is EncoderConfigH264::InitProfileLevel's, + // not this table's. ITU-T H.264 Table A-1 admits High 10 (110) and High + // 4:2:2 (122) to ten bits; the derivation selects them from any depth + // above eight, so a twelve-bit 4:2:0 input derives 110 and a twelve-bit + // 4:2:2 input derives 122, both out of spec. The in-spec answer at those + // geometries is High 4:4:4 Predictive (244), which admits chroma formats + // 0 to 3 and fourteen bits. + // + // Fixing the derivation is out of this change's scope -- no device here + // encodes H.264 above eight bits, so the arm is unreachable in practice + // and correcting it needs hardware nobody has -- but WRITING 110 INTO AN + // ORACLE AS THE EXPECTED ANSWER IS NOT. That documents the defect as + // correct, which is the one thing a test must not do. The rows are + // excluded, counted and named instead, so the exclusion is visible on + // every run rather than being a silently missing assertion. + if (bpp > 10) { + return kNoAdmissibleProfile; + } + if (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR) { + return STD_VIDEO_H264_PROFILE_IDC_HIGH_422; + } + return (bpp > 8) ? (uint32_t)STD_VIDEO_H264_PROFILE_IDC_HIGH_10 + : (uint32_t)STD_VIDEO_H264_PROFILE_IDC_HIGH; +} + +uint32_t WantH265Profile(uint32_t subsampling, uint32_t bpp) +{ + if (subsampling != VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR) { + return STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS; + } + if (bpp == 8) { + return STD_VIDEO_H265_PROFILE_IDC_MAIN; + } + if (bpp <= 10) { + return STD_VIDEO_H265_PROFILE_IDC_MAIN_10; + } + return STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS; +} + +uint32_t WantAv1Profile(uint32_t subsampling, uint32_t bpp) +{ + if ((bpp > 10) || + (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_422_BIT_KHR)) { + return STD_VIDEO_AV1_PROFILE_PROFESSIONAL; + } + if (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR) { + return STD_VIDEO_AV1_PROFILE_HIGH; + } + return STD_VIDEO_AV1_PROFILE_MAIN; +} + +void CaseCodecArmsDeriveTheProfileFromTheEncodeGeometry() +{ + g_currentCase = "the profile each codec arm derives is the one the ENCODE " + "geometry implies"; + + struct CodecRow { + VkVideoCodecOperationFlagBitsKHR codec; + const char* name; + uint32_t (*want)(uint32_t, uint32_t); + }; + // Indexed as well as iterated below, so the per-codec divergence counters + // in the second pass line up with the arms they count. + static const CodecRow kCodecs[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "H.264", + &WantH264Profile }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, "H.265", + &WantH265Profile }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, "AV1", + &WantAv1Profile }, + }; + + uint32_t routableCount = 0; + const VkFormat* const routable = VkEncRoutableInputFormats(routableCount); + uint32_t rows = 0; + uint32_t asserted = 0; + uint32_t excluded = 0; + for (const CodecRow& c : kCodecs) { + for (uint32_t i = 0; i < routableCount; i++) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = c.codec; + cfg.profile = VK_VIDEO_ENCODER_PROFILE_DEFAULT; + cfg.inputFormat = routable[i]; + VkEncBoundConfigProbe probe{}; + const VkResult r = + VkEncBuildAndProbeConfig(cfg, c.codec, &probe); + if (r != VK_SUCCESS) { + continue; + } + rows++; + const uint32_t want = c.want(probe.encodeChromaSubsampling, + probe.encodeBitDepthLuma); + if (want == kNoAdmissibleProfile) { + excluded++; + std::printf(" EXCLUDED [%s] %s enumerant %u: encode geometry " + "(subsampling %u, %u-bit) has no in-spec profile " + "this derivation produces; the arm derived %u. " + "See the TODO on WantH264Profile.\n", + g_currentCase, c.name, (uint32_t)routable[i], + probe.encodeChromaSubsampling, + probe.encodeBitDepthLuma, probe.codecProfile); + continue; + } + asserted++; + // THE EXPECTATION IS CHECKED BEFORE IT IS USED. A table that + // names a profile the standard does not admit at that geometry is + // documenting a defect as the right answer, whatever the code + // then does. + Check(ProfileAdmitsGeometry(c.codec, want, + probe.encodeChromaSubsampling, + probe.encodeBitDepthLuma), + (std::string(c.name) + " enumerant " + + U32((uint32_t)routable[i]) + + ": the profile this case EXPECTS is one the standard " + "admits at that geometry").c_str(), + "expects profile " + U32(want) + " at (subsampling " + + U32(probe.encodeChromaSubsampling) + ", " + + U32(probe.encodeBitDepthLuma) + "-bit)"); + Check(probe.codecProfile == want, + (std::string(c.name) + " enumerant " + + U32((uint32_t)routable[i]) + + ": the profile follows the ENCODE geometry").c_str(), + "encode geometry is (subsampling " + + U32(probe.encodeChromaSubsampling) + ", " + + U32(probe.encodeBitDepthLuma) + + "-bit) which implies profile " + U32(want) + + ", the arm derived " + U32(probe.codecProfile) + + "; the input side was (subsampling " + + U32(probe.inputChromaSubsampling) + ", " + + U32(probe.inputBpp) + "-bit)"); + } + } + Check(rows >= 40u, + "the sweep bound enough rows across the three arms to be read", + "bound " + U32(rows)); + // THE EXCLUSION CANNOT HOLLOW THE CASE OUT. Counted separately from the + // bound rows so that a widening of the excluded geometry shows up here as + // a falling assertion count rather than as a still-green run. + Check(asserted >= 40u, + "and enough of them carried an in-spec expectation to assert", + "asserted " + U32(asserted) + " of " + U32(rows) + ", excluded " + + U32(excluded)); + + // ---- SECOND PASS: THE TWO SIDES MADE TO DIFFER ---- + // + // WHAT THE PASS ABOVE CANNOT SAY. One writer sets the encode side from + // the input side and nothing else touches either, so on every state the + // pass above can reach the two are EQUAL -- and an assertion over equal + // values is satisfied identically whichever side an arm reads. Revert all + // three arms to input.bpp and every row above still passes, with the same + // check count. It is a guard against a wrong derivation and no guard at + // all against a wrong side, which is the regression it was written for. + // + // WHAT MAKES THE DIFFERENCE REACHABLE. EncoderConfig::InitializeParameters + // derives the encode depth under a zero-means-unset guard whose own + // comment calls an explicit encode depth "a request and not a default". + // VkEncBuildAndProbeConfig's fourth argument states one, so the encode + // side becomes the request and the input side stays the caller's format's. + // The two now disagree, and the profile an arm derives says which it + // read: nothing else in the configuration moved. + // + // THE REQUEST IS THE FAR END OF THE DEPTH RANGE from the input, because a + // near one implies the same profile on most rows and a row where both + // sides imply the same answer asserts nothing. How many rows actually + // diverged is COUNTED PER CODEC and asserted non-zero below -- without + // that this pass could go green having compared every row against itself. + uint32_t divergent[3] = {0u, 0u, 0u}; + uint32_t requested = 0u; + for (uint32_t ci = 0; ci < 3u; ci++) { + const CodecRow& c = kCodecs[ci]; + for (uint32_t i = 0; i < routableCount; i++) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = c.codec; + cfg.profile = VK_VIDEO_ENCODER_PROFILE_DEFAULT; + cfg.inputFormat = routable[i]; + + VkEncBoundConfigProbe derived{}; + if (VkEncBuildAndProbeConfig(cfg, c.codec, &derived) != + VK_SUCCESS) { + continue; + } + const uint32_t request = (derived.inputBpp <= 10u) ? 12u : 8u; + + VkEncBoundConfigProbe stated{}; + if (VkEncBuildAndProbeConfig(cfg, c.codec, &stated, request) != + VK_SUCCESS) { + continue; + } + // THE FIXTURE IS ASSERTED BEFORE THE PROPOSITION. If the request + // did not land, or if it moved the input side too, the row below + // would be measuring a broken instrument rather than an arm. + const bool fixtureOk = (stated.encodeBitDepthLuma == request) && + (stated.inputBpp == derived.inputBpp) && + (stated.encodeChromaSubsampling == + derived.encodeChromaSubsampling); + Check(fixtureOk, + Lbl(std::string(c.name) + " enumerant " + + U32((uint32_t)routable[i]) + + ": the stated encode depth landed on the ENCODE side " + "alone"), + "encode " + U32(stated.encodeBitDepthLuma) + " (asked " + + U32(request) + "), input " + U32(stated.inputBpp) + + " (was " + U32(derived.inputBpp) + ")"); + if (!fixtureOk) { + continue; + } + + const uint32_t wantEncode = + c.want(stated.encodeChromaSubsampling, request); + const uint32_t wantInput = + c.want(stated.encodeChromaSubsampling, stated.inputBpp); + if (wantEncode == kNoAdmissibleProfile) { + excluded++; + continue; + } + requested++; + if (wantEncode != wantInput) { + divergent[ci]++; + } + Check(ProfileAdmitsGeometry(c.codec, wantEncode, + stated.encodeChromaSubsampling, + request), + Lbl(std::string(c.name) + " enumerant " + + U32((uint32_t)routable[i]) + + ": the stated-depth expectation is in spec too"), + "expects " + U32(wantEncode) + " at " + U32(request) + + "-bit"); + Check(stated.codecProfile == wantEncode, + Lbl(std::string(c.name) + " enumerant " + + U32((uint32_t)routable[i]) + + ": with the two sides DIFFERENT, the arm follows the " + "ENCODE side"), + "encode side is (subsampling " + + U32(stated.encodeChromaSubsampling) + ", " + + U32(request) + "-bit) implying profile " + + U32(wantEncode) + "; the INPUT side is " + + U32(stated.inputBpp) + "-bit implying " + + ((wantInput == kNoAdmissibleProfile) + ? std::string("no in-spec profile") + : U32(wantInput)) + + "; the arm derived " + U32(stated.codecProfile)); + } + } + std::printf(" STATED-DEPTH PASS [%s]: %u rows asserted, divergent per " + "arm H.264=%u H.265=%u AV1=%u\n", + g_currentCase, requested, divergent[0], divergent[1], + divergent[2]); + Check(requested >= 20u, + "the stated-depth pass bound enough rows to be read", + "asserted " + U32(requested)); + for (uint32_t ci = 0; ci < 3u; ci++) { + // THE ANTI-TAUTOLOGY ASSERTION. Without this the pass above could be + // green because the two sides never once implied different profiles, + // which is precisely the condition that made the first pass a + // non-guard. + Check(divergent[ci] > 0u, + Lbl(std::string(kCodecs[ci].name) + + ": the stated-depth pass contained rows where the two sides " + "imply DIFFERENT profiles, so it can discriminate"), + "divergent rows " + U32(divergent[ci])); + } +} + +void CaseAv1SubsamplingMatchesTheDerivedSeqProfile() +{ + g_currentCase = "the AV1 sequence header's subsampling matches the " + "seq_profile the derivation chose"; + struct Row { + VkFormat fmt; + uint32_t wantProfile; // StdVideoAV1Profile + uint32_t wantX; + uint32_t wantY; + const char* what; + }; + static const Row rows[] = { + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, 0u, 1u, 1u, + "NV12 is Main (0) at (1, 1)" }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, 0u, 1u, 1u, + "P010 is Main (0) at (1, 1) -- ten bits does not change seq_profile" }, + { VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, 1u, 0u, 0u, + "NV24 is High (1), which REQUIRES (0, 0)" }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, 1u, 0u, 0u, + "S410 is High (1) at (0, 0)" }, + { VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, 2u, 1u, 0u, + "NV16 is Professional (2), which at ten bits or fewer is (1, 0)" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + cfg.inputFormat = row.fmt; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, &probe); + Check(r == VK_SUCCESS, Lbl(std::string("bound: ") + row.what), + "VkResult " + U32((uint32_t)r)); + if (r != VK_SUCCESS) { + continue; + } + Check(probe.av1ColorConfigPresent == 1u, + Lbl(std::string("a colour config was attached: ") + row.what), + "av1ColorConfigPresent " + U32(probe.av1ColorConfigPresent)); + Check(probe.codecProfile == row.wantProfile, + Lbl(std::string("the derivation picked its seq_profile: ") + + row.what), + "got " + U32(probe.codecProfile) + ", want " + + U32(row.wantProfile)); + Check((probe.av1SubsamplingX == row.wantX) && + (probe.av1SubsamplingY == row.wantY), + Lbl(row.what), + "got (" + U32(probe.av1SubsamplingX) + ", " + + U32(probe.av1SubsamplingY) + "), want (" + U32(row.wantX) + + ", " + U32(row.wantY) + ")"); + } +} + +// ITU-T H.265 TABLE A.8, WHICH THE 4:4:4 ARM COULD NOT REACH. +// +// GetCpbVclFactor() assigned encodeChromaSubsampling -- a +// VkVideoChromaSubsamplingFlagBitsKHR, 0x2 / 0x4 / 0x8 -- into a variable +// called chroma_format_idc and then tested it against the VALUE 3. No real +// input makes 0x2, 0x4 or 0x8 equal 3, so every stream took the 4:2:0 factor +// of 1000, including the 4:4:4 ones that Table A.8 gives 2000 at eight bits +// and 2500 at ten. +// +// THE FACTOR IS READ DIRECTLY, and that is the point of projecting it. A +// too-high level is still a LEGAL level, which is why nothing caught this, so +// the level alone would be a weak instrument; and the plan's other candidate, +// the default vbvBufferSize, is not readable device-free at all -- the probe +// projects the CONFIG field, and InitRateControl, which is what computes the +// default from the factor, runs later and needs a session. It reads 0 here +// under the broken factor and under the fixed one alike. +// +// THE LEVEL AND TIER ARE STILL READ, at a bitrate that makes them BITE. +// IsSuitableLevel tests averageBitrate against maxBitRateMainTier x cpbFactor, +// and when main tier will not carry the bitrate DetermineLevelTier does not +// climb to the next level -- it takes HIGH TIER at the same one. So at 16 +// Mbit/s on 1080p a 4:4:4 session sits at level 4.0 MAIN tier under the correct +// factor (12000 x 2000 = 24 Mbit/s) and at level 4.0 HIGH tier under the broken +// one (12000 x 1000 = 12 Mbit/s). The LEVEL is 4.0 either way, which is exactly +// why it is read together with the tier and never alone. +// +// At the default 4 Mbit/s neither ceiling binds and both terms are +// picture-size-bound, which is why the rows below carry three bitrates: one +// where the selection cannot move, and two where it must. +// +// THE DEPTH TERM IS NOW LIVE AT THIS CALL SITE TOO, and the rows read it at two +// chroma formats so that "the depth term works" is distinguishable from "the +// 4:4:4 cell was edited". Table A.8, for the formats this tree can reach: +// +// 4:2:0 8-bit Main 1000 +// 4:2:0 10-bit Main 10 1000 +// 4:2:0 12-bit Main 12 1500 +// 4:4:4 8-bit Main 4:4:4 2000 +// 4:4:4 10-bit Main 4:4:4 10 2500 +// +// TWO PAIRS OF ROWS ISOLATE IT AT A FIXED BITRATE AND A FIXED CHROMA. At 16 +// Mbit/s a 4:2:0 stream sits at level 4.0 HIGH tier at eight bits (12000 x 1000 +// = 12 Mbit/s, exceeded) and at level 4.0 MAIN tier at twelve (12000 x 1500 = +// 18 Mbit/s, not exceeded). At 28 Mbit/s a 4:4:4 stream sits at HIGH tier at +// eight bits (12000 x 2000 = 24 Mbit/s, exceeded) and at MAIN tier at ten +// (12000 x 2500 = 30 Mbit/s, not exceeded). In each pair the only variable is +// the depth, so a factor that ignored the depth would read the same tier for +// both members and a factor that had simply been raised everywhere would move +// the eight-bit member too. +// +// MAIN TIER IS THE CORRECT ANSWER WHERE IT IS ASSERTED, not merely a different +// one. H.265 Annex A selects the lowest tier and level whose limits the stream +// satisfies; high tier at level 4.0 is also conformant for these streams and is +// a stricter claim on the decoder than the bitstream needs. +void CaseH265CpbVclFactorFollowsTheChromaFormat() +{ + g_currentCase = "the H.265 CPB VCL factor is Table A.8's, per chroma " + "format and depth"; + // wantLevel is StdVideoH265LevelIdc: 5 is 4.0. wantTier is + // general_tier_flag: 0 main, 1 high. + struct Row { + VkFormat fmt; + uint32_t bitrate; + uint32_t wantFactor; + uint32_t wantLevel; + uint32_t wantTier; + const char* what; + }; + static const Row rows[] = { + // THE CONTROLS, AND THEY RUN FIRST. 4:2:0 reads 1000 and makes the + // same selection at BOTH bitrates. + // Without them, "fixed the 4:4:4 arm" is indistinguishable from + // "changed the factor everywhere". + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, 4000000u, + 1000u, 5u, 0u, "8-bit 4:2:0 is 1000, at level 4.0 main tier" }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, 4000000u, + 1000u, 5u, 0u, + "10-bit 4:2:0 is 1000 -- Table A.8's depth term is +500 per two bits " + "ABOVE ten, so ten bits adds nothing" }, + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, 16000000u, + 1000u, 5u, 1u, + "8-bit 4:2:0 at 16 Mbit/s needs HIGH tier, which the factor does not " + "change" }, + // THE CHROMA ARM. + { VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, 4000000u, + 2000u, 5u, 0u, "8-bit 4:4:4 (NV24) is 2000" }, + { VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, 16000000u, + 2000u, 5u, 0u, + "8-bit 4:4:4 at 16 Mbit/s fits level 4.0 MAIN tier on the right " + "chroma factor" }, + // THE DEPTH ARM, AT 4:2:0. Table A.8 gives Main 12 the factor 1500, + // which is the base 1000 plus one +500 step for the two bits above + // ten. The pair at 16 Mbit/s is where it bites: the eight-bit row + // three above takes HIGH tier at the same bitrate and this one does + // not, and the depth is the only difference between them. + { VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, 4000000u, + 1500u, 5u, 0u, "12-bit 4:2:0 (P012) is 1500" }, + { VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, 16000000u, + 1500u, 5u, 0u, + "12-bit 4:2:0 at 16 Mbit/s fits level 4.0 MAIN tier, where 8-bit " + "4:2:0 at the same bitrate does not" }, + // THE DEPTH ARM, AT 4:4:4, where the base factor moves as well as the + // step: Table A.8 gives Main 4:4:4 10 the factor 2500 against Main + // 4:4:4's 2000. + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, 4000000u, + 2500u, 5u, 0u, "10-bit 4:4:4 (S410) is 2500" }, + // AND THE PAIR THAT MOVES THE TIER. 28 Mbit/s is above main tier's + // ceiling at 2000 (24 Mbit/s) and below it at 2500 (30 Mbit/s), so the + // eight-bit member must still take HIGH tier and the ten-bit member + // must not. Read together they say the depth term moved the selection; + // read alone either would only say the selection is bitrate-sensitive. + { VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, 28000000u, + 2000u, 5u, 1u, + "8-bit 4:4:4 at 28 Mbit/s still needs HIGH tier at level 4.0" }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, 28000000u, + 2500u, 5u, 0u, + "10-bit 4:4:4 at 28 Mbit/s fits level 4.0 MAIN tier, where 8-bit " + "4:4:4 at the same bitrate does not" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + cfg.inputFormat = row.fmt; + cfg.averageBitrate = row.bitrate; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probe); + Check(r == VK_SUCCESS, Lbl(std::string("bound: ") + row.what), + "VkResult " + U32((uint32_t)r)); + if (r != VK_SUCCESS) { + continue; + } + Check(probe.h265CpbVclFactor == row.wantFactor, Lbl(row.what), + "got " + U32(probe.h265CpbVclFactor) + ", want " + + U32(row.wantFactor)); + Check((probe.h265LevelIdc == row.wantLevel) && + (probe.h265GeneralTierFlag == row.wantTier), + Lbl(std::string("and the level and tier it selects: ") + + row.what), + "got StdVideoH265LevelIdc " + U32(probe.h265LevelIdc) + + " tier " + U32(probe.h265GeneralTierFlag) + ", want " + + U32(row.wantLevel) + " tier " + U32(row.wantTier)); + } +} + +// THE BIND SET IS THE STANDARD'S LIMITS TABLE, AND THIS IS THE REACH CHECK. +// +// The derivation already selects H.264 High 4:4:4 Predictive, H.265 Range +// Extensions and AV1 High from 4:4:4 input, and the library emits those +// streams, so naming the same number explicitly must not be refused as +// "unbindable". The bind set covers every number the limits table states, +// and the standard's own limits refuse what the standard forbids. +// +// WHY A PAIR AND NOT AN ACCEPTANCE. VkEncBuildAndProbeConfig returns +// VK_ERROR_INITIALIZATION_FAILED for BOTH the old unbindable refusal and the +// limits refusal, so a bare "it is refused" assertion cannot say which line +// produced it. Each row below is therefore two configs on ONE profile whose +// limits row is NARROW: one input the profile admits, one it does not. Only an +// arm that both binds the number AND calls the limits guard makes both true. +// +// NOT 244, AND NOT H.265 4. Both admit 4:2:0 as well as 4:4:4, so no input +// format makes either refuse on subsampling and the pair would collapse into a +// single assertion. AV1 High is 4:4:4 ONLY and H.264 High 10 is 4:2:0 ONLY, +// which is what makes them discriminating. +void CaseWidenedBindSetStillRunsTheLimitsGuard() +{ + g_currentCase = "a newly bindable profile still refuses what the standard " + "denies it"; + struct Pair { + VkVideoCodecOperationFlagBitsKHR codec; + uint32_t profile; + VkFormat admitted; + const char* admittedWhy; + VkFormat denied; + const char* deniedWhy; + }; + static const Pair pairs[] = { + { VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, + STD_VIDEO_AV1_PROFILE_HIGH, + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + "AV1 High (1) over 4:4:4 (NV24) binds seq_profile 1", + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + "AV1 High (1) over 4:2:0 (NV12) is refused -- High is 4:4:4 only" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + STD_VIDEO_H264_PROFILE_IDC_HIGH_10, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + "H.264 High 10 (110) over 4:2:0 (NV12) binds profile_idc 110", + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, + "H.264 High 10 (110) over 4:4:4 (NV24) is refused -- 110 is 4:2:0 " + "only" }, + }; + for (const Pair& p : pairs) { + VkVideoEncoderConfig okCfg = BaseConfig(); + okCfg.codec = p.codec; + okCfg.profile = p.profile; + okCfg.inputFormat = p.admitted; + VkEncBoundConfigProbe okProbe{}; + const VkResult okR = + VkEncBuildAndProbeConfig(okCfg, p.codec, &okProbe); + Check(okR == VK_SUCCESS, Lbl(std::string("accepted: ") + p.admittedWhy), + "VkResult " + U32((uint32_t)okR)); + Check(okProbe.codecProfile == p.profile, + Lbl(std::string("and binds the number it was given: ") + + p.admittedWhy), + "got " + U32(okProbe.codecProfile)); + + VkVideoEncoderConfig badCfg = BaseConfig(); + badCfg.codec = p.codec; + badCfg.profile = p.profile; + badCfg.inputFormat = p.denied; + VkEncBoundConfigProbe badProbe{}; + const VkResult badR = + VkEncBuildAndProbeConfig(badCfg, p.codec, &badProbe); + Check(badR == VK_ERROR_INITIALIZATION_FAILED, + Lbl(std::string("refused: ") + p.deniedWhy), + "VkResult " + U32((uint32_t)badR)); + } + + // THE MIRROR CONTROL. A change that widened the LIMITS table rather than + // the bind set, or that stopped refusing altogether, would pass everything + // above. H.265 Main (1) is 4:2:0 only and must still refuse 4:4:4, and it + // is bindable. + VkVideoEncoderConfig mainCfg = BaseConfig(); + mainCfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + mainCfg.profile = VK_VIDEO_ENCODER_PROFILE_H265_MAIN; + mainCfg.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_444_UNORM; + VkEncBoundConfigProbe mainProbe{}; + const VkResult mainR = VkEncBuildAndProbeConfig( + mainCfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &mainProbe); + Check(mainR == VK_ERROR_INITIALIZATION_FAILED, + "H.265 Main (1) over 4:4:4 is STILL refused -- the bind set widened, " + "the standard's limits did not", + "VkResult " + U32((uint32_t)mainR)); + + // AND A NUMBER THE TABLE STATES NOTHING ABOUT IS STILL UNBINDABLE, so + // "widened" is not "opened". 88 is not an H.264 profile_idc. + VkVideoEncoderConfig junkCfg = BaseConfig(); + junkCfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + junkCfg.profile = 88u; + junkCfg.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + VkEncBoundConfigProbe junkProbe{}; + const VkResult junkR = VkEncBuildAndProbeConfig( + junkCfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &junkProbe); + Check(junkR == VK_ERROR_INITIALIZATION_FAILED, + "an H.264 profile_idc the standard's table does not state is still " + "unbindable", + "VkResult " + U32((uint32_t)junkR)); +} + +// EVERY PROFILE THE BINDER BINDS IS ONE THE PUBLIC HEADER NAMES, AND THE +// CONVERSE. +// +// The header's profile block says of its constants that "they are what this +// library binds today", and the sentence beside VkVideoEncoderConfig::profile +// routes a caller there "for the numbering and for what this library binds". +// Those were enumeration claims that nothing enforced: the bind set widened +// from six numbers to fourteen and the block did not move, so the header +// advertised a NARROWER set than the library accepted and the only way to ask +// for the difference was a bare integer. The tree's own point-query row for +// High 4:4:4 Predictive wrote 244u for exactly that reason. This case is what +// enforces them now, in both directions. +// +// SWEPT, NOT LISTED. A table of the fourteen checked against a table of the +// fourteen would agree with itself. This walks the whole value space each +// codec's syntax element can carry -- profile_idc and general_profile_idc are +// u(8), seq_profile is f(3) -- and asks the binder about every number in it, +// so a value added to the switch and not to the header fails here on the first +// run rather than on the first consumer. +// +// BINDABILITY IS "SOME INPUT ADMITS IT", because the second half of the +// binder's rule is the standard's limits table and most numbers are refused by +// it on 4:2:0 8-bit input. Six inputs span that table's whole domain: 8, 10 +// and 12 bits at 4:2:0, and 4:2:2 and 4:4:4. A number no input admits is not a +// number a caller can use. +// +// THE OVERLAP IS STATED, NOT WORKED AROUND. On AV1, 0 is seq_profile Main and +// is also VK_VIDEO_ENCODER_PROFILE_DEFAULT, so the binder never sees it and +// the derivation produces it; the sweep reads it as bound because a caller +// that writes 0 does get seq_profile 0, which is what the constant promises. +// On H.264 and H.265, 0 is DEFAULT alone and the derivation lands elsewhere, +// so it reads as unbound there. +struct NamedProfile { + uint32_t value; + const char* spelling; +}; + +// The public header's own enumeration, transcribed. This table is the header's +// claim; the sweep below is the code's answer. +const NamedProfile kNamedH264[] = { + { VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, + "VK_VIDEO_ENCODER_PROFILE_H264_BASELINE" }, + { VK_VIDEO_ENCODER_PROFILE_H264_MAIN, + "VK_VIDEO_ENCODER_PROFILE_H264_MAIN" }, + { VK_VIDEO_ENCODER_PROFILE_H264_HIGH, + "VK_VIDEO_ENCODER_PROFILE_H264_HIGH" }, + { VK_VIDEO_ENCODER_PROFILE_H264_HIGH_10, + "VK_VIDEO_ENCODER_PROFILE_H264_HIGH_10" }, + { VK_VIDEO_ENCODER_PROFILE_H264_HIGH_422, + "VK_VIDEO_ENCODER_PROFILE_H264_HIGH_422" }, + { VK_VIDEO_ENCODER_PROFILE_H264_HIGH_444_PREDICTIVE, + "VK_VIDEO_ENCODER_PROFILE_H264_HIGH_444_PREDICTIVE" }, +}; +const NamedProfile kNamedH265[] = { + { VK_VIDEO_ENCODER_PROFILE_H265_MAIN, + "VK_VIDEO_ENCODER_PROFILE_H265_MAIN" }, + { VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, + "VK_VIDEO_ENCODER_PROFILE_H265_MAIN10" }, + { VK_VIDEO_ENCODER_PROFILE_H265_MAIN_STILL_PICTURE, + "VK_VIDEO_ENCODER_PROFILE_H265_MAIN_STILL_PICTURE" }, + { VK_VIDEO_ENCODER_PROFILE_H265_FORMAT_RANGE_EXTENSIONS, + "VK_VIDEO_ENCODER_PROFILE_H265_FORMAT_RANGE_EXTENSIONS" }, + { VK_VIDEO_ENCODER_PROFILE_H265_SCC_EXTENSIONS, + "VK_VIDEO_ENCODER_PROFILE_H265_SCC_EXTENSIONS" }, +}; +const NamedProfile kNamedAv1[] = { + { VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, + "VK_VIDEO_ENCODER_PROFILE_AV1_MAIN" }, + { VK_VIDEO_ENCODER_PROFILE_AV1_HIGH, + "VK_VIDEO_ENCODER_PROFILE_AV1_HIGH" }, + { VK_VIDEO_ENCODER_PROFILE_AV1_PROFESSIONAL, + "VK_VIDEO_ENCODER_PROFILE_AV1_PROFESSIONAL" }, +}; + +// Does |profile| bind on |codec| for at least one input the standard's limits +// table admits? +bool ProfileBindsOnSomeInput(VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile) +{ + static const VkFormat kSpan[] = { + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, // 8-bit 4:2:0 + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, // 10-bit 4:2:0 + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, // 12-bit 4:2:0 + VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, // 8-bit 4:2:2 + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, // 8-bit 4:4:4 + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, // 10-bit 4:4:4 + }; + for (VkFormat fmt : kSpan) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = codec; + cfg.profile = profile; + cfg.inputFormat = fmt; + VkEncBoundConfigProbe probe{}; + if (VkEncBuildAndProbeConfig(cfg, codec, &probe) != VK_SUCCESS) { + continue; + } + if (probe.codecProfile == profile) { + return true; + } + } + return false; +} + +void SweepOneCodecsProfileSpace(VkVideoCodecOperationFlagBitsKHR codec, + const char* codecName, + const NamedProfile* named, + size_t namedCount, + uint32_t valueSpace) +{ + // Half one: every constant the header names must bind. A name for a value + // the library refuses is the same defect pointing the other way. + for (size_t i = 0; i < namedCount; i++) { + Check(ProfileBindsOnSomeInput(codec, named[i].value), + Lbl(std::string(named[i].spelling) + " (" + + U32(named[i].value) + ") is a profile this library binds"), + "no input in the span bound it"); + } + + // Half two: nothing outside the named set binds. + std::string unnamed; + uint32_t unnamedCount = 0; + for (uint32_t v = 0; v < valueSpace; v++) { + bool isNamed = false; + for (size_t i = 0; i < namedCount; i++) { + if (named[i].value == v) { + isNamed = true; + break; + } + } + if (isNamed || !ProfileBindsOnSomeInput(codec, v)) { + continue; + } + unnamedCount++; + if (!unnamed.empty()) { + unnamed += ", "; + } + unnamed += U32(v); + } + Check(unnamedCount == 0, + Lbl(std::string(codecName) + + ": the header names every profile the binder binds"), + "bindable and unnamed: " + unnamed); +} + +void CaseNamedProfileConstantsAreExactlyTheBoundSet() +{ + g_currentCase = "the named profile constants are exactly what the binder " + "binds"; + + // CALIBRATION, BEFORE THE SWEEP READS ANYTHING. The sweep's verdict is a + // membership test, and a membership test that answered the same for every + // input would report a clean set-equality no matter what the binder did. + // One number known to bind and one known not to, through the same + // function, on the same inputs. + Check(ProfileBindsOnSomeInput( + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH), + "calibration: the sweep reads H.264 High (100) as BOUND", + "the instrument cannot see a bound profile"); + Check(!ProfileBindsOnSomeInput( + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, 88u), + "calibration: the sweep reads 88, which is no profile_idc, as " + "UNBOUND", + "the instrument reports everything bound"); + + // profile_idc and general_profile_idc are u(8); seq_profile is f(3). + SweepOneCodecsProfileSpace(VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + "H.264", kNamedH264, + sizeof(kNamedH264) / sizeof(kNamedH264[0]), + 256u); + SweepOneCodecsProfileSpace(VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + "H.265", kNamedH265, + sizeof(kNamedH265) / sizeof(kNamedH265[0]), + 256u); + SweepOneCodecsProfileSpace(VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, + "AV1", kNamedAv1, + sizeof(kNamedAv1) / sizeof(kNamedAv1[0]), + 8u); +} + +//============================================================================= +// 2b. What the capability probe can ASK a driver +// +// The probe is keyed on (codec, profile, bit depth). These assert the KEY, not +// a device answer: this runner has no encode-capable device, so no capability +// entry point can answer anything but "not present", and a claim about what a +// driver reports would be unfounded. +//============================================================================= + +void CaseProbeNamesAv1MainAtBothDepths() +{ + g_currentCase = "the probe can name AV1 Main at 8 AND at 10 bits"; + // AV1 seq_profile 0 carries 8 or 10 bits at 4:2:0 (AV1 A.2). One profile, + // two VkVideoProfileInfoKHR values, so both have to be nameable or the + // 10-bit half of a profile that encodes on hardware today cannot be asked + // about at all. + Check(VkEncProbeNamesProfileBitDepth( + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, 8), + "AV1 Main at 8 bits", "not named"); + Check(VkEncProbeNamesProfileBitDepth( + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, 10), + "AV1 Main at 10 bits", "not named"); + // The control that makes the two above measure the DEPTH term rather than + // a table that says yes to everything: a depth no arm carries. + Check(!VkEncProbeNamesProfileBitDepth( + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, 12), + "AV1 Main at 12 bits is NOT named", "unexpectedly named"); +} + +void CaseProbeRefusesDepthsItHasNoEvidenceFor() +{ + g_currentCase = "a profile is probed only at the depth it is probed at"; + struct Row { VkVideoCodecOperationFlagBitsKHR codec; uint32_t profile; + uint32_t depth; bool named; const char* why; }; + static const Row rows[] = { + // H.264 Baseline, Main and High are 8-bit (H.264 A.2). + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, 8, true, + "H.264 High at 8 bits" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H264_HIGH, 10, false, + "H.264 High at 10 bits" }, + // High 10 (profile_idc 110) is a row at 10 bits only. That is + // narrower than H.264 A.2.5 allows -- 110 admits 8-bit as well -- + // and is deliberate, the same choice the H.265 Main 10 rows below + // make: 8-bit input has High, so the 8-bit 110 pairing is not a row. + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, 110, 10, true, + "H.264 High 10 at 10 bits" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, 110, 8, false, + "H.264 High 10 at 8 bits" }, + // H.265 Main is 8-bit 4:2:0 (A.3.2); Main 10 is probed at 10 only, + // which is narrower than A.3.3 allows and is deliberate. + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, 8, true, + "H.265 Main at 8 bits" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, 10, false, + "H.265 Main at 10 bits" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, 10, true, + "H.265 Main 10 at 10 bits" }, + { VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, 8, false, + "H.265 Main 10 at 8 bits" }, + // The codec arm still disambiguates a repeated number: 1 is H.265 + // Main and is not an H.264 profile_idc. + { VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_H265_MAIN, 8, false, + "H.265 Main number on the H.264 arm" }, + }; + for (const Row& row : rows) { + const bool named = VkEncProbeNamesProfileBitDepth( + row.codec, row.profile, row.depth); + Check(named == row.named, + (std::string(row.named ? "named: " : "not named: ") + + row.why).c_str(), + named ? "named" : "not named"); + } +} + +void CaseSnapshotCarriesBothAv1Depths() +{ + g_currentCase = "the context snapshot holds a row per probed depth"; + // The rows are what a context build issues one driver query each for, so + // this is what decides whether the AV1 10-bit question ever reaches a + // driver at all. + uint32_t av1Rows = VkEncProbeSnapshotRowCount( + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR); + Check(av1Rows == 2, "AV1 has two probe rows", "got " + U32(av1Rows)); + + bool sawEight = false; + bool sawTen = false; + uint32_t firstProfile = 0; + uint32_t firstDepth = 0; + for (uint32_t slot = 0; slot < av1Rows; slot++) { + uint32_t profile = 0; + uint32_t depth = 0; + if (!VkEncProbeSnapshotRowAt( + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, slot, + &profile, &depth)) { + Check(false, "row readable", "slot " + U32(slot)); + continue; + } + Check(profile == VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, + "every AV1 row is seq_profile 0", "got " + U32(profile)); + if (slot == 0) { + firstProfile = profile; + firstDepth = depth; + } + sawEight = sawEight || (depth == 8); + sawTen = sawTen || (depth == 10); + } + Check(sawEight, "an 8-bit AV1 Main row", "absent"); + Check(sawTen, "a 10-bit AV1 Main row", "absent"); + // Order is load-bearing: the public lookup resolves a profile number to + // the FIRST row carrying it, so the 8-bit row has to be first or every + // published AV1 answer would silently become the 10-bit one. + Check((firstProfile == VK_VIDEO_ENCODER_PROFILE_AV1_MAIN) && + (firstDepth == 8), + "the 8-bit row is first, so published answers are unchanged", + "profile " + U32(firstProfile) + " depth " + U32(firstDepth)); + + // The other two codecs are pinned by count as well, so a row that + // appears or vanishes fails here rather than silently changing what the + // library probes a driver for. + const uint32_t h264Rows = VkEncProbeSnapshotRowCount( + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR); + const uint32_t h265Rows = VkEncProbeSnapshotRowCount( + VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR); + Check(h264Rows == 5, "H.264 has five rows", "got " + U32(h264Rows)); + Check(h265Rows == 3, "H.265 still has three rows", "got " + U32(h265Rows)); + // A codec with no rows answers zero rather than reading off the end. + const uint32_t noneRows = + VkEncProbeSnapshotRowCount(VK_VIDEO_CODEC_OPERATION_NONE_KHR); + Check(noneRows == 0, "an unprobed codec has no rows", + "got " + U32(noneRows)); + Check(!VkEncProbeSnapshotRowAt( + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, av1Rows, + nullptr, nullptr), + "one past the last AV1 row is refused", "accepted"); +} + +//============================================================================= +// 3. The field table itself +//============================================================================= + +void CaseFieldTableClassifiesEveryField() +{ + g_currentCase = "the field table classifies every field, once"; + // Every public config field appears in the table exactly once, with a + // disposition saying what happens to it. A disposition that disagrees + // with the binder is the same defect as a field that is accepted and + // ignored. + // + // The transfer-function declaration is the one VALIDATED field: it is read + // as a requirement and checked, and it reaches EncoderConfig nowhere, + // because this library applies no transfer function and so has nothing to + // forward. + bool foundOtf = false; + bool foundColorModel = false; + size_t nRemoved = 0; + for (const VkVideoEncoderConfigFieldInfo& f : kVkVideoEncoderConfigFields) { + if (std::strcmp(f.name, "inputTransferCharacteristics") == 0) { + foundOtf = true; + Check(f.disposition == VK_ENC_FIELD_VALIDATED, + "inputTransferCharacteristics is VALIDATED", + "disposition " + U32((uint32_t)f.disposition)); + } + // The other half of the pair a session is declared in. BOUND, not + // VALIDATED: the resolved model lands in EncoderConfig as + // input.colorSpace and the input geometry follows from it, which is + // what the binder cases above read back off the probe. A struct field + // with no row here is the one thing the shape assertions below cannot + // see, so the field the whole routing turns on is named explicitly. + if (std::strcmp(f.name, "inputColorModel") == 0) { + foundColorModel = true; + Check(f.disposition == VK_ENC_FIELD_BOUND, + "inputColorModel is BOUND", + "disposition " + U32((uint32_t)f.disposition)); + } + // The preprocess conversion is not a caller-settable field any more. + // Stated as a count over the whole table rather than as an absence, so + // a table that stopped being iterated would not read as a pass. + if (std::strcmp(f.name, "enablePreprocessFilter") == 0) { + nRemoved++; + } + } + Check(foundOtf, "inputTransferCharacteristics appears in the field table", + "absent"); + Check(foundColorModel, "inputColorModel appears in the field table", + "absent"); + Check(nRemoved == 0, + "no field asks the caller to decide the preprocess conversion", + U32((uint32_t)nRemoved) + " of " + + U32((uint32_t)kVkEncCfgFieldCount) + " fields still do"); + + // And the shape assertion the table exists for: exactly one entry per + // field, every offset inside the struct. + Check((size_t)kVkEncCfgFieldCount == + sizeof(kVkVideoEncoderConfigFields) / + sizeof(kVkVideoEncoderConfigFields[0]), + "table length matches the field enum", "mismatch"); + for (const VkVideoEncoderConfigFieldInfo& f : kVkVideoEncoderConfigFields) { + Check(f.offset < sizeof(VkVideoEncoderConfig), + "field offset is inside VkVideoEncoderConfig", + std::string(f.name) + " at " + U32((uint32_t)f.offset)); + } + + // "Exactly once" asserted rather than implied. The length check above + // pairs the table against an enum expanded from the SAME macro, so the two + // move together and can never disagree about a repeat; only comparing the + // names can see one. Tallied and asserted once rather than per pair, so + // the check count stays a count of facts and not of comparisons. + size_t nDuplicates = 0; + size_t nMissingNotes = 0; + const char* firstDuplicate = ""; + for (size_t i = 0; i < (size_t)kVkEncCfgFieldCount; ++i) { + if (kVkVideoEncoderConfigFields[i].note == nullptr) { + nMissingNotes++; + } + for (size_t j = i + 1; j < (size_t)kVkEncCfgFieldCount; ++j) { + if (std::strcmp(kVkVideoEncoderConfigFields[i].name, + kVkVideoEncoderConfigFields[j].name) == 0) { + if (nDuplicates == 0) { + firstDuplicate = kVkVideoEncoderConfigFields[i].name; + } + nDuplicates++; + } + } + } + Check(nMissingNotes == 0, "every row says what happens to its field", + U32((uint32_t)nMissingNotes) + " rows carry no note"); + Check(nDuplicates == 0, "no field is listed twice", + std::string("first repeat: ") + firstDuplicate); +} + + +//============================================================================= +// 3b. The direction the checks above cannot look: a struct field with NO row. +// +// Everything in the case above walks table -> struct. Each check reads a row +// and asks something about it, so a field with NO row is never read and never +// asked about: the table can be one row short and every check above still +// passes. A test that walks the same direction cannot see it either, wherever +// it lives, which is why the direction below is the one that matters. +// +// This case walks the other way -- by arithmetic, since C++ offers no +// reflection to enumerate the struct with. Each row now carries the size and +// the alignment of the field it names, so the rows sorted by offset can be +// laid end to end and measured against the struct they claim to describe: +// +// * no row may start inside the row before it; +// * a gap before a row is alignment padding, and is legal only while it is +// STRICTLY narrower than that row's alignment -- a gap the successor's +// alignment did not force is a field whose row was never written; +// * the last row must close the struct; +// * the gaps must total the pinned padding budget. +// +// Between them these fail for the removal of ANY row in the table, including +// the last. What none of them can see is a field added into padding that +// already exists: such a field moves neither the size nor any offset, so no +// arithmetic over sizes and offsets has anything to count. That case is +// stated as uncovered in the header rather than papered over here. +//============================================================================= + +void CaseFieldTableTilesTheStruct() +{ + g_currentCase = "the field table tiles VkVideoEncoderConfig"; + + // Offset order, not declaration order. The table is hand-maintained and + // the argument below is about the LAYOUT, so the rows are sorted instead + // of being assumed to already be in layout order -- assuming it would put + // the assumption under test rather than the layout. Insertion sort over + // pointers: the table is small, and this keeps the case free of any + // dependency it would otherwise have to bring in. + const VkVideoEncoderConfigFieldInfo* byOffset[kVkEncCfgFieldCount]; + for (size_t i = 0; i < (size_t)kVkEncCfgFieldCount; ++i) { + byOffset[i] = &kVkVideoEncoderConfigFields[i]; + } + for (size_t i = 1; i < (size_t)kVkEncCfgFieldCount; ++i) { + const VkVideoEncoderConfigFieldInfo* key = byOffset[i]; + size_t j = i; + while (j > 0 && byOffset[j - 1]->offset > key->offset) { + byOffset[j] = byOffset[j - 1]; + --j; + } + byOffset[j] = key; + } + + size_t cursor = 0; // first byte no row has claimed yet + size_t gapTotal = 0; // bytes no row claims at all + size_t covered = 0; // bytes the rows do claim + for (size_t i = 0; i < (size_t)kVkEncCfgFieldCount; ++i) { + const VkVideoEncoderConfigFieldInfo& f = *byOffset[i]; + covered += f.size; + + Check(f.offset >= cursor, + "no row starts inside the row before it", + std::string(f.name) + " starts at " + U32((uint32_t)f.offset) + + ", the row before it ends at " + U32((uint32_t)cursor)); + if (f.offset < cursor) { + cursor = f.offset + f.size; + continue; + } + + // THE MISSING-ROW RULE. Bytes between two rows are padding, and + // padding exists for exactly one reason: the member that follows has + // to begin on its own alignment. A gap AT LEAST as wide as that + // alignment is therefore not padding -- the compiler would never have + // inserted it -- and what is sitting in it is a field whose row was + // never written. + const size_t gap = f.offset - cursor; + gapTotal += gap; + Check(gap < f.align, + "the gap before a row is no wider than alignment forces", + U32((uint32_t)gap) + " unclaimed bytes at offset " + + U32((uint32_t)cursor) + " precede " + f.name + + ", which needs only " + U32((uint32_t)f.align) + + "-byte alignment: a field with no row lives there"); + cursor = f.offset + f.size; + } + + // THE END OF THE STRUCT, asserted on its own because it is the one place + // the rule above has nothing to measure against: the last row has no + // successor to take an alignment from, so a missing FINAL row would read + // as trailing padding and pass. It is caught here instead -- and only + // because this struct happens to end on its own alignment and so has no + // trailing padding for a row to hide in. Should it ever acquire some, THIS + // assertion is what says so, and the end of the table stops being covered + // until it is pinned another way. + Check(cursor == sizeof(VkVideoEncoderConfig), + "the last row closes VkVideoEncoderConfig", + "the rows end at " + U32((uint32_t)cursor) + ", sizeof is " + + U32((uint32_t)sizeof(VkVideoEncoderConfig))); + + // THE BUDGET. The per-gap rule passes for a row removed from in front of a + // widely aligned successor, because the bytes it freed fit inside slack + // that successor already had; silenceStdio and matrixCoefficients are both + // of that shape. Pinning the TOTAL is what catches those: the freed bytes + // have to surface somewhere, and this is where. + Check(gapTotal == kVkEncCfgPaddingBytes, + "the gaps total the pinned padding budget", + U32((uint32_t)gapTotal) + " bytes of padding, pinned at " + + U32((uint32_t)kVkEncCfgPaddingBytes)); + + // The same fact reached from the other side, so the two cannot drift: + // every byte of the struct is either claimed by a row or is pinned + // padding, and there is no third kind. + Check(covered + kVkEncCfgPaddingBytes == sizeof(VkVideoEncoderConfig), + "the rows and the pinned padding account for every byte", + "rows claim " + U32((uint32_t)covered) + " bytes + " + + U32((uint32_t)kVkEncCfgPaddingBytes) + " padding, sizeof is " + + U32((uint32_t)sizeof(VkVideoEncoderConfig))); +} + + +//============================================================================= +// 3. The colour description: what the binder writes, and what the RGBA arm's +// matrix contract does with a code point it cannot produce. +// +// Every case here is device-free for the same reason the ones above are: the +// binder, EncoderConfig::ResolveRgbToYcbcrMatrix and +// EncoderConfigAV1::InitSequenceHeader read only their arguments. What they +// CANNOT prove is that the driver then writes those values into the VUI or +// the sequence header; that is the GPU suite's job +// (test/encoder-ext-format-encode, which reads them back with ffprobe). +//============================================================================= + +// ISO/IEC 23091-4 code points used below, named so the expectations read as +// claims about colour rather than about integers. +enum { + kCpUnspecified = 2, + kCpBt709 = 1, + kCpBt2020 = 9, + kTcBt709 = 1, + kTcPq = 16, // SMPTE ST 2084 + kTcHlg = 18, // ARIB STD-B67 + kMcBt709 = 1, + kMcBt2020Ncl = 9, + kMcSmpte240M = 7, +}; + +void CasePartialColourSupplyDoesNotFabricateTheRest() +{ + g_currentCase = "declaring ONLY a transfer function declares only that"; + // THE HDR PATH, EXACTLY. A caller that knows its content is PQ and + // nothing else sets transferCharacteristics = 16. The binder's gate used + // to be "any of the four is set" and its body copied ALL FOUR, so this + // config also emitted colour_primaries 0 (Reserved) and + // matrix_coefficients 0 (Identity/GBR -- "the samples are RGB"). Two + // fabricated declarations from one honest one, both wrong, both on the + // HDR path. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + cfg.transferCharacteristics = kTcPq; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted a transfer-only declaration", + "VkResult " + U32((uint32_t)r)); + Check(probe.transferCharacteristics == kTcPq, "PQ survived", + "got " + U32(probe.transferCharacteristics)); + Check(probe.colourPrimaries == kCpUnspecified, + "unsupplied primaries are Unspecified, NOT Reserved(0)", + "got " + U32(probe.colourPrimaries)); + Check(probe.matrixCoefficients == kCpUnspecified, + "unsupplied matrix is Unspecified, NOT Identity/GBR(0)", + "got " + U32(probe.matrixCoefficients)); + Check(probe.colorDescriptionPresent == 1, "the description is present", + "got " + U32(probe.colorDescriptionPresent)); + Check(probe.videoSignalTypePresent == 1, "video_signal_type is present", + "got " + U32(probe.videoSignalTypePresent)); + Check(probe.videoFullRangeFlag == 0, + "an undeclared range stays studio", "got " + U32(probe.videoFullRangeFlag)); +} + +void CaseFullRangeOnlyDeclaresRangeAndNoColour() +{ + g_currentCase = "declaring ONLY full range raises no colour description"; + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.videoFullRange = VK_TRUE; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted a range-only declaration", + "VkResult " + U32((uint32_t)r)); + Check(probe.videoFullRangeFlag == 1, "full range bound", + "got " + U32(probe.videoFullRangeFlag)); + Check(probe.videoSignalTypePresent == 1, + "video_signal_type carries it", "got " + U32(probe.videoSignalTypePresent)); + Check(probe.colorDescriptionPresent == 0, + "no colour description is invented for it", + "got " + U32(probe.colorDescriptionPresent)); +} + +void CaseAv1SignalsRangeWithoutAColourDescription() +{ + g_currentCase = "AV1 signals full range with no colour description"; + // AV1's color_config carries color_range, BitDepth and subsampling in + // ADDITION to the colour description, and none of those is conditioned on + // color_description_present_flag in the AV1 syntax. The whole struct used + // to sit behind that one flag, so this configuration -- which signals + // range fine on H.264 and H.265, where the range lives under a DIFFERENT + // presence flag -- reached AV1 as nothing at all. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + cfg.videoFullRange = VK_TRUE; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted the AV1 config", + "VkResult " + U32((uint32_t)r)); + Check(probe.av1ColorConfigPresent == 1, + "the sequence header carries a color_config at all", + "got " + U32(probe.av1ColorConfigPresent)); + Check(probe.av1ColorRange == 1, "color_range is full", + "got " + U32(probe.av1ColorRange)); + Check(probe.av1ColorDescriptionPresent == 0, + "and no colour description is invented", + "got " + U32(probe.av1ColorDescriptionPresent)); + // The three idc fields must be UNSPECIFIED, not 0: in AV1's enums 0 is + // BT.709 / BT.709 / IDENTITY, and IDENTITY additionally asserts RGB. + Check(probe.av1ColorPrimaries == kCpUnspecified, + "primaries default to Unspecified, not BT.709(0)", + "got " + U32(probe.av1ColorPrimaries)); + Check(probe.av1MatrixCoefficients == kCpUnspecified, + "matrix defaults to Unspecified, not Identity(0)", + "got " + U32(probe.av1MatrixCoefficients)); + Check(probe.av1BitDepth == 8, "BitDepth still describes the session", + "got " + U32(probe.av1BitDepth)); +} + +void CaseHdr10CodePointsReachBothArms() +{ + g_currentCase = "BT.2020 + PQ + BT.2020-ncl survive to H.265 and AV1"; + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16; + cfg.colourPrimaries = kCpBt2020; + cfg.transferCharacteristics = kTcPq; + cfg.matrixCoefficients = kMcBt2020Ncl; + + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + VkEncBoundConfigProbe h265{}; + Check(VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &h265) == + VK_SUCCESS, + "H.265 accepted HDR10 code points", "init failed"); + Check(h265.colourPrimaries == kCpBt2020 && + h265.transferCharacteristics == kTcPq && + h265.matrixCoefficients == kMcBt2020Ncl && + h265.colorDescriptionPresent == 1, + "H.265 VUI carries 9 / 16 / 9", + U32(h265.colourPrimaries) + " / " + U32(h265.transferCharacteristics) + + " / " + U32(h265.matrixCoefficients) + " present=" + + U32(h265.colorDescriptionPresent)); + + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + VkEncBoundConfigProbe av1{}; + Check(VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, &av1) == + VK_SUCCESS, + "AV1 accepted HDR10 code points", "init failed"); + Check(av1.av1ColorPrimaries == kCpBt2020 && + av1.av1TransferCharacteristics == kTcPq && + av1.av1MatrixCoefficients == kMcBt2020Ncl && + av1.av1ColorDescriptionPresent == 1, + "AV1 color_config carries 9 / 16 / 9", + U32(av1.av1ColorPrimaries) + " / " + + U32(av1.av1TransferCharacteristics) + " / " + + U32(av1.av1MatrixCoefficients) + " present=" + + U32(av1.av1ColorDescriptionPresent)); + Check(av1.av1BitDepth == 10, "at 10 bits", "got " + U32(av1.av1BitDepth)); + + // HLG is the other half of the HDR pair and takes a different code point + // through the same field; it is here so a transfer-specific clamp could + // not pass by hardcoding 16. + cfg.transferCharacteristics = kTcHlg; + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + VkEncBoundConfigProbe hlg{}; + Check(VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &hlg) == + VK_SUCCESS, + "H.265 accepted HLG", "init failed"); + Check(hlg.transferCharacteristics == kTcHlg, "HLG survived", + "got " + U32(hlg.transferCharacteristics)); +} + +void CaseUnexpressibleMatrixIsRefusedOnAnRgbaSession() +{ + g_currentCase = "an RGBA session refuses a matrix the filter cannot produce"; + if (!kFilterCompiledIn) { + return; // an RGBA session cannot be built at all in this build. + } + // SMPTE 240M is a real, different matrix. The filter has no encoding for + // it, and the old behaviour was to log a line, convert as BT.709 and + // leave matrix_coefficients saying 7 -- BT.709 pixels under a 240M label, + // on the SUCCESS path. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + cfg.colourPrimaries = kCpBt709; + cfg.transferCharacteristics = kTcBt709; + cfg.matrixCoefficients = kMcSmpte240M; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + "matrix 7 on an RGBA session is refused, not converted as BT.709", + "VkResult " + U32((uint32_t)r)); + + // Identity/GBR is refused for a different reason -- it asserts the + // samples ARE RGB -- and it is reachable here only because + // matrixCoefficients 0 means "not supplied", so it is driven through the + // config the binder produces rather than through the public field. + cfg.matrixCoefficients = 11; // reserved, outside every expressible set + const VkResult r2 = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r2 == VK_ERROR_INITIALIZATION_FAILED, + "a reserved matrix code point is refused too", + "VkResult " + U32((uint32_t)r2)); +} + +void CaseUnspecifiedMatrixIsDerivedFromPrimaries() +{ + g_currentCase = "an RGBA session signals the matrix it actually applied"; + if (!kFilterCompiledIn) { + return; + } + // The caller declined to NAME a matrix, so there is no request to ignore. + // The filter must pick one, and after it does the samples really ARE the + // matrix it picked -- so the label is written to say so. Leaving it at 2 + // would leave a decoder guessing at something the encoder knows. + // + // WHAT CHANGED, and it is the whole of CC-1's disposition for code point + // 2: this arm used to write 1 unconditionally. That threw away the one + // piece of information the caller DID supply -- the primaries -- and it + // did so on precisely the caller that has them and nothing else. The + // BT.2020 sub-case below is the same configuration an HDR caller + // produces, and it must not emit BT.709 chroma under BT.2020 primaries. + struct Row { uint8_t primaries; uint8_t expectMatrix; const char* why; }; + static const Row rows[] = { + { kCpBt709, kMcBt709, "BT.709 primaries -> BT.709 matrix" }, + { kCpBt2020, kMcBt2020Ncl, "BT.2020 primaries -> BT.2020 NCL matrix" }, + { 6, 6, "SMPTE 170M primaries -> BT.601 matrix" }, + { 5, 6, "BT.470BG primaries -> BT.601 matrix" }, + { kCpUnspecified, kMcBt709, + "Unspecified primaries -> BT.709, the we-were-not-told default" }, + { 12, kMcBt709, "Display-P3 primaries -> BT.709" }, + }; + for (const Row& row : rows) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + cfg.colourPrimaries = row.primaries; + cfg.transferCharacteristics = kTcBt709; + cfg.matrixCoefficients = kCpUnspecified; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, (std::string("accepted: ") + row.why).c_str(), + "VkResult " + U32((uint32_t)r)); + Check(probe.matrixCoefficients == row.expectMatrix, + (std::string("signalled: ") + row.why).c_str(), + "got " + U32(probe.matrixCoefficients) + ", wanted " + + U32(row.expectMatrix)); + } +} + +//----------------------------------------------------------------------------- +// CC-1's table, walked. THE POINT IS NOT COVERAGE, IT IS DRIFT DETECTION. +// +// The same table is implemented on the Chromium side +// (media/gpu/vulkan/vulkan_video_encode_accelerator.cc, and walked by +// VulkanVeaMatrixTableTest). Code cannot be shared across the two +// repositories, so a divergence can only be caught by both sides asserting +// the same rows. If you change a disposition here and the review does not +// also carry a matching change to VulkanVeaMatrixTableTest, one of the two +// is wrong. +// +// READ THIS BEFORE ADDING A ROW: the input here is the PUBLIC ext field, on +// which 0 means "not supplied". A 0 written below is rewritten to 2 by the +// binder before ResolveRgbToYcbcrMatrix sees it, so row 0 measures the +// UNNAMED disposition and NOT the `case 0` refusal arm -- that arm is +// unreachable through every producer in this tree and is asserted by reading +// it, not by running it. +//----------------------------------------------------------------------------- +void CaseMatrixDispositionTable() +{ + g_currentCase = "CC-1 matrix disposition table (FILTER lane)"; + if (!kFilterCompiledIn) { + return; + } + enum Disp { HONOUR, DERIVE, REFUSE }; + struct Row { uint8_t code; Disp disp; uint8_t signalled; const char* name; }; + // Primaries are pinned to BT.709 throughout, so DERIVE rows expect 1. + static const Row table[] = { + { 0, DERIVE, 1, "Identity/GBR -- \"not supplied\" on this surface" }, + { 1, HONOUR, 1, "BT.709" }, + { 2, DERIVE, 1, "Unspecified" }, + { 3, REFUSE, 0, "reserved" }, + { 4, REFUSE, 0, "FCC" }, + { 5, HONOUR, 5, "BT.470BG" }, + { 6, HONOUR, 6, "SMPTE 170M" }, + { 7, REFUSE, 0, "SMPTE 240M" }, + { 8, REFUSE, 0, "YCoCg" }, + { 9, HONOUR, 9, "BT.2020 NCL" }, + { 10, HONOUR, 10, "BT.2020 CL -- accepted, approximated as NCL" }, + { 11, REFUSE, 0, "SMPTE 2085" }, + { 12, REFUSE, 0, "chroma-derived NCL" }, + { 13, REFUSE, 0, "chroma-derived CL" }, + { 14, REFUSE, 0, "ICtCp" }, + { 255, REFUSE, 0, "unknown / INVALID" }, + }; + for (const Row& row : table) { + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + cfg.colourPrimaries = kCpBt709; + cfg.transferCharacteristics = kTcBt709; + cfg.matrixCoefficients = row.code; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + const std::string label = + "matrix " + U32(row.code) + " (" + row.name + ")"; + if (row.disp == REFUSE) { + Check(r == VK_ERROR_INITIALIZATION_FAILED, + (label + " is REFUSED").c_str(), + "VkResult " + U32((uint32_t)r)); + } else { + Check(r == VK_SUCCESS, (label + " is ACCEPTED").c_str(), + "VkResult " + U32((uint32_t)r)); + if (r == VK_SUCCESS) { + Check(probe.matrixCoefficients == row.signalled, + (label + " signals " + U32(row.signalled)).c_str(), + "got " + U32(probe.matrixCoefficients)); + } + } + } + + // THE DIRECT LANE IS NOT SUBJECT TO THE TABLE, asserted here rather than + // left to CaseYcbcrSessionKeepsAMatrixTheFilterCannotProduce alone, so + // the scope travels with the table it scopes. + for (uint8_t code : { (uint8_t)4, (uint8_t)7, (uint8_t)8, (uint8_t)11 }) { + VkVideoEncoderConfig direct = BaseConfig(); // NV12, DIRECT + direct.colourPrimaries = kCpBt709; + direct.transferCharacteristics = kTcBt709; + direct.matrixCoefficients = code; + VkEncBoundConfigProbe dprobe{}; + const VkResult dr = VkEncBuildAndProbeConfig( + direct, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &dprobe); + Check(dr == VK_SUCCESS, + ("DIRECT lane accepts matrix " + U32(code) + + " -- the VUI describes the caller's own samples").c_str(), + "VkResult " + U32((uint32_t)dr)); + Check(dprobe.matrixCoefficients == code, + ("DIRECT lane carries matrix " + U32(code) + + " unaltered").c_str(), + "got " + U32(dprobe.matrixCoefficients)); + } +} + +void CaseYcbcrSessionKeepsAMatrixTheFilterCannotProduce() +{ + g_currentCase = "a YCbCr->YCbCr session is not subject to the RGB contract"; + if (!kFilterCompiledIn) { + return; + } + // THE SCOPING, AS A MEASUREMENT. A 3-plane I420 session also builds the + // preprocess filter, but that filter COPIES chroma -- it applies no + // matrix at all. Refusing SMPTE 240M there would reject a configuration + // that is entirely correct, so the refusal is scoped to the arm that + // actually converts. Nothing else in the suite would notice if that + // scoping were dropped. + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.inputFormat = VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM; + cfg.colourPrimaries = kCpBt709; + cfg.transferCharacteristics = kTcBt709; + cfg.matrixCoefficients = kMcSmpte240M; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "a 3-plane session accepts matrix 7", + "VkResult " + U32((uint32_t)r)); + Check(probe.matrixCoefficients == kMcSmpte240M, + "and carries it through unaltered", + "got " + U32(probe.matrixCoefficients)); +} + +void CaseChromaSitingIsSignalledOnlyWhereItIsKnown() +{ + g_currentCase = "chroma siting is signalled for the converting arm only"; + VkVideoEncoderConfig direct = BaseConfig(); + VkEncBoundConfigProbe dprobe{}; + Check(VkEncBuildAndProbeConfig( + direct, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &dprobe) == + VK_SUCCESS, + "NV12 session builds", "init failed"); + Check(dprobe.chromaLocInfoPresent == 0, + "a DIRECT session signals no siting -- the content's siting is the " + "caller's and this library never learns it", + "got " + U32(dprobe.chromaLocInfoPresent)); + + if (!kFilterCompiledIn) { + return; + } + VkVideoEncoderConfig rgba = BaseConfig(); + rgba.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + rgba.colourPrimaries = kCpBt709; + rgba.transferCharacteristics = kTcBt709; + rgba.matrixCoefficients = kMcBt709; + VkEncBoundConfigProbe rprobe{}; + Check(VkEncBuildAndProbeConfig( + rgba, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &rprobe) == + VK_SUCCESS, + "RGBA session builds", "init failed"); + Check(rprobe.chromaLocInfoPresent == 1, + "a converting session signals its siting", + "got " + U32(rprobe.chromaLocInfoPresent)); + Check(rprobe.chromaSampleLocType == 1, + "and the type is 1 (centre) -- the 2x2 box average the filter runs, " + "NOT the 0 (left) a raised flag used to advertise by default", + "got " + U32(rprobe.chromaSampleLocType)); + + // AND THE SAME ASSERTION ON THE VUI THE ARM BUILDS, on BOTH H.26x + // codecs. Reading the config alone is what let a real defect through: + // EncoderConfigH265::InitVuiParameters wrote chroma_sample_loc_type from + // the config and then re-zeroed it unconditionally two hundred lines + // later, so an H.265 session raised chroma_loc_info_present_flag and + // advertised type 0 (left) for centre-sited samples -- and no encode row + // could see it either, because every H.265 row in the matrix takes a path + // that signals no siting at all. + for (int arm = 0; arm < 2; arm++) { + const VkVideoCodecOperationFlagBitsKHR codec = + (arm == 0) ? VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR + : VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + const char* codecName = (arm == 0) ? "H.264" : "H.265"; + + VkVideoEncoderConfig direct2 = BaseConfig(); + direct2.codec = codec; + VkEncBoundConfigProbe dv{}; + Check(VkEncBuildAndProbeConfig(direct2, codec, &dv) == VK_SUCCESS, + "DIRECT session builds", codecName); + const std::string wNoSiting = + std::string(codecName) + " DIRECT VUI signals no siting"; + Check(dv.vuiChromaLocInfoPresent == 0, wNoSiting.c_str(), + "got " + U32(dv.vuiChromaLocInfoPresent)); + + VkVideoEncoderConfig rgba2 = BaseConfig(); + rgba2.codec = codec; + rgba2.inputFormat = VK_FORMAT_R8G8B8A8_UNORM; + rgba2.colourPrimaries = kCpBt709; + rgba2.transferCharacteristics = kTcBt709; + rgba2.matrixCoefficients = kMcBt709; + VkEncBoundConfigProbe rv{}; + Check(VkEncBuildAndProbeConfig(rgba2, codec, &rv) == VK_SUCCESS, + "converting session builds", codecName); + const std::string wRaised = std::string(codecName) + + " converting VUI raises chroma_loc_info_present_flag"; + Check(rv.vuiChromaLocInfoPresent == 1, wRaised.c_str(), + "got " + U32(rv.vuiChromaLocInfoPresent)); + const std::string wType = std::string(codecName) + + " converting VUI carries type 1 (centre) in BOTH field positions" + " -- not the 0 a raised flag used to advertise"; + Check(rv.vuiChromaSampleLocTypeTop == 1 && + rv.vuiChromaSampleLocTypeBottom == 1, wType.c_str(), + U32(rv.vuiChromaSampleLocTypeTop) + " / " + + U32(rv.vuiChromaSampleLocTypeBottom)); + } + + // AV1 cannot express centre siting: chroma_sample_position offers only + // UNKNOWN, VERTICAL and COLOCATED. UNKNOWN is therefore the correct + // answer and not a gap -- signalling VERTICAL to look decisive would + // assert a position half a chroma sample away from the one written. + rgba.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + VkEncBoundConfigProbe aprobe{}; + Check(VkEncBuildAndProbeConfig( + rgba, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, &aprobe) == + VK_SUCCESS, + "AV1 RGBA session builds", "init failed"); + Check(aprobe.av1ChromaSamplePosition == + (uint32_t)STD_VIDEO_AV1_CHROMA_SAMPLE_POSITION_UNKNOWN, + "AV1 signals chroma_sample_position UNKNOWN", + "got " + U32(aprobe.av1ChromaSamplePosition)); +} + + +//============================================================================= +// 4. HDR10 static metadata: the config surface, and the BYTES. +// +// THE GOLDEN VECTORS ARE A MEASUREMENT, not a transcription of a spec. Both +// payloads below were built by these same functions, spliced into real +// streams and read back by ffprobe 6.1.1 -- the H.265 SEI inside an encode on +// an RTX A4000, the AV1 OBUs inside a working AV1 stream because no device +// available to this work has AV1 encode. Every field of both, and the AV1 +// fixed-point scales and primary permutation in particular, is what that +// read-back reported. A byte array with no such provenance would only pin the +// author's opinion. +//============================================================================= + +// BT.2020 mastering display at 1000 cd/m^2 peak, 0.0001 cd/m^2 floor, in +// SMPTE ST 2086 units (chromaticity x 50000, luminance x 10000) and ST 2086 +// order (green, blue, red). +// green (0.170, 0.797) blue (0.131, 0.046) red (0.708, 0.292) +// white D65 (0.3127, 0.3290) +VkVideoEncoderHdrMetadataInfo Hdr10Bt2020() +{ + VkVideoEncoderHdrMetadataInfo hdr{}; + hdr.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO; + hdr.masteringDisplayPresent = VK_TRUE; + hdr.displayPrimaryX[0] = 8500; hdr.displayPrimaryY[0] = 39850; // G + hdr.displayPrimaryX[1] = 6550; hdr.displayPrimaryY[1] = 2300; // B + hdr.displayPrimaryX[2] = 35400; hdr.displayPrimaryY[2] = 14600; // R + hdr.whitePointX = 15635; + hdr.whitePointY = 16450; + hdr.maxDisplayMasteringLuminance = 10000000; // 1000.0000 cd/m^2 + hdr.minDisplayMasteringLuminance = 1; // 0.0001 cd/m^2 + hdr.contentLightLevelPresent = VK_TRUE; + hdr.maxContentLightLevel = 1000; + hdr.maxFrameAverageLightLevel = 400; + return hdr; +} + +std::string Hex(const uint8_t* p, size_t n) +{ + std::string s; + char b[4]; + for (size_t i = 0; i < n; i++) { + std::snprintf(b, sizeof(b), "%02x", p[i]); + if (i != 0) s += " "; + s += b; + } + return s; +} + +void CheckBytes(const char* what, const uint8_t* got, size_t gotLen, + const uint8_t* want, size_t wantLen) +{ + if ((gotLen == wantLen) && (std::memcmp(got, want, wantLen) == 0)) { + Check(true, what, ""); + return; + } + Check(false, what, + "got [" + Hex(got, gotLen) + "] want [" + Hex(want, wantLen) + "]"); +} + +void CaseH265HdrSeiBytes() +{ + g_currentCase = "the H.265 HDR10 prefix SEI is byte-exact"; + const VkVideoEncoderHdrMetadataInfo hdr = Hdr10Bt2020(); + uint8_t got[128]; + const uint32_t n = VkEncBuildHdrMetadataPayload( + &hdr, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, got, sizeof(got)); + + // 00 00 00 01 4-byte start code + // 4e 01 nal_unit_type 39 (PREFIX_SEI_NUT), tid 0 + // 89 18 payloadType 137, payloadSize 24 + // 2134 9baa G x=8500 y=39850 + // 1996 08fc B x=6550 y=2300 + // 8a48 3908 R x=35400 y=14600 + // 3d13 4042 white point 15635 / 16450 + // 0098 9680 max luminance 10000000 (1000 cd/m^2) + // 0000 0003 01 min luminance 1 -- WITH THE EMULATION + // PREVENTION BYTE. Three zero bytes precede a + // 0x01, so 00 00 00 01 would open a NAL in + // the middle of this one. This is the live + // case, not a formality: a floor of 0.0001 + // cd/m^2 is the commonest HDR10 value there + // is. + // 90 04 payloadType 144, payloadSize 4 + // 03e8 0190 MaxCLL 1000, MaxFALL 400 + // 80 rbsp_trailing_bits + static const uint8_t kWant[] = { + 0x00, 0x00, 0x00, 0x01, + 0x4e, 0x01, + 0x89, 0x18, + 0x21, 0x34, 0x9b, 0xaa, + 0x19, 0x96, 0x08, 0xfc, + 0x8a, 0x48, 0x39, 0x08, + 0x3d, 0x13, 0x40, 0x42, + 0x00, 0x98, 0x96, 0x80, + 0x00, 0x00, 0x03, 0x00, 0x01, + 0x90, 0x04, + 0x03, 0xe8, 0x01, 0x90, + 0x80, + }; + CheckBytes("mastering display + content light SEI", got, n, + kWant, sizeof(kWant)); +} + +void CaseAv1HdrMetadataObuBytes() +{ + g_currentCase = "the AV1 HDR10 metadata OBUs are byte-exact"; + const VkVideoEncoderHdrMetadataInfo hdr = Hdr10Bt2020(); + uint8_t got[128]; + const uint32_t n = VkEncBuildHdrMetadataPayload( + &hdr, VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, got, sizeof(got)); + + // 2a 1a OBU_METADATA, has_size_field, size 26 + // 02 metadata_type 2 = HDR_MDCV + // b53f 4ac1 RED 46399 / 19137 (0.708, 0.292) + // 2b85 cc08 GREEN 11141 / 52232 (0.170, 0.797) + // 2189 0bc7 BLUE 8585 / 3015 (0.131, 0.046) + // 500d 5439 white 20493 / 21561 + // 0003 e800 luminance_max 256000, 24.8 -> 1000.0 + // 0000 0002 luminance_min 2, 18.14 -> 0.000122 + // 80 trailing_bits + // 2a 06 OBU_METADATA, size 6 + // 01 03e8 0190 80 metadata_type 1 = HDR_CLL, 1000, 400 + // + // THE PRIMARY ORDER IS THE POINT OF THIS VECTOR. It is R,G,B here and + // G,B,R in the H.265 vector above, from the SAME input array. ffprobe + // reported BT.2020 for this spelling and reported BT.2020's GREEN as its + // red for the other one. + // + // The luminance_min round trip is lossy and that is a property of the + // format, not a bug: 0.0001 cd/m^2 is 1.6384 in 18.14, which rounds to 2, + // i.e. 0.000122. H.265 carries the same value exactly. + static const uint8_t kWant[] = { + 0x2a, 0x1a, + 0x02, + 0xb5, 0x3f, 0x4a, 0xc1, + 0x2b, 0x85, 0xcc, 0x08, + 0x21, 0x89, 0x0b, 0xc7, + 0x50, 0x0d, 0x54, 0x39, + 0x00, 0x03, 0xe8, 0x00, + 0x00, 0x00, 0x00, 0x02, + 0x80, + 0x2a, 0x06, + 0x01, 0x03, 0xe8, 0x01, 0x90, 0x80, + }; + CheckBytes("MDCV + CLL metadata OBUs", got, n, kWant, sizeof(kWant)); +} + +void CaseEachHdrPayloadIsIndependent() +{ + g_currentCase = "each HDR payload is emitted on its own"; + // A mastering display of all zeros is a CLAIM -- it says the display is + // black -- so a caller that knows only MaxCLL must not get one. + VkVideoEncoderHdrMetadataInfo hdr{}; + hdr.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO; + hdr.contentLightLevelPresent = VK_TRUE; + hdr.maxContentLightLevel = 600; + hdr.maxFrameAverageLightLevel = 120; + uint8_t got[128]; + const uint32_t n = VkEncBuildHdrMetadataPayload( + &hdr, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, got, sizeof(got)); + static const uint8_t kWant[] = { + 0x00, 0x00, 0x00, 0x01, 0x4e, 0x01, + 0x90, 0x04, 0x02, 0x58, 0x00, 0x78, + 0x80, + }; + CheckBytes("content light only -- no mastering display message", got, n, + kWant, sizeof(kWant)); + + // And nothing at all when nothing was declared. + VkVideoEncoderHdrMetadataInfo none{}; + none.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO; + const uint32_t z = VkEncBuildHdrMetadataPayload( + &none, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, got, sizeof(got)); + Check(z == 0, "an empty declaration emits no SEI at all", + "got " + U32(z) + " bytes"); +} + +void CaseHdrPayloadRefusesToTruncate() +{ + g_currentCase = "a payload that does not fit is refused, not clipped"; + // The caller cannot see a dropped SEI: byte counts, completion edges and + // every counter are identical with and without it. So a short buffer must + // produce 0 rather than a prefix. + const VkVideoEncoderHdrMetadataInfo hdr = Hdr10Bt2020(); + uint8_t got[16]; + std::memset(got, 0xAA, sizeof(got)); + const uint32_t n = VkEncBuildHdrMetadataPayload( + &hdr, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, got, sizeof(got)); + Check(n == 0, "a 16-byte buffer yields 0, not a truncated NAL", + "got " + U32(n)); + Check(got[0] == 0xAA, "and nothing was written into it", + "first byte " + U32(got[0])); +} + +void CaseHdrMetadataBindsThroughThePnextChain() +{ + g_currentCase = "the chained HDR struct reaches the encoder config"; + VkVideoEncoderHdrMetadataInfo hdr = Hdr10Bt2020(); + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + cfg.pNext = &hdr; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &probe); + Check(r == VK_SUCCESS, "binder accepted the chained HDR struct", + "VkResult " + U32((uint32_t)r)); + Check(probe.hdrMasteringPresent == 1 && probe.hdrContentLightPresent == 1, + "both payloads bound", + U32(probe.hdrMasteringPresent) + " / " + + U32(probe.hdrContentLightPresent)); + Check(probe.hdrMaxDisplayMasteringLuminance == 10000000 && + probe.hdrMinDisplayMasteringLuminance == 1, + "luminance bounds bound", + U32(probe.hdrMaxDisplayMasteringLuminance) + " / " + + U32(probe.hdrMinDisplayMasteringLuminance)); + Check(probe.hdrMaxContentLightLevel == 1000 && + probe.hdrMaxFrameAverageLightLevel == 400, + "MaxCLL / MaxFALL bound", + U32(probe.hdrMaxContentLightLevel) + " / " + + U32(probe.hdrMaxFrameAverageLightLevel)); + Check(probe.hdrGreenPrimaryX == 8500 && probe.hdrGreenPrimaryY == 39850, + "the ST 2086 green primary bound unpermuted", + U32(probe.hdrGreenPrimaryX) + " / " + U32(probe.hdrGreenPrimaryY)); + + // A config with no chain must land NOTHING -- otherwise the assertions + // above could be satisfied by a default rather than by the binder. + VkVideoEncoderConfig plain = BaseConfig(); + plain.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + VkEncBoundConfigProbe pprobe{}; + Check(VkEncBuildAndProbeConfig( + plain, VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, &pprobe) == + VK_SUCCESS, + "unchained config builds", "init failed"); + Check(pprobe.hdrMasteringPresent == 0 && pprobe.hdrContentLightPresent == 0, + "an unchained config declares no HDR metadata", + U32(pprobe.hdrMasteringPresent) + " / " + + U32(pprobe.hdrContentLightPresent)); +} + +void CaseH264RefusesHdrMetadata() +{ + g_currentCase = "H.264 refuses HDR metadata it cannot carry"; + // There is no standard H.264 mastering-display or content-light SEI, so + // accepting the chain would be accepted-and-ignored -- the defect class + // this API refuses everywhere else. + VkVideoEncoderHdrMetadataInfo hdr = Hdr10Bt2020(); + VkVideoEncoderConfig cfg = BaseConfig(); + cfg.pNext = &hdr; + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig( + cfg, VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + "an H.264 session is refused rather than silently dropping it", + "VkResult " + U32((uint32_t)r)); +} + +//============================================================================= +// 6. The colour-model declaration at the registration gate +//============================================================================= +// +// Section 1 asks the taxonomy what a (format, colorModel) pair MEANS. This +// group asks the registration gate what it DOES about a pair that means +// nothing -- a different question with a different answer, and the only one a +// producer can act on: ValidateImageDescriptor is the negotiation point, +// reached identically from RegisterImageResource and from QueryImageSupport, +// and a declaration refused there is one the producer can still correct +// before it commits an allocation. +// +// The session these run against has a null backend: no device, no worker +// threads, no negotiated encode format. That is not a limitation here, it is +// the point. Whether a declaration can be read against its format is a +// property of the DESCRIPTOR ALONE, so the gate that judges it must answer +// the same on a session that has negotiated nothing, and this group is what +// holds it to that. + +class NullSession { +public: + bool Open() + { + if ((CreateVulkanVideoEncoderExt(m_encoder) != VK_SUCCESS) || + !m_encoder) { + std::printf(" ERROR: CreateVulkanVideoEncoderExt failed\n"); + return false; + } + if (VkEncInstallNullBackend(m_encoder.get(), &m_backend) != + VK_SUCCESS) { + std::printf(" ERROR: VkEncInstallNullBackend failed\n"); + return false; + } + return true; + } + VulkanVideoEncoderExt* Get() const { return m_encoder.get(); } + +private: + VkSharedBaseObj m_encoder; + VkEncNullBackendState m_backend{}; +}; + +// The shape a producer actually hands in: a block-linear image it intends the +// encoder to read directly. Zero-initialised except for the fields named, so +// colorModel is FROM_FORMAT unless a case says otherwise -- which is exactly +// the descriptor an ordinary caller builds. +VkVideoEncoderExternalImageDescriptor DirectDescriptor(VkFormat format) +{ + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = format; + desc.width = 640; + desc.height = 360; + desc.tiling = VK_IMAGE_TILING_OPTIMAL; + desc.imageUsage = VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.existingImage = (VkImage)(uintptr_t)0xA110C8ED; + return desc; +} + +VkVideoEncoderStatusCode Register( + NullSession& s, const VkVideoEncoderExternalImageDescriptor& desc) +{ + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode code = + s.Get()->RegisterImageResource(desc, 0, &resource, nullptr); + if (code == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + s.Get()->UnregisterImageResource(resource); + } + return code; +} + +VkVideoEncoderStatusCode Query( + NullSession& s, const VkVideoEncoderExternalImageDescriptor& desc) +{ + VkVideoEncoderImageSupport support{}; + support.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT; + s.Get()->QueryImageSupport(desc, &support); + return support.status; +} + +void CaseContradictoryColorModelIsRefusedAtRegistration(NullSession& s) +{ + g_currentCase = "a contradictory colorModel is refused at registration"; + // THE CONTRACT. A declaration the format cannot carry is answered + // COLOR_MODEL_UNSUPPORTED at the gate, which names the field that is + // wrong: NV12 below is directly encodable and it is the declaration over + // it that is refused. The alternative -- registering it and + // letting the route fall out of an unresolvable colour model -- is the + // accepted-and-silently-degraded shape: the registration succeeds, the + // direct route is declined for a reason the caller is never told, and the + // frames take a staging copy nobody asked for. + struct Row { + VkFormat format; + VkVideoEncoderColorModel declared; + const char* why; + }; + static const Row rows[] = { + { VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, + "RGB declared over NV12, the directly-encodable case" }, + { VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_RGB, + "RGB declared over 3-plane I420" }, + { VK_FORMAT_B8G8R8A8_UNORM, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR, + "YCbCr declared over BGRA8, which carries no packed reading" }, + }; + for (const Row& row : rows) { + VkVideoEncoderExternalImageDescriptor desc = + DirectDescriptor(row.format); + desc.colorModel = row.declared; + const VkVideoEncoderStatusCode reg = Register(s, desc); + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED, + (std::string("registration refuses it: ") + row.why).c_str(), + "status " + U32((uint32_t)reg)); + // The query is not a second opinion. It runs the same predicate, so a + // producer that negotiates hears the same word it would have heard + // from the registration it was about to attempt. + const VkVideoEncoderStatusCode q = Query(s, desc); + Check(q == VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED, + (std::string("and the query agrees: ") + row.why).c_str(), + "status " + U32((uint32_t)q)); + } +} + +void CaseZeroInitialisedDescriptorStillRegisters(NullSession& s) +{ + g_currentCase = "FROM_FORMAT, the zero-initialised case, still registers"; + // VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT is 0, so this is what every + // caller that never heard of the field declares. It reads the model off + // the format and must pass wherever the format itself does; a gate that + // caught it would refuse the entire installed base. + VkVideoEncoderExternalImageDescriptor desc = + DirectDescriptor(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM); + Check(desc.colorModel == VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, + "the descriptor under test really is the zero-init one", + "colorModel " + U32((uint32_t)desc.colorModel)); + const VkVideoEncoderStatusCode reg = Register(s, desc); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "an NV12 descriptor declaring no colour model registers", + "status " + U32((uint32_t)reg)); + const VkVideoEncoderStatusCode q = Query(s, desc); + Check(q == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "and the query says so too", "status " + U32((uint32_t)q)); +} + +void CaseAgreeingDeclarationStillRegisters(NullSession& s) +{ + g_currentCase = "a declaration that agrees with the format still registers"; + // Naming the model the format already implies is legal and changes + // nothing. This is the other half of the zero-init case: the gate refuses + // a CONTRADICTION, not a declaration. + VkVideoEncoderExternalImageDescriptor desc = + DirectDescriptor(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM); + desc.colorModel = VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + const VkVideoEncoderStatusCode reg = Register(s, desc); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "NV12 declared YCbCr registers", "status " + U32((uint32_t)reg)); + const VkVideoEncoderStatusCode q = Query(s, desc); + Check(q == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "and the query says so too", "status " + U32((uint32_t)q)); +} + +void CasePackedYcbcrDeclarationIsNotAContradiction(NullSession& s) +{ + g_currentCase = "the packed 4:4:4 declaration is not a contradiction"; + // AYUV has no Vulkan enumerant of its own and rides R8G8B8A8_UNORM, so + // YCbCr declared over that format is a statement of fact -- the one + // disagreement the library resolves rather than refuses. The gate must + // let it past, and this is what proves the gate reads the packed table + // instead of comparing two enums. + // + // Whether this session can then ROUTE the input is a separate question + // with a separate answer: a null-backend session built no compute filter, + // so the format needs a conversion this session does not have and the + // honest verdict is CONVERSION_REQUIRED. That it is NOT + // COLOR_MODEL_UNSUPPORTED is the assertion -- the colour-model gate did + // not fire. + VkVideoEncoderExternalImageDescriptor desc = + DirectDescriptor(VK_FORMAT_R8G8B8A8_UNORM); + desc.colorModel = VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + const VkVideoEncoderStatusCode reg = Register(s, desc); + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED, + "AYUV declared YCbCr passes the colour-model gate and is answered " + "on its routing, not on its declaration", + "status " + U32((uint32_t)reg)); +} + +void CaseUnknownColorModelValueIsRefused(NullSession& s) +{ + g_currentCase = "a colorModel outside the enumeration is refused"; + // A value the enumeration does not define cannot agree with any format, + // so it resolves to nothing and is refused by the same gate. Reading it as + // FROM_FORMAT would give a caller that miscomputed the field an encode it + // never asked for and no diagnostic anywhere. + VkVideoEncoderExternalImageDescriptor desc = + DirectDescriptor(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM); + desc.colorModel = (VkVideoEncoderColorModel)7; + const VkVideoEncoderStatusCode reg = Register(s, desc); + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED, + "an undefined colorModel is refused rather than read as 0", + "status " + U32((uint32_t)reg)); +} + +} // namespace + +// The registration gate stands up an encoder session; every other group in +// this binary calls functions that read only their arguments. Selecting it +// keeps the two apart, so a session that cannot be created is reported as +// itself rather than as a taxonomy failure. +bool WantsRegistrationGroup(int argc, char** argv) +{ + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--registration") == 0) { + return true; + } + } + return false; +} + +int RunRegistrationGroup() +{ + std::printf("Encoder-ext colour-model declaration at the " + "registration gate\n"); + std::printf("-------------------------------------------------------\n"); + + NullSession session; + if (!session.Open()) { + return 2; + } + CaseContradictoryColorModelIsRefusedAtRegistration(session); + CaseZeroInitialisedDescriptorStillRegisters(session); + CaseAgreeingDeclarationStillRegisters(session); + CasePackedYcbcrDeclarationIsNotAContradiction(session); + CaseUnknownColorModelValueIsRefused(session); + + std::printf("-------------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} + +int main(int argc, char** argv) +{ + if (WantsRegistrationGroup(argc, argv)) { + return RunRegistrationGroup(); + } + + std::printf("Encoder-ext input-format taxonomy, preprocess decision,\n"); + std::printf("transfer-function declaration and colour description\n"); + std::printf("-------------------------------------------------------\n"); + std::printf("compute filter compiled in: %s\n", + kFilterCompiledIn ? "yes" : "no"); + + CaseSemiPlanarIsDirect(); + CaseTwelveBitSemiPlanarIsViaFilter(); + CaseThreePlaneIsViaFilter(); + CaseRgbaIsViaFilterAndSinglePlane(); + CaseSrgbAndJunkAreUnsupported(); + CaseDeclaredColorModelDecidesTheClass(); + CaseAdvertisedListDropsWhatTheLibraryRefuses(); + CaseAdvertisedListPassesWhatTheLibraryRoutes(); + CaseAdvertisedListReportsOneFormatOnce(); + CaseAdvertisedListStopsAtCapacity(); + CaseAdvertisedListDropsAnUnreachableConversionTarget(); + CaseOptimalityNamesTheEncodersOwnFormat(); + CaseFourFourFourReachesTheListOnlyFromTheDevice(); + CaseFilterlessBuildAdvertisesNoConvertedEntry(); + CaseConversionTargetPreservesSubsamplingAndDepth(); + CaseRoutableListAgreesWithTheClassifier(); + CaseRoutableSetIsDerivedFromTheFormatTables(); + CaseSinglePlaneInterleavedIsRefused(); + CaseFourTwoZeroDeviceAdvertisesTheHistoricalSet(); + CaseYcbcrIsNeverClaimedAsRgba(); + CaseRgbaSessionSurvivesTheSinglePlaneGate(); + + CaseContradictoryColorModelIsRefusedByTheBinder(); + CaseAgreeingColorModelStillBinds(); + CaseSemiPlanarBindsTwoPlanes(); + CaseInputFormatSurvivesTheRoundTripThroughGeometry(); + CaseTenBitBindsBitDepthAndPlanes(); + CaseFourFourFourBindsItsOwnSubsampling(); + CaseFourTwoZeroStillBindsFourTwoZero(); + CaseDirectFormatGetsNoFilter(); + CaseThreePlaneGetsAFilterWithoutAsking(); + CaseRgbaGetsAFilterWithoutAsking(); + CaseUnsupportedFormatStillRefused(); + CasePackedYcbcrIsRoutedWhereItIsDeclared(); + CaseFilterArmReadsTheDeclaredColourModel(); + + CaseUndeclaredInputOtfAssertsNothing(); + CaseAgreeingOtfDeclarationsAreAccepted(); + CaseMismatchedOtfDeclarationIsRefused(); + + CaseProfileNumbersReachTheCodecConfigUnchanged(); + CaseProfileDefaultIsDerivedPerCodec(); + CaseProfileNumbersAreReadAgainstTheCodec(); + CaseProfileMustAdmitTheInputDepth(); + CaseProfileMustAdmitTheInputSubsampling(); + CaseWidenedBindSetStillRunsTheLimitsGuard(); + CaseNamedProfileConstantsAreExactlyTheBoundSet(); + CaseH265CpbVclFactorFollowsTheChromaFormat(); + CaseAv1SubsamplingMatchesTheDerivedSeqProfile(); + CaseCodecArmsDeriveTheProfileFromTheEncodeGeometry(); + CaseInputColourChainBindsEachAxis(); + CaseAbsentInputColourChainChangesNothing(); + CaseInputColourDisagreementIsRefused(); + CaseDeclaredInputRangeDecidesTheStreamsRange(); + CaseInputColourPrimariesDriveTheDerivedMatrix(); + + CaseProbeNamesAv1MainAtBothDepths(); + CaseProbeRefusesDepthsItHasNoEvidenceFor(); + CaseSnapshotCarriesBothAv1Depths(); + + CasePartialColourSupplyDoesNotFabricateTheRest(); + CaseFullRangeOnlyDeclaresRangeAndNoColour(); + CaseAv1SignalsRangeWithoutAColourDescription(); + CaseHdr10CodePointsReachBothArms(); + CaseUnexpressibleMatrixIsRefusedOnAnRgbaSession(); + CaseUnspecifiedMatrixIsDerivedFromPrimaries(); + CaseMatrixDispositionTable(); + CaseYcbcrSessionKeepsAMatrixTheFilterCannotProduce(); + CaseChromaSitingIsSignalledOnlyWhereItIsKnown(); + + CaseH265HdrSeiBytes(); + CaseAv1HdrMetadataObuBytes(); + CaseEachHdrPayloadIsIndependent(); + CaseHdrPayloadRefusesToTruncate(); + CaseHdrMetadataBindsThroughThePnextChain(); + CaseH264RefusesHdrMetadata(); + + CaseProfileNumbersReachTheCodecConfigUnchanged(); + CaseProfileDefaultIsDerivedPerCodec(); + CaseProfileNumbersAreReadAgainstTheCodec(); + CaseProfileMustAdmitTheInputDepth(); + CaseProfileMustAdmitTheInputSubsampling(); + CaseNamedProfileConstantsAreExactlyTheBoundSet(); + CaseInputColourChainBindsEachAxis(); + CaseAbsentInputColourChainChangesNothing(); + CaseInputColourDisagreementIsRefused(); + CaseDeclaredInputRangeDecidesTheStreamsRange(); + CaseInputColourPrimariesDriveTheDerivedMatrix(); + CaseAv1SubsamplingMatchesTheDerivedSeqProfile(); + CaseH265CpbVclFactorFollowsTheChromaFormat(); + CaseWidenedBindSetStillRunsTheLimitsGuard(); + CaseProbeNamesAv1MainAtBothDepths(); + CaseProbeRefusesDepthsItHasNoEvidenceFor(); + CaseSnapshotCarriesBothAv1Depths(); + + CaseCodecArmsDeriveTheProfileFromTheEncodeGeometry(); + + CaseFieldTableClassifiesEveryField(); + CaseFieldTableTilesTheStruct(); + + std::printf("-------------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-format-encode/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-format-encode/CMakeLists.txt new file mode 100644 index 00000000..4f1646b0 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-format-encode/CMakeLists.txt @@ -0,0 +1,273 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_format_encode_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The STATIC library, for the same reason the sibling taxonomy test links it: +# the internal header's free functions -- the format taxonomy and +# VkEncProbeResource, which is how this test reads a slot's resolved input +# path -- are deliberately not exported from libvkvideo-encoder.so. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +# Must branch the same way the library did -- an RGBA session cannot route +# FILTER at all in a build without the filter, and a test that assumed it +# present would fail for the wrong reason. +if(BUILD_ENCODER_COMPUTE_FILTER) + target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED) +endif() + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add test. +# +# THE NAME IS NOT EncoderExtInputFormatMatrix. This block was copied verbatim +# from encoder-ext-format-matrix next door, comment header included, so two +# different binaries were registered under one ctest name: `ctest -R +# EncoderExtInputFormatMatrix` ran both, and a name-keyed CI line was +# ambiguous between them. The CI set-comparison could not catch it either -- +# it compares COUNTS, and 29 == 29 with one name appearing twice. +# +# Rows that are not abandoned by the safety guard create a real encode session +# and a real VkImage on the LIBRARY-OWNED device and then check that BITSTREAM +# BYTES CAME BACK -- and that the GPU was not lost, that no frame was dropped, +# that the submit loop ran to completion and that the output file opened. That +# is what separates this from the routing test next door: that one proves a +# descriptor registers and a slot resolves, and contains no encode call at all. +# +# AND THEN IT LOOKS AT THE PIXELS, which is the half that was missing. Every +# row that encodes is handed to ffmpeg -- a decoder outside this process -- and +# the four quadrant centres of frame 0 are compared against the primaries the +# harness wrote, to a tolerance of 24/255. That assertion is in the DEFAULT +# verdict chain, not behind a flag, because the state it catches is invisible +# to everything else here: delete the four vkCmdDispatch calls in +# VulkanFilterYuvCompute.cpp and this test still reported +# `sessions=8 encoded=8 abandoned=0 failures=0` and exit 0, with dispatch=12 on +# every filter row, while all five filter-routed rows decoded to a flat +# (0,76,0). See kQuadTolerance in src/main.cpp for the measurements. (Those +# counts are the 14-row table of the day; the current bar is in the +# src/main.cpp header.) +# +# ONE ROW DOES NOT TAKE THE QUADRANT GATE, and it is not an exemption. The +# HDR10 row declares BT.2020/PQ, so a decoder inverting the BT.2020 matrix +# legitimately produces different RGB from the BT.709 pattern the harness +# wrote; comparing it against the BT.709 expectations would fail a CORRECT +# encode. CheckHdrSignalling() judges it instead -- the four VUI code points, +# a full decoded frame whose quadrant centres are not all identical, and both +# HDR10 SEI payloads read back field by field by ffprobe -- and it is counted +# in its own summary fields, hdrGated / hdrFailed, so a row judged by neither +# gate cannot hide inside decodeGated. +# +# The reference bar is: +# sessions=9 encoded=9 abandoned=0 failures=0 decodeGated=8 decodeFailed=0 +# deviceLimited=6 av1Unverified=0 hdrGated=1 hdrFailed=0 (exit 0) +# +# A 3-plane row that does not resolve to FILTER is ABANDONED before any submit +# (staging one is a measured GPU hang) and is neither encoded nor failed here; +# the routing test next door is what turns that regression red. +# +# So this is a gpu-label test. It exits 77 (SKIPPED) in exactly two cases, and +# 1 for any other failure: +# +# 1. every row reports VK_ERROR_INCOMPATIBLE_DRIVER, i.e. no usable driver; +# 2. ffmpeg is not on PATH, so the four-quadrant decode assertion could not +# be rendered on the rows that encoded. +# +# The second is deliberate and is NOT a convenience. A gate that passes when +# its decoder is missing is just another test that cannot fail, and the +# counters this suite would be left with are green with the compute filter +# dead. So an unjudged run reports SKIPPED, loudly, with the reason on stdout, +# rather than PASSED. A genuine failure is still 1: main() returns 1 on +# g_failures before it ever considers the skip, so the skip cannot launder one. +# +# Everything else is 1 -- including "no session at all for a reason that is not +# an absent driver", which is a library refusal wearing an environment costume. +enable_testing() +add_test(NAME EncoderExtInputFormatEncode + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtInputFormatEncode PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu") + +# --------------------------------------------------------------------------- +# THE FILTER-ARM RESTORE, WHICH SHIPPED WITH NO COVERAGE AT ALL. +# +# The filter branch of StageInputFrame hands the input image back with +# RestoreStagedInputLayout. For a consumer that declares GENERAL -- which +# equals that arm's residual -- the helper's equal-layout early return fires +# and no barrier is recorded, so the call does nothing. +# +# --declare-tso is the shape that exercises it: the registration declares +# TRANSFER_SRC_OPTIMAL, which differs from the filter arm's residual (GENERAL), +# so the early return does not fire and a real barrier is recorded on every +# frame of every filter-routed row that can legally hold that layout. +# +# In the default mode no restore barrier is recorded at all -- the early +# return fires on every frame -- while under --declare-tso one is recorded +# per frame of every filter-routed row (GENERAL -> TRANSFER_SRC_OPTIMAL). +# +# AND IT IS LOAD-BEARING, which a count could not have shown. Suppressing ONLY +# the CmdPipelineBarrier2KHR inside RestoreStagedInputLayout, leaving its +# return value -- so the registration's residual record still claims +# TRANSFER_SRC_OPTIMAL while the image is really still in GENERAL -- makes +# frame 2's acquire name a layout the image is not in, and the validation +# layer raises VUID-vkCmdDraw-None-09600 for every plane of every affected +# frame under --declare-tso. The default mode reports nothing, because it has +# nothing to suppress -- which is the coverage gap this mode closes. +add_test(NAME EncoderExtInputFormatEncodeDeclaredTransferSrcOptimal + COMMAND ${PROJECT_NAME} --declare-tso) +set_tests_properties(EncoderExtInputFormatEncodeDeclaredTransferSrcOptimal + PROPERTIES SKIP_RETURN_CODE 77 LABELS "gpu") + +# --------------------------------------------------------------------------- +# THE AV1 ARM. This is the only ctest entry that covers it. +# +# AV1 also runs inside EncoderExtInputFormatEncode above -- the four AV1 rows +# are in kRows, so the default gating entry already covers them, and a broken +# AV1 arm turns THAT red on hardware that has AV1. This second entry exists +# because a codec-scoped run reports a codec-scoped verdict: `ctest -R Av1` +# answers "AV1: passed / skipped / failed" instead of leaving the reader to +# find four rows inside thirteen. +# +# HOW IT STAYS GREEN ON THE A4000, WHICH IS THE QUESTION WORTH ANSWERING. +# The RTX A4000 has no VK_KHR_video_encode_av1, so all four rows get no +# session and this entry creates ZERO sessions. That +# combination hit main()'s `nSession == 0` branch, which returns 1 for +# anything other than "every row said VK_ERROR_INCOMPATIBLE_DRIVER" -- and +# these rows say VK_ERROR_FEATURE_NOT_PRESENT, because there IS a driver. +# Registering this entry as-is would therefore have turned A4000 CI RED, +# which is why main() now distinguishes a DEVICE refusal from a LIBRARY +# refusal (IsDeviceLimitedInit) and reports the former as 77. +# +# So on hardware without AV1 this entry is SKIPPED via the SKIP_RETURN_CODE 77 +# already declared for its siblings, with the reason and the per-device +# capability probe on stdout. It is not silent, and it is not green-by-default: +# main() reaches that 77 only with g_failures == 0 AND no unverified row, and +# an AV1 row that gets no session on a device that DOES advertise the +# extension is counted as a failure before the skip is ever considered. +# +# The one thing this entry cannot do on such hardware is pass. That is +# correct: it has encoded nothing. +add_test(NAME EncoderExtInputFormatEncodeAv1 + COMMAND ${PROJECT_NAME} --codec av1) +set_tests_properties(EncoderExtInputFormatEncodeAv1 + PROPERTIES SKIP_RETURN_CODE 77 LABELS "gpu") + +# --------------------------------------------------------------------------- +# THE CONTENT PROBE'S ONLY EXECUTING TEST, ANYWHERE IN THE TREE. +# +# test/encoder-ext-import-content is device-free by construction: it drives the +# scorer and the latch directly and registers against a NULL-BACKEND session +# where no frame is ever submitted. It can therefore assert that a registration +# ARMS, and it cannot assert that an armed registration ever produces a +# VERDICT. Nothing else submitted a frame with the probe chained. +# +# That gap sat directly on top of the change that needed it most. A DIRECT +# (block-linear + VIDEO_ENCODE_SRC) registration only reaches the capture site +# because VkVideoEncoder::SetExternalInputFrameWithNode sends its FIRST frame +# down a staged detour. This entry is what executes that detour. +# +# THE GAP, DEMONSTRATED RATHER THAN ASSERTED. Reverting the detour to +# `if (directlyEncodable)` instead of +# `if (directlyEncodable && !probeStillOwesACapture)` gives: +# +# ctest -L device-free 6/6 passed +# encoder_ext_import_content_test checks: 118, failures: 0 RESULT: PASS +# THIS ENTRY exit 1, row [1/13] NV12: +# "the registration ARMED and NOTHING WAS EVER SCORED (probed=0 ...)" +# +# HEALTHY, same host, row [1/13] NV12 (DIRECT, VK_IMAGE_TILING_OPTIMAL, i.e. +# the block-linear class the periodicity harness measured poisoned): +# armEcho=ARMED(gen=1) final=CLEAN probed=1 damaged=0 armedOutstanding=0 +# meanY=31983 meanU=32768 meanV=32768 (Q8) +# +# THOSE CHROMA NUMBERS ARE THE PROOF THE CAPTURE READ THE REAL PIXELS, not +# merely that it ran. The harness uploads a four-quadrant red/green/blue/white +# bar. In BT.709 limited range the Cb coefficients of R, G and B sum to +# exactly zero (-0.1146 - 0.3854 + 0.5), and so do the Cr coefficients +# (0.5 - 0.4542 - 0.0458), so the three colour quadrants contribute exactly +# 3*128 and the white quadrant 128: mean 128.000 in BOTH chroma channels, or +# 32768 in the Q8 units this field carries. Measured 32768 and 32768. Luma +# predicts 125.500 (Q8 32128) against a measured 31983, the 0.57 gap being +# 8-bit quadrant rounding plus the scorer's every-8th-row sampling. +# +# It is a gpu-label test and shares the 77/SKIPPED contract of its siblings. +add_test(NAME EncoderExtInputFormatEncodeContentProbe + COMMAND ${PROJECT_NAME} --content-probe) +set_tests_properties(EncoderExtInputFormatEncodeContentProbe + PROPERTIES SKIP_RETURN_CODE 77 LABELS "gpu") + +# THE DECODER IS A RUNTIME DEPENDENCY OF THE TEST, NOT A BUILD DEPENDENCY, and +# deliberately not a hard one at configure time. In this project the machine +# that configures and builds (no GPU) is not the machine that runs (the GPU +# host), so a find_program() failure here would say nothing about whether the +# assertion can run where it matters -- and making it REQUIRED would break a +# build that is perfectly capable of producing the binary. The binary probes +# for ffmpeg itself and reports SKIPPED if it is absent. This is a heads-up for +# whoever configures, nothing more. +find_program(FFMPEG_FOR_ENCODE_GATE ffmpeg) +if(NOT FFMPEG_FOR_ENCODE_GATE) + message(STATUS "encoder_ext_format_encode_test: ffmpeg not found on the " + "BUILD host. The four-quadrant decode assertion needs it on " + "the RUN host; without it there, this test reports SKIPPED " + "rather than PASSED.") +else() + message(STATUS "encoder_ext_format_encode_test: decode gate will use " + "${FFMPEG_FOR_ENCODE_GATE} if the run host has it") +endif() + +message(STATUS "encoder_ext_format_encode_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-format-encode/src/main.cpp b/vk_video_encoder/test/encoder-ext-format-encode/src/main.cpp new file mode 100644 index 00000000..4e3094fa --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-format-encode/src/main.cpp @@ -0,0 +1,3352 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * INPUT-FORMAT *ENCODE* MATRIX, on the LIBRARY-OWNED device. + * + * WHY THIS EXISTS BESIDE encoder-ext-format-matrix. That test walks the same + * nine formats and proves REGISTRATION and ROUTING -- a descriptor is + * accepted and a slot resolves to DIRECT/FILTER/STAGED. It contains no + * encode call at all: zero submits, zero bitstreams. "Registers and routes + * FILTER" is a strictly weaker claim than "encodes", and the gap between + * them is exactly where a filter that dispatches nothing, or converts pure + * black, has hidden on this project before (see the filter suite that + * reported 54/54 while converting black). + * + * WHAT THIS ADDS. For every format the device can carry a profile for, it + * 1. fills a real image with a KNOWN four-quadrant primaries pattern, + * written in that format's own layout (RGB direct for the RGBA arm; + * BT.709 limited-range YCbCr for the YCbCr arms, so one comparison + * harness serves both); + * 2. submits frames through SubmitRegisteredFrame; + * 3. drains the bitstream out of the library and writes it to a file for + * an INDEPENDENT decoder (ffmpeg, outside this process) to judge. + * + * WHAT IT CAN FAIL ON -- stated up front: + * - a format whose session initializes, registers and routes correctly can + * still produce ZERO bytes of bitstream. That is the whole point: it is + * the failure the routing test cannot see. + * - the filter observable is read from GetCompletionInfo's chained + * VkVideoEncoderFilterInfo and reported as a PAIR (dispatch, staged) + * against the frame count. A dispatch count that is non-zero but does + * not track frames, or that rises while stagedCopyCount also rises, is + * visible here rather than averaged away. + * - bytes alone prove nothing about colour. Colour is judged by an + * INDEPENDENT DECODER -- ffmpeg, a separate process reading the written + * file -- and this program never decodes its own bitstream in-process. + * What it does do is RUN that decoder and grade the four quadrant centres + * it returns; the promise used to stop at "is judged out of process" with + * nothing anywhere that judged. + * + * WHAT "COLOUR IS JUDGED" MEANT UNTIL CC-1 W8, because the unqualified + * claim was broader than the gate. The decode was to rgb24 at tolerance + * 24, and in the RGB domain that gate judged FILTER LIVENESS and CHANNEL + * ORDER -- both of which move a flat primary by 177 or more -- and NOT + * the conversion matrix. Measured: it passed THREE of the five + * applied/declared matrix confusions, including `applied 709 / declared + * 601` at 23 against a tolerance of 24, i.e. by one code value. The gate + * now decodes to yuv444p and compares in the Y'CbCr domain against the + * matrix the row DECLARED, at tolerance 4, and the LABEL is checked + * explicitly rather than implicitly. See kQuadTolerance for the full + * before/after measurement. + * + * SAFETY -- multi-planar staging is a GPU HANG on this hardware + * (VK_ERROR_DEVICE_LOST, 0-byte bitstream). A 3-plane row whose + * slot does not resolve to FILTER is ABANDONED BEFORE ANY SUBMIT rather than + * encoded, and says so. The guard is unconditional and is not a diagnostic. + * + * THE BAR, IN FULL. Four numbers. The second was missing when this file was + * written; the third was missing until the decode assertion below was wired + * into the default verdict chain, and it is the only one of the first three + * that can see the compute filter stop executing; the fourth arrived with the + * HDR10 row, which takes a different gate and therefore needs its own count: + * + * sessions=9 encoded=9 abandoned=0 failures=0 (default and --declare-tso) + * decodeGated=8 decodeFailed=0 (default and --declare-tso) + * hdrGated=1 hdrFailed=0 (default and --declare-tso) + * 0 "The Vulkan spec states" under --validate (default and --declare-tso) + * + * That is the bar, at exit 0. + * decodeGated is 8 and not 9 ON PURPOSE: the HDR10 row's colour volume is + * BT.2020/PQ, so the BT.709 quadrant comparison would fail on a CORRECT + * encode. It is judged by CheckHdrSignalling() instead and counted + * separately, because folding two different assertions into one number is how + * "eight of nine rows were judged" would read as green. + * + * THE THIRD LINE IS NOT REDUNDANT WITH THE FIRST. Delete the four + * m_vkDevCtx->CmdDispatch calls in VulkanFilterYuvCompute.cpp and the first + * line is UNCHANGED -- measured as sessions=8 encoded=8 abandoned=0 + * failures=0 when this table had 14 rows and no HDR row, exit 0, dispatch=12 + * on every FILTER row -- while all five filter-routed rows decode to a flat + * (0,76,0) in every quadrant. See kQuadTolerance below for the measurement + * and for why the threshold is what it is. + * + * The validation half is stated because leaving it out cost this suite seven + * real messages. `--validate` emitted 7 x + * VUID-vkCmdPipelineBarrier-pImageMemoryBarriers-02820 -- one per session -- + * while the counter half read green, because a counter that only counts + * bitstreams cannot see a barrier. They were not the library's: they came + * from this file's own producer emulation in UploadPattern, whose handover + * barrier named VK_ACCESS_SHADER_READ_BIT on the ENCODE family. Fixed there; + * see the comment on toGen.dstAccessMask. + * + * --foreign-residency is EXCLUDED from the bar and exits non-zero by design: + * its two DIRECT rows drive a driver defect (see g_foreignResidency). + * + * AV1, ADDED FOR THE BLACKWELL SESSION -- AND THE NUMBERS ABOVE ARE FOR THE + * REFERENCE HOST, WHICH CANNOT RUN IT. + * + * Four AV1 rows now sit at the bottom of kRows. On an RTX A4000 (GA104) there + * is no VK_KHR_video_encode_av1 at all, so all four report DEVICE-LIMITED and + * contribute nothing but a deviceLimited count. The full summary line, as + * The full summary line on such a device is: + * + * sessions=9 encoded=9 abandoned=0 failures=0 + * decodeGated=8 decodeFailed=0 deviceLimited=6 av1Unverified=0 + * hdrGated=1 hdrFailed=0 + * + * The eighth session is the HEVC Main 8-bit row and the ninth is the HDR10 + * row. + * + * AV1 IS THE ONE ARM THE REFERENCE HOST CANNOT JUDGE, and the HDR feature + * inherits that ON THAT HOST: the AV1 metadata OBUs are built and appended by + * the library, and their BYTES are pinned device-free in + * test/encoder-ext-filter against values ffprobe read back out of a real AV1 + * stream. + * + * NO LONGER PREDICTED. On an RTX 5070 all four AV1 rows encode and are + * decode-gated, and TWO AV1 HDR ROWS ARE NOW IN THE TRACKED MATRIX -- their + * mastering-display and content-light payloads are read back field by field + * by ffprobe on every run. What had kept them out was not the encoder: it was + * CheckHdrSignalling() string-matching H.265's ST 2086 units, against which a + * correct AV1 stream reported ten MISSING lines. The gate parses and compares + * numerically now, so one assertion serves both codecs. + * + * deviceLimited counts P012 and I420-12 (no 12-bit profile on this device) + * plus the AV1 rows. On hardware that HAS AV1 encode those rows join the + * encoded set -- and the "12/12/12" this line used to predict was simply + * wrong arithmetic (9 + 4 = 13), quite apart from the two AV1 HDR rows added + * since. On a device that HAS AV1 encode the summary line reads: + * + * sessions=15 encoded=15 abandoned=0 failures=0 + * decodeGated=12 decodeFailed=0 deviceLimited=2 av1Unverified=0 + * hdrGated=3 hdrFailed=0 + * + * deviceLimited falls from 6 to 2 there: the AV1 rows run, and only the two + * 12-bit rows remain refused. decodeGated stays at 12 and hdrGated rises to 3 + * because the three HDR rows -- one H.265 and two AV1 -- take + * CheckHdrSignalling() rather than the quadrant gate, for the reason given at + * the HDR row in kRows. + * + * "DEVICE-LIMITED" IS NOT TAKEN ON TRUST, which is the part that matters. + * A no-session row costs nothing and fails nothing, so a new arm added this + * way is a test that cannot fail by construction. Every AV1 row that gets no + * session is therefore checked against the DEVICE -- see + * ProbeAv1EncodeSupport() -- and a device that advertises the extension while + * the session refuses to start is a FAILURE, loudly, not a skip. + * + * WHAT THE AV1 ROWS DO NOT COVER, stated here so nobody reads their green as + * broader than it is: this suite sets disableFileOutput, which takes + * VkVideoEncoderAV1's IN-MEMORY capture arm. The DKIF/'AV01' IVF muxer -- + * BuildFrameObuSequence and FlushBatchedTemporalUnit -- lives on the FILE arm + * and is not reached from here, nor from Chromium. See the AV1 macro below. + */ + +#include "vulkan_video_encoder_ext.h" +#include "vulkan_video_encoder_ext_internal.h" + +#include "vk_video/vulkan_video_codec_h264std.h" +#include "vk_video/vulkan_video_codec_h265std.h" +#include "vk_video/vulkan_video_codec_av1std.h" + +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace { + +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; +const uint32_t kFrames = 12; + +int g_failures = 0; + +// DIAGNOSTIC ONLY. The X6/X4 formats put the sample in the HIGH bits of each +// 16-bit word (that is what the suffix means, and P010 -- same convention, +// same writer -- decodes correctly through the DIRECT path). This flag writes +// the sample in the LOW bits instead. It exists to tell a library-side +// misread apart from a writer-side convention error on the 3-plane 10-bit +// arm. It is not a fix and must not become one. +// +// This flag is a DISCRIMINATOR and not a description of a live bug: the row +// it names decodes out-of-process indistinguishably from the other rows. +// +// WHAT IS STILL TRUE, AND IS WHY THIS ROW'S GREEN IS NOT EVIDENCE FOR +// CHROMIUM: this harness's writer is MSB-aligned by design, and Chromium's +// PIXEL_FORMAT_YUV420P10 is LSB-aligned. The shape Chromium would produce for +// this row is not what this row exercises, which is exactly why +// VulkanVideoEncoderConfigBuilder::MapPixelFormat still answers +// VK_FORMAT_UNDEFINED for it. +bool g_rawBits = false; + +// --declare-tso. THE POINT OF THIS MODE, stated where it is defined. +// +// The filter arm's LOCAL handback -- RestoreStagedInputLayout called from the +// filter branch of StageInputFrame -- does nothing for a consumer that +// declares GENERAL: GENERAL equals that arm's residual, so the helper's +// equal-layout early return fires and no barrier is recorded. +// +// This mode is the shape that exercises it. Declaring TRANSFER_SRC_OPTIMAL +// instead of GENERAL makes targetLayout differ from the filter arm's residual +// (GENERAL), so the early return does NOT fire and a real +// (GENERAL -> TRANSFER_SRC_OPTIMAL) barrier is recorded on every frame. +// +// It is not a contrived declaration: TRANSFER_SRC_OPTIMAL is the ext layer's +// own legacy-wrap default, and it is what a caller that pools a staging image +// and last used it as a copy source must state to be truthful. +// +// The loop closes with no missing arm: frame 1 acquires TRANSFER_SRC_OPTIMAL +// -> GENERAL (an arm that already exists), the handback returns the image to +// TRANSFER_SRC_OPTIMAL and RECORDS that, and frame 2 reads the record and +// names TRANSFER_SRC_OPTIMAL again. +bool g_declareTso = false; + +// --content-probe. THE ONLY PLACE IN THE TREE THAT EXECUTES THE CONTENT +// PROBE'S CAPTURE. +// +// Every other assertion about the probe -- the whole of +// test/encoder-ext-import-content -- is device-free: it drives the scorer and +// the latch directly, and drives RegisterImageResource on a NULL-BACKEND +// session where no frame is ever submitted. So it can assert that a +// registration ARMS, and it cannot assert that an armed registration ever +// produces a VERDICT. +// +// That gap sat exactly on top of the change that needed it most. Arming a +// DIRECT (block-linear + VIDEO_ENCODE_SRC) registration is only useful +// because SetExternalInputFrameWithNode sends its first frame down a staged +// detour so the capture site in StageInputFrame can read it. If that detour +// silently stopped firing, the device-free suite would stay green and the +// registration would sit ARMED forever reporting NOT_EVALUATED -- an +// observable that cannot fail, which is the shape this suite's own C1c case +// declares unacceptable. +// +// THE ROW IS ALREADY THE RIGHT SHAPE, which is why this is a mode and not a +// new binary: DeclFor(ARM_DIRECT) declares VK_IMAGE_TILING_OPTIMAL plus +// VIDEO_ENCODE_SRC plus TRANSFER_SRC -- block-linear and directly encodable, +// i.e. the class the periodicity harness measured poisoned and the class the +// probe can be blind to -- and UploadPattern fills it with a four-quadrant +// colour bar before any frame is submitted. A colour bar is emphatically not +// a dead plane, so CLEAN is the answer a working capture must produce, and +// the arithmetic behind it is printed so the verdict is readable rather than +// merely asserted. +// +// WHAT EACH OUTCOME MEANS on a DIRECT row under this flag: +// echo ARMED + final CLEAN + probed>=1 + armed==0 -> the detour ran, the +// capture was recorded, the fence was waited and the bytes were scored. +// This is the pass. +// echo ARMED + final ARMED + probed==0 + armed==1 -> the arm decision works +// and NOTHING DOWNSTREAM OF IT DOES. This is precisely the regression +// the device-free suite cannot see, and it is a FAIL here. +// echo NOT_APPLICABLE -> the arm predicate +// regressed to reading encodeCapable/tiling again. FAIL. +bool g_contentProbe = false; + +// --validate. THIS SUITE HAD NO WAY TO ENABLE THE VALIDATION LAYER AT ALL. +// Every sibling suite has one; this one did not, so setting VK_LAYER_PATH +// around it did exactly nothing and any "zero validation errors" reading +// taken from it was vacuous -- the layer was never in the instance. Found +// while trying to use this suite as a validation gate for the filter-arm +// restore; the first mutation run came back clean and the clean run was the +// bug. +bool g_validate = false; + +// --foreign-residency. THE TWO RELEASE SITES THAT HAVE NEVER BEEN DRIVEN. +// +// VkVideoEncoder::ReleaseImageToForeignQueue is called from three places and +// only ONE of them had ever executed in any run in this tree: the staging COPY +// arm (StageInputFrame, !useComputeFilter), driven by +// encoder-ext-input-residency --foreign-opaque-fd. The other two -- the staging +// FILTER arm (StageInputFrame, compute-filter else-branch) and Path A +// (RecordVideoCodingCmd, after CmdEndVideoCodingKHR) -- had no traffic at all: +// grepping every log in the tree and on the GPU host for the [QFOT-REL] record +// found zero occurrences carrying old=GENERAL or old=VIDEO_ENCODE_SRC_KHR. +// +// WHY THIS ONE FLAG REACHES BOTH. The path is chosen by the registration's +// declared usage/tiling (ARM_DIRECT resolves DIRECT, ARM_FILTER_* resolve +// FILTER); the RESIDENCY is an independent axis this suite had pinned to LOCAL. +// Flipping it to FOREIGN is therefore the whole difference, and it is honoured +// verbatim: VulkanVideoEncoderExtImpl::SubmitRegisteredFrame only DERIVES +// residency when handleType is not VK_IMAGE, and every row here registers a +// VK_IMAGE. So the same nine rows re-run with FOREIGN drive the filter release +// on the six FILTER rows and Path A's release on the three DIRECT rows. +// +// VK_IMAGE IS THE RIGHT VEHICLE, not a compromise for a missing dma-buf. The +// image is created by this suite on the session device with no external memory, +// so the validation layer tracks its layout completely. A real dma-buf import +// brings VVL's external-image relaxations into play and can MASK exactly the +// layout inconsistency this mode exists to expose. +// +// The only extra requirement declaring FOREIGN imposes is that +// VK_EXT_queue_family_foreign be present (ValidateImageDescriptor); it is, on +// this hardware, which is what lets --foreign-opaque-fd pass today. +// +// WHAT IT REPORTS -- READ THIS BEFORE TREATING A NON-ZERO EXIT AS A +// REGRESSION. At 12 frames per row: +// +// FILTER arm (5 rows, 60 frames): GREEN. The release records +// (old=GENERAL new=GENERAL) on every frame and every row encodes +// BYTE-IDENTICAL output to the RESIDENCY_LOCAL run (1930, 3162, 1930, 1930, +// 1930). This site had never executed before; it is correct as written. +// +// DIRECT arm / Path A (2 rows, 24 frames): RED, AND STILL OPEN. The release +// records (old=VIDEO_ENCODE_SRC_KHR new=GENERAL) and the session then dies: +// VK_ERROR_DEVICE_LOST, 24 x VUID-vkResetFences-pFences-01123, and a +// ZERO-BYTE bitstream on both rows where RESIDENCY_LOCAL produces 2011 and +// 3471 bytes. So --foreign-residency EXITS NON-ZERO BY DESIGN today; the +// two DIRECT failures are the finding, not a broken suite. +// +// WHAT THE HANG IS NOT -- each of these was tried and measured, and none of +// them changed the outcome: +// * NOT the release/acquire layout disagreement. Path A's release used to +// hardcode VIDEO_ENCODE_SRC_KHR as its newLayout while the next acquire +// declares GENERAL. That is handled in VkVideoEncoder::RecordVideoCodingCmd +// (pathAProducerLayout) and confirmed via the [QFOT-REL] record; the +// DEVICE_LOST is unaffected either way. +// * NOT missing external memory. Backing the DIRECT images with +// VkExternalMemoryImageCreateInfo + VkExportMemoryAllocateInfo (OPAQUE_FD) +// changed nothing. That experiment was reverted rather than kept, because +// it bought no behaviour. +// * NOT the acquire. Suppressing ONLY the Path A release and keeping the +// acquire makes the NV12 row encode 2011 bytes with zero errors -- so the +// release barrier alone is the trigger. +// +// IT HAS NOW BEEN CHASED, AND IT IS NOT A LIBRARY DEFECT. The two bullets +// above survive, but they were "what it is not"; the cause is below and it is +// in the DRIVER. Isolated to a standalone program that links no part of this +// library, creates no video session, records no encode, and still loses the +// device. +// +// THE DEFECT, stated as a conjunction. A vkCmdPipelineBarrier2 image barrier +// loses the device iff ALL THREE hold: +// (1) the image was created with VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR or +// VK_IMAGE_USAGE_VIDEO_ENCODE_DPB_BIT_KHR. A VkVideoProfileListInfoKHR +// WITHOUT either usage bit is green, so it is the usage, not the +// profile, and not the multi-planar format (a plain +// TRANSFER_SRC/DST NV12 image is green); +// (2) the barrier is a RELEASE -- dstQueueFamilyIndex == +// VK_QUEUE_FAMILY_FOREIGN_EXT. VK_QUEUE_FAMILY_EXTERNAL in the same +// position is green, a real family index is green, and the opposite +// direction (a FOREIGN acquire) is green; +// (3) it is recorded on a queue family other than graphics (family 0) or +// optical flow (family 5). Families 1 (transfer), 2 (compute), +// 3 (video decode) and 4 (video encode) all die. +// +// INDEPENDENT OF: the layout pair (GENERAL->GENERAL, i.e. no transition at +// all, dies), srcStageMask (VIDEO_ENCODE, TRANSFER, ALL_COMMANDS and NONE all +// die), whether the memory is externally allocated and exportable, whether an +// acquire preceded it, whether any video command was ever recorded, and the +// barrier API generation -- the v1 vkCmdPipelineBarrier spelling of the same +// release dies identically, so this cannot be worked around by re-expressing +// the barrier. +// +// The failure is Xid 32 (invalid/corrupted push-buffer stream) on the +// recording engine's channel, HCE_DBG0 00000124 then 00000800. +// VK_EXT_device_fault is supported and returns an EMPTY fault record -- +// consistent with a method-parse error rather than a memory fault. +// +// WHY THE LIBRARY CANNOT SIMPLY MOVE IT. A release must be recorded on a queue +// of its SOURCE family, and the only family that owns the Path-A input image +// is the encode family. So VkVideoEncoder::RecordVideoCodingCmd's release has +// nowhere legal to go, and the four possible responses -- drop the Path-A +// release, substitute VK_QUEUE_FAMILY_EXTERNAL (semantically wrong: it means +// another Vulkan instance, not a non-Vulkan agent), a two-hop +// encode->graphics->FOREIGN transfer on a second queue, or fix the driver -- +// are a design decision, not a bug fix. NOTHING HAS BEEN CHANGED HERE, and +// --foreign-residency still EXITS NON-ZERO BY DESIGN on the two DIRECT rows. +bool g_foreignResidency = false; + +// TRANSFER_SRC_OPTIMAL is legal only for an image created with +// VK_IMAGE_USAGE_TRANSFER_SRC_BIT (VUID-VkImageMemoryBarrier2-oldLayout-01208 +// and the newLayout equivalent). ARM_FILTER_RGBA is declared STORAGE| +// TRANSFER_DST and has no TRANSFER_SRC, so it keeps declaring GENERAL rather +// than have this mode manufacture a validation error that says nothing about +// the restore. ARM_FILTER_YCBCR -- the three I420 rows, which are the filter +// consumers this mode exists for -- does carry it. +VkImageLayout DeclaredLayoutFor(VkImageUsageFlags usage) +{ + if (g_declareTso && (usage & VK_IMAGE_USAGE_TRANSFER_SRC_BIT)) { + return VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL; + } + return VK_IMAGE_LAYOUT_GENERAL; +} + +// --------------------------------------------------------------------------- +// The four-quadrant primaries pattern. PURE primaries, because a U/V swap +// conserves the byte histogram and a black conversion has the right size -- +// neither a checksum nor a size check can catch either, and a mid-tone ramp +// makes both survivable. These exact values are the calibrated set. +// --------------------------------------------------------------------------- +struct RGB { uint8_t r, g, b; }; +const RGB kQuadTL = {253, 0, 0}; // red +const RGB kQuadTR = { 0, 253, 0}; // green +const RGB kQuadBL = { 0, 0, 252}; // blue +const RGB kQuadBR = {255, 255, 255}; // white + +RGB QuadAt(uint32_t x, uint32_t y) +{ + const bool right = (x >= kWidth / 2); + const bool bottom = (y >= kHeight / 2); + if (!bottom) return right ? kQuadTR : kQuadTL; + return right ? kQuadBR : kQuadBL; +} + +// LIMITED (studio) range, for ANY declared matrix. Writing the YCbCr arms +// with the same matrix the RGBA arm's filter is told to use is what lets ONE +// out-of-process comparison judge every row. +// +// CC-1 W8 GENERALISED THIS FROM A HARDCODED BT.709. The pattern WRITER still +// uses BT.709 -- the uploaded picture is BT.709 by construction and must not +// move -- but the quadrant GATE now compares in the Y'CbCr domain against the +// matrix the row DECLARED, so it needs the other two. Today every tracked row +// declares 1, so this is a generalisation with one live caller value; the +// point is that the gate stops being silently BT.709-only the day a row does +// not. +struct YUV { double y, cb, cr; }; + +struct KrKb { double kr, kb; }; +KrKb MatrixConstants(uint8_t matrix) +{ + switch (matrix) { + case 5: // BT.470BG + case 6: // SMPTE 170M -- the same matrix + return { 0.299, 0.114 }; + case 9: // BT.2020 NCL + case 10: // BT.2020 CL -- approximated as NCL, exactly as the filter does + return { 0.2627, 0.0593 }; + default: // 1 (BT.709), and any unnamed matrix, which resolves to BT.709 + return { 0.2126, 0.0722 }; + } +} + +YUV RgbToYuvLimited(const RGB& c, uint8_t matrix) +{ + const KrKb k = MatrixConstants(matrix); + const double Kr = k.kr, Kb = k.kb; + const double R = c.r, G = c.g, B = c.b; + const double Yf = Kr * R + (1.0 - Kr - Kb) * G + Kb * B; // 0..255 + YUV o; + o.y = 16.0 + 219.0 * Yf / 255.0; + o.cb = 128.0 + 224.0 * (B - Yf) / (2.0 * (1.0 - Kb) * 255.0); + o.cr = 128.0 + 224.0 * (R - Yf) / (2.0 * (1.0 - Kr) * 255.0); + return o; +} + +// The BT.709 spelling the pattern writer uses. Kept as a named wrapper rather +// than open-coded at its dozen call sites, so "what the uploader writes" stays +// one decision. +YUV RgbToYuv709Limited(const RGB& c) +{ + return RgbToYuvLimited(c, 1); +} + +uint8_t Clamp8(double v) +{ + if (v < 0.0) return 0; + if (v > 255.0) return 255; + return (uint8_t)(v + 0.5); +} + +uint16_t ClampN(double v, int bits) +{ + const double maxv = (double)((1 << bits) - 1); + if (v < 0.0) return 0; + if (v > maxv) return (uint16_t)maxv; + return (uint16_t)(v + 0.5); +} + +// --------------------------------------------------------------------------- +// Vulkan entry points, loaded off the LIBRARY's instance/device. +// --------------------------------------------------------------------------- +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkCreateBuffer CreateBuffer = nullptr; + PFN_vkDestroyBuffer DestroyBuffer = nullptr; + PFN_vkGetBufferMemoryRequirements GetBufferMemoryRequirements = nullptr; + PFN_vkBindBufferMemory BindBufferMemory = nullptr; + PFN_vkCreateCommandPool CreateCommandPool = nullptr; + PFN_vkDestroyCommandPool DestroyCommandPool = nullptr; + PFN_vkAllocateCommandBuffers AllocateCommandBuffers = nullptr; + PFN_vkBeginCommandBuffer BeginCommandBuffer = nullptr; + PFN_vkEndCommandBuffer EndCommandBuffer = nullptr; + PFN_vkCmdPipelineBarrier CmdPipelineBarrier = nullptr; + PFN_vkCmdCopyBufferToImage CmdCopyBufferToImage = nullptr; + PFN_vkQueueSubmit QueueSubmit = nullptr; + PFN_vkQueueWaitIdle QueueWaitIdle = nullptr; + PFN_vkGetDeviceQueue GetDeviceQueue = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = nullptr; + PFN_vkGetPhysicalDeviceProperties GetPhysicalDeviceProperties = nullptr; + PFN_vkGetPhysicalDeviceQueueFamilyProperties + GetPhysicalDeviceQueueFamilyProperties = nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) lib = dlopen("libvulkan.so", RTLD_NOW); + if (lib == nullptr) { std::printf(" ERROR: dlopen(libvulkan): %s\n", dlerror()); return false; } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) return false; + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) return false; +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { std::printf(" ERROR: missing vk" #name "\n"); return false; } + LOAD_DEV(CreateImage) LOAD_DEV(DestroyImage) LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) LOAD_DEV(FreeMemory) LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) LOAD_DEV(UnmapMemory) + LOAD_DEV(CreateBuffer) LOAD_DEV(DestroyBuffer) LOAD_DEV(GetBufferMemoryRequirements) + LOAD_DEV(BindBufferMemory) + LOAD_DEV(CreateCommandPool) LOAD_DEV(DestroyCommandPool) LOAD_DEV(AllocateCommandBuffers) + LOAD_DEV(BeginCommandBuffer) LOAD_DEV(EndCommandBuffer) + LOAD_DEV(CmdPipelineBarrier) LOAD_DEV(CmdCopyBufferToImage) + LOAD_DEV(QueueSubmit) LOAD_DEV(QueueWaitIdle) LOAD_DEV(GetDeviceQueue) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa(instance, "vkGetPhysicalDeviceMemoryProperties"); + fns->GetPhysicalDeviceProperties = + (PFN_vkGetPhysicalDeviceProperties)gipa(instance, "vkGetPhysicalDeviceProperties"); + fns->GetPhysicalDeviceQueueFamilyProperties = + (PFN_vkGetPhysicalDeviceQueueFamilyProperties)gipa(instance, "vkGetPhysicalDeviceQueueFamilyProperties"); + return fns->GetPhysicalDeviceMemoryProperties && fns->GetPhysicalDeviceProperties && + fns->GetPhysicalDeviceQueueFamilyProperties; +} + +// --nv12-companion. THE SHAPE CHROMIUM NOW BUILDS, and the one shape this +// matrix cannot express on its own: a session DECLARED in an RGBA format +// that is also handed NV12 descriptors. +// +// Chromium's VEA declares the session from its widest admissible input format +// rather than from the frame format its embedder asked for, because +// VideoEncodeAcceleratorAdapter pins that to NV12 before any frame exists and a +// session declared NV12 has no filter at all. The consequence is that an +// ordinary NV12 stream now runs on a session that HAS a preprocess filter -- +// and GetStagedInputSubmitType() returns COMPUTE for exactly such a +// session, so every staged NV12 frame moves off the encode/transfer family onto +// the compute family. That is a correct consequence of +// VUID-vkQueueSubmit2-commandBuffer-03874, and it is still a behaviour change +// on the busiest lane in the product, on a driver that has already been +// measured losing the device over a queue-family mistake on the input +// acquire/release path. +// +// So this mode exists to MEASURE two claims that were otherwise only reasoned +// about: +// 1. the library's registration gate really does admit BOTH the declared +// filter-input format and the recomputed semi-planar encode-source format +// on ONE session -- and routes them FILTER and DIRECT respectively; and +// 2. an NV12 frame on such a session encodes with zero spec violations under +// --validate, i.e. the compute-family move is clean. +// +// It runs ONLY on the ARM_FILTER_RGBA rows and only under the flag, so every +// standing bar of this suite is byte-identical without it. +bool g_nv12Companion = false; + +// --nv12-staged-companion. THE SHAPE CHROMIUM ACTUALLY PRODUCES, which is NOT +// the shape --nv12-companion measures, and the difference is the whole point. +// +// --nv12-companion declares its NV12 companion OPTIMAL + VIDEO_ENCODE_SRC, so +// the registration resolves DIRECT: the encode reads the caller's image in +// place and StageInputFrame is never entered. Chromium's shipping NV12 lanes +// resolve STAGED instead, for two independently measured reasons: +// +// * tier 2 (CPU dma-buf, ENABLED BY DEFAULT and the only zero-copy-adjacent +// tier X11 can reach): for a modifier-0 NV12 dma-buf, declaring +// VIDEO_ENCODE_SRC makes the library answer MODIFIER_UNSUPPORTED and the +// driver agree with VK_ERROR_FORMAT_NOT_SUPPORTED, so the VEA declares +// TRANSFER_SRC alone and encodeCapable resolves 0; and +// * tier 3 (OPAQUE_FD staging): a LINEAR host-written image, TRANSFER_SRC +// alone by construction. +// +// Either way the descriptor the library sees is LINEAR-or-non-encode-capable +// NV12 with TRANSFER_SRC usage, it routes STAGED, and it takes the plain-COPY +// arm inside StageInputFrame -- and that arm's submit family is read from +// GetStagedInputSubmitType(), which returns COMPUTE for any session carrying a +// preprocess filter OBJECT. A session declared RGBA always carries one. So +// widening the declared session format silently migrates the busiest lane in +// the product, including its FOREIGN queue-family RELEASE, from the encode +// family onto the compute family -- on a driver that loses the device on a +// FOREIGN release off the wrong engine. +// +// This mode is the A/B for that. It attaches the SAME LINEAR/TRANSFER_SRC NV12 +// companion to two sessions: +// arm A -- the NV12-declared row: no filter, so the copy stays on the +// encode (or transfer) family. The control. +// arm B -- each RGBA-declared row: filter present, so the copy moves to the +// compute family. The subject. +// and prints, per arm, the submit-type queue flags and the queue-family index +// the library ACTUALLY used, read back through +// VkVideoEncoderInputResidencyInfo rather than re-derived here. +bool g_stagedCompanion = false; + +// --companion-foreign. RESIDENCY_FOREIGN ON THE COMPANION ONLY, and the +// isolation is the point rather than a convenience. +// +// The staged copy arm's ReleaseImageToForeignQueue only runs for a frame whose +// residency resolves FOREIGN, so the queue-family question cannot be answered +// at all under the default RESIDENCY_LOCAL -- measured, foreign=0 local=12 on +// both arms of the A/B. The obvious lever, --foreign-residency, is unusable +// here: it also flips the PRIMARY registration, and on a DIRECT row that is a +// KNOWN VK_ERROR_DEVICE_LOST plus Xid 32 through Path A's own release (see +// g_foreignResidency). The process would die +// before the companion registered, and a device loss caused by a different +// barrier is not evidence about this one. +// +// So this flag drives the companion's residency alone. The primary stays LOCAL, +// Path A never releases, and the ONLY FOREIGN release in the run is the staged +// copy arm's release of the companion's LINEAR/TRANSFER_SRC image -- on family 4 +// in arm A and family 0 in arm B. That is exactly one variable. +bool g_companionForeign = false; + +enum Arm { ARM_DIRECT, ARM_FILTER_YCBCR, ARM_FILTER_RGBA }; +enum Group { G_8BIT = 0, G_10BIT = 1, G_12BIT = 2 }; + +// A row's COLOUR DECLARATION, when it is not the suite's pinned default. +// +// Every row before the HDR one declares BT.709 studio range, which is the +// pair the four-quadrant pattern is generated with, so the decoded picture +// can be compared against the primaries the harness wrote. An HDR row cannot +// take that gate -- the same YCbCr samples inverted through the BT.2020 +// matrix are different RGB -- so it carries its own colour AND its own +// verdict. See CheckHdrSignalling(). +struct RowColour { + uint8_t primaries; // ISO/IEC 23091-4 code points + uint8_t transfer; + uint8_t matrix; + VkBool32 fullRange; + bool hdr10; // also chain VkVideoEncoderHdrMetadataInfo +}; + +// BT.2020 primaries (9), PQ / SMPTE ST 2084 transfer (16), BT.2020 +// non-constant-luminance matrix (9), studio range -- HDR10's colour volume. +const RowColour kColourBt2020Pq = {9, 16, 9, VK_FALSE, true}; + +struct Row { + const char* name; + const char* shortName; + VkFormat format; + Arm arm; + Group group; + VkVideoCodecOperationFlagBitsKHR codec; + const char* ext; + // nullptr means the pinned BT.709 studio pair; every pre-HDR row leaves + // it unwritten and aggregate initialization supplies the null. + const RowColour* colour; +}; + +#define H264 VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "264" +#define H265 VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, "265" +// AV1 -- AND THE EXTENSION IS "obu", NOT "ivf". THAT IS A MEASUREMENT AND A +// RECORDED COVERAGE GAP, not a preference. +// +// VkVideoEncoderAV1 has TWO output arms and they emit DIFFERENT BYTES: +// +// FILE arm (disableFileOutput == VK_FALSE) -- +// VkVideoEncoderAV1::WriteBitstreamToFileOutput -> BuildFrameObuSequence +// -> FlushBatchedTemporalUnit, which muxes the 32-byte DKIF/'AV01' IVF +// file header and a 12-byte IVF frame header per temporal unit +// (VkVideoEncoderAV1.cpp:854-925, 1059-1079). +// +// CAPTURE arm (disableFileOutput == VK_TRUE) -- +// the in-memory branch of VkVideoEncoderAV1::WriteBitstreamToFile +// (VkVideoEncoderAV1.cpp:1026-1045). It prepends the two-byte Temporal +// Delimiter OBU {0x12,0x00}, appends the sequence-header OBU when the +// frame carries one, appends the frame OBU, and publishes THAT as the +// completion record. It never enters FlushBatchedTemporalUnit and writes +// no IVF header of any kind. +// +// THIS SUITE SETS disableFileOutput = VK_TRUE and drains through +// AcquireNextEncodedFrame, so it takes the CAPTURE arm -- and so does +// Chromium, for the same reason. What reaches the file this harness writes is +// a bare low-overhead OBU stream: TD-delimited temporal units, no container. +// Calling it ".ivf" would be false, and would imply the IVF muxer had run. +// +// SO THE GAP, PLAINLY: BuildFrameObuSequence and FlushBatchedTemporalUnit -- +// the DKIF/AV01 construction the AV1 half of the upstream refactor was +// accepted on a STATIC argument -- are NOT reachable from this harness, +// because they sit on the arm neither this suite nor the product takes. +// Covering them needs a file-output AV1 session, which is a different +// harness. This one must not claim them. +// +// The capture arm's shape needs no demuxer hint: +// * an SVT-AV1 stream muxed with `-f obu` begins 12 00 0a 0e ... -- the +// same {0x12,0x00} TD OBU this arm prepends, so the shapes agree; +// * ffprobe resolves it format_name=obu codec_name=av1 with NO -f flag, and +// STILL resolves it as obu when the identical bytes are renamed .ivf -- +// the content probe outranks the extension in both directions; +// * the decode gate's verbatim command line decodes it, and all twelve +// concatenated temporal units are read back (nb_read_frames=12). +// Hence NO `-f obu` below: it was tried and it is not needed. If a future +// ffmpeg regresses the obu probe, adding it for these rows is the one-line +// fix -- but adding it today would be cargo cult. +#define AV1 VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR, "obu" + +const Row kRows[] = { + {"NV12 G8_B8R8_2PLANE_420_UNORM", "nv12", VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ARM_DIRECT, G_8BIT, H264}, + {"P010 G10X6_B10X6R10X6_2PLANE_420", "p010", VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, ARM_DIRECT, G_10BIT, H265}, + {"P012 G12X4_B12X4R12X4_2PLANE_420", "p012", VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, ARM_FILTER_YCBCR, G_12BIT, H265}, + {"I420 G8_B8_R8_3PLANE_420_UNORM", "i420", VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, ARM_FILTER_YCBCR, G_8BIT, H264}, + {"I420-10 G10X6_B10X6_R10X6_3PLANE_420", "i420p10", VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, ARM_FILTER_YCBCR, G_10BIT, H265}, + {"I420-12 G12X4_B12X4_R12X4_3PLANE_420", "i420p12", VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, ARM_FILTER_YCBCR, G_12BIT, H265}, + {"RGBA8 R8G8B8A8_UNORM", "rgba8", VK_FORMAT_R8G8B8A8_UNORM, ARM_FILTER_RGBA, G_8BIT, H264}, + {"BGRA8 B8G8R8A8_UNORM", "bgra8", VK_FORMAT_B8G8R8A8_UNORM, ARM_FILTER_RGBA, G_8BIT, H264}, + {"ABGR8 A8B8G8R8_UNORM_PACK32", "abgr8", VK_FORMAT_A8B8G8R8_UNORM_PACK32, ARM_FILTER_RGBA, G_8BIT, H264}, + // ---- HEVC MAIN, 8-BIT. The one advertised codec/depth pair that no row + // reached, at either layer, and it was a hole in this table's shape rather + // than a decision: the rows are indexed by INPUT FORMAT and the codec is a + // dependent variable of bit depth, so H.264 took every 8-bit slot and + // every H.265 slot needed a format above 8 bits. Same VkFormat as the NV12 + // row above ON PURPOSE -- the input path is identical and already proven, + // so the ONLY variable this row adds is the codec, which is what makes it + // a clean read. It is also the only row that executes + // STD_VIDEO_H265_PROFILE_IDC_MAIN in CreateImage; that arm is dead + // otherwise. Distinct shortName because the output stem is + // "/." and "nv12" is taken. + {"NV12-265 G8_B8R8_2PLANE_420 (H.265 Main)", "nv12h265", VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ARM_DIRECT, G_8BIT, H265}, + // ---- AV1. THIS CODEC HAS NEVER EXECUTED ANYWHERE IN THIS TREE. ------- + // + // WHICH ROWS, AND WHY EXACTLY THESE FOUR. The input-format taxonomy + // (VkEncClassifyInput) is CODEC-INDEPENDENT -- it takes a VkFormat + // and nothing else -- so AV1 inherits the same nine-format ladder the + // rows above walk. What narrows it is the PROFILE, and the library states + // the constraint itself: AV1 Main is 8/10-bit 4:2:0 only, and 12-bit + // input against it is refused with a reason + // (vulkan_video_encoder_ext.cpp:1403-1428). + // + // * P012 and I420-12 are EXCLUDED. Against AV1 Main they can only ever + // report a PROFILE refusal, which is a fact about the enum and not + // about the encoder or the device. A row whose only possible outcome + // is a refusal measures the row. + // * NV12 (8-bit) and P010 (10-bit) are the whole ENCODABLE_DIRECT arm + // AV1 Main admits, and both are here. 10-bit is not redundant: the + // AV1 sequence header carries its own colour config, whose BitDepth + // and colour fields are built by AV1-only code (the av1BitDepth and + // av1ColorRange fields ON VkEncBoundConfigProbe exist precisely + // because nothing else reads them). Named by SYMBOL and not by line: + // the previous citation pointed at a range that no longer holds + // either the write site or the declarations, which is what a line + // number in a comment eventually does. + // * I420 is the ENCODABLE_VIA_FILTER *YCbCr* arm -- three planes into + // two, through the compute filter's per-plane storage-view read. + // * RGBA8 is the ENCODABLE_VIA_FILTER *RGBA* arm -- one plane, read + // through a single combined storage view and put through the RGB to + // Y'CbCr matrix. That is a different shader path from I420's, so one + // row cannot stand in for both. + // * BGRA8, ABGR8 and I420-10 are EXCLUDED as REDUNDANT, not as + // unsupported. What each adds over RGBA8 / I420 is which VkFormat the + // filter resolves its view from, and that is settled before the codec + // is consulted; it is already covered on the H.26x rows above. + // + // Both routing arms are therefore exercised under AV1 -- DIRECT, and both + // sub-arms of FILTER -- which is the whole point: the AV1 encoder, the + // seven ext-layer AV1 sites and VkVideoEncoderAV1.cpp have never seen a + // frame from any of them. + // ---- HDR10. THE ONLY ROW IN THIS SUITE THAT LOOKS AT A SEI. ---------- + // + // Same VkFormat, same DIRECT path and same codec as the P010 row above, + // ON PURPOSE: the input path is already proven there, so the only + // variables this row adds are the colour volume and the two SEI messages + // -- which is what makes its verdict readable. + // + // It takes the HDR gate instead of the four-quadrant gate, and that is a + // deliberate narrowing rather than a hole. The harness writes its pattern + // as BT.709 limited-range YCbCr; declaring BT.2020/PQ tells the decoder + // to invert a DIFFERENT matrix, so the RGB it produces is legitimately + // not the RGB the pattern started from. Comparing it against the BT.709 + // expectations would fail for a correct encode. What the HDR gate asserts + // instead is stronger where it matters here: the four VUI code points, a + // full frame actually decoded, four quadrant centres that are not all the + // same value (a flat picture is still caught), and both SEI payloads read + // back field by field by ffprobe. + {"HDR10-P010 BT.2020 PQ + ST2086 + MaxCLL", "hdr10", VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, ARM_DIRECT, G_10BIT, H265, &kColourBt2020Pq}, + {"AV1-NV12 G8_B8R8_2PLANE_420_UNORM", "av1nv12", VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ARM_DIRECT, G_8BIT, AV1}, + {"AV1-P010 G10X6_B10X6R10X6_2PLANE_420", "av1p010", VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, ARM_DIRECT, G_10BIT, AV1}, + {"AV1-I420 G8_B8_R8_3PLANE_420_UNORM", "av1i420", VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, ARM_FILTER_YCBCR, G_8BIT, AV1}, + {"AV1-RGBA8 R8G8B8A8_UNORM", "av1rgba", VK_FORMAT_R8G8B8A8_UNORM, ARM_FILTER_RGBA, G_8BIT, AV1}, + // ---- AV1 HDR10, BOTH DEPTHS. Added once CheckHdrSignalling() stopped + // string-matching H.265's ST 2086 units, which is what had kept these two + // out: AV1's metadata_hdr_mdcv carries the same physical values in + // different fixed point, so a correct AV1 HDR stream reported ten MISSING + // lines against the old table while the H.265 control passed in the same + // run. The gate parses and compares numerically now, so the same + // assertion serves both codecs. + // + // BOTH DEPTHS, and 8-bit is not redundant. AV1 Main admits 8 and 10 bits, + // the metadata OBU is emitted by the same code either way, and the + // BitDepth in the sequence header's colour config comes from a DIFFERENT + // field -- so an 8-bit HDR row is the one that would catch the metadata + // being made conditional on depth. + {"HDR10-AV1-NV12 BT.2020 PQ + ST2086 + MaxCLL", "hdr10av1nv12", VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ARM_DIRECT, G_8BIT, AV1, &kColourBt2020Pq}, + {"HDR10-AV1-P010 BT.2020 PQ + ST2086 + MaxCLL", "hdr10av1p010", VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, ARM_DIRECT, G_10BIT, AV1, &kColourBt2020Pq}, +}; +const size_t kNumRows = sizeof(kRows) / sizeof(kRows[0]); + +// The codec, as the string --codec takes. Used only by the row filter and the +// report; kept beside the macros above so the two cannot drift. +const char* CodecName(VkVideoCodecOperationFlagBitsKHR c) +{ + switch ((int)c) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: return "h264"; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: return "h265"; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: return "av1"; + default: return "?"; + } +} + +// --------------------------------------------------------------------------- +// DEVICE LIMITATION vs LIBRARY REFUSAL, and why that has to be decided in +// code rather than left to whoever reads the log. +// +// A row that gets no session prints one line and is not counted as a failure. +// That is right for P012 and I420-12 on hardware with no 12-bit encode +// profile -- and it is exactly the shape in which a NEW arm passes without +// ever running. The AV1 rows are the first arm in this file that the +// reference hardware (RTX A4000, GA104) cannot run AT ALL, so the distinction +// stops being cosmetic and becomes the difference between a gate and a +// decoration. +// +// WHAT THE A4000 ANSWERS FOR AV1, traced through the library rather than +// guessed. VK_KHR_video_encode_av1 is on the OPTIONAL device-extension list +// (vulkan_video_encoder_ext.cpp:1615), so HasAllDeviceExtensions() warns and +// DROPS it rather than failing device selection -- the row does not die +// there. It dies one step later, in VulkanDeviceContext::InitPhysicalDevice: +// the encode-queue clause admits a family only when +// `videoQueue.videoCodecOperations & requestVideoEncodeQueueOperations` +// (VulkanDeviceContext.cpp:857-859); the A4000's encode family advertises +// H.264 and H.265 and not AV1; no family matches, no physical device is +// selected, and the function returns VK_ERROR_FEATURE_NOT_PRESENT +// (VulkanDeviceContext.cpp:972). So -8 is the expected AV1 answer there. +// +// The codes below are the ones that mean THE DEVICE SAID NO. Everything else +// -- notably VK_ERROR_INITIALIZATION_FAILED, which is what every ext-layer +// refusal in BuildEncoderConfig returns -- is the LIBRARY saying no with a +// driver in hand, which is a verdict and not an environment fact. That is the +// sentence CMakeLists.txt already applies to nSession == 0, one level finer. +// +// ONE OF THESE CODES NOW ALSO ARRIVES FROM THE EXT LAYER, and the +// classification is still right. InitializeExt asks the device whether it +// encodes the profile the input derives BEFORE it creates a session, and +// answers VK_ERROR_FORMAT_NOT_SUPPORTED when it does not -- the same code +// VkEncQueryInputFormatSupport gives that verdict. It is the library RELAYING +// the device's answer with the format and its subsampling named, not a library +// rule, so it belongs on this list; what changed for P012 and I420-12 is only +// that the refusal now arrives before the session instead of out of +// vkGetPhysicalDeviceVideoCapabilitiesKHR after it. +bool IsDeviceLimitedInit(VkResult r) +{ + switch ((int)r) { + case VK_ERROR_FEATURE_NOT_PRESENT: // -8 + case VK_ERROR_EXTENSION_NOT_PRESENT: // -7 + case VK_ERROR_FORMAT_NOT_SUPPORTED: // -11 + case VK_ERROR_IMAGE_USAGE_NOT_SUPPORTED_KHR: // -1000023000 + case VK_ERROR_VIDEO_PICTURE_LAYOUT_NOT_SUPPORTED_KHR: // -1000023001 + case VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR: // -1000023002 + case VK_ERROR_VIDEO_PROFILE_FORMAT_NOT_SUPPORTED_KHR: // -1000023003 + case VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR: // -1000023004 + case VK_ERROR_VIDEO_STD_VERSION_NOT_SUPPORTED_KHR: // -1000023005 + return true; + default: + return false; + } +} + +// --------------------------------------------------------------------------- +// THE AV1 ANTI-SILENT-PASS PROBE. This is what makes "device-limited" a +// finding rather than an assumption. +// +// "No session, therefore this device cannot do AV1" is not an inference. It +// is the assumption that turns a new arm into a test that cannot fail: if the +// AV1 path is broken on hardware that HAS AV1, the row still reports no +// session, still is not counted, and the suite still exits 0. That is the +// thirteenth cannot-fail test, pre-built, and this function is the reason it +// is not one. +// +// So the claim is CHECKED against the device, independently of the library: a +// throwaway VkInstance of this test's own, then +// vkEnumerateDeviceExtensionProperties on each physical device, looking for +// VK_KHR_video_encode_av1. The extension STRING is the primary signal on +// purpose -- it is a flat list with no pNext chaining, so it cannot come back +// silently empty the way a feature struct an older driver does not recognise +// can, which is exactly the trap +// VkPhysicalDeviceVideoEncodeAV1FeaturesKHR would set here. +// +// The three answers, and what each licenses: +// present (1) -- a no-session AV1 row is a FAILURE. The device advertises +// the extension, so something between that and the session +// is broken; surfacing that is what this arm is for. +// absent (0) -- a no-session AV1 row carrying a device-limitation code is +// DEVICE-LIMITED, and is not a failure. This is the A4000. +// unknown(-1) -- no instance could be made. Reported UNVERIFIED and +// counted; never a clean pass. +// +// The over-approximation points the safe way ON PURPOSE: this asks whether +// ANY physical device advertises AV1 encode, while the session selects one. +// On a multi-GPU host that can only convert a silent pass into a loud +// failure, never the reverse. +int g_av1Probe = -1; // -1 unknown, 0 absent, 1 present +std::string g_av1ProbeDetail; + +// VK_DRIVER_FILES / VK_ICD_FILENAMES scope the Vulkan loader for the WHOLE +// process. When either is set, EVERY instance in this process -- the probe's +// and the library's alike -- is offered a SUBSET of the machine's drivers, so +// a short device list is the environment speaking, not the probe failing. +// Returns nullptr when the loader is unscoped. +const char* Av1ProbeIcdScope() +{ + const char* s = std::getenv("VK_DRIVER_FILES"); + if ((s == nullptr) || (s[0] == '\0')) s = std::getenv("VK_ICD_FILENAMES"); + return ((s != nullptr) && (s[0] != '\0')) ? s : nullptr; +} + +void ProbeAv1EncodeSupport() +{ + if (!g_av1ProbeDetail.empty()) return; // probed at most once per run + g_av1ProbeDetail = "the probe did not complete"; + + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) lib = dlopen("libvulkan.so", RTLD_NOW); + if (lib == nullptr) { g_av1ProbeDetail = "dlopen(libvulkan) failed"; return; } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) { g_av1ProbeDetail = "no vkGetInstanceProcAddr"; return; } + auto createInstance = (PFN_vkCreateInstance)gipa(nullptr, "vkCreateInstance"); + if (createInstance == nullptr) { g_av1ProbeDetail = "no vkCreateInstance"; return; } + + VkApplicationInfo app{VK_STRUCTURE_TYPE_APPLICATION_INFO}; + app.pApplicationName = "encoder-ext-format-encode av1 probe"; + app.apiVersion = VK_API_VERSION_1_3; + VkInstanceCreateInfo ici{VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO}; + ici.pApplicationInfo = &app; + VkInstance inst = VK_NULL_HANDLE; + const VkResult ir = createInstance(&ici, nullptr, &inst); + if ((ir != VK_SUCCESS) || (inst == VK_NULL_HANDLE)) { + char m[160]; + std::snprintf(m, sizeof(m), + "vkCreateInstance -> %d, so no device could be asked", + (int)ir); + g_av1ProbeDetail = m; + return; + } + + auto destroyInstance = (PFN_vkDestroyInstance)gipa(inst, "vkDestroyInstance"); + auto enumPhys = (PFN_vkEnumeratePhysicalDevices)gipa(inst, "vkEnumeratePhysicalDevices"); + auto enumDevExt = (PFN_vkEnumerateDeviceExtensionProperties)gipa(inst, "vkEnumerateDeviceExtensionProperties"); + auto getProps = (PFN_vkGetPhysicalDeviceProperties)gipa(inst, "vkGetPhysicalDeviceProperties"); + if ((enumPhys == nullptr) || (enumDevExt == nullptr)) { + g_av1ProbeDetail = "instance entry points missing"; + if (destroyInstance != nullptr) destroyInstance(inst, nullptr); + return; + } + + uint32_t nDev = 0; + enumPhys(inst, &nDev, nullptr); + std::vector devs(nDev); + if (nDev != 0) enumPhys(inst, &nDev, devs.data()); + if (nDev == 0) { + { + const char* scope = Av1ProbeIcdScope(); + char z[512]; + std::snprintf(z, sizeof(z), + "the instance enumerated no physical device%s%s%s", + (scope != nullptr) ? " (loader scoped by VK_DRIVER_FILES/VK_ICD_FILENAMES to " : "", + (scope != nullptr) ? scope : "", + (scope != nullptr) ? ")" : ""); + g_av1ProbeDetail = z; + } + if (destroyInstance != nullptr) destroyInstance(inst, nullptr); + return; + } + + std::string detail; + int found = 0; + for (uint32_t d = 0; d < nDev; d++) { + uint32_t nExt = 0; + enumDevExt(devs[d], nullptr, &nExt, nullptr); + std::vector exts(nExt); + if (nExt != 0) enumDevExt(devs[d], nullptr, &nExt, exts.data()); + bool hasAv1 = false; + for (uint32_t e = 0; e < nExt; e++) { + if (std::strcmp(exts[e].extensionName, + VK_KHR_VIDEO_ENCODE_AV1_EXTENSION_NAME) == 0) { + hasAv1 = true; + break; + } + } + if (hasAv1) found = 1; + VkPhysicalDeviceProperties p{}; + if (getProps != nullptr) getProps(devs[d], &p); + char one[320]; + std::snprintf(one, sizeof(one), "%s%s %s VK_KHR_video_encode_av1", + detail.empty() ? "" : "; ", + (getProps != nullptr) ? p.deviceName : "device", + hasAv1 ? "HAS" : "does NOT enumerate"); + detail += one; + } + // THE COUNT IS PART OF THE VERDICT, not decoration. A verdict that names + // one device is ambiguous between "the probe truncated its report" and + // "this process's loader was only ever offered one driver", and that + // ambiguity has already cost a misdiagnosis: a CORRECT llvmpipe-only + // verdict was read as a probe defect and sent someone looking for a bug + // in vkCreateInstance parameters that was never there. Stating the count, + // and stating when the loader was scoped, collapses the ambiguity at the + // point of reading instead of leaving it for a bisect. + { + const char* scope = Av1ProbeIcdScope(); + char head[640]; + std::snprintf(head, sizeof(head), "%u device%s enumerated%s%s%s: ", + nDev, (nDev == 1) ? "" : "s", + (scope != nullptr) ? " (loader scoped by VK_DRIVER_FILES/VK_ICD_FILENAMES to " : "", + (scope != nullptr) ? scope : "", + (scope != nullptr) ? ")" : ""); + detail = std::string(head) + detail; + } + g_av1Probe = found; + g_av1ProbeDetail = detail; + if (destroyInstance != nullptr) destroyInstance(inst, nullptr); +} + +const char* PathName(VkVideoEncoderExternalInputPath p) +{ + switch (p) { + case VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT: return "DIRECT"; + case VK_VIDEO_EXTERNAL_INPUT_PATH_STAGED: return "STAGED"; + case VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER: return "FILTER"; + default: return "?"; + } +} + +// --------------------------------------------------------------------------- +// Pattern writers. One per format family; each fills a tightly packed +// staging buffer laid out plane-after-plane and reports the plane offsets. +// --------------------------------------------------------------------------- +struct PlaneCopy { VkDeviceSize offset; uint32_t w, h; VkImageAspectFlagBits aspect; }; + +size_t BuildPattern(VkFormat fmt, std::vector* buf, + std::vector* planes) +{ + buf->clear(); + planes->clear(); + const uint32_t cw = kWidth / 2, ch = kHeight / 2; + + switch (fmt) { + case VK_FORMAT_R8G8B8A8_UNORM: + case VK_FORMAT_A8B8G8R8_UNORM_PACK32: { + // A8B8G8R8_UNORM_PACK32 is a uint32 A<<24|B<<16|G<<8|R, which on + // a little-endian host is byte order R,G,B,A -- BYTE-IDENTICAL to + // R8G8B8A8_UNORM. Written once for both, deliberately: if these + // two rows ever decode differently the difference is in the view + // format the filter resolves, not in what was uploaded. + buf->resize((size_t)kWidth * kHeight * 4); + for (uint32_t y = 0; y < kHeight; y++) + for (uint32_t x = 0; x < kWidth; x++) { + const RGB c = QuadAt(x, y); + uint8_t* p = buf->data() + ((size_t)y * kWidth + x) * 4; + p[0] = c.r; p[1] = c.g; p[2] = c.b; p[3] = 255; + } + planes->push_back({0, kWidth, kHeight, VK_IMAGE_ASPECT_COLOR_BIT}); + break; + } + case VK_FORMAT_B8G8R8A8_UNORM: { + buf->resize((size_t)kWidth * kHeight * 4); + for (uint32_t y = 0; y < kHeight; y++) + for (uint32_t x = 0; x < kWidth; x++) { + const RGB c = QuadAt(x, y); + uint8_t* p = buf->data() + ((size_t)y * kWidth + x) * 4; + p[0] = c.b; p[1] = c.g; p[2] = c.r; p[3] = 255; + } + planes->push_back({0, kWidth, kHeight, VK_IMAGE_ASPECT_COLOR_BIT}); + break; + } + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: { // NV12 + const size_t ySize = (size_t)kWidth * kHeight; + buf->resize(ySize + (size_t)cw * ch * 2); + for (uint32_t y = 0; y < kHeight; y++) + for (uint32_t x = 0; x < kWidth; x++) + (*buf)[(size_t)y * kWidth + x] = Clamp8(RgbToYuv709Limited(QuadAt(x, y)).y); + for (uint32_t y = 0; y < ch; y++) + for (uint32_t x = 0; x < cw; x++) { + const YUV c = RgbToYuv709Limited(QuadAt(x * 2, y * 2)); + uint8_t* p = buf->data() + ySize + ((size_t)y * cw + x) * 2; + p[0] = Clamp8(c.cb); p[1] = Clamp8(c.cr); + } + planes->push_back({0, kWidth, kHeight, VK_IMAGE_ASPECT_PLANE_0_BIT}); + planes->push_back({(VkDeviceSize)ySize, cw, ch, VK_IMAGE_ASPECT_PLANE_1_BIT}); + break; + } + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16: // P010 + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16: { // P012 + const int bits = (fmt == VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16) ? 10 : 12; + const int shift = g_rawBits ? 0 : (16 - bits); // value sits in the HIGH bits + const double scale = (double)((1 << bits) - 1) / 255.0; + const size_t ySize = (size_t)kWidth * kHeight * 2; + buf->resize(ySize + (size_t)cw * ch * 2 * 2); + uint16_t* yp = (uint16_t*)buf->data(); + for (uint32_t y = 0; y < kHeight; y++) + for (uint32_t x = 0; x < kWidth; x++) + yp[(size_t)y * kWidth + x] = + (uint16_t)(ClampN(RgbToYuv709Limited(QuadAt(x, y)).y * scale, bits) << shift); + uint16_t* cp = (uint16_t*)(buf->data() + ySize); + for (uint32_t y = 0; y < ch; y++) + for (uint32_t x = 0; x < cw; x++) { + const YUV c = RgbToYuv709Limited(QuadAt(x * 2, y * 2)); + cp[((size_t)y * cw + x) * 2 + 0] = (uint16_t)(ClampN(c.cb * scale, bits) << shift); + cp[((size_t)y * cw + x) * 2 + 1] = (uint16_t)(ClampN(c.cr * scale, bits) << shift); + } + planes->push_back({0, kWidth, kHeight, VK_IMAGE_ASPECT_PLANE_0_BIT}); + planes->push_back({(VkDeviceSize)ySize, cw, ch, VK_IMAGE_ASPECT_PLANE_1_BIT}); + break; + } + case VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM: { // I420 + const size_t ySize = (size_t)kWidth * kHeight, cSize = (size_t)cw * ch; + buf->resize(ySize + cSize * 2); + for (uint32_t y = 0; y < kHeight; y++) + for (uint32_t x = 0; x < kWidth; x++) + (*buf)[(size_t)y * kWidth + x] = Clamp8(RgbToYuv709Limited(QuadAt(x, y)).y); + for (uint32_t y = 0; y < ch; y++) + for (uint32_t x = 0; x < cw; x++) { + const YUV c = RgbToYuv709Limited(QuadAt(x * 2, y * 2)); + (*buf)[ySize + (size_t)y * cw + x] = Clamp8(c.cb); + (*buf)[ySize + cSize + (size_t)y * cw + x] = Clamp8(c.cr); + } + planes->push_back({0, kWidth, kHeight, VK_IMAGE_ASPECT_PLANE_0_BIT}); + planes->push_back({(VkDeviceSize)ySize, cw, ch, VK_IMAGE_ASPECT_PLANE_1_BIT}); + planes->push_back({(VkDeviceSize)(ySize + cSize), cw, ch, VK_IMAGE_ASPECT_PLANE_2_BIT}); + break; + } + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16: + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16: { + const int bits = (fmt == VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16) ? 10 : 12; + const int shift = g_rawBits ? 0 : (16 - bits); + const double scale = (double)((1 << bits) - 1) / 255.0; + const size_t ySize = (size_t)kWidth * kHeight * 2, cSize = (size_t)cw * ch * 2; + buf->resize(ySize + cSize * 2); + uint16_t* yp = (uint16_t*)buf->data(); + for (uint32_t y = 0; y < kHeight; y++) + for (uint32_t x = 0; x < kWidth; x++) + yp[(size_t)y * kWidth + x] = + (uint16_t)(ClampN(RgbToYuv709Limited(QuadAt(x, y)).y * scale, bits) << shift); + uint16_t* bp = (uint16_t*)(buf->data() + ySize); + uint16_t* rp = (uint16_t*)(buf->data() + ySize + cSize); + for (uint32_t y = 0; y < ch; y++) + for (uint32_t x = 0; x < cw; x++) { + const YUV c = RgbToYuv709Limited(QuadAt(x * 2, y * 2)); + bp[(size_t)y * cw + x] = (uint16_t)(ClampN(c.cb * scale, bits) << shift); + rp[(size_t)y * cw + x] = (uint16_t)(ClampN(c.cr * scale, bits) << shift); + } + planes->push_back({0, kWidth, kHeight, VK_IMAGE_ASPECT_PLANE_0_BIT}); + planes->push_back({(VkDeviceSize)ySize, cw, ch, VK_IMAGE_ASPECT_PLANE_1_BIT}); + planes->push_back({(VkDeviceSize)(ySize + cSize), cw, ch, VK_IMAGE_ASPECT_PLANE_2_BIT}); + break; + } + default: + return 0; + } + return buf->size(); +} + +// The ext header's own enumerator spellings, so a log line greps straight +// into VkVideoEncoderImportContentState. +const char* ContentStateName(uint32_t state) +{ + switch ((VkVideoEncoderImportContentState)state) { + case VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED: return "NOT_EVALUATED"; + case VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_APPLICABLE: return "NOT_APPLICABLE"; + case VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED: return "ARMED"; + case VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_CLEAN: return "CLEAN"; + case VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_CHROMA: return "DAMAGED_CHROMA"; + case VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_ALL: return "DAMAGED_ALL"; + } + return "unknown"; +} + +struct Decl { VkImageUsageFlags usage; VkImageCreateFlags flags; VkImageTiling tiling; bool profileList; }; + +Decl DeclFor(Arm arm) +{ + Decl d = {}; + d.tiling = VK_IMAGE_TILING_OPTIMAL; + switch (arm) { + case ARM_DIRECT: + // TRANSFER_DST is added to every arm purely so the pattern can be + // uploaded. It is a superset of the declaration the routing test + // validated; each row prints its resolved path so a routing change + // caused by this addition would be visible rather than assumed. + d.usage = VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT; + d.profileList = true; + break; + case ARM_FILTER_YCBCR: + d.usage = VK_IMAGE_USAGE_STORAGE_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT | + VK_IMAGE_USAGE_TRANSFER_DST_BIT; + d.flags = VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT | VK_IMAGE_CREATE_EXTENDED_USAGE_BIT; + break; + case ARM_FILTER_RGBA: + // The filter's single-plane arm binds this image's ONE combined + // view as a VK_DESCRIPTOR_TYPE_STORAGE_IMAGE, so STORAGE is the + // usage it needs -- and no create flags. + d.usage = VK_IMAGE_USAGE_STORAGE_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT; + break; + } + return d; +} + +// The NV12 companion's declaration. Deliberately NOT an Arm case: an Arm +// describes how a SESSION is configured, and this describes a registration +// attached to somebody else's session. +// +// LINEAR + TRANSFER_SRC and NO video profile list, which together are what +// make slot.encodeCapable resolve 0 (it requires OPTIMAL tiling AND +// VIDEO_ENCODE_SRC) and therefore what make the registration route STAGED. +// TRANSFER_DST is present only so UploadPattern can write the pattern, exactly +// as it is on every other arm. +// +// Consequence worth naming, because it is the reason the arm-B measurement is +// not simply a repeat of the Path-A device loss: this image carries NO +// VIDEO_ENCODE_SRC and no VIDEO_ENCODE_DPB usage. The driver defect is gated on +// that usage (`rel --novideo` SURVIVED in the reproducer's matrix), so the +// prediction going in is that the compute-family FOREIGN release here is +// harmless. A prediction is not a measurement, which is why this mode exists. +Decl StagedCompanionDecl() +{ + Decl d = {}; + d.tiling = VK_IMAGE_TILING_LINEAR; + d.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT; + d.flags = 0; + d.profileList = false; + return d; +} + +struct Img { VkImage image = VK_NULL_HANDLE; VkDeviceMemory memory = VK_NULL_HANDLE; }; + +bool CreateImage(const DeviceFns& fns, VkPhysicalDevice phys, VkDevice device, + VkFormat format, const Decl& d, + VkVideoCodecOperationFlagBitsKHR codec, + VkVideoComponentBitDepthFlagBitsKHR depth, Img* out) +{ + VkVideoEncodeH264ProfileInfoKHR h264Profile{VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_PROFILE_INFO_KHR}; + h264Profile.stdProfileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH; + VkVideoEncodeH265ProfileInfoKHR h265Profile{VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_PROFILE_INFO_KHR}; + h265Profile.stdProfileIdc = (depth == VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR) + ? STD_VIDEO_H265_PROFILE_IDC_MAIN : STD_VIDEO_H265_PROFILE_IDC_MAIN_10; + // AV1's profile struct is NOT shaped like the H.26x pair, and the + // difference is not cosmetic: the field is `stdProfile`, of type + // StdVideoAV1Profile, and there is no `stdProfileIdc`. Matched to how the + // LIBRARY builds it rather than invented -- VkVideoCoreProfile's AV1 arm + // populates exactly {sType = ..._VIDEO_ENCODE_AV1_PROFILE_INFO_KHR, + // stdProfile = STD_VIDEO_AV1_PROFILE_MAIN} for a default AV1 encode + // profile (VkVideoCoreProfile.h:166-185 and 288-294). + // + // MAIN serves BOTH AV1 rows, and that is a decision, not a default: AV1 + // Main is 8/10-bit 4:2:0, which is exactly the two depths the AV1 rows + // use -- unlike H.265 above, which has to pick MAIN vs MAIN_10. The + // profile named in this image's VkVideoProfileListInfoKHR must be the one + // the session will use, or vkCreateImage is being asked about an image + // the encoder will not be able to read. + VkVideoEncodeAV1ProfileInfoKHR av1Profile{VK_STRUCTURE_TYPE_VIDEO_ENCODE_AV1_PROFILE_INFO_KHR}; + av1Profile.stdProfile = STD_VIDEO_AV1_PROFILE_MAIN; + VkVideoProfileInfoKHR profile{VK_STRUCTURE_TYPE_VIDEO_PROFILE_INFO_KHR}; + // A SWITCH, NOT THE TWO-CODEC TERNARY THIS REPLACES. That ternary + // answered "H.265" for every codec that was not H.264, so an AV1 row + // would have been handed a VkVideoEncodeH265ProfileInfoKHR chained under + // an AV1 videoCodecOperation -- a mismatch the driver would either refuse + // for the wrong reason or, worse, accept. + switch (codec) { + case VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR: + profile.pNext = (void*)&h264Profile; break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR: + profile.pNext = (void*)&h265Profile; break; + case VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR: + profile.pNext = (void*)&av1Profile; break; + default: + // Unreachable from kRows, and a hard stop rather than a silent + // fallthrough for exactly the reason above. + std::printf(" ERROR: no profile shape for codec 0x%x\n", + (unsigned)codec); + return false; + } + profile.videoCodecOperation = codec; + profile.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + profile.lumaBitDepth = depth; profile.chromaBitDepth = depth; + VkVideoProfileListInfoKHR profileList{VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR}; + profileList.profileCount = 1; profileList.pProfiles = &profile; + + VkFormat viewFormats[4] = {format, VK_FORMAT_UNDEFINED, VK_FORMAT_UNDEFINED, VK_FORMAT_UNDEFINED}; + uint32_t viewFormatCount = 1; + switch (format) { + case VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM: + viewFormats[viewFormatCount++] = VK_FORMAT_R8_UNORM; break; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16: + viewFormats[viewFormatCount++] = VK_FORMAT_R10X6_UNORM_PACK16; + viewFormats[viewFormatCount++] = VK_FORMAT_R16_UNORM; break; + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16: + viewFormats[viewFormatCount++] = VK_FORMAT_R12X4_UNORM_PACK16; + viewFormats[viewFormatCount++] = VK_FORMAT_R16_UNORM; break; + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: + viewFormats[viewFormatCount++] = VK_FORMAT_R8_UNORM; + viewFormats[viewFormatCount++] = VK_FORMAT_R8G8_UNORM; break; + default: break; + } + VkImageFormatListCreateInfo listInfo{VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + listInfo.viewFormatCount = viewFormatCount; listInfo.pViewFormats = viewFormats; + + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + const void* chain = nullptr; + if (d.profileList) chain = &profileList; + if ((d.flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) { listInfo.pNext = chain; chain = &listInfo; } + ci.pNext = chain; ci.flags = d.flags; ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = format; ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; ci.arrayLayers = 1; ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = d.tiling; ci.usage = d.usage; ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) return false; + + VkMemoryRequirements req{}; fns.GetImageMemoryRequirements(device, out->image, &req); + VkPhysicalDeviceMemoryProperties memProps{}; fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT) != 0)) { typeIndex = i; break; } + if (typeIndex == UINT32_MAX) { fns.DestroyImage(device, out->image, nullptr); out->image = VK_NULL_HANDLE; return false; } + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + fns.DestroyImage(device, out->image, nullptr); out->image = VK_NULL_HANDLE; return false; + } + return fns.BindImageMemory(device, out->image, out->memory, 0) == VK_SUCCESS; +} + +void DestroyImg(const DeviceFns& fns, VkDevice device, Img* img) +{ + if (img->image != VK_NULL_HANDLE) { fns.DestroyImage(device, img->image, nullptr); img->image = VK_NULL_HANDLE; } + if (img->memory != VK_NULL_HANDLE) { fns.FreeMemory(device, img->memory, nullptr); img->memory = VK_NULL_HANDLE; } +} + +// Upload the pattern. Runs on the session's VIDEO ENCODE queue family -- +// which on this vendor advertises TRANSFER, and is a family the library +// demonstrably created a queue on (vkGetDeviceQueue on any other family +// would be undefined behavior). It is issued ONCE, after registration and +// BEFORE the first submit, then waited to idle: at that moment no frame is +// in flight, so the library's assembly workers have nothing to submit and +// the queue is not concurrently used. +bool UploadPattern(VkImageUsageFlags declaredUsage, + const DeviceFns& fns, VkPhysicalDevice phys, VkDevice device, + uint32_t queueFamily, VkImage image, VkFormat format, + std::string* err) +{ + std::vector data; std::vector planes; + const size_t size = BuildPattern(format, &data, &planes); + if (size == 0) { *err = "no pattern writer for this format"; return false; } + + VkBuffer buf = VK_NULL_HANDLE; VkDeviceMemory mem = VK_NULL_HANDLE; + VkBufferCreateInfo bci{VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO}; + bci.size = size; bci.usage = VK_BUFFER_USAGE_TRANSFER_SRC_BIT; bci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + if (fns.CreateBuffer(device, &bci, nullptr, &buf) != VK_SUCCESS) { *err = "vkCreateBuffer failed"; return false; } + VkMemoryRequirements req{}; fns.GetBufferMemoryRequirements(device, buf, &req); + VkPhysicalDeviceMemoryProperties mp{}; fns.GetPhysicalDeviceMemoryProperties(phys, &mp); + const VkMemoryPropertyFlags want = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; + uint32_t ti = UINT32_MAX; + for (uint32_t i = 0; i < mp.memoryTypeCount; i++) + if (((req.memoryTypeBits & (1u << i)) != 0) && ((mp.memoryTypes[i].propertyFlags & want) == want)) { ti = i; break; } + if (ti == UINT32_MAX) { fns.DestroyBuffer(device, buf, nullptr); *err = "no host-visible memory type"; return false; } + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; ai.memoryTypeIndex = ti; + if (fns.AllocateMemory(device, &ai, nullptr, &mem) != VK_SUCCESS) { fns.DestroyBuffer(device, buf, nullptr); *err = "vkAllocateMemory failed"; return false; } + fns.BindBufferMemory(device, buf, mem, 0); + void* mapped = nullptr; + if (fns.MapMemory(device, mem, 0, VK_WHOLE_SIZE, 0, &mapped) != VK_SUCCESS) { *err = "vkMapMemory failed"; return false; } + std::memcpy(mapped, data.data(), size); + fns.UnmapMemory(device, mem); + + VkCommandPool pool = VK_NULL_HANDLE; + VkCommandPoolCreateInfo pci{VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO}; + pci.queueFamilyIndex = queueFamily; + if (fns.CreateCommandPool(device, &pci, nullptr, &pool) != VK_SUCCESS) { *err = "vkCreateCommandPool failed"; return false; } + VkCommandBuffer cmd = VK_NULL_HANDLE; + VkCommandBufferAllocateInfo cbai{VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO}; + cbai.commandPool = pool; cbai.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cbai.commandBufferCount = 1; + if (fns.AllocateCommandBuffers(device, &cbai, &cmd) != VK_SUCCESS) { *err = "vkAllocateCommandBuffers failed"; return false; } + VkCommandBufferBeginInfo bi{VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + bi.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + fns.BeginCommandBuffer(cmd, &bi); + + VkImageMemoryBarrier toDst{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + toDst.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; toDst.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + toDst.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; toDst.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + toDst.image = image; toDst.srcAccessMask = 0; toDst.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + toDst.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + fns.CmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + 0, 0, nullptr, 0, nullptr, 1, &toDst); + + std::vector regions; + for (const PlaneCopy& p : planes) { + VkBufferImageCopy r{}; + r.bufferOffset = p.offset; r.bufferRowLength = 0; r.bufferImageHeight = 0; + r.imageSubresource.aspectMask = p.aspect; + r.imageSubresource.mipLevel = 0; r.imageSubresource.baseArrayLayer = 0; r.imageSubresource.layerCount = 1; + r.imageOffset = {0, 0, 0}; r.imageExtent = {p.w, p.h, 1}; + regions.push_back(r); + } + fns.CmdCopyBufferToImage(cmd, buf, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, + (uint32_t)regions.size(), regions.data()); + + // To the layout the descriptor declares as defaultLayout -- GENERAL, or + // TRANSFER_SRC_OPTIMAL under --declare-tso. The producer must actually + // LEAVE the image where the registration says it is, or frame 1's acquire + // names a layout the image was never in + // (VUID-VkImageMemoryBarrier2-oldLayout-01197) and the mode would be + // testing the test rather than the library. + VkImageMemoryBarrier toGen{VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER}; + toGen.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL; + toGen.newLayout = DeclaredLayoutFor(declaredUsage); + toGen.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; toGen.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + toGen.image = image; toGen.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT; + // VK_ACCESS_MEMORY_READ_BIT, not SHADER_READ|TRANSFER_READ, AND THAT IS A + // FIX, NOT A TIDY-UP. This barrier is recorded on |encFamily| -- see the + // sole UploadPattern call in RunRow, which passes the ENCODE family for + // every row, DIRECT and FILTER alike. On this hardware family 4 reports + // TRANSFER|SPARSE|VIDEO_ENCODE, so the expansion of + // VK_PIPELINE_STAGE_ALL_COMMANDS_BIT below contains no stage that supports + // VK_ACCESS_SHADER_READ_BIT, and the pair is invalid under + // VUID-vkCmdPipelineBarrier-pImageMemoryBarriers-02820. + // + // This is the whole of this suite's validation baseline: an uncorrected + // line makes `--validate` report one "The Vulkan spec states" message per + // session on the healthy default (RESIDENCY_LOCAL) run, all of them this + // VUID. Nothing in the LIBRARY + // was emitting them: the message names vkCmdPipelineBarrier (the v1 entry + // point) and VkVideoEncoder::TransitionImageLayout records exclusively + // through vkCmdPipelineBarrier2. It was this helper all along. + // + // WHY IT WAS INVISIBLE. This suite's stated bar was + // `sessions=8 encoded=8 failures=0` at the time -- an earlier row set, and + // with no validation count in it -- so seven real messages sat under a + // green bar. The bar now includes "0 spec-states under --validate"; see the file + // header for the current line in full. + // + // MEMORY_READ is a strict superset of the two flags it replaces and is + // supported by every pipeline stage, so it is legal on every family this + // helper could be asked to record on -- it does not merely move the + // problem to whichever family a future row picks. + toGen.dstAccessMask = VK_ACCESS_MEMORY_READ_BIT; + toGen.subresourceRange = {VK_IMAGE_ASPECT_COLOR_BIT, 0, 1, 0, 1}; + fns.CmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, + 0, 0, nullptr, 0, nullptr, 1, &toGen); + fns.EndCommandBuffer(cmd); + + VkQueue queue = VK_NULL_HANDLE; + fns.GetDeviceQueue(device, queueFamily, 0, &queue); + if (queue == VK_NULL_HANDLE) { *err = "vkGetDeviceQueue returned NULL"; return false; } + VkSubmitInfo si{VK_STRUCTURE_TYPE_SUBMIT_INFO}; + si.commandBufferCount = 1; si.pCommandBuffers = &cmd; + const VkResult sr = fns.QueueSubmit(queue, 1, &si, VK_NULL_HANDLE); + if (sr != VK_SUCCESS) { *err = "vkQueueSubmit failed: " + std::to_string((int)sr); return false; } + const VkResult wr = fns.QueueWaitIdle(queue); + if (wr != VK_SUCCESS) { *err = "vkQueueWaitIdle failed: " + std::to_string((int)wr); return false; } + + fns.DestroyCommandPool(device, pool, nullptr); + fns.DestroyBuffer(device, buf, nullptr); + fns.FreeMemory(device, mem, nullptr); + return true; +} + +struct EncResult { + bool sessionInit = false; + // WHY the session failed, not just that it did. The skip decision below + // turns on this: a device that is absent and a library that refused are + // both "no session", and only the first is an environment condition. + VkResult initResult = VK_ERROR_INITIALIZATION_FAILED; + bool fileOpenFailed = false; + bool imageCreated = false; + bool uploaded = false; + bool registered = false; + bool abandoned = false; // refused to submit (safety guard) + std::string abandonReason; + std::string uploadError; + VkVideoEncoderStatusCode regStatus = VK_VIDEO_ENCODER_STATUS_ERROR_FORMAT_UNSUPPORTED; + VkVideoEncoderExternalInputPath path = VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + VkBool32 planeStorageViews = VK_FALSE; + VkBool32 storageReadView = VK_FALSE; + uint32_t submitted = 0, retrieved = 0; + uint64_t bytes = 0; + int lastSubmitStatus = 0; + // G-07: what the SAME format does on the UNREGISTERED submit lane. The + // advertised list qualifies SUBOPTIMAL entries as registered-lane only, + // and this is where that qualification is measured rather than restated. + bool unregisteredSubmitTried = false; + VkResult unregisteredSubmit = VK_SUCCESS; + bool deviceLost = false; + // The filter observable is read TWICE and both readings are reported. The + // first run of this harness read it only after Flush() and got + // filterCreated=0 on a session that had just encoded 12 frames -- the + // documented "asked too late" answer. A single post-teardown reading is + // therefore not evidence of anything, in either direction. + // --content-probe. Read at TWO points for the same reason the filter + // observable is: the registration echo says what was ARMED, and the + // post-drain completion says what was SCORED. Only the pair can tell + // "the detour never ran" from "the probe was never armed", and those are + // different defects with different fixes. + bool contentChained = false; + uint32_t contentArmEcho = 0; // VkVideoEncoderImportContentState + uint32_t contentArmGeneration = 0; + uint32_t contentFinalState = 0; + uint32_t contentProbedCount = 0; + uint32_t contentDamagedCount = 0; + uint32_t contentArmedCount = 0; + uint32_t contentMeanY = 0, contentMeanU = 0, contentMeanV = 0; + VkBool32 filterCreatedPre = VK_FALSE, filterCreatedPost = VK_FALSE; + uint64_t filterDispatchPre = 0, stagedCopiesPre = 0; + uint64_t filterDispatchPost = 0, stagedCopiesPost = 0; + // --nv12-companion. Read BETWEEN the two submit loops, which is the whole + // point: a single total cannot say which format dispatched. + bool companionAttempted = false; + bool companionRegistered = false; + VkVideoEncoderStatusCode companionRegStatus = VK_VIDEO_ENCODER_STATUS_SUCCESS; + VkVideoEncoderExternalInputPath companionPath = VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + uint32_t companionSubmitted = 0; + std::string companionError; + uint64_t dispatchAfterPrimary = 0, stagedAfterPrimary = 0; + uint64_t dispatchAfterCompanion = 0, stagedAfterCompanion = 0; + // --nv12-staged-companion. THE ANSWER THE MODE EXISTS FOR. Session- + // constant, so one reading per session is the whole fact; read anyway at + // both split points so a mid-session change could not hide. + bool stagedCompanion = false; + VkVideoEncoderExternalInputPath companionExpectPath = + VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + uint32_t submitFlagsPrimary = 0, submitFamilyPrimary = VK_QUEUE_FAMILY_IGNORED; + uint32_t submitFlagsCompanion = 0, submitFamilyCompanion = VK_QUEUE_FAMILY_IGNORED; + uint64_t foreignAcqPrimary = 0, localAcqPrimary = 0; + uint64_t foreignAcqCompanion = 0, localAcqCompanion = 0; + std::string file; +}; + +// The staged-input queue answer, plus the two acquire counters, in one read. +// Chained the way the header documents -- CompletionInfo::pNext -- and read +// BEFORE Flush(), for the same reason ReadFilterInfo is. +void ReadResidencyInfo(VulkanVideoEncoderExt* enc, uint32_t* submitFlags, + uint32_t* submitFamily, uint64_t* foreignAcq, + uint64_t* localAcq) +{ + VkVideoEncoderStagedSubmitInfo si{}; + si.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STAGED_SUBMIT_INFO; + VkVideoEncoderInputResidencyInfo ri{}; + ri.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_RESIDENCY_INFO; + ri.pNext = &si; // two links on one chain, which the walk supports + VkVideoEncoderCompletionInfo ci{}; + ci.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + ci.pNext = &ri; + if (enc->GetCompletionInfo(&ci) == VK_SUCCESS) { + *submitFlags = si.submitTypeQueueFlags; + *submitFamily = si.queueFamilyIndex; + *foreignAcq = ri.foreignAcquireCount; + *localAcq = ri.localAcquireCount; + } +} + +// VK_QUEUE_* bit -> the engine name, so the report does not make the reader +// decode a hex flag. Named from the flag rather than from the library's +// internal enum on purpose: the flag is what the submit used. +const char* SubmitEngineName(uint32_t queueFlags) +{ + switch (queueFlags) { + case VK_QUEUE_GRAPHICS_BIT: return "GRAPHICS"; + case VK_QUEUE_COMPUTE_BIT: return "COMPUTE"; + case VK_QUEUE_TRANSFER_BIT: return "TRANSFER"; + case VK_QUEUE_VIDEO_ENCODE_BIT_KHR: return "VIDEO_ENCODE"; + case VK_QUEUE_VIDEO_DECODE_BIT_KHR: return "VIDEO_DECODE"; + case 0: return "none/no-session"; + default: return "?"; + } +} + +void ReadFilterInfo(VulkanVideoEncoderExt* enc, VkBool32* created, + uint64_t* dispatch, uint64_t* staged) +{ + VkVideoEncoderFilterInfo fi{}; + fi.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FILTER_INFO; + VkVideoEncoderCompletionInfo ci{}; + ci.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + ci.pNext = &fi; + if (enc->GetCompletionInfo(&ci) == VK_SUCCESS) { + *created = fi.filterCreated; *dispatch = fi.filterDispatchCount; *staged = fi.stagedCopyCount; + } +} + +EncResult RunRow(const Row& row, const char* outDir) +{ + EncResult res; + VkSharedBaseObj enc; + if ((CreateVulkanVideoEncoderExt(enc) != VK_SUCCESS) || !enc) return res; + + const VkVideoComponentBitDepthFlagBitsKHR depth = + (row.group == G_8BIT) ? VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR : + (row.group == G_10BIT) ? VK_VIDEO_COMPONENT_BIT_DEPTH_10_BIT_KHR + : VK_VIDEO_COMPONENT_BIT_DEPTH_12_BIT_KHR; + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = row.codec; + config.encodeWidth = kWidth; config.encodeHeight = kHeight; + config.inputFormat = row.format; + config.inputWidth = kWidth; config.inputHeight = kHeight; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 20000000; config.maxBitrate = 20000000; + config.gopLength = 30; config.consecutiveBFrames = 0; config.idrPeriod = 30; + config.frameRateNum = 30; config.frameRateDen = 1; + config.deviceId = -1; + config.disableFileOutput = VK_TRUE; // bitstream comes back in memory + if (g_validate) { config.validate = VK_TRUE; } + // THE FILTER-ENABLE DERIVATION, AND WHY IT IS NOT AN UNCONDITIONAL VK_TRUE + // UNDER --nv12-staged-companion. + // + // No filter request exists to make: the library builds the preprocess + // filter for the formats that need one and for no others, derived from + // inputFormat alone. An NV12-declared session therefore has no filter and + // an RGBA- or I420-declared one does, which is exactly the split the + // staged-companion A/B needs -- the question that mode asks is what + // DECLARING A WIDER SESSION FORMAT costs the NV12 frames riding on it, and + // that cost only exists if the narrow session is measured without a + // filter. + // Colour is PINNED, not defaulted. matrixCoefficients 1 = BT.709 and + // videoFullRange FALSE = studio range are the same pair the YCbCr rows' + // pattern is generated with, so the VUI the bitstream advertises and the + // numbers actually written agree by construction. Leaving this at 0 no + // longer converts silently -- the library refuses an unexpressible matrix + // and signals Unspecified as BT.709 -- but the pin stays, because a row + // whose colour depends on a library default is a row that measures the + // default. + config.matrixCoefficients = 1; + config.videoFullRange = VK_FALSE; + config.colourPrimaries = 1; + config.transferCharacteristics = 1; + // A row may override all four, and an HDR10 row also chains the static + // metadata. hdrInfo must outlive InitializeExt, which reads the chain and + // keeps nothing; this scope covers that. + VkVideoEncoderHdrMetadataInfo hdrInfo{}; + if (row.colour != nullptr) { + config.colourPrimaries = row.colour->primaries; + config.transferCharacteristics = row.colour->transfer; + config.matrixCoefficients = row.colour->matrix; + config.videoFullRange = row.colour->fullRange; + if (row.colour->hdr10) { + hdrInfo.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_HDR_METADATA_INFO; + hdrInfo.masteringDisplayPresent = VK_TRUE; + // BT.2020 mastering display, ST 2086 units and ST 2086 order + // (green, blue, red). 1000 cd/m^2 peak, 0.0001 cd/m^2 floor. + hdrInfo.displayPrimaryX[0] = 8500; hdrInfo.displayPrimaryY[0] = 39850; + hdrInfo.displayPrimaryX[1] = 6550; hdrInfo.displayPrimaryY[1] = 2300; + hdrInfo.displayPrimaryX[2] = 35400; hdrInfo.displayPrimaryY[2] = 14600; + hdrInfo.whitePointX = 15635; + hdrInfo.whitePointY = 16450; + hdrInfo.maxDisplayMasteringLuminance = 10000000; + hdrInfo.minDisplayMasteringLuminance = 1; + hdrInfo.contentLightLevelPresent = VK_TRUE; + hdrInfo.maxContentLightLevel = 1000; + hdrInfo.maxFrameAverageLightLevel = 400; + config.pNext = &hdrInfo; + } + } + + res.initResult = enc->InitializeExt(config); + if (res.initResult != VK_SUCCESS) return res; + res.sessionInit = true; + + VkInstance instance = enc->GetVkInstance(); + VkDevice device = enc->GetVkDevice(); + VkPhysicalDevice phys = enc->GetVkPhysicalDevice(); + DeviceFns fns; + if (!LoadDeviceFns(instance, device, &fns)) return res; + + // The video-encode family: the one family this session certainly has a + // queue on, and on this vendor it advertises TRANSFER. + uint32_t qfCount = 0; + fns.GetPhysicalDeviceQueueFamilyProperties(phys, &qfCount, nullptr); + std::vector qfs(qfCount); + fns.GetPhysicalDeviceQueueFamilyProperties(phys, &qfCount, qfs.data()); + uint32_t encFamily = UINT32_MAX; + for (uint32_t i = 0; i < qfCount; i++) + if ((qfs[i].queueFlags & VK_QUEUE_VIDEO_ENCODE_BIT_KHR) != 0 && + (qfs[i].queueFlags & VK_QUEUE_TRANSFER_BIT) != 0) { encFamily = i; break; } + if (encFamily == UINT32_MAX) { res.uploadError = "no encode family with TRANSFER"; return res; } + + const Decl d = DeclFor(row.arm); + Img img; + if (!CreateImage(fns, phys, device, row.format, d, row.codec, depth, &img)) return res; + res.imageCreated = true; + + VkVideoEncoderExternalImageDescriptor desc{}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = row.format; desc.width = kWidth; desc.height = kHeight; + desc.tiling = d.tiling; desc.imageUsage = d.usage; desc.imageFlags = d.flags; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; desc.planeCount = 0; + desc.residency = g_foreignResidency + ? VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN + : VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.defaultLayout = DeclaredLayoutFor(d.usage); + desc.existingImage = img.image; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + // THE OPT-IN IS THE CHAIN ITSELF. There is no other switch: a library + // whose caller never chains this struct never creates a probe object and + // never takes the detour, which is what makes --content-probe a genuine + // A/B against the default mode of this same binary. + VkVideoEncoderImportContentInfo contentEcho = {}; + contentEcho.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO; + VkVideoEncoderStatus regStatusStruct = {}; + regStatusStruct.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; + regStatusStruct.pNext = &contentEcho; + res.regStatus = enc->RegisterImageResource( + desc, 0, &resource, g_contentProbe ? ®StatusStruct : nullptr); + if (g_contentProbe) { + res.contentChained = true; + res.contentArmEcho = (uint32_t)contentEcho.state; + res.contentArmGeneration = contentEcho.probeGeneration; + } + if (res.regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS || resource == VK_VIDEO_ENCODER_RESOURCE_NULL) { + DestroyImg(fns, device, &img); return res; + } + res.registered = true; + + // ---- G-07: THE UNREGISTERED LANE, ON THE SAME SESSION AND THE SAME + // ---- FORMAT, AS A PAIR. + // + // A SUBOPTIMAL (ENCODABLE_VIA_FILTER) format encodes only through the + // preprocess filter, which reads per-plane views that exist only on a + // REGISTERED resource. SubmitExternalFrame therefore refuses it with + // VK_ERROR_FORMAT_NOT_SUPPORTED, and the header says so; this is what + // holds the two together. + // + // NO IMAGE IS SUPPLIED, DELIBERATELY. The format/route refusal sits ABOVE + // the handle gate on purpose -- the library states that a null image must + // not be able to change the format answer -- so a bare frame reaches the + // gate under test and nothing below it. That also makes the DIRECT rows a + // free control: they get PAST the format gate and fail on the missing + // image instead, with a different code. A build that refused every + // unregistered submit would satisfy the filter rows and fail the direct + // ones. + { + VkVideoEncodeInputFrame ext{}; + ext.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME; + ext.pNext = nullptr; + ext.image = VK_NULL_HANDLE; + ext.format = row.format; + ext.width = kWidth; + ext.height = kHeight; + ext.currentLayout = VK_IMAGE_LAYOUT_GENERAL; + ext.frameId = 0; + ext.pts = 0; + ext.forceIDR = VK_FALSE; + ext.qpOverride = -1; + res.unregisteredSubmit = enc->SubmitExternalFrame(ext, nullptr); + res.unregisteredSubmitTried = true; + } + VkEncResourceProbe probe; + if (VkEncProbeResource(enc.get(), resource, &probe) == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + res.path = probe.inputPath; + res.planeStorageViews = probe.planeStorageViews; + res.storageReadView = probe.storageReadView; + } + + // SAFETY GATE. A 3-plane input that reaches the staging copy is a + // measured GPU hang, not a slow path. Abandon before any submit. + if (row.arm == ARM_FILTER_YCBCR && res.path != VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER) { + res.abandoned = true; + res.abandonReason = std::string("3-plane input resolved to ") + PathName(res.path) + + ", not FILTER; staging a 3-plane input is a measured VK_ERROR_DEVICE_LOST. " + "No frame was submitted."; + enc->UnregisterImageResource(resource); + DestroyImg(fns, device, &img); + return res; + } + + std::string uerr; + if (!UploadPattern(d.usage, fns, phys, device, encFamily, img.image, row.format, &uerr)) { + res.uploadError = uerr; + enc->UnregisterImageResource(resource); + DestroyImg(fns, device, &img); + return res; + } + res.uploaded = true; + + char path[512]; + std::snprintf(path, sizeof(path), "%s/%s.%s", outDir, row.shortName, row.ext); + FILE* out = std::fopen(path, "wb"); + if (out == nullptr) { + // Leave res.file EMPTY. Every write below is guarded on |out|, so an + // unwritable --out directory used to yield "RESULT: ENCODED -> path" + // for nine files that do not exist -- and the colour verdict for this + // suite is delegated to an out-of-process decoder reading exactly + // those files, so the in-process half would claim success while the + // out-of-process half had nothing to judge. + res.fileOpenFailed = true; + } else { + res.file = path; + } + + for (uint32_t i = 0; i < kFrames; i++) { + VkVideoEncoderFrameSubmitInfo info{}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = resource; + info.frameId = i; info.pts = i; + info.forceIDR = (i == 0) ? VK_TRUE : VK_FALSE; + info.isLastFrame = (i == kFrames - 1) ? VK_TRUE : VK_FALSE; + info.qpOverride = -1; + // UNDEFINED is the documented sentinel for "as declared at + // registration", and under --declare-tso it is REQUIRED rather than + // stylistic: a non-UNDEFINED per-frame layout is an explicit + // statement that outranks the library's own residual record, which + // would bypass the very mechanism the restore exists to feed. + info.currentLayout = g_declareTso ? VK_IMAGE_LAYOUT_UNDEFINED + : VK_IMAGE_LAYOUT_GENERAL; + const VkVideoEncoderStatusCode s = enc->SubmitRegisteredFrame(info, nullptr); + res.lastSubmitStatus = (int)s; + if (s == VK_VIDEO_ENCODER_STATUS_SUCCESS) { res.submitted++; } + else if (s == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + // Capacity, not an error: drain and retry this same frame. + VkVideoEncodeResult r{}; + while (enc->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + res.retrieved++; res.bytes += r.bitstreamSize; + if (out && r.pBitstreamData && r.bitstreamSize) std::fwrite(r.pBitstreamData, 1, r.bitstreamSize, out); + enc->ReleaseEncodedFrame(r.frameId); + r = VkVideoEncodeResult{}; + } + i--; continue; + } else { break; } + + VkVideoEncodeResult r{}; + while (enc->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + res.retrieved++; res.bytes += r.bitstreamSize; + if (out && r.pBitstreamData && r.bitstreamSize) std::fwrite(r.pBitstreamData, 1, r.bitstreamSize, out); + enc->ReleaseEncodedFrame(r.frameId); + r = VkVideoEncodeResult{}; + } + } + + // Completion is ASYNCHRONOUS: a submit returns once the CPU-side pipeline + // has issued, not once the bitstream exists. Acquiring in a single pass + // therefore collects only whatever happened to be ready and silently + // reports the rest as missing -- which on the first run of this harness + // read as "12 submitted, 3 retrieved" and would have understated every + // row. Poll until the retrieved count catches the submitted count, with a + // bounded wait so a genuinely stuck frame still ends the run. + // THE SPLIT READING. Taken before the companion submits anything, so + // "did the RGBA frames dispatch" and "did the NV12 frames dispatch" are two + // numbers rather than one sum. Without this the companion could not tell + // per-frame routing from per-session routing, which is the only claim it is + // here to test. + { + VkBool32 createdNow = VK_FALSE; + ReadFilterInfo(enc.get(), &createdNow, &res.dispatchAfterPrimary, + &res.stagedAfterPrimary); + ReadResidencyInfo(enc.get(), &res.submitFlagsPrimary, + &res.submitFamilyPrimary, &res.foreignAcqPrimary, + &res.localAcqPrimary); + } + + // ---- --nv12-companion: an NV12 registration on this RGBA-declared session + Img companion{}; + bool companionCreated = false; + // ARM B is every RGBA-declared row. ARM A -- the control -- is the + // 8-bit NV12-declared row, whose session has NO filter, and it is included + // ONLY for the staged variant, because that is the only variant whose + // answer differs between the two. The 10-bit DIRECT row is deliberately + // excluded: its sessionEncodeFormat is P010, so an NV12 descriptor is a + // format mismatch the registration correctly refuses, and a refusal there + // would read as a failure of this mode rather than of nothing. + // AV1 ROWS ARE EXCLUDED FROM BOTH COMPANION MODES, DELIBERATELY. Those + // two modes are A/Bs about a Chromium-shaped question -- what declaring a + // wider session input format costs the shipping NV12 lane -- and their + // measured bars are stated in H.26x terms. Their answer is decided by + // whether the session carries a filter OBJECT, which is a function of the + // declared input format and not of the codec, so admitting AV1 rows would + // move those numbers without adding an observation. Excluding them keeps + // every existing companion measurement byte-identical. + const bool companionCodecEligible = + (row.codec != VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR); + const bool companionOnThisRow = + companionCodecEligible && + ((g_nv12Companion && row.arm == ARM_FILTER_RGBA) || + (g_stagedCompanion && + ((row.arm == ARM_FILTER_RGBA) || + (row.arm == ARM_DIRECT && row.group == G_8BIT)))); + if (companionOnThisRow) { + res.companionAttempted = true; + res.stagedCompanion = g_stagedCompanion; + res.companionExpectPath = g_stagedCompanion + ? VK_VIDEO_EXTERNAL_INPUT_PATH_STAGED + : VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + const Decl cd = g_stagedCompanion ? StagedCompanionDecl() + : DeclFor(ARM_DIRECT); + if (!CreateImage(fns, phys, device, VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + cd, row.codec, VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR, + &companion)) { + res.companionError = "NV12 companion vkCreateImage failed"; + } else { + companionCreated = true; + VkVideoEncoderExternalImageDescriptor cdesc{}; + cdesc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + cdesc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + cdesc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + cdesc.width = kWidth; cdesc.height = kHeight; + cdesc.tiling = cd.tiling; cdesc.imageUsage = cd.usage; + cdesc.imageFlags = cd.flags; + cdesc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; cdesc.planeCount = 0; + cdesc.residency = (g_foreignResidency || g_companionForeign) + ? VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN + : VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + cdesc.defaultLayout = DeclaredLayoutFor(cd.usage); + cdesc.existingImage = companion.image; + + VkVideoEncoderResource cres = VK_VIDEO_ENCODER_RESOURCE_NULL; + res.companionRegStatus = + enc->RegisterImageResource(cdesc, 0, &cres, nullptr); + if (res.companionRegStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS && + cres != VK_VIDEO_ENCODER_RESOURCE_NULL) { + res.companionRegistered = true; + VkEncResourceProbe cprobe; + if (VkEncProbeResource(enc.get(), cres, &cprobe) == + VK_VIDEO_ENCODER_STATUS_SUCCESS) { + res.companionPath = cprobe.inputPath; + } + std::string cerr; + if (!UploadPattern(cd.usage, fns, phys, device, encFamily, + companion.image, + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, &cerr)) { + res.companionError = "NV12 companion upload: " + cerr; + } else { + for (uint32_t i = 0; i < kFrames; i++) { + VkVideoEncoderFrameSubmitInfo ci{}; + ci.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + ci.resource = cres; + ci.frameId = kFrames + i; ci.pts = kFrames + i; + ci.forceIDR = VK_FALSE; + ci.isLastFrame = (i == kFrames - 1) ? VK_TRUE : VK_FALSE; + ci.qpOverride = -1; + ci.currentLayout = g_declareTso + ? VK_IMAGE_LAYOUT_UNDEFINED + : VK_IMAGE_LAYOUT_GENERAL; + const VkVideoEncoderStatusCode cs = + enc->SubmitRegisteredFrame(ci, nullptr); + if (cs == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + res.companionSubmitted++; + } else if (cs == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + VkVideoEncodeResult r{}; + while (enc->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + res.retrieved++; res.bytes += r.bitstreamSize; + if (out && r.pBitstreamData && r.bitstreamSize) + std::fwrite(r.pBitstreamData, 1, r.bitstreamSize, out); + enc->ReleaseEncodedFrame(r.frameId); + r = VkVideoEncodeResult{}; + } + i--; continue; + } else { + res.companionError = + "NV12 companion submit status " + std::to_string((int)cs); + break; + } + VkVideoEncodeResult r{}; + while (enc->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + res.retrieved++; res.bytes += r.bitstreamSize; + if (out && r.pBitstreamData && r.bitstreamSize) + std::fwrite(r.pBitstreamData, 1, r.bitstreamSize, out); + enc->ReleaseEncodedFrame(r.frameId); + r = VkVideoEncodeResult{}; + } + } + } + VkBool32 createdNow = VK_FALSE; + ReadFilterInfo(enc.get(), &createdNow, + &res.dispatchAfterCompanion, + &res.stagedAfterCompanion); + ReadResidencyInfo(enc.get(), &res.submitFlagsCompanion, + &res.submitFamilyCompanion, + &res.foreignAcqCompanion, + &res.localAcqCompanion); + enc->UnregisterImageResource(cres); + } else { + res.companionError = "NV12 companion registration refused"; + } + } + // The submitted total must include the companion so the retrieved/ + // submitted balance check below still means what it says. + res.submitted += res.companionSubmitted; + } + + // BEFORE Flush: dispatches are RECORDED inline on the submit call, so the + // full count already exists here. + ReadFilterInfo(enc.get(), &res.filterCreatedPre, &res.filterDispatchPre, &res.stagedCopiesPre); + + if (enc->Flush() == VK_ERROR_DEVICE_LOST) res.deviceLost = true; + for (int spin = 0; spin < 2000 && res.retrieved < res.submitted; spin++) { + VkVideoEncodeResult r{}; + while (enc->AcquireNextEncodedFrame(r) == VK_SUCCESS) { + res.retrieved++; res.bytes += r.bitstreamSize; + if (out && r.pBitstreamData && r.bitstreamSize) std::fwrite(r.pBitstreamData, 1, r.bitstreamSize, out); + enc->ReleaseEncodedFrame(r.frameId); + r = VkVideoEncodeResult{}; + } + if (res.retrieved < res.submitted) usleep(5000); + } + if (out) std::fclose(out); + + ReadFilterInfo(enc.get(), &res.filterCreatedPost, &res.filterDispatchPost, &res.stagedCopiesPost); + + // READ BEFORE THE RETIREMENT, and that ordering is load-bearing: + // UnregisterImageResource calls ForgetRegistration, which erases the + // verdict and drops the outstanding count. Reading after it would report + // a clean, empty probe for a session that had just found damage. + if (g_contentProbe) { + VkVideoEncoderImportContentInfo verdict = {}; + verdict.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO; + VkVideoEncoderCompletionInfo completion = {}; + completion.pNext = &verdict; + enc->GetCompletionInfo(&completion); + res.contentFinalState = (uint32_t)verdict.state; + res.contentProbedCount = verdict.probedRegistrationCount; + res.contentDamagedCount = verdict.damagedRegistrationCount; + res.contentArmedCount = verdict.armedRegistrationCount; + res.contentMeanY = verdict.meanY; + res.contentMeanU = verdict.meanU; + res.contentMeanV = verdict.meanV; + } + + enc->UnregisterImageResource(resource); + DestroyImg(fns, device, &img); + if (companionCreated) { DestroyImg(fns, device, &companion); } + enc = nullptr; + return res; +} + + +// --------------------------------------------------------------------------- +// THE FOUR-QUADRANT DECODE ASSERTION. +// +// WHY IT EXISTS. This file's header has promised since it was written that +// "Colour is judged OUT OF PROCESS from the written file". Nothing ever ran +// that judgement: the suite wrote seven elementary streams and no line of code +// -- here or in CMakeLists.txt -- ever looked at a pixel in them. It was a +// documented acceptance criterion with nothing behind it. +// +// THE GAP IS NOT THEORETICAL. Commenting out the four +// m_vkDevCtx->CmdDispatch calls in +// common/libs/VkCodecUtils/VulkanFilterYuvCompute.cpp DESTROYS the decoded +// picture -- every quadrant of frame 0 of all five FILTER-routed rows becomes +// a flat (0,76,0), which is BT.709-limited (16,128,128), i.e. the zero-filled +// output image the filter never wrote to -- while EVERY OTHER OBSERVABLE IN +// THIS SUITE STAYS GREEN. Every session still reports encoded, every filter +// row still reports a dispatch, no staged copy is taken, no device is lost, +// every submitted frame is retrieved and the byte counts do not move. +// +// The dispatch counter cannot see it because it counts command buffers +// RECORDED, not executed: VkVideoEncoder.cpp increments it immediately after +// RecordCommandBuffer returns VK_SUCCESS, and a command buffer with the +// vkCmdDispatch deleted records just as successfully as one without. The +// decoded picture is the only observable in this suite that can tell the +// difference, which is why this runs in the DEFAULT verdict chain and not +// behind an opt-in flag. +// +// WHY A TOLERANCE AND NOT A CHECKSUM. An md5 pin over the bitstream was +// considered and rejected: it goes red on any benign encoder change, it gets +// disabled the first time it does, and then the suite is back where it +// started with a colour criterion nothing enforces. +// +// THE TOLERANCE, AND THE ARITHMETIC BEHIND IT. Every figure below is an +// L-infinity (worst single channel) deviation, in 8-bit RGB, of a decoded +// quadrant centre from the RGB the pattern writer put in that quadrant. +// The figures are: +// +// HEALTHY tree, all 7 encoding rows x 4 quadrants x 3 channels: worst = 2 +// (the 8-bit rows land TL(251,0,0) TR(0,251,0) BL(0,0,252) BR(255,255,255) +// against a written 253/253/252/255; the 10-bit rows land TL(253,0,0) +// TR(0,252,1) BL(0,0,253) BR(255,253,255).) +// HEALTHY tree, spatial spread within a 41x41 box centred on each sample +// point: 0 on every row, quadrant and channel. The decoded quadrants are +// exactly flat, so the single-pixel sample below is a measurement and not +// a lucky draw. +// DEAD FILTER (the flat 0,76,0 above): TL 253, TR 177, BL 252, BR 255. The +// SMALLEST of those -- the number the tolerance actually has to sit under +// -- is 177. +// RED/BLUE CHANNEL SWAP in the RGBA storage-read arm: >= 251 on TL and BL. +// TR (green) and BR (white) are invariant under that swap, which is +// exactly why the assertion is per-quadrant and not an aggregate. +// +// CC-1 W8: THE GATE MOVED FROM THE RGB DOMAIN TO THE Y'CbCr DOMAIN, AND THE +// TOLERANCE MOVED WITH IT. Everything below the "WHAT THE OLD NUMBER WAS" +// heading is history kept on purpose; read the new bar first. +// +// WHAT IS COMPARED NOW. The decoder is asked for `-pix_fmt yuv444p`, which +// performs NO colour conversion and NO inverse matrix and does NO clamping of +// an out-of-gamut RGB triple -- it only un-subsamples chroma. The four +// quadrant centres are then compared, per channel, against +// RgbToYuvLimited(QuadAt(pt), ). So the gate asks +// the question it always claimed to ask -- "are the samples the matrix this +// stream declares?" -- instead of asking it through a lossy round trip. +// +// WHY THE OLD DOMAIN COULD NOT ANSWER IT, measured rather than argued. At +// 1920x1080 with the gate's own pinned decoder invocation, the RGB gate at +// tolerance 24 caught 2 of the 5 applied/declared mismatch directions. It +// missed `applied 709 / declared 601` at 23 against a tolerance of 24 -- it +// passed BY ONE CODE VALUE -- and it missed both directions of the 709/2020 +// confusion. Three structural reasons, none of which a tolerance change +// fixes: +// * the BR (white) quadrant is a mathematical no-op under every matrix; +// * the round trip through the inverse matrix partially CANCELS the error; +// * Clamp8 eats the residual asymmetrically -- 601->709 gives 40 while +// 709->601 gives 23 for the SAME physical error, which is pure clamp +// artefact and not physics. +// A whole-cube search puts a hard physical ceiling of 13.2 on any RGB +// round-trip gate for a 709/2020 confusion, so the TOLERANCE, not the +// palette, was the binding constraint and no palette rescues the domain. +// +// THE NEW BAR. Measured worst |dY'CbCr| across the four quadrant centres, +// 1920x1080, real encode, decoded to yuv444p: +// +// applied / declared worst |dY'CbCr| +// 709 / 709 (healthy) 0.5 +// 2020 / 709 10.8 <- the weakest failure +// 709 / 2020 11.1 +// 601 / 2020 19.3 +// 601 / 709 27.4 +// 709 / 601 27.5 +// +// kQuadTolerance = 4 sits 8x above the healthy figure and 2.7x below the +// WEAKEST failure, and -- unlike the number it replaces -- it is SYMMETRIC, +// as the physics requires. The old gate's 40-versus-23 asymmetry for one +// physical error was the clamp, and it is gone with the domain. +// +// WHAT LIVES IN THE 0.5-TO-4 GAP: rate-control drift, rounding, and decoder +// version differences in the chroma UPSAMPLE (the only conversion yuv444p +// still performs). What lives above 4: every matrix confusion, a dead filter, +// and a channel swap. +// +// L1 WAS THE OTHER OPTION AND IS NOT TAKEN. Keeping the RGB domain and +// retuning 24 -> 6 gives only 3x/2x separation instead of 8x/2.7x, and it does +// not remove the clamp asymmetry or the matrix-invariant white quadrant. +// Retained only as a one-line fallback if the yuv444p decode proves +// unavailable on some host -- and it would need its own rationale, because 6 +// does not survive the paragraphs below either. +// +// THE ONE THING THE OLD DOMAIN GAVE FOR FREE AND THIS DOES NOT. `-pix_fmt +// rgb24` consulted the stream's VUI to choose its inverse matrix, so the old +// gate checked the LABEL implicitly: a stream that signalled the wrong matrix +// decoded to different pixels. `yuv444p` does not consult the VUI at all. The +// label check is therefore now EXPLICIT -- see CheckColourLabel(), which +// ffprobes color_space and color_range and compares them to the row's +// configuration. Losing an implicit check and not replacing it would have +// been a straight downgrade. +// +// EXPECTED VALUES COME FROM QuadAt(), NOT FROM A TABLE OF DECODED NUMBERS. +// Tying the assertion to the pattern the uploader actually writes means the +// two cannot drift apart, and it means the geometry -- which quadrant is +// where -- is stated once in this file rather than twice. +// +// ---- WHAT THE OLD NUMBER WAS, and why its arithmetic is no longer quoted -- +// +// kQuadTolerance was 24, described as "12x the largest deviation a healthy +// encode has ever produced here, and 7.4x below the smallest deviation any +// failure mode in scope produces". Both multipliers were true OF THE RGB +// DOMAIN and of the failure set considered then -- a dead filter (smallest +// deviation 177) and a red/blue channel swap (>= 251). They were never +// measured against a MATRIX confusion, and the sentence "there is no +// legitimate encoder change that moves a flat primary by 24/255 and is still +// the same picture" is true and beside the point: a matrix confusion moves it +// by 23, and 23 < 24. +// +// The AV1 paragraph that followed is also superseded, and its finding is +// preserved because it still holds: measured on the A4000 host with the SAME +// pattern writer at 1920x1080, H.264 (libx264), AV1 (libsvtav1, raw OBU) and +// AV1 (libsvtav1, real IVF) all produced worst channel delta 2 with 41x41 +// spread 0 -- identical to the count. The quadrants decode exactly flat on +// AV1 as they do on H.26x, so the single-pixel sample is as sound on AV1 as +// on H.26x. That bounded the CODEC and the colour conversion, not the rate +// control of the Vulkan AV1 encoder; if a Blackwell run shows AV1 quadrant +// deltas materially above the healthy figure, the correct response is still a +// separate AV1-specific constant here rather than a wider shared tolerance. +const int kQuadTolerance = 4; + +// THE DECODER INVOCATION, PINNED. Frame 0 only, chroma un-subsampled to +// yuv444p on stdout: +// +// ffmpeg -v error -i -frames:v 1 -pix_fmt yuv444p -f rawvideo - +// +// NO COLOUR CONVERSION AT ALL. yuv444p is the decoder's own output format +// with the chroma planes upsampled; there is no inverse matrix, no gamut +// clamp and no dependence on the stream's VUI. That last part is why +// CheckColourLabel() exists -- the old rgb24 invocation checked the label +// implicitly by consulting the VUI, and this one cannot. +enum QuadStatus { + QUAD_PASS, + QUAD_MISMATCH, // the decoder ran and the picture is wrong -- FAILURE + QUAD_DECODE_FAILED, // the decoder is present and produced no frame -- FAILURE + QUAD_NO_DECODER, // there is no ffmpeg at all -- a SKIP, and a loud one +}; + +int g_quadChecked = 0; +int g_quadFailed = 0; +bool g_quadNoDecoder = false; + +// Probed ONCE per run. -1 unknown, 0 absent, 1 present. +int g_ffmpegProbe = -1; +std::string g_ffmpegBanner; + +void DrainToEof(FILE* p) +{ + char scratch[65536]; + while (std::fread(scratch, 1, sizeof(scratch), p) != 0) { } +} + +bool HaveDecoder() +{ + if (g_ffmpegProbe < 0) { + g_ffmpegProbe = 0; + FILE* p = popen("ffmpeg -version 2>/dev/null", "r"); + if (p != nullptr) { + char line[256]; + line[0] = '\0'; + const bool got = (std::fgets(line, sizeof(line), p) != nullptr); + DrainToEof(p); + const int rc = pclose(p); + if (rc == 0 && got) { + g_ffmpegProbe = 1; + std::string b(line); + while (!b.empty() && (b[b.size() - 1] == '\n' || b[b.size() - 1] == '\r')) { + b.erase(b.size() - 1); + } + g_ffmpegBanner = b; + } + } + } + return g_ffmpegProbe == 1; +} + +int AbsDiff(int a, int b) { return (a > b) ? (a - b) : (b - a); } + +struct QuadPoint { const char* name; uint32_t x, y; }; + +// CC-1 W8: the LABEL check, which the rgb24 decode used to make implicitly. +// yuv444p does not consult the VUI, so a stream that signalled the wrong +// matrix would decode to the SAME samples and the quadrant comparison could +// not see it. This asks ffprobe directly. +// +// SCOPED TO THE NON-HDR ROWS by its caller: the HDR rows are judged by +// CheckHdrSignalling(), which reads the same two fields plus the static +// metadata, and running both would double-count the row. +bool CheckColourLabel(const std::string& file, + uint8_t matrix, + bool fullRange, + std::string* detail) +{ + if (file.find('\'') != std::string::npos) { + *detail = "the output path contains a single quote"; + return false; + } + char cmd[1024]; + const int need = std::snprintf( + cmd, sizeof(cmd), + "ffprobe -v error -select_streams v:0 -show_entries " + "stream=color_space,color_range -of default=nw=1 '%s'", + file.c_str()); + if (need < 0 || (size_t)need >= sizeof(cmd)) { + *detail = "the ffprobe command line does not fit"; + return false; + } + FILE* p = popen(cmd, "r"); + if (p == nullptr) { + *detail = "popen(ffprobe) failed"; + return false; + } + std::string out; + char buf[512]; + while (std::fgets(buf, sizeof(buf), p) != nullptr) out += buf; + const int rc = pclose(p); + if (rc != 0) { + *detail = "ffprobe failed on a file this suite reported as ENCODED"; + return false; + } + + // ffprobe's spelling of each code point. Only the values this suite can + // configure are listed; anything else is reported verbatim and fails, + // rather than being silently accepted. + const char* wantSpace = + (matrix == 1) ? "bt709" : + (matrix == 5 || matrix == 6) ? "smpte170m" : + (matrix == 9) ? "bt2020nc" : nullptr; + const char* wantRange = fullRange ? "pc" : "tv"; + + std::string gotSpace = "", gotRange = ""; + size_t pos = 0; + while (pos < out.size()) { + const size_t eol = out.find('\n', pos); + const std::string line = out.substr(pos, (eol == std::string::npos) ? std::string::npos : eol - pos); + const size_t eq = line.find('='); + if (eq != std::string::npos) { + const std::string k = line.substr(0, eq), v = line.substr(eq + 1); + if (k == "color_space") gotSpace = v; + if (k == "color_range") gotRange = v; + } + if (eol == std::string::npos) break; + pos = eol + 1; + } + + if (wantSpace == nullptr) { + char m[256]; + std::snprintf(m, sizeof(m), + "this gate has no ffprobe spelling for matrix_coefficients %u; " + "add one rather than skipping the label check", + (unsigned)matrix); + *detail = m; + return false; + } + if (gotSpace != wantSpace || gotRange != wantRange) { + char m[320]; + std::snprintf(m, sizeof(m), + "the stream LABEL disagrees with the configuration: " + "ffprobe says color_space=%s color_range=%s, the session " + "declared matrix_coefficients %u (%s) and %s range", + gotSpace.c_str(), gotRange.c_str(), (unsigned)matrix, + wantSpace, wantRange); + *detail = m; + return false; + } + return true; +} + +QuadStatus CheckQuadrants(const std::string& file, + uint8_t matrix, + bool fullRange, + std::string* detail) +{ + if (!HaveDecoder()) { + *detail = "ffmpeg is not on PATH"; + return QUAD_NO_DECODER; + } + // --out is caller-controlled and the path is interpolated into a shell + // command. A single quote in it would let the argument rewrite the command + // rather than name a file; refuse instead of running something else. + if (file.find('\'') != std::string::npos) { + *detail = "the output path contains a single quote and cannot be handed to the decoder"; + return QUAD_DECODE_FAILED; + } + + char cmd[1024]; + const int need = std::snprintf( + cmd, sizeof(cmd), + "ffmpeg -v error -i '%s' -frames:v 1 -pix_fmt yuv444p -f rawvideo -", + file.c_str()); + // A silently truncated command line would name a DIFFERENT file, and the + // decode would fail for a reason that has nothing to do with the picture. + // A gate that can go red for a reason it does not name is a gate that gets + // switched off. + if (need < 0 || (size_t)need >= sizeof(cmd)) { + *detail = "the decoder command line does not fit; --out path is too long"; + return QUAD_DECODE_FAILED; + } + FILE* p = popen(cmd, "r"); + if (p == nullptr) { + *detail = "popen(ffmpeg) failed"; + return QUAD_DECODE_FAILED; + } + + // yuv444p is THREE FULL-SIZE PLANES: Y, then Cb, then Cr. Same total size + // as the packed rgb24 buffer this replaces, but the layout is planar, so + // a sample is at plane*kWidth*kHeight + y*kWidth + x rather than at + // (y*kWidth + x)*3 + channel. + std::vector rgb((size_t)kWidth * (size_t)kHeight * 3); + size_t got = 0; + while (got < rgb.size()) { + const size_t n = std::fread(rgb.data() + got, 1, rgb.size() - got, p); + if (n == 0) break; + got += n; + } + DrainToEof(p); + const int rc = pclose(p); + if (rc != 0 || got != rgb.size()) { + char msg[256]; + std::snprintf(msg, sizeof(msg), + "the decoder produced %zu of %zu bytes for frame 0 " + "(ffmpeg exit status %d). A file this suite reported as " + "ENCODED that ffmpeg cannot decode is a failure, not a skip", + got, rgb.size(), rc); + *detail = msg; + return QUAD_DECODE_FAILED; + } + + const QuadPoint pts[4] = { + {"TL", kWidth / 4, kHeight / 4}, + {"TR", 3 * kWidth / 4, kHeight / 4}, + {"BL", kWidth / 4, 3 * kHeight / 4}, + {"BR", 3 * kWidth / 4, 3 * kHeight / 4}, + }; + + const size_t kPlane = (size_t)kWidth * (size_t)kHeight; + std::string seen; + std::string bad; + int worst = 0; + int nBad = 0; + for (int q = 0; q < 4; q++) { + const size_t o = (size_t)pts[q].y * (size_t)kWidth + (size_t)pts[q].x; + const int gy = (int)rgb[o]; + const int gu = (int)rgb[kPlane + o]; + const int gv = (int)rgb[2 * kPlane + o]; + // THE EXPECTATION IS BUILT WITH THE MATRIX THE ROW DECLARED, which is + // the whole of CC-1 W8: a stream whose samples were converted with a + // DIFFERENT matrix from the one it declares now lands outside the + // tolerance instead of being partially cancelled by an inverse matrix + // and then clamped. + const YUV want = RgbToYuvLimited(QuadAt(pts[q].x, pts[q].y), matrix); + const int wy = (int)Clamp8(want.y); + const int wu = (int)Clamp8(want.cb); + const int wv = (int)Clamp8(want.cr); + const int dy = AbsDiff(gy, wy); + const int du = AbsDiff(gu, wu); + const int dv = AbsDiff(gv, wv); + const int dmax = (dy > du) ? ((dy > dv) ? dy : dv) : ((du > dv) ? du : dv); + if (dmax > worst) worst = dmax; + + char one[64]; + std::snprintf(one, sizeof(one), " %s Y'CbCr(%d,%d,%d)", + pts[q].name, gy, gu, gv); + seen += one; + + if (dmax > kQuadTolerance) { + nBad++; + // NAME THE QUADRANT AND THE AMOUNT. A gate that says only "colour + // wrong" is undebuggable from a CI log, and which quadrants moved + // is the diagnosis: all four flat means the filter is not running, + // TL and BL swapped with TR and BR intact means red and blue are + // crossed, and a uniform CHROMA offset with Y' intact is a matrix + // confusion. + const char* ch = (dmax == dy) ? "Y'" : ((dmax == du) ? "Cb" : "Cr"); + char m[360]; + std::snprintf(m, sizeof(m), + "\n %s centre (%u,%u): got Y'CbCr(%d,%d,%d) " + "want (%d,%d,%d) for matrix %u -- delta (%d,%d,%d), " + "worst channel %s off by %d, tolerance %d", + pts[q].name, pts[q].x, pts[q].y, gy, gu, gv, + wy, wu, wv, (unsigned)matrix, dy, du, dv, + ch, dmax, kQuadTolerance); + bad += m; + } + } + + if (nBad != 0) { + char head[160]; + std::snprintf(head, sizeof(head), + "%d of 4 quadrant centres of frame 0 are outside +/-%d", + nBad, kQuadTolerance); + *detail = std::string(head) + bad; + return QUAD_MISMATCH; + } + + // THE LABEL, CHECKED LAST AND CHECKED SEPARATELY. The samples being right + // and the stream saying so are two claims, and after the move to yuv444p + // this gate can no longer conflate them. + std::string labelDetail; + if (!CheckColourLabel(file, matrix, fullRange, &labelDetail)) { + *detail = "the samples match matrix " + std::to_string((unsigned)matrix) + + " but " + labelDetail; + return QUAD_MISMATCH; + } + + char ok[360]; + std::snprintf(ok, sizeof(ok), + "%s worst channel delta %d (tolerance %d, matrix %u, " + "label verified)", + seen.c_str(), worst, kQuadTolerance, (unsigned)matrix); + *detail = ok; + return QUAD_PASS; +} + +//============================================================================= +// THE HDR GATE. Out of process, like the quadrant gate beside it, and for the +// same reason: nothing inside this binary can tell whether the bytes it wrote +// mean what it thinks they mean. +// +// It reads three things, and each one is a separate way for the HDR feature +// to be broken while every counter stays green: +// +// 1. the four VUI code points (-show_streams). A colour description that +// never reached the SPS looks identical from here to one that did. +// 2. a full decoded frame whose four quadrant centres are not all equal. +// Weaker than the BT.709 quadrant comparison, deliberately -- see the +// note on the HDR row in kRows -- but it still catches a stream that +// decodes to nothing, or to a flat picture. +// 3. both SEI payloads, field by field (-show_frames side_data_list). This +// is the half that needs a real decoder to judge: a SEI the encoder does +// not emit is invisible to every other check in this suite. +// +// The primary values are asserted individually rather than as a set, because +// the failure worth catching is a PERMUTED one: ST 2086 orders the primaries +// green, blue, red and it is the easiest thing in this feature to get wrong. +// red_x=35400 and green_y=39850 cannot both hold under a permutation. +int g_hdrGated = 0; +int g_hdrFailed = 0; + +bool RunProbe(const char* cmd, std::string* out) +{ + FILE* p = popen(cmd, "r"); + if (p == nullptr) return false; + char buf[4096]; + out->clear(); + size_t n; + while ((n = std::fread(buf, 1, sizeof(buf), p)) != 0) { + out->append(buf, n); + } + return pclose(p) == 0; +} + +QuadStatus CheckHdrSignalling(const std::string& file, std::string* detail) +{ + if (!HaveDecoder()) { + *detail = "ffmpeg/ffprobe is not on PATH"; + return QUAD_NO_DECODER; + } + if (file.find('\'') != std::string::npos) { + *detail = "the output path contains a single quote and cannot be handed to the prober"; + return QUAD_DECODE_FAILED; + } + + char cmd[1024]; + std::string streams; + std::snprintf(cmd, sizeof(cmd), + "ffprobe -v error -select_streams v:0 -show_entries " + "stream=color_range,color_space,color_transfer,color_primaries " + "-of default=nw=1 '%s'", file.c_str()); + if (!RunProbe(cmd, &streams)) { + *detail = "ffprobe -show_streams failed"; + return QUAD_DECODE_FAILED; + } + + std::string frames; + std::snprintf(cmd, sizeof(cmd), + "ffprobe -v error -select_streams v:0 -read_intervals '%%+#1' " + "-show_frames -show_entries frame=side_data_list " + "-of default=nw=1 '%s'", file.c_str()); + if (!RunProbe(cmd, &frames)) { + *detail = "ffprobe -show_frames failed"; + return QUAD_DECODE_FAILED; + } + + // PRESENCE, matched as strings, because these have no numeric value: the + // payload is either in the stream or it is not. + struct Want { const char* what; const char* needle; const std::string* in; }; + const Want wants[] = { + {"color_primaries bt2020", "color_primaries=bt2020", &streams}, + {"color_transfer smpte2084","color_transfer=smpte2084", &streams}, + {"color_space bt2020nc", "color_space=bt2020nc", &streams}, + {"color_range tv", "color_range=tv", &streams}, + {"mastering display SEI", "Mastering display metadata", &frames}, + {"content light SEI", "Content light level metadata", &frames}, + }; + + // CC-1 W9: THE NUMBERS ARE PARSED AND COMPARED, NOT STRING-MATCHED, AND + // THAT IS WHAT MAKES THIS GATE CODEC-INDEPENDENT. + // + // The sixteen entries this replaces were exact strings like + // "red_x=35400/50000", which is H.265's ST 2086 SEI in its own units + // (chromaticity in 1/50000, max luminance in 1/10000). AV1's + // metadata_hdr_mdcv carries THE SAME PHYSICAL VALUES in DIFFERENT fixed + // point -- 0.16 for chromaticity (/65536), 24.8 for max luminance (/256), + // 18.14 for min luminance (/16384) -- so ffprobe prints + // "red_x=46399/65536" and "max_luminance=256000/256". Those AGREE + // numerically (46399/65536 = 0.707993 against 35400/50000 = 0.70800; + // 256000/256 = 1000 exactly against 10000000/10000 = 1000) and a string + // match cannot see it. Without this, both AV1 HDR rows report HDR GATE: FAIL + // with ten MISSING lines while the H.265 control passes in the same run. + // + // THE TOLERANCES ARE QUANTISATION STEPS, NOT SLACK. Each is one step of + // the COARSER of the two representations, so a real disagreement of one + // step still fails: + // chromaticity 1/16384 -- AV1's 0.16 field is finer (1/65536) and + // H.265's is 1/50000; 1/16384 = 6.1e-5 bounds both with + // room for the rounding, and the values differ by 0.7 + // in every direction if a primary is actually wrong. + // max luminance 1 cd/m2 out of 1000 -- AV1's step is 1/256. + // min luminance 1/16384 -- this is the one that genuinely differs: + // 0.0001 cd/m2 cannot be represented in 18.14 and AV1 + // emits 2/16384 = 0.000122. That is AV1's coarser + // quantisation, not an error, and encoding it as a + // tolerance is the only honest way to assert it. + // MaxCLL/MaxFALL exact -- both codecs carry plain integers. + struct WantNum { const char* what; const char* key; double value; double tol; }; + const double kChromaTol = 1.0 / 16384.0; + const WantNum nums[] = { + {"red primary x", "red_x", 0.708, kChromaTol}, + {"red primary y", "red_y", 0.292, kChromaTol}, + {"green primary x", "green_x", 0.170, kChromaTol}, + {"green primary y", "green_y", 0.797, kChromaTol}, + {"blue primary x", "blue_x", 0.131, kChromaTol}, + {"blue primary y", "blue_y", 0.046, kChromaTol}, + {"white point x", "white_point_x", 0.3127, kChromaTol}, + {"white point y", "white_point_y", 0.3290, kChromaTol}, + {"max luminance", "max_luminance", 1000.0, 1.0}, + {"min luminance", "min_luminance", 0.0001, 1.0 / 16384.0}, + {"MaxCLL", "max_content", 1000.0, 0.0}, + {"MaxFALL", "max_average", 400.0, 0.0}, + }; + + std::string missing; + for (const Want& w : wants) { + if (w.in->find(w.needle) == std::string::npos) { + missing += std::string("\n MISSING ") + w.what + " (" + + w.needle + ")"; + } + } + for (const WantNum& w : nums) { + // "=" anchored at a line start, so red_x cannot match inside + // some other key and a value cannot be found in a neighbouring field. + const std::string needle = std::string(w.key) + "="; + size_t at = std::string::npos; + for (size_t p = 0; p + needle.size() <= frames.size(); ++p) { + if ((p == 0 || frames[p - 1] == '\n') && + frames.compare(p, needle.size(), needle) == 0) { + at = p + needle.size(); + break; + } + } + if (at == std::string::npos) { + missing += std::string("\n MISSING ") + w.what + " (no " + + w.key + " in the decoded side data at all)"; + continue; + } + const size_t eol = frames.find('\n', at); + const std::string raw = + frames.substr(at, (eol == std::string::npos) ? std::string::npos : eol - at); + // ffprobe prints these as "num/den" for the fixed-point fields and as + // a plain integer for MaxCLL/MaxFALL. Accept both. + double got = 0.0; + const size_t slash = raw.find('/'); + bool parsed = false; + if (slash != std::string::npos) { + const double num = std::atof(raw.substr(0, slash).c_str()); + const double den = std::atof(raw.substr(slash + 1).c_str()); + if (den != 0.0) { got = num / den; parsed = true; } + } else if (!raw.empty()) { + got = std::atof(raw.c_str()); + parsed = true; + } + if (!parsed) { + missing += std::string("\n UNPARSEABLE ") + w.what + " (" + + w.key + "=" + raw + ")"; + continue; + } + const double d = (got > w.value) ? (got - w.value) : (w.value - got); + if (d > w.tol) { + char m[256]; + std::snprintf(m, sizeof(m), + "\n WRONG %s: %s=%s parses to %.6f, wanted " + "%.6f +/- %.6f", + w.what, w.key, raw.c_str(), got, w.value, w.tol); + missing += m; + } + } + if (!missing.empty()) { + *detail = "the stream does not carry the HDR10 signalling it was " + "configured with:" + missing + + "\n --- ffprobe -show_streams ---\n" + streams + + " --- ffprobe -show_frames ---\n" + frames; + return QUAD_MISMATCH; + } + + // And it must still be a picture. A header-only gate would pass on a + // stream that carries perfect metadata and decodes to nothing. + std::snprintf(cmd, sizeof(cmd), + "ffmpeg -v error -i '%s' -frames:v 1 -pix_fmt rgb24 " + "-f rawvideo -", file.c_str()); + FILE* p = popen(cmd, "r"); + if (p == nullptr) { + *detail = "popen(ffmpeg) failed"; + return QUAD_DECODE_FAILED; + } + std::vector rgb((size_t)kWidth * (size_t)kHeight * 3); + size_t got = 0; + while (got < rgb.size()) { + const size_t n = std::fread(rgb.data() + got, 1, rgb.size() - got, p); + if (n == 0) break; + got += n; + } + DrainToEof(p); + const int rc = pclose(p); + if ((rc != 0) || (got != rgb.size())) { + char msg[256]; + std::snprintf(msg, sizeof(msg), + "the decoder produced %zu of %zu bytes for frame 0 " + "(ffmpeg exit status %d)", got, rgb.size(), rc); + *detail = msg; + return QUAD_DECODE_FAILED; + } + const QuadPoint pts[4] = { + {"TL", kWidth / 4, kHeight / 4}, + {"TR", 3 * kWidth / 4, kHeight / 4}, + {"BL", kWidth / 4, 3 * kHeight / 4}, + {"BR", 3 * kWidth / 4, 3 * kHeight / 4}, + }; + std::string seen; + bool allSame = true; + int first[3] = {0, 0, 0}; + for (int q = 0; q < 4; q++) { + const size_t o = ((size_t)pts[q].y * (size_t)kWidth + (size_t)pts[q].x) * 3; + const int r = (int)rgb[o], g = (int)rgb[o + 1], b = (int)rgb[o + 2]; + if (q == 0) { first[0] = r; first[1] = g; first[2] = b; } + else if ((r != first[0]) || (g != first[1]) || (b != first[2])) { + allSame = false; + } + char one[64]; + std::snprintf(one, sizeof(one), " %s(%d,%d,%d)", pts[q].name, r, g, b); + seen += one; + } + if (allSame) { + *detail = "all four quadrant centres decoded to the same value" + seen + + " -- the metadata is right and the picture is not"; + return QUAD_MISMATCH; + } + *detail = "VUI bt2020/smpte2084/bt2020nc/tv, ST 2086 + MaxCLL/MaxFALL " + "read back by ffprobe, picture" + seen; + return QUAD_PASS; +} + +} // namespace + +int main(int argc, char** argv) +{ + const char* outDir = "."; + const char* only = nullptr; + // --codec. A CODEC-WIDE row filter, which --only cannot express: --only + // takes ONE shortName and the AV1 arm is four rows. It exists so ctest + // can register the AV1 arm as its own entry -- see CMakeLists.txt, which + // also states how that entry stays GREEN (SKIPPED, 77) on hardware + // without AV1 rather than reddening the gating set. + const char* onlyCodec = nullptr; + std::vector rowResults; + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--out") == 0 && i + 1 < argc) outDir = argv[++i]; + else if (std::strcmp(argv[i], "--only") == 0 && i + 1 < argc) only = argv[++i]; + else if (std::strcmp(argv[i], "--codec") == 0 && i + 1 < argc) onlyCodec = argv[++i]; + else if (std::strcmp(argv[i], "--rawbits") == 0) g_rawBits = true; + else if (std::strcmp(argv[i], "--declare-tso") == 0) g_declareTso = true; + else if (std::strcmp(argv[i], "--content-probe") == 0) g_contentProbe = true; + else if (std::strcmp(argv[i], "--validate") == 0) g_validate = true; + else if (std::strcmp(argv[i], "--foreign-residency") == 0) g_foreignResidency = true; + else if (std::strcmp(argv[i], "--nv12-companion") == 0) g_nv12Companion = true; + else if (std::strcmp(argv[i], "--nv12-staged-companion") == 0) g_stagedCompanion = true; + else if (std::strcmp(argv[i], "--companion-foreign") == 0) g_companionForeign = true; + } + + std::printf("Encoder-ext INPUT FORMAT *ENCODE* MATRIX -- library-owned device\n"); + std::printf("================================================================\n"); + std::printf("pattern: TL(253,0,0) TR(0,253,0) BL(0,0,252) BR(255,255,255)\n"); + std::printf("colour : matrixCoefficients=1 (BT.709), videoFullRange=FALSE (studio)\n"); + std::printf("frames : %u per format\n", kFrames); + if (g_stagedCompanion) { + std::printf("companion: LINEAR/TRANSFER_SRC NV12 descriptor + %u NV12 frames,\n" + " on the NV12-declared row (arm A, no filter) AND on each\n" + " RGBA-declared row (arm B, filter present). Reports the\n" + " staged-input SUBMIT ENGINE for each -- the queue-family\n" + " move this widening causes on the shipping NV12 lane.\n", + kFrames); + } + if (g_nv12Companion) { + std::printf("companion: NV12 descriptor + %u NV12 frames on each " + "RGBA-DECLARED session (the shape Chromium now builds)\n", + kFrames); + } + std::printf("declared input layout: %s\n\n", + g_declareTso ? "TRANSFER_SRC_OPTIMAL where usage permits " + "(exercises the filter-arm restore)" + : "GENERAL"); + std::printf("declared residency : %s\n\n", + g_foreignResidency + ? "FOREIGN (drives the filter-arm and Path A releases)" + : "LOCAL"); + + size_t nSession = 0, nEncoded = 0, nAbandoned = 0; + // Rows the DEVICE refused, kept apart from failures because they are not + // one -- and apart from silence, because a new arm that is only ever + // silent is a new arm that never ran. See IsDeviceLimitedInit(). + size_t nDeviceLimited = 0, nAv1Unverified = 0, nSelected = 0; + for (size_t i = 0; i < kNumRows; i++) { + const Row& row = kRows[i]; + if (only && std::strcmp(only, row.shortName) != 0) continue; + if (onlyCodec && std::strcmp(onlyCodec, CodecName(row.codec)) != 0) continue; + nSelected++; + std::printf("[%zu/%zu] %s\n", i + 1, kNumRows, row.name); + std::fflush(stdout); + + const EncResult r = RunRow(row, outDir); + + if (!r.sessionInit) { + rowResults.push_back(r.initResult); + const bool noDriver = (r.initResult == VK_ERROR_INCOMPATIBLE_DRIVER); + const bool deviceLimited = IsDeviceLimitedInit(r.initResult); + + // THE AV1 ARM'S VERDICT, which is the half of this change that is + // worth more than the rows themselves. "No session" is free; it + // is what a new codec arm reports on the day it is added and on + // every day after, whether or not it works. So for AV1 the claim + // is put to the device -- see ProbeAv1EncodeSupport(). + if ((row.codec == VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR) && + !noDriver) { + ProbeAv1EncodeSupport(); + if (g_av1Probe == 1) { + g_failures++; + std::printf(" RESULT: FAIL -- no session " + "(InitializeExt -> %d), but the device DOES " + "advertise AV1 encode [%s]. That is a refusal, " + "not a device limitation, and calling it one " + "is how this arm would pass without ever " + "encoding a frame.\n\n", + (int)r.initResult, g_av1ProbeDetail.c_str()); + } else if (g_av1Probe == 0) { + if (deviceLimited) { + nDeviceLimited++; + std::printf(" RESULT: DEVICE-LIMITED -- this " + "hardware has no AV1 encode " + "(InitializeExt -> %d; %s). Not a " + "failure.\n\n", + (int)r.initResult, g_av1ProbeDetail.c_str()); + } else { + g_failures++; + std::printf(" RESULT: FAIL -- no session " + "(InitializeExt -> %d). The device does " + "not advertise AV1 encode, but %d is not " + "a device-limitation code either: this is " + "a library refusal wearing an environment " + "costume.\n\n", + (int)r.initResult, (int)r.initResult); + } + } else { + nAv1Unverified++; + std::printf(" RESULT: UNVERIFIED -- no session " + "(InitializeExt -> %d) and the AV1 capability " + "probe could not run (%s), so whether this is " + "a device limitation is UNJUDGED.\n\n", + (int)r.initResult, g_av1ProbeDetail.c_str()); + } + continue; + } + + // The nine H.26x rows KEEP the verdict they have always had: no + // session is reported and is not a failure. What is new is that + // the classification is PRINTED, which it was not -- a library + // refusal here is otherwise indistinguishable in the log from an absent + // profile. + // + // NOT TIGHTENED, and that is scoped rather than overlooked. + // Turning a non-device-limited no-session into a failure for + // those nine would change the verdict of a suite in the CI gating + // set on a prediction this change cannot measure -- the reference + // host is not available to this work, and a red suite on it + // blocks everyone. The residual is real and is one measured A4000 + // run away from closing exactly the way the AV1 arm above already + // has. + std::printf(" RESULT: no session (InitializeExt -> %d)%s\n\n", + (int)r.initResult, + noDriver ? " -- no usable driver" + : (deviceLimited + ? " -- DEVICE-LIMITED" + : " -- NOT a device-limitation code, " + "i.e. a library refusal")); + if (deviceLimited) nDeviceLimited++; + continue; + } + nSession++; + std::printf(" registered=%d status=%d ROUTED=%s planeStorage=%d storageRead=%d\n", + (int)r.registered, (int)r.regStatus, PathName(r.path), + (int)r.planeStorageViews, (int)r.storageReadView); + if (r.abandoned) { + nAbandoned++; + std::printf(" ABANDONED (safety): %s\n\n", r.abandonReason.c_str()); + continue; + } + if (!r.uploaded) { + std::printf(" RESULT: upload failed: %s\n\n", r.uploadError.c_str()); + g_failures++; + continue; + } + std::printf(" submitted=%u retrieved=%u bytes=%llu lastSubmitStatus=%d deviceLost=%d\n", + r.submitted, r.retrieved, (unsigned long long)r.bytes, + r.lastSubmitStatus, (int)r.deviceLost); + std::printf(" filter PRE-flush : created=%d dispatch=%llu stagedCopies=%llu\n", + (int)r.filterCreatedPre, (unsigned long long)r.filterDispatchPre, + (unsigned long long)r.stagedCopiesPre); + std::printf(" filter POST-flush: created=%d dispatch=%llu stagedCopies=%llu\n", + (int)r.filterCreatedPost, (unsigned long long)r.filterDispatchPost, + (unsigned long long)r.stagedCopiesPost); + + // G-07: THE ADVERTISED LIST'S LANE QUALIFICATION, AS A PAIR. + // + // The header says a SUBOPTIMAL entry is taken only through + // RegisterImageResource + SubmitRegisteredFrame. The registered half + // is this row's own encode, above. This is the other half, and it is + // the DIRECT rows that make it mean anything: they take the same + // unregistered call, get past the same format gate, and fail on the + // missing image with a DIFFERENT code. A build that refused every + // unregistered submit would satisfy the filter rows alone. + if (r.unregisteredSubmitTried) { + const bool viaFilter = + (r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER); + const bool refusedOnFormat = + (r.unregisteredSubmit == VK_ERROR_FORMAT_NOT_SUPPORTED); + if (viaFilter != refusedOnFormat) { + g_failures++; + std::printf(" VERDICT: FAIL -- unregistered submit of " + "this row's own format returned %d; a %s row " + "must%s be refused with " + "VK_ERROR_FORMAT_NOT_SUPPORTED\n", + (int)r.unregisteredSubmit, PathName(r.path), + viaFilter ? "" : " NOT"); + } else { + std::printf(" unregistered lane: %s route -> %d (%s)\n", + PathName(r.path), (int)r.unregisteredSubmit, + viaFilter + ? "refused on format, as the registered-lane " + "qualification states" + : "past the format gate, refused below it"); + } + } + if (r.contentChained) { + std::printf(" CONTENT PROBE: armEcho=%s(gen=%u) final=%s " + "probed=%u damaged=%u armedOutstanding=%u " + "meanY=%u meanU=%u meanV=%u (Q8)\n", + ContentStateName(r.contentArmEcho), + r.contentArmGeneration, + ContentStateName(r.contentFinalState), + r.contentProbedCount, r.contentDamagedCount, + r.contentArmedCount, + r.contentMeanY, r.contentMeanU, r.contentMeanV); + // THE VERDICT, and it is scoped to the rows that can carry it. + // + // A DIRECT row is the whole point: it is block-linear plus + // VIDEO_ENCODE_SRC, so it is the class D2 unblinded, and it is + // the ONLY class that needs the staged detour to be probed at + // all. A STAGED row already passes through the capture site + // without any detour, so it proves the plumbing but not the + // detour. A FILTER row is a storage read with no copy, so + // NOT_APPLICABLE is its correct answer and anything else is the + // arm predicate having regressed into "arm everything". + // WHAT THIS ROW SHOULD ANSWER, computed from the same two facts + // the library's arm predicate reads -- deliberately restated here + // rather than imported, so a change to the predicate has to be + // made twice and cannot silently redefine its own test. + // + // FILTER route -> a storage read, never a copy, so there is no + // capture site at all. + // not 8-bit 420 -> the scorer reads plane means byte-wise and + // would read the wrong half of a 10/12-bit + // word, so there are no bytes it can judge. + // + // The P010 row is the reason this is here. Before the format + // clause was added to the arm predicate it echoed ARMED and then + // scored nothing, and this suite reported that as a failure -- + // which is how the defect was found. + const bool probeScorableFormat = + (row.format == VK_FORMAT_G8_B8R8_2PLANE_420_UNORM); + const bool expectArm = + (r.path != VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER) && + probeScorableFormat; + if (!expectArm) { + if (r.contentArmEcho != + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_APPLICABLE) { + g_failures++; + std::printf(" VERDICT: FAIL -- a registration the " + "probe cannot ride (%s route, format %s) " + "echoed %s, not NOT_APPLICABLE. An ARMED echo " + "is a PROMISE of a verdict; promising one " + "here leaves the registration ARMED forever " + "reporting NOT_EVALUATED, which is an " + "observable that cannot fail.\n", + PathName(r.path), + probeScorableFormat ? "scorable" + : "not 8-bit 420", + ContentStateName(r.contentArmEcho)); + } else { + std::printf(" VERDICT: content probe correctly " + "REFUSED this registration up front (%s " + "route, format %s)\n", + PathName(r.path), + probeScorableFormat ? "scorable" + : "not 8-bit 420"); + } + } else if (r.contentArmEcho != + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED) { + g_failures++; + std::printf(" VERDICT: FAIL -- a %s registration echoed " + "%s, not ARMED. The arm predicate is reading " + "something other than 'is a transfer-readable " + "copy of the producer's pixels reachable'.\n", + PathName(r.path), + ContentStateName(r.contentArmEcho)); + } else if (r.contentProbedCount == 0) { + // THE ONE THIS MODE EXISTS FOR. + g_failures++; + std::printf(" VERDICT: FAIL -- the registration ARMED " + "and NOTHING WAS EVER SCORED (probed=0, " + "armedOutstanding=%u) after %u submitted and %u " + "retrieved frames. On a %s row that means the " + "one-frame staged detour did not fire or the " + "capture site was not reached. This is the exact " + "state the device-free suite reports as a PASS.\n", + r.contentArmedCount, r.submitted, r.retrieved, + PathName(r.path)); + } else if (r.contentArmedCount != 0) { + g_failures++; + std::printf(" VERDICT: FAIL -- the session ended with %u " + "registration(s) still armed and unscored.\n", + r.contentArmedCount); + } else if (r.contentFinalState != + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_CLEAN) { + // NOT automatically a library bug -- DAMAGED_* here would be + // the driver defect firing on a locally-allocated VkImage, + // which has never been observed (the defect is GBM-import + // gated). Either way it is not the expected result of + // encoding a four-quadrant colour bar, so it is reported + // loudly rather than absorbed. + g_failures++; + std::printf(" VERDICT: FAIL -- scored %s on an image the " + "harness filled with a four-quadrant colour bar. " + "A colour bar has no dead plane, so either the " + "capture read the wrong image or the predicate " + "regressed.\n", + ContentStateName(r.contentFinalState)); + } else { + std::printf(" VERDICT: content probe OK -- ARMED at " + "registration, and the capture RAN and scored " + "CLEAN on a %s row (probed=%u, nothing left " + "outstanding)\n", + PathName(r.path), r.contentProbedCount); + } + } + if (r.companionAttempted) { + std::printf(" NV12 COMPANION on this RGBA-declared session:\n"); + std::printf(" registered=%d status=%d ROUTED=%s submitted=%u\n", + (int)r.companionRegistered, (int)r.companionRegStatus, + PathName(r.companionPath), r.companionSubmitted); + std::printf(" dispatch after RGBA frames = %llu (stagedCopies=%llu)\n", + (unsigned long long)r.dispatchAfterPrimary, + (unsigned long long)r.stagedAfterPrimary); + std::printf(" dispatch after NV12 frames = %llu (stagedCopies=%llu)\n", + (unsigned long long)r.dispatchAfterCompanion, + (unsigned long long)r.stagedAfterCompanion); + std::printf(" STAGED-INPUT SUBMIT ENGINE = %s (VK_QUEUE flags 0x%x) " + "queueFamilyIndex=%u\n", + SubmitEngineName(r.submitFlagsCompanion), + r.submitFlagsCompanion, r.submitFamilyCompanion); + std::printf(" staging acquires: foreign=%llu local=%llu " + "(after primary: foreign=%llu local=%llu)\n", + (unsigned long long)r.foreignAcqCompanion, + (unsigned long long)r.localAcqCompanion, + (unsigned long long)r.foreignAcqPrimary, + (unsigned long long)r.localAcqPrimary); + if (!r.companionError.empty()) { + std::printf(" companionError: %s\n", r.companionError.c_str()); + } + // THE TWO VERDICTS, each a failure and not a note. + if (!r.companionRegistered) { + g_failures++; + std::printf(" VERDICT: FAIL -- the session refused an NV12 descriptor, " + "so declaring RGBA COSTS the direct lane\n"); + } else if (r.companionPath != r.companionExpectPath) { + g_failures++; + std::printf(" VERDICT: FAIL -- NV12 routed %s, not %s\n", + PathName(r.companionPath), + PathName(r.companionExpectPath)); + } else if (r.stagedCompanion && + (r.stagedAfterCompanion - r.stagedAfterPrimary) != kFrames) { + g_failures++; + std::printf(" VERDICT: FAIL -- %llu of %u NV12 frames took the staging " + "COPY arm; a staged companion that does not copy is measuring " + "nothing\n", + (unsigned long long)(r.stagedAfterCompanion - + r.stagedAfterPrimary), + kFrames); + } else if (r.companionSubmitted != kFrames) { + g_failures++; + std::printf(" VERDICT: FAIL -- only %u of %u NV12 frames submitted\n", + r.companionSubmitted, kFrames); + } else if (r.dispatchAfterCompanion != r.dispatchAfterPrimary) { + g_failures++; + std::printf(" VERDICT: FAIL -- the NV12 frames DISPATCHED the filter " + "(%llu -> %llu); routing is per-session, not per-frame\n", + (unsigned long long)r.dispatchAfterPrimary, + (unsigned long long)r.dispatchAfterCompanion); + } else if (r.dispatchAfterPrimary == 0 && row.arm == ARM_FILTER_RGBA) { + g_failures++; + std::printf(" VERDICT: FAIL -- the RGBA frames did NOT dispatch the " + "filter, so the companion comparison proves nothing\n"); + } else if (g_companionForeign && + (r.foreignAcqCompanion - r.foreignAcqPrimary) != kFrames) { + g_failures++; + std::printf(" VERDICT: FAIL -- --companion-foreign asked for %u FOREIGN " + "acquires and got %llu; with no foreign acquire there is no " + "foreign RELEASE, so the queue-family claim is untested\n", + kFrames, + (unsigned long long)(r.foreignAcqCompanion - + r.foreignAcqPrimary)); + } else if (r.stagedCompanion && row.arm == ARM_DIRECT) { + // ARM A. There is no filter on this session and there is not + // meant to be, so "RGBA dispatched" is not a claim this row can + // make. What it establishes is the CONTROL VALUE of the submit + // engine, and it is a failure if it is not the encode or + // transfer family -- that would mean the control and the + // subject were never different and the A/B measures nothing. + if (r.submitFlagsCompanion != VK_QUEUE_VIDEO_ENCODE_BIT_KHR && + r.submitFlagsCompanion != VK_QUEUE_TRANSFER_BIT) { + g_failures++; + std::printf(" VERDICT: FAIL -- arm A (no filter) submitted staged " + "input on %s; expected VIDEO_ENCODE or TRANSFER\n", + SubmitEngineName(r.submitFlagsCompanion)); + } else { + std::printf(" VERDICT: PASS (arm A control) -- no filter, %u staged " + "copies, 0 dispatches, staged input on %s family %u\n", + kFrames, SubmitEngineName(r.submitFlagsCompanion), + r.submitFamilyCompanion); + } + } else { + std::printf(" VERDICT: PASS -- RGBA dispatched %llu, NV12 dispatched 0, " + "one session, per-frame routing\n", + (unsigned long long)r.dispatchAfterPrimary); + } + } + // THE VERDICT MUST READ EVERY SIGNAL THE ROW ALREADY PRODUCED. + // "bytes > 0 && retrieved > 0" alone passed a row that lost the GPU + // mid-encode, that dropped nine of twelve frames in the assembly + // pipeline, that broke out of the submit loop on an error status, or + // that wrote its bitstream nowhere. Each of those was measured and + // printed one line above and then discarded. + if (r.deviceLost) { + g_failures++; + std::printf(" RESULT: DEVICE LOST during encode -- bytes=%llu is not a pass\n\n", + (unsigned long long)r.bytes); + } else if (r.fileOpenFailed) { + g_failures++; + std::printf(" RESULT: could not open the output file; nothing was written " + "for a decoder to judge\n\n"); + } else if (r.retrieved < r.submitted) { + g_failures++; + std::printf(" RESULT: LOST FRAMES -- submitted=%u retrieved=%u after the " + "bounded wait\n\n", r.submitted, r.retrieved); + } else if (r.submitted < kFrames) { + g_failures++; + std::printf(" RESULT: SUBMIT ABORTED after %u of %u frames " + "(lastSubmitStatus=%d)\n\n", + r.submitted, kFrames, r.lastSubmitStatus); + } else if (r.bytes > 0 && r.retrieved > 0) { + nEncoded++; + std::printf(" RESULT: ENCODED -> %s\n", r.file.c_str()); + // THE DECODE ASSERTION, IN THE DEFAULT CHAIN. Not behind a flag: + // the state it exists to catch -- the compute filter recording its + // dispatches and never executing them -- is green on every other + // observable this suite has, so an opt-in gate would simply not be + // opted into on the run that mattered. + const bool hdrRow = (row.colour != nullptr) && row.colour->hdr10; + std::string qdetail; + // The DECLARED matrix and range, which the gate now needs because + // it compares in the Y'CbCr domain. Every tracked row pins 1 and + // studio range (see the "Colour is PINNED" block), and a row that + // overrides them is honoured here rather than silently judged as + // BT.709. + const uint8_t rowMatrix = + (row.colour != nullptr) ? row.colour->matrix : (uint8_t)1; + const bool rowFullRange = + (row.colour != nullptr) && (row.colour->fullRange == VK_TRUE); + const QuadStatus qs = + hdrRow ? CheckHdrSignalling(r.file, &qdetail) + : CheckQuadrants(r.file, rowMatrix, rowFullRange, + &qdetail); + const char* gateName = hdrRow ? "HDR GATE" : "DECODE GATE"; + int* gated = hdrRow ? &g_hdrGated : &g_quadChecked; + int* failed = hdrRow ? &g_hdrFailed : &g_quadFailed; + if (qs == QUAD_PASS) { + (*gated)++; + std::printf(" %s: PASS --%s\n\n", gateName, qdetail.c_str()); + } else if (qs == QUAD_NO_DECODER) { + // NOT A PASS AND NOT SILENT. The run is downgraded to SKIPPED + // at exit; see the bottom of main(). + g_quadNoDecoder = true; + std::printf(" %s: NOT RUN -- %s. This row's colour " + "is UNJUDGED.\n\n", gateName, qdetail.c_str()); + } else { + (*gated)++; + (*failed)++; + g_failures++; + std::printf(" %s: FAIL -- %s\n\n", gateName, qdetail.c_str()); + } + } else { + g_failures++; + std::printf(" RESULT: NO BITSTREAM (registered and routed, produced nothing)\n\n"); + } + } + + std::printf("================================================================\n"); + // decodeGated IS PART OF THE BAR, not decoration. A run in which the + // colour assertion did not execute on every encoded row is not the same + // run as one in which it did, and without this number on the summary line + // the two are indistinguishable from a CI log. + // deviceLimited and av1Unverified are APPENDED, not inserted: every + // existing CI line that greps the first six fields still matches. They + // are on the bar for the same reason decodeGated is -- a run in which an + // arm was refused by the device is not the same run as one in which it + // encoded, and without the number on this line the two are + // indistinguishable from a log. + // hdrGated / hdrFailed are APPENDED for the same reason deviceLimited and + // av1Unverified were: every CI line that greps the first eight fields + // still matches, and a run in which the HDR row's SEI was not read back + // is not the same run as one in which it was. decodeGated does NOT count + // the HDR row -- it takes a different gate, and folding two different + // assertions into one counter is how "8 of 9 rows were judged" would read + // as green. + std::printf("sessions=%zu encoded=%zu abandoned=%zu failures=%d " + "decodeGated=%d decodeFailed=%d deviceLimited=%zu " + "av1Unverified=%zu hdrGated=%d hdrFailed=%d\n", + nSession, nEncoded, nAbandoned, g_failures, + g_quadChecked, g_quadFailed, nDeviceLimited, nAv1Unverified, + g_hdrGated, g_hdrFailed); + if (!g_av1ProbeDetail.empty()) { + std::printf("av1 capability probe: %s\n", g_av1ProbeDetail.c_str()); + } + if (g_ffmpegProbe == 1) { + std::printf("decoder: %s\n", g_ffmpegBanner.c_str()); + } + + // THIS USED TO `return 0;` UNCONDITIONALLY, WHICH MADE THE ONLY TEST IN + // THE SUITE THAT ENCODES REAL PIXELS INCAPABLE OF GOING RED. + // + // g_failures was counted at two sites -- an upload failure and a row that + // registered, routed, and produced NO BITSTREAM -- printed on the line + // above, and then discarded. The second of those is the exact failure this + // file's own header says it exists to catch, the one the routing test + // cannot see. It carries LABELS "gpu" and is therefore in the CI gating + // set, so a green run here has meant nothing. + // + // THE SKIP MUST BE JUSTIFIED BY A DEVICE SIGNAL, NOT BY AN AGGREGATE. + // + // An earlier version skipped on nSession == 0 alone. That is not "the + // GPU-less host": it is also "the library refused every session", and the + // two are trivially confusable. Configure with + // -DBUILD_ENCODER_COMPUTE_FILTER=OFF -- a supported option -- and + // InitializeExt hard-refuses every row whose format needs a conversion; + // nSession collapses on a box with a working GPU, and a compute-shader + // regression, which is the single thing this suite exists to catch, would + // have reported SKIPPED and left CI green. Replacing "cannot fail" with + // "can be silenced" is not progress. + // + // VK_ERROR_INCOMPATIBLE_DRIVER is the loader's answer when no usable + // driver is present -- it is what this binary returns for every row on a + // GPU-less host, measured. Any OTHER failure came from the library with a + // driver in hand and is a verdict, not an environment fact. + if (nSession == 0) { + bool everyRowSaysNoDriver = !rowResults.empty(); + // THE SECOND ENVIRONMENT CASE, which `--codec av1` on hardware + // without AV1 produces: + // every selected row was refused BY THE DEVICE. That is a skip for + // the same reason an absent driver is, and treating it as a failure + // would make the AV1 ctest entry red on every pre-Blackwell box. + // + // IT DOES NOT WEAKEN THE GUARD THE ORIGINAL COMMENT EXISTS FOR. + // Configuring -DBUILD_ENCODER_COMPUTE_FILTER=OFF still turns this + // suite RED, not skipped: that refusal comes out of the config binder + // as VK_ERROR_INITIALIZATION_FAILED, which IsDeviceLimitedInit() + // deliberately does not admit. And the skip requires g_failures == 0 + // and nAv1Unverified == 0, so it cannot launder a verdict or an + // unjudged row -- the same invariant the ffmpeg skip below keeps. + bool everyRowRefusedByDevice = + !rowResults.empty() && (g_failures == 0) && (nAv1Unverified == 0); + for (const VkResult rr : rowResults) { + if (rr != VK_ERROR_INCOMPATIBLE_DRIVER) { + everyRowSaysNoDriver = false; + if (!IsDeviceLimitedInit(rr)) { + everyRowRefusedByDevice = false; + } + } + } + if (everyRowSaysNoDriver) { + std::printf("SKIP: no usable Vulkan driver -- every row returned " + "VK_ERROR_INCOMPATIBLE_DRIVER\n"); + return 77; + } + if (everyRowRefusedByDevice) { + std::printf("SKIP: every selected row was refused by the DEVICE, " + "not by the library -- %zu of %zu selected row(s) " + "device-limited. There is a working driver here; it " + "does not implement what these rows ask for.\n", + nDeviceLimited, nSelected); + if (!g_av1ProbeDetail.empty()) { + std::printf(" %s\n", g_av1ProbeDetail.c_str()); + } + return 77; + } + std::printf("FAILED: no session was created, and at least one row " + "failed for a reason that is neither a missing driver nor " + "a device limitation -- that is a library refusal, not an " + "absent GPU\n"); + for (size_t i = 0; i < rowResults.size(); i++) { + std::printf(" row %zu InitializeExt -> %d\n", i, (int)rowResults[i]); + } + return 1; + } + + if (g_failures != 0) { + return 1; + } + + // A GATE THAT PASSES WHEN ITS DECODER IS MISSING IS ANOTHER CANNOT-FAIL + // TEST. Every counter in this suite is green with the compute filter dead, + // so "we encoded seven files and looked at none of them" must not be + // reported as a pass. It is reported as SKIPPED -- 77, which CMakeLists.txt + // already declares as this test's SKIP_RETURN_CODE -- with the reason on + // stdout. Real failures above still return 1 first, so this cannot be used + // to launder one. + if (g_quadNoDecoder) { + std::printf("SKIP: ffmpeg is not on PATH, so the four-quadrant decode " + "assertion did not run on %zu encoded row(s). That assertion " + "is the ONLY observable in this suite that can see the " + "compute filter stop executing -- with the filter's four " + "vkCmdDispatch calls deleted, every counter above still " + "reads green. An unjudged run is not a pass.\n", nEncoded); + return 77; + } + + // Belt and braces: rows encoded but nothing graded means the wiring above + // was bypassed, which is the failure this whole change exists to remove. + if (nEncoded != 0 && (g_quadChecked + g_hdrGated) == 0) { + std::printf("FAILED: %zu row(s) encoded and neither the four-quadrant " + "decode assertion nor the HDR gate ran on any of them\n", + nEncoded); + return 1; + } + return 0; +} diff --git a/vk_video_encoder/test/encoder-ext-format-matrix/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-format-matrix/CMakeLists.txt new file mode 100644 index 00000000..f0d84594 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-format-matrix/CMakeLists.txt @@ -0,0 +1,114 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_format_matrix_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The STATIC library, for the same reason the sibling taxonomy test links it: +# the internal header's free functions -- the format taxonomy and +# VkEncProbeResource, which is how this test reads a slot's resolved input +# path -- are deliberately not exported from libvkvideo-encoder.so. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +# Must branch the same way the library did -- an RGBA session cannot route +# FILTER at all in a build without the filter, and a test that assumed it +# present would fail for the wrong reason. +if(BUILD_ENCODER_COMPUTE_FILTER) + target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED) +endif() + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add test. +# +# Every row creates a real encode session and a real VkImage on the +# LIBRARY-OWNED device, so this is a gpu-label test: it exits 77 (SKIPPED) +# when no encode-capable Vulkan device is present, and 1 when an assertion +# fails. There is no arm of it that is meaningful without a device -- the +# device-free half of the taxonomy is already covered by +# EncoderExtInputFormatTaxonomy next door. +enable_testing() +# +# The matrix is walked by PROFILE FAMILY, one invocation each, because a row +# costs a whole encoder session and a process cannot hold an unbounded number +# of them. Splitting is not a convenience: past that limit a row reports "this +# device has no such encode profile" for a profile the device does support, +# and the suite would go green over a wrong finding. +# +# GATED ON VALIDATION. Walking the matrix creates a session, an image pool and +# an encode per row, so a spec violation introduced anywhere on the input path +# surfaces here first. vvs_add_validation_gated_test() forces the validation +# layer on and fails the arm that produced a message; no ceiling is declared, +# so any message at all is a failure. +vvs_add_validation_gated_test(EncoderExtInputFormatMatrix + TARGET ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtInputFormatMatrix PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +# Semi-planar 4:4:4, which belongs to a different profile in both codecs +# (H.264 High 4:4:4 Predictive, H.265 Range Extensions) and so to a different +# family. Carries the same two controls, so a refusal is observed on the same +# engine that ran the 4:4:4 rows. +vvs_add_validation_gated_test(EncoderExtInputFormatMatrix444 + TARGET ${PROJECT_NAME} + ARGS --444) +set_tests_properties(EncoderExtInputFormatMatrix444 PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +message(STATUS "encoder_ext_format_matrix_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-format-matrix/src/main.cpp b/vk_video_encoder/test/encoder-ext-format-matrix/src/main.cpp new file mode 100644 index 00000000..fc720139 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-format-matrix/src/main.cpp @@ -0,0 +1,1273 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * INPUT-FORMAT MATRIX, on the LIBRARY-OWNED device. + * + * WHAT THIS ANSWERS. VkEncClassifyInput is keyed on a PAIR -- a format and a + * colour model -- and the set of accepted pairs is DERIVED from the + * multi-planar Y'CbCr format table rather than listed, so it is described here + * by its rule and not copied: semi-planar at 8 or 10 bits is + * ENCODABLE_DIRECT; 3-plane, and semi-planar at 12 bits, is + * ENCODABLE_VIA_FILTER wherever the table names a semi-planar sibling at the + * same depth and subsampling to convert into; the packed 4:4:4 layouts AYUV + * and Y410 are ENCODABLE_VIA_FILTER when declared + * VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; and three RGB spellings + * (R8G8B8A8_UNORM, B8G8R8A8_UNORM, A8B8G8R8_UNORM_PACK32) are + * ENCODABLE_VIA_FILTER. At the table this tree carries that is 23 pairs over + * 22 distinct enumerants -- a count stated as a measurement of today's table + * and not as a contract, since the derivation moves with the table. The pair + * count exceeds the enumerant count because R8G8B8A8_UNORM carries two + * accepted readings, AYUV and RGBA, and only the declared colour model + * separates them. + * "Claims to support" is a statement about that rule. Whether a format + * REGISTERS, and onto WHICH INPUT PATH, is a statement about a device, a + * session and a descriptor -- three things the rule does not see. NOTHING HERE + * WALKS A 4:2:2 OR A 3-PLANE 4:4:4 ROW: the derived set admits them and no row + * below encodes one, so this file is not evidence about them. + * + * This walks eleven rows against a real device and reports both, with the + * denominator, so a format that does NOT encode is a recorded result rather + * than a missing row: by default the nine 4:2:0-and-RGB rows plus the two + * controls, and under --444 the two 4:4:4 rows plus the same two controls. + * No ROW here carries a packed Y'CbCr reading: every row is a session input + * format, R8G8B8A8_UNORM appears once and under its RGB reading, and + * A2B10G10R10_UNORM_PACK32 is not a row at all. The file does vary the + * declared colour model -- at registration, and across Reconfigure, including + * the AYUV reading of R8G8B8A8_UNORM -- but it varies it to check what is + * REFUSED and how, not to encode from it. So no packed reading is exercised as + * an input format by this matrix, which is the narrower thing the rows above + * can be read for. + * + * WHAT IT CAN FAIL ON -- stated up front, because a suite whose assertions + * cannot discriminate has shipped on this project before: + * + * - It asserts the RGBA rows REGISTER and resolve to + * VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER. A library that refuses them + * answers ERROR_CONVERSION_REQUIRED at registration and this fails. + * - It asserts the routing from VkEncProbeResource -- the slot's own + * resolved inputPath -- and NOT from the config flag, from + * ComputeFilterActive(), or from a log line. Those can all be true of a + * registration that resolved to STAGED. + * - It asserts the gate still DISCRIMINATES: the same RGBA image registered + * with a descriptor that does not grant VK_IMAGE_USAGE_STORAGE_BIT must + * still be refused CONVERSION_REQUIRED. A fix that simply deleted the + * refusal would pass every "RGBA registers" assertion and fail this one. + * - It asserts the multi-planar arm is UNCHANGED: a 3-plane descriptor + * without MUTABLE_FORMAT|EXTENDED_USAGE|STORAGE is still refused, and + * with them still resolves to FILTER through planeStorageViews, not + * through the new single-plane fact. + * - It asserts that Reconfigure REFUSES a config whose inputColorModel + * resolves to the other model, with two controls that must still be + * accepted: a rate-control change, and the same model stated explicitly + * rather than left to the format. A library that discards the field + * answers VK_SUCCESS to all three. + * + * WHY REGISTRATION AND ROUTING ARE REPORTED SEPARATELY. Registration tests + * the DESCRIPTOR; routing tests the VIEW that was actually built. They can + * disagree, and the disagreement is not academic: it is how a filter-only + * format could reach the staging copy, which for such a format is a measured + * VK_ERROR_DEVICE_LOST. Every row prints both. + */ + +#include "vulkan_video_encoder_ext.h" +#include "vulkan_video_encoder_ext_internal.h" + +#include "vk_video/vulkan_video_codec_h264std.h" +#include "vk_video/vulkan_video_codec_h265std.h" + +#include + +#include +#include +#include +#include +#include + +namespace { + +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; + +int g_failures = 0; +int g_checks = 0; + +void Check(const char* row, bool ok, const char* what, + const std::string& detail) +{ + g_checks++; + if (!ok) { + g_failures++; + std::printf(" FAIL [%s] %s\n %s\n", row, what, + detail.c_str()); + } else { + std::printf(" ok [%s] %s\n", row, what); + } +} + +std::string U32(unsigned long long v) { return std::to_string(v); } + +// --------------------------------------------------------------------------- +// Vulkan entry points, loaded off the LIBRARY's instance/device. Nothing here +// creates a device of its own: the whole point is the library-owned one. +// --------------------------------------------------------------------------- +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = + nullptr; + PFN_vkGetPhysicalDeviceProperties GetPhysicalDeviceProperties = nullptr; + PFN_vkGetPhysicalDeviceFormatProperties GetPhysicalDeviceFormatProperties = + nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) { + lib = dlopen("libvulkan.so", RTLD_NOW); + } + if (lib == nullptr) { + std::printf(" ERROR: dlopen(libvulkan) failed: %s\n", dlerror()); + return false; + } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) { + return false; + } + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + fns->GetPhysicalDeviceProperties = + (PFN_vkGetPhysicalDeviceProperties)gipa( + instance, "vkGetPhysicalDeviceProperties"); + fns->GetPhysicalDeviceFormatProperties = + (PFN_vkGetPhysicalDeviceFormatProperties)gipa( + instance, "vkGetPhysicalDeviceFormatProperties"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr) && + (fns->GetPhysicalDeviceProperties != nullptr) && + (fns->GetPhysicalDeviceFormatProperties != nullptr); +} + +// --------------------------------------------------------------------------- +// The nine formats the library claims, in taxonomy order, plus two controls. +// --------------------------------------------------------------------------- +enum Arm { ARM_DIRECT, ARM_FILTER_YCBCR, ARM_FILTER_RGBA, ARM_CONTROL }; + +// Rows are GROUPED by the video PROFILE their bit depth needs, and every +// group's DIRECT member is its control. "This format does not encode" then +// splits into two different findings that must not be conflated: +// +// * the DEVICE has no such profile -- the group's DIRECT control fails to +// initialize too, and the library never got a say; or +// * the LIBRARY refuses it -- the control initializes and the filter +// member does not, or registers and routes wrongly. +// +// Measured on the A4000 the first time this ran: all four 10/12-bit rows +// failed, and every one of them was +// GetPhysicalDeviceVideoCapabilitiesKHR answering +// VK_ERROR_VIDEO_PROFILE_FORMAT_NOT_SUPPORTED_KHR. Attributing that to the +// registration gate would have been simply wrong. +// The 4:4:4 groups are separate profiles, not a bit-depth variant of the +// 4:2:0 ones: H.264 High 4:4:4 Predictive and H.265 Range Extensions are +// what a semi-planar 4:4:4 encode source belongs to, and a device may +// expose one subsampling and not the other. +enum Group { + G_8BIT = 0, + G_10BIT = 1, + G_12BIT = 2, + G_444_8 = 3, + G_444_10 = 4, + G_NONE = 5 +}; + +// The video profile a group's images and sessions belong to. Everything the +// profile list needs is a function of the group, so a row states its group +// and nothing else. +VkVideoChromaSubsamplingFlagBitsKHR GroupSubsampling(Group g) +{ + return ((g == G_444_8) || (g == G_444_10)) + ? VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR + : VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; +} + +VkVideoComponentBitDepthFlagBitsKHR GroupDepth(Group g) +{ + switch (g) { + case G_10BIT: + case G_444_10: return VK_VIDEO_COMPONENT_BIT_DEPTH_10_BIT_KHR; + case G_12BIT: return VK_VIDEO_COMPONENT_BIT_DEPTH_12_BIT_KHR; + default: return VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + } +} + +struct Row { + const char* name; + VkFormat format; + Arm arm; + Group group; + VkVideoCodecOperationFlagBitsKHR codec; + const char* codecName; +}; + +// H.265 carries the 10- and 12-bit rows: H.264 High has no 10-bit 4:2:0 +// profile at all on this stack, so testing depth through it measures the +// codec rather than the format. +#define H264 VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR, "H264" +#define H265 VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR, "H265" + +const Row kRows[] = { + {"NV12 G8_B8R8_2PLANE_420_UNORM", + VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ARM_DIRECT, G_8BIT, H264}, + {"P010 G10X6_B10X6R10X6_2PLANE_420...", + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16, ARM_DIRECT, + G_10BIT, H265}, + {"P012 G12X4_B12X4R12X4_2PLANE_420...", + VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, ARM_FILTER_YCBCR, + G_12BIT, H265}, + {"I420 G8_B8_R8_3PLANE_420_UNORM", + VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, ARM_FILTER_YCBCR, G_8BIT, H264}, + {"I420-10 G10X6_B10X6_R10X6_3PLANE_420", + VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, ARM_FILTER_YCBCR, + G_10BIT, H265}, + {"I420-12 G12X4_B12X4_R12X4_3PLANE_420", + VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, ARM_FILTER_YCBCR, + G_12BIT, H265}, + {"RGBA8 R8G8B8A8_UNORM", + VK_FORMAT_R8G8B8A8_UNORM, ARM_FILTER_RGBA, G_8BIT, H264}, + {"BGRA8 B8G8R8A8_UNORM", + VK_FORMAT_B8G8R8A8_UNORM, ARM_FILTER_RGBA, G_8BIT, H264}, + {"ABGR8-packed A8B8G8R8_UNORM_PACK32", + VK_FORMAT_A8B8G8R8_UNORM_PACK32, ARM_FILTER_RGBA, G_8BIT, H264}, + // Semi-planar 4:4:4. Same layout and same depth as the 4:2:0 pair above, + // a different profile: these belong to H.264 High 4:4:4 Predictive and + // H.265 Range Extensions, and each is the only member of its group, so + // its own initialization is what separates "this device has no 4:4:4 + // encode profile" from "the library refuses the format". + {"NV24 G8_B8R8_2PLANE_444_UNORM", + VK_FORMAT_G8_B8R8_2PLANE_444_UNORM, ARM_DIRECT, G_444_8, H264}, + {"S410 G10X6_B10X6R10X6_2PLANE_444...", + VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16, ARM_DIRECT, + G_444_10, H265}, + // Controls. Neither is in the taxonomy; both MUST be refused, and by the + // format gate rather than by the arm gates. + {"CONTROL R8G8B8A8_SRGB (excluded)", + VK_FORMAT_R8G8B8A8_SRGB, ARM_CONTROL, G_NONE, H264}, + {"CONTROL R16G16B16A16_SFLOAT (excl)", + VK_FORMAT_R16G16B16A16_SFLOAT, ARM_CONTROL, G_NONE, H264}, +}; +const size_t kNumRows = sizeof(kRows) / sizeof(kRows[0]); + +const char* PathName(VkVideoEncoderExternalInputPath p) +{ + switch (p) { + case VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT: return "DIRECT"; + case VK_VIDEO_EXTERNAL_INPUT_PATH_STAGED: return "STAGED"; + case VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER: return "FILTER"; + default: return "?"; + } +} + +const char* ClassName(VkEncInputFormatClass c) +{ + switch (c) { + case VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT: return "DIRECT"; + case VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER: return "VIA_FILTER"; + default: return "UNSUPPORTED"; + } +} + +struct Img { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; +}; + +bool CreateImage(const DeviceFns& fns, VkPhysicalDevice phys, VkDevice device, + VkFormat format, VkImageTiling tiling, + VkImageUsageFlags usage, VkImageCreateFlags flags, + bool withProfileList, + VkVideoCodecOperationFlagBitsKHR codec, + VkVideoComponentBitDepthFlagBitsKHR depth, + VkVideoChromaSubsamplingFlagBitsKHR subsampling, Img* out) +{ + // The codec-specific profile struct is NOT optional: + // VUID-VkVideoProfileInfoKHR-videoCodecOperation-07181/07182 require it, + // and without it the whole profile-list query fails -- which then shows + // up as -06811 and -02251 on the same create, three VUIDs deep, none of + // them naming the actual omission. + VkVideoEncodeH264ProfileInfoKHR h264Profile{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_PROFILE_INFO_KHR}; + // 4:4:4 is carried by a different profile in both codecs, and the image + // has to name the same one its session will negotiate or the profile + // list describes an image the encoder never sees. + const bool is444 = + (subsampling == VK_VIDEO_CHROMA_SUBSAMPLING_444_BIT_KHR); + h264Profile.stdProfileIdc = + is444 ? STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE + : STD_VIDEO_H264_PROFILE_IDC_HIGH; + VkVideoEncodeH265ProfileInfoKHR h265Profile{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H265_PROFILE_INFO_KHR}; + h265Profile.stdProfileIdc = + is444 ? STD_VIDEO_H265_PROFILE_IDC_FORMAT_RANGE_EXTENSIONS + : (depth == VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR) + ? STD_VIDEO_H265_PROFILE_IDC_MAIN + : STD_VIDEO_H265_PROFILE_IDC_MAIN_10; + + VkVideoProfileInfoKHR profile{VK_STRUCTURE_TYPE_VIDEO_PROFILE_INFO_KHR}; + profile.pNext = + (codec == VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR) + ? (void*)&h264Profile + : (void*)&h265Profile; + profile.videoCodecOperation = codec; + profile.chromaSubsampling = subsampling; + profile.lumaBitDepth = depth; + profile.chromaBitDepth = depth; + VkVideoProfileListInfoKHR profileList{ + VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR}; + profileList.profileCount = 1; + profileList.pProfiles = &profile; + + // MUTABLE_FORMAT obliges a format list naming every view format + // (VUID-VkImageCreateInfo-tiling-02353 / -pNext-01585), and every entry + // must be compatible with a PLANE of this format + // (VUID-VkImageCreateInfo-pNext-10062). So the list is derived from the + // format rather than hardcoded: a 3-plane 420 image's planes are all + // single-component, and offering it R8G8_UNORM is invalid even though + // the library never asks for such a view. + VkFormat viewFormats[4] = {format, VK_FORMAT_UNDEFINED, + VK_FORMAT_UNDEFINED, VK_FORMAT_UNDEFINED}; + uint32_t viewFormatCount = 1; + switch (format) { + case VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM: + viewFormats[viewFormatCount++] = VK_FORMAT_R8_UNORM; + break; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16: + // BOTH spellings. The plane format of a G10X6 image is + // R10X6_UNORM_PACK16, but VkImageResourceView::Create derives its + // plane views as R16_UNORM -- class-compatible (both 16-bit + // single-component) and therefore legal, but a format list naming + // only the plane spelling makes the view VUID-...-pNext-01585. + // Measured, not assumed: that VUID fired here three times before + // R16_UNORM was added. + viewFormats[viewFormatCount++] = VK_FORMAT_R10X6_UNORM_PACK16; + viewFormats[viewFormatCount++] = VK_FORMAT_R16_UNORM; + break; + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16: + viewFormats[viewFormatCount++] = VK_FORMAT_R12X4_UNORM_PACK16; + viewFormats[viewFormatCount++] = VK_FORMAT_R16_UNORM; + break; + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: + case VK_FORMAT_G8_B8R8_2PLANE_444_UNORM: + viewFormats[viewFormatCount++] = VK_FORMAT_R8_UNORM; + viewFormats[viewFormatCount++] = VK_FORMAT_R8G8_UNORM; + break; + default: + break; + } + VkImageFormatListCreateInfo listInfo{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + listInfo.viewFormatCount = viewFormatCount; + listInfo.pViewFormats = viewFormats; + + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + const void* chain = nullptr; + if (withProfileList) { + chain = &profileList; + } + if ((flags & VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT) != 0) { + listInfo.pNext = chain; + chain = &listInfo; + } + ci.pNext = chain; + ci.flags = flags; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = format; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = tiling; + ci.usage = usage; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = (tiling == VK_IMAGE_TILING_LINEAR) + ? VK_IMAGE_LAYOUT_PREINITIALIZED + : VK_IMAGE_LAYOUT_UNDEFINED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + const VkMemoryPropertyFlags want = + (tiling == VK_IMAGE_TILING_LINEAR) + ? (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT) + : (VkMemoryPropertyFlags)VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT; + uint32_t typeIndex = UINT32_MAX; + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + fns.DestroyImage(device, out->image, nullptr); + out->image = VK_NULL_HANDLE; + return false; + } + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + fns.DestroyImage(device, out->image, nullptr); + out->image = VK_NULL_HANDLE; + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != VK_SUCCESS) { + return false; + } + return true; +} + +void DestroyImg(const DeviceFns& fns, VkDevice device, Img* img) +{ + if (img->image != VK_NULL_HANDLE) { + fns.DestroyImage(device, img->image, nullptr); + img->image = VK_NULL_HANDLE; + } + if (img->memory != VK_NULL_HANDLE) { + fns.FreeMemory(device, img->memory, nullptr); + img->memory = VK_NULL_HANDLE; + } +} + +// One row of the matrix. Everything is per-session because the format is a +// SESSION property: the filter's input format and the registration gate +// both read m_encoderConfig. +struct RowResult { + bool sessionInit = false; + bool imageCreated = false; + VkBool32 filterCapableOptimal = VK_FALSE; + VkBool32 filterCapableLinear = VK_FALSE; + VkVideoEncoderStatusCode queryStatus = + VK_VIDEO_ENCODER_STATUS_ERROR_FORMAT_UNSUPPORTED; + VkBool32 querySupported = VK_FALSE; + VkVideoEncoderStatusCode regStatus = + VK_VIDEO_ENCODER_STATUS_ERROR_FORMAT_UNSUPPORTED; + VkVideoEncoderExternalInputPath path = + VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT; + VkBool32 planeStorageViews = VK_FALSE; + VkBool32 storageReadView = VK_FALSE; + // The negative control: the same image with the arm's declaration + // withheld must still be refused. + bool ranWithheld = false; + VkVideoEncoderStatusCode withheldStatus = + VK_VIDEO_ENCODER_STATUS_SUCCESS; + // The COLOUR-MODEL DECLARATION, asked on the same session and the same + // image, so the declaration is the only thing that differs between these + // and the row above. + bool ranDeclared = false; + VkVideoEncoderStatusCode declaredAgreeing = + VK_VIDEO_ENCODER_STATUS_SUCCESS; + VkVideoEncoderStatusCode declaredYcbcr = + VK_VIDEO_ENCODER_STATUS_SUCCESS; + VkVideoEncoderStatusCode declaredYcbcrNoExtent = + VK_VIDEO_ENCODER_STATUS_SUCCESS; + VkVideoEncoderStatusCode fromFormatNoExtent = + VK_VIDEO_ENCODER_STATUS_SUCCESS; + // The SESSION-level counterpart of those declarations: what Reconfigure + // does with the same pair. The two positives default to a value that + // fails their check, so a leg that did not run cannot read as a pass. + bool ranReconfig = false; + VkResult reconfigBitrate = VK_ERROR_UNKNOWN; + VkResult reconfigAgreeing = VK_ERROR_UNKNOWN; + VkResult reconfigYcbcr = VK_SUCCESS; +}; + +// The declaration each arm's descriptor makes. This is the whole content of +// the fix under test, expressed as data. +struct Decl { + VkImageUsageFlags usage; + VkImageCreateFlags flags; + VkImageTiling tiling; + bool profileList; +}; + +Decl DeclFor(Arm arm, bool withhold) +{ + Decl d = {}; + switch (arm) { + case ARM_DIRECT: + d.tiling = VK_IMAGE_TILING_OPTIMAL; + d.usage = VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + d.profileList = true; + if (withhold) { // no VIDEO_ENCODE_SRC -> not encodeCapable + d.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + d.profileList = false; + } + break; + case ARM_FILTER_YCBCR: + d.tiling = VK_IMAGE_TILING_OPTIMAL; + d.usage = VK_IMAGE_USAGE_STORAGE_BIT | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + d.flags = VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT | + VK_IMAGE_CREATE_EXTENDED_USAGE_BIT; + if (withhold) { + // Withhold the CREATE FLAGS, which is this arm's own + // declaration. STORAGE goes with them, because a 3-plane + // 420 image with STORAGE and no MUTABLE|EXTENDED is + // VK_ERROR_FORMAT_NOT_SUPPORTED on this device -- that is + // proof 4 restated, and creating it anyway makes the control + // emit VUID-VkImageCreateInfo-imageCreateMaxMipLevels-02251 + // (this driver creates it regardless, which is its own + // finding). The registration is still refused + // CONVERSION_REQUIRED, and now for a reason the descriptor + // states rather than one the create smuggled in. + d.flags = 0; + d.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + } + break; + case ARM_FILTER_RGBA: + d.tiling = VK_IMAGE_TILING_OPTIMAL; + // STORAGE and nothing else is what this arm needs. NO create + // flags: asserting that is half the point of the row. + d.usage = VK_IMAGE_USAGE_STORAGE_BIT | + VK_IMAGE_USAGE_TRANSFER_DST_BIT; + if (withhold) { // the storage read the filter's RGBA arm binds + d.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + } + break; + case ARM_CONTROL: + default: + d.tiling = VK_IMAGE_TILING_OPTIMAL; + d.usage = VK_IMAGE_USAGE_SAMPLED_BIT | + VK_IMAGE_USAGE_TRANSFER_DST_BIT; + break; + } + return d; +} + +void FillDescriptor(VkVideoEncoderExternalImageDescriptor* desc, + VkFormat format, const Decl& d, VkImage image) +{ + std::memset(desc, 0, sizeof(*desc)); + desc->sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc->handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc->format = format; + desc->width = kWidth; + desc->height = kHeight; + desc->tiling = d.tiling; + desc->imageUsage = d.usage; + desc->imageFlags = d.flags; + desc->sharingMode = VK_SHARING_MODE_EXCLUSIVE; + desc->planeCount = 0; + desc->residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc->defaultLayout = VK_IMAGE_LAYOUT_GENERAL; + desc->existingImage = image; +} + +RowResult RunRow(const Row& row, bool verbose) +{ + RowResult res; + + VkSharedBaseObj enc; + if ((CreateVulkanVideoEncoderExt(enc) != VK_SUCCESS) || !enc) { + return res; + } + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = row.codec; + config.encodeWidth = kWidth; + config.encodeHeight = kHeight; + config.inputFormat = row.format; + config.inputWidth = kWidth; + config.inputHeight = kHeight; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 5000000; + config.maxBitrate = 5000000; + config.gopLength = 30; + config.consecutiveBFrames = 0; + config.idrPeriod = 30; + config.frameRateNum = 30; + config.frameRateDen = 1; + config.deviceId = -1; + config.disableFileOutput = VK_TRUE; + // No filter request: the library derives the preprocess conversion from + // inputFormat, so each row's session is configured exactly as a shipping + // caller would configure it for that format. + + if (enc->InitializeExt(config) != VK_SUCCESS) { + return res; + } + res.sessionInit = true; + + VkInstance instance = enc->GetVkInstance(); + VkDevice device = enc->GetVkDevice(); + VkPhysicalDevice phys = enc->GetVkPhysicalDevice(); + DeviceFns fns; + if (!LoadDeviceFns(instance, device, &fns)) { + return res; + } + + // Device format features, reported for BOTH tilings. This is the raw + // material the registration gate derives its single-plane arm from, + // printed beside the verdict so a refusal can be attributed to the device + // rather than to the library. + VkFormatProperties fp{}; + fns.GetPhysicalDeviceFormatProperties(phys, row.format, &fp); + res.filterCapableOptimal = + ((fp.optimalTilingFeatures & VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0) + ? VK_TRUE : VK_FALSE; + res.filterCapableLinear = + ((fp.linearTilingFeatures & VK_FORMAT_FEATURE_STORAGE_IMAGE_BIT) != 0) + ? VK_TRUE : VK_FALSE; + if (verbose) { + std::printf(" optimalTilingFeatures 0x%08X " + "linearTilingFeatures 0x%08X\n", + (unsigned)fp.optimalTilingFeatures, + (unsigned)fp.linearTilingFeatures); + } + + const VkVideoComponentBitDepthFlagBitsKHR depth = GroupDepth(row.group); + const VkVideoChromaSubsamplingFlagBitsKHR subsampling = + GroupSubsampling(row.group); + const Decl d = DeclFor(row.arm, false); + Img img; + if (!CreateImage(fns, phys, device, row.format, d.tiling, d.usage, d.flags, + d.profileList, row.codec, depth, subsampling, &img)) { + return res; + } + res.imageCreated = true; + + VkVideoEncoderExternalImageDescriptor desc; + FillDescriptor(&desc, row.format, d, img.image); + + VkVideoEncoderImageSupport support = {}; + support.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMAGE_SUPPORT; + enc->QueryImageSupport(desc, &support); + res.queryStatus = support.status; + res.querySupported = support.supported; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + res.regStatus = enc->RegisterImageResource(desc, 0, &resource, nullptr); + if ((res.regStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS) && + (resource != VK_VIDEO_ENCODER_RESOURCE_NULL)) { + VkEncResourceProbe probe; + if (VkEncProbeResource(enc.get(), resource, &probe) == + VK_VIDEO_ENCODER_STATUS_SUCCESS) { + res.path = probe.inputPath; + res.planeStorageViews = probe.planeStorageViews; + res.storageReadView = probe.storageReadView; + } + enc->UnregisterImageResource(resource); + } + + // WHAT THE DESCRIPTOR DECLARES, on the session and the image the row + // already built. Nothing here creates a Vulkan object: the only variable + // is VkVideoEncoderExternalImageDescriptor::colorModel. + if (row.arm == ARM_FILTER_RGBA) { + res.ranDeclared = true; + VkVideoEncoderExternalImageDescriptor cdesc; + VkVideoEncoderResource cres = VK_VIDEO_ENCODER_RESOURCE_NULL; + + // Naming the model the format already carries is legal and changes + // nothing. + FillDescriptor(&cdesc, row.format, d, img.image); + cdesc.colorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + res.declaredAgreeing = + enc->RegisterImageResource(cdesc, 0, &cres, nullptr); + if (cres != VK_VIDEO_ENCODER_RESOURCE_NULL) { + enc->UnregisterImageResource(cres); + cres = VK_VIDEO_ENCODER_RESOURCE_NULL; + } + + // Y'CbCr declared over the same enumerant. On R8G8B8A8_UNORM that is + // the packed 4:4:4 AYUV reading, which is a statement of fact the + // library resolves; on the other RGBA spellings it is a contradiction + // the format cannot carry. Either way this session routes R'G'B' + // through its filter and cannot route what the descriptor declares. + FillDescriptor(&cdesc, row.format, d, img.image); + cdesc.colorModel = VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + res.declaredYcbcr = + enc->RegisterImageResource(cdesc, 0, &cres, nullptr); + if (cres != VK_VIDEO_ENCODER_RESOURCE_NULL) { + enc->UnregisterImageResource(cres); + cres = VK_VIDEO_ENCODER_RESOURCE_NULL; + } + + // The same two declarations over a descriptor that also states no + // extent. A descriptor's colour model is a property of the descriptor + // alone, so it is judged before the geometry is read: the pair below + // differs only in the declaration and must therefore answer + // differently. + FillDescriptor(&cdesc, row.format, d, img.image); + cdesc.colorModel = VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + cdesc.width = 0; + cdesc.height = 0; + res.declaredYcbcrNoExtent = + enc->RegisterImageResource(cdesc, 0, &cres, nullptr); + if (cres != VK_VIDEO_ENCODER_RESOURCE_NULL) { + enc->UnregisterImageResource(cres); + cres = VK_VIDEO_ENCODER_RESOURCE_NULL; + } + + FillDescriptor(&cdesc, row.format, d, img.image); + cdesc.width = 0; + cdesc.height = 0; + res.fromFormatNoExtent = + enc->RegisterImageResource(cdesc, 0, &cres, nullptr); + if (cres != VK_VIDEO_ENCODER_RESOURCE_NULL) { + enc->UnregisterImageResource(cres); + } + + // THE SAME DECLARATION ONE LEVEL UP. Reconfigure carries a whole + // VkVideoEncoderConfig, and the session's input pair is half of what + // its preprocess filter was built from -- so a call that re-declares + // the model has to be refused, not answered VK_SUCCESS with the arm + // the session already holds. Nothing here reaches the device: a + // rate-control request is state the encoder thread folds in at the + // next frame, and this test submits none. + VkVideoEncoderConfig rcfg = config; + rcfg.averageBitrate = config.averageBitrate + 1000000; + rcfg.maxBitrate = rcfg.averageBitrate; + res.reconfigBitrate = enc->Reconfigure(rcfg); + + // The control on the COMPARISON rather than on the call: the model + // this format already carries, stated explicitly instead of left to + // be read off the format. It resolves to what is in force and must + // still be accepted -- a gate that compared the spelling would + // refuse it and refuse the caller nothing useful. + rcfg = config; + rcfg.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_RGB; + res.reconfigAgreeing = enc->Reconfigure(rcfg); + + // The flip. On R8G8B8A8_UNORM this is AYUV, a reading that enumerant + // really carries and one this session's filter was not built for; on + // the other RGBA spellings it resolves to nothing at all. Either way + // it is not the model the session encodes from. + rcfg = config; + rcfg.inputColorModel = VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR; + res.reconfigYcbcr = enc->Reconfigure(rcfg); + res.ranReconfig = true; + } + + // The negative control: same format, same session, the arm's own + // declaration withheld. + if (row.arm != ARM_CONTROL) { + const Decl w = DeclFor(row.arm, true); + Img wimg; + if (CreateImage(fns, phys, device, row.format, w.tiling, w.usage, + w.flags, w.profileList, row.codec, depth, + subsampling, &wimg)) { + VkVideoEncoderExternalImageDescriptor wdesc; + FillDescriptor(&wdesc, row.format, w, wimg.image); + VkVideoEncoderResource wres = VK_VIDEO_ENCODER_RESOURCE_NULL; + res.withheldStatus = + enc->RegisterImageResource(wdesc, 0, &wres, nullptr); + res.ranWithheld = true; + if (wres != VK_VIDEO_ENCODER_RESOURCE_NULL) { + enc->UnregisterImageResource(wres); + } + } + DestroyImg(fns, device, &wimg); + } + + DestroyImg(fns, device, &img); + enc = nullptr; + return res; +} + +} // namespace + +int main(int argc, char** argv) +{ + bool verbose = false; + // Which profile family this invocation walks. Each row costs a whole + // encoder session -- its own VkInstance and VkDevice -- and a single + // process cannot hold an unbounded number of those, so the matrix is + // split by family and each family gets its own process. Sharing one + // would not merely be slower: the rows past the limit report "this + // device has no such encode profile" for a profile the device does + // support, which is a wrong finding rather than a missing one. + bool only444 = false; + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--verbose") == 0) { + verbose = true; + } else if (std::strcmp(argv[i], "--444") == 0) { + only444 = true; + } + } + + std::printf("Encoder-ext INPUT FORMAT MATRIX -- library-owned device\n"); + std::printf("=======================================================\n"); + + // One throwaway session purely to name the device the rest of the matrix + // ran on. A matrix that silently fell through to a software ICD would be + // meaningless, and this is what makes that visible rather than assumed. + { + VkSharedBaseObj probe; + if ((CreateVulkanVideoEncoderExt(probe) != VK_SUCCESS) || !probe) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + return 77; + } + VkVideoEncoderConfig c = {}; + c.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + c.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + c.encodeWidth = kWidth; + c.encodeHeight = kHeight; + c.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + c.inputWidth = kWidth; + c.inputHeight = kHeight; + c.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + c.averageBitrate = 5000000; + c.maxBitrate = 5000000; + c.gopLength = 30; + c.idrPeriod = 30; + c.frameRateNum = 30; + c.frameRateDen = 1; + c.deviceId = -1; + c.disableFileOutput = VK_TRUE; + if (probe->InitializeExt(c) != VK_SUCCESS) { + std::printf("SKIP: InitializeExt failed -- no encode-capable " + "Vulkan device on this host\n"); + return 77; + } + DeviceFns fns; + if (LoadDeviceFns(probe->GetVkInstance(), probe->GetVkDevice(), + &fns)) { + VkPhysicalDeviceProperties props{}; + fns.GetPhysicalDeviceProperties(probe->GetVkPhysicalDevice(), + &props); + std::printf("device: %s\n", props.deviceName); + std::printf(" vendorID 0x%04X deviceID 0x%04X\n", + props.vendorID, props.deviceID); + // driverVersion is printed BOTH ways on purpose. NVIDIA packs it + // as (major<<22)|(minor<<14)|(secondary<<6)|tertiary; the standard + // VK_API_VERSION macros unpack (22/12/0), so one driver reads as + // two different version strings depending on who decodes it. + std::printf(" driverVersion raw 0x%08X" + " -> standard-decode %u.%u.%u" + " -> NVIDIA-decode %u.%02u\n", + props.driverVersion, + props.driverVersion >> 22, + (props.driverVersion >> 12) & 0x3FF, + props.driverVersion & 0xFFF, + props.driverVersion >> 22, + (props.driverVersion >> 14) & 0xFF); + } + probe = nullptr; + } + std::printf("\n"); + + // ---- THE INPUT-COLOUR CHAIN, AT InitializeExt ---- + // + // WHY IT IS HERE AND NOT IN THE DEVICE-FREE SUITE. Two walks read + // VkVideoEncoderConfig::pNext: the BINDER's, which the device-free tests + // drive, and InitializeExt's, which rejects any sType it does not name. + // The device-free suite cannot reach the second one -- its session is + // installed already-initialized -- so a chain that the binder consumes + // perfectly could still be rejected outright before the binder ever sees + // it, and nothing would say so. These three sessions are the only place + // that walk is exercised. + // + // THE PAIR IS THE POINT. Every binder refusal is + // VK_ERROR_INITIALIZATION_FAILED, so a refusal on its own attributes + // nothing. The accepted config and the refused one differ in ONE FIELD. + { + struct ChainCase { + const char* what; + uint8_t bitstreamPrimaries; + uint8_t inputPrimaries; + bool wantSuccess; + }; + static const ChainCase chainCases[] = { + // An all-zero chain is "undeclared on every axis" and must behave + // exactly like no chain at all. + { "an all-zero input-colour chain initializes", 0u, 0u, true }, + // CONTROL, and it runs before the refusal is read. + { "input primaries 9 with bitstream primaries 9 initializes", + 9u, 9u, true }, + { "input primaries 1 with bitstream primaries 9 is refused", + 9u, 1u, false }, + }; + for (const ChainCase& cc : chainCases) { + VkSharedBaseObj enc; + if ((CreateVulkanVideoEncoderExt(enc) != VK_SUCCESS) || !enc) { + Check("input-colour", false, cc.what, "create failed"); + continue; + } + VkVideoEncoderInputColourInfo ic = {}; + ic.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO; + ic.inputColourPrimaries = cc.inputPrimaries; + VkVideoEncoderConfig c = {}; + c.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + c.pNext = ⁣ + c.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + c.encodeWidth = kWidth; + c.encodeHeight = kHeight; + c.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + c.inputWidth = kWidth; + c.inputHeight = kHeight; + c.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + c.averageBitrate = 5000000; + c.maxBitrate = 5000000; + c.gopLength = 30; + c.idrPeriod = 30; + c.frameRateNum = 30; + c.frameRateDen = 1; + c.deviceId = -1; + c.disableFileOutput = VK_TRUE; + c.colourPrimaries = cc.bitstreamPrimaries; + const VkResult r = enc->InitializeExt(c); + Check("input-colour", + cc.wantSuccess ? (r == VK_SUCCESS) + : (r == VK_ERROR_INITIALIZATION_FAILED), + cc.what, "InitializeExt returned " + U32((uint32_t)r)); + enc = nullptr; + } + } + std::printf("\n"); + + // The rows this invocation walks. The controls belong to every family: + // a refusal only ever observed beside 4:2:0 rows says nothing about the + // engine the 4:4:4 rows actually ran on. + size_t selected[kNumRows]; + size_t nSelected = 0; + for (size_t i = 0; i < kNumRows; i++) { + const bool is444 = + (kRows[i].group == G_444_8) || (kRows[i].group == G_444_10); + if ((is444 == only444) || (kRows[i].arm == ARM_CONTROL)) { + selected[nSelected++] = i; + } + } + std::printf("walking the %s rows: %zu of %zu\n", + only444 ? "4:4:4" : "4:2:0", nSelected, kNumRows); + + size_t nSessions = 0, nRegistered = 0, nFilter = 0; + size_t nDirect = 0, nStaged = 0, nDeviceLimited = 0; + + // PASS 1 -- run every row. A group's DIRECT control has to be known + // before any of its siblings can be judged. + RowResult results[kNumRows]; + bool groupHasDevice[G_NONE + 1] = {false, false, false, + false, false, true}; + for (size_t s = 0; s < nSelected; s++) { + const size_t i = selected[s]; + results[i] = RunRow(kRows[i], verbose); + if ((kRows[i].arm == ARM_DIRECT) && results[i].sessionInit) { + groupHasDevice[kRows[i].group] = true; + } + } + + // PASS 2 -- report and assert. + for (size_t s = 0; s < nSelected; s++) { + const size_t i = selected[s]; + const Row& row = kRows[i]; + const VkEncInputFormatClass cls = + VkEncClassifyInput(row.format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT); + const bool deviceHasProfile = groupHasDevice[row.group]; + std::printf("[%zu/%zu] %s [%s]\n", s + 1, nSelected, row.name, + row.codecName); + std::printf(" taxonomy=%s planeCount=%u isRgba=%u\n", + ClassName(cls), VkEncInputFormatPlaneCount(row.format), + (unsigned)(VkEncResolveColorModel( + row.format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT) == + VK_VIDEO_ENCODER_COLOR_MODEL_RGB)); + + const RowResult r = results[i]; + + std::printf(" session=%d image=%d\n", + (int)r.sessionInit, (int)r.imageCreated); + std::printf(" STORAGE_IMAGE feature: optimal=%d linear=%d\n", + (int)r.filterCapableOptimal, (int)r.filterCapableLinear); + std::printf(" query: supported=%d status=%d | " + "register: status=%d\n", + (int)r.querySupported, (int)r.queryStatus, + (int)r.regStatus); + std::printf(" ROUTED: %s planeStorageViews=%d " + "storageReadView=%d\n", + PathName(r.path), (int)r.planeStorageViews, + (int)r.storageReadView); + if (r.ranWithheld) { + std::printf(" negative control (declaration withheld): " + "status=%d\n", (int)r.withheldStatus); + } + if (r.ranDeclared) { + std::printf(" colorModel declared: RGB=%d YCBCR=%d ; " + "no extent: YCBCR=%d FROM_FORMAT=%d\n", + (int)r.declaredAgreeing, (int)r.declaredYcbcr, + (int)r.declaredYcbcrNoExtent, + (int)r.fromFormatNoExtent); + } + if (r.ranReconfig) { + std::printf(" Reconfigure: bitrate=%d RGB=%d YCBCR=%d\n", + (int)r.reconfigBitrate, (int)r.reconfigAgreeing, + (int)r.reconfigYcbcr); + } + + if (r.sessionInit) nSessions++; + if (r.regStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + nRegistered++; + if (r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER) nFilter++; + if (r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT) nDirect++; + if (r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_STAGED) nStaged++; + } + + // ------------------------------------------------------------------ + // DEVICE-LIMITED rows short-circuit. This is a RESULT, not a + // failure: the group's DIRECT control could not initialize either, + // so this device has no such video profile and the library was never + // consulted. The one thing still assertable -- and it is a real + // assertion, it fails if the library refuses a FILTER format on a + // profile whose DIRECT sibling works -- is that the two AGREE. + // ------------------------------------------------------------------ + // The FORMAT TAXONOMY is a pure function of the format: no device, + // no session, no profile takes part in it. So it is asserted for + // every row before anything can short-circuit -- otherwise a row + // whose class regressed would be reported as "this device has no + // such profile", which is a different finding and a false one. + if (row.arm == ARM_DIRECT) { + Check(row.name, cls == VK_ENC_INPUT_FORMAT_ENCODABLE_DIRECT, + "taxonomy says DIRECT", ClassName(cls)); + } else if ((row.arm == ARM_FILTER_YCBCR) || + (row.arm == ARM_FILTER_RGBA)) { + Check(row.name, cls == VK_ENC_INPUT_FORMAT_ENCODABLE_VIA_FILTER, + "taxonomy says VIA_FILTER", ClassName(cls)); + } + + if ((row.arm != ARM_CONTROL) && !deviceHasProfile) { + nDeviceLimited++; + std::printf(" DEVICE-LIMITED: this device has no encode " + "profile at this bit depth; the group's DIRECT " + "control did not initialize either.\n"); + Check(row.name, !r.sessionInit, + "and the FILTER member agrees with its DIRECT control " + "(no library-side refusal hiding behind a device limit)", + "sessionInit " + U32(r.sessionInit ? 1 : 0)); + std::printf("\n"); + continue; + } + + // Whether a Y'CbCr declaration can be READ against this row's + // format. The three packed 4:4:4 layouts have no Vulkan format of + // their own and ride RGBA enumerants, so on those the declaration + // resolves; on every other RGBA spelling it contradicts the format. + // Asking the library rather than listing the formats is what keeps + // this from becoming a second copy of that table. + const bool ycbcrDeclarable = + (VkEncResolveColorModel(row.format, + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR) == + VK_VIDEO_ENCODER_COLOR_MODEL_YCBCR); + const VkVideoEncoderStatusCode expectDeclaredYcbcr = + ycbcrDeclarable + ? VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED + : VK_VIDEO_ENCODER_STATUS_ERROR_COLOR_MODEL_UNSUPPORTED; + + // ------------------------------------------------------------------ + // Assertions. Per arm, because the arms make different promises. + // ------------------------------------------------------------------ + switch (row.arm) { + case ARM_DIRECT: + Check(row.name, r.sessionInit, "session initializes", ""); + Check(row.name, + r.regStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "registers", "status " + U32(r.regStatus)); + Check(row.name, + r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_DIRECT, + "routes DIRECT", PathName(r.path)); + Check(row.name, + r.ranWithheld && + (r.withheldStatus == + VK_VIDEO_ENCODER_STATUS_SUCCESS), + "and without VIDEO_ENCODE_SRC still registers " + "(it stages)", + "status " + U32(r.withheldStatus)); + break; + case ARM_FILTER_YCBCR: + Check(row.name, r.sessionInit, "session initializes", ""); + Check(row.name, + r.regStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "registers with MUTABLE|EXTENDED|STORAGE", + "status " + U32(r.regStatus)); + Check(row.name, + r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER, + "routes FILTER", PathName(r.path)); + Check(row.name, r.planeStorageViews == VK_TRUE, + "through PLANE STORAGE views (unchanged arm)", + "planeStorageViews " + U32(r.planeStorageViews)); + Check(row.name, r.storageReadView == VK_FALSE, + "and NOT through the single-plane fact", + "storageReadView " + U32(r.storageReadView)); + Check(row.name, + r.ranWithheld && + (r.withheldStatus == + VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED), + "without the create flags still CONVERSION_REQUIRED", + "status " + U32(r.withheldStatus)); + break; + case ARM_FILTER_RGBA: + Check(row.name, r.sessionInit, "session initializes", ""); + Check(row.name, + r.regStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "registers with STORAGE and NO create flags", + "status " + U32(r.regStatus)); + Check(row.name, + r.path == VK_VIDEO_EXTERNAL_INPUT_PATH_FILTER, + "routes FILTER (from the slot, not the config flag)", + PathName(r.path)); + Check(row.name, r.storageReadView == VK_TRUE, + "through the SINGLE-PLANE storage-read fact", + "storageReadView " + U32(r.storageReadView)); + Check(row.name, r.planeStorageViews == VK_FALSE, + "and NOT through plane storage views", + "planeStorageViews " + U32(r.planeStorageViews)); + Check(row.name, + r.ranWithheld && + (r.withheldStatus == + VK_VIDEO_ENCODER_STATUS_ERROR_CONVERSION_REQUIRED), + "without STORAGE still CONVERSION_REQUIRED " + "(the gate still discriminates)", + "status " + U32(r.withheldStatus)); + // WHAT THE DESCRIPTOR DECLARES IS WHAT IT IS JUDGED ON. + // The registration gate is a negotiation point: it answers + // for the descriptor in front of it, and the colour model is + // part of that descriptor. Reading the model off the SESSION + // instead answers about a different picture -- and the packed + // 4:4:4 layouts are why it matters, because they ride these + // very enumerants and the format alone cannot tell one of + // them from an ordinary R'G'B' image. + Check(row.name, + r.ranDeclared && + (r.declaredAgreeing == + VK_VIDEO_ENCODER_STATUS_SUCCESS), + "declaring the model the format carries changes nothing", + "status " + U32(r.declaredAgreeing)); + // A Y'CbCr declaration over an RGBA enumerant is one of two + // different things, and the library answers each as itself. + // Where the enumerant carries a packed 4:4:4 reading the + // declaration RESOLVES, and what is left is a routing answer: + // this session converts R'G'B' and cannot route what the + // descriptor declares. Where it carries none the declaration + // cannot be read against the format at all, and that is a + // COLOUR MODEL answer, not a routing one. + Check(row.name, + r.ranDeclared && (r.declaredYcbcr == expectDeclaredYcbcr), + ycbcrDeclarable + ? "a resolvable Y'CbCr declaration this session " + "cannot route is answered on its routing" + : "an unresolvable Y'CbCr declaration is answered " + "on the declaration itself", + "status " + U32(r.declaredYcbcr)); + // The pair that differs ONLY in the declaration. A gate that + // discards the declared model judges both of these on the + // R'G'B' route, passes the class question on both, and + // answers both with the geometry -- which is what makes this + // the calibration rather than a second positive. + Check(row.name, + r.ranDeclared && + (r.fromFormatNoExtent == + VK_VIDEO_ENCODER_STATUS_ERROR_PLANE_LAYOUT_INVALID), + "an extentless descriptor that declares nothing is " + "answered on its geometry", + "status " + U32(r.fromFormatNoExtent)); + Check(row.name, + r.ranDeclared && + (r.declaredYcbcrNoExtent == expectDeclaredYcbcr), + "and the same descriptor declaring Y'CbCr is answered " + "on its declaration, before its geometry is read", + "status " + U32(r.declaredYcbcrNoExtent)); + // AND THE SESSION IS DECLARED IN THE SAME PAIR. Reconfigure + // forwards a rate-control change and refuses everything the + // session was built around; the input colour model is half + // of what the preprocess filter was built from, so it + // belongs to the second set. The two positives below are + // what keep this from passing on a Reconfigure that refuses + // everything. + Check(row.name, + r.ranReconfig && (r.reconfigBitrate == VK_SUCCESS), + "Reconfigure still carries a rate-control change", + "VkResult " + std::to_string((int)r.reconfigBitrate)); + Check(row.name, + r.ranReconfig && (r.reconfigAgreeing == VK_SUCCESS), + "and accepts the model the format already carries " + "when it is stated rather than inferred", + "VkResult " + std::to_string((int)r.reconfigAgreeing)); + Check(row.name, + r.ranReconfig && + (r.reconfigYcbcr == VK_ERROR_INITIALIZATION_FAILED), + "and REFUSES a declaration that resolves to the other " + "model instead of answering VK_SUCCESS and keeping " + "the arm it was built with", + "VkResult " + std::to_string((int)r.reconfigYcbcr)); + break; + case ARM_CONTROL: + default: + Check(row.name, cls == VK_ENC_INPUT_FORMAT_UNSUPPORTED, + "taxonomy says UNSUPPORTED", ClassName(cls)); + Check(row.name, + r.regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS, + "and it does not register", + "status " + U32(r.regStatus)); + break; + } + std::printf("\n"); + } + + std::printf("=======================================================\n"); + std::printf("DENOMINATOR: %zu formats walked (%zu claimed by the " + "taxonomy + 2 controls)\n", nSelected, nSelected - 2); + std::printf(" sessions initialized : %zu of %zu\n", nSessions, nSelected); + std::printf(" registered : %zu of %zu\n", nRegistered, nSelected); + std::printf(" routed DIRECT/FILTER/STAGED : %zu / %zu / %zu\n", + nDirect, nFilter, nStaged); + std::printf(" device-limited (no such encode profile HERE) : %zu\n", + nDeviceLimited); + std::printf("%d checks, %d failures\n", g_checks, g_failures); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-import-content/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-import-content/CMakeLists.txt new file mode 100644 index 00000000..467d85b1 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-import-content/CMakeLists.txt @@ -0,0 +1,102 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_import_content_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, not the shared one: this reaches both the +# internal null-backend seam and VkVideoEncoderContentProbe, neither of which +# is exported from libvkvideo-encoder.so. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + # VkVideoEncoder/VkVideoEncoderContentProbe.h -- the probe class lives in + # the encoder library, below the ext layer, and this suite drives its pure + # scorer and predicate directly. + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# CTest semantics, matching the sibling library tests: 0 means every assertion +# held, 1 means an assertion failed, and 2 means the session could not be stood +# up at all -- deliberately a FAILURE and not a skip. +# +# WHAT A GREEN HERE DOES NOT MEAN, stated at the gate and not only in the +# source: it does NOT mean the workaround fires on a buffer the driver +# actually poisoned. That needs a driver that exhibits the defect, a GBM +# exporter and a real dma-buf import, and is proven on hardware or not at +# all. +# +# WHAT IT DOES GATE, and it is the part that would otherwise be provable on +# exactly one machine: the DECISION. Given the bytes a poisoned import +# produces -- which this suite synthesises, laid out with the plane offsets, +# padded row pitch and skipped rows a real HOST_VISIBLE LINEAR image has -- +# the predicate, the scorer, the per-registration latch, the +# oldest-damaged-first report ordering, the drain on retirement and the +# chained carrier are all exercised with no GPU. The predicate in particular +# is asserted over EVERY (Y,U,V) liveness quadrant rather than spot-checked, +# because the chroma-only version of it has produced a wrong conclusion on +# this defect twice, and a fully legal BLACK frame (Y=16, U=V=128) is +# asserted CLEAN, because that is the false positive that would reroute +# working buffers. +enable_testing() +add_test(NAME EncoderExtImportContentProbe + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtImportContentProbe PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_ext_import_content_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-import-content/src/main.cpp b/vk_video_encoder/test/encoder-ext-import-content/src/main.cpp new file mode 100644 index 00000000..91430d68 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-import-content/src/main.cpp @@ -0,0 +1,1265 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * The dma-buf IMPORT CONTENT PROBE: the detection predicate, the scorer that + * feeds it, the per-registration latch, and the carrier the verdict travels + * on. Device-free. + * + * WHAT THIS COVERS, AND -- FIRST -- WHAT IT DOES NOT. + * + * NOT COVERED, and it is the interesting half: whether the probe fires on a + * REAL buffer the driver actually poisoned. That needs a driver that + * exhibits the defect, a GBM exporter and a real dma-buf import, and no + * device-free test + * can manufacture one. A green here is NOT evidence the workaround works on + * hardware. It is evidence that, GIVEN the bytes a poisoned import produces, + * this code reaches the right verdict and reports it -- which is the half + * that can be regression-tested on every host, and the half that would + * otherwise be provable only on one machine. + * + * WHAT IS COVERED: + * + * 1. THE PREDICATE, over the whole 2x2x2 quadrant space of (luma dead, + * U dead, V dead) plus both sides of the threshold. This is the + * requirement that has been got wrong twice on this defect: a + * CHROMA-ONLY scorer conflates a chroma-dead buffer with an all-dead + * one, so the union is asserted quadrant by quadrant rather than by + * spot-check. + * + * 2. THE SCORER, over SYNTHETIC NV12 buffers laid out the way a real + * HOST_VISIBLE LINEAR image is -- non-zero plane offsets, row pitch + * wider than the width, and DIFFERENT VALUES in the rows the row-stride + * skips and in the padding past the width. A scorer that used the wrong + * pitch, forgot an offset, walked every row, or read into the padding + * produces a different mean and turns these red. The four buffers are + * the two measured damage modes, ordinary content, and -- the one that + * matters most for false positives -- a FULLY LEGAL BLACK FRAME, which + * is Y=16 U=V=128 and must score CLEAN. + * + * 3. THE LATCH: one verdict per REGISTRATION and not per frame, the + * oldest-damaged-first report ordering, and the drain -- retiring a + * damaged registration is what lets the next one surface, and is the + * whole reason a consumer can act on this channel without losing a + * verdict between two polls. + * + * 4. THE CARRIER: VkVideoEncoderImportContentInfo chained onto + * RegisterImageResource (where chaining it IS the opt-in) and onto + * GetCompletionInfo (where the verdict comes back), the widened chain + * gate that now accepts TWO known link types, and the writer proof. + * + * WHY A NULL BACKEND, for group 4. VkEncInstallNullBackend gives a session + * that reports initialized with no device, no worker threads and no + * VkVideoEncoder at all, on which a VK_IMAGE registration runs the real + * RegisterImageResource -- the real pStatus gate, the real chain walk, the + * real arming. That the probe object is owned by the EXT layer rather than by + * the encoder is what makes this reachable; see the note on m_contentProbe. + */ + +#include "vulkan_video_encoder_ext_internal.h" +#include "VkVideoEncoder/VkVideoEncoderContentProbe.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Scrub them before anything else sees them -- the +// same block, for the same reason, as the sibling library test TUs. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; +const char* g_currentCase = ""; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + return; + } + g_failures++; + std::printf(" FAIL [%s] %s : %s\n", g_currentCase, what, detail.c_str()); +} + +std::string U64(uint64_t v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%llu", (unsigned long long)v); + return buf; +} + +using Probe = VkVideoEncoderContentProbe; + +const char* StateName(Probe::State s) +{ + switch (s) { + case Probe::STATE_NOT_EVALUATED: return "NOT_EVALUATED"; + case Probe::STATE_NOT_APPLICABLE: return "NOT_APPLICABLE"; + case Probe::STATE_ARMED: return "ARMED"; + case Probe::STATE_CLEAN: return "CLEAN"; + case Probe::STATE_DAMAGED_CHROMA: return "DAMAGED_CHROMA"; + case Probe::STATE_DAMAGED_ALL: return "DAMAGED_ALL"; + } + return "?"; +} + +// Q8 helper: a plane whose every sampled byte is |v|. +constexpr uint32_t Q8(uint32_t v) { return v * 256u; } + +// =========================================================================== +// GROUP 1 -- THE PREDICATE +// =========================================================================== +// +// The full quadrant space. DEAD is any mean strictly below 2.0/255; ALIVE is +// at or above it. The two damage modes this driver defect produces are +// (alive, dead, dead) and (dead, dead, dead); the single-channel chroma cases +// are included because a partially-written chroma plane is the shape a +// half-completed import would take and the requirement names U OR V, not +// U AND V. +void CasePredicateQuadrants() +{ + g_currentCase = "predicate: every (Y,U,V) liveness quadrant"; + + struct Row { + uint32_t y, u, v; + Probe::State want; + const char* why; + }; + const uint32_t DEAD = 0; + const uint32_t ALIVE = Q8(100); + const Row rows[] = { + { ALIVE, ALIVE, ALIVE, Probe::STATE_CLEAN, + "ordinary content" }, + { ALIVE, DEAD, DEAD, Probe::STATE_DAMAGED_CHROMA, + "CHROMA_ZERO: the measured mode -- luma correct, chroma plane dead" }, + { ALIVE, DEAD, ALIVE, Probe::STATE_DAMAGED_CHROMA, + "U alone dead: the requirement says U OR V, not U AND V" }, + { ALIVE, ALIVE, DEAD, Probe::STATE_DAMAGED_CHROMA, + "V alone dead" }, + { DEAD, DEAD, DEAD, Probe::STATE_DAMAGED_ALL, + "ALL_ZERO: the other measured mode. A CHROMA-ONLY scorer reports " + "this identically to the row above, which is the conflation that " + "has produced a wrong conclusion here twice" }, + { DEAD, ALIVE, ALIVE, Probe::STATE_CLEAN, + "luma dead over LIVE chroma is deliberately NOT in the union: it " + "is not a mode this defect produces, and a black-luma frame is a " + "thing a producer can legitimately send" }, + { DEAD, DEAD, ALIVE, Probe::STATE_CLEAN, + "same quadrant family: no all-dead, and the chroma clause requires " + "live luma" }, + { DEAD, ALIVE, DEAD, Probe::STATE_CLEAN, + "same" }, + }; + for (const Row& r : rows) { + const Probe::State got = Probe::ClassifyPlaneMeansQ8(r.y, r.u, r.v); + Check(got == r.want, r.why, + std::string("Y=") + U64(r.y) + " U=" + U64(r.u) + " V=" + + U64(r.v) + " -> " + StateName(got) + ", expected " + + StateName(r.want)); + } +} + +// The threshold itself, from both sides. 512 is 2.0 in Q8; the predicate is +// STRICTLY below, so 511 is dead and 512 is alive. A scorer that used <= or +// that scaled by 255 instead of 256 lands on the wrong side of one of these. +void CasePredicateThresholdBoundary() +{ + g_currentCase = "predicate: the dead-plane threshold, both sides"; + + Check(Probe::kDeadPlaneMeanQ8 == 512u, + "the threshold is 2.0/255 expressed in Q8", + "kDeadPlaneMeanQ8 = " + U64(Probe::kDeadPlaneMeanQ8)); + Check(Probe::ClassifyPlaneMeansQ8(Q8(100), 511u, 511u) == + Probe::STATE_DAMAGED_CHROMA, + "a chroma mean of 511 (just under 2.0) is DEAD", + "got " + std::string(StateName( + Probe::ClassifyPlaneMeansQ8(Q8(100), 511u, 511u)))); + Check(Probe::ClassifyPlaneMeansQ8(Q8(100), 512u, 512u) == + Probe::STATE_CLEAN, + "a chroma mean of exactly 512 (2.0) is ALIVE -- strictly below", + "got " + std::string(StateName( + Probe::ClassifyPlaneMeansQ8(Q8(100), 512u, 512u)))); + Check(Probe::ClassifyPlaneMeansQ8(511u, 511u, 511u) == + Probe::STATE_DAMAGED_ALL, + "all three just under the threshold is ALL, not CHROMA", + "got " + std::string(StateName( + Probe::ClassifyPlaneMeansQ8(511u, 511u, 511u)))); +} + +// =========================================================================== +// GROUP 2 -- THE SCORER, over synthetic NV12 images +// =========================================================================== + +// A host-visible LINEAR NV12 image the way a driver actually hands one back: +// a non-zero plane offset, a row pitch wider than the width, and the chroma +// plane after the luma one at its own offset. The three "poison" values below +// are what make this a test of the ADDRESSING and not just of the arithmetic. +struct SyntheticNv12 { + static constexpr uint32_t kWidth = 64; + static constexpr uint32_t kHeight = 32; + static constexpr uint32_t kLumaPitch = kWidth + 37; // padded, odd + static constexpr uint32_t kChromaPitch = (kWidth / 2) * 2 + 23; + static constexpr uint64_t kLumaOffset = 64; // not 0 + static constexpr uint64_t kChromaOffset = + kLumaOffset + (uint64_t)kLumaPitch * kHeight; + + std::vector bytes; + VkSubresourceLayout luma{}; + VkSubresourceLayout chroma{}; + + // |sampledY| is written into the rows the kRowStride=8 walk READS; + // |skippedY| into the rows it must SKIP; padding past the width gets + // 0xFF. Same for chroma. If the scorer walks every row, or misreads the + // pitch, or runs off the end of a row, it picks up |skippedY| or 0xFF and + // the asserted mean moves. + SyntheticNv12(uint8_t sampledY, uint8_t skippedY, + uint8_t sampledU, uint8_t sampledV, + uint8_t skippedU, uint8_t skippedV) + { + const uint32_t cw = kWidth / 2; + const uint32_t ch = kHeight / 2; + bytes.assign((size_t)kChromaOffset + (size_t)kChromaPitch * ch + 64, + 0xFFu); + for (uint32_t y = 0; y < kHeight; y++) { + uint8_t* r = bytes.data() + kLumaOffset + (size_t)y * kLumaPitch; + const uint8_t v = ((y % 8u) == 0u) ? sampledY : skippedY; + memset(r, v, kWidth); + // Row padding stays 0xFF from the fill above. + } + for (uint32_t y = 0; y < ch; y++) { + uint8_t* r = bytes.data() + kChromaOffset + (size_t)y * kChromaPitch; + const bool sampled = ((y % 8u) == 0u); + for (uint32_t x = 0; x < cw; x++) { + r[(2 * x) + 0] = sampled ? sampledU : skippedU; + r[(2 * x) + 1] = sampled ? sampledV : skippedV; + } + } + luma.offset = kLumaOffset; + luma.rowPitch = kLumaPitch; + chroma.offset = kChromaOffset; + chroma.rowPitch = kChromaPitch; + } + + Probe::State Score(uint32_t& y, uint32_t& u, uint32_t& v) const + { + std::vector scratch; + Probe::ScorePlanesQ8(bytes.data(), luma, chroma, kWidth, kHeight, + scratch, y, u, v); + return Probe::ClassifyPlaneMeansQ8(y, u, v); + } +}; + +void CaseScorerOnSyntheticBuffers() +{ + g_currentCase = "scorer: synthetic NV12, padded pitch and plane offsets"; + + struct Buf { + const char* name; + uint8_t sy, ky, su, sv, ku, kv; + uint32_t wantY, wantU, wantV; + Probe::State want; + }; + const Buf bufs[] = { + // Ordinary content. Skipped rows and padding carry values that would + // move every one of these means if they were read. + { "ordinary content", 100, 200, 110, 120, 10, 20, + Q8(100), Q8(110), Q8(120), Probe::STATE_CLEAN }, + // THE FALSE-POSITIVE CASE. A fully legal black NV12 frame: Y=16, + // U=V=128. Zeroed is NOT black, and this is the assertion that says + // so -- if it ever goes red, the workaround has started rerouting + // buffers that were working. + { "fully legal BLACK frame (Y=16, U=V=128)", 16, 16, 128, 128, 128, 128, + Q8(16), Q8(128), Q8(128), Probe::STATE_CLEAN }, + // CHROMA_ZERO: measured mode 1. Luma alive, chroma plane dead. The + // skipped chroma rows are ALIVE, so a scorer that walked every row + // would find live chroma and miss the damage entirely. + { "CHROMA_ZERO (measured mode 1)", 100, 200, 0, 0, 200, 200, + Q8(100), 0, 0, Probe::STATE_DAMAGED_CHROMA }, + // ALL_ZERO: measured mode 2. + { "ALL_ZERO (measured mode 2)", 0, 200, 0, 0, 200, 200, + 0, 0, 0, Probe::STATE_DAMAGED_ALL }, + }; + for (const Buf& b : bufs) { + SyntheticNv12 img(b.sy, b.ky, b.su, b.sv, b.ku, b.kv); + uint32_t y = 0, u = 0, v = 0; + const Probe::State got = img.Score(y, u, v); + Check(y == b.wantY, + "luma mean is exact (pitch, offset and row stride all right)", + std::string(b.name) + ": meanY=" + U64(y) + ", expected " + + U64(b.wantY)); + Check(u == b.wantU, "U mean is exact", + std::string(b.name) + ": meanU=" + U64(u) + ", expected " + + U64(b.wantU)); + Check(v == b.wantV, "V mean is exact", + std::string(b.name) + ": meanV=" + U64(v) + ", expected " + + U64(b.wantV)); + Check(got == b.want, "the verdict for this buffer", + std::string(b.name) + ": " + StateName(got) + ", expected " + + StateName(b.want)); + } +} + +// =========================================================================== +// GROUP 3 -- THE LATCH, THE ORDERING, THE DRAIN +// =========================================================================== + +VkSharedBaseObj MakeProbe() +{ + VkSharedBaseObj p; + Probe::Create(p); + return p; +} + +void CaseArmingAndApplicability() +{ + g_currentCase = "latch: arming, applicability, once-per-registration"; + + VkSharedBaseObj p = MakeProbe(); + Check(p != nullptr, "the probe object is creatable", "Create returned null"); + if (!p) { + return; + } + Check(p->GetRegistrationState(1) == Probe::STATE_NOT_EVALUATED, + "an unknown registration is NOT_EVALUATED", + StateName(p->GetRegistrationState(1))); + + p->ArmRegistration(1, Probe::CaptureSite::kReachable); + Check(p->GetRegistrationState(1) == Probe::STATE_ARMED, + "a registration whose capture site is reachable arms", + StateName(p->GetRegistrationState(1))); + Check(p->NeedsCapture(1), "an armed registration wants a capture", "it did not"); + + // A DIRECTLY ENCODABLE REGISTRATION IS NOT EXCUSED FROM PROBING, and is + // deliberately not NOT_APPLICABLE. + // Directly-encodable registrations are the BLOCK-LINEAR ones, i.e. the + // measured damage class, so excusing them is precisely the blindness the + // probe exists to avoid. What IS NOT_APPLICABLE is a registration the probe + // genuinely cannot ride: no legal copy out of the image, or a FILTER route + // that samples rather than copies. Arming one of those would leave it + // ARMED forever, reporting NOT_EVALUATED -- an observable that cannot + // fail, which is worse than an honest refusal. + p->ArmRegistration(2, Probe::CaptureSite::kUnreachable); + Check(p->GetRegistrationState(2) == Probe::STATE_NOT_APPLICABLE, + "a registration with no reachable capture site is NOT_APPLICABLE", + StateName(p->GetRegistrationState(2))); + Check(!p->NeedsCapture(2), "and it is never asked for a capture", "it was"); + + // ONCE PER BUFFER. The second measurement is the one that would arrive if + // the capture were per frame instead of per registration; it must not + // move the state or the counters. + p->ApplyVerdict(1, Q8(100), 0, 0); + Check(p->GetRegistrationState(1) == Probe::STATE_DAMAGED_CHROMA, + "the first measurement latches", + StateName(p->GetRegistrationState(1))); + Check(!p->NeedsCapture(1), "and the registration stops asking", + "it still wants a capture"); + p->ApplyVerdict(1, Q8(100), Q8(110), Q8(120)); + Check(p->GetRegistrationState(1) == Probe::STATE_DAMAGED_CHROMA, + "a SECOND measurement for the same registration is ignored", + StateName(p->GetRegistrationState(1))); + + Probe::Verdict verdict; + uint32_t probed = 0, damaged = 0, armedNow = 0; + p->GetSnapshot(verdict, probed, damaged, armedNow); + Check(probed == 1, + "the session counts one PROBED registration, not two measurements", + "probed=" + U64(probed)); + Check(damaged == 1, "and one damaged", "damaged=" + U64(damaged)); + // Registration 1 was scored and registration 2 was refused honestly, so + // nothing is still waiting. This is the value the case below shows is + // NOT automatic. + Check(armedNow == 0, + "and nothing is left waiting for a verdict", + "armed=" + U64(armedNow)); +} + +// F1/F2. THE ASSERTION THAT MAKES THE PROBE ITSELF FAIL-ABLE. +// +// The verifier's finding, restated so it stays fixed: arming a registration +// whose capture never runs leaves it STATE_ARMED forever, and the session +// then reports probed=0 damaged=0 with a NOT_EVALUATED verdict -- BYTE-FOR- +// BYTE what a session that armed nothing reports, and one reading away from +// "every buffer was checked and every buffer was clean". That is the same +// disguised-inertness shape this whole probe exists to expose, reproduced one +// layer down inside the probe. +// +// It matters most for exactly the class D2 added. A DIRECT registration only +// reaches the capture site through the one-frame staged detour in +// VkVideoEncoder::SetExternalInputFrameWithNode; if that detour stops firing, +// every block-linear buffer arms and none is ever scored, and WITHOUT THIS +// COUNTER the session reports clean. With it, the session reports "N buffers +// were promised a verdict and never got one", which is a failure a caller can +// see and a test can assert. +void CaseArmedButNeverScoredIsCounted() +{ + g_currentCase = "latch: ARMED-and-never-scored is counted, not silent"; + + VkSharedBaseObj p = MakeProbe(); + if (!p) { + Check(false, "probe creatable", "null"); + return; + } + + Probe::Verdict v; + uint32_t probed = 0, damaged = 0, armed = 0; + + // Three buffers imported and armed. No capture has landed yet -- which is + // the honest mid-session state, and also exactly what a permanently + // broken capture site looks like. + p->ArmRegistration(21, Probe::CaptureSite::kReachable); + p->ArmRegistration(22, Probe::CaptureSite::kReachable); + p->ArmRegistration(23, Probe::CaptureSite::kReachable); + + p->GetSnapshot(v, probed, damaged, armed); + Check(probed == 0 && damaged == 0, + "with nothing yet scored the two session totals are zero -- which is " + "ALSO what a session that armed nothing reports, and is why they " + "cannot carry this question", + "probed=" + U64(probed) + " damaged=" + U64(damaged)); + Check(v.state == Probe::STATE_NOT_EVALUATED, + "and the reported verdict is the NOT_EVALUATED default", + StateName(v.state)); + Check(armed == 3, + "but THREE registrations are recorded as still waiting for a " + "verdict, which is the fact the totals above cannot express", + "armed=" + U64(armed)); + + // A capture lands for one of them. The count falls -- it is a live + // outstanding count, not a session total, and that direction is the + // useful one: it is what lets 'still non-zero at the end' mean something. + p->ApplyVerdict(22, Q8(90), Q8(115), Q8(125)); + p->GetSnapshot(v, probed, damaged, armed); + Check(probed == 1, "scoring one moves the probed total", + "probed=" + U64(probed)); + Check(armed == 2, "and drops the outstanding count to two", + "armed=" + U64(armed)); + + p->ApplyVerdict(21, Q8(100), 0, 0); + p->ApplyVerdict(23, 0, 0, 0); + p->GetSnapshot(v, probed, damaged, armed); + Check(armed == 0, + "a session in which every armed buffer was scored ends with nothing " + "outstanding -- this zero is the discriminating value, and the 3 " + "above is what a dead capture site would leave here instead", + "armed=" + U64(armed)); + Check(damaged == 2, "and the damage it found is still reported", + "damaged=" + U64(damaged)); + + // A REGISTRATION REFUSED HONESTLY IS NOT OUTSTANDING. NOT_APPLICABLE is + // an answer; it must not be counted as a buffer awaiting one, or the + // number would never reach zero on any session with a FILTER route in it + // and would stop discriminating. + p->ArmRegistration(24, Probe::CaptureSite::kUnreachable); + p->GetSnapshot(v, probed, damaged, armed); + Check(armed == 0, + "a NOT_APPLICABLE registration is an ANSWER, not an outstanding " + "promise, and does not inflate the count", + "armed=" + U64(armed)); + + // AND RETIREMENT DRAINS IT. A consumer that reroutes a buffer and + // unregisters it must not leave a permanent phantom in this count, or a + // long session accumulates one and the signal decays to noise. + p->ArmRegistration(25, Probe::CaptureSite::kReachable); + p->GetSnapshot(v, probed, damaged, armed); + Check(armed == 1, "a freshly armed buffer is outstanding", + "armed=" + U64(armed)); + p->ForgetRegistration(25); + p->GetSnapshot(v, probed, damaged, armed); + Check(armed == 0, + "and retiring it drains the count rather than leaving a phantom", + "armed=" + U64(armed)); +} + +// F1. THE DETOUR'S CONTROLLING PREDICATE, PINNED. +// +// This does not execute the detour -- that needs a device, and +// test/encoder-ext-format-encode --content-probe does it on real hardware. +// What it pins is the ONE fact the detour reads, NeedsCapture(), including +// the property that makes the detour affordable at all: it is TRUE for +// exactly one frame per registration. +// +// VkVideoEncoder::SetExternalInputFrameWithNode routes a directly-encodable +// frame down the staged path iff `directlyEncodable && probeStillOwesA- +// Capture`, and probeStillOwesACapture IS NeedsCapture(). So "the detour +// fires once and then stops" and "NeedsCapture goes true then false" are the +// same statement, and the second one is testable on any host. +void CaseDetourBudgetIsOneFramePerRegistration() +{ + g_currentCase = "latch: the staged detour is one frame per registration"; + + VkSharedBaseObj p = MakeProbe(); + if (!p) { + Check(false, "probe creatable", "null"); + return; + } + + // Not armed at all: a caller that never chained the struct must never be + // detoured. This is the "costs nothing when you did not ask" property. + Check(!p->NeedsCapture(31), + "an unregistered buffer never owes a capture, so an un-opted-in " + "session takes no detour at all", + "it claimed to"); + + p->ArmRegistration(31, Probe::CaptureSite::kReachable); + Check(p->NeedsCapture(31), + "frame 1 of an armed DIRECT registration owes a capture, which is " + "what sends it down the staged path", + "it did not"); + + // The verdict is what the scored capture produces. After it, the budget + // is spent and every later frame of this buffer goes DIRECT again. + p->ApplyVerdict(31, Q8(90), Q8(115), Q8(125)); + Check(!p->NeedsCapture(31), + "and frame 2 does NOT -- the detour is bounded at one frame per " + "registration, so a 5125-frame session over 5 buffers pays 5 " + "detours and not 5125", + "it still owes one"); + Check(p->GetRegistrationState(31) == Probe::STATE_CLEAN, + "with the verdict latched", + StateName(p->GetRegistrationState(31))); + + // A registration the probe cannot ride must never detour even once -- + // otherwise every FILTER-routed frame in a session pays a staged copy for + // a capture that can never happen. + p->ArmRegistration(32, Probe::CaptureSite::kUnreachable); + Check(!p->NeedsCapture(32), + "a registration with no reachable capture site is never detoured", + "it was"); +} + +void CaseOldestDamagedFirstAndDrain() +{ + g_currentCase = "latch: oldest-damaged-first, and the drain on retirement"; + + VkSharedBaseObj p = MakeProbe(); + if (!p) { + Check(false, "probe creatable", "null"); + return; + } + for (uint64_t id = 10; id <= 14; id++) { + p->ArmRegistration(id, Probe::CaptureSite::kReachable); + } + // Scored in this order: clean, chroma-dead, clean, all-dead, clean. + p->ApplyVerdict(10, Q8(100), Q8(110), Q8(120)); + p->ApplyVerdict(11, Q8(100), 0, 0); + p->ApplyVerdict(12, Q8(90), Q8(115), Q8(125)); + p->ApplyVerdict(13, 0, 0, 0); + p->ApplyVerdict(14, Q8(80), Q8(100), Q8(100)); + + Probe::Verdict v; + uint32_t probed = 0, damaged = 0, armedNow = 0; + p->GetSnapshot(v, probed, damaged, armedNow); + Check(probed == 5, "five registrations reached a verdict", + "probed=" + U64(probed)); + Check(damaged == 2, "two of them were damaged", "damaged=" + U64(damaged)); + Check(v.registrationId == 11, + "the report names the OLDEST damaged registration, not the newest -- " + "a consumer polls between frames and must not lose the older one", + "reported id " + U64(v.registrationId)); + Check(v.state == Probe::STATE_DAMAGED_CHROMA, "with its own verdict", + StateName(v.state)); + Check((v.meanY == Q8(100)) && (v.meanU == 0) && (v.meanV == 0), + "and the arithmetic behind it", + "Y=" + U64(v.meanY) + " U=" + U64(v.meanU) + " V=" + U64(v.meanV)); + + // IDEMPOTENT: reading does not consume. A consumer that reads twice + // before acting must see the same answer both times. + Probe::Verdict again; + p->GetSnapshot(again, probed, damaged, armedNow); + Check(again.registrationId == 11, "reading the report does not consume it", + "second read gave id " + U64(again.registrationId)); + + // THE DRAIN. Retiring the damaged registration is what surfaces the next + // one -- this is the mechanism that lets a consumer work through several + // damaged buffers one poll at a time. + p->ForgetRegistration(11); + Check(p->GetRegistrationState(11) == Probe::STATE_NOT_EVALUATED, + "retirement forgets the registration itself, not just its place in " + "the damaged order -- the two erasures inside ForgetRegistration " + "are redundant for the REPORT (either alone drains it) and this is " + "the assertion that pins both", + StateName(p->GetRegistrationState(11))); + p->GetSnapshot(v, probed, damaged, armedNow); + Check(v.registrationId == 13, + "retiring the reported registration surfaces the NEXT damaged one", + "reported id " + U64(v.registrationId)); + Check(v.state == Probe::STATE_DAMAGED_ALL, "with its own verdict", + StateName(v.state)); + Check(damaged == 2, + "the SESSION total does not fall when a damaged buffer is retired -- " + "'two of this session's buffers were damaged' stays true", + "damaged=" + U64(damaged)); + + p->ForgetRegistration(13); + p->GetSnapshot(v, probed, damaged, armedNow); + Check(v.state == Probe::STATE_CLEAN, + "with no damaged registration outstanding the report falls back to " + "the most recent CLEAN verdict", + StateName(v.state)); + Check(v.registrationId == 14, "which is the last one scored clean", + "reported id " + U64(v.registrationId)); +} + +// =========================================================================== +// GROUP 4 -- THE CARRIER +// =========================================================================== + +class Session { +public: + bool Open() + { + if ((CreateVulkanVideoEncoderExt(m_encoder) != VK_SUCCESS) || !m_encoder) { + std::printf(" ERROR: CreateVulkanVideoEncoderExt failed\n"); + return false; + } + if (VkEncInstallNullBackend(m_encoder.get(), &m_backend) != VK_SUCCESS) { + std::printf(" ERROR: VkEncInstallNullBackend failed\n"); + return false; + } + return true; + } + VulkanVideoEncoderExt* Get() const { return m_encoder.get(); } + + static VkVideoEncoderExternalImageDescriptor Descriptor() + { + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = 640; + desc.height = 360; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.existingImage = (VkImage)(uintptr_t)0xA110C8ED; + return desc; + } + +private: + VkSharedBaseObj m_encoder; + VkEncNullBackendState m_backend{}; +}; + +VkVideoEncoderStatus FreshStatus() +{ + VkVideoEncoderStatus status = {}; + status.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; + return status; +} + +VkVideoEncoderImportContentInfo FreshContentInfo() +{ + VkVideoEncoderImportContentInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_CONTENT_INFO; + return info; +} + +VkVideoEncoderImportGuardInfo FreshGuardInfo() +{ + VkVideoEncoderImportGuardInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO; + return info; +} + +// Values no library path produces, so "the library wrote this" and "the +// library did not" stay distinguishable. probeGeneration is deliberately +// poisoned too: it is the writer proof, and a writer proof that is only +// checked against 0 cannot tell a stamped 0 from an untouched one. +void PoisonContentInfo(VkVideoEncoderImportContentInfo& info) +{ + info.state = VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_ALL; + info.resource = 0xDEADBEEFu; + info.probeGeneration = 0xBADC0DEu; + info.meanY = 0xAAAAu; + info.meanU = 0xBBBBu; + info.meanV = 0xCCCCu; + info.probedRegistrationCount = 0x1234u; + info.damagedRegistrationCount = 0x5678u; +} + +// C1. Chaining the struct onto a registration is accepted, arms the probe, +// and the echo reports ARMED with the writer proof stamped. +void CaseRegistrationEchoArmsAndReports(Session& s) +{ + g_currentCase = "carrier: chaining onto RegisterImageResource arms it"; + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "a chained VkVideoEncoderImportContentInfo is ACCEPTED", + "status " + U64((uint64_t)reg)); + Check(content.probeGeneration == + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION, + "probeGeneration is stamped: the writer proof, and it is NON-ZERO " + "by contract so it cannot decay into 'same as an untouched struct' " + "the way the guard's requestedCount did when the guard was retired", + "probeGeneration " + U64(content.probeGeneration)); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED, + "the registration echo reports ARMED -- at registration time the " + "producer has written nothing, so there is no verdict to give yet", + "state " + U64((uint64_t)content.state)); + Check(content.resource == resource, + "and names the registration it armed", + "echo said " + U64(content.resource) + ", id was " + U64(resource)); + Check(content.meanY == 0 && content.meanU == 0 && content.meanV == 0, + "the echo carries no measurement, and says so with zeros (the " + "poison was cleared)", + "Y=" + U64(content.meanY) + " U=" + U64(content.meanU) + " V=" + + U64(content.meanV)); +} + +// C1b. THE BLOCK-LINEAR REGISTRATION, which is the shape a probe gated on +// encodeCapable is structurally blind to. +// +// encodeCapable is `encodableFormat && (tiling != VK_IMAGE_TILING_LINEAR) && +// (usage & VIDEO_ENCODE_SRC)`. Deciding the arm on it means a BLOCK-LINEAR +// NV12 import that also declares +// VIDEO_ENCODE_SRC -- i.e. the exact descriptor a compositor hands over on a +// modifier that carries encode-src -- came out encodeCapable and was latched +// NOT_APPLICABLE: never probed, on a class of buffer that is not +// hypothetical. The periodicity harness measured block-linear GBM buffers +// (modifier 0x0300000000606014) poisoned on 8 of 64 import ordinals, four +// ALL_ZERO and four CHROMA_ZERO. +// +// This case pins the fix at the ONE line that decides it. Against the +// previous library it reports state 1 (NOT_APPLICABLE) and fails; the only +// thing that changes it is dropping tiling/encodeCapable out of the arm +// decision. +void CaseBlockLinearDirectRegistrationArms(Session& s) +{ + g_currentCase = "carrier: a BLOCK-LINEAR encode-src registration ARMS"; + + VkVideoEncoderExternalImageDescriptor desc = Session::Descriptor(); + // Block-linear. VK_IMAGE_TILING_OPTIMAL is 0, so this is also what the + // base descriptor already says -- named explicitly because it is the + // clause under test and a reader must not have to know that 0 is OPTIMAL. + desc.tiling = VK_IMAGE_TILING_OPTIMAL; + // ...and the usage that tips encodeCapable to true. TRANSFER_SRC is + // present because the probe's capture is a copy out of the image and the + // new predicate requires exactly that bit; VIDEO_ENCODE_SRC is what makes + // the registration DIRECT. + desc.imageUsage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = + s.Get()->RegisterImageResource(desc, 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "a block-linear VIDEO_ENCODE_SRC registration is accepted", + "status " + U64((uint64_t)reg)); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED, + "and it ARMS -- the probe's arm decision no longer reads tiling, so " + "a block-linear import is covered instead of being latched " + "NOT_APPLICABLE for being directly encodable", + "state " + U64((uint64_t)content.state)); + Check(content.resource == resource, "and names the registration it armed", + "echo said " + U64(content.resource) + ", id was " + U64(resource)); +} + +// C1c. THE NEGATIVE CONTROL, and it is what keeps C1b from degenerating into +// "arm everything". A registration the probe genuinely cannot ride must still +// answer NOT_APPLICABLE, because arming it would leave it ARMED forever and +// reporting NOT_EVALUATED -- an observable that cannot fail. +// +// The case chosen is an import with no TRANSFER_SRC. The capture is a +// vkCmdCopyImage out of the imported image; without that usage bit the copy +// is a VUID violation, so "the probe can ride it" is false as a matter of +// spec, not of policy. +void CaseUncopyableRegistrationStaysNotApplicable(Session& s) +{ + g_currentCase = "carrier: an import with no TRANSFER_SRC is NOT_APPLICABLE"; + + VkVideoEncoderExternalImageDescriptor desc = Session::Descriptor(); + desc.tiling = VK_IMAGE_TILING_OPTIMAL; + desc.imageUsage = VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = + s.Get()->RegisterImageResource(desc, 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "the registration itself is fine", "status " + U64((uint64_t)reg)); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_APPLICABLE, + "but the probe refuses it honestly: no TRANSFER_SRC means no legal " + "copy out of the image, so there is no capture site and " + "NOT_APPLICABLE is the true answer rather than a permanent ARMED", + "state " + U64((uint64_t)content.state)); +} + +// C2. NOT opting in must cost nothing and must leave the channel silent. +// This is the assertion that the default-off path is really off. +// C1d. A FORMAT THE SCORER CANNOT READ IS REFUSED AT REGISTRATION, NOT A +// FRAME LATER. +// +// FOUND BY THE HARDWARE TEST, NOT BY REVIEW. encoder-ext-format-encode +// --content-probe row [2/13] P010 on an RTX A4000 reported, before the fix: +// +// armEcho=ARMED(gen=1) final=NOT_EVALUATED probed=0 damaged=0 armedOutstanding=0 +// +// The arm predicate asked whether a copy could be RECORDED out of the image +// and never whether the bytes it produced could be READ. RecordCapture asked +// the second question on the first frame and latched NOT_APPLICABLE quietly, +// so the session ended reporting the same three zeros a fully-probed, fully- +// clean session reports. +// +// The rule this restores is the one C1c states: an ARMED echo is a PROMISE of +// a verdict. A promise that is withdrawn a frame later is worse than an +// honest refusal, because armedRegistrationCount -- the count that exists to +// catch exactly this -- cannot see a registration that has already been +// downgraded. +void CaseUnscorableFormatIsRefusedAtRegistration(Session& s) +{ + g_currentCase = "carrier: a 10-bit import is NOT_APPLICABLE at REGISTRATION"; + + VkVideoEncoderExternalImageDescriptor desc = Session::Descriptor(); + // Everything else is the shape that DOES arm -- block-linear, encode-src, + // transfer-src -- so the ONLY thing this case varies is the format. Same + // discipline as C1b/C1c: one clause at a time. + desc.tiling = VK_IMAGE_TILING_OPTIMAL; + desc.imageUsage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + desc.format = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16; + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = + s.Get()->RegisterImageResource(desc, 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "the registration itself is fine -- the probe's opinion is not a " + "registration error", + "status " + U64((uint64_t)reg)); + Check(content.state == + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_APPLICABLE, + "a 10-bit import is refused AT REGISTRATION, where the caller can " + "still see it, rather than echoing ARMED and being downgraded on " + "the first frame", + "state " + U64((uint64_t)content.state)); + + // AND THE 8-BIT CONTROL, so this is not "refuse everything". Without it a + // predicate hardwired to false would pass the assertion above. + VkVideoEncoderExternalImageDescriptor ok = desc; + ok.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + VkVideoEncoderImportContentInfo okContent = FreshContentInfo(); + PoisonContentInfo(okContent); + VkVideoEncoderStatus okStatus = FreshStatus(); + okStatus.pNext = &okContent; + VkVideoEncoderResource okResource = VK_VIDEO_ENCODER_RESOURCE_NULL; + s.Get()->RegisterImageResource(ok, 0, &okResource, &okStatus); + Check(okContent.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED, + "while the same descriptor in 8-bit 420 still ARMS -- the clause " + "added is the format, not a blanket refusal", + "state " + U64((uint64_t)okContent.state)); +} + +void CaseNoOptInLeavesTheChannelSilent() +{ + g_currentCase = "carrier: a session that never chains it reports nothing"; + + Session s; + if (!s.Open()) { + Check(false, "session", "setup failed"); + return; + } + VkVideoEncoderStatus status = FreshStatus(); + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "an unchained registration still works", "status " + U64((uint64_t)reg)); + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderCompletionInfo info = {}; + info.pNext = &content; + const VkResult r = s.Get()->GetCompletionInfo(&info); + Check(r == VK_SUCCESS, "GetCompletionInfo accepts the struct", + "VkResult " + U64((uint64_t)r)); + Check(content.probeGeneration == + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION, + "probeGeneration is stamped even with nothing armed -- 'the library " + "has the observable and has nothing to say' must be readable apart " + "from 'the library predates the observable'", + "probeGeneration " + U64(content.probeGeneration)); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED, + "and the verdict is NOT_EVALUATED: nothing was armed, so nothing " + "was probed", + "state " + U64((uint64_t)content.state)); + Check(content.probedRegistrationCount == 0 && + content.damagedRegistrationCount == 0, + "with zero probed and zero damaged (the poison was cleared)", + "probed " + U64(content.probedRegistrationCount) + " damaged " + + U64(content.damagedRegistrationCount)); +} + +// C3. The chain gate, WIDENED not relaxed. Two known link types are now +// accepted, in either order; everything the narrower rule refused it still +// refuses. +void CaseChainGateWidenedNotRelaxed(Session& s) +{ + g_currentCase = "carrier: the widened chain gate still refuses skew"; + + // Both known links, content first. + { + VkVideoEncoderImportGuardInfo guard = FreshGuardInfo(); + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + content.pNext = &guard; + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "content-then-guard is accepted", + "status " + U64((uint64_t)reg)); + Check(content.probeGeneration == + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION, + "and both links are written: content", "content not written"); + Check(guard.state == VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED, + "and both links are written: guard", + "guard state " + U64((uint64_t)guard.state)); + } + // Both known links, guard first. + { + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + VkVideoEncoderImportGuardInfo guard = FreshGuardInfo(); + guard.pNext = &content; + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &guard; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "guard-then-content is accepted too (order does not matter)", + "status " + U64((uint64_t)reg)); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED, + "and the content link is still armed and written", + "state " + U64((uint64_t)content.state)); + } + // A REPEATED content link is version skew and is refused. + { + VkVideoEncoderImportContentInfo a = FreshContentInfo(); + VkVideoEncoderImportContentInfo b = FreshContentInfo(); + a.pNext = &b; + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &a; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "a REPEATED content link is refused -- two links of one type " + "mean the caller believes it is getting two different things", + "status " + U64((uint64_t)reg)); + } + // An unknown sType behind a known one is still refused. + { + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + VkVideoEncoderImportGuardInfo bogus = FreshGuardInfo(); + bogus.sType = (VkVideoEncoderStructureType)0x56450FFF; + content.pNext = &bogus; + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "an unknown sType chained behind a known one is still refused", + "status " + U64((uint64_t)reg)); + } + // A chain hanging off a MIS-STAMPED status is refused AND not written. + { + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderStatus status = FreshStatus(); + status.sType = (VkVideoEncoderStructureType)0x56450FFE; + status.pNext = &content; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "a chain on a mis-stamped status is refused", + "status " + U64((uint64_t)reg)); + Check(content.probeGeneration == 0xBADC0DEu, + "and is NOT written through -- writing into a struct whose gate " + "rejected the request is the exact version-skew failure the " + "gate exists to prevent", + "probeGeneration " + U64(content.probeGeneration)); + } +} + +// C4. The completion channel: the same struct, the verdict side, and its own +// refusal rules unchanged. +void CaseCompletionChannel(Session& s) +{ + g_currentCase = "carrier: GetCompletionInfo is the verdict channel"; + + // The session |s| has armed several registrations by now (C1 and C3), but + // nothing has been SCORED -- a null-backend session runs no frames. That + // is exactly the ARMED-but-unscored state, and NOT_EVALUATED is its + // honest report. + { + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderCompletionInfo info = {}; + info.pNext = &content; + const VkResult r = s.Get()->GetCompletionInfo(&info); + Check(r == VK_SUCCESS, "the completion chain accepts the struct", + "VkResult " + U64((uint64_t)r)); + Check(content.probeGeneration == + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION, + "probeGeneration stamped here too", + "probeGeneration " + U64(content.probeGeneration)); + Check(content.state == + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED, + "armed but never scored reports NOT_EVALUATED, not ARMED: the " + "verdict channel answers about VERDICTS", + "state " + U64((uint64_t)content.state)); + Check(content.probedRegistrationCount == 0, + "and nothing has been probed", + "probed " + U64(content.probedRegistrationCount)); + } + // Repeated link, refused, exactly as every other known link on this call. + { + VkVideoEncoderImportContentInfo a = FreshContentInfo(); + VkVideoEncoderImportContentInfo b = FreshContentInfo(); + a.pNext = &b; + VkVideoEncoderCompletionInfo info = {}; + info.pNext = &a; + Check(s.Get()->GetCompletionInfo(&info) == VK_ERROR_INITIALIZATION_FAILED, + "a repeated content link on GetCompletionInfo is refused", + "it was accepted"); + } + // Alongside the other known links, all of which must still work. + { + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + VkVideoEncoderImportGuardInfo guard = FreshGuardInfo(); + VkVideoEncoderDiagnosticInfo diag = {}; + diag.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_DIAGNOSTIC_INFO; + diag.pNext = &guard; + guard.pNext = &content; + VkVideoEncoderCompletionInfo info = {}; + info.pNext = &diag; + Check(s.Get()->GetCompletionInfo(&info) == VK_SUCCESS, + "diagnostic + guard + content chained together are all accepted", + "the chain was refused"); + Check(content.probeGeneration == + VK_VIDEO_ENCODER_IMPORT_CONTENT_PROBE_GENERATION, + "and the content link is written", "it was not"); + } +} + +// C5. THE WHOLE EXT-SIDE PATH, END TO END, minus the GPU readback: register +// with the struct chained, feed the probe the measurement a poisoned import +// produces, and read the verdict back off GetCompletionInfo -- then retire the +// buffer, which is the consumer's reaction, and watch the report drain. +// +// The measurement is injected through VkEncInjectImportContentMeasurement +// (the internal seam) because the capture that would otherwise produce it +// needs a dma-buf import, a staging command buffer and a fence. What is +// injected is a MEASUREMENT and never a verdict: every state asserted below +// is the production predicate's own answer. +void CaseVerdictReachesTheCompletionChannel() +{ + g_currentCase = "carrier: a damaged verdict reaches GetCompletionInfo"; + + Session s; + if (!s.Open()) { + Check(false, "session", "setup failed"); + return; + } + // Two armed registrations, so the ordering and the drain are visible + // across the ext boundary and not only inside the probe object. + VkVideoEncoderResource first = VK_VIDEO_ENCODER_RESOURCE_NULL; + VkVideoEncoderResource second = VK_VIDEO_ENCODER_RESOURCE_NULL; + for (VkVideoEncoderResource* out : { &first, &second }) { + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &content; + s.Get()->RegisterImageResource(Session::Descriptor(), 0, out, &status); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_ARMED, + "armed", "state " + U64((uint64_t)content.state)); + } + + // CHROMA_ZERO on the first buffer, ALL_ZERO on the second -- the two + // damage modes the driver produces, in that order. + Check(VkEncInjectImportContentMeasurement(s.Get(), first, Q8(100), 0, 0) == + VK_SUCCESS, + "the measurement seam reaches the armed probe", "it did not"); + Check(VkEncInjectImportContentMeasurement(s.Get(), second, 0, 0, 0) == + VK_SUCCESS, + "and the second one too", "it did not"); + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderCompletionInfo info = {}; + info.pNext = &content; + Check(s.Get()->GetCompletionInfo(&info) == VK_SUCCESS, + "the completion call succeeds", "it did not"); + Check(content.state == + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_CHROMA, + "the CHROMA_ZERO verdict crosses the ext boundary intact", + "state " + U64((uint64_t)content.state)); + Check(content.resource == first, + "naming the registration a consumer has to reroute -- WHICH BUFFER " + "is the entire actionable content of this channel", + "reported " + U64(content.resource) + ", expected " + U64(first)); + Check(content.meanY == Q8(100) && content.meanU == 0 && content.meanV == 0, + "with the arithmetic behind it", + "Y=" + U64(content.meanY) + " U=" + U64(content.meanU) + " V=" + + U64(content.meanV)); + Check(content.probedRegistrationCount == 2 && + content.damagedRegistrationCount == 2, + "and the session totals", + "probed " + U64(content.probedRegistrationCount) + " damaged " + + U64(content.damagedRegistrationCount)); + + // THE CONSUMER'S REACTION, and the assertion that it is wired: retiring + // the buffer named above must surface the NEXT damaged one. Without + // UnregisterImageResource reaching the probe, this call reports |first| + // forever and a consumer can never work past the first damaged buffer. + Check(s.Get()->UnregisterImageResource(first) == + VK_VIDEO_ENCODER_STATUS_SUCCESS, + "the damaged registration retires", "unregister refused"); + content = FreshContentInfo(); + info = VkVideoEncoderCompletionInfo{}; + info.pNext = &content; + s.Get()->GetCompletionInfo(&info); + Check(content.state == VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_DAMAGED_ALL, + "retiring it surfaces the NEXT damaged buffer, with ITS verdict", + "state " + U64((uint64_t)content.state)); + Check(content.resource == second, "and its id", + "reported " + U64(content.resource) + ", expected " + U64(second)); + + Check(s.Get()->UnregisterImageResource(second) == + VK_VIDEO_ENCODER_STATUS_SUCCESS, + "the second one retires too", "unregister refused"); + content = FreshContentInfo(); + info = VkVideoEncoderCompletionInfo{}; + info.pNext = &content; + s.Get()->GetCompletionInfo(&info); + Check(content.state == + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED, + "with every damaged buffer retired and none ever scored clean, the " + "channel goes quiet again", + "state " + U64((uint64_t)content.state)); + Check(content.damagedRegistrationCount == 2, + "but the SESSION total still remembers both -- 'two of this " + "session's buffers were damaged' does not stop being true when they " + "are retired", + "damaged " + U64(content.damagedRegistrationCount)); +} + +// C6. The once-per-registration rule holds ACROSS the ext boundary too, and +// a measurement for a retired registration is dropped rather than resurrecting +// a verdict the consumer has already acted on. +void CaseMeasurementForARetiredRegistrationIsDropped() +{ + g_currentCase = "carrier: a measurement for a retired registration is dropped"; + + Session s; + if (!s.Open()) { + Check(false, "session", "setup failed"); + return; + } + VkVideoEncoderImportContentInfo armed = FreshContentInfo(); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &armed; + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + s.Get()->RegisterImageResource(Session::Descriptor(), 0, &resource, &status); + s.Get()->UnregisterImageResource(resource); + + // A capture recorded before the retirement can still land after it: the + // fence wait that scores it is on another thread. It must not resurrect a + // verdict for a buffer the consumer has already finished with. + VkEncInjectImportContentMeasurement(s.Get(), resource, Q8(100), 0, 0); + + VkVideoEncoderImportContentInfo content = FreshContentInfo(); + PoisonContentInfo(content); + VkVideoEncoderCompletionInfo info = {}; + info.pNext = &content; + s.Get()->GetCompletionInfo(&info); + Check(content.state == + VK_VIDEO_ENCODER_IMPORT_CONTENT_STATE_NOT_EVALUATED, + "a late measurement for a retired registration is dropped", + "state " + U64((uint64_t)content.state)); + Check(content.probedRegistrationCount == 0, + "and does not move the session totals", + "probed " + U64(content.probedRegistrationCount)); +} + +} // namespace + +int main(int argc, char** argv) +{ + (void)argc; + (void)argv; + std::printf("Encoder-ext dma-buf import CONTENT probe\n"); + std::printf("----------------------------------------\n"); + + CasePredicateQuadrants(); + CasePredicateThresholdBoundary(); + CaseScorerOnSyntheticBuffers(); + CaseArmingAndApplicability(); + CaseArmedButNeverScoredIsCounted(); + CaseDetourBudgetIsOneFramePerRegistration(); + CaseOldestDamagedFirstAndDrain(); + + Session session; + if (!session.Open()) { + std::printf("RESULT: COULD-NOT-RUN (session setup failed)\n"); + return 2; + } + CaseRegistrationEchoArmsAndReports(session); + CaseBlockLinearDirectRegistrationArms(session); + CaseUncopyableRegistrationStaysNotApplicable(session); + CaseUnscorableFormatIsRefusedAtRegistration(session); + CaseChainGateWidenedNotRelaxed(session); + CaseCompletionChannel(session); + CaseNoOptInLeavesTheChannelSilent(); + CaseVerdictReachesTheCompletionChannel(); + CaseMeasurementForARetiredRegistrationIsDropped(); + + std::printf("----------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-import-guard/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-import-guard/CMakeLists.txt new file mode 100644 index 00000000..5137506e --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-import-guard/CMakeLists.txt @@ -0,0 +1,100 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_import_guard_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, not the shared one: the internal header's +# seam functions are deliberately not exported from libvkvideo-encoder.so, so +# only the archive can satisfy them. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + # The public encoder header reaches VkCodecUtils/VkVideoRefCountBase.h, + # which lives under the shared common-libs root. + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# CTest semantics, matching the sibling library tests: 0 means every assertion +# held, 1 means an assertion failed, and 2 means the session could not be stood +# up at all -- deliberately a FAILURE and not a skip, because a host that could +# not run this must not report the guard's reporting channel as verified. There +# is no GPU, driver or display dependence: the session has no device by +# construction. +# +# WHAT A GREEN HERE DOES NOT MEAN, stated at the gate rather than only in the +# source: this covers the CARRIER, not the workaround. The guard's COMPLETE, +# INCOMPLETE and DISABLED verdicts need a real dma-buf import on an NVIDIA +# device and are not reachable from any device-free session, so they are proven +# on hardware or not at all. +# +# WHAT IT DOES GATE ABOUT THE RETIREMENT, and it is not nothing: the library's +# guard count. The guard is retired (kVkEncImportOrdinalGuardCount = 0) because +# retaining imports was measured to move which imports the driver damages and +# not how many, and C1 pins that 0 device-free, in CI, on any host. A build +# that re-arms the guard turns this entry red and has to come and say why. +# Retiring it also cost this suite its old writer proof -- a zeroed build +# constant is indistinguishable from an untouched struct -- so every +# guard-info case now poisons the caller's struct first and asserts whether +# the poison was cleared or survived. +enable_testing() +add_test(NAME EncoderExtImportGuardReport + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtImportGuardReport PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_ext_import_guard_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-import-guard/src/main.cpp b/vk_video_encoder/test/encoder-ext-import-guard/src/main.cpp new file mode 100644 index 00000000..247e89bb --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-import-guard/src/main.cpp @@ -0,0 +1,576 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * The dma-buf import-ordinal guard's REPORTING CHANNEL, device-free half. + * + * WHAT THIS COVERS, AND -- FIRST -- WHAT IT DOES NOT. + * + * The guard is RETIRED BY DEFAULT: the library builds with + * kVkEncImportOrdinalGuardCount = 0, takes no sacrificial dma-buf import and + * reports DISABLED. Armed, it takes that many imports on an NVIDIA VkDevice. + * Nothing here can produce one either way: a dma-buf import needs a real + * exporter, a real driver and a real device, and no encoder-ext test in this + * tree reaches VkEncImportExternalImage with HANDLE_TYPE_DMA_BUF except the + * hardware sibling. So COMPLETE and INCOMPLETE -- the verdicts only an armed + * build can produce -- are NOT covered by this file, and neither is DISABLED + * on a real import. They are covered by a hardware run or not at all. Read + * the honest gap statement in the commit message before treating a green + * here as "the guard works"; on the shipped build the guard does nothing, + * deliberately, and that is what the hardware sibling asserts. + * + * WHAT IS COVERED is the carrier the guard's verdict now travels on, which + * is the part that was missing entirely and which is fully device-free: + * + * * VkVideoEncoderStatus::pNext accepts exactly one + * VkVideoEncoderImportGuardInfo, and the library WRITES it. Before this + * change a chained pStatus was refused outright, so a caller could not + * ask the question at all. + * * An unknown chained sType, and a REPEATED known one, are still refused + * -- the relaxation must not have turned into "ignore the chain". + * * A chain hanging off a MIS-STAMPED pStatus is refused AND is not + * written through. Writing into a struct whose gate rejected the + * request is the version-skew failure the gate exists to prevent. + * * VkVideoEncoderCompletionInfo::pNext accepts the same struct, and its + * own unknown-sType refusal is unchanged. + * + * WHY EVERY GUARD-INFO CASE POISONS ITS STRUCT FIRST. Every field's + * "nothing happened" value is 0, which is exactly what a caller's + * zero-initialised struct already holds -- so asserting 0 proves nothing + * about whether the library wrote anything. requestedCount escapes that only + * while it is non-zero: it is a BUILD constant stamped on every path, so a + * non-zero count read back proves the writer ran. This build's count is 0, so + * that proof is not available here, and the poison below is what supplies it. + * + * What replaces it does not depend on the constant at all, and is stronger: + * hand the library values it MUST overwrite (C1) or MUST leave alone + * (C2-C4), and assert which happened. A library that dropped the chain, or + * that wrote through a chain its own gate refused, leaves the poison + * standing or clears it -- and either way one of these cases goes red. C1 + * additionally pins the build count itself, so the retirement is a claim + * this suite makes rather than one it merely tolerates. + * + * WHY A NULL BACKEND. VkEncInstallNullBackend gives a session that reports + * initialized with no device, no worker threads and no VkVideoEncoder, on + * which a VK_IMAGE registration runs the real RegisterImageResource -- + * including the real pStatus gate and the real chain walk -- and stops + * before the one step that needs a device. No hardware dependence in either + * direction, which is what makes it deterministic. + */ + +#include "vulkan_video_encoder_ext_internal.h" +// C9 needs the CONCRETE VulkanDeviceContext, not the forward declaration the +// internal header is content with -- it constructs one and writes a dispatch +// member. Same include, same position, as the sibling ordinal-guard suite. +#include "VkCodecUtils/VulkanDeviceContext.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Scrub them before anything else sees them -- the +// same block, for the same reason, as the sibling library test TUs. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#if defined(__linux__) +#include +#include +#endif + +namespace { + +int g_failures = 0; +int g_checks = 0; +const char* g_currentCase = ""; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + return; + } + g_failures++; + std::printf(" FAIL [%s] %s : %s\n", g_currentCase, what, detail.c_str()); +} + +std::string U64(uint64_t v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%llu", (unsigned long long)v); + return buf; +} + +// A session with no device. VK_IMAGE registrations skip the device-touching +// view build; everything ahead of it -- the ownership echo, the structure +// gate, the chain walk, the guard verdict -- is the production code. +class Session { +public: + bool Open() + { + if ((CreateVulkanVideoEncoderExt(m_encoder) != VK_SUCCESS) || + !m_encoder) { + std::printf(" ERROR: CreateVulkanVideoEncoderExt failed\n"); + return false; + } + if (VkEncInstallNullBackend(m_encoder.get(), &m_backend) != + VK_SUCCESS) { + std::printf(" ERROR: VkEncInstallNullBackend failed\n"); + return false; + } + return true; + } + + VulkanVideoEncoderExt* Get() const { return m_encoder.get(); } + + // The descriptor every case registers. A VK_IMAGE slot with a sentinel + // handle: stored and compared, never dereferenced, because this session + // has no device to dereference it with. + static VkVideoEncoderExternalImageDescriptor Descriptor() + { + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = 640; + desc.height = 360; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.existingImage = (VkImage)(uintptr_t)0xA110C8ED; + return desc; + } + +private: + VkSharedBaseObj m_encoder; + VkEncNullBackendState m_backend{}; +}; + +VkVideoEncoderImportGuardInfo FreshGuardInfo() +{ + VkVideoEncoderImportGuardInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_IMPORT_GUARD_INFO; + return info; +} + +// The library's kVkEncImportOrdinalGuardCount, mirrored -- it lives in an +// anonymous namespace and cannot be read from here. 0 is the retired default: +// the guard takes no sacrificial import and reports DISABLED on the imports +// it would once have moved. Pinning it is deliberate. The count is a claim +// about a driver defect (it buys a phase shift of the damage pattern and not +// a repair, measured), so a build that changes it is changing that claim and +// has to come here and say so. The hardware sibling takes --guard-count=N for +// the same reason, and is where a non-zero count is actually exercised. +constexpr uint32_t kExpectedRequestedCount = 0; + +// Values no library path can produce, written into a caller's struct before +// the call so that "the library wrote this" and "the library did not" are +// distinguishable at all. See WHY EVERY GUARD-INFO CASE POISONS ITS STRUCT +// FIRST at the top of this file: with the guard retired, every honest field +// value on every path this suite can reach is 0, which is also what an +// untouched struct holds. +void PoisonGuardInfo(VkVideoEncoderImportGuardInfo& info) +{ + info.state = VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE; + info.requestedCount = 0xBADC0DEu; + info.retainedCount = 99u; + info.failureStatus = VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED; + info.failureErrno = -4242; +} + +VkVideoEncoderStatus FreshStatus() +{ + VkVideoEncoderStatus status = {}; + status.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; + return status; +} + +// --------------------------------------------------------------------------- +// C1. The channel exists at all: a chained guard-info is accepted and FILLED. +// +// This is the case the whole change is for. Before it, the same call answered +// STRUCTURE_TYPE_UNKNOWN -- a caller with silenceStdio set had no way, at +// all, to learn what the workaround had done. +// --------------------------------------------------------------------------- +void CaseChainedGuardInfoIsAcceptedAndWritten(Session& s) +{ + g_currentCase = "chained guard-info is accepted and written"; + + VkVideoEncoderImportGuardInfo guardInfo = FreshGuardInfo(); + PoisonGuardInfo(guardInfo); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &guardInfo; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "a chained VkVideoEncoderImportGuardInfo is ACCEPTED", + "status " + U64((uint64_t)reg)); + Check(resource != VK_VIDEO_ENCODER_RESOURCE_NULL, + "the registration still produced an id", "id was NULL"); + Check(status.handlesConsumed == VK_FALSE, + "the ownership echo still answers for a VK_IMAGE registration", + "handlesConsumed was VK_TRUE"); + +#if defined(__linux__) + // THE BUILD-COUNT PIN. requestedCount is stamped by the library on every + // path, and it is the guard's build constant. The library ships 0 -- the + // guard is retired, because retaining imports was measured to move which + // imports the driver damages and not how many. A build that re-arms it + // fails here, which is the point: the count is a claim about a driver + // defect and it does not get to change silently. This is ALSO half of the + // writer gate, since the struct was poisoned with 0xBADC0DE and only the + // library can have put a 0 there. + Check(guardInfo.requestedCount == kExpectedRequestedCount, + "requestedCount is this build's guard count (retired default 0)", + "requestedCount was " + U64(guardInfo.requestedCount) + + ", expected " + U64(kExpectedRequestedCount)); +#endif + // THE REST OF THE WRITER GATE. Each of these fields was poisoned above + // with a value no library path produces, so every one of them is now a + // proof that the library WROTE the caller's struct -- not merely that + // the struct still holds zeroes. Before the guard was retired that job + // belonged to requestedCount alone, and a zeroed build constant can no + // longer do it. + // + // The values themselves are the honest answer for a VK_IMAGE + // registration: it performs no import, so the guard never ran for it, + // which is a distinct answer from "it ran and did nothing" (DISABLED). + Check(guardInfo.state == VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED, + "a VK_IMAGE registration reports NOT_EVALUATED (poison cleared)", + "state " + U64((uint64_t)guardInfo.state)); + Check(guardInfo.retainedCount == 0, + "no sacrificial import is claimed on a registration that did none", + "retainedCount " + U64(guardInfo.retainedCount)); + Check(guardInfo.failureStatus == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "no failure is claimed where nothing failed", + "failureStatus " + U64((uint64_t)guardInfo.failureStatus)); + Check(guardInfo.failureErrno == 0, + "no errno is claimed where nothing failed", + "failureErrno " + U64((uint64_t)(int64_t)guardInfo.failureErrno)); +} + +// --------------------------------------------------------------------------- +// C2. The relaxation did not become "ignore the chain". +// --------------------------------------------------------------------------- +void CaseUnknownChainedSTypeIsRefused(Session& s) +{ + g_currentCase = "unknown chained sType is refused"; + + VkVideoEncoderImportGuardInfo alien = FreshGuardInfo(); + alien.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FILTER_INFO; // not ours + // Poisoned so the "not written through" assertion below can fail. At the + // retired guard count the library's own written value for requestedCount + // is 0, so asserting 0 on a refusal path stopped discriminating. + PoisonGuardInfo(alien); + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &alien; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "an extension this build does not understand is REFUSED", + "status " + U64((uint64_t)reg)); + Check(resource == VK_VIDEO_ENCODER_RESOURCE_NULL, + "a refused registration mints no id", + "id " + U64((uint64_t)resource)); + Check((alien.requestedCount == 0xBADC0DEu) && + (alien.retainedCount == 99u) && + (alien.state == + VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE), + "a refused link is not written through (the poison SURVIVES)", + "requestedCount " + U64(alien.requestedCount) + ", retainedCount " + + U64(alien.retainedCount) + ", state " + + U64((uint64_t)alien.state)); +} + +// --------------------------------------------------------------------------- +// C3. Two links of one type mean the caller believes it is getting two +// different things. Refused, as everywhere else in this file's chain walks. +// --------------------------------------------------------------------------- +void CaseRepeatedGuardInfoLinkIsRefused(Session& s) +{ + g_currentCase = "repeated guard-info link is refused"; + + VkVideoEncoderImportGuardInfo second = FreshGuardInfo(); + VkVideoEncoderImportGuardInfo first = FreshGuardInfo(); + // Poisoned for the same reason as C2: 0 is what the library itself would + // write for these fields now, so only a value it cannot write can prove + // it wrote nothing. + PoisonGuardInfo(first); + PoisonGuardInfo(second); + first.pNext = &second; + VkVideoEncoderStatus status = FreshStatus(); + status.pNext = &first; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "a repeated known sType is REFUSED", + "status " + U64((uint64_t)reg)); + Check((first.requestedCount == 0xBADC0DEu) && + (second.requestedCount == 0xBADC0DEu), + "neither link of a refused chain is written through (poison " + "SURVIVES in both)", + "first " + U64(first.requestedCount) + ", second " + + U64(second.requestedCount)); +} + +// --------------------------------------------------------------------------- +// C4. A chain behind a MIS-STAMPED pStatus. The status gate refuses first, +// and nothing may be written through either struct: a consumer built against +// a different header is exactly who owns that memory. +// --------------------------------------------------------------------------- +void CaseMisStampedStatusRefusesAndWritesNothing(Session& s) +{ + g_currentCase = "mis-stamped pStatus refuses and writes nothing"; + + VkVideoEncoderImportGuardInfo guardInfo = FreshGuardInfo(); + // Poisoned, or this case cannot fail: the library's own written value for + // requestedCount is 0 on every path this build takes, so "still 0" + // does not separate "not written" from "written". + PoisonGuardInfo(guardInfo); + VkVideoEncoderStatus status = FreshStatus(); + status.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_UNDEFINED; + status.pNext = &guardInfo; + status.handlesConsumed = VK_TRUE; // a sentinel the library must not clear + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "a mis-stamped pStatus is still version skew", + "status " + U64((uint64_t)reg)); + Check(status.handlesConsumed == VK_TRUE, + "a mis-stamped pStatus is not written through", + "handlesConsumed was cleared"); + Check((guardInfo.requestedCount == 0xBADC0DEu) && + (guardInfo.retainedCount == 99u), + "a chain behind a mis-stamped pStatus is not written through (the " + "poison SURVIVES)", + "requestedCount " + U64(guardInfo.requestedCount) + + ", retainedCount " + U64(guardInfo.retainedCount)); +} + +// --------------------------------------------------------------------------- +// C5. The pre-existing shape -- pStatus with no chain -- is untouched. This +// is the ABI-compatibility case: every caller built against the header that +// had nothing to chain here passes exactly this. +// --------------------------------------------------------------------------- +void CaseUnchainedStatusStillWorks(Session& s) +{ + g_currentCase = "unchained pStatus is unaffected"; + + VkVideoEncoderStatus status = FreshStatus(); + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode reg = s.Get()->RegisterImageResource( + Session::Descriptor(), 0, &resource, &status); + + Check(reg == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "an unchained pStatus still registers", + "status " + U64((uint64_t)reg)); + Check(status.handlesConsumed == VK_FALSE, + "the ownership echo is unchanged", + "handlesConsumed was VK_TRUE"); +} + +// --------------------------------------------------------------------------- +// C6. The second read site: GetCompletionInfo. This is the surface a consumer +// already calls, so it is the one that can report the verdict without the +// consumer changing its registration code at all. +// --------------------------------------------------------------------------- +void CaseCompletionInfoChainAcceptsGuardInfo(Session& s) +{ + g_currentCase = "GetCompletionInfo accepts the guard-info link"; + + VkVideoEncoderImportGuardInfo guardInfo = FreshGuardInfo(); + guardInfo.state = VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_COMPLETE; // sentinel + guardInfo.retainedCount = 99; // sentinel + VkVideoEncoderCompletionInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + info.pNext = &guardInfo; + + const VkResult r = s.Get()->GetCompletionInfo(&info); + Check(r == VK_SUCCESS, + "the snapshot call accepts a chained guard-info", + "VkResult " + U64((uint64_t)r)); + // The sentinels must have been OVERWRITTEN. No registration on this + // session was ever evaluated by the guard, so the honest answer is + // NOT_EVALUATED / 0 -- and a library that ignored the link would leave + // COMPLETE / 99 standing, which is the exact shape of the over-claim + // this channel exists to make impossible. + Check(guardInfo.state == VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED, + "the snapshot overwrote the caller's sentinel state", + "state " + U64((uint64_t)guardInfo.state)); + Check(guardInfo.retainedCount == 0, + "the snapshot overwrote the caller's sentinel count", + "retainedCount " + U64(guardInfo.retainedCount)); +} + +void CaseCompletionInfoStillRefusesUnknownSType(Session& s) +{ + g_currentCase = "GetCompletionInfo still refuses an unknown sType"; + + VkVideoEncoderImportGuardInfo alien = FreshGuardInfo(); + alien.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; // not chainable here + VkVideoEncoderCompletionInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + info.pNext = &alien; + + const VkResult r = s.Get()->GetCompletionInfo(&info); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + "an unknown chained sType is still refused, not ignored", + "VkResult " + U64((uint64_t)r)); +} + +void CaseCompletionInfoRefusesRepeatedGuardInfo(Session& s) +{ + g_currentCase = "GetCompletionInfo refuses a repeated guard-info"; + + VkVideoEncoderImportGuardInfo second = FreshGuardInfo(); + VkVideoEncoderImportGuardInfo first = FreshGuardInfo(); + first.pNext = &second; + VkVideoEncoderCompletionInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + info.pNext = &first; + + const VkResult r = s.Get()->GetCompletionInfo(&info); + Check(r == VK_ERROR_INITIALIZATION_FAILED, + "two guard-info links are refused", + "VkResult " + U64((uint64_t)r)); +} + +} // namespace + +#if defined(__linux__) +// --------------------------------------------------------------------------- +// C9. The guard must not require dispatch the import never needed. +// +// REGRESSION, and it shipped: the guard calls VkEncIsNvidiaDevice on the way +// in to EVERY dma-buf import, and that probe called +// vkDevCtx.GetPhysicalDeviceProperties with no null check. Until the guard +// existed VkEncImportExternalImage touched no INSTANCE-level dispatch at all, +// so a consumer that populated only the device-level entries the import +// actually uses was fine -- and then was not. It segfaulted all ten of +// Chromium's VulkanVideoEncoderImportOwnershipTest cases (SEGV_MAPERR, ip=0), +// which is precisely that shape: a bare VulkanDeviceContext with device-level +// stubs and no instance table. The library's own CI could not see it, because +// nothing here had ever called the import with a partly-populated context. +// +// So this case builds the minimal such context -- ONE dispatch entry, a +// CreateImage that refuses -- and asserts the import RETURNS. The vendor probe +// must answer "not NVIDIA" when it cannot tell, which skips the guard and +// leaves the import behaving exactly as it did before the guard existed. +// +// MUTATION: drop either clause of the null check in VkEncIsNvidiaDevice and +// this case does not fail, it CRASHES -- ctest reports the whole binary dead, +// which is a louder red than a FAIL line and is the correct one here. +VkResult VKAPI_PTR RefusingCreateImage(VkDevice, + const VkImageCreateInfo*, + const VkAllocationCallbacks*, + VkImage*) +{ + return VK_ERROR_OUT_OF_HOST_MEMORY; +} + +void CaseImportSurvivesAContextWithNoInstanceDispatch() +{ + g_currentCase = "import survives a context with no instance dispatch"; + + // Everything null except the one entry the import reaches after the + // guard declines. This is the point: it is SUPPOSED to be under-populated. + VulkanDeviceContext ctx{}; + ctx.CreateImage = &RefusingCreateImage; + + int fds[2] = {-1, -1}; + if (pipe(fds) != 0) { + Check(false, "pipe(2) for the import fd", "pipe failed"); + return; + } + close(fds[1]); + const int fd = fds[0]; + + VkVideoEncoderExternalImageDescriptor desc{}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = 64; + desc.height = 64; + desc.tiling = VK_IMAGE_TILING_LINEAR; + desc.imageUsage = VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; + desc.allocationSize = 8192; + desc.memoryTypeBits = 0; + desc.memoryTypeIndex = UINT32_MAX; + + VkEncImportedImage imported{}; + const VkVideoEncoderStatusCode status = + VkEncImportExternalImage(ctx, desc, static_cast(fd), + &imported); + + // Reaching this line at all is the assertion that matters; the rest + // pins the behaviour so the crash cannot be "fixed" by changing it. + Check(status == VK_VIDEO_ENCODER_STATUS_ERROR_IMPORT_FAILED, + "a dma-buf import on an instance-less context returns, not crashes", + "status " + U64((uint64_t)status)); + Check(imported.result == VK_ENC_IMPORT_FAILED_BEFORE_ALLOCATE, + "the refusal is classified as pre-handoff", + "result " + U64((uint64_t)imported.result)); + // Design section 2.3: nothing reached the driver, so the LIBRARY owns the + // close. fcntl on a closed fd is the only way to ask without racing. + Check(fcntl(fd, F_GETFD) == -1, + "the library closed the fd it never handed over", + "fd " + U64((uint64_t)fd) + " is still open"); +} +#endif // __linux__ + +int main(int argc, char** argv) +{ + (void)argc; + (void)argv; + std::printf("Encoder-ext import-ordinal guard reporting channel\n"); + std::printf("--------------------------------------------------\n"); + + Session session; + if (!session.Open()) { + std::printf("RESULT: COULD-NOT-RUN (session setup failed)\n"); + return 2; + } + + CaseChainedGuardInfoIsAcceptedAndWritten(session); + CaseUnknownChainedSTypeIsRefused(session); + CaseRepeatedGuardInfoLinkIsRefused(session); + CaseMisStampedStatusRefusesAndWritesNothing(session); + CaseUnchainedStatusStillWorks(session); + CaseCompletionInfoChainAcceptsGuardInfo(session); + CaseCompletionInfoStillRefusesUnknownSType(session); + CaseCompletionInfoRefusesRepeatedGuardInfo(session); +#if defined(__linux__) + CaseImportSurvivesAContextWithNoInstanceDispatch(); +#endif + + std::printf("--------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-import-ordinal-guard/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-import-ordinal-guard/CMakeLists.txt new file mode 100644 index 00000000..7e970f94 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-import-ordinal-guard/CMakeLists.txt @@ -0,0 +1,266 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_import_ordinal_guard_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The STATIC archive, for the same reason encoder-ext-format-encode links it: +# the subject here is VkEncImportExternalImage, a free function declared in +# the INTERNAL header and deliberately not exported from +# libvkvideo-encoder.so. The archive also carries VulkanDeviceContext, which +# this test builds one of -- no encode session is created, because the guard +# is a property of the import and needs no encoder, no codec and no encode +# queue to exercise. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +# Must branch the same way the library did. Nothing in this test touches the +# filter, but the internal header is shared with the library's own TUs and a +# test compiled with a different view of it is a layout skew waiting to be +# blamed on something else. +if(BUILD_ENCODER_COMPUTE_FILTER) + target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_VIDEO_SAMPLES_COMPUTE_FILTER_SUPPORTED) +endif() + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +enable_testing() + +# --------------------------------------------------------------------------- +# THE SUBJECT, and what it asserts follows one library constant. +# +# THE GUARD IS RETIRED: kVkEncImportOrdinalGuardCount is 0, so this entry +# asserts it is INERT. A caller's first dma-buf import on a VkDevice must be +# the ONLY image created, nothing may be retained, the release must find +# nothing, and the verdict must be DISABLED. That is a gate and not a +# formality -- a build that re-arms the guard without telling this suite fails +# it in four independent places (image count, image identity, report, release). +# +# WHY IT IS DISABLED BY DEFAULT. The guard retains K sacrificial dma-buf +# imports so that caller imports land K live-positions later. Over enough +# import ordinals to cover several periods of the damage pattern, that changes +# WHICH imports the driver damages and never HOW MANY -- the damage rate is +# the same for every K. A short run that registers two or three buffers +# samples one period at one phase and reads the shift as a fix. There is also +# no consumer-side repair to fall back on: a write through a damaged import is +# swallowed, and an immediate re-import at the next ordinal does not rescue +# it. The defect is in the driver's dma-buf import path. +# +# THE ARMED BEHAVIOUR HAS NOT LOST ITS COVERAGE. The same binary takes +# --guard-count=N: +# +# ./encoder_ext_import_ordinal_guard_test --guard-count=2 +# +# against a library rebuilt with kVkEncImportOrdinalGuardCount = 2 runs every +# positional, retention and release assertion below exactly as it did when 2 +# was the default. It is deliberately NOT registered as a ctest entry: the +# count is a build constant, so such an entry would be red against the library +# this tree actually builds. A knob whose test cannot run is not a tested +# knob, and this is how it stays testable in one command. +# +# WHAT MADE THIS SUITE NECESSARY. The workaround it covers shipped with a +# coverage floor of exactly zero: no test, CMake entry or CI file referenced +# it, and -- the reason a passing suite could not have caught that -- +# VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF reached +# VkEncImportExternalImage from NO encoder-ext test at all. Every registration +# under vk_video_encoder/test was VK_IMAGE or OPAQUE_FD, on both of which +# VkEncEnsureImportOrdinalGuard returns early. The green suite was green with +# the guard deleted. It is also, now, what makes the retirement checkable: +# "the guard does nothing" is only a claim worth anything if something can +# tell the difference between that and "the guard is gone". +# +# WHAT IT ASSERTS, read off the DEVICE and not off a log line or a +# self-report: counting thunks are swapped into this test's own +# VulkanDeviceContext dispatch table, so the real driver entry points run and +# the library's calls are counted and ORDERED. With K the expected count (0 by +# default, --guard-count=N otherwise): +# +# * K+1 vkCreateImage and K+1 fd-importing vkAllocateMemory for ONE caller +# import, and the caller's VkImage is the one created last. At K = 0 that +# is "exactly one image, and it is the caller's" -- the inertness claim. +# At K > 0 it is the positional claim, "caller imports now start at +# live-position K+1", stated in the only terms the driver understands. +# * 0 vkDestroyImage and 0 vkFreeMemory before the release. At K > 0 this is +# what separates the mechanism from a decoration: the original cross-tab +# has a row for "2, created then DESTROYED" and it is 747/747 BAD, so a +# guard that freed its pair would satisfy every count above while not even +# delivering the phase shift it claims. +# * a SECOND caller import costs exactly one more image, not K+1 -- the +# documented "no-op after the first one, no per-frame cost". +# * and the guard comes DOWN: VkEncReleaseImportOrdinalGuard returns K AND K +# vkDestroyImage / K vkFreeMemory actually reach the driver, and a second +# call returns 0 and destroys nothing. That function had no caller on any +# test path outside this suite -- its production call site is inside a full +# encode session's Deinitialize(), and no session in the corpus performs a +# dma-buf import -- so the vkDestroyDevice-with-live-children leak it +# closes could have come back with CI green. At K = 0 it must return 0 and +# destroy nothing, which is the same assertion doing the retired build's +# half of the job. +# +# AND IT CHECKS THE LIBRARY'S OWN VERDICT AGAINST THOSE COUNTS. +# VkEncImportOrdinalGuardReport must read DISABLED with retainedCount 0 at +# K = 0, COMPLETE with retainedCount == K above it -- and retainedCount must +# EQUAL (images created - the caller's one) either way. The DISABLED +# expectation is load-bearing rather than cosmetic: 0 == 0 satisfies the +# arithmetic behind COMPLETE, whose public definition is that caller imports +# landed past a position they did not, at 0, move to. A library that reports a +# working workaround on a build which removed it fails here. +# +# WHAT IT DOES NOT ASSERT, stated so a green run is not over-read: nothing +# about the pixels, and nothing about the driver. It does not measure whether +# imports are still damaged. They are -- that is why K is 0 -- and this suite +# would be green either way. It gates the MECHANISM and the build's claim +# about it, which is all a 900-second CI binary can honestly gate. +# +# SKIP_RETURN_CODE 77, like every sibling, and the skip is narrow: no Vulkan +# device, no dma-buf export, or a non-NVIDIA GPU -- where the guard refuses to +# run BY DESIGN, so there is nothing to prove and PASSED would be a lie. A +# failed assertion is 1 and can never be laundered into a skip: main() returns +# 1 on g_failures before it considers anything else. +# --------------------------------------------------------------------------- +add_test(NAME EncoderExtImportOrdinalGuardIsInertWhenRetired + COMMAND ${PROJECT_NAME} --guard-count=0) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +# --guard-count=0 is passed explicitly although it is the default, so the +# expectation this entry encodes is readable at the gate and not only in the +# binary. +set_tests_properties(EncoderExtImportOrdinalGuardIsInertWhenRetired + PROPERTIES + LABELS "gpu" + TIMEOUT 900 + SKIP_RETURN_CODE 77) + +# --------------------------------------------------------------------------- +# THE KILL SWITCH, which had no coverage either -- and this entry is also the +# PERMANENT form of this suite's mutation proof, WITH ONE CAVEAT NOW: against +# the retired default it and the subject assert the same observable, because a +# build count of 0 and the kill switch produce the same inert shape. It +# regains its discriminating power the moment the library is built with a +# non-zero count, which is exactly when the guard is doing something worth +# switching off. It is kept registered because the kill switch is a shipped, +# documented control a support engineer will be told to set, and "setting it +# still stops the workaround" is a claim with no other test. +# +# The mutation that proves the subject above can go red is to disable the +# guard, and the cheapest lever for that is the documented escape hatch +# VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD=1. Against a library built with +# guard count 2, the four arms answer as follows; run --guard-count=2 against +# such a build to reproduce it: +# +# arm environment exit checks +# ------------------------------------------------------------------------ +# (subject) - 0 0 of 21 failed +# (subject) KILL=1 1 8 of 21 failed +# --expect-disabled KILL=1 0 0 of 20 failed +# --expect-disabled - 1 7 of 20 failed +# +# The guard-off failures name the property that broke rather than reporting a +# bare non-zero exit: +# FAIL vkCreateImage call count ... expected 3, got 1 +# FAIL fd-importing vkAllocateMemory count ... expected 3, got 1 +# FAIL the caller's VkImage is the one created LAST, at live-position 3 +# FAIL the guard report's state is COMPLETE : got DISABLED +# +# AND A SOURCE-LEVEL MUTATION, because the env var only proves sensitivity to +# the env var. Deleting the VkEncEnsureImportOrdinalGuard() call from the top +# of VkEncImportExternalImage and rebuilding turns BOTH entries red -- +# `ctest -R EncoderExtImportOrdinalGuard` exits 8, 0 of 2 passed, with the +# report reading NOT_EVALUATED on both. Restoring the file (md5 back to the +# original, binary md5 changed, so the restore really did rebuild) returns +# 2 of 2 passed, exit 0. +# +# The RELEASE half is mutation-proven separately, because a return value is +# not a destroy: neutering only the destroy loop inside +# VkEncReleaseImportOrdinalGuard -- so it still returns 2 while calling +# nothing -- turns the subject red with +# FAIL the release destroyed exactly the guard's images and memories : +# vkDestroyImage 0 -> 0, vkFreeMemory 0 -> 0, expected +2 each +# and nothing else, which is the assertion doing exactly its own job. +# Restoring (source md5 981b738a2656c7a61bda426608ace93f, test-binary md5 +# 8be4cfb7... -> 62489f58..., so the rebuild happened) returns 2 of 2 passed. +# +# Note what the second row of that table means for THIS entry on its own: with +# the guard's call site deleted, "one import and it is the first" is trivially +# true, so the kill-switch arm's COUNT assertions cannot catch a deleted guard. +# Its report assertion can and does (DISABLED vs NOT_EVALUATED), and that is +# now the load-bearing half of BOTH entries: at the retired count the guard is +# supposed to create nothing, so only the verdict separates "retired" from +# "deleted". Deleting the VkEncEnsureImportOrdinalGuard() call from the top of +# VkEncImportExternalImage still turns both entries red, on +# NOT_EVALUATED-instead-of-DISABLED rather than on a count. +# +# This entry sets the variable through the test's ENVIRONMENT property and +# asserts the inert shape -- one image, and it is the first created. Against +# an ARMED library that is the OPPOSITE of the subject's assertions; run this +# binary with --guard-count=2 --expect-disabled and NO variable set, against +# such a library, and it fails, which is what stops the arm from being green +# for the wrong reason. Against the retired default there is no opposite left +# to assert, and the honest statement of what this entry then covers is: the +# kill-switch code path is still reached and still reports DISABLED. +# +# The variable is read once into a function-local static, so guard-on and +# guard-off cannot be exercised in one process -- hence two ctest entries over +# one binary rather than two arms in one run. +# --------------------------------------------------------------------------- +add_test(NAME EncoderExtImportOrdinalGuardKillSwitchDisablesIt + COMMAND ${PROJECT_NAME} --expect-disabled) +set_tests_properties(EncoderExtImportOrdinalGuardKillSwitchDisablesIt + PROPERTIES + LABELS "gpu" + TIMEOUT 900 + SKIP_RETURN_CODE 77 + ENVIRONMENT "VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD=1") + +message(STATUS "encoder_ext_import_ordinal_guard_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-import-ordinal-guard/src/main.cpp b/vk_video_encoder/test/encoder-ext-import-ordinal-guard/src/main.cpp new file mode 100644 index 00000000..1d03200d --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-import-ordinal-guard/src/main.cpp @@ -0,0 +1,1266 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +//============================================================================= +// The dma-buf IMPORT-ORDINAL GUARD, under test. +// +// WHAT IT IS NOW. The guard (VkEncEnsureImportOrdinalGuard, at the top of +// VkEncImportExternalImage) is RETIRED: the library builds with +// kVkEncImportOrdinalGuardCount = 0, so it takes no sacrificial import and +// reports DISABLED -- retaining K imports shifts the PHASE of the driver's +// damage pattern and nothing else, the damage rate being the same for every +// K. The mechanism is kept, so this suite keeps testing it, in both +// directions: +// * at the shipped count it asserts the guard is INERT -- one image for one +// caller import, nothing retained, nothing released, verdict DISABLED; +// * with --guard-count=N, against a library rebuilt with that N, it asserts +// the full positional behaviour exactly as it did when N was 2. +// +// WHY THIS SUITE EXISTS: without it nothing in the tree can regress the +// guard. No test, CMake +// file or CI entry referenced it, and -- the load-bearing half -- +// VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF did not reach +// VkEncImportExternalImage from ANY encoder-ext test. Every registration +// under vk_video_encoder/test was VK_IMAGE or OPAQUE_FD, and the guard +// returns early on both. So the workaround shipped with a coverage floor of +// zero, on a code path whose failure mode is green frames. +// +// WHAT THIS BINARY DOES. It exports a REAL dma-buf from a real NVIDIA VkDevice +// (NV12, a DRM format modifier the driver advertises as EXPORTABLE and +// IMPORTABLE, the same shape the sibling linux_dmabuf_import test measures) +// and hands it to VkEncImportExternalImage as HANDLE_TYPE_DMA_BUF. That is the +// first time that combination is driven anywhere in this tree. +// +// WHAT IT ASSERTS, AND WHY THAT AND NOT SOMETHING ELSE. +// +// The guard's claim is not "an import succeeded" -- an import succeeds with +// the guard deleted. Its claim is POSITIONAL, it is stated in terms of the +// build's count K, and it has four parts: +// +// 1. K sacrificial imports are taken BEFORE the caller's, so the caller's +// VkImage is the (K+1)-th image created on the device by this path. At +// the shipped K = 0 that same assertion is the INERTNESS claim: the +// caller's import is the first and only image on the device. +// 2. They are RETAINED, not created and thrown away. The original +// measurement table has a row for "2, created then DESTROYED" and it is +// 100% bad, so a guard that freed them would pass any "did it import" +// test while not even delivering the phase shift it claims. +// 3. It is a no-op after the first import: the second and every later +// caller import costs exactly one image, not K+1. +// 4. And it comes DOWN with the device: VkEncReleaseImportOrdinalGuard +// really destroys K images and frees K allocations, and a second call +// destroys nothing. That function has no caller on any other test +// path -- its production call site is inside a full encode +// session's Deinitialize(), and no session in the corpus performs a +// dma-buf import -- so the leak it fixes could return with CI green. +// +// All three are read off the DEVICE, not off a log line and not off a +// self-report: the test swaps counting thunks into its own +// VulkanDeviceContext's dispatch table (vk::VkInterfaceFunctions is a plain +// struct of function pointers) and counts the vkCreateImage, +// import-chained vkAllocateMemory, vkDestroyImage and vkFreeMemory calls the +// library actually makes, and in what order. The identity assertion -- +// "the caller's VkImage is the one created third" -- is the positional claim +// stated in the only terms the driver understands. +// +// AND THE REPORT IS CHECKED AGAINST THEM, not instead of them. A report is a +// claim about behaviour; the counters are the behaviour. A guard that +// reported COMPLETE while taking one import would pass a report-based test +// and fail this one -- so this file asserts the library's own +// VkEncImportOrdinalGuardReport AND that its retainedCount agrees with what +// the driver was actually asked to do. The sibling encoder-ext-import-guard +// suite gates that report's CARRIER device-free and says in as many words +// that the COMPLETE and INCOMPLETE verdicts "are covered only by a hardware +// run". This is that run. +// +// WHY NOT THE PIXELS. The defect's symptom is a dead plane on one registered +// buffer, and reproducing THAT needs a producer feeding a real encode session +// for hundreds of frames per registration, swept over enough import ordinals +// to cover several periods of the damage pattern. It is not something a CI +// binary can do in 900 seconds, and a per-run sample of it would be a +// coin-flip gate. This binary asserts the MECHANISM, never the pixels. +// Stated plainly so a green run is not over-read: it means the guard does +// exactly what its build count says -- nothing at all, at the shipped 0 -- +// and says nothing whatever about whether the driver in front of it still +// damages imports. It does. That is why the count is 0. +// +// EXIT CODES, and the skip is deliberately narrow: +// 0 every assertion held. +// 1 an assertion failed -- including every library refusal once the host +// question has been settled. g_failures is checked BEFORE any skip, so +// a skip cannot launder a failure. +// 77 the host cannot run this at all: no Vulkan device, no dma-buf export, +// or a non-NVIDIA GPU. The guard is NVIDIA-gated by design, so on any +// other vendor there is nothing to prove and reporting PASSED would be +// a lie. CTest reports these SKIPPED. +// +// ARMS +// (default) the library's shipped count, which is 0: asserts the +// guard is INERT -- ONE image for the caller's import, +// nothing retained, nothing released, verdict DISABLED. +// --guard-count=N the same assertions against a library rebuilt with +// kVkEncImportOrdinalGuardCount = N. This is what keeps +// the ARMED behaviour covered while the shipped build +// does not use it: with N > 0 the positional, retention +// and release claims above are all live again. +// --expect-disabled VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD is set in the +// environment. Asserts the kill switch forces the inert +// shape whatever the build count is. It is that +// control's only coverage, and the permanent form of +// this suite's mutation proof -- but only against a +// build with a non-zero count, since at the shipped 0 +// both arms assert the same observable. See the +// CMakeLists. +//============================================================================= + +#if !defined(__linux__) + +#include +int main() +{ + std::printf("SKIP: a dma-buf is a Linux kernel object; there is no " + "import-ordinal guard to test on this platform (it is not " + "even compiled).\n"); + return 77; +} + +#else // __linux__ + +#include "vulkan_video_encoder_ext.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include "vulkan_video_encoder_ext_internal.h" +#include "VkCodecUtils/VulkanDeviceContext.h" + +#undef Status +#undef None +#undef Bool +#undef Window + +#include + +#include +#include +#include +#include +#include +#include + +namespace { + +//============================================================================= +// Verdict plumbing. Identical in shape to the sibling encoder-ext suites. +//============================================================================= + +int g_failures = 0; +int g_checks = 0; +bool g_verbose = false; + +void Check(bool ok, const std::string& what, + const std::string& detail = std::string()) +{ + g_checks++; + if (ok) { + std::printf(" ok %s\n", what.c_str()); + return; + } + g_failures++; + std::printf(" FAIL %s : %s\n", what.c_str(), detail.c_str()); +} + +std::string U32(uint32_t v) +{ + char b[32]; + std::snprintf(b, sizeof(b), "%u", v); + return b; +} + +std::string Hex64(uint64_t v) +{ + char b[32]; + std::snprintf(b, sizeof(b), "0x%llx", (unsigned long long)v); + return b; +} + +std::string Ptr(const void* p) +{ + char b[32]; + std::snprintf(b, sizeof(b), "%p", p); + return b; +} + +const char* GuardStateName(VkVideoEncoderImportGuardState s) +{ + switch (s) { + case VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_EVALUATED: + return "NOT_EVALUATED"; + case VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_APPLICABLE: + return "NOT_APPLICABLE"; + case VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_NOT_NVIDIA: + return "NOT_NVIDIA"; + case VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_DISABLED: + return "DISABLED"; + case VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_COMPLETE: + return "COMPLETE"; + case VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_INCOMPLETE: + return "INCOMPLETE"; + default: + return "(unknown)"; + } +} + +//============================================================================= +// THE OBSERVABLE: counting thunks over the device dispatch table. +// +// VulkanDeviceContext derives from vk::VkInterfaceFunctions, which is a plain +// struct of PFN_vk* members, so a test that OWNS the context can put its own +// function in front of the driver's and forward. Nothing about the library is +// recompiled or stubbed -- the real driver entry point runs, and the only +// thing added is a counter and an ordered record of the handles it minted. +// +// The export half of this test runs BEFORE these are installed, so nothing +// the harness itself allocates is ever counted. +//============================================================================= + +struct DispatchCounters { + uint32_t createImage = 0; + uint32_t importAlloc = 0; // vkAllocateMemory carrying an fd import chain + uint32_t plainAlloc = 0; // any other vkAllocateMemory + uint32_t destroyImage = 0; + uint32_t freeMemory = 0; + // In creation order. images[0] is the first VkImage the library created + // after the thunks went in -- which, with the guard on, is a guard image + // and NOT the caller's. + std::vector images; + std::vector imports; +}; + +DispatchCounters g_c; + +PFN_vkCreateImage g_realCreateImage = nullptr; +PFN_vkAllocateMemory g_realAllocateMemory = nullptr; +PFN_vkDestroyImage g_realDestroyImage = nullptr; +PFN_vkFreeMemory g_realFreeMemory = nullptr; + +bool ChainHasFdImport(const void* pNext) +{ + for (const VkBaseInStructure* p = (const VkBaseInStructure*)pNext; + p != nullptr; p = p->pNext) { + if (p->sType == VK_STRUCTURE_TYPE_IMPORT_MEMORY_FD_INFO_KHR) { + return true; + } + } + return false; +} + +VKAPI_ATTR VkResult VKAPI_CALL CountingCreateImage( + VkDevice device, const VkImageCreateInfo* pCreateInfo, + const VkAllocationCallbacks* pAllocator, VkImage* pImage) +{ + const VkResult r = g_realCreateImage(device, pCreateInfo, pAllocator, pImage); + if (r == VK_SUCCESS) { + g_c.createImage++; + g_c.images.push_back(*pImage); + if (g_verbose) { + std::printf(" [dispatch] vkCreateImage #%u -> %p\n", + g_c.createImage, (void*)(uintptr_t)*pImage); + } + } + return r; +} + +VKAPI_ATTR VkResult VKAPI_CALL CountingAllocateMemory( + VkDevice device, const VkMemoryAllocateInfo* pAllocateInfo, + const VkAllocationCallbacks* pAllocator, VkDeviceMemory* pMemory) +{ + const bool isImport = ChainHasFdImport(pAllocateInfo->pNext); + const VkResult r = + g_realAllocateMemory(device, pAllocateInfo, pAllocator, pMemory); + if (r == VK_SUCCESS) { + if (isImport) { + g_c.importAlloc++; + g_c.imports.push_back(*pMemory); + } else { + g_c.plainAlloc++; + } + if (g_verbose) { + std::printf(" [dispatch] vkAllocateMemory %s #%u\n", + isImport ? "(fd import)" : "(plain) ", + isImport ? g_c.importAlloc : g_c.plainAlloc); + } + } + return r; +} + +VKAPI_ATTR void VKAPI_CALL CountingDestroyImage( + VkDevice device, VkImage image, const VkAllocationCallbacks* pAllocator) +{ + if (image != VK_NULL_HANDLE) { + g_c.destroyImage++; + } + g_realDestroyImage(device, image, pAllocator); +} + +VKAPI_ATTR void VKAPI_CALL CountingFreeMemory( + VkDevice device, VkDeviceMemory memory, + const VkAllocationCallbacks* pAllocator) +{ + if (memory != VK_NULL_HANDLE) { + g_c.freeMemory++; + } + g_realFreeMemory(device, memory, pAllocator); +} + +bool InstallCountingDispatch(VulkanDeviceContext& ctx) +{ + if ((ctx.CreateImage == nullptr) || (ctx.AllocateMemory == nullptr) || + (ctx.DestroyImage == nullptr) || (ctx.FreeMemory == nullptr)) { + return false; + } + g_realCreateImage = ctx.CreateImage; + g_realAllocateMemory = ctx.AllocateMemory; + g_realDestroyImage = ctx.DestroyImage; + g_realFreeMemory = ctx.FreeMemory; + + ctx.CreateImage = CountingCreateImage; + ctx.AllocateMemory = CountingAllocateMemory; + ctx.DestroyImage = CountingDestroyImage; + ctx.FreeMemory = CountingFreeMemory; + return true; +} + +void RemoveCountingDispatch(VulkanDeviceContext& ctx) +{ + if (g_realCreateImage != nullptr) { + ctx.CreateImage = g_realCreateImage; + ctx.AllocateMemory = g_realAllocateMemory; + ctx.DestroyImage = g_realDestroyImage; + ctx.FreeMemory = g_realFreeMemory; + } +} + +//============================================================================= +// The harness. +//============================================================================= + +// The subject's own count, mirrored. The library's +// kVkEncImportOrdinalGuardCount lives in an anonymous namespace, so it cannot +// be read from here -- and mirroring it is the more useful behaviour anyway, +// because the count is a claim about a driver defect and a change to it +// should have to come here and say so. +// +// THE DEFAULT IS 0 BECAUSE THE GUARD IS RETIRED. A non-zero K here would be +// an OBSERVATION, not a fix. A run that registers only a few buffers samples +// one period of the damage pattern at one phase, so a K that moves the +// damaged ordinals off those buffers reads as a fix. Sweeping the import +// ordinal far enough shows that the damage is periodic in the ordinal and +// that its RATE does not move with K: K changes which imports are damaged, +// never how many. +// +// --guard-count=N is what keeps this suite a test of the KNOB and not only of +// its retirement. Point it at a library built with +// kVkEncImportOrdinalGuardCount = N and every positional assertion below runs +// as it did when N was the default. At the shipped N = 0 those same +// assertions say the guard is INERT. +constexpr uint32_t kDefaultGuardCount = 0; +uint32_t g_guardCount = kDefaultGuardCount; + +// 4:2:0, so both dimensions must be even. Small on purpose: nothing here +// looks at a pixel, and a 1920x1080 NV12 allocation per import x3 buys +// nothing but runtime. +constexpr uint32_t kWidth = 640; +constexpr uint32_t kHeight = 480; +constexpr VkFormat kFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; // NV12 + +// Exactly the exporter shape the sibling linux_dmabuf_import test measures +// green on this host, and the shape a Chromium/GBM producer presents: usage +// 0x4007, flags 0x100108. The library copies both into its own +// VkImageCreateInfo verbatim (it never invents a usage), so the importer's +// create info can only be right if the exporter's was. +constexpr VkImageUsageFlags kExportUsage = + VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT | + VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR; +constexpr VkImageCreateFlags kExportFlags = + VK_IMAGE_CREATE_EXTENDED_USAGE_BIT | VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT | + VK_IMAGE_CREATE_VIDEO_PROFILE_INDEPENDENT_BIT_KHR; + +struct Exported { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + uint64_t modifier = 0; + uint32_t memPlaneCount = 0; + VkSubresourceLayout planes[VK_VIDEO_ENCODER_MAX_PLANES]{}; +}; + +class Harness { +public: + // 0 ok, 1 assertion failed, 77 cannot run here. + int Run(bool expectDisabled); + ~Harness() { Teardown(); } + +private: + int Setup(); // 0 ok, 77 cannot run here + bool ExportDmaBuf(); + bool PickModifier(uint64_t* outModifier, uint32_t* outPlaneCount); + int ExportFd(); // a new dma-buf fd, or -1 + void BuildDescriptor(VkVideoEncoderExternalImageDescriptor* outDesc) const; + void Teardown(); + + VulkanDeviceContext m_ctx; + VkDevice m_device = VK_NULL_HANDLE; + bool m_deviceUp = false; + bool m_thunksIn = false; + Exported m_exp; + VkPhysicalDeviceMemoryProperties m_memProps{}; + std::string m_why; // why Setup() gave up + + // The caller-visible imports, destroyed by Teardown() with the thunks + // already removed so the cleanup cannot move a counter an assertion read. + std::vector m_callerImports; +}; + +bool Harness::PickModifier(uint64_t* outModifier, uint32_t* outPlaneCount) +{ + VkPhysicalDevice phys = m_ctx.getPhysicalDevice(); + + VkDrmFormatModifierPropertiesListEXT list{ + VK_STRUCTURE_TYPE_DRM_FORMAT_MODIFIER_PROPERTIES_LIST_EXT}; + VkFormatProperties2 fmtProps{VK_STRUCTURE_TYPE_FORMAT_PROPERTIES_2}; + fmtProps.pNext = &list; + m_ctx.GetPhysicalDeviceFormatProperties2(phys, kFormat, &fmtProps); + if (list.drmFormatModifierCount == 0) { + m_why = "the driver reports no DRM format modifiers for NV12"; + return false; + } + std::vector props( + list.drmFormatModifierCount); + list.pDrmFormatModifierProperties = props.data(); + m_ctx.GetPhysicalDeviceFormatProperties2(phys, kFormat, &fmtProps); + + VkFormat viewFormat = kFormat; + for (const auto& p : props) { + if ((p.drmFormatModifierPlaneCount == 0) || + (p.drmFormatModifierPlaneCount > VK_VIDEO_ENCODER_MAX_PLANES)) { + continue; + } + uint64_t mod = p.drmFormatModifier; + + VkPhysicalDeviceExternalImageFormatInfo extInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_IMAGE_FORMAT_INFO}; + extInfo.handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + VkPhysicalDeviceImageDrmFormatModifierInfoEXT modInfo{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_DRM_FORMAT_MODIFIER_INFO_EXT}; + modInfo.pNext = &extInfo; + modInfo.drmFormatModifier = mod; + modInfo.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + + VkImageFormatListCreateInfo formatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + formatListCI.pNext = &modInfo; + formatListCI.viewFormatCount = 1; + formatListCI.pViewFormats = &viewFormat; + + VkPhysicalDeviceImageFormatInfo2 info{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_IMAGE_FORMAT_INFO_2}; + info.pNext = (void*)&formatListCI; + info.format = kFormat; + info.type = VK_IMAGE_TYPE_2D; + info.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + info.usage = kExportUsage; + info.flags = kExportFlags; + + VkExternalImageFormatProperties extProps{ + VK_STRUCTURE_TYPE_EXTERNAL_IMAGE_FORMAT_PROPERTIES}; + VkImageFormatProperties2 out{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_PROPERTIES_2}; + out.pNext = &extProps; + + if (m_ctx.GetPhysicalDeviceImageFormatProperties2(phys, &info, &out) != + VK_SUCCESS) { + continue; + } + const VkExternalMemoryFeatureFlags f = + extProps.externalMemoryProperties.externalMemoryFeatures; + if (((f & VK_EXTERNAL_MEMORY_FEATURE_EXPORTABLE_BIT) == 0) || + ((f & VK_EXTERNAL_MEMORY_FEATURE_IMPORTABLE_BIT) == 0)) { + continue; + } + *outModifier = mod; + *outPlaneCount = p.drmFormatModifierPlaneCount; + return true; + } + m_why = "no DRM modifier for NV12 is advertised as both EXPORTABLE and " + "IMPORTABLE for dma-buf with the encode-source usage/flags"; + return false; +} + +bool Harness::ExportDmaBuf() +{ + if (!PickModifier(&m_exp.modifier, &m_exp.memPlaneCount)) { + return false; + } + std::printf("[INFO] exporting NV12 %ux%u modifier %s (%u memory plane%s)\n", + kWidth, kHeight, Hex64(m_exp.modifier).c_str(), + m_exp.memPlaneCount, (m_exp.memPlaneCount == 1) ? "" : "s"); + + VkFormat viewFormat = kFormat; + uint64_t modifier = m_exp.modifier; + + VkExternalMemoryImageCreateInfo extMemCI{ + VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_IMAGE_CREATE_INFO}; + extMemCI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + VkImageDrmFormatModifierListCreateInfoEXT drmList{ + VK_STRUCTURE_TYPE_IMAGE_DRM_FORMAT_MODIFIER_LIST_CREATE_INFO_EXT}; + drmList.pNext = &extMemCI; + drmList.drmFormatModifierCount = 1; + drmList.pDrmFormatModifiers = &modifier; + + VkImageFormatListCreateInfo formatListCI{ + VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO}; + formatListCI.pNext = &drmList; + formatListCI.viewFormatCount = 1; + formatListCI.pViewFormats = &viewFormat; + + VkImageCreateInfo imageCI{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + imageCI.pNext = (void*)&formatListCI; + imageCI.flags = kExportFlags; + imageCI.imageType = VK_IMAGE_TYPE_2D; + imageCI.format = kFormat; + imageCI.extent = {kWidth, kHeight, 1}; + imageCI.mipLevels = 1; + imageCI.arrayLayers = 1; + imageCI.samples = VK_SAMPLE_COUNT_1_BIT; + imageCI.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + imageCI.usage = kExportUsage; + imageCI.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + imageCI.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + + if (m_ctx.CreateImage(m_device, &imageCI, nullptr, &m_exp.image) != + VK_SUCCESS) { + m_why = "vkCreateImage rejected the modifier the driver had just " + "advertised as exportable"; + return false; + } + + VkMemoryRequirements memReqs{}; + m_ctx.GetImageMemoryRequirements(m_device, m_exp.image, &memReqs); + + uint32_t typeIndex = UINT32_MAX; + for (uint32_t i = 0; i < m_memProps.memoryTypeCount; i++) { + if (((memReqs.memoryTypeBits & (1u << i)) != 0) && + ((m_memProps.memoryTypes[i].propertyFlags & + VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT) != 0)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + m_why = "no DEVICE_LOCAL memory type in the export image's " + "requirements mask"; + return false; + } + + VkMemoryDedicatedAllocateInfo dedicated{ + VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO}; + dedicated.image = m_exp.image; + + VkExportMemoryAllocateInfo exportAI{ + VK_STRUCTURE_TYPE_EXPORT_MEMORY_ALLOCATE_INFO}; + exportAI.pNext = &dedicated; + exportAI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + + VkMemoryAllocateInfo allocInfo{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + allocInfo.pNext = &exportAI; + allocInfo.allocationSize = memReqs.size; + allocInfo.memoryTypeIndex = typeIndex; + + if (m_ctx.AllocateMemory(m_device, &allocInfo, nullptr, &m_exp.memory) != + VK_SUCCESS) { + m_why = "the exportable dedicated allocation failed"; + return false; + } + if (m_ctx.BindImageMemory(m_device, m_exp.image, m_exp.memory, 0) != + VK_SUCCESS) { + m_why = "vkBindImageMemory failed on the export image"; + return false; + } + + // For a DRM-modifier image the layouts come from the MEMORY_PLANE aspects + // and the count is the modifier's drmFormatModifierPlaneCount -- not the + // colour plane count, which for NV12 happens to agree but need not. + for (uint32_t p = 0; p < m_exp.memPlaneCount; p++) { + VkImageSubresource sub{}; + sub.aspectMask = + (VkImageAspectFlags)(VK_IMAGE_ASPECT_MEMORY_PLANE_0_BIT_EXT << p); + m_ctx.GetImageSubresourceLayout(m_device, m_exp.image, &sub, + &m_exp.planes[p]); + } + return true; +} + +int Harness::ExportFd() +{ + if (m_ctx.GetMemoryFdKHR == nullptr) { + return -1; + } + VkMemoryGetFdInfoKHR getFd{VK_STRUCTURE_TYPE_MEMORY_GET_FD_INFO_KHR}; + getFd.memory = m_exp.memory; + getFd.handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_DMA_BUF_BIT_EXT; + int fd = -1; + if ((m_ctx.GetMemoryFdKHR(m_device, &getFd, &fd) != VK_SUCCESS) || + (fd < 0)) { + return -1; + } + return fd; +} + +void Harness::BuildDescriptor( + VkVideoEncoderExternalImageDescriptor* outDesc) const +{ + VkVideoEncoderExternalImageDescriptor& d = *outDesc; + d = VkVideoEncoderExternalImageDescriptor{}; + d.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + // THE POINT OF THE WHOLE FILE. This is the only encoder-ext test that sets + // this member to DMA_BUF; without it VkEncEnsureImportOrdinalGuard returns + // at its handleType check on every test in the tree. + d.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_DMA_BUF; + d.format = kFormat; + d.width = kWidth; + d.height = kHeight; + d.tiling = VK_IMAGE_TILING_DRM_FORMAT_MODIFIER_EXT; + d.imageUsage = kExportUsage; + d.imageFlags = kExportFlags; + d.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + + d.hasDrmFormatModifier = VK_TRUE; + d.drmFormatModifier = m_exp.modifier; + d.planeCount = m_exp.memPlaneCount; + for (uint32_t p = 0; p < m_exp.memPlaneCount; p++) { + d.planeLayouts[p].offset = m_exp.planes[p].offset; + // MUST be 0 on the explicit DRM path + // (VUID-VkImageDrmFormatModifierExplicitCreateInfoEXT-size-02267); + // the import zeroes it anyway, and passing the exporter's value here + // would make this descriptor a poor model of a real producer's. + d.planeLayouts[p].size = 0; + d.planeLayouts[p].rowPitch = m_exp.planes[p].rowPitch; + d.planeLayouts[p].arrayPitch = m_exp.planes[p].arrayPitch; + d.planeLayouts[p].depthPitch = m_exp.planes[p].depthPitch; + } + + // Left UNKNOWN on purpose, which is both what the public header + // prescribes for dma-buf ("Deriving allocationSize from + // vkGetImageMemoryRequirements is wrong for dma-buf on NVIDIA. 0 means + // unknown") and what a GBM/Ozone producer actually has: it holds a + // gfx::NativePixmapHandle, which carries no Vulkan memory type at all. + d.allocationSize = 0; + d.memoryTypeBits = 0; + d.memoryTypeIndex = UINT32_MAX; + + VkPhysicalDeviceIDProperties idProps{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_ID_PROPERTIES}; + VkPhysicalDeviceProperties2 props2{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2}; + props2.pNext = &idProps; + m_ctx.GetPhysicalDeviceProperties2(m_ctx.getPhysicalDevice(), &props2); + std::memcpy(d.deviceUUID, idProps.deviceUUID, VK_UUID_SIZE); + std::memcpy(d.driverUUID, idProps.driverUUID, VK_UUID_SIZE); +} + +int Harness::Setup() +{ + static const char* const instanceExtensions[] = { + VK_KHR_GET_PHYSICAL_DEVICE_PROPERTIES_2_EXTENSION_NAME, + VK_KHR_EXTERNAL_MEMORY_CAPABILITIES_EXTENSION_NAME, + nullptr + }; + m_ctx.AddReqInstanceExtensions(instanceExtensions, g_verbose); + + // The precondition set the library itself gates the dma-buf import on, + // plus the video pair the encode-source usage needs. Requiring them here + // means a host that cannot possibly run this says so by name at init + // rather than failing an import later for an unrelated-looking reason. + static const char* const requiredDeviceExtensions[] = { + VK_KHR_EXTERNAL_MEMORY_FD_EXTENSION_NAME, + VK_EXT_EXTERNAL_MEMORY_DMA_BUF_EXTENSION_NAME, + VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME, + VK_EXT_IMAGE_DRM_FORMAT_MODIFIER_EXTENSION_NAME, + VK_KHR_EXTERNAL_MEMORY_EXTENSION_NAME, + VK_KHR_IMAGE_FORMAT_LIST_EXTENSION_NAME, + VK_KHR_BIND_MEMORY_2_EXTENSION_NAME, + VK_KHR_SAMPLER_YCBCR_CONVERSION_EXTENSION_NAME, + VK_KHR_MAINTENANCE_1_EXTENSION_NAME, + VK_KHR_GET_MEMORY_REQUIREMENTS_2_EXTENSION_NAME, + VK_KHR_DEDICATED_ALLOCATION_EXTENSION_NAME, + nullptr + }; + m_ctx.AddReqDeviceExtensions(requiredDeviceExtensions, g_verbose); + + // VIDEO_ENCODE_SRC usage and VIDEO_PROFILE_INDEPENDENT are what a real + // encode producer declares; optional here so the arm degrades to a named + // skip rather than an unexplained device-selection failure. + static const char* const videoExtensions[] = { + VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME, + VK_KHR_TIMELINE_SEMAPHORE_EXTENSION_NAME, + VK_KHR_VIDEO_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME, + VK_KHR_VIDEO_MAINTENANCE_1_EXTENSION_NAME, + nullptr + }; + m_ctx.AddOptDeviceExtensions(videoExtensions, g_verbose); + + if (m_ctx.InitVulkanDevice("EncoderExtImportOrdinalGuardTest", + VK_NULL_HANDLE, g_verbose) != VK_SUCCESS) { + m_why = "InitVulkanDevice failed (no Vulkan loader or no ICD)"; + return 77; + } + + vk::DeviceUuidUtils deviceUuid; + if (m_ctx.InitPhysicalDevice(-1, deviceUuid, + VK_QUEUE_TRANSFER_BIT | VK_QUEUE_COMPUTE_BIT, + nullptr, + 0, VK_VIDEO_CODEC_OPERATION_NONE_KHR, + 0, VK_VIDEO_CODEC_OPERATION_NONE_KHR) != + VK_SUCCESS) { + m_why = "InitPhysicalDevice found no device offering the dma-buf / " + "DRM-modifier extension set"; + return 77; + } + for (const char* const* p = requiredDeviceExtensions; *p != nullptr; ++p) { + if (m_ctx.FindRequiredDeviceExtension(*p) == nullptr) { + m_why = std::string("required device extension not available: ") + *p; + return 77; + } + } + + VkPhysicalDeviceProperties props{}; + m_ctx.GetPhysicalDeviceProperties(m_ctx.getPhysicalDevice(), &props); + std::printf("[INFO] physical device: %s vendorID 0x%04X driver %u.%u.%u\n", + props.deviceName, props.vendorID, + VK_VERSION_MAJOR(props.driverVersion), + VK_VERSION_MINOR(props.driverVersion), + VK_VERSION_PATCH(props.driverVersion)); + + // THE VENDOR GATE, AND WHY IT IS A SKIP AND NOT A PASS. The guard refuses + // to run on anything but 0x10DE, deliberately -- "a workaround must not + // run where the defect it works around has never been observed". On any + // other vendor every assertion below would be asserting the ABSENCE of + // the behaviour under test, which is not evidence that the behaviour is + // right where it does run. Reporting PASSED there is precisely how a gate + // stops being able to fail. + if (props.vendorID != 0x10DE) { + m_why = std::string("this host's GPU is vendor 0x") + + Hex64(props.vendorID) + + ", and the import-ordinal guard is gated to NVIDIA (0x10DE) " + "by design -- there is nothing here to prove either way"; + return 77; + } + + if (m_ctx.CreateVulkanDevice(0, 0, VK_VIDEO_CODEC_OPERATION_NONE_KHR, + /*transfer*/ true, /*graphics*/ false, + /*present*/ false, /*compute*/ true) != + VK_SUCCESS) { + m_why = "CreateVulkanDevice failed"; + return 77; + } + m_deviceUp = true; + m_device = m_ctx.getDevice(); + if (m_device == VK_NULL_HANDLE) { + m_why = "the device context produced a null VkDevice"; + return 77; + } + m_ctx.GetPhysicalDeviceMemoryProperties(m_ctx.getPhysicalDevice(), + &m_memProps); + + if (m_ctx.GetMemoryFdKHR == nullptr) { + m_why = "the device has no vkGetMemoryFdKHR, so no dma-buf can be " + "exported to import"; + return 77; + } + if (!ExportDmaBuf()) { + return 77; // m_why already set + } + return 0; +} + +void Harness::Teardown() +{ + if (!m_deviceUp) { + return; + } + // Thunks out FIRST. Cleanup must not be able to move a counter that an + // assertion has already read, and more importantly must not be able to + // make a later assertion true by accident. + RemoveCountingDispatch(m_ctx); + m_thunksIn = false; + + m_ctx.DeviceWaitIdle(); + for (auto& imp : m_callerImports) { + if (imp.image != VK_NULL_HANDLE) { + m_ctx.DestroyImage(m_device, imp.image, nullptr); + } + if (imp.memory != VK_NULL_HANDLE) { + m_ctx.FreeMemory(m_device, imp.memory, nullptr); + } + } + m_callerImports.clear(); + if (m_exp.image != VK_NULL_HANDLE) { + m_ctx.DestroyImage(m_device, m_exp.image, nullptr); + m_exp.image = VK_NULL_HANDLE; + } + if (m_exp.memory != VK_NULL_HANDLE) { + m_ctx.FreeMemory(m_device, m_exp.memory, nullptr); + m_exp.memory = VK_NULL_HANDLE; + } + // The guard's own two imports are not destroyed HERE and could not be -- + // their handles never leave the library. Run() has already asked the + // library to take them down, through the same + // VkEncReleaseImportOrdinalGuard() call its Deinitialize() makes, and + // asserted that it did. Nothing is papered over on this path: if Run() + // exited early the guards are still live, ~VulkanDeviceContext destroys + // the device under them, and that is the spec violation the release + // exists to close -- visible to a validation layer, not hidden by this + // teardown. + m_deviceUp = false; +} + +int Harness::Run(bool expectDisabled) +{ + const int setup = Setup(); + if (setup != 0) { + std::printf("SKIP: %s\n", m_why.c_str()); + return 77; + } + + if (!InstallCountingDispatch(m_ctx)) { + std::printf("SKIP: the device dispatch table is missing one of " + "vkCreateImage / vkAllocateMemory / vkDestroyImage / " + "vkFreeMemory\n"); + return 77; + } + m_thunksIn = true; + + VkVideoEncoderExternalImageDescriptor desc{}; + BuildDescriptor(&desc); + + // What the library is expected to do on THIS run: the build's count, or 0 + // when the environment kill switch has taken it out. Both zeroes produce + // the same observable -- one image, nothing retained, verdict DISABLED -- + // and that is not a coincidence to paper over: the retired default IS the + // kill switch, permanently, which is exactly why the count-0 path reports + // DISABLED rather than a vacuously-satisfied COMPLETE. + const uint32_t effectiveCount = expectDisabled ? 0u : g_guardCount; + const uint32_t expectedFirst = effectiveCount + 1u; + + std::printf("\n=== import 1: the first DMA_BUF import on this VkDevice " + "===\n"); + if (effectiveCount == 0) { + std::printf("[INFO] guard expectation: INERT (%s) -- the caller's " + "import must be the ONLY one and the verdict DISABLED\n", + expectDisabled + ? "VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD is set" + : "this build's guard count is 0, the retired " + "default"); + } else { + std::printf("[INFO] guard expectation: ARMED -- %u sacrificial " + "import(s) must be taken and RETAINED ahead of the " + "caller's, and the verdict must be COMPLETE\n", + effectiveCount); + } + + int fd1 = ExportFd(); + if (fd1 < 0) { + Check(false, "exported a dma-buf fd for import 1", + "vkGetMemoryFdKHR(DMA_BUF) failed after the export image was " + "already built, which is a driver fact and not a host one"); + return 1; + } + std::printf("[INFO] exported dma-buf fd %d\n", fd1); + + // The thread's verdict record, cleared exactly as RegisterImageResource + // clears it at the top of every registration -- so what is read back + // below is this import's answer and not a previous one's. + VkEncResetImportOrdinalGuardReport(); + + VkEncImportedImage first{}; + const VkVideoEncoderStatusCode s1 = + VkEncImportExternalImage(m_ctx, desc, (uint64_t)fd1, &first); + // TRANSFER ownership: the fd is consumed on EVERY exit, success or not. + fd1 = -1; + + Check(s1 == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "the caller's DMA_BUF import succeeded", + "VkEncImportExternalImage returned status " + + U32((uint32_t)s1)); + if (s1 != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + // Nothing below can mean anything if the subject never ran. This is a + // FAILURE and not a skip: the host question was settled in Setup(), + // which built this very image on this very device and exported this + // very fd from it. + std::printf("\n%d/%d checks failed\n", g_failures, g_checks); + return 1; + } + m_callerImports.push_back(first); + + Check(g_c.createImage == expectedFirst, + "vkCreateImage call count for the first caller import", + "expected " + U32(expectedFirst) + " (the guard's " + + U32(effectiveCount) + + " sacrificial import(s) plus the caller's), got " + + U32(g_c.createImage)); + + Check(g_c.importAlloc == expectedFirst, + "fd-importing vkAllocateMemory call count for the first caller " + "import", + "expected " + U32(expectedFirst) + ", got " + U32(g_c.importAlloc)); + + // THE POSITIONAL CLAIM, stated in the only terms the driver understands. + // With the guard on, images[0] and images[1] are the sacrificial pair and + // the caller's is images[2] -- i.e. the caller's import is at + // live-position 3, which is the entire content of the workaround. + Check((g_c.images.size() == expectedFirst) && + (g_c.images.back() == first.image), + "the caller's VkImage is the one created LAST, at live-position " + + U32(expectedFirst), + "created " + U32((uint32_t)g_c.images.size()) + + " image(s); the caller got " + Ptr((void*)(uintptr_t)first.image) + + " and the last created was " + + (g_c.images.empty() + ? std::string("(none)") + : Ptr((void*)(uintptr_t)g_c.images.back()))); + + if (effectiveCount > 0) { + bool distinct = (g_c.images.size() >= effectiveCount); + for (uint32_t i = 0; distinct && (i < effectiveCount); i++) { + distinct = (g_c.images[i] != first.image); + } + Check(distinct, + "the " + U32(effectiveCount) + + " image(s) created before the caller's are NOT the caller's", + "the guard did not take a distinct set ahead of the import"); + } else { + // THE INERTNESS CLAIM, which is the one the SHIPPED build makes: a + // retired guard creates nothing of its own. This is not the same + // assertion as the count above -- that one would still hold if the + // library created an extra image and handed it to the caller -- and + // it is what goes red if someone re-arms the guard without telling + // this suite. + Check(g_c.images.size() == 1, + "a retired guard created exactly ONE image for one import", + "created " + U32((uint32_t)g_c.images.size()) + " image(s)"); + } + + // THE RETENTION CLAIM. "2, created then DESTROYED" is a row in the + // original measurement table and it is 100% bad, so a guard that freed + // its imports would satisfy every count above while not even delivering + // the phase shift it claims. That is what this assertion separates: a + // mechanism that does what it says from a decoration. It says nothing + // about whether doing what it says is worth anything -- measurement says + // it is not, which is why the shipped count is 0. + Check((g_c.destroyImage == 0) && (g_c.freeMemory == 0), + (effectiveCount == 0) + ? std::string("nothing has been destroyed, and a retired guard " + "had nothing to destroy") + : std::string("nothing has been destroyed -- the sacrificial " + "imports are RETAINED"), + "vkDestroyImage x" + U32(g_c.destroyImage) + ", vkFreeMemory x" + + U32(g_c.freeMemory) + + "; a guard whose imports are not LIVE alongside the caller's " + "does not shift the caller's import ordinal at all"); + + // --------------------------------------------------------------------- + // THE LIBRARY'S OWN VERDICT, CHECKED AGAINST WHAT THE DRIVER WAS ASKED. + // + // The sibling encoder-ext-import-guard suite gates this record's carrier + // device-free, and states plainly that COMPLETE and INCOMPLETE "are + // covered only by a hardware run". This is that run: it is the only place + // in the tree where a real dma-buf import can produce either verdict. + // + // The cross-check is the load-bearing line. A report that said COMPLETE + // while one import had been taken would satisfy any report-only test; it + // cannot satisfy retainedCount == (images created - the caller's one). + // --------------------------------------------------------------------- + VkEncImportOrdinalGuardReport report{}; + VkEncGetImportOrdinalGuardReport(&report); + std::printf("[INFO] guard report: state=%s requested=%u retained=%u " + "failureStatus=%u errno=%d\n", + GuardStateName(report.state), report.requestedCount, + report.retainedCount, (uint32_t)report.failureStatus, + report.failureErrno); + + Check(report.requestedCount == g_guardCount, + "the guard report's requestedCount is this build's count", + "expected " + U32(g_guardCount) + ", got " + + U32(report.requestedCount) + + "; requestedCount is a BUILD constant, stamped even when the " + "kill switch is set, so it does NOT follow --expect-disabled. " + "If this is the only failing line, the library's " + "kVkEncImportOrdinalGuardCount and this run's --guard-count " + "disagree"); + + // DISABLED covers BOTH zeroes: the environment kill switch returns before + // the count is consulted, and a build count of 0 returns just after the + // vendor probe with the same verdict. COMPLETE is reserved for a build + // that actually retained something. 0 == 0 must never report COMPLETE -- + // vulkan_video_encoder_ext_internal.h defines that verdict as caller + // imports landing past a position they did not, at 0, move to, and a + // workaround that reports + // success on a build which removed it is worse than one that reports + // nothing. + const VkVideoEncoderImportGuardState expectedState = + (effectiveCount == 0) ? VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_DISABLED + : VK_VIDEO_ENCODER_IMPORT_GUARD_STATE_COMPLETE; + Check(report.state == expectedState, + std::string("the guard report's state is ") + + GuardStateName(expectedState), + std::string("got ") + GuardStateName(report.state)); + + const uint32_t expectedRetained = effectiveCount; + Check(report.retainedCount == expectedRetained, + "the guard report's retainedCount", + "expected " + U32(expectedRetained) + ", got " + + U32(report.retainedCount)); + + Check(report.retainedCount == (g_c.createImage - 1), + "the report AGREES with the driver: retainedCount == images created " + "minus the caller's one", + "the library reported retaining " + U32(report.retainedCount) + + " sacrificial import(s) but " + U32(g_c.createImage) + + " image(s) were created for one caller import; a report that " + "does not match the calls is worse than no report"); + + std::printf("\n=== import 2: the guard must NOT run again ===\n"); + const uint32_t createBefore = g_c.createImage; + const uint32_t importBefore = g_c.importAlloc; + + int fd2 = ExportFd(); + if (fd2 < 0) { + Check(false, "exported a second dma-buf fd", "vkGetMemoryFdKHR failed"); + std::printf("\n%d/%d checks failed\n", g_failures, g_checks); + return 1; + } + VkEncResetImportOrdinalGuardReport(); + VkEncImportedImage second{}; + const VkVideoEncoderStatusCode s2 = + VkEncImportExternalImage(m_ctx, desc, (uint64_t)fd2, &second); + fd2 = -1; + Check(s2 == VK_VIDEO_ENCODER_STATUS_SUCCESS, + "the second caller DMA_BUF import succeeded", + "status " + U32((uint32_t)s2)); + if (s2 == VK_VIDEO_ENCODER_STATUS_SUCCESS) { + m_callerImports.push_back(second); + } + + Check(g_c.createImage == (createBefore + 1), + "the second caller import cost exactly ONE vkCreateImage", + "expected " + U32(createBefore + 1) + ", got " + + U32(g_c.createImage) + + "; the guard is documented as a no-op after the first import " + "and carries no per-frame cost"); + Check(g_c.importAlloc == (importBefore + 1), + "the second caller import cost exactly ONE fd import", + "expected " + U32(importBefore + 1) + ", got " + + U32(g_c.importAlloc)); + Check((s2 != VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (g_c.images.back() == second.image), + "the second caller import is the last image created", + "the library minted an image after the caller's"); + Check((g_c.destroyImage == 0) && (g_c.freeMemory == 0), + "still nothing destroyed after the second import", + "vkDestroyImage x" + U32(g_c.destroyImage) + ", vkFreeMemory x" + + U32(g_c.freeMemory)); + + // The second import takes the guard's already-satisfied early return, and + // that return still owes the caller a verdict -- a registration that + // reported NOT_EVALUATED here would tell a consumer the guard had not run + // on an import it is in fact protecting. + VkEncImportOrdinalGuardReport report2{}; + VkEncGetImportOrdinalGuardReport(&report2); + std::printf("[INFO] guard report: state=%s requested=%u retained=%u\n", + GuardStateName(report2.state), report2.requestedCount, + report2.retainedCount); + Check(report2.state == expectedState, + std::string("the second import still reports ") + + GuardStateName(expectedState), + std::string("got ") + GuardStateName(report2.state) + + " -- the already-guarded early return must not drop the verdict"); + Check(report2.retainedCount == expectedRetained, + "the second import reports the SAME retainedCount", + "expected " + U32(expectedRetained) + ", got " + + U32(report2.retainedCount) + + "; the guard is documented as a no-op after the first import"); + + // --------------------------------------------------------------------- + // THE OTHER END OF THE GUARD'S LIFETIME. + // + // VkEncReleaseImportOrdinalGuard is what stops vkDestroyDevice running + // with two VkImage and two VkDeviceMemory still alive on the device + // (VUID-vkDestroyDevice-device-05137). Nothing else in the tree calls it: + // the production call site is + // VulkanVideoEncoderExtImpl::Deinitialize(), reachable only through a + // full encode session, and no session in the corpus performs a dma-buf + // import -- so the function ran on exactly zero test paths and the leak + // it fixes could come back with the suite green. + // + // It is asserted the same way as everything else here: through the + // driver. Two destroys and two frees must actually reach the device, not + // just a return value saying they did. + // + // THIS IS ALSO WHY THE TEST OWNS ITS OWN DEVICE. Releasing the guard is + // only safe where no further import on this device is possible; here that + // is true by construction, because the next thing that happens is + // teardown. + // --------------------------------------------------------------------- + std::printf("\n=== release: the guard must come down with the device ===\n"); + const uint32_t destroyBefore = g_c.destroyImage; + const uint32_t freeBefore = g_c.freeMemory; + + const uint32_t released = VkEncReleaseImportOrdinalGuard(m_ctx); + Check(released == expectedRetained, + "VkEncReleaseImportOrdinalGuard returned the guard count", + "expected " + U32(expectedRetained) + ", got " + U32(released) + + "; the count is a RETURN VALUE and not only a log line because " + "the shipping embedder sets silenceStdio"); + Check((g_c.destroyImage == (destroyBefore + expectedRetained)) && + (g_c.freeMemory == (freeBefore + expectedRetained)), + "the release destroyed exactly the guard's images and memories", + "vkDestroyImage " + U32(destroyBefore) + " -> " + + U32(g_c.destroyImage) + ", vkFreeMemory " + U32(freeBefore) + + " -> " + U32(g_c.freeMemory) + ", expected +" + + U32(expectedRetained) + " each"); + + // Idempotent and null-safe, per its own contract: a second call must + // return 0 and destroy nothing. Without the take-it-off-the-registry- + // first ordering this is a double free. + const uint32_t destroyAfterFirst = g_c.destroyImage; + const uint32_t freeAfterFirst = g_c.freeMemory; + const uint32_t releasedAgain = VkEncReleaseImportOrdinalGuard(m_ctx); + Check(releasedAgain == 0, + "a second release returns 0", + "got " + U32(releasedAgain)); + Check((g_c.destroyImage == destroyAfterFirst) && + (g_c.freeMemory == freeAfterFirst), + "a second release destroys nothing", + "vkDestroyImage " + U32(destroyAfterFirst) + " -> " + + U32(g_c.destroyImage) + ", vkFreeMemory " + U32(freeAfterFirst) + + " -> " + U32(g_c.freeMemory) + " -- that is a double free"); + + std::printf("\n=== summary ===\n"); + std::printf(" vkCreateImage %u\n", g_c.createImage); + std::printf(" vkAllocateMemory(import) %u\n", g_c.importAlloc); + std::printf(" vkAllocateMemory(plain) %u\n", g_c.plainAlloc); + std::printf(" vkDestroyImage %u\n", g_c.destroyImage); + std::printf(" vkFreeMemory %u\n", g_c.freeMemory); + std::printf(" caller imports 2\n"); + std::printf(" guard imports (derived) %u\n", + (g_c.createImage >= 2) ? (g_c.createImage - 2) : 0); + std::printf(" guard count expected %u (%s)\n", effectiveCount, + (effectiveCount == 0) ? "INERT" : "ARMED"); + std::printf(" (the destroy/free counts above are the guard RELEASE; the " + "two caller imports are torn down after the thunks come " + "out)\n"); + + std::printf("\n%d/%d checks failed\n", g_failures, g_checks); + return (g_failures == 0) ? 0 : 1; +} + +void Usage(const char* argv0) +{ + std::printf( + "usage: %s [--guard-count=N] [--expect-disabled] [--verbose]\n" + " (default) assert the import-ordinal guard is INERT, which\n" + " is what the shipped library does: its build\n" + " constant kVkEncImportOrdinalGuardCount is 0, so\n" + " a caller's dma-buf import must be the ONLY image\n" + " created and the verdict must be DISABLED.\n" + " --guard-count=N assert the ARMED behaviour of a library rebuilt\n" + " with kVkEncImportOrdinalGuardCount = N: N\n" + " sacrificial imports taken and retained ahead of\n" + " the caller's, verdict COMPLETE, and N destroyed\n" + " on release. N must match the library under test\n" + " and must fit kVkEncImportOrdinalGuardCapacity.\n" + " --expect-disabled assert the kill switch forces the inert shape.\n" + " That is what setting the environment variable\n" + " VK_VIDEO_ENCODER_NO_IMPORT_ORDINAL_GUARD=1 must\n" + " produce. Against an ARMED library\n" + " (--guard-count=N, N>0) run WITHOUT the variable\n" + " set, this arm fails -- which is the point;\n" + " against the retired default both arms agree.\n", + argv0); +} + +} // namespace + +int main(int argc, char** argv) +{ + bool expectDisabled = false; + for (int i = 1; i < argc; i++) { + const std::string a = argv[i]; + if (a == "--expect-disabled") { + expectDisabled = true; + } else if (a.rfind("--guard-count=", 0) == 0) { + // Hand-parsed: no new include, and a silently-misread count would + // turn every positional assertion below into a different claim. + const std::string v = a.substr(sizeof("--guard-count=") - 1); + if (v.empty() || + (v.find_first_not_of("0123456789") != std::string::npos)) { + std::printf("--guard-count needs a non-negative integer, got " + "'%s'\n", v.c_str()); + return 1; + } + uint32_t n = 0; + for (size_t k = 0; k < v.size(); k++) { + n = (n * 10u) + (uint32_t)(v[k] - '0'); + } + g_guardCount = n; + } else if (a == "--verbose") { + g_verbose = true; + } else if ((a == "-h") || (a == "--help")) { + Usage(argv[0]); + return 0; + } else { + std::printf("unrecognised argument '%s'\n", a.c_str()); + Usage(argv[0]); + return 1; + } + } + + std::printf("=====================================================\n"); + std::printf(" encoder-ext import-ordinal guard\n"); + std::printf("=====================================================\n"); + + Harness h; + const int rc = h.Run(expectDisabled); + + // A failure is never reported as a skip. Run() already returns 1 whenever + // g_failures is non-zero, and this restates it so a future edit to Run() + // cannot quietly relax it. + if ((g_failures > 0) && (rc != 1)) { + std::printf("INTERNAL: %d check(s) failed but the harness returned " + "%d; reporting FAILURE\n", g_failures, rc); + return 1; + } + return rc; +} + +#endif // __linux__ diff --git a/vk_video_encoder/test/encoder-ext-input-format-query/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-input-format-query/CMakeLists.txt new file mode 100644 index 00000000..4e2b58ee --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-input-format-query/CMakeLists.txt @@ -0,0 +1,135 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_input_format_query_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, as the sibling tests do, so the test calls +# the same object files the library ships rather than a copy. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # The descriptor API is an internal header: the public surface of this + # library is the encoder interface, and a test that drives the layer + # beneath it names the internal directory to say so. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + # The public encoder header reaches VkCodecUtils/VkVideoRefCountBase.h, + # which lives under the shared common-libs root. + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# This file deliberately uses the PUBLIC header only. It is the one test in +# this tree that reaches the query the way a client would, so an internal +# include directory here would weaken exactly what it is measuring. + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add test. +# +# A gpu-label test: every row puts a real profile to a real driver, and there +# is no arm of it that means anything without one. It exits 77 (SKIPPED) when +# no encode-capable NVIDIA device is enumerated, 1 when an assertion fails. +# +# VALIDATION-GATED, ALL THREE ROWS. A gate is easy to decline here on the +# ground that "nothing creates a session, an image or a command buffer -- the +# whole file is physical-device-level capability queries". That holds for the +# two query rows only. The --init-gate row drives one InitializeExt per advertised +# entry, and each of those runs CreateVideoEncoder, InitEncoder, the device +# format selection, the preprocess-filter build and the image-pool +# allocation. It is the highest-risk path in this directory, and the one that +# most needs a layer under it. +# +# THE TWO QUERY ROWS ARE GATED TOO, and that is not gate-everything reflex. +# vvs_add_validation_gated_test() reads the layer output ALONGSIDE the exit +# code, so it only adds a way to fail; a row with no spec-violation surface +# gates at zero messages and stays there, and the summary line it writes is +# what shows the layer was actually loaded. The old note's worry -- that a +# gate advertises coverage the test does not have -- is answered by the +# summary file rather than by declining to measure: a run with no layer is +# recorded as NOT-MEASURED, never as clean. +enable_testing() + +vvs_add_validation_gated_test(EncoderExtInputFormatPointQuery + TARGET ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +# No SKIP_RETURN_CODE: the gate carries the skip itself, through +# SKIP_REGULAR_EXPRESSION, because the registered command is now the cmake -P +# wrapper whose exit code is 0 or 1 and never 77. Its own note says why that +# ordering matters -- SKIP_RETURN_CODE outranks every other verdict, so a run +# that emitted validation errors on its way to a skip would be recorded as +# Skipped. +set_tests_properties(EncoderExtInputFormatPointQuery PROPERTIES + LABELS "gpu" + TIMEOUT 300) + +# The cross-route sweep, registered separately because it asserts a different +# proposition: not "is this pair accepted" but "do the two surfaces that answer +# that question give the same answer". Same binary, same context bring-up, same +# SKIP-77 harness -- a second executable would duplicate all three to assert +# something neither copy owns. +vvs_add_validation_gated_test(EncoderExtInputFormatCrossRoute + TARGET ${PROJECT_NAME} + ARGS --cross-route) +set_tests_properties(EncoderExtInputFormatCrossRoute PROPERTIES + LABELS "gpu" + TIMEOUT 300) + +# The initialisation gate, registered separately for the same reason and one +# step further out: --cross-route asserts that the two PRE-SESSION surfaces +# agree with each other, and this asserts that the SESSION agrees with them. +# It is the only row in this binary that creates encoder sessions -- one per +# advertised entry -- which is why it carries a longer timeout than its two +# siblings and why the note above no longer says this binary creates none. +vvs_add_validation_gated_test(EncoderExtInputFormatInitGate + TARGET ${PROJECT_NAME} + ARGS --init-gate) +set_tests_properties(EncoderExtInputFormatInitGate PROPERTIES + LABELS "gpu" + TIMEOUT 900) + +message(STATUS "encoder_ext_input_format_query_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-input-format-query/src/main.cpp b/vk_video_encoder/test/encoder-ext-input-format-query/src/main.cpp new file mode 100644 index 00000000..314b9a19 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-input-format-query/src/main.cpp @@ -0,0 +1,1189 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * THE PRE-SESSION INPUT-FORMAT POINT QUERY, on a real device. + * + * WHAT THIS ANSWERS. VkEncQueryInputFormatSupport is the only surface that + * takes a (format, colour model) PAIR and a (codec, profile) and answers + * before a session exists. QueryImageSupport reads a colour model but is a + * session method; VkEncEnumerateInputFormats is context-level and carries no + * colour model, so it answers the same question for the whole routable set at + * once. This file measures both on hardware. + * + * THE CASE THIS EXISTS FOR. An A4000 encodes NV24 at H.264 High 4:4:4 + * Predictive and S410 at H.265 Range Extensions. A session declared in either + * one works -- the codec config derives the 4:4:4 profile from the input's own + * chroma subsampling -- so those are ACCEPTED, and the assertions below are + * that the point query says so before a frame pool is allocated. They were + * once also UNDISCOVERABLE, because the enumerator answered from a capability + * snapshot probed at a fixed 4:2:0 envelope; section 3 now asserts the + * opposite, and --cross-route asserts that the two surfaces give the same + * answer rather than merely both saying yes. + * + * --cross-route IS A SEPARATE MODE AND A SEPARATE CTEST ROW. It sweeps every + * publishable (codec, profile) key, puts each advertised entry back through + * the point query, and asserts equality of encodeFormat and optimality -- with + * the OPTIMAL arm as its calibration and a refusal control that keeps + * "accurate" distinguishable from "permissive". + * + * --init-gate IS A THIRD MODE AND A THIRD ROW, and it asserts the proposition + * neither of the other two can: that what InitializeExt ACCEPTS is what the + * enumerator ADVERTISES. Both surfaces above are answers ABOUT a session; this + * one builds the session. It is the only mode here that does -- which used to + * be offered as the reason this binary could decline a validation gate, and + * is not: the gate is per binary, and one mode that creates twenty-five + * sessions is the whole binary creating sessions. All three rows are gated. + * + * CALIBRATION ORDER, AND IT IS NOT COSMETIC. The known-good 4:2:0 rows run + * FIRST. An instrument that answered "no" to everything would pass every + * refusal assertion in this file, and a bare refusal proves nothing until the + * same instrument has been seen to say yes to something. So the 4:2:0 rows + * are a precondition for reading anything below them, and they are asserted, + * not merely printed. + * + * WHAT IT CAN FAIL ON: + * - The device leg could be inert -- a query that consulted the 4:2:0 + * capability snapshot instead of asking the device at the input's own + * subsampling answers "no" for NV24 and fails the acceptance rows. + * - The library leg could over-promise -- a query that skipped the profile + * guard would accept H.264 High (100) over NV24, which the standard + * forbids and this device refuses. That row is here. + * - The guard could be indiscriminate -- a library that refused every + * explicit profile would fail the negative control, which asks for High + * (100) over NV12 and requires a yes. + * + * DEVICE IDENTITY IS ASSERTED, NOT ASSUMED. Both lab machines carry a second + * Vulkan device (llvmpipe needs no render node), and a capability verdict read + * off a software rasteriser is worthless. The identity block of every device + * the context enumerated is printed, and the run refuses to proceed on + * anything but an NVIDIA one. + * + * Exits 77 (CTest SKIP) with no encode-capable NVIDIA device, 0 when every + * assertion held, 1 when one did not. + */ + +#include "vulkan_video_encoder_ext.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; + +void Check(bool ok, const std::string& what, const std::string& detail) +{ + g_checks++; + if (ok) { + std::printf(" ok %s\n", what.c_str()); + return; + } + g_failures++; + std::printf(" FAIL %s : %s\n", what.c_str(), detail.c_str()); +} + +std::string U32(uint32_t v) +{ + char buf[24]; + std::snprintf(buf, sizeof(buf), "%u", v); + return buf; +} + +const char* ResultName(VkResult r) +{ + switch (r) { + case VK_SUCCESS: return "VK_SUCCESS"; + case VK_ERROR_FORMAT_NOT_SUPPORTED: return "VK_ERROR_FORMAT_NOT_SUPPORTED"; + case VK_ERROR_INITIALIZATION_FAILED: return "VK_ERROR_INITIALIZATION_FAILED"; + case VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR: + return "VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR"; + case VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR: + return "VK_ERROR_VIDEO_PROFILE_OPERATION_NOT_SUPPORTED_KHR"; + case VK_ERROR_VIDEO_PROFILE_FORMAT_NOT_SUPPORTED_KHR: + return "VK_ERROR_VIDEO_PROFILE_FORMAT_NOT_SUPPORTED_KHR"; + case VK_ERROR_FEATURE_NOT_PRESENT: return "VK_ERROR_FEATURE_NOT_PRESENT"; + default: { + // NAMED BY NUMBER RATHER THAN "other". A code this switch does not + // spell is exactly the interesting case -- the late driver refusal + // this file's --init-gate mode exists to move -- and "other" makes + // two different late refusals read identically in a failure line. + static char buf[32]; + std::snprintf(buf, sizeof(buf), "VkResult %d", (int)r); + return buf; + } + } +} + +// Vulkan enumerants the rows below name. Spelled once so a row reads as the +// format a producer would recognise it by. +const VkFormat kNv12 = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; +const VkFormat kP010 = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16; +const VkFormat kNv24 = VK_FORMAT_G8_B8R8_2PLANE_444_UNORM; +const VkFormat kS410 = VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16; +const VkFormat kNv16 = VK_FORMAT_G8_B8R8_2PLANE_422_UNORM; +const VkFormat kP012 = VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16; + +const VkVideoCodecOperationFlagBitsKHR kH264 = + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; +const VkVideoCodecOperationFlagBitsKHR kH265 = + VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; +const VkVideoCodecOperationFlagBitsKHR kAV1 = + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + +// Spelling for the report only. A format this list does not name prints as its +// enumerant, which is what makes an unexpected entry readable rather than +// silently anonymous. +const char* FormatName(VkFormat f) +{ + switch ((uint32_t)f) { + case VK_FORMAT_G8_B8R8_2PLANE_420_UNORM: return "NV12"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_420_UNORM_3PACK16: return "P010"; + case VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16: return "P012"; + case VK_FORMAT_G16_B16R16_2PLANE_420_UNORM: return "P016"; + case VK_FORMAT_G8_B8R8_2PLANE_422_UNORM: return "NV16"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_422_UNORM_3PACK16: return "P210"; + case VK_FORMAT_G8_B8R8_2PLANE_444_UNORM: return "NV24"; + case VK_FORMAT_G10X6_B10X6R10X6_2PLANE_444_UNORM_3PACK16: return "S410"; + case VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM: return "I420"; + case VK_FORMAT_G8_B8_R8_3PLANE_422_UNORM: return "I422"; + case VK_FORMAT_G8_B8_R8_3PLANE_444_UNORM: return "I444"; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16: + return "I420-10"; + case VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_444_UNORM_3PACK16: + return "I444-10"; + case VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16: + return "I420-12"; + case VK_FORMAT_R8G8B8A8_UNORM: return "RGBA8"; + case VK_FORMAT_B8G8R8A8_UNORM: return "BGRA8"; + case VK_FORMAT_A8B8G8R8_UNORM_PACK32: return "ABGR8"; + case VK_FORMAT_A2B10G10R10_UNORM_PACK32: return "A2BGR10"; + case VK_FORMAT_A2R10G10B10_UNORM_PACK32: return "A2RGB10"; + case VK_FORMAT_UNDEFINED: return "UNDEFINED"; + default: return "(other)"; + } +} + +const char* OptimalityName(VkVideoEncoderInputFormatOptimality o) +{ + return (o == VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) ? "OPTIMAL" + : "SUBOPTIMAL"; +} + +// One row of the sweep. |wantEncodeFormat| is checked only when the row is +// expected to succeed; VK_FORMAT_UNDEFINED means "do not check it". +struct Row { + VkVideoCodecOperationFlagBitsKHR codec; + const char* codecName; + uint32_t profile; + VkFormat format; + const char* formatName; + VkResult want; + VkFormat wantEncodeFormat; + VkVideoEncoderInputFormatOptimality wantOptimality; + const char* why; +}; + +// Ask the device, through the same enumeration surface a caller would use, +// whether |format| is encodable for (codec, profile), and hand back the +// advertised properties when it is. +// +// The 4:2:2 rows below used to assert a refusal unconditionally. That is not a +// library rule -- it is a per-device capability: Blackwell and later encode +// 4:2:2, earlier parts do not. Hardcoding the refusal makes this test report a +// hardware CAPABILITY as a library DEFECT on every part that has it. Derive the +// expectation instead, and take the expected properties from the enumerated +// entry, so the point query is still held to agreeing with the list. +static bool DeviceAdvertisesFormat(VulkanVideoEncoderContext* ctx, + uint32_t deviceIndex, + VkVideoCodecOperationFlagBitsKHR codec, + uint32_t profile, + VkFormat format, + VkVideoEncoderInputFormatProperties* pProps) +{ + uint32_t count = 0; + if (VkEncEnumerateInputFormats(ctx, deviceIndex, codec, profile, + &count, nullptr) != VK_SUCCESS) { + return false; + } + if (count == 0) { + return false; + } + VkVideoEncoderInputFormatProperties entries[64] = {}; + uint32_t fetched = (count < 64u) ? count : 64u; + const VkResult r = VkEncEnumerateInputFormats(ctx, deviceIndex, codec, + profile, &fetched, entries); + if ((r != VK_SUCCESS) && (r != VK_INCOMPLETE)) { + return false; + } + for (uint32_t i = 0; i < fetched; i++) { + if (entries[i].format == format) { + if (pProps != nullptr) { + *pProps = entries[i]; + } + return true; + } + } + return false; +} + +void RunRows(VulkanVideoEncoderContext* ctx, uint32_t deviceIndex, + const Row* rows, size_t count) +{ + for (size_t i = 0; i < count; i++) { + const Row& row = rows[i]; + VkVideoEncoderInputFormatProperties props = {}; + const VkResult r = VkEncQueryInputFormatSupport( + ctx, deviceIndex, row.codec, row.profile, row.format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props); + const std::string what = + std::string(row.codecName) + " profile " + U32(row.profile) + + " + " + row.formatName + " -- " + row.why; + Check(r == row.want, what, + std::string("got ") + ResultName(r) + ", wanted " + + ResultName(row.want)); + if ((r != VK_SUCCESS) || (row.want != VK_SUCCESS)) { + continue; + } + Check(props.format == row.format, + what + ": echoes the format asked about", + "got " + U32((uint32_t)props.format)); + if (row.wantEncodeFormat != VK_FORMAT_UNDEFINED) { + Check(props.encodeFormat == row.wantEncodeFormat, + what + ": names what the bitstream is coded from", + "got " + U32((uint32_t)props.encodeFormat)); + } + Check(props.optimality == row.wantOptimality, + what + ": reports the right optimality", + "got " + U32((uint32_t)props.optimality)); + } +} + +// --------------------------------------------------------------------------- +// THE CROSS-ROUTE SWEEP (--cross-route). +// +// Two entry points answer the same question by different routes: +// VkEncEnumerateInputFormats hands back a list, VkEncQueryInputFormatSupport +// answers a point. For one context, one device and one (codec, profile) they +// must agree, and the field that matters most is encodeFormat -- the header +// makes it load-bearing because it, and not `format`, is what decides the +// chroma subsampling and bit depth of the bitstream. +// +// WHY IT LIVES IN THIS BINARY. This file already calls both entry points +// against one context, one device and one bring-up. A second binary would +// duplicate the context creation, the device choice and the SKIP-77 harness to +// assert something neither half of the duplication owns. +// +// THE CALIBRATION IS THE OPTIMAL ARM AND IT IS NOT COSMETIC. On an OPTIMAL +// entry both surfaces set encodeFormat to the format itself, by construction, +// through code they do not share. So every OPTIMAL row must agree BEFORE any +// SUBOPTIMAL row is read: a sweep that is red on OPTIMAL rows too is not +// measuring a cross-route gap, it is measuring a broken harness, and the run +// has to be discarded rather than interpreted. +// +// THE SECOND CONTROL IS A REFUSAL. 4:2:2 and 12-bit must be on NO list, before +// or after any change here. Without it, "every format the caller hoped for is +// now advertised" cannot be told apart from "the enumerator became permissive". + +struct Key { + VkVideoCodecOperationFlagBitsKHR codec; + const char* codecName; + uint32_t profile; + const char* profileName; +}; + +// The eight PUBLISHABLE keys -- the (codec, profile) pairs the context probes +// and a caller can name. AV1 Main appears once: the context probes it at two +// depths but a profile number resolves to the first row, so 8-bit Main is the +// only AV1 answer any public entry point can return. +const Key kPublishableKeys[] = { + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, "66 Baseline" }, + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_H264_MAIN, "77 Main" }, + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_H264_HIGH, "100 High" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_H265_MAIN, "1 Main" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_H265_MAIN10, "2 Main 10" }, + { kAV1, "AV1", VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, "0 Main" }, +}; +const size_t kPublishableKeyCount = + sizeof(kPublishableKeys) / sizeof(kPublishableKeys[0]); + +// Wide enough for every routable format plus slack, so a list that grew is +// reported as it is rather than clipped into agreement. +enum { kMaxEntries = 64 }; + +struct KeyList { + VkVideoEncoderInputFormatProperties entries[kMaxEntries]; + uint32_t count; + VkResult result; + bool answered; +}; + +void FetchList(VulkanVideoEncoderContext* ctx, uint32_t deviceIndex, + const Key& key, KeyList& out) +{ + out.count = 0; + out.answered = false; + uint32_t count = 0; + out.result = VkEncEnumerateInputFormats(ctx, deviceIndex, key.codec, + key.profile, &count, nullptr); + if (out.result != VK_SUCCESS) { + return; + } + out.answered = true; + if (count == 0) { + return; + } + uint32_t fetched = (count < (uint32_t)kMaxEntries) ? count + : (uint32_t)kMaxEntries; + const VkResult r = + VkEncEnumerateInputFormats(ctx, deviceIndex, key.codec, key.profile, + &fetched, out.entries); + if ((r != VK_SUCCESS) && (r != VK_INCOMPLETE)) { + out.result = r; + out.answered = false; + return; + } + out.count = fetched; +} + +std::string KeyName(const Key& key) +{ + return std::string(key.codecName) + " " + key.profileName; +} + +// One advertised entry, put back through the point query. Three assertions, +// each its own Check because they fail for different reasons: acceptance, +// encodeFormat equality (the one the header makes load-bearing), and +// optimality. +void CrossCheckEntry(VulkanVideoEncoderContext* ctx, uint32_t deviceIndex, + const Key& key, + const VkVideoEncoderInputFormatProperties& entry) +{ + VkVideoEncoderInputFormatProperties props = {}; + const VkResult r = VkEncQueryInputFormatSupport( + ctx, deviceIndex, key.codec, key.profile, entry.format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props); + const std::string what = KeyName(key) + " + " + FormatName(entry.format) + + " [" + OptimalityName(entry.optimality) + "]"; + Check(r == VK_SUCCESS, what + ": an advertised format is accepted by the " + "point query", + std::string("got ") + ResultName(r)); + if (r != VK_SUCCESS) { + return; + } + Check(props.encodeFormat == entry.encodeFormat, + what + ": both routes name the same encodeFormat", + std::string("enumerator says ") + FormatName(entry.encodeFormat) + + " (" + U32((uint32_t)entry.encodeFormat) + "), point query says " + + FormatName(props.encodeFormat) + " (" + + U32((uint32_t)props.encodeFormat) + ")"); + Check(props.optimality == entry.optimality, + what + ": both routes name the same optimality", + std::string("enumerator says ") + OptimalityName(entry.optimality) + + ", point query says " + OptimalityName(props.optimality)); +} + +bool ListContains(const KeyList& list, VkFormat format) +{ + for (uint32_t i = 0; i < list.count; i++) { + if (list.entries[i].format == format) { + return true; + } + } + return false; +} + +int RunCrossRoute(VulkanVideoEncoderContext* ctx, uint32_t deviceIndex) +{ + std::printf("== cross-route agreement: enumeration vs point query ==\n"); + + static KeyList lists[kPublishableKeyCount]; + uint32_t answeredKeys = 0; + uint32_t optimalRows = 0; + uint32_t suboptimalRows = 0; + + std::printf("-- advertised lists, per publishable key --\n"); + for (size_t k = 0; k < kPublishableKeyCount; k++) { + FetchList(ctx, deviceIndex, kPublishableKeys[k], lists[k]); + if (!lists[k].answered) { + std::printf(" %-18s : no list (%s)\n", + KeyName(kPublishableKeys[k]).c_str(), + ResultName(lists[k].result)); + continue; + } + answeredKeys++; + uint32_t opt = 0; + uint32_t sub = 0; + for (uint32_t i = 0; i < lists[k].count; i++) { + if (lists[k].entries[i].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) { + opt++; + } else { + sub++; + } + } + optimalRows += opt; + suboptimalRows += sub; + std::printf(" %-18s : %u entries (%u OPTIMAL, %u SUBOPTIMAL) :", + KeyName(kPublishableKeys[k]).c_str(), lists[k].count, opt, + sub); + for (uint32_t i = 0; i < lists[k].count; i++) { + std::printf(" %s->%s", FormatName(lists[k].entries[i].format), + FormatName(lists[k].entries[i].encodeFormat)); + } + std::printf("\n"); + } + std::printf(" totals: %u keys answered, %u OPTIMAL rows, %u SUBOPTIMAL " + "rows\n", answeredKeys, optimalRows, suboptimalRows); + + Check(answeredKeys >= 2u, + "at least two publishable keys advertise a list, so the sweep has " + "something to read", + "answeredKeys=" + U32(answeredKeys)); + + // ---- The two-call idiom's stability, which is now a DIFFERENT claim ---- + // + // The header used to rest this list's stability on the context's snapshot + // being immutable. It is not read out of the snapshot any more -- it is + // recomputed live, per candidate, against the device, on every call -- + // so the rule survives for a different reason: the resolve is a + // deterministic function of (context, device, codec, profile). That is a + // published caller contract and nothing asserted it, which is exactly how + // a rationale outlives its mechanism. + // + // CALIBRATION FIRST. "Every call agrees" is worth nothing from an + // observable that reads the same for everything, so the counts are shown + // to VARY ACROSS KEYS before they are required to be STABLE ACROSS CALLS. + { + uint32_t distinctCounts = 0; + for (size_t k = 0; k < kPublishableKeyCount; k++) { + if (!lists[k].answered) { + continue; + } + bool seen = false; + for (size_t j = 0; j < k; j++) { + if (lists[j].answered && (lists[j].count == lists[k].count)) { + seen = true; + break; + } + } + if (!seen) { + distinctCounts++; + } + } + Check(distinctCounts >= 2u, + "calibration: the advertised counts differ across keys, so " + "'the same on every call' is an observation and not a constant", + "distinct counts " + U32(distinctCounts)); + + for (size_t k = 0; k < kPublishableKeyCount; k++) { + if (!lists[k].answered) { + continue; + } + const Key& key = kPublishableKeys[k]; + uint32_t countA = 0; + const VkResult rA = VkEncEnumerateInputFormats( + ctx, deviceIndex, key.codec, key.profile, &countA, nullptr); + uint32_t countB = 0; + const VkResult rB = VkEncEnumerateInputFormats( + ctx, deviceIndex, key.codec, key.profile, &countB, nullptr); + Check((rA == VK_SUCCESS) && (rB == VK_SUCCESS) && + (countA == countB) && (countA == lists[k].count), + KeyName(key) + ": two counting calls and the fetch agree, " + "on a list recomputed live each time", + "counts " + U32(countA) + ", " + U32(countB) + ", " + + U32(lists[k].count)); + + static VkVideoEncoderInputFormatProperties again[kMaxEntries]; + uint32_t fetched = (countA < (uint32_t)kMaxEntries) + ? countA : (uint32_t)kMaxEntries; + const VkResult rC = VkEncEnumerateInputFormats( + ctx, deviceIndex, key.codec, key.profile, &fetched, again); + bool identical = ((rC == VK_SUCCESS) || (rC == VK_INCOMPLETE)) && + (fetched == lists[k].count); + for (uint32_t i = 0; identical && (i < fetched); i++) { + identical = (again[i].format == lists[k].entries[i].format) && + (again[i].encodeFormat == + lists[k].entries[i].encodeFormat) && + (again[i].optimality == + lists[k].entries[i].optimality); + } + Check(identical, + KeyName(key) + ": and a second fetch is entry-for-entry the " + "same list, ordering included", + "fetched " + U32(fetched) + " of " + U32(lists[k].count)); + } + } + if (g_failures != 0) { + std::printf("\nNO LISTS -- nothing below can be read. checks: %d, " + "failures: %d\nRESULT: FAIL\n", g_checks, g_failures); + return 1; + } + + // ---- CONTROL 1. The OPTIMAL arm, where the two surfaces must agree by + // ---- construction. Read before any SUBOPTIMAL row. + std::printf("-- control 1: the OPTIMAL arm, which must already agree --\n"); + for (size_t k = 0; k < kPublishableKeyCount; k++) { + for (uint32_t i = 0; i < lists[k].count; i++) { + if (lists[k].entries[i].optimality != + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) { + continue; + } + CrossCheckEntry(ctx, deviceIndex, kPublishableKeys[k], + lists[k].entries[i]); + } + } + if (g_failures != 0) { + std::printf("\nCALIBRATION FAILED on the OPTIMAL arm -- this run is " + "measuring a broken harness and not a cross-route gap. " + "Discard it. checks: %d, failures: %d\nRESULT: FAIL\n", + g_checks, g_failures); + return 1; + } + + // ---- CONTROL 2. The refusal. 12-bit is refused by every part this + // ---- library targets, at every key including DEFAULT, so it belongs on no + // ---- list -- before or after any change to how the list is built, and it + // ---- is what keeps this control able to catch an enumerator that simply + // ---- became permissive. + // ---- + // ---- 4:2:2 is NOT such an absolute. Every NAMED profile here is 4:2:0 + // ---- only, so NV16 must be absent from those lists on any device. But + // ---- DEFAULT derives the profile from the input's own subsampling, and + // ---- whether 4:2:2 appears there is a per-device capability -- Blackwell + // ---- and later encode it. Asserting its absence made this control report + // ---- a hardware capability as an enumerator defect. + std::printf("-- control 2: what must be on no list at all --\n"); + for (size_t k = 0; k < kPublishableKeyCount; k++) { + if (!lists[k].answered) { + continue; + } + if (kPublishableKeys[k].profile != VK_VIDEO_ENCODER_PROFILE_DEFAULT) { + Check(!ListContains(lists[k], kNv16), + KeyName(kPublishableKeys[k]) + + ": NV16 (4:2:2) is not advertised, because this profile " + "is 4:2:0 only", + "it was"); + } else { + // Derive it: the list must say exactly what the point query says. + VkVideoEncoderInputFormatProperties props = {}; + const VkResult r = VkEncQueryInputFormatSupport( + ctx, deviceIndex, kPublishableKeys[k].codec, + kPublishableKeys[k].profile, kNv16, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props); + const bool listed = ListContains(lists[k], kNv16); + Check(listed == (r == VK_SUCCESS), + KeyName(kPublishableKeys[k]) + + ": NV16 (4:2:2) is listed exactly when the point query " + "accepts it", + std::string("the list says ") + (listed ? "yes" : "no") + + " and the point query says " + ResultName(r)); + } + Check(!ListContains(lists[k], kP012), + KeyName(kPublishableKeys[k]) + ": P012 (12-bit) is not advertised", + "it was"); + } + if (g_failures != 0) { + std::printf("\nREFUSAL CONTROL FAILED -- the enumerator is advertising " + "what the point query refuses, so no presence assertion " + "below distinguishes accurate from permissive. checks: %d, " + "failures: %d\nRESULT: FAIL\n", g_checks, g_failures); + return 1; + } + + // ---- THE ASSERTION. The SUBOPTIMAL arm, where the enumerator computes + // ---- encodeFormat from the DEVICE's list and the point query computes it + // ---- from the SESSION's own bound configuration. + std::printf("-- cross-route equality on the SUBOPTIMAL arm --\n"); + for (size_t k = 0; k < kPublishableKeyCount; k++) { + for (uint32_t i = 0; i < lists[k].count; i++) { + if (lists[k].entries[i].optimality == + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL) { + continue; + } + CrossCheckEntry(ctx, deviceIndex, kPublishableKeys[k], + lists[k].entries[i]); + } + } + + // ---- PRESENCE. A different observable from equality, so it is asserted + // ---- separately: these are formats this library encodes on this device + // ---- and the point query accepts, which no advertised list reported. + std::printf("-- presence: what the library encodes and must advertise --\n"); + struct Presence { + size_t keyIndex; + VkFormat format; + const char* why; + }; + static const Presence kPresence[] = { + { 0u, kNv24, "8-bit 4:4:4 derives H.264 High 4:4:4 Predictive" }, + { 4u, kNv24, "8-bit 4:4:4 derives H.265 Range Extensions" }, + { 4u, kS410, "10-bit 4:4:4 derives H.265 Range Extensions" }, + { 4u, kP010, "10-bit 4:2:0 derives H.265 Main 10" }, + { 7u, kP010, "10-bit 4:2:0 is AV1 Main at ten bits" }, + }; + for (size_t i = 0; i < sizeof(kPresence) / sizeof(kPresence[0]); i++) { + const Presence& p = kPresence[i]; + if (!lists[p.keyIndex].answered) { + std::printf(" (skipped: %s has no list on this device)\n", + KeyName(kPublishableKeys[p.keyIndex]).c_str()); + continue; + } + Check(ListContains(lists[p.keyIndex], p.format), + KeyName(kPublishableKeys[p.keyIndex]) + " advertises " + + FormatName(p.format) + " -- " + p.why, + "it does not"); + } + + std::printf("\nchecks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} + + +// --------------------------------------------------------------------------- +// THE INITIALISATION GATE (--init-gate). +// +// WHAT THIS ASSERTS, AND IT IS ONE SENTENCE: what InitializeExt accepts is what +// VkEncEnumerateInputFormats advertises. Not a superset of it. +// +// WHY THAT IS NOT A RESTATEMENT OF --cross-route. That sweep puts the two +// PRE-SESSION surfaces against each other -- the enumerator and the point query +// -- and both of them are answers ABOUT a session. This one puts the answer +// against the SESSION ITSELF. A library whose two queries agreed perfectly with +// each other and disagreed with what init does would pass --cross-route on +// every row and fail every row here. +// +// THE TWO HALVES ARE DIFFERENT PROPOSITIONS AND FAIL FOR DIFFERENT REASONS, so +// they are asserted separately and in this order: +// +// 1. EVERY ADVERTISED FORMAT STILL INITIALISES. This is the negative control +// and it runs first, because a gate that refused everything would satisfy +// half 2 completely. It is also the half that costs a real session per +// entry, which is the price of asserting the proposition rather than a +// proxy for it. +// 2. A FORMAT THE DEVICE REFUSES IS REFUSED AT InitializeExt. 4:2:2 and +// 12-bit are on no advertised list and the point query refuses them at +// every profile, so they are the pairs where "advertised" and "accepted" +// used to differ: the binder took them, the session was created, and the +// driver refused at video-session creation -- with a message naming +// neither the format nor its subsampling. +// +// AND THE CALIBRATION COMES BEFORE BOTH. NV12 must initialise. An instrument +// that could not create a session at all would report every row of half 1 as a +// failure and every row of half 2 as a pass, and the second reading is the +// dangerous one: it looks exactly like a working gate. +// +// THE REFUSAL CODE IS THE POINT QUERY'S. VK_ERROR_FORMAT_NOT_SUPPORTED is what +// VkEncQueryInputFormatSupport returns for a pair this device will not take, so +// the session boundary returning anything else would be a second vocabulary for +// one verdict. + +struct InitVerdict { + VkResult result; + bool created; +}; + +// One session, built exactly as a shipping caller builds one for that format, +// and destroyed before the next: the sweep walks tens of formats and a process +// that held them all open would measure the driver's session limit instead of +// the library's acceptance. +InitVerdict RunInit(VkVideoCodecOperationFlagBitsKHR codec, uint32_t profile, + VkFormat format) +{ + InitVerdict v = { VK_ERROR_INITIALIZATION_FAILED, false }; + VkSharedBaseObj enc; + if ((CreateVulkanVideoEncoderExt(enc) != VK_SUCCESS) || !enc) { + return v; + } + v.created = true; + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = codec; + config.profile = profile; + config.encodeWidth = 1920; + config.encodeHeight = 1080; + config.inputFormat = format; + config.inputWidth = 1920; + config.inputHeight = 1080; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 4000000; + config.maxBitrate = 4000000; + config.gopLength = 30; + config.idrPeriod = 30; + config.frameRateNum = 30; + config.frameRateDen = 1; + config.deviceId = -1; + config.disableFileOutput = VK_TRUE; + v.result = enc->InitializeExt(config); + return v; +} + +// The keys the gate sweeps. DEFAULT on each codec, which is what a caller that +// has not named a profile gets and the only key whose derivation reaches the +// 4:4:4 and 10-bit profiles at all. +const Key kInitGateKeys[] = { + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, "DEFAULT" }, + { kAV1, "AV1", VK_VIDEO_ENCODER_PROFILE_AV1_MAIN, "0 Main" }, +}; +const size_t kInitGateKeyCount = + sizeof(kInitGateKeys) / sizeof(kInitGateKeys[0]); + +// What the DEVICE refuses, on both measured architectures, at every profile: +// the 4:2:2 family and the 12-bit family. These are exactly the formats the +// library ROUTES -- the taxonomy admits them and the binder builds a config for +// them -- and that the device then will not encode. They are on no advertised +// list, which --cross-route's control 2 already asserts for two of them; what +// is asserted here is that InitializeExt agrees. +struct RefusalRow { + VkFormat format; + const char* name; + const char* why; +}; +const RefusalRow kMustRefuse[] = { + { VK_FORMAT_G8_B8R8_2PLANE_422_UNORM, "NV16", + "8-bit 4:2:2, semi-planar" }, + { VK_FORMAT_G8_B8_R8_3PLANE_422_UNORM, "I422", + "8-bit 4:2:2, 3-plane, converted rung" }, + { VK_FORMAT_G10X6_B10X6R10X6_2PLANE_422_UNORM_3PACK16, "P210", + "10-bit 4:2:2, semi-planar" }, + { VK_FORMAT_G12X4_B12X4R12X4_2PLANE_420_UNORM_3PACK16, "P012", + "12-bit 4:2:0, semi-planar" }, + { VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, "I420-12", + "12-bit 4:2:0, 3-plane, converted rung" }, + { VK_FORMAT_G12X4_B12X4R12X4_2PLANE_444_UNORM_3PACK16, "S412", + "12-bit 4:4:4, semi-planar" }, +}; +const size_t kMustRefuseCount = + sizeof(kMustRefuse) / sizeof(kMustRefuse[0]); + +int RunInitGate(VulkanVideoEncoderContext* ctx, uint32_t deviceIndex) +{ + std::printf("== initialisation gate: accepted == advertised ==\n"); + + // ---- CALIBRATION. One session on the format every encoder takes. Read + // ---- before anything below, because a harness that can create no session + // ---- reports every refusal assertion as a pass. + std::printf("-- calibration: a session the device certainly takes --\n"); + { + const InitVerdict v = + RunInit(kH264, VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv12); + Check(v.created, "an encoder object is creatable at all", "it is not"); + Check(v.result == VK_SUCCESS, + "H.264 DEFAULT + NV12 initialises -- the instrument can produce " + "an ACCEPTANCE, not only a refusal", + std::string("got ") + ResultName(v.result)); + } + if (g_failures != 0) { + std::printf("\nCALIBRATION FAILED -- this harness cannot create a " + "session, so every refusal below would pass for the wrong " + "reason. checks: %d, failures: %d\nRESULT: FAIL\n", + g_checks, g_failures); + return 1; + } + + // ---- HALF 1, THE NEGATIVE CONTROL. Every advertised entry, at the key it + // ---- was advertised for, must still initialise. + std::printf("-- half 1 (negative control): every ADVERTISED format still " + "initialises --\n"); + static KeyList lists[kInitGateKeyCount]; + uint32_t swept = 0; + for (size_t k = 0; k < kInitGateKeyCount; k++) { + FetchList(ctx, deviceIndex, kInitGateKeys[k], lists[k]); + if (!lists[k].answered) { + std::printf(" %-14s : no list on this device (%s) -- skipped\n", + KeyName(kInitGateKeys[k]).c_str(), + ResultName(lists[k].result)); + continue; + } + for (uint32_t i = 0; i < lists[k].count; i++) { + const VkFormat f = lists[k].entries[i].format; + const InitVerdict v = + RunInit(kInitGateKeys[k].codec, kInitGateKeys[k].profile, f); + swept++; + Check(v.result == VK_SUCCESS, + KeyName(kInitGateKeys[k]) + " + " + FormatName(f) + " [" + + OptimalityName(lists[k].entries[i].optimality) + + "]: an ADVERTISED format still initialises", + std::string("got ") + ResultName(v.result)); + } + } + std::printf(" %u advertised entries put through InitializeExt\n", swept); + Check(swept >= 10u, + "the negative control swept a list worth reading", + "only " + U32(swept) + " entries"); + if (g_failures != 0) { + std::printf("\nNEGATIVE CONTROL FAILED -- acceptance has been narrowed " + "past the advertised set, which is worse than the gap it " + "was closing. checks: %d, failures: %d\nRESULT: FAIL\n", + g_checks, g_failures); + return 1; + } + + // ---- HALF 2, THE ASSERTION. What the device refuses is refused here, + // ---- with the same code the point query gives it. + std::printf("-- half 2: a DEVICE-REFUSED format is refused at " + "InitializeExt --\n"); + for (size_t k = 0; k < kInitGateKeyCount; k++) { + if (!lists[k].answered) { + continue; + } + for (size_t r = 0; r < kMustRefuseCount; r++) { + const RefusalRow& row = kMustRefuse[r]; + const std::string what = KeyName(kInitGateKeys[k]) + " + " + + row.name + " (" + row.why + ")"; + // The point query first, so the row is only read as evidence about + // InitializeExt when the pre-session surfaces already say no. A + // device that DID encode 4:2:2 would make this row vacuous rather + // than false, and it would say so here. + const VkResult q = VkEncQueryInputFormatSupport( + ctx, deviceIndex, kInitGateKeys[k].codec, + kInitGateKeys[k].profile, row.format, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, nullptr); + if (q == VK_SUCCESS) { + std::printf(" (vacuous: %s is ACCEPTED by the point query on " + "this device, so it is not a refusal row here)\n", + what.c_str()); + continue; + } + Check(!ListContains(lists[k], row.format), + what + ": is on no advertised list", "it was advertised"); + std::printf(" [InitializeExt refusal message follows for %s]\n", + what.c_str()); + std::fflush(stdout); + const InitVerdict v = + RunInit(kInitGateKeys[k].codec, kInitGateKeys[k].profile, + row.format); + Check(v.result != VK_SUCCESS, + what + ": InitializeExt REFUSES what no list advertises", + "InitializeExt returned VK_SUCCESS -- the accepted set is " + "wider than the advertised one"); + Check(v.result == VK_ERROR_FORMAT_NOT_SUPPORTED, + what + ": refused with the point query's own code", + std::string("got ") + ResultName(v.result) + + ", wanted VK_ERROR_FORMAT_NOT_SUPPORTED"); + } + } + + std::printf("\nchecks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} + +} // namespace + +int main(int argc, char** argv) +{ + // Two flags, following the sibling binaries' pattern: --cross-route runs + // the enumeration-versus-point-query sweep and --init-gate the + // advertisement-versus-InitializeExt sweep, each and nothing else, each + // registered as its own ctest row. Without either the behaviour is + // unchanged. + bool crossRoute = false; + bool initGate = false; + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--cross-route") == 0) { + crossRoute = true; + } else if (std::strcmp(argv[i], "--init-gate") == 0) { + initGate = true; + } + } + + std::printf("== pre-session input-format point query ==\n"); + + VkVideoEncoderContextCreateInfo ci = {}; + ci.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONTEXT_CREATE_INFO; + ci.mode = VK_VIDEO_ENCODER_CONTEXT_MODE_OWN; + + VkSharedBaseObj context; + if (CreateVulkanVideoEncoderContext(&ci, context) != VK_SUCCESS) { + std::printf("SKIP: no Vulkan encoder context could be created\n"); + return 77; + } + VulkanVideoEncoderContext* const ctx = context.get(); + + // ---- Device identity, printed for every device and asserted for the one + // ---- used. A capability answer off llvmpipe is not a capability answer. + const uint32_t deviceCount = VkEncGetPhysicalDeviceCount(ctx); + std::printf(" devices enumerated: %u\n", deviceCount); + + uint32_t chosen = UINT32_MAX; + for (uint32_t i = 0; i < deviceCount; i++) { + VkVideoEncoderDeviceIdentity id = {}; + id.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_DEVICE_IDENTITY; + if (VkEncGetPhysicalDeviceIdentity(ctx, i, &id) != VK_SUCCESS) { + continue; + } + std::printf(" device[%u]: name='%s' vendorID=0x%04x deviceID=0x%04x\n", + i, id.deviceName, id.vendorID, id.deviceID); + // 0x10DE is NVIDIA. Mesa's llvmpipe reports 0x10005 and is excluded + // by the same test that would exclude any other vendor's device. + if ((chosen == UINT32_MAX) && (id.vendorID == 0x10DEu)) { + // Only if it can actually answer for an encode profile: a + // display-only NVIDIA device would pass the vendor test and + // report nothing. + VkVideoEncoderCapabilities caps = {}; + caps.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CAPABILITIES; + if (VkEncGetEncodeCapabilities(ctx, i, kH264, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, + &caps) == VK_SUCCESS) { + chosen = i; + } + } + } + if (chosen == UINT32_MAX) { + std::printf("SKIP: no encode-capable NVIDIA device enumerated\n"); + return 77; + } + std::printf(" using device[%u]\n", chosen); + + if (crossRoute) { + return RunCrossRoute(ctx, chosen); + } + if (initGate) { + return RunInitGate(ctx, chosen); + } + + // ---- 1. CALIBRATION. Known-good 4:2:0 rows, run before anything else is + // ---- believed. If these do not say yes, no refusal below means anything. + std::printf("-- calibration: known-good 4:2:0 --\n"); + static const Row kCalibration[] = { + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv12, "NV12", + VK_SUCCESS, kNv12, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "8-bit 4:2:0, the format every encoder takes" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv12, "NV12", + VK_SUCCESS, kNv12, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "8-bit 4:2:0" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kP010, "P010", + VK_SUCCESS, kP010, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "10-bit 4:2:0 derives Main 10" }, + }; + RunRows(ctx, chosen, kCalibration, + sizeof(kCalibration) / sizeof(kCalibration[0])); + if (g_failures != 0) { + std::printf("\nCALIBRATION FAILED -- nothing below this line can be " + "read. checks: %d, failures: %d\nRESULT: FAIL\n", + g_checks, g_failures); + return 1; + } + + // ---- 2. THE ACCEPTANCE CASE. 4:4:4, which no enumeration reports. + std::printf("-- acceptance: 4:4:4, accepted and previously " + "undiscoverable --\n"); + static const Row kAcceptance[] = { + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv24, "NV24", + VK_SUCCESS, kNv24, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "8-bit 4:4:4 derives High 4:4:4 Predictive (244)" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kS410, "S410", + VK_SUCCESS, kS410, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "10-bit 4:4:4 derives Range Extensions (4)" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv24, "NV24", + VK_SUCCESS, kNv24, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "8-bit 4:4:4 derives Range Extensions (4)" }, + }; + RunRows(ctx, chosen, kAcceptance, + sizeof(kAcceptance) / sizeof(kAcceptance[0])); + + // ---- 3. THE GAP IS CLOSED, MEASURED. The same NV24 the point query just + // ---- accepted is on the advertised list too. + // ---- + // ---- THE INVERSE OF THIS ASSERTION -- "NV24 is NOT on the advertised + // ---- H.264 list" -- holds only when the enumerator builds its + // ---- answer from a capability snapshot that probes every profile at + // ---- 4:2:0, so no 4:4:4 encode source could reach any advertised list on + // ---- any device however capable. Both surfaces now resolve through one + // ---- function, so a format this library encodes on this device is + // ---- discoverable before a session exists, which is what the entry point + // ---- was for. + // ---- + // ---- MEMBERSHIP ONLY, HERE. That the two surfaces AGREE -- same + // ---- encodeFormat, same optimality, for every entry of every publishable + // ---- key -- is a different proposition and is asserted in --cross-route, + // ---- where it has the calibration it needs. + std::printf("-- the gap this closes --\n"); + { + uint32_t count = 0; + VkResult r = VkEncEnumerateInputFormats( + ctx, chosen, kH264, VK_VIDEO_ENCODER_PROFILE_DEFAULT, &count, + nullptr); + bool advertised = false; + if ((r == VK_SUCCESS) && (count > 0)) { + VkVideoEncoderInputFormatProperties entries[64] = {}; + uint32_t fetched = (count < 64u) ? count : 64u; + r = VkEncEnumerateInputFormats(ctx, chosen, kH264, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, + &fetched, entries); + if ((r == VK_SUCCESS) || (r == VK_INCOMPLETE)) { + for (uint32_t i = 0; i < fetched; i++) { + if (entries[i].format == kNv24) { + advertised = true; + } + } + } + } + std::printf(" VkEncEnumerateInputFormats(H.264, DEFAULT) advertises " + "%u formats\n", count); + Check(advertised, + "NV24 IS on the advertised H.264 list, so what the point query " + "accepts is discoverable without asking about it by name", + "NV24 was not advertised"); + } + + // ---- 4. REFUSALS THE DEVICE OWNS. The sweep on this part refuses 4:2:2 + // ---- at every profile and 12-bit at every profile. + std::printf("-- refusals the device owns --\n"); + // 12-bit is refused by every part this library targets, so it stays a + // fixed expectation. + static const Row kDeviceRefusals[] = { + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_DEFAULT, kP012, "P012", + VK_ERROR_FORMAT_NOT_SUPPORTED, VK_FORMAT_UNDEFINED, + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "12-bit, which this device does not encode at any profile" }, + }; + RunRows(ctx, chosen, kDeviceRefusals, + sizeof(kDeviceRefusals) / sizeof(kDeviceRefusals[0])); + + // 4:2:2 is a per-device capability. Ask, then assert what the answer + // obliges: a refusal where the device has no 4:2:2, and agreement with the + // advertised entry where it does. + { + static const struct { + VkVideoCodecOperationFlagBitsKHR codec; + const char* codecName; + const char* refusedWhy; + const char* acceptedWhy; + } k422Rows[] = { + { kH264, "H.264", + "4:2:2 derives High 4:2:2 (122), which this device does not encode", + "4:2:2 derives High 4:2:2 (122), which this device encodes, and " + "the point query agrees with the advertised list" }, + { kH265, "H.265", + "4:2:2 at Range Extensions, which this device does not encode", + "4:2:2 at Range Extensions, which this device encodes, and the " + "point query agrees with the advertised list" }, + }; + for (size_t i = 0; i < sizeof(k422Rows) / sizeof(k422Rows[0]); i++) { + VkVideoEncoderInputFormatProperties advertised = {}; + const bool supported = DeviceAdvertisesFormat( + ctx, chosen, k422Rows[i].codec, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv16, &advertised); + const Row row = { + k422Rows[i].codec, k422Rows[i].codecName, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv16, "NV16", + supported ? VK_SUCCESS : VK_ERROR_FORMAT_NOT_SUPPORTED, + supported ? advertised.encodeFormat : VK_FORMAT_UNDEFINED, + supported ? advertised.optimality + : VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + supported ? k422Rows[i].acceptedWhy : k422Rows[i].refusedWhy + }; + RunRows(ctx, chosen, &row, 1); + } + } + + // ---- 5. REFUSALS THE LIBRARY OWNS, and the negative control that says + // ---- the guard discriminates rather than refusing everything. + std::printf("-- refusals the library owns, and the negative control --\n"); + static const Row kLibraryRows[] = { + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_H264_HIGH, kNv24, "NV24", + VK_ERROR_FORMAT_NOT_SUPPORTED, VK_FORMAT_UNDEFINED, + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "High (100) is a 4:2:0 profile and cannot carry 4:4:4" }, + { kH265, "H.265", VK_VIDEO_ENCODER_PROFILE_H265_MAIN, kNv24, "NV24", + VK_ERROR_FORMAT_NOT_SUPPORTED, VK_FORMAT_UNDEFINED, + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "Main (1) is a 4:2:0 profile and cannot carry 4:4:4" }, + // EXPRESSIBLE, DERIVABLE, AND NOW BINDABLE. 244 is the profile the + // DEFAULT derivation reaches for this very input and this device + // encodes it, so naming it explicitly must not be refused. This query + // answers what the binder + // answers. The bindable set has widened to every number the standard's + // limits table states, so the two now agree: what the derivation + // selects, a caller may also name. What the DEVICE will encode is a + // separate question and the rows above and below are where it is put. + { kH264, "H.264", + (uint32_t)VK_VIDEO_ENCODER_PROFILE_H264_HIGH_444_PREDICTIVE, + kNv24, "NV24", + VK_SUCCESS, kNv24, + VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "244 is expressible, derivable and now bindable" }, + // THE NEGATIVE CONTROL. A guard that refused every explicitly named + // profile would pass all three rows above. This one requires a yes. + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_H264_HIGH, kNv12, "NV12", + VK_SUCCESS, kNv12, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "High (100) over 4:2:0 is still accepted -- the guard " + "discriminates" }, + { kH264, "H.264", VK_VIDEO_ENCODER_PROFILE_H264_BASELINE, kNv12, + "NV12", VK_SUCCESS, kNv12, VK_VIDEO_ENCODER_INPUT_FORMAT_OPTIMAL, + "Baseline (66) over 4:2:0 is still accepted" }, + }; + RunRows(ctx, chosen, kLibraryRows, + sizeof(kLibraryRows) / sizeof(kLibraryRows[0])); + + // ---- 6. ARGUMENT GATES, and the optional out-parameter. + std::printf("-- argument gates --\n"); + { + VkVideoEncoderInputFormatProperties props = {}; + Check(VkEncQueryInputFormatSupport( + nullptr, 0u, kH264, VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv12, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props) == + VK_ERROR_INITIALIZATION_FAILED, + "a null context is VK_ERROR_INITIALIZATION_FAILED", ""); + Check(VkEncQueryInputFormatSupport( + ctx, deviceCount + 7u, kH264, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv12, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props) == + VK_ERROR_INITIALIZATION_FAILED, + "an out-of-range deviceIndex is VK_ERROR_INITIALIZATION_FAILED", + ""); + Check(VkEncQueryInputFormatSupport( + ctx, chosen, VK_VIDEO_CODEC_OPERATION_DECODE_H264_BIT_KHR, + VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv12, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props) == + VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR, + "a decode codec is VK_ERROR_VIDEO_PROFILE_CODEC_NOT_SUPPORTED_KHR", + ""); + // The out-parameter is optional and the verdict must not depend on it. + const VkResult withOut = VkEncQueryInputFormatSupport( + ctx, chosen, kH264, VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv24, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, &props); + const VkResult withoutOut = VkEncQueryInputFormatSupport( + ctx, chosen, kH264, VK_VIDEO_ENCODER_PROFILE_DEFAULT, kNv24, + VK_VIDEO_ENCODER_COLOR_MODEL_FROM_FORMAT, nullptr); + Check(withOut == withoutOut, + "a NULL pProperties gives the same verdict", + std::string("with ") + ResultName(withOut) + ", without " + + ResultName(withoutOut)); + } + + std::printf("\nchecks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-input-residency/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-input-residency/CMakeLists.txt new file mode 100644 index 00000000..141c7212 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-input-residency/CMakeLists.txt @@ -0,0 +1,306 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_input_residency_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# The PUBLIC API only, like the sibling real-device tests. This suite's whole +# claim is about what an EMBEDDER can observe through the shipped header -- +# reaching an internal seam would let it assert on a fact no caller can see, +# which is the failure mode it exists to close. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# CTest semantics, matching the siblings: 0 every assertion held, 1 an +# assertion failed, 77 no encode-capable GPU and therefore nothing proved +# either way -- reported as SKIPPED, never as a pass. +enable_testing() + +configure_file(vk_layer_settings.txt + ${CMAKE_CURRENT_BINARY_DIR}/vk_layer_settings.txt COPYONLY) + +# -------------------------------------------------------------------------- +# VALIDATION GATING. Same wrapper, same cache variables and same rationale as +# encoder-ext-adopt-device; validation_gate.cmake and vk_layer_settings.txt +# are copied verbatim from that suite (both are self-contained). +# +# WHY THIS IS A NEW BINARY AND NOT A SIXTH ARM ON encoder-ext-adopt-device: +# +# 1. ATTRIBUTION. That suite's subject is DEVICE ADOPTION and its own CMake +# argues that the control/subject split is what makes a red arm +# attributable. A residency arm inside it would be red for a reason the +# suite's name denies. +# 2. THE PINNED BASELINE. encoder_ext_adopt_device_test --own-validate is the +# number the CF-02 work is pinned to (46 -> 0). New arms in the same +# binary invite conflation of counts. +# 3. DIFFERENT FLOOR. This binary's validation floor is NOT zero and cannot +# be -- see below. Mixing it into a suite whose floor IS zero would force +# one of the two numbers to be explained away. +# +# THE FLOOR IS ZERO, AND THAT IS NOT WHAT THE REASONING BELOW PREDICTS. +# +# This suite was designed expecting a non-zero validation floor. The reasoning +# was sound as far as it went: on an OS-handle registration frame 1's acquire +# names a layout the image is not in, because the library creates the imported +# VkImage with initialLayout = UNDEFINED (forced by +# VUID-VkImageCreateInfo-pNext-01443), performs NO transition at registration +# (the public header: the declaration "must ALREADY be true when the first +# frame is submitted"), and then overwrites the one truthful declaration in +# VkVideoEncoder.cpp's `srcOldLayout = VK_IMAGE_LAYOUT_GENERAL; // Fallback +# for compute output`. The prediction was 2 messages per arm, one per NV12 +# plane. +# +# The prediction is wrong: the layer reports nothing on any of the five +# arms, whether or not the residency declaration is honoured. +# +# THAT ZERO IS NOT A BLIND HARNESS. A temporary probe arm declaring +# VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL against a TRANSFER_SRC-only image +# raises VUID-VkImageMemoryBarrier2-oldLayout-01213 out of +# vkCmdPipelineBarrier2KHR on the library's own device, captured in this +# harness's own output. So the layer is loaded, it inspects this workload's +# barriers, and its messages reach the gate. +# +# WHAT THAT PROBE DOES NOT ESTABLISH, stated so the zero is not over-read: it +# fired a USAGE-compatibility VUID, not the layout-TRACKING VUID (01197). So +# "0 messages" is what this layer build reports; it is NOT proof that a +# declared-vs-actual layout mismatch would be reported here. Do not convert +# this zero into "the declarations are all true" -- see the mirror arm below, +# where the declaration is measurably false on frames 2..8 and still reports +# nothing. +# +# NOTHING IS ALLOWLISTED. VVS_VALIDATION_ALLOW_VUIDS keeps its inherited +# default and the frame-1 VUID is deliberately absent from it: allowlisting a +# real defect to make a number look better is the one thing this project's +# rules forbid outright. The suite is carried by the COUNTER assertions, which +# need no layer at all -- which is the point, since the property under test is +# invisible to the layer either way. +# -------------------------------------------------------------------------- +set(VVS_VALIDATION_ALLOW_VUIDS "unknown VkStructureType" CACHE STRING + "Regex matched against a validation message BODY; matches are tolerated") +option(VVS_REQUIRE_VALIDATION_LAYER + "Fail, rather than skip, the validation arms when no layer is installed" + OFF) +set(VVS_VALIDATION_LAYER_PATH "" CACHE PATH + "Directory holding a validation-layer manifest; becomes the tests' VK_LAYER_PATH") +set(VVS_VALIDATION_LAYER_LIBDIR "" CACHE PATH + "Directory added to the tests' LD_LIBRARY_PATH.") + +set(_vvs_test_env + "VK_LAYER_SETTINGS_PATH=${CMAKE_CURRENT_BINARY_DIR}/vk_layer_settings.txt") +if(VVS_VALIDATION_LAYER_PATH) + list(APPEND _vvs_test_env "VK_LAYER_PATH=${VVS_VALIDATION_LAYER_PATH}") +endif() +if(VVS_VALIDATION_LAYER_LIBDIR) + list(APPEND _vvs_test_env "LD_LIBRARY_PATH=${VVS_VALIDATION_LAYER_LIBDIR}") +endif() + +# The COUNTER entries. These are the suite's load-bearing tests and they need +# no validation layer, which is deliberate: the property under test is +# invisible to the layer (both barrier programs are spec-clean), so tying it +# to a layer would make it skip on exactly the fleet it guards. +function(vvs_add_counter_test _name) + add_test(NAME ${_name} COMMAND ${PROJECT_NAME} ${ARGN}) + set_tests_properties(${_name} PROPERTIES + LABELS "gpu" + TIMEOUT 900 + SKIP_RETURN_CODE 77) +endfunction() + +function(vvs_add_gated_test _name) + # SPACE-separated, not ${ARGN} directly. validation_gate.cmake documents + # TEST_ARGS as "ONE argument or a space-separated string" and splits it with + # separate_arguments(NATIVE_COMMAND), which splits on WHITESPACE. ${ARGN} is + # a CMake list, so it interpolates SEMICOLON-separated, and add_test records + # the whole thing as one argv token: "-DTEST_ARGS=--arm;--validate". With no + # whitespace in it the split was a no-op and the test binary got a single + # unrecognised argument, printed its usage and exited 1 -- so every gated + # test taking TWO arguments never executed a line of its subject. That was + # both of them, and they had been red long enough to be filed as expected. + string(JOIN " " _gated_args ${ARGN}) + add_test(NAME ${_name} + COMMAND ${CMAKE_COMMAND} + -DTEST_EXE=$ + "-DTEST_ARGS=${_gated_args}" + -DREQUIRE_LAYER=${VVS_REQUIRE_VALIDATION_LAYER} + "-DALLOW_VUIDS=${VVS_VALIDATION_ALLOW_VUIDS}" + "-DGATE_SUMMARY=${CMAKE_BINARY_DIR}/validation_gate_summary.txt" + -P ${CMAKE_CURRENT_SOURCE_DIR}/validation_gate.cmake) + set_tests_properties(${_name} PROPERTIES + LABELS "gpu" + TIMEOUT 900 + SKIP_REGULAR_EXPRESSION "VALIDATION_GATE_RESULT=SKIP" + ENVIRONMENT "${_vvs_test_env}") +endfunction() + +# --------------------------------------------------------------------------- +# THE SUBJECT. An OPAQUE_FD registration -- self-exported from the library's +# own VkDevice and re-imported into it, which is exactly Chromium's shipping +# CPU staging tier -- declaring VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL. +# The declaration must be HONOURED: no queue-family acquire, and the local +# HOST|TRANSFER availability barrier instead, on all 8 reused frames. +# +# THIS IS THE ARM THAT WAS RED. Before the residency fix it reported +# foreignAcquireCount = 8 / localAcquireCount = 0. +# --------------------------------------------------------------------------- +vvs_add_counter_test(EncoderExtInputResidencyLocalHonouredOnOpaqueFd + --local-opaque-fd) + +# --------------------------------------------------------------------------- +# THE CONTROL, and it is what makes a red subject attributable. The IDENTICAL +# VkImage -- same create info, same exportable dedicated allocation, same +# persistent mapping, same 8-frame reuse -- registered as VK_IMAGE, the one +# handle type where residency was already honoured. Subject-red with this +# green isolates handleType and nothing else: not the host, not the format, +# not LINEAR tiling, not reuse, not the self-import. +# --------------------------------------------------------------------------- +vvs_add_counter_test(EncoderExtInputResidencyLocalControlOnVkImage + --local-vk-image) + +# --------------------------------------------------------------------------- +# THE NEGATIVE CONTROL. An explicit RESIDENCY_FOREIGN on the same OPAQUE_FD +# shape must STILL take the queue-family acquire. Without this entry the fix +# could degenerate into "stop taking the foreign acquire", which would turn +# the subject green while silently breaking every genuine import. +# --------------------------------------------------------------------------- +vvs_add_counter_test(EncoderExtInputResidencyForeignStillAcquires + --foreign-opaque-fd) + +# --------------------------------------------------------------------------- +# THE PIN, and it guards two live consumers by name. residency left ZERO +# (AUTO) on an OS handle must still DERIVE FOREIGN. The Wayland/GBM zero-copy +# lane is host-written and correctly FOREIGN -- its exporter is GBM/DRM -- and +# the out-of-tree Vulkan renderer app never declares residency at all. Both +# inherit AUTO, and a fix that generalised to "honour LOCAL, and while we are +# here stop deriving FOREIGN" would break both without this entry noticing. +# --------------------------------------------------------------------------- +vvs_add_counter_test(EncoderExtInputResidencyAutoStillDerivesForeign + --auto-opaque-fd) + +# --------------------------------------------------------------------------- +# THE CHROMIUM MIRROR. This arm declares VK_IMAGE_LAYOUT_PREINITIALIZED as +# both the registration default AND the per-frame currentLayout, which is what +# media/gpu/vulkan/vulkan_video_encode_accelerator.cc did before the UNDEFINED +# sentinel change -- OPAQUE_FD, RESIDENCY_LOCAL, one reused host-written +# staging image. +# +# IT PASSES, AND IT WAS EXPECTED TO FAIL. The design registered it WILL_FAIL +# on the reasoning below; the hardware said otherwise and the registration was +# corrected rather than the measurement explained away. +# +# THE REASONING, WHICH IS STILL CORRECT ABOUT THE CODE: a non-UNDEFINED +# per-frame currentLayout is an explicit statement of fact that outranks the +# library's own record, so the residual-layout mechanism is bypassed on every +# frame and the caller re-declares PREINITIALIZED each time -- while the +# library's own handback left the image in GENERAL, because PREINITIALIZED is +# not a legal barrier destination +# (VUID-VkImageMemoryBarrier2-newLayout-01198) and RestoreStagedInputLayout +# substitutes GENERAL and records it. From frame 2 on, the declaration is +# FALSE. That is what VUID-VkImageMemoryBarrier2-oldLayout-01197 forbids. +# +# THE MEASUREMENT: 0 validation messages, exit 0, on this host and layer. So +# the declaration is false and the layer does not say so. Both halves of that +# sentence matter -- it is precisely why this defect class keeps surviving +# review, and it is why this suite's real assertions are counters. +# +# SO WHAT IS THIS ENTRY FOR. Two things, neither of them "it is red": +# * it pins that the pre-sentinel Chromium shape still REGISTERS, encodes 8 +# reused frames and routes LOCAL, so the library half of the fix did not +# break the caller it was aimed at before that caller has moved; +# * it is the standing record that a green layer run over this shape does +# NOT mean the declaration is true. Anyone tempted to keep declaring +# PREINITIALIZED because "validation is clean" should read this block +# first. +# The caller-side fix is the UNDEFINED sentinel, which is the truthful answer: +# the layout being described belongs to the LIBRARY's own imported VkImage, +# created initialLayout = UNDEFINED, and the caller genuinely never knew it. +# --------------------------------------------------------------------------- +vvs_add_gated_test(EncoderExtInputResidencyPreinitializedChromiumMirror + --preinit-opaque-fd --validate) + +# --------------------------------------------------------------------------- +# THE LAYOUT-TABLE HOLE. OPAQUE_FD + RESIDENCY_LOCAL declaring +# VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL as the registration default. +# +# THIS ARM DID NOT FAIL BEFORE THE FIX -- IT ABORTED. The staging copy's +# source acquire transitions TO TRANSFER_SRC_OPTIMAL, so a caller declaring +# TRANSFER_SRC_OPTIMAL produces the pair +# (TRANSFER_SRC_OPTIMAL -> TRANSFER_SRC_OPTIMAL), which the transition table +# had no arm for. The table's terminal else is `throw std::invalid_argument` +# under __cpp_exceptions, this standalone CMake build defines it, and nothing +# on StageInputFrame's call stack catches -- so the result was +# std::terminate/SIGABRT, not a test failure. In Chromium's -fno-exceptions +# build the same pair is instead a SILENT fall-through that keeps the barrier +# struct's VIDEO_ENCODE defaults in front of a TRANSFER read. +# +# Nothing about this declaration is exotic: it is what a caller reusing a +# staging image it last used as a copy source would naturally state, it is +# what the ext layer's own legacy wrap uses as its default, and +# VkVideoEncoder.cpp:1290-1294 and :1513-1517 both promise in as many words +# that such a caller round-trips. +# --------------------------------------------------------------------------- +vvs_add_counter_test(EncoderExtInputResidencyTransferSrcOptimalDeclaration + --local-tso-opaque-fd) + +# The same arm under the layer. The counter entry above proves the process +# survives; this one proves the barrier it now records is spec-clean -- +# specifically that the arm's stage masks are legal on the family this batch +# is submitted to (VUID-vkCmdPipelineBarrier2-srcStageMask-09675) and that +# the handback leaves the image where frame 2's acquire says it is +# (VUID-VkImageMemoryBarrier2-oldLayout-01197). Neither is visible to a +# counter. +vvs_add_gated_test(EncoderExtInputResidencyTransferSrcOptimalValidated + --local-tso-opaque-fd --validate) + +message(STATUS "encoder_ext_input_residency_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-input-residency/src/main.cpp b/vk_video_encoder/test/encoder-ext-input-residency/src/main.cpp new file mode 100644 index 00000000..10fcdf31 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-input-residency/src/main.cpp @@ -0,0 +1,827 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * INPUT RESIDENCY ON AN OS-HANDLE REGISTRATION: does an explicit + * VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL survive the ext layer? + * + * ------------------------------------------------------------------------ + * THE COVERAGE HOLE THIS FILLS, stated first because it is why the defect + * this suite covers survived three rounds of review. + * ------------------------------------------------------------------------ + * `grep -rn handleType vk_video_encoder/test --include=*.cpp` returned EIGHT + * registration sites outside this file, and every one of them declares + * VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE. Zero OPAQUE_FD, zero + * DMA_BUF -- so the entire OS-handle import path, and with it every rule the + * ext layer applies ONLY to an OS handle, had no test of any kind. The + * residency read in SubmitRegisteredFrame is exactly such a rule: + * + * residency = (slot->handleType == ..._VK_IMAGE) ? slot->residency + * : ..._FOREIGN; + * + * A suite that only ever registers VK_IMAGE takes the left arm every time and + * cannot see the right one at all. That is the shape this file adds. + * + * It is also the shape Chromium's SHIPPING CPU staging tier uses -- OPAQUE_FD, + * self-exported from the library's own VkDevice and re-imported into it, + * declaring RESIDENCY_LOCAL -- and that tier has no library-side coverage + * whatsoever. + * + * ------------------------------------------------------------------------ + * WHY THE ASSERTION IS A COUNTER AND NOT A VALIDATION-ERROR COUNT. + * ------------------------------------------------------------------------ + * This was settled by construction before the file was written, because a + * test that cannot fail is worse than no test. The two barrier programs the + * residency decision selects between are BOTH spec-clean and BOTH leave the + * image in the same layout: + * + * derived FOREIGN acquire GENERAL -> TRANSFER_SRC_OPTIMAL with + * srcQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT, + * srcStageMask NONE; handback releases back to FOREIGN + * with newLayout = srcOldLayout, residual CLEARED. + * honoured LOCAL the same layout pair with both families IGNORED and + * srcStageMask HOST|TRANSFER; handback RESTORES to + * srcOldLayout and RECORDS it. + * + * Both round-trip to the same layout, so frames 2..N are validation-clean + * either way. The three real differences -- a frame-1 acquire from FOREIGN + * that nothing released, N ownership transfers that transfer nothing, and the + * LOSS OF THE AVAILABILITY OPERATION for the caller's host writes -- are none + * of them core VUIDs. Arm 1 emits no validation message whether its LOCAL + * declaration is honoured or discarded, while its counters move from + * foreign-only to local-only. The layer is live for both readings -- a + * temporary probe arm declaring a usage-incompatible layout does raise + * messages through the same harness -- so the silence is the layer's answer + * and not a dead harness. + * + * So the observable is VkVideoEncoderInputResidencyInfo, chained onto + * GetCompletionInfo()'s pNext, which reports the two sides of that decision + * directly. It was added for this suite and is what makes arm 1 red. + * + * ------------------------------------------------------------------------ + * THE ARMS. One registration each, EIGHT frames through it -- reuse is not + * incidental, it is the case the residency declaration exists for -- with a + * different host-written pattern mapped in between frames. + * ------------------------------------------------------------------------ + * --local-opaque-fd SUBJECT. OPAQUE_FD + explicit LOCAL. Expects the + * local acquire. RED before the fix (8 foreign), + * green after. + * --local-vk-image CONTROL. The IDENTICAL image and the identical + * declaration registered as VK_IMAGE, the one arm + * where residency is already honoured. Subject-red + * with control-green isolates handleType and nothing + * else -- not the host, not the format, not reuse. + * --foreign-opaque-fd NEGATIVE CONTROL. Explicit FOREIGN must still take + * the foreign acquire. + * --auto-opaque-fd PIN. residency left zero (AUTO) must still DERIVE + * foreign. Together with the arm above this is what + * stops the fix degenerating into "never acquire": + * deleting the ternary's FOREIGN branch outright would + * turn the subject green and break the Wayland/GBM + * zero-copy lane and the renderer app, both of which + * inherit AUTO. + * --preinit-opaque-fd THE CHROMIUM-TODAY MIRROR. OPAQUE_FD + LOCAL, but + * declaring VK_IMAGE_LAYOUT_PREINITIALIZED as both the + * registration default and the per-frame layout, which + * is what media/gpu/vulkan/vulkan_video_encode_ + * accelerator.cc did before the sentinel change. It + * PASSES -- counters and validation both -- and it was + * expected to fail. Its declaration is nonetheless + * FALSE from frame 2 on (the library's handback leaves + * the image in GENERAL, because PREINITIALIZED is not + * a legal barrier destination), and the layer does not + * say so. See the CMake entry: that combination is why + * this defect class keeps surviving review, and why + * this suite asserts on counters. + * --local-tso-opaque-fd THE LAYOUT-TABLE HOLE. OPAQUE_FD + LOCAL declaring + * VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL. Before the + * transition table gained a + * (TRANSFER_SRC_OPTIMAL -> TRANSFER_SRC_OPTIMAL) arm + * this did not fail -- it ABORTED, via the table's + * terminal `throw` in a build with no handler. + * + * Exits 0 all assertions held, 1 an assertion failed, 77 (CTest SKIP) with no + * encode-capable Vulkan device -- same contract as the sibling suites. + */ + +#include "vulkan_video_encoder_ext_internal.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + std::printf(" ok %s\n", what); + return; + } + g_failures++; + std::printf(" FAIL %s : %s\n", what, detail.c_str()); +} + +std::string U64(unsigned long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%llu", v); + return buf; +} + +std::string I64(long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%lld", v); + return buf; +} + +const uint32_t kWidth = 1280; +const uint32_t kHeight = 720; + +// EIGHT, and the number is load-bearing rather than arbitrary. The residency +// declaration exists for a REUSED input image -- the header says so in as many +// words ("A caller pooling/reusing input images must state the residency +// explicitly") -- and a single-frame registration cannot distinguish the two +// handbacks at all, because nothing ever reads back what they left behind. +const uint32_t kFrames = 8; + +// --------------------------------------------------------------------------- +// Device-level entry points, resolved off the LIBRARY's own device. +// +// GetMemoryFdKHR is the one this suite adds over its siblings: the test +// EXPORTS from the library's device and hands the fd straight back to the +// library, which re-imports it into that same device. That self-import is not +// a contrivance -- it is precisely what Chromium's CPU staging tier does +// (vulkan_video_encode_accelerator.cc takes `device = lib_state_->encoder-> +// GetVkDevice()`), and it is the reason the "the memory is foreign to the +// encode device" premise behind the derived FOREIGN is false on that tier. +// --------------------------------------------------------------------------- +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetMemoryFdKHR GetMemoryFdKHR = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = + nullptr; + PFN_vkGetPhysicalDeviceProperties2 GetPhysicalDeviceProperties2 = nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + // vkGetInstanceProcAddr is linked, not dlopen'd: this binary links the + // Vulkan loader like its siblings and never stands up an instance of its + // own -- every handle it uses came out of the library. + auto gipa = (PFN_vkGetInstanceProcAddr)vkGetInstanceProcAddr; + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, + "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + std::printf(" ERROR: no vkGetDeviceProcAddr\n"); + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) + // NOT under LOAD_DEV's hard failure: a device without + // VK_KHR_external_memory_fd cannot run the OS-handle arms at all, and + // that is a HOST fact, not a library defect. main() turns a null here + // into a 77, so it can never be mistaken for a pass. + fns->GetMemoryFdKHR = + (PFN_vkGetMemoryFdKHR)gdpa(device, "vkGetMemoryFdKHR"); +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + fns->GetPhysicalDeviceProperties2 = + (PFN_vkGetPhysicalDeviceProperties2)gipa( + instance, "vkGetPhysicalDeviceProperties2"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr) && + (fns->GetPhysicalDeviceProperties2 != nullptr); +} + +struct InputImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; + void* mapped = nullptr; + VkDeviceSize size = 0; + uint32_t typeIndex = UINT32_MAX; + uint32_t typeBits = 0; +}; + +// A host-written LINEAR NV12 image on the library's own device, allocated +// EXPORTABLE as OPAQUE_FD and DEDICATED. +// +// EVERY FIELD HERE IS CONSTRAINED BY THE IMPORT SIDE, not chosen for taste. +// VkEncImportExternalImage rebuilds the VkImage from the descriptor as 2D / +// desc.format / extent / mip 1 / layers 1 / samples 1 / desc.tiling / +// desc.imageUsage / desc.imageFlags / desc.sharingMode, chaining +// VkExternalMemoryImageCreateInfo{OPAQUE_FD} and forcing +// initialLayout = VK_IMAGE_LAYOUT_UNDEFINED. If the exporting create info +// disagrees with that in any field, the import is a different image over the +// same bytes and the aliasing is not defined. +// +// initialLayout is UNDEFINED here for a second, independent reason: an image +// created with external memory CANNOT be PREINITIALIZED +// (VUID-VkImageCreateInfo-pNext-01443). The VK_IMAGE control arm uses this +// same image, deliberately, so that the two arms differ in handleType and in +// nothing else -- a control that used a differently-created image would leave +// initialLayout as an alternative explanation for a red subject. +bool CreateInputImage(const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, bool exportable, InputImage* out) +{ + VkExternalMemoryImageCreateInfo extCI{ + VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_IMAGE_CREATE_INFO}; + extCI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT; + + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.pNext = exportable ? &extCI : nullptr; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = VK_IMAGE_TILING_LINEAR; + ci.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + std::printf(" SKIP-CAUSE: vkCreateImage(LINEAR NV12, exportable=%d) " + "failed\n", (int)exportable); + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + const VkMemoryPropertyFlags want = + (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + std::printf(" SKIP-CAUSE: no host-visible+coherent memory type\n"); + fns.DestroyImage(device, out->image, nullptr); + out->image = VK_NULL_HANDLE; + return false; + } + + // DEDICATED, and it must be: the library performs every OS-handle import + // as a dedicated allocation (the VkMemoryDedicatedAllocateInfo chain is + // what avoids a known NVIDIA zero-fill defect on imported encode images), + // so an import of a non-dedicated export is a mismatch the driver + // reports late and unhelpfully. + VkMemoryDedicatedAllocateInfo dedicated{ + VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO}; + dedicated.image = out->image; + + VkExportMemoryAllocateInfo exportAI{ + VK_STRUCTURE_TYPE_EXPORT_MEMORY_ALLOCATE_INFO}; + exportAI.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT; + exportAI.pNext = &dedicated; + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.pNext = exportable ? (const void*)&exportAI + : (const void*)&dedicated; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + std::printf(" SKIP-CAUSE: vkAllocateMemory(exportable=%d) failed\n", + (int)exportable); + fns.DestroyImage(device, out->image, nullptr); + out->image = VK_NULL_HANDLE; + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != + VK_SUCCESS) { + std::printf(" SKIP-CAUSE: vkBindImageMemory failed\n"); + fns.DestroyImage(device, out->image, nullptr); + fns.FreeMemory(device, out->memory, nullptr); + out->image = VK_NULL_HANDLE; + out->memory = VK_NULL_HANDLE; + return false; + } + + // PERSISTENTLY mapped, which is the shape the residency declaration is + // about: a caller that keeps a mapping and host-writes the same staging + // image between submits. HOST_COHERENT, so there is nothing to flush. + if (fns.MapMemory(device, out->memory, 0, req.size, 0, &out->mapped) != + VK_SUCCESS) { + std::printf(" SKIP-CAUSE: vkMapMemory failed\n"); + fns.DestroyImage(device, out->image, nullptr); + fns.FreeMemory(device, out->memory, nullptr); + out->image = VK_NULL_HANDLE; + out->memory = VK_NULL_HANDLE; + return false; + } + out->size = req.size; + out->typeIndex = typeIndex; + out->typeBits = req.memoryTypeBits; + return true; +} + +void DestroyInputImage(const DeviceFns& fns, VkDevice device, InputImage* in) +{ + if (in->mapped != nullptr) { + fns.UnmapMemory(device, in->memory); + in->mapped = nullptr; + } + if (in->image != VK_NULL_HANDLE) { + fns.DestroyImage(device, in->image, nullptr); + in->image = VK_NULL_HANDLE; + } + if (in->memory != VK_NULL_HANDLE) { + fns.FreeMemory(device, in->memory, nullptr); + in->memory = VK_NULL_HANDLE; + } +} + +// Real content, and a DIFFERENT pattern per frame. Two reasons, both about +// this suite specifically: a flat surface encodes to a degenerate bitstream, +// and identical frames would let a run in which the staging copy silently read +// stale bytes look exactly like a correct one. +void WriteFramePattern(InputImage* in, uint32_t frame) +{ + uint8_t* bytes = (uint8_t*)in->mapped; + for (VkDeviceSize i = 0; i < in->size; i++) { + bytes[i] = (uint8_t)(((i * 7u) ^ (i >> 9)) + (frame * 37u)); + } +} + +void FillConfig(VkVideoEncoderConfig* config) +{ + config->sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config->codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + config->encodeWidth = kWidth; + config->encodeHeight = kHeight; + config->inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + config->inputWidth = kWidth; + config->inputHeight = kHeight; + config->rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config->averageBitrate = 5000000; + config->maxBitrate = 5000000; + config->gopLength = 30; + config->consecutiveBFrames = 0; + config->idrPeriod = 30; + config->frameRateNum = 30; + config->frameRateDen = 1; + config->deviceId = -1; + config->disableFileOutput = VK_TRUE; +} + +struct Arm { + const char* flag; + const char* what; + VkVideoEncoderExternalHandleType handleType; + VkVideoEncoderInputResidency residency; + VkImageLayout defaultLayout; + // UNDEFINED here is THE SENTINEL, which the public contract defines as + // "as declared at registration" -- not a missing value. + VkImageLayout perFrameLayout; + uint64_t expectForeign; + uint64_t expectLocal; +}; + +const Arm kArms[] = { + {"--local-opaque-fd", + "SUBJECT -- OPAQUE_FD self-import declaring RESIDENCY_LOCAL", + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD, + VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL, + VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_UNDEFINED, + /*foreign=*/0, /*local=*/kFrames}, + + {"--local-vk-image", + "CONTROL -- the IDENTICAL image and declaration, registered as VK_IMAGE", + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE, + VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL, + VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_UNDEFINED, + /*foreign=*/0, /*local=*/kFrames}, + + {"--foreign-opaque-fd", + "NEGATIVE CONTROL -- OPAQUE_FD declaring RESIDENCY_FOREIGN", + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD, + VK_VIDEO_ENCODER_INPUT_RESIDENCY_FOREIGN, + VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_UNDEFINED, + /*foreign=*/kFrames, /*local=*/0}, + + {"--auto-opaque-fd", + "PIN -- OPAQUE_FD leaving residency zero (AUTO); FOREIGN is still DERIVED", + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD, + VK_VIDEO_ENCODER_INPUT_RESIDENCY_AUTO, + VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_UNDEFINED, + /*foreign=*/kFrames, /*local=*/0}, + + {"--preinit-opaque-fd", + "CHROMIUM MIRROR -- OPAQUE_FD + LOCAL declaring PREINITIALIZED per frame", + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD, + VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL, + VK_IMAGE_LAYOUT_PREINITIALIZED, VK_IMAGE_LAYOUT_PREINITIALIZED, + /*foreign=*/0, /*local=*/kFrames}, + + // THE LAYOUT-TABLE HOLE. Everything above declares GENERAL or + // PREINITIALIZED, and those are the only two input layouts the transition + // table was ever exercised with on this lane. A caller that declares + // TRANSFER_SRC_OPTIMAL -- which the public header invites, which + // VkVideoEncoder.cpp:1290-1294 and :1513-1517 both explicitly promise + // round-trips, and which is the natural declaration for a staging source + // the caller last used as a copy source -- produces the acquire pair + // (TRANSFER_SRC_OPTIMAL -> TRANSFER_SRC_OPTIMAL), because this arm's + // newLayout for the staging copy IS TRANSFER_SRC_OPTIMAL. + // + // That pair had NO ARM. It fell to the table's terminal else, which is + // `throw std::invalid_argument` wherever __cpp_exceptions is defined -- + // and this standalone CMake build defines it (flags.make carries no + // -fno-exceptions), with no handler anywhere on StageInputFrame's call + // stack. So this arm did not fail an assertion: the process ABORTED. + // + // The counters are the same 0/kFrames as the LOCAL arms above because the + // residency decision is not what is under test here -- reaching the end + // of eight frames AT ALL is. Kept identical on purpose: if a future change + // makes this arm take a foreign acquire, that is also a defect and this + // catches it too. + {"--local-tso-opaque-fd", + "LAYOUT HOLE -- OPAQUE_FD + LOCAL declaring TRANSFER_SRC_OPTIMAL", + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_OPAQUE_FD, + VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL, + VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, VK_IMAGE_LAYOUT_UNDEFINED, + /*foreign=*/0, /*local=*/kFrames}, +}; + +} // namespace + +int main(int argc, const char** argv) +{ + const Arm* arm = nullptr; + bool validate = false; + for (int i = 1; i < argc; i++) { + if (std::strcmp(argv[i], "--validate") == 0) { + validate = true; + continue; + } + for (const Arm& a : kArms) { + if (std::strcmp(argv[i], a.flag) == 0) { + arm = &a; + } + } + } + if (arm == nullptr) { + std::printf("usage: %s [--validate]\n arms:\n", argv[0]); + for (const Arm& a : kArms) { + std::printf(" %-22s %s\n", a.flag, a.what); + } + // NOT 77. An unrecognised argument is a harness defect, and a harness + // defect that reports SKIP is how a suite comes to sit green having + // run nothing. + return 1; + } + + std::printf("Encoder-ext INPUT RESIDENCY: %s\n", arm->what); + std::printf(" handleType=%d residency=%d defaultLayout=%d " + "perFrameLayout=%d frames=%u\n", + (int)arm->handleType, (int)arm->residency, + (int)arm->defaultLayout, (int)arm->perFrameLayout, kFrames); + // Layer provenance, echoed for the same reason the adopt suite echoes it: + // a validation claim without the layer configuration that produced it is + // not a measurement. The wording avoids the bare token used by the CTest + // skip regex. + { + const char* layerPath = std::getenv("VK_LAYER_PATH"); + const char* layerSettings = std::getenv("VK_LAYER_SETTINGS_PATH"); + std::printf(" VK_LAYER_PATH=%s\n", + layerPath ? layerPath : "(unset)"); + std::printf(" VK_LAYER_SETTINGS_PATH=%s\n", + layerSettings ? layerSettings : "(unset)"); + } + std::printf("------------------------------------------------\n"); + + int rc = 0; + { + VkSharedBaseObj encoder; + if ((CreateVulkanVideoEncoderExt(encoder) != VK_SUCCESS) || !encoder) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + return 77; + } + + VkVideoEncoderConfig config = {}; + FillConfig(&config); + if (validate) { + config.validate = VK_TRUE; + } + const VkResult init = encoder->InitializeExt(config); + if (init != VK_SUCCESS) { + // A SKIP, and only here: no encode-capable device, no codec, no + // encode queue. Every failure BELOW this line is an assertion, + // because by then the library has stood a working device up. + std::printf("SKIP: InitializeExt failed (%d)\n", (int)init); + encoder.reset(); + return 77; + } + + const VkDevice device = encoder->GetVkDevice(); + const VkPhysicalDevice phys = encoder->GetVkPhysicalDevice(); + const VkInstance inst = encoder->GetVkInstance(); + DeviceFns fns; + if ((device == VK_NULL_HANDLE) || (phys == VK_NULL_HANDLE) || + (inst == VK_NULL_HANDLE) || !LoadDeviceFns(inst, device, &fns)) { + std::printf("SKIP: could not resolve device entry points\n"); + encoder.reset(); + return 77; + } + + const bool isOsHandle = + (arm->handleType != + VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE); + if (isOsHandle && (fns.GetMemoryFdKHR == nullptr)) { + std::printf("SKIP: the library's device has no vkGetMemoryFdKHR; " + "the OS-handle arms cannot run here\n"); + encoder.reset(); + return 77; + } + + // THE IMAGE IS ALWAYS EXPORTABLE, on every arm including VK_IMAGE. + // That is what makes the control a control: the two registrations + // describe the SAME VkImage with the SAME create info, and differ + // only in which door they come through. + InputImage input; + if (!CreateInputImage(fns, phys, device, /*exportable=*/true, + &input)) { + std::printf("SKIP: could not create the input image\n"); + encoder.reset(); + return 77; + } + WriteFramePattern(&input, 0); + + int fd = -1; + if (isOsHandle) { + VkMemoryGetFdInfoKHR getFd{ + VK_STRUCTURE_TYPE_MEMORY_GET_FD_INFO_KHR}; + getFd.memory = input.memory; + getFd.handleType = VK_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD_BIT; + const VkResult fr = fns.GetMemoryFdKHR(device, &getFd, &fd); + if ((fr != VK_SUCCESS) || (fd < 0)) { + std::printf("SKIP: vkGetMemoryFdKHR failed (%d)\n", (int)fr); + DestroyInputImage(fns, device, &input); + encoder.reset(); + return 77; + } + std::printf(" exported OPAQUE_FD %d from the LIBRARY's own " + "device (self-import)\n", fd); + } + + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = arm->handleType; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = kWidth; + desc.height = kHeight; + desc.tiling = VK_IMAGE_TILING_LINEAR; + desc.imageUsage = (VkImageUsageFlags)VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + desc.planeCount = 0; + desc.residency = arm->residency; + desc.defaultLayout = arm->defaultLayout; + if (isOsHandle) { + // The exporter's own allocation parameters, which an OPAQUE_FD + // import REQUIRES rather than prefers + // (VUID-VkMemoryAllocateInfo-allocationSize-01742): the ext layer + // sets exporterIndexExact for this handle type, so a guessed + // index or a zero size is a refusal, not a fallback. This test + // has all three because it performed the export. + desc.allocationSize = (uint64_t)input.size; + desc.memoryTypeIndex = input.typeIndex; + desc.memoryTypeBits = input.typeBits; + // PROVENANCE. Propagated rather than left zero so the descriptor + // is the shape a real consumer sends; a zeroed deviceUUID is + // accepted with a warning and would make the multi-GPU check + // vacuous on this arm. + VkPhysicalDeviceIDProperties idProps{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_ID_PROPERTIES}; + VkPhysicalDeviceProperties2 props2{ + VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2}; + props2.pNext = &idProps; + fns.GetPhysicalDeviceProperties2(phys, &props2); + std::memcpy(desc.deviceUUID, idProps.deviceUUID, VK_UUID_SIZE); + std::memcpy(desc.driverUUID, idProps.driverUUID, VK_UUID_SIZE); + } else { + desc.existingImage = input.image; + } + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + VkVideoEncoderStatus regStatus = {}; + regStatus.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_STATUS; + const VkVideoEncoderStatusCode reg = encoder->RegisterImageResource( + desc, (uint64_t)(int64_t)(isOsHandle ? fd : -1), &resource, + ®Status); + + // AN ASSERTION, NOT A SKIP -- the same reclassification the adopt + // suite argues at length. The host question was settled above: + // CreateInputImage built this very image on this very device, and + // (on the OS arms) vkGetMemoryFdKHR already exported from it. A + // library refusal from here on is a library fact. + Check((reg == VK_VIDEO_ENCODER_STATUS_SUCCESS) && + (resource != VK_VIDEO_ENCODER_RESOURCE_NULL), + "RegisterImageResource", "status " + I64((long long)reg)); + if (isOsHandle) { + // The ownership echo, asserted rather than assumed: ownership is + // TRANSFER by zero-init, so the library consumes the fd on EVERY + // exit including failure, and the test must not close it again. + Check(regStatus.handlesConsumed == VK_TRUE, + "the library consumed the transferred fd", + "handlesConsumed was VK_FALSE"); + fd = -1; + } + + uint32_t captured = 0; + uint32_t submitted = 0; + uint64_t bytes = 0; + if (resource != VK_VIDEO_ENCODER_RESOURCE_NULL) { + for (uint32_t f = 0; f < kFrames; f++) { + // THE REUSE. A new pattern is host-written into the SAME + // mapping before every frame, which is the whole point of the + // shape: the image the library staged from last frame is the + // image being rewritten now. + WriteFramePattern(&input, f); + + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = resource; + info.frameId = f; + info.pts = f; + info.qpOverride = -1; + info.currentLayout = arm->perFrameLayout; + + VkVideoEncoderStatusCode status = + encoder->SubmitRegisteredFrame(info, nullptr); + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && + (retry < 2000); + retry++) { + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == + VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + } + status = encoder->SubmitRegisteredFrame(info, nullptr); + if (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + struct timespec ts = {0, 1000000}; // 1 ms + nanosleep(&ts, nullptr); + } + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + Check(false, "SubmitRegisteredFrame", + "frame " + I64(f) + " status " + + I64((long long)status)); + break; + } + submitted++; + + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == + VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + } + } + } + + // DRAIN BEFORE READING THE COUNTERS, and before freeing anything the + // submitted work reads. DrainPendingFrames joins the encoder and + // assembly threads, so every submitted frame has been through + // StageInputFrame -- and therefore counted -- by the time it returns. + // It does NOT clear m_encoder, so the side-channel below still has + // something to answer from; Flush() and teardown DO, which is why + // neither is called first. + Check(encoder->DrainPendingFrames() == VK_SUCCESS, + "DrainPendingFrames", + "the encoder could not flush its in-flight work"); + { + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == VK_SUCCESS) { + captured++; + bytes += drained.bitstreamSize; + encoder->ReleaseEncodedFrame(drained.frameId); + } + } + + // ---- THE OBSERVABLE ---- + VkVideoEncoderInputResidencyInfo residencyInfo = {}; + residencyInfo.sType = + VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_RESIDENCY_INFO; + residencyInfo.pNext = nullptr; + VkVideoEncoderCompletionInfo completion = {}; + completion.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_COMPLETION_INFO; + completion.pNext = &residencyInfo; + const VkResult ci = encoder->GetCompletionInfo(&completion); + Check(ci == VK_SUCCESS, + "GetCompletionInfo accepts a chained " + "VkVideoEncoderInputResidencyInfo", + "returned " + I64((long long)ci)); + + std::printf(" submitted=%u captured=%u bytes=%llu " + "foreignAcquireCount=%llu localAcquireCount=%llu\n", + submitted, captured, (unsigned long long)bytes, + (unsigned long long)residencyInfo.foreignAcquireCount, + (unsigned long long)residencyInfo.localAcquireCount); + + Check(submitted == kFrames, "every frame was submitted", + U64(submitted) + " of " + U64(kFrames)); + Check(captured == kFrames, "every frame produced a bitstream", + U64(captured) + " of " + U64(kFrames)); + Check(bytes > 0, "the bitstreams are non-empty", "0 bytes"); + + // THE TWO ASSERTIONS THIS SUITE EXISTS FOR. Both sides, not one: an + // arm that only checked foreignAcquireCount == 0 would also pass on a + // build where no frame reached the staging tier at all. + Check(residencyInfo.foreignAcquireCount == arm->expectForeign, + "foreignAcquireCount", + "expected " + U64(arm->expectForeign) + ", got " + + U64(residencyInfo.foreignAcquireCount)); + Check(residencyInfo.localAcquireCount == arm->expectLocal, + "localAcquireCount", + "expected " + U64(arm->expectLocal) + ", got " + + U64(residencyInfo.localAcquireCount)); + + // UNREGISTER BEFORE DESTROYING, which the public header requires in + // as many words: a driver may recycle the handle value and a + // surviving registration would then name freed memory. + if (resource != VK_VIDEO_ENCODER_RESOURCE_NULL) { + Check(encoder->UnregisterImageResource(resource) == + VK_VIDEO_ENCODER_STATUS_SUCCESS, + "UnregisterImageResource", + "the registration id did not resolve"); + } + DestroyInputImage(fns, device, &input); + if (fd >= 0) { + // Only reachable if registration never happened, i.e. the library + // never took the transfer. + close(fd); + } + encoder.reset(); + } + + std::printf("------------------------------------------------\n"); + if (g_failures == 0) { + std::printf("PASSED : %d checks, 0 failures\n", g_checks); + } else { + std::printf("FAILED : %d checks, %d failures\n", g_checks, g_failures); + rc = 1; + } + return rc; +} diff --git a/vk_video_encoder/test/encoder-ext-input-residency/validation_gate.cmake b/vk_video_encoder/test/encoder-ext-input-residency/validation_gate.cmake new file mode 100644 index 00000000..33753ccd --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-input-residency/validation_gate.cmake @@ -0,0 +1,239 @@ +# Run a test binary and decide pass / fail / skip on the exit code AND the +# validation-layer output together. +# +# WHY A WRAPPER RATHER THAN set_tests_properties(FAIL_REGULAR_EXPRESSION): +# SKIP_RETURN_CODE outranks FAIL_REGULAR_EXPRESSION in CTest, so a run that +# emitted validation errors on its way to a 77 exit was recorded as Skipped. +# main.cpp has two 77 returns reached AFTER device creation, i.e. after the +# layer can already have spoken. A property-only gate cannot see both signals. +# +# ------------------------------------------------------------------------ +# THE ECHO IS SANITISED, AND THAT IS LOAD-BEARING, NOT TIDINESS. +# ------------------------------------------------------------------------ +# CTest decides SKIP by regex-matching the WHOLE captured output, and this +# script replays the child's stdout+stderr into that same space. An earlier +# version claimed the skip token "does not exist on any failing path" because +# only this script emits it. That guarantee was void: the child's text is in the +# match space too. Demonstrated end-to-end -- a directory named +# .../VALIDATION_GATE_RESULT=SKIP passed to VK_LAYER_PATH is echoed verbatim by +# main.cpp's provenance line, and a real run with 46 genuine validation errors +# was recorded "100% tests passed ... ***Skipped". +# +# So every occurrence of the token is defanged in the child's text before it is +# printed. The token then appears in the output only when THIS script writes it, +# on a path that has already established there were no validation errors. +# +# ------------------------------------------------------------------------ +# WHAT IS COUNTED: MESSAGES, NOT VUID TOKENS. +# ------------------------------------------------------------------------ +# The same defect yields a different token count depending on which reporter is +# active, with the library untouched: +# * arms where the harness installs its own debug-utils messenger print a +# "Validation Error: [ VUID-x ]" header AND a "The Vulkan spec states: +# ...(VUID-x)" trailer -- TWO tokens per message, on stdout; +# * --own-validate has no harness callback (the library owns the instance), so +# the layer's default reporter prints ONE token per message, on stderr. +# Counting tokens made five arms look like "92 occurrences" against +# --own-validate's 46 when all five in fact have exactly 46 errors. A count that +# doubles under a reporter swap cannot distinguish "removed a defect" from +# "changed a message format". What both reporters emit exactly once per message +# is the "The Vulkan spec states:" trailer, so that closes a message block and +# the block carries the body the allowlist needs. Verified against both: +# default reporter tokens=46 headers=0 -> messages=46; messenger reporter +# tokens=92 headers=46 -> messages=46. +# +# CONTRACT +# -DTEST_EXE= required +# -DTEST_ARGS= optional, ONE argument or a space-separated string +# -DREQUIRE_LAYER= optional. ON: "no layer" is a failure, not a skip, +# so a fleet cannot sit green purely because the layer +# was missing everywhere. +# -DALLOW_VUIDS= optional, matched against the MESSAGE BODY. +# -DRUN_TIMEOUT= optional, default 600. + +if(NOT DEFINED TEST_EXE) + message(FATAL_ERROR "validation_gate: TEST_EXE is required") +endif() +if(NOT DEFINED RUN_TIMEOUT OR RUN_TIMEOUT STREQUAL "") + set(RUN_TIMEOUT 600) +endif() +if(NOT DEFINED ALLOW_VUIDS OR ALLOW_VUIDS STREQUAL "") + # Matched against the BODY, deliberately, because the VUID NAME cannot carry + # this distinction. VUID-Vk-pNext-pNext fires for two unrelated things: + # a struct type the layer does not recognise (version skew, benign), and a + # struct the layer knows perfectly well but which is NOT PERMITTED in that + # chain -- a genuine defect, and exactly the kind a video-encode library that + # chains many extension structs is at risk of. Only the body separates them. + # An earlier version allowlisted on the name and would have passed a run whose + # only message was "...which is not allowed here". + # + # SCOPE, measured, because it is easy to overstate: with the Chrome-bundled + # layer there are ZERO skew messages on all seven arms. With the Vulkan SDK + # layer they appear on six, but are the SOLE cause of redness on exactly ONE + # (--context-conflict); the others are red from the genuine defect anyway. So + # this allowlist keeps one arm honest, not six. "Older layer" is not the + # explanation either -- both manifests declare api_version 1.4.304, and it is + # the SDK build that reports skew while the newer Chrome-bundled one does not. + set(ALLOW_VUIDS "unknown VkStructureType") +endif() + +set(_args "") +if(DEFINED TEST_ARGS AND NOT TEST_ARGS STREQUAL "") + # Accept the list spelling too. A caller that passes ${ARGN} straight + # through hands us "--arm;--validate", which has no whitespace, so + # separate_arguments would return it unsplit as ONE argument -- and an + # unrecognised argument makes the subject print usage and exit non-zero, + # which reads as a test failure rather than as a harness bug. Normalising + # first means both spellings work and neither fails silently. + string(REPLACE ";" " " _test_args_norm "${TEST_ARGS}") + separate_arguments(_args NATIVE_COMMAND "${_test_args_norm}") +endif() + +# TIMEOUT rather than streaming. execute_process buffers, so a hang would +# otherwise reach CTest's own TIMEOUT and take every captured byte with it -- +# measured: "", zero diagnostics, on a suite whose most plausible +# hang is a GPU encode. Timing out INSIDE the script keeps what was captured. +# Streaming (ECHO_*_VARIABLE) is not the answer: it would put the child's raw +# text back into CTest's match space and re-open the token collision above. +execute_process( + COMMAND "${TEST_EXE}" ${_args} + TIMEOUT ${RUN_TIMEOUT} + RESULT_VARIABLE _rc + OUTPUT_VARIABLE _out + ERROR_VARIABLE _err) + +set(_all "${_out}${_err}") +set(_safe "${_all}") +string(REPLACE "VALIDATION_GATE_RESULT" "VALIDATION_GATE_RESULT_FROM_CHILD" _safe "${_safe}") +message("${_safe}") + +# Message count. Prefer the header, which exists exactly once per message when a +# debug-utils messenger is installed; fall back to raw tokens for the default +# reporter, which emits one per message. +string(REGEX MATCHALL "Validation Error" _headers "${_all}") +list(LENGTH _headers _n_headers) +string(REGEX MATCHALL "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+" _tokens "${_all}") +list(LENGTH _tokens _n_tokens) + +# BLOCK-BASED, and the reason matters -- a line-based count is wrong for one of +# the two reporters and an earlier attempt silently passed a 46-error run +# because of it. +# +# messenger reporter (arms where the harness installs its own callback): +# "Validation Error: [ VUID-x ] ... " <- token here +# "The Vulkan spec states: ... (...#VUID-x)" <- token here too +# default reporter (--own-validate; the library owns the instance so there is +# no harness callback): +# "vkQueueSubmit2KHR(): ... " <- NO token +# "The Vulkan spec states: ... (...#VUID-x)" <- token ONLY here +# +# So "skip the trailer" discards every message on the default reporter, and +# "count every token" double-counts on the messenger one. What both emit exactly +# once per message is the TRAILER, so the trailer closes a block and the block +# carries the body the allowlist needs. +string(REPLACE ";" "\\;" _lines "${_all}") +string(REPLACE "\n" ";" _lines "${_lines}") +set(_real 0) +set(_allowed 0) +set(_names "") +set(_block "") +foreach(_line IN LISTS _lines) + set(_block "${_block}\n${_line}") + if(_line MATCHES "The Vulkan spec states") + if(_block MATCHES "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+") + set(_vuid "${CMAKE_MATCH_0}") + if(_block MATCHES "${ALLOW_VUIDS}") + math(EXPR _allowed "${_allowed}+1") + else() + math(EXPR _real "${_real}+1") + list(APPEND _names "${_vuid}") + endif() + endif() + set(_block "") + endif() +endforeach() +# A layer that emits a VUID with no trailer at all would leave a dangling block; +# count it rather than lose it. +if(_block MATCHES "VUID-[A-Za-z0-9_]+-[A-Za-z0-9_-]+") + set(_vuid "${CMAKE_MATCH_0}") + if(_block MATCHES "${ALLOW_VUIDS}") + math(EXPR _allowed "${_allowed}+1") + else() + math(EXPR _real "${_real}+1") + list(APPEND _names "${_vuid}") + endif() +endif() +if(_names) + list(REMOVE_DUPLICATES _names) + string(REPLACE ";" " " _names "${_names}") +endif() + +message("VALIDATION_GATE: exit=${_rc} messages=${_real} allowlisted=${_allowed} " + "(raw tokens=${_n_tokens}, headers=${_n_headers})") + +# Allowlisted messages are tolerated but never silent. On a green run CTest +# shows no output at all, so an accumulation of them would otherwise be +# invisible forever; this line is appended to a file that survives the run. +if(DEFINED GATE_SUMMARY AND NOT GATE_SUMMARY STREQUAL "") + file(APPEND "${GATE_SUMMARY}" + "${TEST_EXE} ${TEST_ARGS}: exit=${_rc} messages=${_real} allowlisted=${_allowed}\n") +endif() + +# ------------------------------------------------------------------------ +# THE ENCODE-SOURCE LAYOUT REGRESSION, WHICH NO VALIDATION MESSAGE CAN CARRY. +# ------------------------------------------------------------------------ +# VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811 is checked against the image-layout +# map of the command buffer the encode is recorded into. The staged input's +# barriers are recorded into a DIFFERENT command buffer, so that map has no +# entry for the encode-source image and the check returns true without +# comparing anything -- and the submit-time sweep reads the same registry, so it +# is blind in the same way. CF-02a and CF-02b lived in that blind spot through +# a run reported as "144 -> 0 validation messages"; zero was the correct count +# for a check that never ran. +# +# So the counter above CANNOT see this defect class, and a gate built only on it +# is a gate that passes an encode reading its source in TRANSFER_DST_OPTIMAL. +# VkVideoEncoder::RecordVideoCodingCmd emits its own unconditional diagnostic +# when the staging arm left the image in anything other than +# VIDEO_ENCODE_SRC_KHR; this promotes that from loud to gating. +# +# DEMONSTRATED IN BOTH DIRECTIONS, because a gate that has never been red is not +# known to be a gate: with the hand-off barriers present the two staged arms +# emit zero of these, and with them deleted the copy arm emits 8 and the filter +# arm 60. +string(REGEX MATCHALL "staged encode-source image is in layout" _layout_hits "${_all}") +list(LENGTH _layout_hits _n_layout) +if(_n_layout GREATER 0) + message(FATAL_ERROR + "validation_gate: ${_n_layout} frame(s) reached vkCmdEncodeVideoKHR" + " with the staged encode-source image in the wrong layout" + " (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-10811). A StageInputFrame arm" + " is missing its hand-off barrier to VIDEO_ENCODE_SRC_KHR. This is" + " invisible to the validation layer -- see the note above -- so this" + " check is the only thing that reports it.") +endif() + +if(_real GREATER 0) + message(FATAL_ERROR + "validation_gate: ${_real} validation message(s): ${_names}" + " (exit=${_rc}, ${_allowed} allowlisted by body /${ALLOW_VUIDS}/)") +endif() + +if(_rc EQUAL 77) + if(REQUIRE_LAYER) + message(FATAL_ERROR + "validation_gate: the run skipped (77) but REQUIRE_LAYER is ON." + " A skipped validation arm proves nothing; set" + " VVS_VALIDATION_LAYER_PATH, or configure with" + " -DVVS_REQUIRE_VALIDATION_LAYER=OFF to allow skipping.") + endif() + message("VALIDATION_GATE_RESULT=SKIP") + return() +endif() + +# _rc is a STRING on abnormal exit ("Segmentation fault", "Process terminated +# due to timeout", "no such file or directory"). EQUAL comparisons are correctly +# false for those, so both branches above fall through to here and fail. +if(NOT _rc EQUAL 0) + message(FATAL_ERROR "validation_gate: test did not exit cleanly: ${_rc}") +endif() diff --git a/vk_video_encoder/test/encoder-ext-input-residency/vk_layer_settings.txt b/vk_video_encoder/test/encoder-ext-input-residency/vk_layer_settings.txt new file mode 100644 index 00000000..21623b8b --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-input-residency/vk_layer_settings.txt @@ -0,0 +1,27 @@ +# Layer settings for the encoder-ext-adopt-device suite. +# +# duplicate_message_limit = 0 means "print forever". +# +# WHY IT IS PINNED, and this is a real cap, not a theoretical one. With no +# settings file reachable the layer applies its documented default of 10 +# messages; with this file in effect the run reports every message; with a +# small explicit limit it reports that many plus the layer's own "reported N +# times ... last time" notice. +# +# The cap is keyed on the VUID string alone, with no per-object component, so a +# single repeated VUID truncates and the run silently understates itself. +# +# METHOD WARNING. The layer also searches the CURRENT WORKING DIRECTORY for +# vk_layer_settings.txt. A control run that merely unsets +# VK_LAYER_SETTINGS_PATH while sitting in a directory that holds this file is +# NOT uncapped, and will report the full count while appearing to prove the cap +# does not apply. Run controls from a clean directory. +# +# WHAT THIS FILE DOES NOT BUY: portability of the count across LAYER BUILDS. +# Different validation-layer builds report different totals and different sets +# of distinct VUIDs for the same binary on the same host. One of them reports +# VUID-vkCmdEncodeVideoKHR-pEncodeInfo-08206 -- an ENCODE-specific VUID, the +# class this suite exists to catch -- which another never emits. So no single +# count is canonical, and the lower one must not be read as the answer. Which +# layer runs is controlled by VVS_VALIDATION_LAYER_PATH in CMakeLists.txt. +khronos_validation.duplicate_message_limit = 0 diff --git a/vk_video_encoder/test/encoder-ext-reconfigure/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-reconfigure/CMakeLists.txt new file mode 100644 index 00000000..11ff2591 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-reconfigure/CMakeLists.txt @@ -0,0 +1,84 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_reconfigure_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, not the shared one: the internal header's +# seam functions are deliberately not exported from libvkvideo-encoder.so, so +# only the archive can satisfy them. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + # The public encoder header reaches VkCodecUtils/VkVideoRefCountBase.h, + # which lives under the shared common-libs root. + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# NOTE ON CTest SEMANTICS, matching the sibling library tests: 0 means every +# assertion held, 1 means an assertion failed, and 2 means the session could +# not be stood up at all -- deliberately a CTest FAILURE and not a skip, +# because a host that could not run this must not report the Reconfigure +# field dispositions as verified. There is no GPU, driver or display +# dependence here: the session has no device by construction. +enable_testing() +add_test(NAME EncoderExtReconfigureFieldDispositions + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtReconfigureFieldDispositions PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_ext_reconfigure_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-reconfigure/src/main.cpp b/vk_video_encoder/test/encoder-ext-reconfigure/src/main.cpp new file mode 100644 index 00000000..2a847644 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-reconfigure/src/main.cpp @@ -0,0 +1,1140 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Device-free coverage for WHAT Reconfigure() CARRIES, WHAT IT REFUSES, AND + * WHAT IT STILL IGNORES ON PURPOSE. + * + * THE CONTRACT UNDER TEST, stated by Reconfigure itself: "Everything this + * call cannot carry is REFUSED rather than discarded. Returning VK_SUCCESS + * for a field that was ignored is how a stream ends up encoded one way and + * described another." VkVideoEncoderConfig carries 44 members, and they + * sort four ways: + * + * GATES (2) -- sType and pNext. + * + * APPLIED (9) -- averageBitrate, maxBitrate, frameRateNum, frameRateDen, + * constQpI/P/B, and minQp/maxQp. Each is asserted below as a CHANGE the + * session then holds, never as a return code: a VK_SUCCESS that altered + * nothing is the defect this file exists to pin, so asserting the return + * code alone would assert nothing at all. + * + * REFUSED ON A CHANGE (21) -- the sequence-header and input-routing + * fields, the input extent, vbvBufferSize, the GOP structure and the + * quality controls. Every one is settled at InitializeExt and can reach + * the bitstream, so accepting a change is the misdescription the contract + * forbids. + * + * NEITHER (12) -- the seven session-creation-only members and the five + * diagnostic ones. None can change an encoded bit, so none can + * misdescribe the stream. They stay accepted, and the group below is the + * control that says so: a fix that refused these too would break a caller + * that builds a fresh minimal config for the reconfigure rather than + * copying its stored one, and would buy nothing. + * + * THE RECORD IS ALSO UNDER TEST, and it is the one thing here that is not + * about a config field at all. Reconfigure keeps a copy of the + * configuration in force, both to compare a later call against and as the + * session statement of what it is running. Two members are COERCED on the + * way in and a third can be DROPPED -- a zero maxBitrate becomes the + * average bitrate, a zero frameRateDen becomes 1, and a zero frameRateNum + * leaves the frame rate untouched -- so a record of what was PASSED is not + * a record of what is in force. The cases below read the record and the + * live rate-control layer through two separate seams and assert they agree, + * which is the only form of that claim that is not just inspection. + * + * WHY THE COMPARISON IS AGAINST THE INIT CONFIG rather than an absolute + * refusal: a caller hands Reconfigure a WHOLE config, not a minimal one. + * Two shapes of that are known: retaining a baseline and editing only its + * rate fields in place, and copying the init config and editing a couple of + * fields -- the second is what the sibling format-matrix test does. Refusing + * on a CHANGE rather than on presence is what keeps both shapes working, and + * the first group below is the control that proves an unchanged config is + * still accepted. + * + * WHY THIS RUNS WITHOUT A GPU. Reconfigure gates on + * (m_initialized && m_encoder) before it compares anything. + * VkEncInstallNullBackend sets the first; VkEncPushCapture installs the + * second device-free. Both are required -- with only the null backend every + * call here would stop at that gate and answer NOT_PERMITTED, and the file + * would pass for the wrong reason. The session is never InitializeExt-ed, so + * m_initConfig holds a default-constructed VkVideoEncoderConfig and a + * default-constructed config in this file agrees with it on every compared + * field. That is what makes each case single-variable: the only thing that + * differs is the one field the case sets. + * + * CTest semantics, matching the sibling library tests: 0 means every + * assertion held, 1 means an assertion failed, 2 means the session could not + * be stood up at all -- deliberately a FAILURE and not a skip. There is no + * GPU, driver or display dependence here. + */ + +#include "vulkan_video_encoder_ext_internal.h" + +#include +#include +#include + +namespace { + +int g_checks = 0; +int g_failures = 0; +const char* g_currentCase = ""; + +std::string I64(int64_t v) +{ + char b[32]; + std::snprintf(b, sizeof(b), "%lld", (long long)v); + return b; +} + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + std::printf(" ok %s\n", what); + } else { + g_failures++; + std::printf(" FAIL %s [%s] (%s)\n", what, g_currentCase, + detail.c_str()); + } +} + +// A null-backend session that Reconfigure can actually reach. See the file +// comment: the capture push is what installs m_encoder, and without it every +// assertion below would be answered by the gate instead of by the field +// comparison under test. +class Session { +public: + bool Open() + { + if ((CreateVulkanVideoEncoderExt(m_encoder) != VK_SUCCESS) || + !m_encoder) { + std::printf(" ERROR: CreateVulkanVideoEncoderExt failed\n"); + return false; + } + if (VkEncInstallNullBackend(m_encoder.get(), &m_backend) != + VK_SUCCESS) { + std::printf(" ERROR: VkEncInstallNullBackend failed\n"); + return false; + } + if (VkEncPushCapture(m_encoder.get(), 1, VK_SUCCESS) != VK_SUCCESS) { + std::printf(" ERROR: VkEncPushCapture failed\n"); + return false; + } + return true; + } + + VulkanVideoEncoderExt* Get() const { return m_encoder.get(); } + +private: + VkEncNullBackendState m_backend{}; + VkSharedBaseObj m_encoder; +}; + +// Agrees with the never-initialized m_initConfig on every compared field. +// averageBitrate must be non-zero or RequestRateControlUpdate refuses the +// call for a reason that has nothing to do with what is being tested. +VkVideoEncoderConfig Base() +{ + VkVideoEncoderConfig cfg{}; + cfg.averageBitrate = 5000000; + // NEGATIVE is how this API spells "this config names no quantizer", and + // it is the spelling InitializeExt reads too. ZERO would NOT mean that: + // 0 is a valid lossless QP, so a zero-initialized config asks for QP 0 + // rather than asking for nothing. A default-constructed config therefore + // does NOT mean "leave the quantizers alone", which is why this says so + // explicitly. + cfg.constQpI = -1; + cfg.constQpP = -1; + cfg.constQpB = -1; + return cfg; +} + +void ExpectRefused(Session& s, const char* field, VkVideoEncoderConfig cfg) +{ + const VkResult r = s.Get()->Reconfigure(cfg); + Check(r == VK_ERROR_INITIALIZATION_FAILED, field, + "Reconfigure returned " + I64((int64_t)r) + + ", want VK_ERROR_INITIALIZATION_FAILED (-3)"); +} + +void ExpectAccepted(Session& s, const char* what, VkVideoEncoderConfig cfg) +{ + const VkResult r = s.Get()->Reconfigure(cfg); + Check(r == VK_SUCCESS, what, + "Reconfigure returned " + I64((int64_t)r) + ", want VK_SUCCESS (0)"); +} + +//============================================================================= +// 1. THE CONTROL. An unchanged config is still accepted. +//============================================================================= +// +// Without this, a change that refused everything would look like a fix. Both +// in-tree callers pass a whole config that differs from the init config in +// the rate fields alone, so this is the shape that actually ships. +void CaseUnchangedConfigIsAccepted(Session& s) +{ + g_currentCase = "an unchanged config is accepted"; + ExpectAccepted(s, "an unchanged config is accepted", Base()); +} + +// The four fields the call exists to carry. Changing them must stay accepted. +void CaseCarriedRateFieldsAreAccepted(Session& s) +{ + g_currentCase = "the four carried rate fields are accepted"; + VkVideoEncoderConfig cfg = Base(); + cfg.averageBitrate = 9000000; + cfg.maxBitrate = 12000000; + cfg.frameRateNum = 60; + cfg.frameRateDen = 1; + ExpectAccepted(s, "a rate-control change is carried", cfg); +} + +//============================================================================= +// 2. THE RED LEG. Encoding-affecting fields are REFUSED, not ignored. +//============================================================================= +// +// Every case here answered VK_SUCCESS before the fix while the encoder went +// on using the value it was initialized with. +void CaseEncodingAffectingFieldsAreRefused(Session& s) +{ + g_currentCase = "an encoding-affecting change is refused"; + + // The input extent: what the session conversion was sized around. + { VkVideoEncoderConfig c = Base(); c.inputWidth = 1920; + ExpectRefused(s, "inputWidth is refused", c); } + { VkVideoEncoderConfig c = Base(); c.inputHeight = 1080; + ExpectRefused(s, "inputHeight is refused", c); } + + // The rate-control lever this call still does not carry. NOT for the + // reason minQp and maxQp were once refused beside it: the codec fill + // can be re-invoked, and for the clamps it is. vbvBufferSize is + // refused because it is not an independent input to that fill -- what + // reaches the command is a duration in milliseconds computed from it + // AND from a vbvInitialDelay derived from the OLD buffer size in a + // finalize step this call does not re-run, both divided by a config + // bitrate a mid-stream bitrate change deliberately leaves alone. + { VkVideoEncoderConfig c = Base(); c.vbvBufferSize = 4000000; + ExpectRefused(s, "vbvBufferSize is refused", c); } + + // The GOP structure the session sequences to. + { VkVideoEncoderConfig c = Base(); c.gopLength = 60; + ExpectRefused(s, "gopLength is refused", c); } + { VkVideoEncoderConfig c = Base(); c.consecutiveBFrames = 2; + ExpectRefused(s, "consecutiveBFrames is refused", c); } + { VkVideoEncoderConfig c = Base(); c.idrPeriod = 120; + ExpectRefused(s, "idrPeriod is refused", c); } + { VkVideoEncoderConfig c = Base(); c.closedGop = VK_TRUE; + ExpectRefused(s, "closedGop is refused", c); } + + // The encode-quality controls. + { VkVideoEncoderConfig c = Base(); c.qualityLevel = 2; + ExpectRefused(s, "qualityLevel is refused", c); } + { VkVideoEncoderConfig c = Base(); + c.tuningMode = VK_VIDEO_ENCODE_TUNING_MODE_HIGH_QUALITY_KHR; + ExpectRefused(s, "tuningMode is refused", c); } +} + + +//============================================================================= +// 2b. THE APPLIED GROUP. constQp is CHANGED, not merely accepted. +//============================================================================= +// +// The sharpest case in the whole config, and the one field here that is +// APPLIED rather than refused. On a DISABLED (constant-QP) session the +// constant-QP defaults are the only session-level rate lever there is: the +// per-layer bitrates a rate-control command carries are dropped outright on +// such a session, because that mode commands layerCount 0. So before this +// change a constant-QP caller had NO working session-level control at all, +// and Reconfigure answered VK_SUCCESS to every attempt. +// +// WHAT THIS ASSERTS, AND WHAT IT DOES NOT. It reads back the value the +// session now holds after folding the update -- the exact member +// EncodeFrameCommon copies into the next frame it processes, +// unconditionally, on the encoder thread. A VK_SUCCESS that changed nothing +// is the defect this file exists to pin, so asserting the return code alone +// would assert nothing. What it CANNOT assert is that the driver then +// honours the value in the produced bitstream: that needs a GPU and a decode +// comparison, and is deliberately not claimed here. +// +// IT ALSO CANNOT SEE WHICH FRAME THE VALUE LANDS ON, and that is a +// structural limit of the seam rather than a gap in this case. The seam +// forces a fold and then reads m_encoderConfig->constQp; the ordering that +// decides the landing frame is between that fold and the per-frame copy at +// the top of EncodeFrameCommon, which no device-free session reaches -- the +// null backend stubs out EncodeFrame, the frame-info pool and the image +// resources EncodeFrameCommon asserts on. A build with the fold moved back +// after the copy reads green here and encodes the change one frame late. +// The witness for that is a device run: encode a DISABLED-mode stream, +// Reconfigure before frame N, and read SliceQPy back out of frame N. +void CaseConstQpIsAppliedNotIgnored(Session& s) +{ + g_currentCase = "constQp is applied to the session"; + + int32_t qpI = -1, qpP = -1, qpB = -1; + + // The control FIRST, so the observable is calibrated against a known + // no-change input before it is trusted on a changing one: a config that + // names no QP at all must leave the session exactly as it was. + ExpectAccepted(s, "a config naming no constQp is accepted", Base()); + Check(VkEncApplyAndGetSessionConstQp(s.Get(), &qpI, &qpP, &qpB) == + VK_SUCCESS, + "the session constQp is readable", "seam refused"); + Check((qpI == 0) && (qpP == 0) && (qpB == 0), + "a config naming no constQp leaves the session untouched", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + ", want 0/0/0"); + + // The red leg: a named QP must actually reach the session. + VkVideoEncoderConfig c = Base(); + c.constQpI = 33; + c.constQpP = 34; + c.constQpB = 35; + ExpectAccepted(s, "a constQp change is accepted", c); + Check(VkEncApplyAndGetSessionConstQp(s.Get(), &qpI, &qpP, &qpB) == + VK_SUCCESS, + "the session constQp is readable after the update", "seam refused"); + Check((qpI == 33) && (qpP == 34) && (qpB == 35), + "the constQp change reached the session", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 33/34/35"); + + // A partial declaration touches only what it names. -1 means "not + // specified" at InitializeExt and must keep meaning that here, or a + // caller that copies a config built for a non-CQP session -- where the + // three are left at -1 -- would have its QPs silently rewritten. + VkVideoEncoderConfig p = Base(); + p.constQpI = 40; // named + p.constQpP = -1; // not named -- must survive untouched + p.constQpB = -1; // not named -- must survive untouched + ExpectAccepted(s, "a partial constQp change is accepted", p); + Check(VkEncApplyAndGetSessionConstQp(s.Get(), &qpI, &qpP, &qpB) == + VK_SUCCESS, + "the session constQp is readable after the partial update", + "seam refused"); + Check((qpI == 40) && (qpP == 34) && (qpB == 35), + "an unnamed constQp member is left alone", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 40/34/35"); +} + +//============================================================================= +//============================================================================= +// 2c. THE RECORD AGREES WITH WHAT IS IN FORCE. +//============================================================================= +// +// Reconfigure keeps a copy of the configuration so a later call compares +// against what is actually running. It was updated from the RAW config, and +// the raw config is not always what the session ends up using: a zero +// maxBitrate is coerced to the average bitrate, a zero frameRateDen becomes +// 1, and a zero frameRateNum leaves the frame rate alone entirely. The +// record said 0 in each case while the session ran on something else. +// +// HOW THIS IS ASSERTED. Not by re-deriving the coercion in the test -- that +// would only check the test against itself. Both halves are read back +// through separate seams: VkEncGetRecordedConfig for the record, and +// VkEncApplyAndGetRateControl for the LIVE rate-control layer, which is the +// struct HandleCtrlCmd copies verbatim into the next control command. The +// assertion is that the two agree. +// +// WHAT THIS DOES NOT CLAIM. Nothing compared by Reconfigure reads these +// members, so no refusal decision moves either way -- see the note in +// Reconfigure. This is a truthfulness fix to the record, not a behaviour +// change, and the cases are written to show exactly that. +void CaseRecordAgreesWithWhatIsInForce() +{ + Session s; + if (!s.Open()) { + Check(false, "record-vs-in-force session opens", "session setup"); + return; + } + VkEncRateControlObservation live{}; + VkVideoEncoderConfig rec{}; + + auto Read = [&](const char* what) -> bool { + const bool ok = + (VkEncApplyAndGetRateControl(s.Get(), &live) == VK_SUCCESS) && + (VkEncGetRecordedConfig(s.Get(), &rec) == VK_SUCCESS); + Check(ok, what, "an observation seam refused"); + return ok; + }; + + // THE CONTROL, FIRST. A config where nothing is coerced: every value + // passed is the value applied. The record and the live layer must agree + // here too, and that agreement is the point: an + // observable that only ever reported "they agree" would prove nothing + // below, and one that reported "they differ" on THIS input would be + // measuring something other than the coercion. + g_currentCase = "an uncoerced config: record and session already agree"; + { + VkVideoEncoderConfig c = Base(); + c.averageBitrate = 9000000; + c.maxBitrate = 12000000; + c.frameRateNum = 60; + c.frameRateDen = 1; + ExpectAccepted(s, "an uncoerced rate change is accepted", c); + if (Read("the record and the live layer are both readable")) { + Check((live.layerMaxBitrate == 12000000) && + (live.layerFrameRateNumerator == 60) && + (live.layerFrameRateDenominator == 1), + "the uncoerced values are in force", + "live " + I64((int64_t)live.layerMaxBitrate) + " " + + I64(live.layerFrameRateNumerator) + "/" + + I64(live.layerFrameRateDenominator)); + Check((rec.maxBitrate == live.layerMaxBitrate) && + (rec.frameRateNum == live.layerFrameRateNumerator) && + (rec.frameRateDen == live.layerFrameRateDenominator), + "the record agrees with the session on an uncoerced config", + "record " + I64(rec.maxBitrate) + " " + + I64(rec.frameRateNum) + "/" + I64(rec.frameRateDen)); + } + } + + // RED LEG 1. maxBitrate 0 means "track averageBitrate". The session + // runs at 7000000, so the record must not say 0. + g_currentCase = "a coerced maxBitrate is recorded as what is in force"; + { + VkVideoEncoderConfig c = Base(); + c.averageBitrate = 7000000; + c.maxBitrate = 0; + c.frameRateNum = 60; + c.frameRateDen = 1; + ExpectAccepted(s, "a zero maxBitrate is accepted", c); + if (Read("the seams are readable after the coerced maxBitrate")) { + Check(live.layerMaxBitrate == 7000000, + "a zero maxBitrate runs at the average bitrate", + "live max " + I64((int64_t)live.layerMaxBitrate)); + Check(rec.maxBitrate == live.layerMaxBitrate, + "the record holds the coerced maxBitrate, not the zero", + "record " + I64(rec.maxBitrate) + ", live " + + I64((int64_t)live.layerMaxBitrate)); + } + } + + // RED LEG 2. frameRateNum 0 leaves the frame rate alone -- BOTH halves + // of it. The session is still at 60/1, so the record must not say 0/99. + g_currentCase = "a dropped frame rate is not recorded as passed"; + { + VkVideoEncoderConfig c = Base(); + c.averageBitrate = 7000000; + c.maxBitrate = 7000000; + c.frameRateNum = 0; + c.frameRateDen = 99; + ExpectAccepted(s, "a zero frameRateNum is accepted", c); + if (Read("the seams are readable after the dropped frame rate")) { + Check((live.layerFrameRateNumerator == 60) && + (live.layerFrameRateDenominator == 1), + "a zero frameRateNum leaves the frame rate in force", + "live " + I64(live.layerFrameRateNumerator) + "/" + + I64(live.layerFrameRateDenominator)); + Check((rec.frameRateNum == live.layerFrameRateNumerator) && + (rec.frameRateDen == live.layerFrameRateDenominator), + "the record keeps the frame rate that is in force", + "record " + I64(rec.frameRateNum) + "/" + + I64(rec.frameRateDen) + ", live " + + I64(live.layerFrameRateNumerator) + "/" + + I64(live.layerFrameRateDenominator)); + } + } + + // RED LEG 3. frameRateDen 0 beside a non-zero numerator becomes 1. The + // session runs 24/1, so the record must not say 24/0, which is not a frame + // rate at all. + g_currentCase = "a coerced frameRateDen is recorded as what is in force"; + { + VkVideoEncoderConfig c = Base(); + c.averageBitrate = 7000000; + c.maxBitrate = 7000000; + c.frameRateNum = 24; + c.frameRateDen = 0; + ExpectAccepted(s, "a zero frameRateDen is accepted", c); + if (Read("the seams are readable after the coerced frameRateDen")) { + Check((live.layerFrameRateNumerator == 24) && + (live.layerFrameRateDenominator == 1), + "a zero frameRateDen runs as 1", + "live " + I64(live.layerFrameRateNumerator) + "/" + + I64(live.layerFrameRateDenominator)); + Check(rec.frameRateDen == live.layerFrameRateDenominator, + "the record holds the coerced denominator, not the zero", + "record den " + I64(rec.frameRateDen)); + } + } +} + +//============================================================================= +// 2d. THE QP CLAMPS ARE APPLIED, not refused and not ignored. +//============================================================================= +// +// minQp and maxQp were refused on the grounds that they reach the driver +// only through the codec-specific rate-control structs, which are filled +// once at codec-init. Half of that is true and half is not: the fill, +// EncoderConfig::GetRateControlParameters, is a pure function of config +// state, and CodecHandleRateControlCmd chains its output onto EVERY +// ENCODE_RATE_CONTROL command. So re-invoking it on the encoder thread puts +// a new clamp on the very command this call already causes. +// +// WHAT IS ASSERTED, AND AT WHAT DEPTH. Not the return code, and not the +// config field either -- a config field that moved while the codec struct +// did not would be a clamp that reaches nothing. The assertion is on +// resolvedUseMinQp / resolvedMinQpI, read out of the codec rate-control +// layer struct itself, which is the far end of the chain the library owns. +// Whether the driver then honours it needs a GPU and is not claimed. +void CaseQpClampIsAppliedNotIgnored() +{ + Session s; + if (!s.Open()) { + Check(false, "QP-clamp session opens", "session setup"); + return; + } + VkEncRateControlObservation live{}; + auto Read = [&](const char* what) -> bool { + const bool ok = + (VkEncApplyAndGetRateControl(s.Get(), &live) == VK_SUCCESS); + Check(ok, what, "the observation seam refused"); + return ok; + }; + + // THE CONTROL, FIRST, so the observable is calibrated on a known + // no-change input before it is trusted on a changing one. A config + // naming no clamp must leave the resolved struct reporting no clamp AND + // must not re-invoke the codec fill at all. + g_currentCase = "a config naming no QP clamp changes nothing"; + VkVideoEncoderConfig base = Base(); + base.averageBitrate = 8000000; + base.maxBitrate = 10000000; + base.frameRateNum = 50; + base.frameRateDen = 1; + ExpectAccepted(s, "a config naming no QP clamp is accepted", base); + if (Read("the rate-control observation is readable")) { + Check((live.configMinQpSet == 0) && (live.configMaxQpSet == 0), + "no clamp is requested on the session", + "set flags " + I64(live.configMinQpSet) + "/" + + I64(live.configMaxQpSet)); + Check((live.resolvedUseMinQp == 0) && (live.resolvedUseMaxQp == 0), + "the resolved codec struct carries no clamp", + "use flags " + I64(live.resolvedUseMinQp) + "/" + + I64(live.resolvedUseMaxQp)); + Check(live.codecRefreshCount == 0, + "no clamp change means no codec refresh", + "refresh count " + I64(live.codecRefreshCount)); + } + + // THE RED LEG. A named clamp must reach the codec struct a control + // command is built from. + // + // THE FRAME RATE IS DELIBERATELY LEFT UNNAMED HERE. The codec fill + // rewrites the live layer frame rate from the CONFIG, which still holds + // the value the session was built with -- and with frameRateNum 0 the + // update carries no frame rate to repair it afterwards. So the frame + // rate surviving this call at 50/1 is the assertion that the refresh + // did not reach past what it is for. + g_currentCase = "a QP clamp change reaches the codec rate-control struct"; + VkVideoEncoderConfig clamped = base; + clamped.frameRateNum = 0; + clamped.frameRateDen = 0; + clamped.minQp = 20; + clamped.maxQp = 44; + ExpectAccepted(s, "a QP clamp change is accepted", clamped); + if (Read("the observation is readable after the clamp change")) { + Check((live.configMinQp == 20) && (live.configMinQpSet == 1) && + (live.configMaxQp == 44) && (live.configMaxQpSet == 1), + "the clamp request reached the session config", + "config " + I64(live.configMinQp) + "/" + + I64(live.configMaxQp)); + Check((live.resolvedUseMinQp == 1) && (live.resolvedMinQpI == 20), + "the minQp clamp resolved into the codec struct", + "use " + I64(live.resolvedUseMinQp) + ", qpI " + + I64(live.resolvedMinQpI) + ", want 1 and 20"); + Check((live.resolvedUseMaxQp == 1) && (live.resolvedMaxQpI == 44), + "the maxQp clamp resolved into the codec struct", + "use " + I64(live.resolvedUseMaxQp) + ", qpI " + + I64(live.resolvedMaxQpI) + ", want 1 and 44"); + Check(live.codecRefreshCount == 1, + "the codec fill was re-invoked exactly once", + "refresh count " + I64(live.codecRefreshCount)); + Check((live.layerFrameRateNumerator == 50) && + (live.layerFrameRateDenominator == 1), + "the refresh did not revert the live frame rate", + "live " + I64(live.layerFrameRateNumerator) + "/" + + I64(live.layerFrameRateDenominator) + ", want 50/1"); + Check((live.layerAverageBitrate == 8000000) && + (live.layerMaxBitrate == 10000000), + "the refresh did not revert the live bitrates", + "live " + I64((int64_t)live.layerAverageBitrate) + "/" + + I64((int64_t)live.layerMaxBitrate)); + } + + // A CLAMP THAT CAN BE SET MUST BE CLEARABLE. Zero means "no clamp" at + // InitializeExt, and it has to keep meaning that here or these two + // fields would carry two different contracts. + g_currentCase = "a zero QP clamp clears the clamp"; + VkVideoEncoderConfig cleared = clamped; + cleared.minQp = 0; + ExpectAccepted(s, "clearing the minQp clamp is accepted", cleared); + if (Read("the observation is readable after the clear")) { + Check((live.configMinQpSet == 0) && (live.resolvedUseMinQp == 0), + "the cleared clamp no longer reaches the codec struct", + "set " + I64(live.configMinQpSet) + ", use " + + I64(live.resolvedUseMinQp)); + Check((live.resolvedUseMaxQp == 1) && (live.resolvedMaxQpI == 44), + "clearing one clamp leaves the other alone", + "use " + I64(live.resolvedUseMaxQp) + ", qpI " + + I64(live.resolvedMaxQpI)); + } + + // THE SCOPE CONTROL. The refresh is not a general recompute: an update + // that changes no clamp must not run it. Without this a fix that + // re-invoked the fill on every rate change would look identical. + g_currentCase = "a bitrate-only change does not re-invoke the codec fill"; + const uint32_t refreshBefore = live.codecRefreshCount; + VkVideoEncoderConfig rateOnly = cleared; + rateOnly.averageBitrate = 6000000; + ExpectAccepted(s, "a bitrate-only change is accepted", rateOnly); + if (Read("the observation is readable after the bitrate-only change")) { + Check(live.codecRefreshCount == refreshBefore, + "no clamp change means no codec refresh", + "count " + I64(live.codecRefreshCount) + ", want " + + I64(refreshBefore)); + Check((live.resolvedUseMaxQp == 1) && (live.resolvedMaxQpI == 44), + "the standing clamp survives a bitrate-only change", + "use " + I64(live.resolvedUseMaxQp) + ", qpI " + + I64(live.resolvedMaxQpI)); + Check(live.layerAverageBitrate == 6000000, + "the bitrate-only change is still in force", + "live " + I64((int64_t)live.layerAverageBitrate)); + } +} + +//============================================================================= +// 2e. WHERE A QP CLAMP CHANGE IS STILL REFUSED. +//============================================================================= +// +// Three cases, and on each of them the codec fill would run and carry +// nothing -- which is the accepted-and-ignored shape the contract forbids, +// not a milder version of it. Plus the two validations InitializeExt +// already applies, so a value refused at init cannot arrive here instead. +void CaseQpClampRefusals() +{ + Session s; + if (!s.Open()) { + Check(false, "QP-clamp refusal session opens", "session setup"); + return; + } + + g_currentCase = "an invalid QP clamp is refused"; + { VkVideoEncoderConfig c = Base(); c.minQp = 60; + ExpectRefused(s, "a minQp outside 0..51 is refused", c); } + { VkVideoEncoderConfig c = Base(); c.maxQp = 52; + ExpectRefused(s, "a maxQp outside 0..51 is refused", c); } + { VkVideoEncoderConfig c = Base(); c.minQp = 40; c.maxQp = 20; + ExpectRefused(s, "an inverted clamp window is refused", c); } + + // THE DEVICE QP WINDOW. With useMinQp raised the spec requires the + // value inside the window the device reports, and the init path + // refuses one that is not. A mid-stream write that skipped the check + // would be a hole straight past it. + g_currentCase = "a clamp outside the device QP window is refused"; + Check(VkEncSetDeviceQpWindow(s.Get(), 10, 40) == VK_SUCCESS, + "the device QP window is declarable", "seam refused"); + { VkVideoEncoderConfig c = Base(); c.minQp = 5; + ExpectRefused(s, "a minQp below the device window is refused", c); } + { VkVideoEncoderConfig c = Base(); c.maxQp = 45; + ExpectRefused(s, "a maxQp above the device window is refused", c); } + // The control for that check: inside the window it is still accepted, + // so the refusals above are the window firing and not the clamp path + // being broken outright. + { VkVideoEncoderConfig c = Base(); c.minQp = 12; c.maxQp = 38; + ExpectAccepted(s, "a clamp inside the device window is accepted", c); } + + // AV1 HAS NO QP-UNIT CLAMP AT ALL. Its fill reads quantizer indices + // derived from the device capability limits and never these fields, so + // a clamp here would reach nothing. InitializeExt rejects one rather + // than ignoring it, and so does this. + g_currentCase = "a QP clamp is refused on an AV1 session"; + { + Session av1; + if (av1.Open()) { + VkVideoEncoderConfig seed = Base(); + seed.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + Check(VkEncSeedRecordedConfig(av1.Get(), &seed) == VK_SUCCESS, + "an AV1 session record is seedable", "seam refused"); + VkVideoEncoderConfig c = seed; + c.minQp = 30; + ExpectRefused(av1, "minQp is refused on an AV1 session", c); + // The control: the same session still takes a rate change, so + // the refusal is the AV1 rule and not a dead session. + ExpectAccepted(av1, "an AV1 session still takes a rate change", + seed); + } else { + Check(false, "AV1 session opens", "session setup"); + } + } + + // A CONSTANT-QP SESSION IGNORES THE CLAMPS BY CONSTRUCTION. The + // DISABLED arm of the H.26x fill sets the codec clamp from the + // quality-level constant QP and never reads the caller request, so a + // clamp change there is ignored however it is delivered. constQp is + // that session's lever, and it is applied. + g_currentCase = "a QP clamp is refused on a constant-QP session"; + { + Session cqp; + if (cqp.Open()) { + VkVideoEncoderConfig seed = Base(); + seed.rateControlMode = + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + Check(VkEncSeedRecordedConfig(cqp.Get(), &seed) == VK_SUCCESS, + "a constant-QP session record is seedable", "seam refused"); + VkVideoEncoderConfig c = seed; + c.minQp = 30; + ExpectRefused(cqp, "minQp is refused on a constant-QP session", c); + // The control, and the point of refusing rather than accepting: + // the lever that session DOES have still works. + VkVideoEncoderConfig q = seed; + q.constQpI = 28; + ExpectAccepted(cqp, "a constant-QP session still takes constQp", q); + int32_t qpI = -1, qpP = -1, qpB = -1; + Check(VkEncApplyAndGetSessionConstQp(cqp.Get(), &qpI, &qpP, + &qpB) == VK_SUCCESS, + "the constant-QP session is readable", "seam refused"); + Check(qpI == 28, + "the constant-QP change reached the constant-QP session", + "got " + I64(qpI) + ", want 28"); + } else { + Check(false, "constant-QP session opens", "session setup"); + } + } +} + +//============================================================================= +// 2f. THE CONSTANT QUANTIZER'S RANGE, WHICH IS CODEC-DEPENDENT. +//============================================================================= +// +// constQpI/P/B are stated in the CODEC'S OWN units -- a QP on 0..51 for +// H.264/H.265, a quantizer INDEX on 0..255 for AV1 -- and nothing between the +// ext boundary and the bitstream narrows the value: the fields are int32_t +// and ConstQpSettings holds uint32_t. So an out-of-range value was refused +// nowhere. It was TRUNCATED at the (uint8_t) cast in +// VkVideoEncoderAV1::EncodeFrame, modulo 256, with VK_SUCCESS answered to the +// caller: constQpI 300 encoded at quantizer index 44. +// +// EVERY REFUSAL BELOW IS PAIRED WITH A NEGATIVE CONTROL AT THE BOUNDARY -- +// 51 for H.26x, 255 for AV1 -- which must still be ACCEPTED and must still +// READ BACK AS THE VALUE GIVEN. A guard that refused everything would pass a +// refusal-only test, and a guard that CLAMPED rather than refused would pass +// one that read the return code alone. Both halves therefore assert the value +// in force, not the VkResult. +// +// AND THE CODEC-DEPENDENCE IS ITSELF ASSERTED, from the same harness: 52 is +// refused on H.264 and accepted on AV1. One range applied to both codecs +// cannot pass this file. +// +// BOTH ENTRY POINTS ARE COVERED, because a refusal that guarded only one is +// worse than none: it teaches the caller the value is validated. The binder +// half below is the InitializeExt path, reached device-free through +// VkEncBuildAndProbeConfig, which builds a config and reads it back with no +// device anywhere. + +VkVideoEncoderConfig InitBase(VkVideoCodecOperationFlagBitsKHR codec) +{ + VkVideoEncoderConfig cfg{}; + cfg.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + cfg.codec = codec; + cfg.encodeWidth = 1920; + cfg.encodeHeight = 1080; + cfg.inputWidth = 1920; + cfg.inputHeight = 1080; + cfg.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + cfg.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + cfg.averageBitrate = 4000000; + cfg.frameRateNum = 30; + cfg.frameRateDen = 1; + cfg.gopLength = 30; + cfg.constQpI = -1; + cfg.constQpP = -1; + cfg.constQpB = -1; + return cfg; +} + +// Accepted AND carried through unaltered. Reading the triple back is the +// whole point of the control: a guard that clamped an out-of-range value +// would answer VK_SUCCESS here too, and so would one that quietly rewrote an +// in-range one. +void BinderCarries(const std::string& what, VkVideoEncoderConfig cfg, + VkVideoCodecOperationFlagBitsKHR codecOp, uint32_t wantI, + uint32_t wantP, uint32_t wantB) +{ + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, codecOp, &probe); + Check(r == VK_SUCCESS, what.c_str(), + "the binder returned " + I64((int64_t)r) + ", want VK_SUCCESS (0)"); + if (r != VK_SUCCESS) { + return; + } + Check((probe.constQpIntra == wantI) && (probe.constQpInterP == wantP) && + (probe.constQpInterB == wantB) && (probe.constQpSet == 1u), + (what + " -- and is carried unaltered").c_str(), + "got " + I64(probe.constQpIntra) + "/" + I64(probe.constQpInterP) + + "/" + I64(probe.constQpInterB) + " constQpSet " + + I64(probe.constQpSet) + ", want " + I64((int64_t)wantI) + "/" + + I64((int64_t)wantP) + "/" + I64((int64_t)wantB) + + " constQpSet 1"); +} + +void BinderRefuses(const std::string& what, VkVideoEncoderConfig cfg, + VkVideoCodecOperationFlagBitsKHR codecOp) +{ + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(cfg, codecOp, &probe); + Check(r == VK_ERROR_INITIALIZATION_FAILED, what.c_str(), + "the binder returned " + I64((int64_t)r) + + ", want VK_ERROR_INITIALIZATION_FAILED (-3)"); +} + +void CaseConstQpRangeAtTheBinder() +{ + const VkVideoCodecOperationFlagBitsKHR kH264 = + VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + const VkVideoCodecOperationFlagBitsKHR kH265 = + VK_VIDEO_CODEC_OPERATION_ENCODE_H265_BIT_KHR; + const VkVideoCodecOperationFlagBitsKHR kAv1 = + VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + + //------------------------------------------------------------------- + // H.26x. The boundary is 51. + //------------------------------------------------------------------- + g_currentCase = "the H.26x constant-QP range is 0..51 at the binder"; + // CALIBRATED ON A KNOWN-CLEAN INPUT FIRST. Without a legal value that + // passes, every refusal below could be the binder failing for a reason + // that has nothing to do with the range. + { VkVideoEncoderConfig c = InitBase(kH264); + c.constQpI = 51; c.constQpP = 51; c.constQpB = 51; + BinderCarries("H.264 constQp 51 is accepted", c, kH264, 51, 51, 51); } + { VkVideoEncoderConfig c = InitBase(kH264); + c.constQpI = 0; c.constQpP = 0; c.constQpB = 0; + BinderCarries("H.264 constQp 0 is accepted", c, kH264, 0, 0, 0); } + { VkVideoEncoderConfig c = InitBase(kH264); c.constQpI = 52; + BinderRefuses("H.264 constQpI 52 is refused", c, kH264); } + { VkVideoEncoderConfig c = InitBase(kH264); c.constQpP = 52; + BinderRefuses("H.264 constQpP 52 is refused", c, kH264); } + { VkVideoEncoderConfig c = InitBase(kH264); c.constQpB = 52; + BinderRefuses("H.264 constQpB 52 is refused", c, kH264); } + // The pinned value: 300 & 0xFF is 44, a perfectly plausible QP, which is + // why nothing downstream could ever have noticed. + { VkVideoEncoderConfig c = InitBase(kH264); c.constQpI = 300; + BinderRefuses("H.264 constQpI 300 is refused, not truncated to 44", c, + kH264); } + // H.265 takes the same range through the same rule. + { VkVideoEncoderConfig c = InitBase(kH265); c.constQpI = 51; + BinderCarries("H.265 constQpI 51 is accepted", c, kH265, 51, 51, 51); } + { VkVideoEncoderConfig c = InitBase(kH265); c.constQpI = 52; + BinderRefuses("H.265 constQpI 52 is refused", c, kH265); } + + //------------------------------------------------------------------- + // AV1. The boundary is 255, and 52 is a legal input. + //------------------------------------------------------------------- + g_currentCase = "the AV1 quantizer-index range is 0..255 at the binder"; + // THE CODEC-DEPENDENCE, asserted directly: the value refused two blocks + // above is accepted here, because the unit is not the same unit. + { VkVideoEncoderConfig c = InitBase(kAv1); + c.constQpI = 52; c.constQpP = 52; c.constQpB = 52; + BinderCarries("AV1 constQp 52 is accepted -- 52 is a legal quantizer " + "index", c, kAv1, 52, 52, 52); } + { VkVideoEncoderConfig c = InitBase(kAv1); + c.constQpI = 255; c.constQpP = 255; c.constQpB = 255; + BinderCarries("AV1 constQp 255 is accepted at the boundary", c, kAv1, + 255, 255, 255); } + { VkVideoEncoderConfig c = InitBase(kAv1); c.constQpI = 256; + BinderRefuses("AV1 constQpI 256 is refused, not truncated to 0", c, + kAv1); } + { VkVideoEncoderConfig c = InitBase(kAv1); c.constQpP = 256; + BinderRefuses("AV1 constQpP 256 is refused", c, kAv1); } + { VkVideoEncoderConfig c = InitBase(kAv1); c.constQpB = 256; + BinderRefuses("AV1 constQpB 256 is refused", c, kAv1); } + { VkVideoEncoderConfig c = InitBase(kAv1); c.constQpI = 300; + BinderRefuses("AV1 constQpI 300 is refused, not truncated to 44", c, + kAv1); } + + //------------------------------------------------------------------- + // AV1 quantizer index 0: what the library guarantees about it. + //------------------------------------------------------------------- + g_currentCase = "an explicit AV1 quantizer index 0 is carried, not " + "substituted"; + // 0 is a LEGAL AV1 quantizer index -- the lossless one -- so it is + // neither refused nor normalised, and constQpSet must be raised so the + // driver-preference substitution in + // EncoderConfigAV1::InitDeviceCapabilities stays out. This is the + // library side of the contract, and it is the half the library actually + // controls: what the DRIVER then does with base_q_idx 0 is not asserted + // here and cannot be, device-free. + { VkVideoEncoderConfig c = InitBase(kAv1); + c.constQpI = 0; c.constQpP = 0; c.constQpB = 0; + BinderCarries("AV1 constQp 0 is carried and marked resolved", c, kAv1, + 0, 0, 0); } + // THE CONTROL FOR THAT, and the reason constQpSet is not decoration: a + // config that names NOTHING must leave it clear, so the substitution + // does run. If constQpSet read 1 in both cases the assertion above + // would be asserting nothing. + { + VkVideoEncoderConfig c = InitBase(kAv1); + VkEncBoundConfigProbe probe{}; + const VkResult r = VkEncBuildAndProbeConfig(c, kAv1, &probe); + Check(r == VK_SUCCESS, "an AV1 config naming no quantizer is accepted", + "the binder returned " + I64((int64_t)r)); + Check(probe.constQpSet == 0u, + "and leaves constQpSet clear, so the substitution still runs", + "constQpSet " + I64(probe.constQpSet) + ", want 0"); + } +} + +//============================================================================= +// 2g. THE SAME RANGE, ON THE OTHER ENTRY POINT. +//============================================================================= +void CaseConstQpRangeOnReconfigure() +{ + //------------------------------------------------------------------- + // H.26x. + //------------------------------------------------------------------- + g_currentCase = "the H.26x constant-QP range is 0..51 on Reconfigure"; + { + Session h26x; + if (h26x.Open()) { + VkVideoEncoderConfig seed = Base(); + seed.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + seed.rateControlMode = + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + Check(VkEncSeedRecordedConfig(h26x.Get(), &seed) == VK_SUCCESS, + "an H.264 constant-QP session record is seedable", + "seam refused"); + // Calibration: the boundary value must be ACCEPTED and must be + // the value in force afterwards. + { VkVideoEncoderConfig c = seed; + c.constQpI = 51; c.constQpP = 51; c.constQpB = 51; + ExpectAccepted(h26x, "H.264 constQp 51 is accepted", c); } + int32_t qpI = -1, qpP = -1, qpB = -1; + Check(VkEncApplyAndGetSessionConstQp(h26x.Get(), &qpI, &qpP, + &qpB) == VK_SUCCESS, + "the H.264 session quantizers are readable", "seam refused"); + Check((qpI == 51) && (qpP == 51) && (qpB == 51), + "constQp 51 is the value in force", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 51/51/51"); + { VkVideoEncoderConfig c = seed; c.constQpI = 52; + ExpectRefused(h26x, "H.264 constQpI 52 is refused", c); } + { VkVideoEncoderConfig c = seed; c.constQpP = 52; + ExpectRefused(h26x, "H.264 constQpP 52 is refused", c); } + { VkVideoEncoderConfig c = seed; c.constQpB = 52; + ExpectRefused(h26x, "H.264 constQpB 52 is refused", c); } + { VkVideoEncoderConfig c = seed; c.constQpI = 300; + ExpectRefused(h26x, + "H.264 constQpI 300 is refused, not truncated", + c); } + // A REFUSAL MUST NOT HALF-APPLY. The four calls above must have + // left the session on the last value it accepted. + qpI = qpP = qpB = -1; + Check(VkEncApplyAndGetSessionConstQp(h26x.Get(), &qpI, &qpP, + &qpB) == VK_SUCCESS, + "the H.264 session is still readable after the refusals", + "seam refused"); + Check((qpI == 51) && (qpP == 51) && (qpB == 51), + "a refused quantizer left the last accepted one in force", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 51/51/51"); + // And the session is not dead: it still takes a legal change. + { VkVideoEncoderConfig c = seed; c.constQpI = 30; + ExpectAccepted(h26x, "the H.264 session still takes constQp 30", + c); } + } else { + Check(false, "H.264 constant-QP session opens", "session setup"); + } + } + + //------------------------------------------------------------------- + // AV1. + //------------------------------------------------------------------- + g_currentCase = "the AV1 quantizer-index range is 0..255 on Reconfigure"; + { + Session av1; + if (av1.Open()) { + VkVideoEncoderConfig seed = Base(); + seed.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_AV1_BIT_KHR; + seed.rateControlMode = + VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + Check(VkEncSeedRecordedConfig(av1.Get(), &seed) == VK_SUCCESS, + "an AV1 constant-QP session record is seedable", + "seam refused"); + // The codec-dependence again, on this entry point: the value the + // H.264 session above refused is accepted here and applied. + { VkVideoEncoderConfig c = seed; + c.constQpI = 52; c.constQpP = 52; c.constQpB = 52; + ExpectAccepted(av1, "AV1 constQp 52 is accepted", c); } + int32_t qpI = -1, qpP = -1, qpB = -1; + Check(VkEncApplyAndGetSessionConstQp(av1.Get(), &qpI, &qpP, + &qpB) == VK_SUCCESS, + "the AV1 session quantizers are readable", "seam refused"); + Check((qpI == 52) && (qpP == 52) && (qpB == 52), + "AV1 constQp 52 is the value in force", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 52/52/52"); + { VkVideoEncoderConfig c = seed; + c.constQpI = 255; c.constQpP = 255; c.constQpB = 255; + ExpectAccepted(av1, "AV1 constQp 255 is accepted at the " + "boundary", c); } + qpI = qpP = qpB = -1; + Check(VkEncApplyAndGetSessionConstQp(av1.Get(), &qpI, &qpP, + &qpB) == VK_SUCCESS, + "the AV1 session is readable at the boundary", + "seam refused"); + Check((qpI == 255) && (qpP == 255) && (qpB == 255), + "AV1 constQp 255 is the value in force, not wrapped", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 255/255/255"); + { VkVideoEncoderConfig c = seed; c.constQpI = 256; + ExpectRefused(av1, + "AV1 constQpI 256 is refused, not truncated to 0", + c); } + { VkVideoEncoderConfig c = seed; c.constQpP = 256; + ExpectRefused(av1, "AV1 constQpP 256 is refused", c); } + { VkVideoEncoderConfig c = seed; c.constQpB = 256; + ExpectRefused(av1, "AV1 constQpB 256 is refused", c); } + { VkVideoEncoderConfig c = seed; c.constQpI = 300; + ExpectRefused(av1, + "AV1 constQpI 300 is refused, not truncated to 44", + c); } + qpI = qpP = qpB = -1; + Check(VkEncApplyAndGetSessionConstQp(av1.Get(), &qpI, &qpP, + &qpB) == VK_SUCCESS, + "the AV1 session is still readable after the refusals", + "seam refused"); + Check((qpI == 255) && (qpP == 255) && (qpB == 255), + "a refused AV1 quantizer left the last accepted one in " + "force", + "got " + I64(qpI) + "/" + I64(qpP) + "/" + I64(qpB) + + ", want 255/255/255"); + } else { + Check(false, "AV1 constant-QP session opens", "session setup"); + } + } +} + +//============================================================================= +// 3. THE OVER-REFUSAL CONTROL. What cannot change the stream stays accepted. +//============================================================================= +// +// The counterweight to group 2. These are the twelve of the twenty-six that +// are deliberately still ignored: refusing them would buy no correctness -- +// none can change an encoded bit -- and would break a caller that builds a +// fresh minimal config for the reconfigure. A blanket refusal of all +// twenty-six is what this group exists to catch. +void CaseNonEncodingFieldsAreStillAccepted(Session& s) +{ + g_currentCase = "a field that cannot change the stream is still accepted"; + + // Diagnostic. + { VkVideoEncoderConfig c = Base(); c.verbose = VK_TRUE; + ExpectAccepted(s, "verbose is still accepted", c); } + { VkVideoEncoderConfig c = Base(); c.validate = VK_TRUE; + ExpectAccepted(s, "validate is still accepted", c); } + { VkVideoEncoderConfig c = Base(); c.outputPath = "/tmp/reconfigure.h264"; + ExpectAccepted(s, "outputPath is still accepted", c); } + { VkVideoEncoderConfig c = Base(); c.disableFileOutput = VK_TRUE; + ExpectAccepted(s, "disableFileOutput is still accepted", c); } + { VkVideoEncoderConfig c = Base(); c.silenceStdio = VK_TRUE; + ExpectAccepted(s, "silenceStdio is still accepted", c); } + + // Session-creation-only. + { VkVideoEncoderConfig c = Base(); c.deviceId = 3; + ExpectAccepted(s, "deviceId is still accepted", c); } + { VkVideoEncoderConfig c = Base(); c.gpuUUID[0] = 0xAB; + ExpectAccepted(s, "gpuUUID is still accepted", c); } + { VkVideoEncoderConfig c = Base(); + c.externalEncodeQueueFamilyIndex = 2; + ExpectAccepted(s, "externalEncodeQueueFamilyIndex is still accepted", + c); } + { VkVideoEncoderConfig c = Base(); + c.externalComputeQueueFamilyIndex = 3; + ExpectAccepted(s, "externalComputeQueueFamilyIndex is still accepted", + c); } +} + +//============================================================================= +// 4. REGRESSION GUARD on the twelve that were already refused. +//============================================================================= +void CasePreexistingRefusalsStillHold(Session& s) +{ + g_currentCase = "the already-refused fields are still refused"; + { VkVideoEncoderConfig c = Base(); c.encodeWidth = 1920; + ExpectRefused(s, "encodeWidth is still refused", c); } + { VkVideoEncoderConfig c = Base(); c.encodeHeight = 1080; + ExpectRefused(s, "encodeHeight is still refused", c); } + { VkVideoEncoderConfig c = Base(); c.videoFullRange = VK_TRUE; + ExpectRefused(s, "videoFullRange is still refused", c); } + { VkVideoEncoderConfig c = Base(); c.colourPrimaries = 9; + ExpectRefused(s, "colourPrimaries is still refused", c); } + + g_currentCase = "the structure-type gate still fires"; + { VkVideoEncoderConfig c = Base(); + c.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_FRAME; + ExpectRefused(s, "a wrong sType is refused", c); } + { VkVideoEncoderConfig c = Base(); c.pNext = (const void*)&g_checks; + ExpectRefused(s, "a chained pNext is refused", c); } + // A RECOGNISED sType IS STILL A pNext. VkVideoEncoderInputColourInfo is + // chainable at InitializeExt and every axis it carries is immutable for + // the life of the session, so a caller that reuses one config for both + // entry points must clear pNext for this one -- and finding out by having + // the declaration silently ignored is exactly the accepted-and-dropped + // class this call refuses. The case above uses a junk pointer, which + // cannot tell "any pNext" from "an unrecognised one". + { VkVideoEncoderConfig c = Base(); + VkVideoEncoderInputColourInfo ic = {}; + ic.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_INPUT_COLOUR_INFO; + c.pNext = ⁣ + ExpectRefused(s, "a chained VkVideoEncoderInputColourInfo is refused " + "too -- Reconfigure refuses ANY pNext, recognised or " + "not", c); } +} + +} // namespace + +int main(int argc, char** argv) +{ + (void)argc; + (void)argv; + std::printf("Encoder-ext Reconfigure field dispositions\n"); + std::printf("------------------------------------------\n"); + + Session session; + if (!session.Open()) { + std::printf("RESULT: COULD-NOT-RUN (session setup failed)\n"); + return 2; + } + + CaseUnchangedConfigIsAccepted(session); + CaseCarriedRateFieldsAreAccepted(session); + CaseEncodingAffectingFieldsAreRefused(session); + CaseConstQpIsAppliedNotIgnored(session); + CaseRecordAgreesWithWhatIsInForce(); + CaseQpClampIsAppliedNotIgnored(); + CaseQpClampRefusals(); + CaseConstQpRangeAtTheBinder(); + CaseConstQpRangeOnReconfigure(); + CaseNonEncodingFieldsAreStillAccepted(session); + CasePreexistingRefusalsStillHold(session); + + std::printf("------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-release-fence/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-release-fence/CMakeLists.txt new file mode 100644 index 00000000..a639c10b --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-release-fence/CMakeLists.txt @@ -0,0 +1,97 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_release_fence_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Only the PUBLIC API is used here -- unlike the sibling encoder-ext-sync +# test, this one has a real device and therefore needs no internal seams. +# The static archive is still what it links, so the binary can be staged to +# a GPU host on its own. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # The descriptor API is an internal header: the public surface of this + # library is the encoder interface, and a test that drives the layer + # beneath it names the internal directory to say so. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# CTest semantics, DELIBERATELY different from the sibling encoder-ext-sync +# test. That one has no device by construction, so "could not stand up the +# session" is a real failure there. This one needs an encode-capable GPU, and +# a host without one has proved nothing either way -- so it exits 77 and is +# reported as SKIPPED, never as a pass. 0 means every assertion held, 1 means +# an assertion failed. +enable_testing() +# Two cases, because the two registration routings put the release fence on +# two DIFFERENT submissions: the staging copy (staged) and vkCmdEncodeVideoKHR +# (direct). A fence that only works on the staged arm silently answers -1 for +# every zero-copy producer, which is the arm this API exists for. +add_test(NAME EncoderExtReleaseFenceHwStaged + COMMAND ${PROJECT_NAME}) +add_test(NAME EncoderExtReleaseFenceHwDirect + COMMAND ${PROJECT_NAME} direct) +# A third case about the pNext CHAIN rather than the queue: the same staged +# frame with the fence descriptor hung off a VkVideoEncoderFrameSyncDescriptor +# instead of off the submit info. The header documents a flat, order- +# independent chain; this is the only thing that proves the trailing position +# is really read, because a descriptor accepted by the sType gate and then +# never read is dead code that looks wired. +add_test(NAME EncoderExtReleaseFenceHwChained + COMMAND ${PROJECT_NAME} chained) +set_tests_properties(EncoderExtReleaseFenceHwStaged + EncoderExtReleaseFenceHwDirect + EncoderExtReleaseFenceHwChained PROPERTIES + SKIP_RETURN_CODE 77 + LABELS "gpu" + TIMEOUT 900) + +message(STATUS "encoder_ext_release_fence_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-release-fence/src/main.cpp b/vk_video_encoder/test/encoder-ext-release-fence/src/main.cpp new file mode 100644 index 00000000..04a517ae --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-release-fence/src/main.cpp @@ -0,0 +1,711 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Per-frame RELEASE fence coverage, on a real device. + * + * WHAT IS UNDER TEST. VkVideoEncoderFrameFenceDescriptor::pReleaseFenceFd -- + * the export half of the per-frame acquire/release fence entry point. The + * library appends a binary, SYNC_FD-exportable semaphore to the frame's + * signal list so the submission that CONSUMES the input image signals it, + * then exports a SYNC_FD from it once that submission has been issued. + * + * WHY THIS CANNOT BE A NULL-BACKEND TEST, unlike its sibling + * encoder-ext-sync. The three things worth proving are all properties of a + * real queue: that the fd is a live sync_fd (>= 0), that it is NOT already + * signalled when it is handed over (which is what "exported from a PENDING + * signal" means, and what a fd of -1 would have told us instead), and that it + * becomes signalled once the GPU has read the input. A session with no device + * can express none of them. So this one needs a GPU, and says so: with no + * encode-capable device it exits 77, which CTest is configured to read as a + * SKIP rather than a pass. + * + * THE LEAK ASSERTION IS THE POINT OF THE 300-FRAME LOOP. The exported fd is + * the CALLER's to close -- the reverse of every other fd rule in this API, + * all of which govern handles the library is GIVEN. A rule stated in a header + * comment and not exercised is a rule that drifts, so this counts entries in + * /proc/self/fd across the whole run: an over-retaining library shows up as + * growth even though every fd this test itself receives is closed. + */ + +#include "vulkan_video_encoder_ext.h" + +#include "vk_video/vulkan_video_codec_h264std.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Same scrub, same reason, as the sibling tests. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + return; + } + g_failures++; + std::printf(" FAIL %s : %s\n", what, detail.c_str()); +} + +std::string I64(long long v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%lld", v); + return buf; +} + +// Live entries in /proc/self/fd, minus the handle the scan itself holds open. +int OpenFdCount() +{ + DIR* d = opendir("/proc/self/fd"); + if (d == nullptr) { + return -1; + } + int n = 0; + while (struct dirent* e = readdir(d)) { + if ((std::strcmp(e->d_name, ".") == 0) || + (std::strcmp(e->d_name, "..") == 0)) { + continue; + } + n++; + } + closedir(d); + return n - 1; +} + +// poll() for readability. -1 timeout blocks; 0 polls. Returns 1 signalled, +// 0 not yet, <0 error. A sync_fd becomes readable when its fence signals. +int PollSignalled(int fd, int timeoutMs) +{ + struct pollfd p = {}; + p.fd = fd; + p.events = POLLIN; + const int r = poll(&p, 1, timeoutMs); + if (r < 0) { + return -1; + } + if (r == 0) { + return 0; + } + return ((p.revents & POLLIN) != 0) ? 1 : -1; +} + +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; +const uint32_t kFrames = 300; + +struct DeviceFns { + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = + nullptr; + PFN_vkCreateSemaphore CreateSemaphore = nullptr; + PFN_vkDestroySemaphore DestroySemaphore = nullptr; + PFN_vkGetSemaphoreCounterValue GetSemaphoreCounterValue = nullptr; +}; + +bool LoadDeviceFns(VkInstance instance, VkDevice device, DeviceFns* fns) +{ + void* lib = dlopen("libvulkan.so.1", RTLD_NOW); + if (lib == nullptr) { + lib = dlopen("libvulkan.so", RTLD_NOW); + } + if (lib == nullptr) { + std::printf(" ERROR: dlopen(libvulkan) failed: %s\n", dlerror()); + return false; + } + auto gipa = (PFN_vkGetInstanceProcAddr)dlsym(lib, "vkGetInstanceProcAddr"); + if (gipa == nullptr) { + std::printf(" ERROR: no vkGetInstanceProcAddr\n"); + return false; + } + auto gdpa = (PFN_vkGetDeviceProcAddr)gipa(instance, "vkGetDeviceProcAddr"); + if (gdpa == nullptr) { + std::printf(" ERROR: no vkGetDeviceProcAddr\n"); + return false; + } +#define LOAD_DEV(name) \ + fns->name = (PFN_vk##name)gdpa(device, "vk" #name); \ + if (fns->name == nullptr) { \ + std::printf(" ERROR: missing vk" #name "\n"); \ + return false; \ + } + LOAD_DEV(CreateImage) + LOAD_DEV(DestroyImage) + LOAD_DEV(GetImageMemoryRequirements) + LOAD_DEV(AllocateMemory) + LOAD_DEV(FreeMemory) + LOAD_DEV(BindImageMemory) + LOAD_DEV(MapMemory) + LOAD_DEV(UnmapMemory) + LOAD_DEV(CreateSemaphore) + LOAD_DEV(DestroySemaphore) + LOAD_DEV(GetSemaphoreCounterValue) +#undef LOAD_DEV + fns->GetPhysicalDeviceMemoryProperties = + (PFN_vkGetPhysicalDeviceMemoryProperties)gipa( + instance, "vkGetPhysicalDeviceMemoryProperties"); + return (fns->GetPhysicalDeviceMemoryProperties != nullptr); +} + +// A host-written LINEAR NV12 image on the encoder's own device, registered as +// VK_IMAGE. Transfer-source usage only, so the registration routes STAGED -- +// the staging copy is then the submission that reads the input, and therefore +// the one that must signal the release fence. +struct InputImage { + VkImage image = VK_NULL_HANDLE; + VkDeviceMemory memory = VK_NULL_HANDLE; +}; + +bool CreateInputImage(const DeviceFns& fns, VkPhysicalDevice phys, + VkDevice device, InputImage* out, bool direct) +{ + // The direct arm needs an image the ENCODER can read: OPTIMAL tiling, + // VIDEO_ENCODE_SRC usage, a profile list at create time, and therefore + // device-local memory this test cannot host-fill. It encodes undefined + // content, deliberately -- the subject here is WHEN the release fence + // signals, not what the bitstream contains, and the staged arm already + // covers real pixels. + // THE CODEC-SPECIFIC PROFILE STRUCT IS PART OF THE PROFILE, not an + // optional decoration on it. A VkVideoProfileInfoKHR naming an H.264 + // encode operation is only a complete profile once a + // VkVideoEncodeH264ProfileInfoKHR is chained onto it + // (VUID-VkVideoProfileInfoKHR-videoCodecOperation-07181). Without it the + // profile list below describes no profile the session can be matched + // against, so vkCreateImage is asked about an image no session can read + // (VUID-VkImageCreateInfo-pNext-06811) and every encode that names the + // resulting view is incompatible with the bound session + // (VUID-vkCmdEncodeVideoKHR-pEncodeInfo-08206). + // + // HIGH because that is the profile the SESSION will use, not because it + // is the richest one available. The config below leaves |profile| at + // VK_VIDEO_ENCODER_PROFILE_DEFAULT, and the library derives profile_idc + // 100 for 8-bit 4:2:0 input under the default adaptive-transform mode. + // Naming a profile here that the session does not use is the same + // mismatch as naming none. + VkVideoEncodeH264ProfileInfoKHR h264Profile{ + VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_PROFILE_INFO_KHR}; + h264Profile.stdProfileIdc = STD_VIDEO_H264_PROFILE_IDC_HIGH; + VkVideoProfileInfoKHR profile{VK_STRUCTURE_TYPE_VIDEO_PROFILE_INFO_KHR}; + profile.pNext = &h264Profile; + profile.videoCodecOperation = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + profile.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + profile.lumaBitDepth = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + profile.chromaBitDepth = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + VkVideoProfileListInfoKHR profileList{ + VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR}; + profileList.profileCount = 1; + profileList.pProfiles = &profile; + + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.pNext = direct ? (const void*)&profileList : nullptr; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = direct ? VK_IMAGE_TILING_OPTIMAL + : VK_IMAGE_TILING_LINEAR; + ci.usage = direct + ? (VkImageUsageFlags)( + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT) + : (VkImageUsageFlags)VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ci.initialLayout = direct ? VK_IMAGE_LAYOUT_UNDEFINED + : VK_IMAGE_LAYOUT_PREINITIALIZED; + if (fns.CreateImage(device, &ci, nullptr, &out->image) != VK_SUCCESS) { + std::printf(" ERROR: vkCreateImage(LINEAR NV12) failed\n"); + return false; + } + + VkMemoryRequirements req{}; + fns.GetImageMemoryRequirements(device, out->image, &req); + + VkPhysicalDeviceMemoryProperties memProps{}; + fns.GetPhysicalDeviceMemoryProperties(phys, &memProps); + uint32_t typeIndex = UINT32_MAX; + const VkMemoryPropertyFlags want = + direct ? (VkMemoryPropertyFlags)VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT + : (VkMemoryPropertyFlags)(VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | + VK_MEMORY_PROPERTY_HOST_COHERENT_BIT); + for (uint32_t i = 0; i < memProps.memoryTypeCount; i++) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((memProps.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + std::printf(" ERROR: no host-visible memory type for a linear NV12 " + "image\n"); + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (fns.AllocateMemory(device, &ai, nullptr, &out->memory) != VK_SUCCESS) { + std::printf(" ERROR: vkAllocateMemory failed\n"); + return false; + } + if (fns.BindImageMemory(device, out->image, out->memory, 0) != VK_SUCCESS) { + std::printf(" ERROR: vkBindImageMemory failed\n"); + return false; + } + + // Real content, not zeros: a flat surface encodes to a degenerate + // bitstream and would make "the encode actually ran" hard to assert. + if (direct) { + return true; // device-local: nothing to map + } + void* mapped = nullptr; + if (fns.MapMemory(device, out->memory, 0, req.size, 0, &mapped) == + VK_SUCCESS) { + uint8_t* bytes = (uint8_t*)mapped; + for (VkDeviceSize i = 0; i < req.size; i++) { + bytes[i] = (uint8_t)((i * 7u) ^ (i >> 9)); + } + fns.UnmapMemory(device, out->memory); + } + return true; +} + +} // namespace + +int main(int argc, char** argv) +{ + // Which submission ends up consuming the input image is a REGISTRATION- + // time routing decision, and the two answers exercise different halves of + // the library's "was the input-consuming submit issued?" test: + // + // staged (default) transfer-source usage only -> the staging copy + // reads the input, and its submit is issued inline + // from StageInputFrame; the check reads + // inputCmdBuffer. + // direct VIDEO_ENCODE_SRC usage -> vkCmdEncodeVideoKHR + // reads the input directly and there is no staging + // copy at all, so the check has to fall through to + // encodeCmdBuffer->IsCommandBufferSubmitted(). + // + // Both are run, as two CTest cases, because a release fence that works + // only on the staged arm is a release fence that silently answers -1 for + // every zero-copy producer -- which is the arm this API exists for. + const bool direct = (argc > 1) && (std::strcmp(argv[1], "direct") == 0); + // A THIRD routing, and this one is about the pNext CHAIN rather than the + // queue: "chained" submits the same staged frame but hangs the fence + // descriptor off a VkVideoEncoderFrameSyncDescriptor's pNext instead of + // off the submit info directly. The header documents a flat, + // order-independent chain, so a fence honoured only in the leading + // position would make that documentation false -- and a node the sType + // gate accepts but the walk never reads is dead code that looks wired. + // Only a real export can tell those apart: fd >= 0, not yet signalled at + // handover, signalled afterwards. + const bool chained = (argc > 1) && (std::strcmp(argv[1], "chained") == 0); + std::printf("Encoder-ext per-frame release fence (real device, %s)\n", + direct ? "DIRECT encode input" + : (chained + ? "STAGED input, fence chained BEHIND a sync " + "descriptor" + : "STAGED input")); + std::printf("------------------------------------------------\n"); + + VkSharedBaseObj encoder; + if ((CreateVulkanVideoEncoderExt(encoder) != VK_SUCCESS) || !encoder) { + std::printf("SKIP: CreateVulkanVideoEncoderExt failed\n"); + return 77; + } + + VkVideoEncoderConfig config = {}; + config.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_CONFIG; + config.codec = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + config.encodeWidth = kWidth; + config.encodeHeight = kHeight; + config.inputFormat = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + config.inputWidth = kWidth; + config.inputHeight = kHeight; + config.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_CBR_BIT_KHR; + config.averageBitrate = 5000000; + config.maxBitrate = 5000000; + config.gopLength = 30; + // No reordering: the direct-encode submit is then issued inline with the + // frame that produced it, which keeps this test's expectations about WHEN + // the fd exists free of GOP scheduling. + config.consecutiveBFrames = 0; + config.idrPeriod = 30; + config.frameRateNum = 30; + config.frameRateDen = 1; + // -1 is auto-select. 0 is NOT a default here -- it names physical device + // index 0, which on a multi-ICD host is whatever enumerated first. + config.deviceId = -1; + // In-memory capture. Left FALSE the library writes out.264 and delivers + // EMPTY VkVideoEncodeResult records -- a real encode with nothing for + // this test to weigh, which is how a first run of this file mistook a + // working 6.4 MB bitstream for a broken one. + config.disableFileOutput = VK_TRUE; + + if (encoder->InitializeExt(config) != VK_SUCCESS) { + std::printf("SKIP: InitializeExt failed -- no encode-capable Vulkan " + "device on this host\n"); + return 77; + } + + VkInstance instance = encoder->GetVkInstance(); + VkDevice device = encoder->GetVkDevice(); + VkPhysicalDevice phys = encoder->GetVkPhysicalDevice(); + DeviceFns fns; + if (!LoadDeviceFns(instance, device, &fns)) { + std::printf("SKIP: could not load the Vulkan entry points this test " + "needs\n"); + return 77; + } + + InputImage input; + if (!CreateInputImage(fns, phys, device, &input, direct)) { + std::printf("SKIP: could not create the input image\n"); + return 77; + } + + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = kWidth; + desc.height = kHeight; + desc.tiling = direct ? VK_IMAGE_TILING_OPTIMAL + : VK_IMAGE_TILING_LINEAR; + // Declaring VIDEO_ENCODE_SRC is what routes the registration DIRECT: the + // library never grants access the caller did not declare, so leaving this + // at transfer-source is also how the staged arm gets selected. + desc.imageUsage = direct + ? (VkImageUsageFlags)( + VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR | + VK_IMAGE_USAGE_TRANSFER_SRC_BIT) + : (VkImageUsageFlags) + VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + desc.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + // Zero for a VK_IMAGE registration: the library did not perform the + // import, so it has no plane layouts to validate, and two declared planes + // both at offset 0 is exactly the shape it refuses as un-allocatable. + desc.planeCount = 0; + // A reused, locally allocated, host-written image: declaring FOREIGN here + // would make the staging copy take a queue-family acquire with no + // matching release. + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.defaultLayout = direct ? VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR + : VK_IMAGE_LAYOUT_PREINITIALIZED; + desc.existingImage = input.image; + + VkVideoEncoderResource resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + const VkVideoEncoderStatusCode regStatus = + encoder->RegisterImageResource(desc, 0, &resource, nullptr); + if ((regStatus != VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (resource == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + std::printf("SKIP: RegisterImageResource failed, status %d\n", + (int)regStatus); + return 77; + } + + // THE CALLER'S OWN SIGNAL SEMAPHORE -- the half of this entry point that + // has to be exercised against a real queue to mean anything. + // + // The library appends its release-fence semaphore to whatever signal list + // the frame ended up with, reading frame.pSignalSemaphores to copy the + // caller's entries across first. Every existing check on that block was a + // null-backend one (encoder-ext-sync drives the arm that returns before + // SetExternalInputFrame*), and this test supplied no signal semaphores at + // all -- so a clamp, a reorder or a dropped entry in the append would have + // turned nothing red anywhere in the tree, which is exactly the shape of + // the signal-zeroing defect this API's tests exist to catch. + // + // TIMELINE, because a timeline signal is observable from the host with no + // queue of our own (vkGetSemaphoreCounterValue) and is harmless left + // unwaited, whereas a binary semaphore signalled and never waited leaves + // the queue in an invalid state at teardown. + VkSemaphoreTypeCreateInfo semType{ + VK_STRUCTURE_TYPE_SEMAPHORE_TYPE_CREATE_INFO}; + semType.semaphoreType = VK_SEMAPHORE_TYPE_TIMELINE; + semType.initialValue = 0; + VkSemaphoreCreateInfo semCi{VK_STRUCTURE_TYPE_SEMAPHORE_CREATE_INFO}; + semCi.pNext = &semType; + VkSemaphore callerSignal = VK_NULL_HANDLE; + if (fns.CreateSemaphore(device, &semCi, nullptr, &callerSignal) != + VK_SUCCESS) { + std::printf("SKIP: could not create a timeline semaphore\n"); + return 77; + } + + // Baseline AFTER every one-time allocation, so the delta below is the + // per-frame behaviour and nothing else. + const int fdBefore = OpenFdCount(); + std::printf(" /proc/self/fd before the loop: %d\n", fdBefore); + + uint32_t submitted = 0; + uint32_t fencesReceived = 0; + uint32_t fencesUnsignalledAtHandover = 0; + uint32_t fencesSignalled = 0; + uint32_t capturedFrames = 0; + uint64_t capturedBytes = 0; + int firstBadFd = 0; + bool pollFailed = false; + + uint64_t lastSignalValue = 0; + for (uint32_t f = 0; f < kFrames; f++) { + int releaseFenceFd = 0x5EED; // must be overwritten by the library + // Monotonic, one step per frame, so the final counter value names + // exactly how many frames' signal lists survived the walk. + const uint64_t callerSignalValue = (uint64_t)f + 1; + + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + // Names NEITHER direction, so it changes no sync decision and cannot + // account for any difference in the numbers below. Its only job is to + // push the fence descriptor into the TRAILING chain position. + VkVideoEncoderFrameSyncDescriptor sync; + sync.pNext = &fence; + + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.pNext = chained ? (const void*)&sync : (const void*)&fence; + info.resource = resource; + info.frameId = f; + info.pts = f; + info.qpOverride = -1; + // UNDEFINED means "as declared at registration", i.e. PREINITIALIZED + // on every frame. That is the reused host-written LINEAR staging + // image shape the library's PREINITIALIZED -> TRANSFER_SRC_OPTIMAL + // barrier exists for, and the same shape Chromium's shmem staging + // path submits. + info.currentLayout = VK_IMAGE_LAYOUT_UNDEFINED; + info.signalSemaphoreCount = 1; + info.pSignalSemaphores = &callerSignal; + info.pSignalSemaphoreValues = &callerSignalValue; + + VkVideoEncoderStatusCode status = + encoder->SubmitRegisteredFrame(info, nullptr); + + // Admission-control backpressure is transient by contract: drain and + // retry the SAME call. + for (int retry = 0; + (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) && (retry < 2000); + retry++) { + VkVideoEncodeResult drained; + while (encoder->AcquireNextEncodedFrame(drained) == VK_SUCCESS) { + capturedFrames++; + capturedBytes += drained.bitstreamSize; + if (capturedFrames <= 3) { + std::printf(" [diag] frame %llu status=%d size=%u ptr=%p\n", + (unsigned long long)drained.frameId, + (int)drained.status, drained.bitstreamSize, + (const void*)drained.pBitstreamData); + } + encoder->ReleaseEncodedFrame(drained.frameId); + } + status = encoder->SubmitRegisteredFrame(info, nullptr); + if (status == VK_VIDEO_ENCODER_STATUS_NOT_READY) { + // BACK OFF. The drain above returns instantly when nothing has + // completed yet, so an unpaced retry burns all 2000 attempts + // in well under the time one 1080p frame takes to encode. The + // real build never noticed, because its per-frame + // PollSignalled(fd, 5000) blocks until the release fence + // signals and paces the loop for free -- which meant the + // pacing lived in the very thing a negative control removes. + // A control that dies of its own retry budget at frame 10 of + // 300 cannot tell -1-always from -1-for-ten-frames, which is + // the only thing it exists to tell apart. + struct timespec ts = {0, 1000000}; // 1 ms + nanosleep(&ts, nullptr); + } + } + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + Check(false, "SubmitRegisteredFrame", + "frame " + I64(f) + " status " + I64((long long)status)); + break; + } + submitted++; + lastSignalValue = callerSignalValue; + + // A BRANCH, not a `continue`. The per-iteration drain below is what + // keeps admission control fed, and skipping it starved the loop: a + // build with the export disabled -- the negative control this test's + // discriminating power rests on -- died at frame 10 of 300 with + // VK_VIDEO_ENCODER_STATUS_NOT_READY instead of demonstrating -1 across + // the whole run. A control that stops at 3% cannot tell "answers -1 + // always" apart from "answers -1 for ten frames", which is the only + // thing it was there to tell apart. + if (releaseFenceFd < 0) { + if (firstBadFd == 0) { + firstBadFd = (int)f + 1; + } + } else { + Check(releaseFenceFd != 0x5EED, "pReleaseFenceFd was written", + "library left the caller's sentinel in place"); + fencesReceived++; + + // Not-yet-signalled at handover is the strong form of "exported + // from a PENDING signal, not from an already-signalled + // semaphore". It is inherently a race with the GPU, so it is + // COUNTED rather than asserted per frame; the assertion is a + // floor on the count. + if (PollSignalled(releaseFenceFd, 0) == 0) { + fencesUnsignalledAtHandover++; + } + + const int signalled = PollSignalled(releaseFenceFd, 5000); + if (signalled == 1) { + fencesSignalled++; + } else { + pollFailed = true; + Check(false, "release fence signalled", + "frame " + I64(f) + " poll returned " + I64(signalled)); + } + + // The exported fd is the CALLER's. Closing it here is not + // tidiness, it is the contract the header states, and the + // fd-count assertion below is what keeps that statement honest. + close(releaseFenceFd); + } + + VkVideoEncodeResult result; + while (encoder->AcquireNextEncodedFrame(result) == VK_SUCCESS) { + capturedFrames++; + capturedBytes += result.bitstreamSize; + if (capturedFrames <= 3) { + std::printf(" [diag] frame %llu status=%d size=%u ptr=%p\n", + (unsigned long long)result.frameId, + (int)result.status, result.bitstreamSize, + (const void*)result.pBitstreamData); + } + encoder->ReleaseEncodedFrame(result.frameId); + } + if (pollFailed) { + break; + } + } + + encoder->DrainPendingFrames(); + VkVideoEncodeResult tail; + while (encoder->AcquireNextEncodedFrame(tail) == VK_SUCCESS) { + capturedFrames++; + capturedBytes += tail.bitstreamSize; + encoder->ReleaseEncodedFrame(tail.frameId); + } + + uint64_t callerSignalCounter = 0; + if (fns.GetSemaphoreCounterValue(device, callerSignal, + &callerSignalCounter) != VK_SUCCESS) { + callerSignalCounter = UINT64_MAX; // reported, never silently passed + } + + const int fdAfter = OpenFdCount(); + std::printf(" /proc/self/fd after the loop: %d\n", fdAfter); + std::printf(" submitted=%u fences>=0=%u signalled=%u " + "unsignalled-at-handover=%u\n", + submitted, fencesReceived, fencesSignalled, + fencesUnsignalledAtHandover); + std::printf(" unsignalled-at-handover ratio: %u/%u (%.1f%%)\n", + fencesUnsignalledAtHandover, fencesReceived, + (fencesReceived != 0) + ? (100.0 * (double)fencesUnsignalledAtHandover / + (double)fencesReceived) + : 0.0); + std::printf(" caller signal timeline: value=%llu, last requested=%llu\n", + (unsigned long long)callerSignalCounter, + (unsigned long long)lastSignalValue); + std::printf(" captured frames=%u bitstream bytes=%llu\n", + capturedFrames, (unsigned long long)capturedBytes); + if (firstBadFd != 0) { + std::printf(" first frame answering -1: %d\n", firstBadFd - 1); + } + + Check(submitted == kFrames, "all frames submitted", + I64(submitted) + " of " + I64(kFrames)); + Check(fencesReceived == kFrames, "every frame returned a release fd >= 0", + I64(fencesReceived) + " of " + I64(kFrames)); + Check(fencesSignalled == fencesReceived, + "every release fence became signalled", + I64(fencesSignalled) + " of " + I64(fencesReceived)); + // A FLOOR, not "> 0". The count exists to rule out "exported after the + // fact" -- an export moved behind a fence wait would come back already + // signalled on nearly every frame, and "> 0" passes that as long as ONE + // frame races ahead, i.e. it passes the exact state it was written to + // forbid. Half the run is far below what the property actually produces + // (298/300 staged, 299/300 direct, as measured) and far above what a + // regressed export could reach. + Check(fencesUnsignalledAtHandover >= (kFrames / 2), + "most fences were still unsignalled at handover", + I64(fencesUnsignalledAtHandover) + " of " + I64(fencesReceived) + + ", floor " + I64(kFrames / 2) + + " -- an export that did not ride a pending submit comes back " + "already signalled"); + // The caller's raw signal array reached the REAL queue and was not + // clamped, reordered or dropped by the release-fence append. Zero here + // means the library submitted without it: the silent-hang defect. + Check(callerSignalCounter == lastSignalValue, + "the caller's signal semaphore was signalled on every frame", + "timeline at " + I64((long long)callerSignalCounter) + + ", last requested " + I64((long long)lastSignalValue)); + Check(capturedFrames > 0, "the encode actually produced frames", + I64(capturedFrames) + " captured"); + Check(capturedBytes > 0, "the encode actually produced a bitstream", + I64((long long)capturedBytes) + " bytes"); + Check((fdBefore >= 0) && (fdAfter >= 0) && (fdAfter <= fdBefore), + "no fd growth across the run", + "before " + I64(fdBefore) + ", after " + I64(fdAfter)); + + // Order matters and the header states it: unregister BEFORE destroying + // the image, and destroy the image BEFORE releasing the encoder -- the + // VkDevice these objects live on is the encoder's, and it goes away with + // it. + encoder->UnregisterImageResource(resource); + fns.DestroySemaphore(device, callerSignal, nullptr); + fns.DestroyImage(device, input.image, nullptr); + fns.FreeMemory(device, input.memory, nullptr); + encoder = nullptr; + + std::printf("------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-ext-sync/CMakeLists.txt b/vk_video_encoder/test/encoder-ext-sync/CMakeLists.txt new file mode 100644 index 00000000..7f07a87e --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-sync/CMakeLists.txt @@ -0,0 +1,84 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_ext_sync_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, not the shared one: the internal header's +# seam functions are deliberately not exported from libvkvideo-encoder.so, so +# only the archive can satisfy them. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # This test includes vulkan_video_encoder_ext_internal.h, which is not on + # the library target's interface. Naming the directory here is what a + # legitimate internal consumer does, and what a client cannot. + ${VULKAN_VIDEO_ENCODER_INTERNAL_INCLUDE} + # The public encoder header reaches VkCodecUtils/VkVideoRefCountBase.h, + # which lives under the shared common-libs root. + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# NOTE ON CTest SEMANTICS, matching the sibling library tests: 0 means every +# assertion held, 1 means an assertion failed, and 2 means the session could +# not be stood up at all -- deliberately a CTest FAILURE and not a skip, +# because a host that could not run this must not report the per-direction +# override rule as verified. There is no GPU, driver or display dependence +# here: the session has no device by construction. +enable_testing() +add_test(NAME EncoderExtPerDirectionSyncOverride + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderExtPerDirectionSyncOverride PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_ext_sync_test: Configured") diff --git a/vk_video_encoder/test/encoder-ext-sync/src/main.cpp b/vk_video_encoder/test/encoder-ext-sync/src/main.cpp new file mode 100644 index 00000000..7fac8f45 --- /dev/null +++ b/vk_video_encoder/test/encoder-ext-sync/src/main.cpp @@ -0,0 +1,729 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * Per-direction semaphore-override coverage for SubmitRegisteredFrame's + * chained-descriptor walk. + * + * THE DEFECT THIS PINS. The walk resolves a frame's registered wait ids and + * signal ids into two lists, then decides whether the submit gets those or the + * caller's own raw VkSemaphore arrays. THE DECISION IS PER DIRECTION. Deciding + * once for both together -- `if (!resolvedWait.empty() || !resolvedSignal.empty())` + * -- and overwriting BOTH makes a chain naming only waits set + * signalSemaphoreCount to zero and throw the caller's signal semaphores away. + * + * That is not a corner. It is the shape Chromium submits on every fenced + * frame: media/gpu/vulkan/vulkan_video_encode_accelerator.cc chains a + * VkVideoEncoderFrameFenceDescriptor carrying an acquire fence and nothing + * else, and separately fills the submit info's pSignalSemaphores from its own + * end_semaphores. Whoever waits on those end semaphores waits forever, which + * presents as a hung tab rather than an error. + * + * WHY A NULL BACKEND. The rule is a decision about two pointers and two + * counts, taken before any Vulkan object is touched, and the public surface + * cannot see it: SubmitRegisteredFrame consumes the arrays and answers a + * status. So this drives the REAL walk on a null-backend session + * (vulkan_video_encoder_ext_internal.h) and reads back what the submit was + * handed. No device, no queue, no encoder -- and therefore no hardware + * dependence in either direction, which is what makes it deterministic. + * + * A registered semaphore is what makes a chain resolve to something, and the + * public RegisterSemaphore needs a device to create and import one, so the + * ids here come from VkEncInstallTestSemaphore -- sentinel handles that are + * stored, compared and submitted, never dereferenced. Each is uninstalled + * before the session drops, because Deinitialize would otherwise try to + * destroy it through a dispatch table this session never populated. + */ + +#include "vulkan_video_encoder_ext_internal.h" + +// The public header reaches the Xlib platform headers, whose macros collide +// with ordinary identifiers. Scrub them before anything else sees them -- the +// same block, for the same reason, as the sibling library test TUs. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include + +namespace { + +int g_failures = 0; +int g_checks = 0; +const char* g_currentCase = ""; + +void Check(bool ok, const char* what, const std::string& detail) +{ + g_checks++; + if (ok) { + return; + } + g_failures++; + std::printf(" FAIL [%s] %s : %s\n", g_currentCase, what, detail.c_str()); +} + +std::string U64(uint64_t v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%llu", (unsigned long long)v); + return buf; +} + +void CheckEqU32(uint32_t got, uint32_t want, const char* what) +{ + Check(got == want, what, "got " + U64(got) + ", want " + U64(want)); +} + +// A NON-DISPATCHABLE HANDLE IS NOT A POINTER, and only looks like one on +// 64-bit. VK_DEFINE_NON_DISPATCHABLE_HANDLE resolves to a pointer type there +// and to uint64_t on a 32-bit build, so formatting one with %p compiles on the +// one ABI and fails on the other. Print the 64-bit value, which is what the +// handle is on both. +std::string Handle(uint64_t h) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "0x%llx", (unsigned long long)h); + return buf; +} + +void CheckEqSem(VkSemaphore got, VkSemaphore want, const char* what) +{ + Check(got == want, what, + "got " + Handle((uint64_t)got) + ", want " + Handle((uint64_t)want)); +} + +void CheckEqU64(uint64_t got, uint64_t want, const char* what) +{ + Check(got == want, what, "got " + U64(got) + ", want " + U64(want)); +} + +// Sentinel VkSemaphore values. Never dereferenced: the submit terminates at +// the null backend, which records the handles and enqueues. +VkSemaphore Sentinel(uintptr_t v) +{ + return (VkSemaphore)v; +} + +// One null-backend session with a VK_IMAGE registration, torn down in the +// order the internal header requires. +class Session { +public: + bool Open() + { + if ((CreateVulkanVideoEncoderExt(m_encoder) != VK_SUCCESS) || + !m_encoder) { + std::printf(" ERROR: CreateVulkanVideoEncoderExt failed\n"); + return false; + } + if (VkEncInstallNullBackend(m_encoder.get(), &m_backend) != VK_SUCCESS) { + std::printf(" ERROR: VkEncInstallNullBackend failed\n"); + return false; + } + VkVideoEncoderExternalImageDescriptor desc = {}; + desc.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_EXTERNAL_IMAGE_DESCRIPTOR; + desc.handleType = VK_VIDEO_ENCODER_EXTERNAL_HANDLE_TYPE_VK_IMAGE; + desc.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + desc.width = 640; + desc.height = 360; + desc.residency = VK_VIDEO_ENCODER_INPUT_RESIDENCY_LOCAL; + desc.existingImage = (VkImage)(uintptr_t)0xA110C8ED; + if ((m_encoder->RegisterImageResource(desc, 0, &m_resource, nullptr) != + VK_VIDEO_ENCODER_STATUS_SUCCESS) || + (m_resource == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + std::printf(" ERROR: RegisterImageResource failed\n"); + return false; + } + return true; + } + + ~Session() + { + for (size_t i = 0; i < m_testSemaphores.size(); i++) { + VkEncUninstallTestSemaphore(m_encoder.get(), m_testSemaphores[i]); + } + } + + // A registered-semaphore id a FrameSyncDescriptor can name. + VkVideoEncoderResource InstallSemaphore(VkSemaphore sentinel) + { + VkVideoEncoderResource id = VK_VIDEO_ENCODER_RESOURCE_NULL; + if (VkEncInstallTestSemaphore(m_encoder.get(), sentinel, &id) != VK_SUCCESS) { + std::printf(" ERROR: VkEncInstallTestSemaphore failed\n"); + return VK_VIDEO_ENCODER_RESOURCE_NULL; + } + m_testSemaphores.push_back(id); + return id; + } + + VkVideoEncoderFrameSubmitInfo SubmitInfoFor(uint64_t frameId) const + { + VkVideoEncoderFrameSubmitInfo info = {}; + info.sType = VK_VIDEO_ENCODER_STRUCTURE_TYPE_FRAME_PARAMS; + info.resource = m_resource; + info.frameId = frameId; + info.qpOverride = -1; + return info; + } + + VulkanVideoEncoderExt* Get() const { return m_encoder.get(); } + +private: + VkEncNullBackendState m_backend; + VkSharedBaseObj m_encoder; + VkVideoEncoderResource m_resource = VK_VIDEO_ENCODER_RESOURCE_NULL; + std::vector m_testSemaphores; +}; + +// Submit |info| and hand back what the submit was actually given. +bool SubmitAndProbe(Session& s, + VkVideoEncoderFrameSubmitInfo& info, + VkEncSubmitSyncProbe* outProbe) +{ + const VkVideoEncoderStatusCode status = + s.Get()->SubmitRegisteredFrame(info, nullptr); + if (status != VK_VIDEO_ENCODER_STATUS_SUCCESS) { + Check(false, "SubmitRegisteredFrame", + "status " + U64((uint64_t)status)); + return false; + } + if (VkEncProbeLastSubmitSync(s.Get(), outProbe) != + VK_VIDEO_ENCODER_STATUS_SUCCESS) { + Check(false, "VkEncProbeLastSubmitSync", "call failed"); + return false; + } + Check(outProbe->recorded == VK_TRUE, "probe.recorded", + "no submit reached the null backend"); + return outProbe->recorded == VK_TRUE; +} + +// Caller-supplied raw arrays, shared by the cases below. +VkSemaphore g_callerWait[2] = {Sentinel(0x1001), Sentinel(0x1002)}; +uint64_t g_callerWaitValues[2] = {11, 12}; +VkSemaphore g_callerSignal[2] = {Sentinel(0x2001), Sentinel(0x2002)}; +uint64_t g_callerSignalValues[2] = {21, 22}; + +void FillCallerArrays(VkVideoEncoderFrameSubmitInfo& info) +{ + info.waitSemaphoreCount = 2; + info.pWaitSemaphores = g_callerWait; + info.pWaitSemaphoreValues = g_callerWaitValues; + info.signalSemaphoreCount = 2; + info.pSignalSemaphores = g_callerSignal; + info.pSignalSemaphoreValues = g_callerSignalValues; +} + +void ExpectCallerWaitSurvived(const VkEncSubmitSyncProbe& p) +{ + CheckEqU32(p.waitCount, 2, "wait count is the caller's"); + CheckEqSem(p.waitSemaphores[0], g_callerWait[0], "wait[0] is the caller's"); + CheckEqSem(p.waitSemaphores[1], g_callerWait[1], "wait[1] is the caller's"); + CheckEqU64(p.waitValues[0], g_callerWaitValues[0], "waitValue[0]"); + CheckEqU64(p.waitValues[1], g_callerWaitValues[1], "waitValue[1]"); +} + +void ExpectCallerSignalSurvived(const VkEncSubmitSyncProbe& p) +{ + CheckEqU32(p.signalCount, 2, "signal count is the caller's"); + CheckEqSem(p.signalSemaphores[0], g_callerSignal[0], + "signal[0] is the caller's"); + CheckEqSem(p.signalSemaphores[1], g_callerSignal[1], + "signal[1] is the caller's"); + CheckEqU64(p.signalValues[0], g_callerSignalValues[0], "signalValue[0]"); + CheckEqU64(p.signalValues[1], g_callerSignalValues[1], "signalValue[1]"); +} + +//============================================================================= +// Case 1 -- THE REGRESSION. A chain that names only waits must not touch the +// signal direction. This is the shape Chromium submits on every fenced frame, +// and it is the case the combined test broke. +//============================================================================= +void CaseAcquireOnlyChainKeepsCallerSignals(Session& s) +{ + g_currentCase = "AcquireOnlyChainKeepsCallerSignals"; + const VkSemaphore kRegistered = Sentinel(0x3001); + const VkVideoEncoderResource id = s.InstallSemaphore(kRegistered); + if (id == VK_VIDEO_ENCODER_RESOURCE_NULL) { + Check(false, "InstallSemaphore", "setup failed"); + return; + } + const uint64_t kWaitValue = 77; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.waitCount = 1; + sync.pWaitSemaphores = &id; + sync.pWaitValues = &kWaitValue; + // signalCount stays 0: the chain says nothing about the signal direction. + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(101); + FillCallerArrays(info); + info.pNext = &sync; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + // The wait direction WAS named, so the registered id replaces the raw one. + CheckEqU32(p.waitCount, 1, "wait count is the resolved one"); + CheckEqSem(p.waitSemaphores[0], kRegistered, "wait[0] is the resolved one"); + CheckEqU64(p.waitValues[0], kWaitValue, "waitValue[0] is the resolved one"); + // The signal direction was NOT named. The caller's list must survive + // intact -- this is the assertion the defect failed. + ExpectCallerSignalSurvived(p); +} + +//============================================================================= +// Case 2 -- the mirror image. A chain naming only signals must not touch the +// wait direction. +//============================================================================= +void CaseSignalOnlyChainKeepsCallerWaits(Session& s) +{ + g_currentCase = "SignalOnlyChainKeepsCallerWaits"; + const VkSemaphore kRegistered = Sentinel(0x3002); + const VkVideoEncoderResource id = s.InstallSemaphore(kRegistered); + if (id == VK_VIDEO_ENCODER_RESOURCE_NULL) { + Check(false, "InstallSemaphore", "setup failed"); + return; + } + const uint64_t kSignalValue = 88; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.signalCount = 1; + sync.pSignalSemaphores = &id; + sync.pSignalValues = &kSignalValue; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(102); + FillCallerArrays(info); + info.pNext = &sync; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + CheckEqU32(p.signalCount, 1, "signal count is the resolved one"); + CheckEqSem(p.signalSemaphores[0], kRegistered, + "signal[0] is the resolved one"); + CheckEqU64(p.signalValues[0], kSignalValue, + "signalValue[0] is the resolved one"); + ExpectCallerWaitSurvived(p); +} + +//============================================================================= +// Case 3 -- both directions named: both are replaced outright. The registered +// ids win; nothing is merged with the raw arrays. +//============================================================================= +void CaseBothSuppliedReplacesBothDirections(Session& s) +{ + g_currentCase = "BothSuppliedReplacesBothDirections"; + const VkSemaphore kRegWait = Sentinel(0x3003); + const VkSemaphore kRegSignal = Sentinel(0x3004); + const VkVideoEncoderResource waitId = s.InstallSemaphore(kRegWait); + const VkVideoEncoderResource signalId = s.InstallSemaphore(kRegSignal); + if ((waitId == VK_VIDEO_ENCODER_RESOURCE_NULL) || + (signalId == VK_VIDEO_ENCODER_RESOURCE_NULL)) { + Check(false, "InstallSemaphore", "setup failed"); + return; + } + const uint64_t kWaitValue = 55; + const uint64_t kSignalValue = 66; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.waitCount = 1; + sync.pWaitSemaphores = &waitId; + sync.pWaitValues = &kWaitValue; + sync.signalCount = 1; + sync.pSignalSemaphores = &signalId; + sync.pSignalValues = &kSignalValue; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(103); + FillCallerArrays(info); + info.pNext = &sync; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + CheckEqU32(p.waitCount, 1, "wait count is the resolved one"); + CheckEqSem(p.waitSemaphores[0], kRegWait, "wait[0] is the resolved one"); + CheckEqU64(p.waitValues[0], kWaitValue, "waitValue[0]"); + CheckEqU32(p.signalCount, 1, "signal count is the resolved one"); + CheckEqSem(p.signalSemaphores[0], kRegSignal, + "signal[0] is the resolved one"); + CheckEqU64(p.signalValues[0], kSignalValue, "signalValue[0]"); +} + +//============================================================================= +// Case 4 -- control. No chain at all: both raw arrays reach the submit +// untouched. Without this, cases 1 and 2 could pass on a build that ignored +// the chain entirely. +//============================================================================= +void CaseNoChainKeepsBothCallerArrays(Session& s) +{ + g_currentCase = "NoChainKeepsBothCallerArrays"; + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(104); + FillCallerArrays(info); + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + ExpectCallerWaitSurvived(p); + ExpectCallerSignalSurvived(p); +} + +//============================================================================= +// Case 5 -- an acquire-only chain with NO caller signals must not invent one. +// The per-direction rule leaves the untouched direction exactly as it found +// it, empty included. +//============================================================================= +void CaseAcquireOnlyChainWithNoCallerSignals(Session& s) +{ + g_currentCase = "AcquireOnlyChainWithNoCallerSignals"; + const VkSemaphore kRegistered = Sentinel(0x3005); + const VkVideoEncoderResource id = s.InstallSemaphore(kRegistered); + if (id == VK_VIDEO_ENCODER_RESOURCE_NULL) { + Check(false, "InstallSemaphore", "setup failed"); + return; + } + const uint64_t kWaitValue = 99; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.waitCount = 1; + sync.pWaitSemaphores = &id; + sync.pWaitValues = &kWaitValue; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(105); + info.pNext = &sync; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + CheckEqU32(p.waitCount, 1, "wait count is the resolved one"); + CheckEqU32(p.signalCount, 0, "signal count stays empty"); +} + +//============================================================================= +// Case 6 -- the Chromium chain SHAPE: a fence descriptor ahead of the sync +// descriptor. The walk must traverse past the fence descriptor and still apply +// the per-direction rule to what follows. The fence carries acquireFenceFd = +// -1 (no fence) because importing a real one needs a device this session does +// not have; what this pins is the walk and the override, not the import. +//============================================================================= +void CaseFenceThenSyncChainKeepsCallerSignals(Session& s) +{ + g_currentCase = "FenceThenSyncChainKeepsCallerSignals"; + const VkSemaphore kRegistered = Sentinel(0x3006); + const VkVideoEncoderResource id = s.InstallSemaphore(kRegistered); + if (id == VK_VIDEO_ENCODER_RESOURCE_NULL) { + Check(false, "InstallSemaphore", "setup failed"); + return; + } + const uint64_t kWaitValue = 123; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.waitCount = 1; + sync.pWaitSemaphores = &id; + sync.pWaitValues = &kWaitValue; + + int releaseFenceFd = 12345; // must be overwritten by the library + VkVideoEncoderFrameFenceDescriptor fence; + fence.pNext = &sync; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(106); + FillCallerArrays(info); + info.pNext = &fence; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + CheckEqU32(p.waitCount, 1, "wait count is the resolved one"); + CheckEqSem(p.waitSemaphores[0], kRegistered, "wait[0] is the resolved one"); + ExpectCallerSignalSurvived(p); + // Not the subject of this file, but the walk reached the fence descriptor + // only if this was answered. + // + // -1 here is the CORRECT answer and stays correct now that the release + // export is implemented: this is a null-backend session, which has no + // queue, so nothing can carry the signal a SYNC_FD export needs to be + // pending. The library declines to create a release semaphore at all on + // that arm. What proves the export itself works is the real-device + // sibling, vk_video_encoder/test/encoder-ext-release-fence, which asserts + // fd >= 0 and that it becomes signalled -- a claim no device-free test + // can make. + Check(releaseFenceFd == -1, "pReleaseFenceFd answered", + "got " + U64((uint64_t)(int64_t)releaseFenceFd) + ", want -1"); +} + +//============================================================================= +// Case 7 -- THE CHAINING CONTRACT, TRAILING POSITION. The fence descriptor +// hangs off a VkVideoEncoderFrameSyncDescriptor's pNext. That is the shape the +// header names, and this is its only in-tree coverage: +// case 6 puts the fence FIRST, the hardware sibling puts it alone. +// +// The assertion is deliberately an EFFECT and not a status. A descriptor the +// sType gate accepts and the walk then never reads is dead code that looks +// wired, and it would pass any check that only asked whether the submit +// succeeded. The store through pReleaseFenceFd happens in the fence branch of +// the walk and nowhere else, so an unchanged sentinel means this node was +// skipped. +//============================================================================= +void CaseSyncThenFenceChainIsRead(Session& s) +{ + g_currentCase = "SyncThenFenceChainIsRead"; + const VkSemaphore kRegistered = Sentinel(0x3007); + const VkVideoEncoderResource id = s.InstallSemaphore(kRegistered); + if (id == VK_VIDEO_ENCODER_RESOURCE_NULL) { + Check(false, "InstallSemaphore", "setup failed"); + return; + } + const uint64_t kWaitValue = 321; + + int releaseFenceFd = 12345; // must be overwritten by the library + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.pNext = &fence; + sync.waitCount = 1; + sync.pWaitSemaphores = &id; + sync.pWaitValues = &kWaitValue; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(107); + FillCallerArrays(info); + info.pNext = &sync; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + // The sync descriptor was honoured in the LEADING position ... + CheckEqU32(p.waitCount, 1, "wait count is the resolved one"); + CheckEqSem(p.waitSemaphores[0], kRegistered, "wait[0] is the resolved one"); + CheckEqU64(p.waitValues[0], kWaitValue, "waitValue[0] is the resolved one"); + ExpectCallerSignalSurvived(p); + // ... and the fence descriptor BEHIND it was read, not walked past. -1 is + // the correct answer on a session with no queue, for the reason spelled + // out in case 6; what matters here is that 12345 did not survive. + Check(releaseFenceFd == -1, "the fence behind the sync descriptor was read", + "pReleaseFenceFd left at " + U64((uint64_t)(int64_t)releaseFenceFd) + + " -- the walk never read this node"); +} + +//============================================================================= +// Case 8 -- the shape the only real caller sends: a fence descriptor ALONE on +// the submit info's pNext, no sync descriptor anywhere in the chain. That is +// media/gpu/vulkan/vulkan_video_encode_accelerator.cc, which sets +// input.pNext = &fence_desc. Read literally, the header's old wording made +// this illegal. It is not, and neither raw array may be disturbed by it. +//============================================================================= +void CaseFenceAloneChainIsRead(Session& s) +{ + g_currentCase = "FenceAloneChainIsRead"; + int releaseFenceFd = 12345; // must be overwritten by the library + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(108); + FillCallerArrays(info); + info.pNext = &fence; + + VkEncSubmitSyncProbe p; + if (!SubmitAndProbe(s, info, &p)) { + return; + } + Check(releaseFenceFd == -1, "the lone fence descriptor was read", + "pReleaseFenceFd left at " + U64((uint64_t)(int64_t)releaseFenceFd) + + " -- the walk never read this node"); + // A chain that names NEITHER direction is not a chain that names both as + // empty: both caller arrays reach the submit intact. + ExpectCallerWaitSurvived(p); + ExpectCallerSignalSurvived(p); +} + +//============================================================================= +// Case 9 -- the fence node's pNext is genuinely FOLLOWED, and the ABI's +// refuse-don't-skip rule holds in the trailing position. An unknown sType put +// behind the fence descriptor is reachable only through fence->pNext, so the +// refusal proves that field was read rather than the walk stopping at the +// fence. The store through pReleaseFenceFd must still have happened: the +// header promises a defined value on every exit path, the failing ones +// included, so that a caller cannot mistake stack garbage for an fd. +//============================================================================= +void CaseUnknownSTypeBehindFenceIsRefused(Session& s) +{ + g_currentCase = "UnknownSTypeBehindFenceIsRefused"; + // Every public struct opens with {sType, pNext}, which is what the walk + // reads a link through -- so an unknown link needs nothing more than that. + struct BogusLink { + VkVideoEncoderStructureType sType; + const void* pNext; + }; + BogusLink bogus; + bogus.sType = (VkVideoEncoderStructureType)0x5645FFFF; // in the private + bogus.pNext = nullptr; // 'VE' block, unused + + int releaseFenceFd = 12345; // must be overwritten by the library + VkVideoEncoderFrameFenceDescriptor fence; + fence.pNext = &bogus; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.pNext = &fence; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(109); + FillCallerArrays(info); + info.pNext = &sync; + + const VkVideoEncoderStatusCode status = + s.Get()->SubmitRegisteredFrame(info, nullptr); + Check(status == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "unknown sType behind the fence descriptor is refused", + "status " + U64((uint64_t)status) + ", want " + + U64((uint64_t)VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN)); + Check(releaseFenceFd == -1, + "the fence node was read before that refusal", + "pReleaseFenceFd left at " + U64((uint64_t)(int64_t)releaseFenceFd)); +} + +//============================================================================= +// Case 10 -- the MIRROR of case 9, and the ordering in which the header's +// promise is easiest to break. Case 9 puts the bogus link BEHIND the fence descriptor, +// which is the one ordering in which "pReleaseFenceFd holds a defined value on +// every exit path" happened to hold by accident: the walk reached the fence +// node, stored -1, and only then refused. Put the bogus link AHEAD of the +// fence -- a shape this header explicitly blesses, since the chain is flat and +// order-independent -- and the walk refuses before it ever sees the fence +// node. A caller who declared `int fd;` on the strength of the header then +// close(2)s stack garbage, shutting an unrelated descriptor in its own +// process. The store is a pre-pass over the chain for exactly this reason. +//============================================================================= +void CaseUnknownSTypeAheadOfFenceStillWritesTheFd(Session& s) +{ + g_currentCase = "UnknownSTypeAheadOfFenceStillWritesTheFd"; + struct BogusLink { + VkVideoEncoderStructureType sType; + const void* pNext; + }; + + int releaseFenceFd = 12345; // stands in for an uninitialised stack local + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + BogusLink bogus; + bogus.sType = (VkVideoEncoderStructureType)0x5645FFFE; + bogus.pNext = &fence; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(110); + FillCallerArrays(info); + info.pNext = &bogus; + + const VkVideoEncoderStatusCode status = + s.Get()->SubmitRegisteredFrame(info, nullptr); + Check(status == VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN, + "unknown sType ahead of the fence descriptor is refused", + "status " + U64((uint64_t)status) + ", want " + + U64((uint64_t)VK_VIDEO_ENCODER_STATUS_ERROR_STRUCTURE_TYPE_UNKNOWN)); + Check(releaseFenceFd == -1, + "pReleaseFenceFd was written despite refusing before the fence node", + "pReleaseFenceFd left at " + U64((uint64_t)(int64_t)releaseFenceFd) + + " -- a caller following the header would close(2) that"); +} + +//============================================================================= +// Case 11 -- an UNRESOLVABLE registered id ahead of the fence descriptor. Same +// promise, a different refusal: this one is raised inside the sync node's own +// resolve loop rather than by the sType gate, so it proves the pre-pass covers +// the RESOURCE_UNKNOWN exit too and not just the STRUCTURE_TYPE_UNKNOWN one. +//============================================================================= +void CaseUnresolvableIdAheadOfFenceStillWritesTheFd(Session& s) +{ + g_currentCase = "UnresolvableIdAheadOfFenceStillWritesTheFd"; + int releaseFenceFd = 12345; + VkVideoEncoderFrameFenceDescriptor fence; + fence.acquireFenceFd = -1; + fence.pReleaseFenceFd = &releaseFenceFd; + + // Never installed, so it cannot resolve. Carries the semaphore tag bit so + // it is rejected by the registry lookup rather than by the tag check. + const VkVideoEncoderResource bogusId = + (VkVideoEncoderResource)(0x8000000000000000ull | 0x0000000100000777ull); + const uint64_t kWaitValue = 5; + + VkVideoEncoderFrameSyncDescriptor sync; + sync.pNext = &fence; + sync.waitCount = 1; + sync.pWaitSemaphores = &bogusId; + sync.pWaitValues = &kWaitValue; + + VkVideoEncoderFrameSubmitInfo info = s.SubmitInfoFor(111); + FillCallerArrays(info); + info.pNext = &sync; + + const VkVideoEncoderStatusCode status = + s.Get()->SubmitRegisteredFrame(info, nullptr); + Check(status == VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN, + "an unresolvable wait id is refused", + "status " + U64((uint64_t)status) + ", want " + + U64((uint64_t)VK_VIDEO_ENCODER_STATUS_ERROR_RESOURCE_UNKNOWN)); + Check(releaseFenceFd == -1, + "pReleaseFenceFd was written despite refusing in the node ahead", + "pReleaseFenceFd left at " + U64((uint64_t)(int64_t)releaseFenceFd)); +} + +} // namespace + +int main(int argc, char** argv) +{ + (void)argc; + (void)argv; + std::printf("Encoder-ext per-direction sync override\n"); + std::printf("---------------------------------------\n"); + + Session session; + if (!session.Open()) { + std::printf("RESULT: COULD-NOT-RUN (session setup failed)\n"); + return 2; + } + + CaseAcquireOnlyChainKeepsCallerSignals(session); + CaseSignalOnlyChainKeepsCallerWaits(session); + CaseBothSuppliedReplacesBothDirections(session); + CaseNoChainKeepsBothCallerArrays(session); + CaseAcquireOnlyChainWithNoCallerSignals(session); + CaseFenceThenSyncChainKeepsCallerSignals(session); + CaseSyncThenFenceChainIsRead(session); + CaseFenceAloneChainIsRead(session); + CaseUnknownSTypeBehindFenceIsRefused(session); + CaseUnknownSTypeAheadOfFenceStillWritesTheFd(session); + CaseUnresolvableIdAheadOfFenceStillWritesTheFd(session); + + std::printf("---------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-h264-baseline-entropy/CMakeLists.txt b/vk_video_encoder/test/encoder-h264-baseline-entropy/CMakeLists.txt new file mode 100644 index 00000000..2def942b --- /dev/null +++ b/vk_video_encoder/test/encoder-h264-baseline-entropy/CMakeLists.txt @@ -0,0 +1,97 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_h264_baseline_entropy_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library: EncoderConfigH264 and its parameter-set +# writers live below the ext layer and are not exported from +# libvkvideo-encoder.so. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# CTest semantics, matching the sibling library tests: 0 means every assertion +# held, 1 means an assertion failed, and 2 means the session could not be stood +# up at all -- deliberately a FAILURE and not a skip. +# +# WHAT THIS GATES. H.264 Baseline and Constrained Baseline are both profile_idc +# 66 and neither admits CABAC; entropy_coding_mode_flag must be 0. Nothing tied +# the entropy coder to the profile, and the entropy coder is not chosen by the +# caller at all -- there is no switch for it, it is taken from the device +# preferredStdEntropyCodingModeFlag, which is CABAC on this vendor hardware. So +# an explicit --profile baseline emitted profile_idc 66 with +# entropy_coding_mode_flag 1: a non-conformant bitstream, and reachable from the +# WebRTC default profile. +# +# Three parts: the rule over the whole profile x entropy matrix, the emitted +# SPS/PPS pair, and the composition with the profile auto-upgrade in +# InitProfileLevel() -- which does NOT cover this case, because it is guarded by +# profileIdc == INVALID and so never runs on an explicit request. That last part +# is the one worth having: it is the reason the guard is not redundant with the +# upgrade, and it fails if someone later widens the upgrade to fire on requests +# and assumes it subsumes the clamp. +# +# WHAT A GREEN HERE DOES NOT MEAN: it does not mean a decoder accepted the +# stream. This asserts the syntax elements the writers produce, not a decode. +# Conformance against a real decoder needs hardware and is proven there. +enable_testing() +add_test(NAME EncoderH264BaselineRejectsCabac + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderH264BaselineRejectsCabac PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_h264_baseline_entropy_test: Configured") diff --git a/vk_video_encoder/test/encoder-h264-baseline-entropy/src/main.cpp b/vk_video_encoder/test/encoder-h264-baseline-entropy/src/main.cpp new file mode 100644 index 00000000..06e19c5c --- /dev/null +++ b/vk_video_encoder/test/encoder-h264-baseline-entropy/src/main.cpp @@ -0,0 +1,290 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// H.264 Baseline must not emit CABAC. +// +// Baseline and Constrained Baseline are both profile_idc 66 and neither admits +// CABAC: entropy_coding_mode_flag must be 0. Main (77) and the High profiles do +// admit it. Before the guard this suite covers, an explicit --profile baseline +// session took its entropy coder from the device preferredStdEntropyCodingMode +// flag, which is CABAC on this vendor hardware, and emitted profile_idc 66 with +// entropy_coding_mode_flag 1 -- a non-conformant bitstream, reachable from the +// WebRTC default profile. +// +// Device-free by construction. All three parts drive the profile decision and +// the parameter-set writers, neither of which needs a VkDevice. +// +// Exit codes match the sibling library suites: 0 every assertion held, 1 an +// assertion failed, 2 the session could not be stood up at all. + +#include +#include +#include + +#include "VkVideoEncoder/VkEncoderConfigH264.h" + +namespace { + +int g_failures = 0; +int g_checks = 0; + +void Check(bool ok, const char* what) +{ + ++g_checks; + if (ok) { + printf(" ok %s\n", what); + } else { + printf(" FAIL %s\n", what); + ++g_failures; + } +} + +const char* EntropyName(EncoderConfigH264::EntropyCodingMode m) +{ + return (m == EncoderConfigH264::ENTROPY_CODING_MODE_CABAC) ? "CABAC" : "CAVLC"; +} + +// High 10 is profile_idc 110, which the Vulkan std enum does not name; the +// library itself reaches it by the same cast in InitProfileLevel(). +const StdVideoH264ProfileIdc kProfileHigh10 = static_cast(110); + +// PART A -- the rule itself, over the whole profile x requested matrix. +// +// Asserted exhaustively rather than spot-checked on the one interesting cell, +// because the failure mode that matters second-most here is over-reach: a clamp +// that also stripped CABAC from Main or High would silently cost every +// non-Baseline session its entropy coder, and that regression is invisible to a +// test that only checks the Baseline cell. +void PartA_Rule() +{ + printf("PART A -- ConformantEntropyCodingMode(profile, requested)\n"); + + struct Case { + StdVideoH264ProfileIdc profile; + EncoderConfigH264::EntropyCodingMode requested; + EncoderConfigH264::EntropyCodingMode expected; + const char* name; + }; + + const Case cases[] = { + // The defect: Baseline may not carry CABAC, so it is clamped. + { STD_VIDEO_H264_PROFILE_IDC_BASELINE, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + EncoderConfigH264::ENTROPY_CODING_MODE_CAVLC, "Baseline(66) + CABAC -> CAVLC" }, + { STD_VIDEO_H264_PROFILE_IDC_BASELINE, EncoderConfigH264::ENTROPY_CODING_MODE_CAVLC, + EncoderConfigH264::ENTROPY_CODING_MODE_CAVLC, "Baseline(66) + CAVLC -> CAVLC" }, + // Every profile that DOES admit CABAC must keep it. + { STD_VIDEO_H264_PROFILE_IDC_MAIN, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, "Main(77) + CABAC -> CABAC" }, + { STD_VIDEO_H264_PROFILE_IDC_MAIN, EncoderConfigH264::ENTROPY_CODING_MODE_CAVLC, + EncoderConfigH264::ENTROPY_CODING_MODE_CAVLC, "Main(77) + CAVLC -> CAVLC" }, + { STD_VIDEO_H264_PROFILE_IDC_HIGH, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, "High(100) + CABAC -> CABAC" }, + { kProfileHigh10, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, "High10(110) + CABAC -> CABAC" }, + { STD_VIDEO_H264_PROFILE_IDC_HIGH_444_PREDICTIVE, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, "High444P(244) + CABAC -> CABAC" }, + }; + + for (const Case& c : cases) { + const EncoderConfigH264::EntropyCodingMode got = + EncoderConfigH264::ConformantEntropyCodingMode(c.profile, c.requested); + if (got != c.expected) { + printf(" got %s, expected %s\n", EntropyName(got), EntropyName(c.expected)); + } + Check(got == c.expected, c.name); + } +} + +// Stands up a config far enough to drive the parameter-set writers, with no +// device. Only the fields InitSpsPpsParameters() and InitProfileLevel() read +// are set; everything else keeps its constructed default. +void PrimeConfig(EncoderConfigH264& cfg) +{ + // 176x144 is mb-aligned in both dimensions, so no cropping arithmetic runs + // and the level search lands on the lowest entry. + cfg.encodeWidth = 176; + cfg.encodeHeight = 144; + cfg.pic_width_in_mbs = 11; + cfg.pic_height_in_map_units = 9; + cfg.encodeChromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + cfg.input.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + cfg.input.bpp = 8; + cfg.dpbCount = 2; + cfg.numRefL0 = 1; + cfg.numRefL1 = 0; + // Rate control DISABLED with hrdBitrate 0 makes DetermineLevel() skip its + // bitrate and CPB tests, so the level search cannot fall off the table on + // an unrelated default. + cfg.rateControlMode = VK_VIDEO_ENCODE_RATE_CONTROL_MODE_DISABLED_BIT_KHR; + cfg.hrdBitrate = 0; + cfg.vbvBufferSize = 0; + cfg.tuningMode = VK_VIDEO_ENCODE_TUNING_MODE_DEFAULT_KHR; + // Lossless would force qpprime_y_zero_transform_bypass_flag and pull the + // profile to High 4:4:4; keep this suite off that path. + cfg.qpprime_y_zero_transform_bypass_flag = 0; +} + +// PART B -- what actually lands in the emitted SPS/PPS pair. +// +// This is the part that answers the conformance question, because +// entropy_coding_mode_flag in the PPS and profile_idc in the SPS are the two +// syntax elements a decoder reads. Part A can pass while this fails, if the +// writer stops consulting the rule. +void PartB_EmittedParameterSets() +{ + printf("PART B -- emitted SPS/PPS\n"); + + struct Case { + StdVideoH264ProfileIdc profile; + EncoderConfigH264::EntropyCodingMode entropy; + bool expectFlag; + const char* name; + }; + + const Case cases[] = { + { STD_VIDEO_H264_PROFILE_IDC_BASELINE, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + false, "Baseline + CABAC requested -> entropy_coding_mode_flag 0" }, + { STD_VIDEO_H264_PROFILE_IDC_BASELINE, EncoderConfigH264::ENTROPY_CODING_MODE_CAVLC, + false, "Baseline + CAVLC requested -> entropy_coding_mode_flag 0" }, + { STD_VIDEO_H264_PROFILE_IDC_MAIN, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + true, "Main + CABAC requested -> entropy_coding_mode_flag 1" }, + { STD_VIDEO_H264_PROFILE_IDC_HIGH, EncoderConfigH264::ENTROPY_CODING_MODE_CABAC, + true, "High + CABAC requested -> entropy_coding_mode_flag 1" }, + }; + + for (const Case& c : cases) { + VkSharedBaseObj cfg(new EncoderConfigH264()); + if (!cfg) { + printf(" FAIL could not allocate EncoderConfigH264\n"); + ++g_failures; + return; + } + PrimeConfig(*cfg); + cfg->profileIdc = c.profile; + cfg->entropyCodingMode = c.entropy; + + StdVideoH264SequenceParameterSet sps; + StdVideoH264PictureParameterSet pps; + memset(&sps, 0, sizeof(sps)); + memset(&pps, 0, sizeof(pps)); + + if (!cfg->InitSpsPpsParameters(&sps, &pps, nullptr)) { + printf(" FAIL InitSpsPpsParameters returned false for %s\n", c.name); + ++g_failures; + continue; + } + + const bool flag = (pps.flags.entropy_coding_mode_flag != 0); + if (flag != c.expectFlag) { + printf(" entropy_coding_mode_flag=%d, expected %d\n", + flag ? 1 : 0, c.expectFlag ? 1 : 0); + } + Check(flag == c.expectFlag, c.name); + + // The profile is NOT rewritten to buy the entropy coder back. A + // downgrade is only honest if the SPS still says 66, so that what the + // decoder is told matches what the encoder emitted. + Check(sps.profile_idc == c.profile, + " sps.profile_idc is the requested profile, unchanged"); + + if (c.profile == STD_VIDEO_H264_PROFILE_IDC_BASELINE) { + // 66 with constraint_set1_flag set is Constrained Baseline. Both + // readings of the stream forbid CABAC, and both are satisfied. + Check(sps.flags.constraint_set0_flag != 0, + " Baseline: constraint_set0_flag set"); + Check(sps.flags.constraint_set1_flag != 0, + " Baseline: constraint_set1_flag set (Constrained Baseline)"); + // Control on the sibling per-profile tool restriction this fix is + // modelled on: it must still be enforced, and unchanged. + Check(pps.flags.transform_8x8_mode_flag == 0, + " Baseline: transform_8x8_mode_flag still 0 (sibling rule intact)"); + } + } +} + +// PART C -- composition with the profile auto-upgrade in InitProfileLevel(). +// +// The auto-upgrade turns an UNSPECIFIED profile into Main when B-frames or +// CABAC are in play. The two cases below pin why that upgrade neither replaces +// this fix nor collides with it: +// +// C1 An EXPLICIT Baseline request is not upgraded -- the upgrade block is +// guarded by profileIdc == INVALID, so it never runs on a request. This +// is the hole: without the clamp, this config reaches the PPS writer as +// profile_idc 66 with the device CABAC preference still set. +// C2 An UNSPECIFIED profile never resolves to Baseline, so the clamp and the +// upgrade operate on disjoint inputs and cannot both fire. +// +// C2 also records a live oddity: InitProfileLevel() runs at config +// construction, strictly BEFORE InitDeviceCapabilities() assigns +// entropyCodingMode from the device, so the CABAC term of the upgrade condition +// reads the constructed default -- which is CABAC. The upgrade therefore always +// leaves Baseline, making the profileIdc = BASELINE line that opens that block +// dead. That ordering is also why the clamp has to sit at the device-capability +// assignment: placed with the profile decision it would be overwritten, and +// inert. +void PartC_AutoUpgradeComposition() +{ + printf("PART C -- composition with the InitProfileLevel auto-upgrade\n"); + + { + VkSharedBaseObj cfg(new EncoderConfigH264()); + PrimeConfig(*cfg); + cfg->profileIdc = STD_VIDEO_H264_PROFILE_IDC_BASELINE; + cfg->entropyCodingMode = EncoderConfigH264::ENTROPY_CODING_MODE_CABAC; + cfg->InitProfileLevel(); + if (cfg->profileIdc != STD_VIDEO_H264_PROFILE_IDC_BASELINE) { + printf(" profileIdc became %u\n", static_cast(cfg->profileIdc)); + } + Check(cfg->profileIdc == STD_VIDEO_H264_PROFILE_IDC_BASELINE, + "C1 explicit Baseline request survives InitProfileLevel (not auto-upgraded)"); + } + + { + VkSharedBaseObj cfg(new EncoderConfigH264()); + PrimeConfig(*cfg); + cfg->profileIdc = STD_VIDEO_H264_PROFILE_IDC_INVALID; + cfg->entropyCodingMode = EncoderConfigH264::ENTROPY_CODING_MODE_CABAC; + cfg->InitProfileLevel(); + printf(" unspecified profile resolved to %u\n", + static_cast(cfg->profileIdc)); + Check(cfg->profileIdc != STD_VIDEO_H264_PROFILE_IDC_BASELINE, + "C2 unspecified profile never resolves to Baseline"); + Check(cfg->profileIdc != STD_VIDEO_H264_PROFILE_IDC_INVALID, + "C2 unspecified profile is resolved to something concrete"); + } +} + +} // namespace + +int main(int, char**) +{ + printf("encoder_h264_baseline_entropy_test -- H.264 Baseline must not emit CABAC\n\n"); + + PartA_Rule(); + printf("\n"); + PartB_EmittedParameterSets(); + printf("\n"); + PartC_AutoUpgradeComposition(); + + printf("\n%d checks, %d failures\n", g_checks, g_failures); + if (g_checks == 0) { + printf("RESULT: nothing ran\n"); + return 2; + } + printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-interface/CMakeLists.txt b/vk_video_encoder/test/encoder-interface/CMakeLists.txt new file mode 100644 index 00000000..fd3eb4e5 --- /dev/null +++ b/vk_video_encoder/test/encoder-interface/CMakeLists.txt @@ -0,0 +1,55 @@ +# The v3 interface, driven the way a host drives it. +# +# Links the encoder statically: the test exercises the interface, and a static +# link keeps it independent of whichever shared library happens to be found +# first on the loader path. + +set(TEST_NAME vk-video-enc-interface-test) + +add_executable(${TEST_NAME} src/main.cpp) + +target_include_directories(${TEST_NAME} PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/../../include + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} +) + +target_link_libraries(${TEST_NAME} PRIVATE vkvideo-encoder-static ${CMAKE_DL_LIBS}) + +if (TARGET GenVulkanDispatchTable) + add_dependencies(${TEST_NAME} GenVulkanDispatchTable) +endif() + +add_test(NAME ${TEST_NAME} COMMAND ${TEST_NAME}) +# A run that skips every check reports SKIP, not PASS: see Report::Summarise. +set_tests_properties(${TEST_NAME} PROPERTIES + LABELS "gpu" + SKIP_RETURN_CODE 77) + +# The C++17 floor is a claim about the public header, and a claim that nothing +# checks decays the first time someone writes std::span in it. The rest of the +# tree builds at C++20, so compiling this test at C++20 proves nothing about +# the floor; this compiles the header alone, at exactly -std=c++17, and fails +# the build if it needs anything newer. +# +# Syntax-only: there is nothing to link and nothing to run. +set(V3_FLOOR_PROBE ${CMAKE_CURRENT_BINARY_DIR}/v3_cxx17_floor_probe.cpp) +file(WRITE ${V3_FLOOR_PROBE} + "#include \"vulkan_video_encoder.h\"\nint main() { return 0; }\n") + +add_test(NAME vk-video-enc-header-cxx17-floor + COMMAND ${CMAKE_CXX_COMPILER} + -std=c++17 -fsyntax-only + -Wall -Wextra -Werror + -I${CMAKE_CURRENT_SOURCE_DIR}/../../include + -I${VULKAN_VIDEO_APIS_INCLUDE} + -I${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + -DVK_USE_PLATFORM_XLIB_KHR + -DVK_USE_PLATFORM_XCB_KHR + -DVK_USE_PLATFORM_WAYLAND_KHR + ${V3_FLOOR_PROBE}) + +# The window-system defines are the point of the probe as much as the standard +# is: Xlib's Success/None/Status macros only bite a consumer that has included +# it, which is every browser and compositor on Linux. +set_tests_properties(vk-video-enc-header-cxx17-floor PROPERTIES LABELS "device-free") diff --git a/vk_video_encoder/test/encoder-interface/src/main.cpp b/vk_video_encoder/test/encoder-interface/src/main.cpp new file mode 100644 index 00000000..c64b8399 --- /dev/null +++ b/vk_video_encoder/test/encoder-interface/src/main.cpp @@ -0,0 +1,1132 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// The encoder interface, driven the way a host drives it. +// +// The checks that need no device run unconditionally and are the ones that +// gate a build. The checks that need a real encode device are skipped, loudly, +// when there is none -- a suite that silently selects nothing must not pass. + +#include "vulkan_video_encoder.h" + +#include +#include +#include +#include +#include +#include + +using namespace vk::video::enc; + +namespace { + +int g_failures = 0; +int g_checks = 0; +int g_skipped = 0; + +void Check(bool condition, const char* what) +{ + ++g_checks; + if (!condition) { + ++g_failures; + fprintf(stderr, " FAIL %s\n", what); + } else { + fprintf(stderr, " ok %s\n", what); + } +} + +void Skip(const char* what, const char* why) +{ + ++g_skipped; + fprintf(stderr, " SKIP %s (%s)\n", what, why); +} + +//============================================================================= +// Checks that need no device +//============================================================================= + +// Result must stay cheap: it is returned once per frame. +void TestResultIsCheap() +{ + Check(sizeof(Result) <= 2 * sizeof(void*), "Result fits in two words"); + + Result ok; + Check(ok.ok() && static_cast(ok), "a default Result is success"); + Check(ok.code() == ResultCode::Ok, "a default Result carries Ok"); + + Result bad(ResultCode::InvalidArgument, "resolution unset"); + Check(!bad.ok() && !static_cast(bad), "an error Result is falsey"); + Check(bad == ResultCode::InvalidArgument, "the code compares directly"); + Check(strcmp(bad.detail(), "resolution unset") == 0, "the detail survives"); +} + +void TestExpectedCarriesEither() +{ + Expected value(42); + Check(static_cast(value) && *value == 42, "Expected carries a value"); + + Expected error(Result(ResultCode::NotReady, "not yet")); + Check(!error, "Expected carries a failure"); + Check(error.status().code() == ResultCode::NotReady, "the failure keeps its code"); + Check(strcmp(error.status().detail(), "not yet") == 0, "the failure keeps its detail"); +} + +void TestArrayView() +{ + const uint32_t backing[4] = {10, 20, 30, 40}; + ArrayView view(backing, 4); + + Check(view.size() == 4 && !view.empty(), "a view reports its size"); + Check(view[2] == 30, "a view indexes"); + + uint32_t sum = 0; + for (uint32_t v : view) { + sum += v; + } + Check(sum == 100, "a view iterates"); + Check(ArrayView().empty(), "a default view is empty"); +} + +// Format classification, which needs no device. +// +// These expectations are the tables a host would otherwise carry as its own +// format->bool predicates. Pinning them here is what lets a host delete that +// duplicate set and route the question to VkEncClassifyFormat instead: the +// answers stay asserted somewhere that can fail. +void TestFormatClassification() +{ + struct Row { + VkFormat format; + ColorModel model; + bool supported; + InputPath path; + FilterAccess access; + const char* what; + }; + static const Row kRows[] = { + // Semi-planar 8-bit is what the encoder reads: no filter. + {VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ColorModel::YCbCr, true, + InputPath::Copy, FilterAccess::NoFilter, "NV12 is taken directly"}, + + // Three-plane I420 is addressed a plane at a time. + {VK_FORMAT_G8_B8_R8_3PLANE_420_UNORM, ColorModel::YCbCr, true, + InputPath::ComputeFilter, FilterAccess::PlaneStorage, + "I420 needs the filter, addressing planes"}, + {VK_FORMAT_G10X6_B10X6_R10X6_3PLANE_420_UNORM_3PACK16, ColorModel::YCbCr, + true, InputPath::ComputeFilter, FilterAccess::PlaneStorage, + "I420 10-bit needs the filter, addressing planes"}, + {VK_FORMAT_G12X4_B12X4_R12X4_3PLANE_420_UNORM_3PACK16, ColorModel::YCbCr, + true, InputPath::ComputeFilter, FilterAccess::PlaneStorage, + "I420 12-bit needs the filter, addressing planes"}, + + // RGB is one plane, read as a storage image. + {VK_FORMAT_R8G8B8A8_UNORM, ColorModel::Rgb, true, + InputPath::ComputeFilter, FilterAccess::StorageRead, + "RGBA needs the filter, read as storage"}, + {VK_FORMAT_B8G8R8A8_UNORM, ColorModel::Rgb, true, + InputPath::ComputeFilter, FilterAccess::StorageRead, + "BGRA needs the filter, read as storage"}, + {VK_FORMAT_A8B8G8R8_UNORM_PACK32, ColorModel::Rgb, true, + InputPath::ComputeFilter, FilterAccess::StorageRead, + "packed RGBA needs the filter, read as storage"}, + + // Not a picture format on any route. + {VK_FORMAT_D32_SFLOAT, ColorModel::FromFormat, false, + InputPath::Copy, FilterAccess::NoFilter, "a depth format is refused"}, + }; + + for (const Row& row : kRows) { + FormatRouting routing; + const VkResult result = VkEncClassifyFormat(row.format, row.model, &routing); + + if (!row.supported) { + Check(result != VK_SUCCESS && !routing.supported, row.what); + continue; + } + const bool ok = (result == VK_SUCCESS) && routing.supported && + (routing.path == row.path) && + (routing.filterAccess == row.access); + Check(ok, row.what); + if (!ok) { + fprintf(stderr, " format=0x%x path=%u access=%u supported=%d\n", + static_cast(row.format), + static_cast(routing.path), + static_cast(routing.filterAccess), + static_cast(routing.supported)); + } + } + + // A null out-parameter is a caller bug, refused rather than dereferenced. + Check(VkEncClassifyFormat(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, + ColorModel::YCbCr, nullptr) != VK_SUCCESS, + "classification refuses a null out-parameter"); +} + +//============================================================================= +// A real input frame +// +// The image is LINEAR and host-visible, and the pattern is written by mapping +// it. That is not a shortcut: the library owns the device here, and it exposes +// no queue family, so there is no queue on which a staging copy could be +// submitted. Mapping needs none. +// +// A linear image is not directly encodable, so this exercises the path a +// caller most often takes anyway -- the library stages it into an encodable +// image -- and proves the frame reaches the encoder. +//============================================================================= + +struct DeviceFns { + PFN_vkGetDeviceProcAddr GetDeviceProcAddr = nullptr; + PFN_vkCreateImage CreateImage = nullptr; + PFN_vkDestroyImage DestroyImage = nullptr; + PFN_vkGetImageMemoryRequirements GetImageMemoryRequirements = nullptr; + PFN_vkGetPhysicalDeviceMemoryProperties GetPhysicalDeviceMemoryProperties = nullptr; + PFN_vkAllocateMemory AllocateMemory = nullptr; + PFN_vkFreeMemory FreeMemory = nullptr; + PFN_vkBindImageMemory BindImageMemory = nullptr; + PFN_vkMapMemory MapMemory = nullptr; + PFN_vkUnmapMemory UnmapMemory = nullptr; + PFN_vkGetImageSubresourceLayout GetImageSubresourceLayout = nullptr; + PFN_vkDeviceWaitIdle DeviceWaitIdle = nullptr; + + bool Load(PFN_vkGetInstanceProcAddr gipa, VkInstance instance, VkDevice device) + { + if (gipa == nullptr || instance == VK_NULL_HANDLE || device == VK_NULL_HANDLE) { + return false; + } + GetDeviceProcAddr = + reinterpret_cast(gipa(instance, "vkGetDeviceProcAddr")); + GetPhysicalDeviceMemoryProperties = + reinterpret_cast( + gipa(instance, "vkGetPhysicalDeviceMemoryProperties")); + if (GetDeviceProcAddr == nullptr || GetPhysicalDeviceMemoryProperties == nullptr) { + return false; + } + #define V3_LOAD(name) \ + name = reinterpret_cast(GetDeviceProcAddr(device, "vk" #name)) + V3_LOAD(CreateImage); + V3_LOAD(DestroyImage); + V3_LOAD(GetImageMemoryRequirements); + V3_LOAD(AllocateMemory); + V3_LOAD(FreeMemory); + V3_LOAD(BindImageMemory); + V3_LOAD(MapMemory); + V3_LOAD(UnmapMemory); + V3_LOAD(GetImageSubresourceLayout); + V3_LOAD(DeviceWaitIdle); + #undef V3_LOAD + return CreateImage && DestroyImage && GetImageMemoryRequirements && + AllocateMemory && FreeMemory && BindImageMemory && MapMemory && + UnmapMemory && GetImageSubresourceLayout && DeviceWaitIdle; + } +}; + +const uint32_t kWidth = 320; +const uint32_t kHeight = 240; + +class HostImage { +public: + HostImage(const DeviceFns& fns, VkPhysicalDevice phys, VkDevice device) + : m_fns(fns), m_phys(phys), m_device(device) { } + + ~HostImage() { Destroy(); } + + HostImage(const HostImage&) = delete; + HostImage& operator=(const HostImage&) = delete; + + bool Create(const char** why) + { + VkImageCreateInfo ci{VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO}; + ci.imageType = VK_IMAGE_TYPE_2D; + ci.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ci.extent = {kWidth, kHeight, 1}; + ci.mipLevels = 1; + ci.arrayLayers = 1; + ci.samples = VK_SAMPLE_COUNT_1_BIT; + ci.tiling = VK_IMAGE_TILING_LINEAR; + ci.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + ci.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + // Written from the host before anything else touches it, so its + // contents must survive the first transition. + ci.initialLayout = VK_IMAGE_LAYOUT_PREINITIALIZED; + + if (m_fns.CreateImage(m_device, &ci, nullptr, &m_image) != VK_SUCCESS) { + *why = "vkCreateImage failed for a linear NV12 image"; + return false; + } + + VkMemoryRequirements req{}; + m_fns.GetImageMemoryRequirements(m_device, m_image, &req); + + VkPhysicalDeviceMemoryProperties props{}; + m_fns.GetPhysicalDeviceMemoryProperties(m_phys, &props); + + const VkMemoryPropertyFlags want = + VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; + uint32_t typeIndex = UINT32_MAX; + for (uint32_t i = 0; i < props.memoryTypeCount; ++i) { + if (((req.memoryTypeBits & (1u << i)) != 0) && + ((props.memoryTypes[i].propertyFlags & want) == want)) { + typeIndex = i; + break; + } + } + if (typeIndex == UINT32_MAX) { + *why = "no host-visible memory type accepts a linear NV12 image"; + return false; + } + + VkMemoryAllocateInfo ai{VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO}; + ai.allocationSize = req.size; + ai.memoryTypeIndex = typeIndex; + if (m_fns.AllocateMemory(m_device, &ai, nullptr, &m_memory) != VK_SUCCESS) { + *why = "vkAllocateMemory failed"; + return false; + } + if (m_fns.BindImageMemory(m_device, m_image, m_memory, 0) != VK_SUCCESS) { + *why = "vkBindImageMemory failed"; + return false; + } + m_size = req.size; + return true; + } + + // A moving pattern, so successive frames are not identical and the encoder + // has something to predict. A flat image would let a broken submit path + // still produce plausible-looking output. + bool WritePattern(uint32_t frameIndex, const char** why) + { + VkImageSubresource luma{VK_IMAGE_ASPECT_PLANE_0_BIT, 0, 0}; + VkImageSubresource chroma{VK_IMAGE_ASPECT_PLANE_1_BIT, 0, 0}; + VkSubresourceLayout lumaLayout{}, chromaLayout{}; + m_fns.GetImageSubresourceLayout(m_device, m_image, &luma, &lumaLayout); + m_fns.GetImageSubresourceLayout(m_device, m_image, &chroma, &chromaLayout); + + void* mapped = nullptr; + if (m_fns.MapMemory(m_device, m_memory, 0, VK_WHOLE_SIZE, 0, &mapped) != VK_SUCCESS) { + *why = "vkMapMemory failed"; + return false; + } + uint8_t* base = static_cast(mapped); + + const uint32_t bar = (frameIndex * 24) % kWidth; + for (uint32_t y = 0; y < kHeight; ++y) { + uint8_t* row = base + lumaLayout.offset + y * lumaLayout.rowPitch; + for (uint32_t x = 0; x < kWidth; ++x) { + const bool inBar = (x >= bar) && (x < bar + 24); + row[x] = inBar ? 235 : static_cast(16 + (x * 200) / kWidth); + } + } + for (uint32_t y = 0; y < kHeight / 2; ++y) { + uint8_t* row = base + chromaLayout.offset + y * chromaLayout.rowPitch; + for (uint32_t x = 0; x < kWidth / 2; ++x) { + row[2 * x] = 128; + row[2 * x + 1] = 128; + } + } + + m_fns.UnmapMemory(m_device, m_memory); + return true; + } + + void Destroy() + { + if (m_image != VK_NULL_HANDLE) { + m_fns.DestroyImage(m_device, m_image, nullptr); + m_image = VK_NULL_HANDLE; + } + if (m_memory != VK_NULL_HANDLE) { + m_fns.FreeMemory(m_device, m_memory, nullptr); + m_memory = VK_NULL_HANDLE; + } + } + + VkImage Image() const { return m_image; } + uint64_t Size() const { return m_size; } + +private: + const DeviceFns& m_fns; + VkPhysicalDevice m_phys; + VkDevice m_device; + VkImage m_image = VK_NULL_HANDLE; + VkDeviceMemory m_memory = VK_NULL_HANDLE; + uint64_t m_size = 0; +}; + +// Describe a created image to the interface. +ExternalImage DescribeImage(const HostImage& image) +{ + ExternalImage described; + described.handleType = ExternalHandleType::VkImageHandle; + described.existingImage = image.Image(); + described.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + described.width = kWidth; + described.height = kHeight; + described.tiling = VK_IMAGE_TILING_LINEAR; + described.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + described.colorModel = ColorModel::YCbCr; + // Must match vkCreateImage exactly: the library cannot read these back + // from a VkImage handle and uses them to decide how to reach the encoder. + described.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + described.createFlags = 0; + described.residency = Residency::Local; + described.allocationSize = image.Size(); + return described; +} + +//============================================================================= +// Checks that need a device +//============================================================================= + +// Every role a session offers, and the one it does not. +void TestSessionRoles(const Ref& session) +{ + Check(Query(session) != nullptr, "IFrameSubmitter resolves"); + Check(Query(session) != nullptr, "IBitstreamSource resolves"); + Check(Query(session) != nullptr, "ICompletionSignal resolves"); + Check(Query(session) != nullptr, "IResourceRegistry resolves"); + Check(Query(session) != nullptr, "IDiagnostics resolves"); + + // A session knows the device it runs on. A caller holding only a session + // must not have to keep the platform alive to ask. + Ref bound = Query(session); + Check(bound != nullptr, "IDeviceBinding resolves from the session"); + if (bound) { + Check(bound->Device() != VK_NULL_HANDLE, + "the session reports the device it runs on"); + Check(bound->GetInstanceProcAddr() != nullptr, + "the session reports the loader it was built with"); + + // The identity, not the handle. A host that adopted a device confirms + // the encoder bound the one it meant, and two instances give different + // handles for the same hardware. + uint8_t uuid[VK_UUID_SIZE] = {}; + uint8_t zero[VK_UUID_SIZE] = {}; + const bool identified = bound->DeviceUuid(uuid); + Check(identified, "the session identifies its device"); + Check(!identified || memcmp(uuid, zero, sizeof(uuid)) != 0, + "the device identity is not all zero"); + } + + // EVERY ROLE THE INTERFACE DEFINES IS REACHABLE FROM A SESSION. + // + // Listed here rather than checked one by one above so that adding a role + // to the interface and forgetting to answer its id fails here. That + // failure is otherwise silent in the worst way: a host queries the role, + // gets null, and takes whatever fallback it has -- which is how an + // unreachable IDeviceBinding turned off a whole zero-copy input tier + // while every frame still encoded correctly, by the slowest path. + { + struct RoleRow { const char* id; const char* name; }; + static const RoleRow kRoles[] = { + {IEncoderSession::kId.data(), "IEncoderSession"}, + {IFrameSubmitter::kId.data(), "IFrameSubmitter"}, + {IBitstreamSource::kId.data(), "IBitstreamSource"}, + {ICompletionSignal::kId.data(), "ICompletionSignal"}, + {IResourceRegistry::kId.data(), "IResourceRegistry"}, + {IDeviceBinding::kId.data(), "IDeviceBinding"}, + {IDiagnostics::kId.data(), "IDiagnostics"}, + }; + uint32_t reachable = 0; + for (const RoleRow& role : kRoles) { + if (session->QueryInterface(role.id) != nullptr) { + ++reachable; + } else { + fprintf(stderr, " unreachable role: %s\n", role.name); + } + } + Check(reachable == sizeof(kRoles) / sizeof(kRoles[0]), + "every role the interface defines is reachable from a session"); + } + + // A role the session vends outlives the caller's reference to the session: + // the aliasing Ref shares the owner's count. + Ref bitstream = Query(session); + Ref local = session; + local.reset(); + Check(bitstream != nullptr && bitstream->Acquire(9999).status().code() != + ResultCode::Ok, + "a role keeps its owner alive after the session ref is dropped"); +} + +void TestConfigRules(const Ref& platform) +{ + // A configuration refuses a profile from another codec, before anything + // is allocated. + Expected> wrong = platform->CreateConfig(Codec::H264, Profile::H265Main); + Check(!wrong && wrong.status().code() == ResultCode::UnsupportedProfile, + "a profile from another codec is refused at CreateConfig"); + + Expected> h264 = platform->CreateConfig(Codec::H264, Profile::H264Main); + if (!h264) { + Skip("H.264 configuration checks", h264.status().detail()); + return; + } + + // An unconfigured configuration says why it is not usable, without a + // session having been created. + Result unset = (*h264)->Validate(); + Check(!unset && unset.code() == ResultCode::InvalidArgument, + "an empty configuration fails Validate with a reason"); + + (*h264)->SetCodedExtent(1920, 1080) + .SetFrameRate(60, 1) + .SetInputFormat(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ColorModel::YCbCr); + Check(static_cast((*h264)->Validate()), "a complete configuration validates"); + Check((*h264)->GetCodec() == Codec::H264, "the configuration remembers its codec"); + + // The rule that matters: the library answers which input path a + // configuration resolves to. NV12 is what the encoder reads, so no filter. + Check((*h264)->ResolveInputPath() == InputPath::Copy, + "NV12 resolves to a copy, not a compute filter"); + + // H.264 carries no mastering-display or content-light SEI, so the role is + // absent rather than present-and-refusing. + Check(Query(*h264) == nullptr, + "an H.264 configuration offers no HDR metadata role"); + + Expected> h265 = platform->CreateConfig(Codec::H265, Profile::H265Main10); + if (h265) { + Ref hdr = Query(*h265); + Check(hdr != nullptr, "an H.265 configuration offers the HDR metadata role"); + if (hdr) { + HdrMetadata meta; + meta.contentLightLevelPresent = true; + meta.maxContentLightLevel = 1000; + meta.maxFrameAverageLightLevel = 400; + hdr->SetHdrMetadata(meta); + } + } else { + Skip("H.265 HDR role check", h265.status().detail()); + } + + // A format the encoder cannot read on any path is refused by Validate, + // not at session creation. + Expected> bad = platform->CreateConfig(Codec::H264, Profile::H264Main); + if (bad) { + (*bad)->SetCodedExtent(320, 240) + .SetFrameRate(30, 1) + .SetInputFormat(VK_FORMAT_D32_SFLOAT, ColorModel::FromFormat); + Result r = (*bad)->Validate(); + Check(!r && r.code() == ResultCode::UnsupportedFormat, + "a depth format is refused with UnsupportedFormat"); + } +} + +// Returns false when the platform exists but no device on it can encode, +// which is a property of the machine and not a result about the interface. +bool TestCaps(const Ref& platform) +{ + Ref caps = platform->Caps(); + Check(caps != nullptr, "the platform vends capabilities"); + if (!caps) { + return false; + } + + ArrayView codecs = caps->Codecs(); + fprintf(stderr, " codecs advertised: %zu\n", codecs.size()); + if (codecs.empty()) { + Skip("codec and profile enumeration", "no encode-capable device"); + return false; + } + + for (Codec codec : codecs) { + ArrayView profiles = caps->Profiles(codec); + fprintf(stderr, " codec %u: %zu profiles\n", + static_cast(codec), profiles.size()); + } + + // Enumeration is repeatable: a second call returns the same view rather + // than re-querying into different storage. + Check(caps->Codecs().data() == codecs.data(), "capability views are stable"); + return true; +} + +//============================================================================= +// Completion signalling +//============================================================================= + +// What the callback records. Touched from a library thread, so every member is +// atomic and nothing here allocates or blocks -- the contract forbids both. +struct CallbackState { + std::atomic invocations{0}; + std::atomic lastFrameId{0}; + std::atomic wrongCookie{false}; +}; + +void OnFrameComplete(uint64_t frameId, void* userData) +{ + CallbackState* state = static_cast(userData); + if (state == nullptr) { + return; + } + state->invocations.fetch_add(1, std::memory_order_relaxed); + state->lastFrameId.store(frameId, std::memory_order_relaxed); +} + +// Drive the callback, the counter and the timeline semaphore against a real +// encode, and check each for what it actually promises rather than for having +// merely fired. +void TestCompletionSignal(const Ref& platform, + const Ref& session, + const DeviceFns& fns, + HostImage& image, + uint64_t firstFrameId) +{ + Ref completion = Query(session); + Ref submitter = Query(session); + Ref bitstream = Query(session); + if (!completion || !submitter || !bitstream) { + Skip("completion signalling", "the session does not offer the roles"); + return; + } + + const uint64_t before = completion->CompletedCount(); + + CallbackState state; + Check(static_cast(completion->SetCallback(&OnFrameComplete, &state)), + "a completion callback installs"); + + // The semaphore is live for the length of the session and is the encoder's + // to own; a caller never destroys it. + const VkSemaphore semaphore = completion->CompletionSemaphore(); + Check(semaphore != VK_NULL_HANDLE, "a live session vends a completion semaphore"); + + const uint32_t kFrames = 6; + uint64_t lastId = firstFrameId; + for (uint32_t i = 0; i < kFrames; ++i) { + const char* why = ""; + if (!image.WritePattern(i, &why)) { + Skip("completion signalling", why); + return; + } + FrameSubmit frame; + // The timeline currency requires monotonically increasing ids, and + // reaching frameId + 1 is what it means by "this one has retired". + frame.frameId = firstFrameId + i; + frame.pts = i * 3000; + frame.image = DescribeImage(image); + lastId = frame.frameId; + if (!submitter->SubmitFrame(frame)) { + Skip("completion signalling", "submission failed"); + return; + } + } + + Check(static_cast(session->Drain()), "the drain completes"); + + const uint64_t after = completion->CompletedCount(); + fprintf(stderr, " completions %llu -> %llu, %u callback invocations\n", + (unsigned long long)before, (unsigned long long)after, + state.invocations.load()); + + Check(after >= before + kFrames, + "the completion counter advances by at least the frames submitted"); + + // Notifications coalesce -- one invocation may cover several frames -- so + // the check is that the callback fired at all and never more often than + // there were completions. Asserting one-per-frame would encode a promise + // the library explicitly does not make. + Check(state.invocations.load() > 0, "the completion callback fires"); + Check(state.invocations.load() <= after - before, + "the callback never fires more often than frames complete"); + Check(!state.wrongCookie.load(), "the callback receives its own user data"); + + // The documented reconciliation: drain retrieval until the caller's own + // count matches the counter, rather than assuming one callback per frame. + uint32_t retrieved = 0; + while (retrieved < (after - before)) { + Expected got = bitstream->AcquireNext(); + if (!got) { + break; + } + ++retrieved; + bitstream->Release(got->frameId); + } + Check(retrieved == after - before, + "retrieval reconciles against the completion counter"); + + // The timeline says every frame with an id at or below lastId has had its + // encode work retire. It is not a readiness signal, so it is checked after + // the drain, where readiness is already established by other means. + PFN_vkGetSemaphoreCounterValue getValue = + reinterpret_cast( + fns.GetDeviceProcAddr(platform->DeviceBinding()->Device(), + "vkGetSemaphoreCounterValue")); + if (getValue != nullptr && semaphore != VK_NULL_HANDLE) { + uint64_t value = 0; + if (getValue(platform->DeviceBinding()->Device(), semaphore, &value) == VK_SUCCESS) { + fprintf(stderr, " timeline value %llu, last frame id %llu\n", + (unsigned long long)value, (unsigned long long)lastId); + Check(value >= lastId + 1, + "the timeline reaches lastFrameId + 1 once the encodes retire"); + } + } + + // An OS handle a non-Vulkan loop can wait on. The caller owns what it gets. + Expected handle = completion->ExportCompletionHandle(); + if (handle) { + Check(*handle >= 0, "the completion handle exports a usable descriptor"); + close(*handle); + } else { + Skip("completion handle export", handle.status().detail()); + } + + // Detaching is a quiesce point: once it returns no invocation is in + // flight, which is what makes destroying the cookie afterwards safe. The + // check that matters is that it returns at all -- a deadlock here would + // hang rather than fail. + Check(static_cast(completion->SetCallback(nullptr, nullptr)), + "the completion callback detaches"); + + // Detaching has to actually stop the notifications, and the only way to + // show that is to make more of them: encode further frames and require the + // count not to move. Comparing the counter with itself would pass on a + // detach that did nothing. + const uint32_t atDetach = state.invocations.load(); + uint32_t moreCompleted = 0; + for (uint32_t i = 0; i < 2; ++i) { + const char* why = ""; + if (!image.WritePattern(kFrames + i, &why)) { + break; + } + FrameSubmit frame; + frame.frameId = lastId + 1 + i; + frame.pts = (kFrames + i) * 3000; + frame.image = DescribeImage(image); + if (!submitter->SubmitFrame(frame)) { + break; + } + ++moreCompleted; + } + if (moreCompleted > 0) { + session->Drain(); + for (uint32_t i = 0; i < moreCompleted; ++i) { + Expected got = bitstream->AcquireNext(); + if (!got) { + break; + } + bitstream->Release(got->frameId); + } + Check(completion->CompletedCount() > after, + "frames still complete after the callback is detached"); + Check(state.invocations.load() == atDetach, + "a detached callback is not invoked again"); + } +} + +//============================================================================= +// Reconfiguration +//============================================================================= + +void TestReconfigure(const Ref& platform, + const Ref& session, + const Ref& config) +{ + Ref diag = Query(session); + + // What moves: the rate. Handing back the very configuration the session + // was built from, with the rate changed, is the ordinary case. + RateControl faster; + faster.mode = RateControlMode::Cbr; + faster.averageBitrate = 4 * 1000 * 1000; + faster.maxBitrate = 4 * 1000 * 1000; + config->SetRateControl(faster); + + Result applied = session->Reconfigure(config); + if (!applied) { + Skip("reconfiguration", applied.detail()); + return; + } + Check(true, "a rate change is applied mid-stream"); + + // A fresh configuration carrying nothing but the change must work too: + // an unset field means "leave it alone", so the caller is not made to + // restate the resolution and format it is not touching. + Expected> minimal = + platform->CreateConfig(config->GetCodec(), config->GetProfile()); + if (minimal) { + RateControl slower; + slower.mode = RateControlMode::Cbr; + slower.averageBitrate = 1500 * 1000; + (*minimal)->SetRateControl(slower); + Check(static_cast(session->Reconfigure(*minimal)), + "a configuration carrying only the change is accepted"); + } + + // What does not move: a resolution change is refused, by name, and the + // session keeps running. + Expected> resized = + platform->CreateConfig(config->GetCodec(), config->GetProfile()); + if (resized) { + (*resized)->SetCodedExtent(640, 480); + RateControl same; + same.mode = RateControlMode::Cbr; + same.averageBitrate = 1500 * 1000; + (*resized)->SetRateControl(same); + + Result refused = session->Reconfigure(*resized); + Check(!refused && refused.code() == ResultCode::UnsupportedFeature, + "a resolution change is refused rather than half-applied"); + if (diag) { + fprintf(stderr, " refusal named: %s\n", diag->LastErrorDetail()); + Check(strstr(diag->LastErrorDetail(), "width") != nullptr, + "the refusal names the field that cannot change"); + } + } + + // The session still encodes after a refusal. + Check(static_cast(session->Drain()), + "the session still works after a refused reconfiguration"); +} + +// Frames in, bitstream out, through the interface only. +// +// Returns false only when the machine could not be asked -- a device that +// refuses a linear NV12 image is a property of the driver, not a result about +// the interface. +bool TestFrameTransfer(const Ref& platform, + const Ref& session, + const Ref& config) +{ + Ref binding = platform->DeviceBinding(); + if (!binding || binding->Device() == VK_NULL_HANDLE) { + Skip("frame transfer", "no device binding"); + return false; + } + + DeviceFns fns; + if (!fns.Load(binding->GetInstanceProcAddr(), binding->Instance(), binding->Device())) { + Skip("frame transfer", "device entry points unavailable"); + return false; + } + + HostImage image(fns, binding->PhysicalDevice(), binding->Device()); + const char* why = ""; + if (!image.Create(&why)) { + Skip("frame transfer", why); + return false; + } + + Ref submitter = Query(session); + Ref bitstream = Query(session); + Ref diag = Query(session); + if (!submitter || !bitstream) { + Skip("frame transfer", "the session does not offer submit and bitstream roles"); + return false; + } + + const uint32_t kFrames = 8; + + // ---- inline submission ---- + uint32_t encoded = 0; + uint32_t idrSeen = 0; + size_t totalBytes = 0; + bool firstIsIdr = false; + bool submitOk = true; + + for (uint32_t i = 0; i < kFrames; ++i) { + if (!image.WritePattern(i, &why)) { + Skip("frame transfer", why); + return false; + } + FrameSubmit frame; + frame.frameId = 1000 + i; + frame.pts = i * 3000; + frame.image = DescribeImage(image); + frame.forceIdr = (i == 0); + // The stream is not ended here. isLastFrame finalises the session, and + // the registered-submission checks below still have frames to send. + + Result r = submitter->SubmitFrame(frame); + if (!r) { + Check(false, "SubmitFrame accepts an inline image"); + fprintf(stderr, " detail: %s\n", r.detail()); + submitOk = false; + break; + } + } + if (!submitOk) { + return true; // a real failure, already counted + } + Check(true, "SubmitFrame accepts an inline image"); + + Check(static_cast(session->Drain()), "Drain returns success"); + + for (uint32_t i = 0; i < kFrames; ++i) { + Expected got = bitstream->AcquireNext(); + if (!got) { + break; + } + if (encoded == 0) { + firstIsIdr = got->isIdr; + } + if (got->isIdr) { + ++idrSeen; + } + totalBytes += got->bitstream.size(); + ++encoded; + bitstream->Release(got->frameId); + } + + fprintf(stderr, " %u frames encoded, %zu bitstream bytes, %u IDR\n", + encoded, totalBytes, idrSeen); + + Check(encoded == kFrames, "every submitted frame comes back encoded"); + Check(totalBytes > 0, "the bitstream is not empty"); + Check(firstIsIdr, "the first frame is an IDR"); + if (diag) { + Check(diag->FramesSubmitted() == kFrames, "IDiagnostics counts the submissions"); + Check(diag->FramesEncoded() == encoded, "IDiagnostics counts the encodes"); + } + + // ---- registered submission: the same image, by index ---- + Ref registry = Query(session); + if (!registry) { + Skip("registered submission", "no resource registry role"); + return true; + } + + bool handle_consumed = false; + Expected registered = + registry->RegisterImage(DescribeImage(image), &handle_consumed); + if (!registered) { + Skip("registered submission", registered.status().detail()); + return true; + } + Check(*registered != kNoResource, "an image registers and yields an id"); + + // Registering does not encode; submitting by that index does, and it must + // reach the encoder without describing the image again. + uint32_t registeredEncoded = 0; + bool registeredSubmitOk = true; + for (uint32_t i = 0; i < 4; ++i) { + if (!image.WritePattern(kFrames + i, &why)) { + break; + } + FrameSubmit frame; + frame.frameId = 2000 + i; + frame.pts = (kFrames + i) * 3000; + frame.registeredImage = *registered; + frame.image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + + Result r = submitter->SubmitFrame(frame); + if (!r) { + Check(false, "SubmitFrame accepts a registered image by index"); + fprintf(stderr, " detail: %s\n", r.detail()); + if (diag) { + fprintf(stderr, " diagnostic: %s\n", diag->LastErrorDetail()); + } + registeredSubmitOk = false; + break; + } + } + if (registeredSubmitOk) { + Check(true, "SubmitFrame accepts a registered image by index"); + Check(static_cast(session->Drain()), + "Drain leaves the session usable between batches"); + for (uint32_t i = 0; i < 4; ++i) { + Expected got = bitstream->AcquireNext(); + if (!got) { + break; + } + ++registeredEncoded; + bitstream->Release(got->frameId); + } + fprintf(stderr, " %u registered frames encoded\n", registeredEncoded); + Check(registeredEncoded == 4, "every registered frame comes back encoded"); + } + + // Completion signalling and reconfiguration, while the session is live. + // Both must run BEFORE the stream is ended below: the completion semaphore + // dies with it, and a reconfiguration needs a session to reconfigure. + TestCompletionSignal(platform, session, fns, image, 4000); + TestReconfigure(platform, session, config); + + // An id that was never handed out is refused, not silently encoded. Checked + // while the session is live: once the stream has ended every submission is + // refused, and a refusal for the wrong reason proves nothing. + { + FrameSubmit bogus; + bogus.frameId = 3000; + bogus.registeredImage = 0xDEADBEEF; + const Result refused = submitter->SubmitFrame(bogus); + Check(!refused, "an unknown registered id is refused"); + if (diag) { + fprintf(stderr, " unknown-id refusal: %s\n", + diag->LastErrorDetail()); + } + } + + // WHO CLOSES THE FD. + // + // Under Borrow the caller keeps its handle and the library works from a + // private duplicate, so a failed registration cannot touch it. That is + // checkable and is checked. + // + // Under Transfer the caller must not close, and what becomes of the + // descriptor NUMBER is deliberately not asserted here. Past the + // vkAllocateMemory handoff the driver owns it -- and owns it even when the + // allocation fails -- so the library does not close it and a test that + // demanded a closed number would be demanding a double close. Short of + // that handoff the library closes it itself. Both are correct; which one + // happened is not a property a caller can or should observe. + // + // The descriptor here is deliberately not importable: the registration + // must fail, and the question is only whether the caller's own fd survived. + { + ExternalImage borrowed = DescribeImage(image); + borrowed.handleType = ExternalHandleType::OpaqueFd; + borrowed.existingImage = VK_NULL_HANDLE; + borrowed.ownership = HandleOwnership::Borrow; + borrowed.fd = dup(STDIN_FILENO); + if (borrowed.fd >= 0) { + Check(!registry->RegisterImage(borrowed), + "an unimportable descriptor is refused"); + Check(fcntl(borrowed.fd, F_GETFD) != -1, + "a borrowed fd survives a failed registration"); + close(borrowed.fd); + } + + // The echo is written on a refusal too, which is the only path where + // a caller cannot infer it from a returned id. + bool refused_consumed = true; + ExternalImage transferred = DescribeImage(image); + transferred.handleType = ExternalHandleType::OpaqueFd; + transferred.existingImage = VK_NULL_HANDLE; + transferred.ownership = HandleOwnership::Transfer; + transferred.fd = dup(STDIN_FILENO); + if (transferred.fd >= 0) { + Check(!registry->RegisterImage(transferred, &refused_consumed), + "an unimportable descriptor is refused under Transfer too"); + Check(refused_consumed, + "the ownership echo is written on a refused registration"); + } + } + + // Ending the stream is a real state change, so it is checked rather than + // assumed: a submission after the last frame is refused. + { + FrameSubmit last; + last.frameId = 2999; + last.registeredImage = *registered; + last.image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + last.isLastFrame = true; + if (submitter->SubmitFrame(last)) { + Check(static_cast(session->Finish()), "Finish ends the stream"); + Expected tail = bitstream->AcquireNext(); + if (tail) { + bitstream->Release(tail->frameId); + } + FrameSubmit after; + after.frameId = 3001; + after.registeredImage = *registered; + after.image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + Check(!submitter->SubmitFrame(after), + "a submission after the stream ends is refused"); + + // The encoder owns the completion semaphore and releases it with + // the stream, so the currency degrades to VK_NULL_HANDLE rather + // than leaving a caller waiting on a destroyed handle. + Ref completion = Query(session); + if (completion) { + Check(completion->CompletionSemaphore() == VK_NULL_HANDLE, + "the completion semaphore is withdrawn once the stream ends"); + } + } + } + + Check(static_cast(registry->UnregisterImage(*registered)), + "a registered image unregisters"); + Check(!registry->UnregisterImage(kNoResource), + "unregistering nothing is refused rather than ignored"); + + fns.DeviceWaitIdle(binding->Device()); + return true; +} + +} // namespace + +int main(int argc, const char** argv) +{ + (void)argc; + (void)argv; + + fprintf(stderr, "-- checks that need no device --\n"); + TestResultIsCheap(); + TestExpectedCarriesEither(); + TestArrayView(); + TestFormatClassification(); + + fprintf(stderr, "-- checks that need an encode device --\n"); + PlatformCreateInfo info; + info.silenceStdio = false; + + Ref platform; + const VkResult created = VkEncCreatePlatform(info, platform); + if (created != VK_SUCCESS || !platform) { + Skip("every device-backed check", "no Vulkan encode device"); + fprintf(stderr, "\n%d checks, %d failures, %d skipped\n", + g_checks, g_failures, g_skipped); + // A missing device is not a test failure, but it is not a pass of the + // device-backed checks either, and the count above says so. + return g_failures == 0 ? 0 : 1; + } + + Check(Query(platform) != nullptr, "the platform answers its own id"); + Check(platform->DeviceBinding() != nullptr, "the platform vends a device binding"); + + const bool haveEncodeDevice = TestCaps(platform); + + // The configuration rules are the library's own and hold with or without a + // device, so they are checked either way. + TestConfigRules(platform); + + if (!haveEncodeDevice) { + Skip("session role checks", "no encode-capable device"); + fprintf(stderr, "\n%d checks, %d failures, %d skipped\n", + g_checks, g_failures, g_skipped); + return g_failures == 0 ? 0 : 1; + } + + Expected> config = + platform->CreateConfig(Codec::H264, Profile::H264Main); + if (config) { + // A rate-controlled session, because rateControlMode is immutable: + // a session created without one cannot be moved into CBR later, and + // the reconfiguration checks below are about the rate, not the mode. + RateControl rate; + rate.mode = RateControlMode::Cbr; + rate.averageBitrate = 2 * 1000 * 1000; + rate.maxBitrate = 2 * 1000 * 1000; + + (*config)->SetCodedExtent(320, 240) + .SetFrameRate(30, 1) + .SetInputFormat(VK_FORMAT_G8_B8R8_2PLANE_420_UNORM, ColorModel::YCbCr) + .SetRateControl(rate); + Expected> session = platform->CreateSession(*config); + if (session) { + TestSessionRoles(*session); + TestFrameTransfer(platform, *session, *config); + } else { + Skip("session role checks", session.status().detail()); + } + } + + fprintf(stderr, "\n%d checks, %d failures, %d skipped\n", + g_checks, g_failures, g_skipped); + return g_failures == 0 ? 0 : 1; +} diff --git a/vk_video_encoder/test/encoder-release-fence/CMakeLists.txt b/vk_video_encoder/test/encoder-release-fence/CMakeLists.txt new file mode 100644 index 00000000..ffc17620 --- /dev/null +++ b/vk_video_encoder/test/encoder-release-fence/CMakeLists.txt @@ -0,0 +1,28 @@ +# The release fence: when the caller may write the buffer again. +# +# Drives the encoder interface only. Skips without an encode-capable GPU, and +# because the property under test is one the interface promises and a caller +# can observe. Skips (exit 0 with a skip count) without an encode-capable GPU. + +set(TEST_NAME vk-video-enc-release-fence-test) + +add_executable(${TEST_NAME} src/main.cpp) + +target_include_directories(${TEST_NAME} PRIVATE + ${CMAKE_CURRENT_SOURCE_DIR}/../common + ${CMAKE_CURRENT_SOURCE_DIR}/../../include + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} +) + +target_link_libraries(${TEST_NAME} PRIVATE vkvideo-encoder-static ${CMAKE_DL_LIBS}) + +if (TARGET GenVulkanDispatchTable) + add_dependencies(${TEST_NAME} GenVulkanDispatchTable) +endif() + +add_test(NAME ${TEST_NAME} COMMAND ${TEST_NAME}) +# A run that skips every check reports SKIP, not PASS: see Report::Summarise. +set_tests_properties(${TEST_NAME} PROPERTIES + LABELS "gpu" + SKIP_RETURN_CODE 77) diff --git a/vk_video_encoder/test/encoder-release-fence/src/main.cpp b/vk_video_encoder/test/encoder-release-fence/src/main.cpp new file mode 100644 index 00000000..db04f7b1 --- /dev/null +++ b/vk_video_encoder/test/encoder-release-fence/src/main.cpp @@ -0,0 +1,252 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// The release fence: when may the caller write the buffer again. +// +// A producer that hands the encoder a buffer needs to know when the encoder +// has finished READING it, which is earlier than when the bitstream appears +// and is the only thing that lets the buffer be recycled. FrameSubmit asks for +// that answer by naming storage; the library writes a descriptor into it. +// +// FOUR PROPERTIES, AND THE THIRD IS THE ONE THAT MATTERS. +// +// * The storage is written on every submission, before anything can refuse. +// A caller that re-submits must not inherit the previous attempt's fd. +// * What arrives is usable, and belongs to the caller, who closes it. +// * SOME OF THEM ARE UNSIGNALLED WHEN HANDED OVER. A fence that is already +// signalled every time is indistinguishable from no fence at all: the +// test would pass against an encoder that returns a pre-signalled +// descriptor and never orders anything. Requiring that a fair share are +// still pending at handover is what makes the rest of this a measurement. +// * Every one of them eventually signals. A fence that never does is worse +// than none: the producer waits forever on a buffer it owns. +// +// And across all of it the process must not accumulate descriptors, which is +// checked by counting them rather than by trusting the arithmetic. + +#include "encoder_test_support.h" + +#include +#include +#include +#include +#include + +using namespace vk::video::enc; + +namespace { + +// FULL HD, AND ENOUGH FRAMES TO MEASURE. The handover check below asks +// whether a fence is still pending when the caller receives it, and at a small +// enough frame the encode finishes inside the submission call -- so every +// fence comes back signalled and the check reports a property of the frame +// size rather than of the encoder. +const uint32_t kWidth = 1920; +const uint32_t kHeight = 1080; +const uint32_t kFrames = 64; + +// A value no fence descriptor can be, so "was it written" is answerable +// without trusting the library to have written something plausible. +const int kSentinel = 0x5EED; + +// How many descriptors this process holds. Counted, not inferred: an +// arithmetic argument about closes is exactly the thing a leak defeats. +int OpenDescriptorCount() +{ + DIR* dir = opendir("/proc/self/fd"); + if (dir == nullptr) { + return -1; + } + int count = 0; + while (readdir(dir) != nullptr) { + ++count; + } + closedir(dir); + return count; +} + +// Whether a sync descriptor has signalled, without waiting for it. +bool HasSignalled(int fd) +{ + struct pollfd p = {fd, POLLIN, 0}; + return poll(&p, 1, 0) > 0; +} + +// Wait for one, bounded. A fence that never signals is a defect, not a reason +// to hang the suite. +bool WaitSignalled(int fd, int timeoutMs) +{ + struct pollfd p = {fd, POLLIN, 0}; + return poll(&p, 1, timeoutMs) > 0; +} + +} // namespace + +int main(int argc, const char** argv) +{ + (void)argc; + (void)argv; + + enctest::Report report; + + enctest::Session s; + const char* why = ""; + if (!s.Open(kWidth, kHeight, &why)) { + report.Skip("every check", why); + return report.Summarise("encoder-release-fence"); + } + + enctest::DeviceFns fns; + if (!fns.Load(s.binding)) { + report.Skip("every check", "device entry points unavailable"); + return report.Summarise("encoder-release-fence"); + } + + enctest::HostImage image(fns, s.binding->PhysicalDevice(), s.binding->Device(), + kWidth, kHeight); + if (!image.Create(&why)) { + report.Skip("every check", why); + return report.Summarise("encoder-release-fence"); + } + + Expected registered = s.registry->RegisterImage(image.Describe()); + if (!registered) { + report.Skip("every check", registered.status().detail()); + return report.Summarise("encoder-release-fence"); + } + + const int fdBefore = OpenDescriptorCount(); + + uint32_t submitted = 0; + uint32_t retiredDuringSubmit = 0; + uint32_t written = 0; + uint32_t usable = 0; + uint32_t pendingAtHandover = 0; + std::vector fences; + + for (uint32_t i = 0; i < kFrames; ++i) { + if (!image.WritePattern(i, &why)) { + break; + } + + // Armed with the sentinel, so "written" is a fact about this call and + // not a leftover from the last one. + int releaseFenceFd = kSentinel; + + FrameSubmit frame; + frame.frameId = 1000 + i; + frame.pts = i * 3000; + frame.registeredImage = *registered; + frame.image.layout = VK_IMAGE_LAYOUT_PREINITIALIZED; + frame.releaseFenceFd = &releaseFenceFd; + + // BACKPRESSURE IS NOT A FAILURE. A full pipeline refuses with NotReady, + // which means park the frame and re-submit once something has been + // taken out -- so that is what a caller does, and what this does. + // Treating it as a refusal would make this test a measurement of the + // queue depth instead of the fence. + Result sent = s.submitter->SubmitFrame(frame); + for (int attempt = 0; !sent && sent.code() == ResultCode::NotReady && + attempt < 64; ++attempt) { + retiredDuringSubmit += s.DrainRetrievable(); + releaseFenceFd = kSentinel; + sent = s.submitter->SubmitFrame(frame); + } + if (!sent) { + fprintf(stderr, " submission refused: %s\n", sent.detail()); + break; + } + ++submitted; + + if (releaseFenceFd != kSentinel) { + ++written; + } + if (releaseFenceFd >= 0) { + ++usable; + if (!HasSignalled(releaseFenceFd)) { + ++pendingAtHandover; + } + fences.push_back(releaseFenceFd); + } + } + + report.Check(submitted == kFrames, "every frame is accepted"); + report.Check(written == submitted, + "the release-fence slot is written on every submission"); + + if (usable == 0) { + // A library that answers "no fence" for every frame is a legal + // implementation of the interface, and nothing below can be measured + // against it. That is a property of this encoder, not a failure. + report.Skip("release-fence behaviour", + "this encoder returns no release fence for any frame"); + } else { + report.Check(usable == submitted, + "every accepted frame yields a usable release fence"); + + fprintf(stderr, " %u of %u fences were still pending at handover\n", + pendingAtHandover, usable); + // A FLOOR, not "> 0". The count exists to rule out a fence exported + // after the fact, which would come back already signalled on nearly + // every frame -- and "> 0" passes that as long as a single frame races + // ahead, which is the exact state the check is written to forbid. + report.Check(pendingAtHandover >= (usable / 2), + "release fences are handed over before they have signalled"); + + s.session->Drain(); + + uint32_t signalled = 0; + for (int fd : fences) { + if (WaitSignalled(fd, 5000)) { + ++signalled; + } + } + report.Check(signalled == usable, "every release fence signals"); + } + + // The bitstream is checked too: a run that ordered its fences perfectly and + // encoded nothing has measured the wrong thing. + uint32_t frames = 0; + size_t bytes = 0; + for (;;) { + Expected got = s.bitstream->AcquireNext(); + if (!got) { + break; + } + ++frames; + bytes += got->bitstream.size(); + s.bitstream->Release(got->frameId); + } + frames += retiredDuringSubmit; + fprintf(stderr, " %u frames (%u retired under backpressure), " + "%zu bitstream bytes\n", frames, retiredDuringSubmit, bytes); + report.Check(frames > 0, "the encode produced frames"); + report.Check(bytes > 0, "the encode produced a bitstream"); + + for (int fd : fences) { + close(fd); + } + s.registry->UnregisterImage(*registered); + fns.DeviceWaitIdle(s.binding->Device()); + image.Destroy(); + + const int fdAfter = OpenDescriptorCount(); + fprintf(stderr, " descriptors: %d before, %d after\n", fdBefore, fdAfter); + report.Check((fdBefore >= 0) && (fdAfter >= 0) && (fdAfter <= fdBefore), + "the run holds no more descriptors than it started with"); + + return report.Summarise("encoder-release-fence"); +} diff --git a/vk_video_encoder/test/encoder-sync-assembly/CMakeLists.txt b/vk_video_encoder/test/encoder-sync-assembly/CMakeLists.txt new file mode 100644 index 00000000..6039d446 --- /dev/null +++ b/vk_video_encoder/test/encoder-sync-assembly/CMakeLists.txt @@ -0,0 +1,85 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) + +project(encoder_sync_assembly_test LANGUAGES CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +add_executable(${PROJECT_NAME} src/main.cpp) + +# Links the STATIC encoder library, not the shared one: this TU subclasses +# VkVideoEncoder itself, and none of that class is exported from +# libvkvideo-encoder.so. +target_link_libraries(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_STATIC_LIB} +) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + # VkVideoEncoder/VkVideoEncoder.h and VkVideoEncoder/VkEncoderConfigH264.h + # are library-internal headers, not part of the installed include dir. + # This is the same root the library's own TUs compile against. + ${VULKAN_VIDEO_ENCODER_INCLUDE}/../libs + # The encoder header reaches VkCodecUtils/* and mio/mio.hpp under the + # shared common-libs root. + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + # nvidia_utils/vulkan/ycbcrvkinfo.h resolves from the repository root. + ${CMAKE_CURRENT_LIST_DIR}/../../.. + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +# Vulkan is loaded at runtime (VK_NO_PROTOTYPES), so only headers are needed. +find_package(Vulkan QUIET) +if(Vulkan_FOUND AND TARGET Vulkan::Vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE Vulkan::Vulkan) +elseif(TARGET vulkan) + target_link_libraries(${PROJECT_NAME} PRIVATE vulkan) +endif() + +if(UNIX AND NOT APPLE) + target_link_libraries(${PROJECT_NAME} PRIVATE pthread dl) +endif() + +target_compile_definitions(${PROJECT_NAME} PRIVATE + VK_NO_PROTOTYPES + VK_ENABLE_BETA_EXTENSIONS + VK_USE_VIDEO_QUEUE + VK_USE_VIDEO_DECODE_QUEUE + VK_USE_VIDEO_ENCODE_QUEUE +) + +install(TARGETS ${PROJECT_NAME} + RUNTIME DESTINATION bin +) + +# Add tests. +# +# NOTE ON CTest SEMANTICS, matching the sibling library tests: 0 means every +# assertion held and 1 means an assertion failed. There is NO skip code here +# and deliberately so -- the subject has no GPU, driver, display or input-file +# dependence of any kind, so a host that could not run it is a broken host, +# not an unsupported one. +enable_testing() +add_test(NAME EncoderSyncAssemblyPublishesNoCompletionRecord + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderSyncAssemblyPublishesNoCompletionRecord PROPERTIES + LABELS "device-free") + +message(STATUS "encoder_sync_assembly_test: Configured") diff --git a/vk_video_encoder/test/encoder-sync-assembly/src/main.cpp b/vk_video_encoder/test/encoder-sync-assembly/src/main.cpp new file mode 100644 index 00000000..fc36a98a --- /dev/null +++ b/vk_video_encoder/test/encoder-sync-assembly/src/main.cpp @@ -0,0 +1,427 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +/* + * VkVideoEncoder::AssembleBitstreamData -- the SYNCHRONOUS assembly path -- + * must publish NO completion record. + * + * THE CONTRACT THIS PINS. This tree has exactly one producer of a + * CapturedBitstream: PushCapturedBitstream (VkVideoEncoder.h). It is reached + * from four places, and NONE of them is the synchronous assembly path -- + * WriteBitstreamToFile (called only by the assembly worker), the worker's own + * readback-failure arm which calls it directly, the AV1 override, and a + * null-backend test seam in the ext layer. The sibling block in + * test/encoder-ext-drain-assembly/src/main.cpp enumerates the same four. That + * is not an incidental arrangement; seven sites in the tree are built on it + * and say so in as many words: + * + * VkVideoEncoder.cpp ProcessOrderedFrames hard "return VK_ERROR_UNKNOWN" + * guard, and that guard user-facing error text + * VkVideoEncoder.h the DrainAndRestartThreads contract; and the + * WriteDataToFile contract, which names this exact CLI + * pair -- "under --syncAssembly --disableFileOutput + * they reach neither ... encode-and-discard by + * construction" + * vulkan_video_encoder_ext.cpp the justification for pinning + * cfg->asyncAssembly = 1, and the drain-restart note + * test/encoder-ext-drain-assembly/src/main.cpp its documented premise, and + * its record of why the alternative cannot work: + * publishing from the synchronous path makes records + * arrive before their PendingFrame exists, so + * DrainCapturesLocked discards them and m_lateCaptures + * fills with noise. + * + * WHY IT NEEDS ITS OWN TEST. The sync path is unreachable from the ext public + * surface by three independent closures -- asyncAssembly is pinned on, a + * completion subscriber is always registered, and ProcessOrderedFrames refuses + * the sync fallback whenever a subscriber exists -- so no encoder-ext harness + * can execute the function at all, and the file-based CLI that CAN reach it + * hands out no VkVideoEncoder to observe. encoder-ext-drain-assembly is green + * on a tree where this contract is broken, twice over: it runs both its + * batches with async assembly ON, and its only assertions are LOWER bounds + * (retired > 0), which are structurally blind to an EXTRA publish. + * + * WHY NO DEVICE. Every step of AssembleBitstreamData except the readback is + * pure bookkeeping over a frame-info struct. So this subclasses the real + * VkVideoEncoder with a null device context -- the shape the ext layer own + * capture-funnel seam already uses -- overrides ReadbackBitstreamData to hand + * back a synthetic payload, and calls the PRODUCTION AssembleBitstreamData. + * Nothing else is stubbed or reimplemented, which is what makes a failure here + * attributable to that function rather than to the harness. No GPU, no driver, + * no display, no worker thread, no input file: it is deterministic on any + * host, which is why it is labelled device-free and not skippable. + * + * ARM 2 IS NOT DECORATION. Arm 1 alone ("publish nothing") is satisfied by + * simply deleting the write, which would be a far worse bug than the one being + * pinned. Arm 2 runs the same call with real file output and asserts every + * coded byte still reaches the file, so that mutation fails. It passes in both + * states by design -- it is a guard rail, not a detector. + * + * The edge counters are separate from the record counters because + * PushCapturedBitstream raises the completion edge OUTSIDE its own store + * condition: an arm that stores nothing can still raise the edge, and only a + * separate counter sees that. + */ + +#include "VkVideoEncoder/VkVideoEncoder.h" +#include "VkVideoEncoder/VkEncoderConfigH264.h" + +// The encoder headers reach the Xlib platform headers, whose macros collide +// with ordinary identifiers. Scrub them before anything else sees them -- the +// same block, for the same reason, as the sibling library test TUs. +#undef Status +#undef None +#undef Bool +#undef Window + +#include +#include +#include +#include +#include +#include + +namespace { + +int g_checks = 0; +int g_failures = 0; + +std::string U64(uint64_t v) +{ + char buf[32]; + std::snprintf(buf, sizeof(buf), "%llu", (unsigned long long)v); + return std::string(buf); +} + +std::string Hex32(VkResult v) +{ + char buf[16]; + std::snprintf(buf, sizeof(buf), "0x%x", (unsigned)v); + return std::string(buf); +} + +void Check(bool cond, const char* what, const std::string& detail) +{ + ++g_checks; + if (cond) { + std::printf(" ok %s\n", what); + } else { + ++g_failures; + std::printf(" FAIL %s -- %s\n", what, detail.c_str()); + } +} + +// The synthetic frame. Sizes are arbitrary but fixed, so the bytes-on-disk +// assertion in arm 2 is an exact equality rather than a bound. +const uint32_t kPayloadBytes = 4096; +const uint32_t kHeaderBytes = 32; + +// The real VkVideoEncoder, with a null device context and the codec pure +// virtuals stubbed. None of the stubs is reachable from AssembleBitstreamData; +// they exist only because the class is abstract. The base constructor is pure +// member-init and the destruction path guards the missing device and joins no +// threads (none were started), so the null context is tolerated -- the same +// shape VkEncNullBackendCaptureSource relies on in the ext layer. +class SyncAssemblyProbe : public VkVideoEncoder { +public: + SyncAssemblyProbe() : VkVideoEncoder(nullptr) {} + + // Configure as "--syncAssembly --disableFileOutput" (captureMode) or as + // "--syncAssembly -o outputPath". Returns false if the file could not be + // opened, which is a harness failure, not an assertion failure. + bool Configure(bool captureMode, const char* outputPath) + { + VkSharedBaseObj cfg(new EncoderConfigH264()); + cfg->disableFileOutput = captureMode ? 1 : 0; + if (!captureMode) { + if (cfg->outputFileHandler.SetFileName(outputPath) == 0) { + return false; + } + } + m_encoderConfig = cfg; + return true; + } + + // Count the completion edge. Deliberately attached in BOTH arms: the edge + // is raised outside the PushCapturedBitstream store condition, so it is + // the only observable that reports a publish in file-output mode. + void CountEdges(uint32_t* counter) + { + SetOnBitstreamCaptured([counter](uint64_t) { ++(*counter); }); + } + + // Close the output file so its size can be read back. The config own + // destructor would do this, but the assertion needs it flushed first. + void CloseOutput() + { + if (m_encoderConfig) { + m_encoderConfig->outputFileHandler.Destroy(); + } + } + + // THE ONLY device-touching step of the synchronous path, replaced by a + // synthetic payload. The bitstreamCopy arm is what lets the funnel and the + // file-output arm both read a payload with no outputBitstreamBuffer bound; + // both already honour it, for the assembly worker sake. + VkResult ReadbackBitstreamData(VkSharedBaseObj&, + BitstreamReadback& readback) override + { + readback.bitstreamStartOffset = 0; + readback.bitstreamSize = kPayloadBytes; + readback.status = VK_QUERY_RESULT_STATUS_COMPLETE_KHR; + readback.bitstreamCopy.assign(kPayloadBytes, 0xA5); + readback.readbackDone = true; + return VK_SUCCESS; + } + + // Drain the completion FIFO through the PRODUCTION accessor -- the same + // call the ext layer DrainCapturesLocked makes -- and report how many + // records the synchronous path left behind. + uint32_t DrainRecords(size_t* outBytes) + { + uint32_t n = 0; + *outBytes = 0; + uint64_t frameId = 0; + std::vector bytes; + bool isIdr = false; + uint32_t pictureType = 0; + VkResult status = VK_SUCCESS; + while (TryPopCapturedBitstream(&frameId, &bytes, &isIdr, + &pictureType, &status)) { + ++n; + *outBytes += bytes.size(); + bytes.clear(); + } + return n; + } + + // --- codec pure virtuals: never reached from AssembleBitstreamData --- + VkResult CreateFrameInfoBuffersQueue(uint32_t) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + bool GetAvailablePoolNode(VkSharedBaseObj&) override { + return false; + } + VkResult InitEncoderCodec(VkSharedBaseObj&) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult EncodeFrame(VkSharedBaseObj&) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult CodecHandleRateControlCmd( + VkSharedBaseObj&) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult InitRateControl(VkCommandBuffer, uint32_t) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } + VkResult ProcessDpb(VkSharedBaseObj&, + uint32_t, uint32_t) override { + return VK_ERROR_FEATURE_NOT_PRESENT; + } +}; + +VkSharedBaseObj MakeFrame() +{ + VkSharedBaseObj frame( + new VkVideoEncoder::VkVideoEncodeFrameInfo()); + frame->bitstreamHeaderBufferSize = kHeaderBytes; + frame->bitstreamHeaderOffset = 0; + std::memset(frame->bitstreamHeaderBuffer, 0x5A, kHeaderBytes); + frame->frameEncodeInputOrderNum = 0; + return frame; +} + +long FileSize(const char* path) +{ + FILE* f = std::fopen(path, "rb"); + if (f == nullptr) { + return -1; + } + std::fseek(f, 0, SEEK_END); + long n = std::ftell(f); + std::fclose(f); + return n; +} + +// ARM 1 -- capture mode: the "--syncAssembly --disableFileOutput" shape. +// disableFileOutput alone satisfies the PushCapturedBitstream store condition, +// so this arm observes exactly what the CLI would retain, with no +// subscriber-induced distortion. +void CaseCaptureModePublishesNothing() +{ + std::printf("capture mode (--syncAssembly --disableFileOutput):\n"); + + SyncAssemblyProbe probe; + if (!probe.Configure(/*captureMode=*/true, nullptr)) { + Check(false, "harness: capture-mode config built", "Configure failed"); + return; + } + uint32_t edges = 0; + probe.CountEdges(&edges); + + VkSharedBaseObj frame = MakeFrame(); + VkResult result = probe.AssembleBitstreamData(frame, 0, 1); + + size_t retainedBytes = 0; + const uint32_t records = probe.DrainRecords(&retainedBytes); + std::printf(" assembled result=0x%x records published=%u" + " retained bytes=%zu completion edges=%u\n", + (unsigned)result, records, retainedBytes, edges); + + Check(result == VK_SUCCESS, + "the synchronous assembly itself still succeeds", + "result " + Hex32(result)); + Check(records == 0, + "AssembleBitstreamData leaves NO CapturedBitstream in the FIFO", + "records=" + U64(records) + " holding " + U64(retainedBytes) + + " bytes that nothing on this path will ever drain"); + Check(edges == 0, + "AssembleBitstreamData raises NO completion edge", + "edges=" + U64(edges)); +} + +// ARM 2 -- file output: the anti-mutation guard. Passes in both states. +void CaseFileOutputStillWritesEveryByte() +{ + std::printf("file output (--syncAssembly -o ):\n"); + + const char* path = "encoder_sync_assembly_out.bin"; + std::remove(path); + + uint32_t edges = 0; + { + SyncAssemblyProbe probe; + if (!probe.Configure(/*captureMode=*/false, path)) { + Check(false, "harness: output file opened", path); + return; + } + probe.CountEdges(&edges); + + VkSharedBaseObj frame = MakeFrame(); + VkResult result = probe.AssembleBitstreamData(frame, 0, 1); + probe.CloseOutput(); + + Check(result == VK_SUCCESS, + "the file-output assembly still succeeds", + "result " + Hex32(result)); + } + + const long onDisk = FileSize(path); + std::printf(" bytes on disk=%ld completion edges=%u\n", onDisk, edges); + + Check(onDisk == (long)(kHeaderBytes + kPayloadBytes), + "the file-output arm still writes header + every coded byte", + "wrote " + U64((uint64_t)(int64_t)onDisk) + ", want " + + U64(kHeaderBytes + kPayloadBytes)); + Check(edges == 0, + "the file-output arm raises NO completion edge either", + "edges=" + U64(edges)); + + std::remove(path); +} + +// GUARD 1 -- ProcessOrderedFrames refuses the synchronous fallback while a +// completion subscriber is registered. +// +// This is the guard that ENFORCES the contract arm 1 pins: arm 1 shows that +// AssembleBitstreamData publishes nothing, and THIS is what keeps a +// subscriber-bearing session off AssembleBitstreamData in the first place. It +// is quoted as a load-bearing site by five comments in the tree and, until +// now, nothing executed it. +// +// The preconditions are free here: m_asyncAssemblyEnabled defaults false, so +// the probe is already in the "async off" shape, and attaching a subscriber is +// the only other one. The guard returns before any device is touched. +void CaseOrderedSubscriberGuard() +{ + std::printf("ProcessOrderedFrames refuses a subscriber with async off:\n"); + + SyncAssemblyProbe probe; + if (!probe.Configure(/*captureMode=*/true, nullptr)) { + Check(false, "harness: capture-mode config built", "Configure failed"); + return; + } + uint32_t edges = 0; + probe.CountEdges(&edges); + Check(probe.HasCompletionSubscriber(), + "harness: a completion subscriber is registered", + "HasCompletionSubscriber() is false after SetOnBitstreamCaptured"); + + VkSharedBaseObj frame = MakeFrame(); + const VkResult result = probe.ProcessOrderedFrames(frame, 1); + std::printf(" ProcessOrderedFrames -> %s completion edges=%u\n", + Hex32(result).c_str(), edges); + + Check(result == VK_ERROR_UNKNOWN, + "ProcessOrderedFrames refuses rather than encoding unreportably", + "got " + Hex32(result) + ", want VK_ERROR_UNKNOWN " + + Hex32(VK_ERROR_UNKNOWN) + " -- without the guard the call falls " + "through into the callback sequence instead of refusing"); + Check(edges == 0, + "the refused call publishes nothing", + "edges=" + U64(edges)); +} + +// GUARD 2 -- the mirror in ProcessOutOfOrderFrames, added by the same commit +// as the fix and, until now, shipped with a comment asserting no test could +// construct the state that fires it. +void CaseOutOfOrderSubscriberGuard() +{ + std::printf("ProcessOutOfOrderFrames refuses a subscriber with async off:\n"); + + SyncAssemblyProbe probe; + if (!probe.Configure(/*captureMode=*/true, nullptr)) { + Check(false, "harness: capture-mode config built", "Configure failed"); + return; + } + uint32_t edges = 0; + probe.CountEdges(&edges); + + VkSharedBaseObj frame = MakeFrame(); + const VkResult result = probe.ProcessOutOfOrderFrames(frame, 1); + std::printf(" ProcessOutOfOrderFrames -> %s completion edges=%u\n", + Hex32(result).c_str(), edges); + + Check(result == VK_ERROR_UNKNOWN, + "ProcessOutOfOrderFrames refuses rather than encoding unreportably", + "got " + Hex32(result) + ", want VK_ERROR_UNKNOWN " + + Hex32(VK_ERROR_UNKNOWN) + " -- without the guard the call falls " + "through into the callback sequence instead of refusing"); + Check(edges == 0, + "the refused call publishes nothing", + "edges=" + U64(edges)); +} + +} // namespace + +int main(int argc, char** argv) +{ + (void)argc; + (void)argv; + std::printf("Synchronous AssembleBitstreamData publishes no completion record\n"); + std::printf("---------------------------------------------------------------\n"); + + CaseCaptureModePublishesNothing(); + CaseFileOutputStillWritesEveryByte(); + CaseOrderedSubscriberGuard(); + CaseOutOfOrderSubscriberGuard(); + + std::printf("---------------------------------------------------------------\n"); + std::printf("checks: %d, failures: %d\n", g_checks, g_failures); + std::printf("RESULT: %s\n", (g_failures == 0) ? "PASS" : "FAIL"); + return (g_failures == 0) ? 0 : 1; +} diff --git a/vk_video_encoder/test/gop-structure/CMakeLists.txt b/vk_video_encoder/test/gop-structure/CMakeLists.txt new file mode 100644 index 00000000..0c796a32 --- /dev/null +++ b/vk_video_encoder/test/gop-structure/CMakeLists.txt @@ -0,0 +1,64 @@ +# Copyright 2026 NVIDIA Corporation. +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +cmake_minimum_required(VERSION 3.20) +project(gop_structure_test LANGUAGES CXX) +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) + +# VkVideoGopStructure.cpp is compiled in rather than linked from the encoder +# library on purpose: the class it defines includes no Vulkan header and needs +# no device, and compiling it here keeps this test free of the library's link +# closure. That is what lets it carry the device-free label honestly. +add_executable(${PROJECT_NAME} + src/main.cpp + ${CMAKE_SOURCE_DIR}/vk_video_encoder/libs/VkVideoEncoder/VkVideoGopStructure.cpp) + +target_include_directories(${PROJECT_NAME} PRIVATE + ${VULKAN_VIDEO_ENCODER_INCLUDE} + ${VK_VIDEO_ENCODER_LIBS_SOURCE_ROOT} + ${VK_VIDEO_COMMON_LIBS_SOURCE_ROOT} + ${VULKAN_VIDEO_APIS_INCLUDE} + ${VULKAN_HEADERS_INCLUDE_DIR} + ${Vulkan_INCLUDE_DIR} +) + +install(TARGETS ${PROJECT_NAME} RUNTIME DESTINATION bin) + +# Add tests. +# +# WHAT THIS GATES. The GOP machine's frame sequencing for one configuration -- +# gopFrameCount 11, idrPeriod 25, 3 consecutive B frames, open GOP -- checked +# frame by frame over 30 frames against three recorded tables: the encode +# order, the B-frame count at each position, and the B-frame position within +# each mini-GOP. It also asserts that FLAGS_CLOSE_GOP is never set on an IDR +# in an open-GOP configuration. +# +# WHAT A GREEN HERE DOES NOT MEAN. Nothing is encoded and no device is +# involved. This is the sequencing decision only, taken before any picture +# exists, so a green says the order the encoder intends is the order recorded +# -- not that the encoder coded it, and not that any other GOP configuration +# sequences correctly. printGopTable() and testEdgeCases() in the same binary +# print and assert nothing; they are not part of the gate. +# +# CTest semantics, matching the sibling library tests: 0 means every +# expectation held, 1 means one missed. +enable_testing() +add_test(NAME EncoderGopStructureSequencing + COMMAND ${PROJECT_NAME}) +# LABELS: this is the CI gating set. See the top-level CMakeLists.txt note. +set_tests_properties(EncoderGopStructureSequencing PROPERTIES + LABELS "device-free") + +message(STATUS "gop_structure_test: Configured") diff --git a/vk_video_encoder/test/gop_structure_test.cpp b/vk_video_encoder/test/gop-structure/src/main.cpp similarity index 92% rename from vk_video_encoder/test/gop_structure_test.cpp rename to vk_video_encoder/test/gop-structure/src/main.cpp index cd844a0d..31ed95d8 100644 --- a/vk_video_encoder/test/gop_structure_test.cpp +++ b/vk_video_encoder/test/gop-structure/src/main.cpp @@ -105,7 +105,10 @@ void printGopTable(uint32_t gopFrameCount, uint32_t idrPeriod, uint32_t consecut std::cout << "\n"; } -void verifyExpectedValues() { +// Returns false if any expectation missed, so main() can carry it to the +// exit code. Returning void and only printing the result would make every +// expectation below unable to fail anything. +bool verifyExpectedValues() { std::cout << "\n=== Verifying Expected Values ===\n"; VkVideoGopStructure gop(11, 25, 3, 1, @@ -176,6 +179,7 @@ void verifyExpectedValues() { } else { std::cout << "\n=== SOME TESTS FAILED ===\n"; } + return allPassed; } void testEdgeCases() { @@ -203,10 +207,13 @@ int main() { std::cout << "Testing: GOP=11, IDR=25, B=3, Open GOP\n"; printGopTable(11, 25, 3, 30); - - verifyExpectedValues(); - + + const bool passed = verifyExpectedValues(); + testEdgeCases(); - - return 0; + + // CTest semantics, as the sibling suites use them: 0 every expectation + // held, 1 an expectation missed. printGopTable and testEdgeCases print + // only and assert nothing, so verifyExpectedValues is the whole gate. + return passed ? 0 : 1; } diff --git a/vk_video_encoder/test/qfot_foreign_release_repro.c b/vk_video_encoder/test/qfot_foreign_release_repro.c new file mode 100644 index 00000000..8ae82223 --- /dev/null +++ b/vk_video_encoder/test/qfot_foreign_release_repro.c @@ -0,0 +1,423 @@ +/* + * Copyright 2026 NVIDIA Corporation. + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +// qfot_foreign_release_repro.c -- a DRIVER-DEFECT reproducer, not a test. +// +// --------------------------------------------------------------------------- +// WHY THIS IS NOT WIRED INTO CMake OR ctest, AND MUST NOT BE. +// --------------------------------------------------------------------------- +// Every RED mode here DESTROYS THE VULKAN DEVICE and raises an Xid on the +// running GPU. Under a test runner that is not a failing assertion, it is a +// machine-wide event that takes out whatever else is using the card. Build it +// by hand, on a host you are willing to disturb: +// +// gcc -O0 -g -o qfot_foreign_release_repro qfot_foreign_release_repro.c -lvulkan +// +// It links no part of this library. It creates no VkVideoSessionKHR, records +// no CmdBeginVideoCodingKHR and no CmdEncodeVideoKHR, and runs no encode. One +// image, one command buffer, one submit, one fence wait. The ONLY thing that +// varies between modes is which image memory barriers are recorded. +// +// --------------------------------------------------------------------------- +// WHAT IT ISOLATES. +// --------------------------------------------------------------------------- +// VkVideoEncoder::RecordVideoCodingCmd's Path-A queue-family RELEASE (see +// VkVideoEncoder::ReleaseImageToForeignQueue) loses the device on an +// affected driver. The encoder-ext-format-encode --foreign-residency DIRECT +// rows are +// where it shows: VK_ERROR_DEVICE_LOST, a zero-byte bitstream, and the +// validation layer silent throughout. That barrier is spec-legal -- VVL 1.4.355 +// raises nothing about it, the acquire/release are balanced, the transfer is +// recorded OUTSIDE the video coding scope, and the aspect mask is right for a +// non-disjoint multi-planar image. +// +// --------------------------------------------------------------------------- +// THE DEFECT, as a conjunction. +// --------------------------------------------------------------------------- +// The device is lost iff ALL THREE hold: +// +// (1) the image was created with VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR or +// VK_IMAGE_USAGE_VIDEO_ENCODE_DPB_BIT_KHR; +// (2) the barrier is a RELEASE -- dstQueueFamilyIndex is +// VK_QUEUE_FAMILY_FOREIGN_EXT; +// (3) it is recorded on a queue family other than graphics or optical flow. +// +// Failure signature: Xid 32 (invalid or corrupted push buffer stream) on the +// recording engine's channel, "HCE_DBG0 00000124 HCE_DBG1 00000002" followed +// immediately by "HCE_DBG0 00000800". VK_EXT_device_fault is supported and +// returns addressInfoCount=0, vendorInfoCount=0 and an empty description -- +// consistent with a method-parse error rather than a memory fault. +// +// RESULT MATRIX (each row one process; verdict is vkWaitForFences): +// +// ctl no ownership transfer SURVIVED +// acq FOREIGN -> encode, acquire direction SURVIVED +// rel encode -> FOREIGN, no layout change LOST +// rel-xition encode -> FOREIGN, ENCODE_SRC->GENERAL LOST +// pair acq then rel-xition (Path-A shape) LOST +// rel --home-general GENERAL -> GENERAL, no transition LOST +// rel --stage-none srcStageMask NONE LOST +// rel --stage-transfer srcStageMask TRANSFER LOST +// rel --legacy same release via v1 vkCmdPipelineBarrier LOST +// rel --dpb VIDEO_ENCODE_DPB usage instead of SRC LOST +// rel-external dstQF = VK_QUEUE_FAMILY_EXTERNAL SURVIVED +// rel-gfx dstQF = a real family (graphics) SURVIVED +// +// FULL RECORDING-FAMILY SWEEP, one process each, `rel` in every row. The +// device's six families on this part are 0 graphics|compute|transfer|sparse, +// 1 transfer|sparse, 2 compute|transfer|sparse, 3 transfer|sparse|video +// decode, 4 transfer|sparse|video encode, 5 transfer|sparse|optical flow: +// +// --fam 0 (graphics) SURVIVED +// --fam 1 (transfer) LOST +// --fam 2 (compute) LOST +// --fam 3 (video decode) LOST +// --fam 4 (video encode) LOST <- the default, and Path A's family +// --fam 5 (optical flow) SURVIVED +// rel --novideo no video usage bits on the image SURVIVED +// rel --profile-only video profile list, NO video usage SURVIVED +// rel --rgba R8G8B8A8 TRANSFER image SURVIVED +// +// READ THE GREEN ROWS AS CAREFULLY AS THE RED ONES: +// * --profile-only green and --novideo green isolate the USAGE BIT. It is +// the NVENC input-surface allocation, not the video profile and not the +// multi-planar format. +// * --home-general LOST is strictly stronger than "changing newLayout does +// not help": that mode performs no layout transition at all. +// * --legacy LOST rules out the barrier API generation, so the library +// cannot dodge this by re-expressing the release. +// * --fam 5 SURVIVED while --fam 1 LOST, and family 5 (optical flow) has no +// VK_QUEUE_GRAPHICS_BIT either -- both are plain transfer|sparse plus one +// engine bit. So the rule is NOT "any non-graphics family"; graphics and +// optical flow are exempt and the other four engines are not. Stated as +// measured, because no theory offered so far predicts that split. +// +// --------------------------------------------------------------------------- +// WHY THE LIBRARY CANNOT ROUTE AROUND IT. +// --------------------------------------------------------------------------- +// A release must be recorded on a queue of its SOURCE family, and on Path A +// the only family that owns the input image is the encode family. Condition +// (3) is therefore not a choice the library gets to make. The four responses +// -- drop the Path-A release, substitute VK_QUEUE_FAMILY_EXTERNAL (which means +// "another Vulkan instance", not "a non-Vulkan agent", so it would be a +// correctness regression dressed as a fix), a two-hop encode->graphics->FOREIGN +// transfer costing a second per-frame submit on a second queue, or fix the +// driver -- are a design decision. None of them is applied in this tree. +// +// Attach this file as-is to a driver bug; it is self-contained. + +#include +#include +#include +#include + +#define CHK(x) do { VkResult _r = (x); if (_r != VK_SUCCESS) { \ + fprintf(stderr, "FAIL %s -> %d (line %d)\n", #x, (int)_r, __LINE__); exit(2);} } while(0) + +static const char* ResName(VkResult r) { + switch (r) { + case VK_SUCCESS: return "VK_SUCCESS"; + case VK_TIMEOUT: return "VK_TIMEOUT"; + case VK_ERROR_DEVICE_LOST: return "VK_ERROR_DEVICE_LOST"; + default: return "other"; + } +} + +int main(int argc, char** argv) +{ + const char* mode = (argc > 1) ? argv[1] : "ctl"; + int useRgba = 0, useGfxQueue = 0, noVideoUsage = 0, homeGeneral = 0, famOverride = -1; + int relStage = 0; // 0 = home stage, 1 = TRANSFER, 2 = NONE + int profileOnly = 0, dpbUsage = 0, legacyBarrier = 0; + for (int i = 1; i < argc; i++) { + if (!strcmp(argv[i], "--rgba")) useRgba = 1; + if (!strcmp(argv[i], "--gfxq")) useGfxQueue = 1; + if (!strcmp(argv[i], "--novideo")) noVideoUsage = 1; + if (!strcmp(argv[i], "--home-general")) homeGeneral = 1; + if (!strcmp(argv[i], "--fam") && i + 1 < argc) famOverride = atoi(argv[i + 1]); + if (!strcmp(argv[i], "--stage-transfer")) relStage = 1; + if (!strcmp(argv[i], "--stage-none")) relStage = 2; + if (!strcmp(argv[i], "--profile-only")) profileOnly = 1; + if (!strcmp(argv[i], "--dpb")) dpbUsage = 1; + if (!strcmp(argv[i], "--legacy")) legacyBarrier = 1; + } + + VkApplicationInfo app = { VK_STRUCTURE_TYPE_APPLICATION_INFO }; + app.apiVersion = VK_API_VERSION_1_3; + app.pApplicationName = "qfot_repro"; + VkInstanceCreateInfo ici = { VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO }; + ici.pApplicationInfo = &app; + const char* instLayers[1]; uint32_t nInstLayers = 0; + if (getenv("QFOT_VALIDATE")) instLayers[nInstLayers++] = "VK_LAYER_KHRONOS_validation"; + ici.enabledLayerCount = nInstLayers; ici.ppEnabledLayerNames = instLayers; + VkInstance inst; CHK(vkCreateInstance(&ici, NULL, &inst)); + + uint32_t nPhys = 0; vkEnumeratePhysicalDevices(inst, &nPhys, NULL); + VkPhysicalDevice phys[8]; if (nPhys > 8) nPhys = 8; + vkEnumeratePhysicalDevices(inst, &nPhys, phys); + VkPhysicalDevice pd = VK_NULL_HANDLE; + uint32_t encFam = UINT32_MAX, gfxFam = UINT32_MAX; + for (uint32_t i = 0; i < nPhys && pd == VK_NULL_HANDLE; i++) { + uint32_t nq = 0; vkGetPhysicalDeviceQueueFamilyProperties(phys[i], &nq, NULL); + VkQueueFamilyProperties qf[16]; if (nq > 16) nq = 16; + vkGetPhysicalDeviceQueueFamilyProperties(phys[i], &nq, qf); + uint32_t e = UINT32_MAX, g = UINT32_MAX; + for (uint32_t q = 0; q < nq; q++) { + if ((qf[q].queueFlags & VK_QUEUE_VIDEO_ENCODE_BIT_KHR) && e == UINT32_MAX) e = q; + if ((qf[q].queueFlags & VK_QUEUE_GRAPHICS_BIT) && g == UINT32_MAX) g = q; + } + if (e != UINT32_MAX) { pd = phys[i]; encFam = e; gfxFam = g; + VkPhysicalDeviceProperties p; vkGetPhysicalDeviceProperties(pd, &p); + printf("device=%s encodeFamily=%u gfxFamily=%u\n", p.deviceName, e, g); + for (uint32_t q = 0; q < nq; q++) + printf(" family[%u] flags=0x%x count=%u\n", q, qf[q].queueFlags, qf[q].queueCount); + } + } + if (pd == VK_NULL_HANDLE) { fprintf(stderr, "no video-encode queue family\n"); return 3; } + + uint32_t submitFam = useGfxQueue ? gfxFam : encFam; + if (famOverride >= 0) submitFam = (uint32_t)famOverride; + + // Device extensions + uint32_t nExt = 0; vkEnumerateDeviceExtensionProperties(pd, NULL, &nExt, NULL); + VkExtensionProperties* ext = malloc(sizeof(*ext) * nExt); + vkEnumerateDeviceExtensionProperties(pd, NULL, &nExt, ext); + int hasFault = 0; + for (uint32_t i = 0; i < nExt; i++) + if (!strcmp(ext[i].extensionName, VK_EXT_DEVICE_FAULT_EXTENSION_NAME)) hasFault = 1; + printf("VK_EXT_device_fault present=%d\n", hasFault); + + const char* devExts[8]; uint32_t nDevExts = 0; + devExts[nDevExts++] = VK_KHR_VIDEO_QUEUE_EXTENSION_NAME; + devExts[nDevExts++] = VK_KHR_VIDEO_ENCODE_QUEUE_EXTENSION_NAME; + devExts[nDevExts++] = VK_KHR_VIDEO_ENCODE_H264_EXTENSION_NAME; + devExts[nDevExts++] = VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME; + if (hasFault) devExts[nDevExts++] = VK_EXT_DEVICE_FAULT_EXTENSION_NAME; + + float prio = 1.0f; + VkDeviceQueueCreateInfo dq[2]; uint32_t nDq = 0; + dq[nDq] = (VkDeviceQueueCreateInfo){ VK_STRUCTURE_TYPE_DEVICE_QUEUE_CREATE_INFO }; + dq[nDq].queueFamilyIndex = submitFam; dq[nDq].queueCount = 1; dq[nDq].pQueuePriorities = &prio; nDq++; + + VkPhysicalDeviceFaultFeaturesEXT faultF = { VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_FAULT_FEATURES_EXT }; + faultF.deviceFault = VK_TRUE; + VkPhysicalDeviceVulkan13Features f13 = { VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES }; + f13.synchronization2 = VK_TRUE; + if (hasFault) f13.pNext = &faultF; + VkDeviceCreateInfo dci = { VK_STRUCTURE_TYPE_DEVICE_CREATE_INFO }; + dci.pNext = &f13; + dci.queueCreateInfoCount = nDq; dci.pQueueCreateInfos = dq; + dci.enabledExtensionCount = nDevExts; dci.ppEnabledExtensionNames = devExts; + VkDevice dev; CHK(vkCreateDevice(pd, &dci, NULL, &dev)); + VkQueue queue; vkGetDeviceQueue(dev, submitFam, 0, &queue); + + PFN_vkGetDeviceFaultInfoEXT pfnFault = hasFault + ? (PFN_vkGetDeviceFaultInfoEXT)vkGetDeviceProcAddr(dev, "vkGetDeviceFaultInfoEXT") : NULL; + + // ---- image ------------------------------------------------------------- + VkVideoEncodeH264ProfileInfoKHR h264 = { VK_STRUCTURE_TYPE_VIDEO_ENCODE_H264_PROFILE_INFO_KHR }; + h264.stdProfileIdc = 77; // STD_VIDEO_H264_PROFILE_IDC_MAIN + VkVideoProfileInfoKHR prof = { VK_STRUCTURE_TYPE_VIDEO_PROFILE_INFO_KHR }; + prof.pNext = &h264; + prof.videoCodecOperation = VK_VIDEO_CODEC_OPERATION_ENCODE_H264_BIT_KHR; + prof.chromaSubsampling = VK_VIDEO_CHROMA_SUBSAMPLING_420_BIT_KHR; + prof.lumaBitDepth = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + prof.chromaBitDepth = VK_VIDEO_COMPONENT_BIT_DEPTH_8_BIT_KHR; + VkVideoProfileListInfoKHR plist = { VK_STRUCTURE_TYPE_VIDEO_PROFILE_LIST_INFO_KHR }; + plist.profileCount = 1; plist.pProfiles = &prof; + + VkImageCreateInfo ic = { VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO }; + ic.imageType = VK_IMAGE_TYPE_2D; + ic.extent = (VkExtent3D){ 1920, 1088, 1 }; + ic.mipLevels = 1; ic.arrayLayers = 1; + ic.samples = VK_SAMPLE_COUNT_1_BIT; + ic.tiling = VK_IMAGE_TILING_OPTIMAL; + ic.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + ic.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED; + if (useRgba) { + ic.format = VK_FORMAT_R8G8B8A8_UNORM; + ic.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + } else if (profileOnly) { + // Multi-planar NV12 carrying the video PROFILE LIST at creation but no + // video usage: separates "created against a video profile" from + // "created with VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR". + ic.pNext = &plist; + ic.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ic.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + } else if (noVideoUsage) { + // Same multi-planar NV12 format, NO video-encode usage and no profile + // list: separates "multi-planar image" from "video-encode resource". + ic.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ic.usage = VK_IMAGE_USAGE_TRANSFER_DST_BIT | VK_IMAGE_USAGE_TRANSFER_SRC_BIT; + } else { + ic.pNext = &plist; + ic.format = VK_FORMAT_G8_B8R8_2PLANE_420_UNORM; + ic.usage = (dpbUsage ? VK_IMAGE_USAGE_VIDEO_ENCODE_DPB_BIT_KHR + : VK_IMAGE_USAGE_VIDEO_ENCODE_SRC_BIT_KHR) | VK_IMAGE_USAGE_TRANSFER_DST_BIT; + } + VkImage img; CHK(vkCreateImage(dev, &ic, NULL, &img)); + + VkMemoryRequirements mr; vkGetImageMemoryRequirements(dev, img, &mr); + VkPhysicalDeviceMemoryProperties mp; vkGetPhysicalDeviceMemoryProperties(pd, &mp); + uint32_t mt = UINT32_MAX; + for (uint32_t i = 0; i < mp.memoryTypeCount; i++) + if ((mr.memoryTypeBits & (1u << i)) && + (mp.memoryTypes[i].propertyFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT)) { mt = i; break; } + VkMemoryAllocateInfo mai = { VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO }; + mai.allocationSize = mr.size; mai.memoryTypeIndex = mt; + VkDeviceMemory mem; CHK(vkAllocateMemory(dev, &mai, NULL, &mem)); + CHK(vkBindImageMemory(dev, img, mem, 0)); + + // ---- command buffer ---------------------------------------------------- + VkCommandPoolCreateInfo cpi = { VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO }; + cpi.queueFamilyIndex = submitFam; + VkCommandPool pool; CHK(vkCreateCommandPool(dev, &cpi, NULL, &pool)); + VkCommandBufferAllocateInfo cbai = { VK_STRUCTURE_TYPE_COMMAND_BUFFER_ALLOCATE_INFO }; + cbai.commandPool = pool; cbai.level = VK_COMMAND_BUFFER_LEVEL_PRIMARY; cbai.commandBufferCount = 1; + VkCommandBuffer cb; CHK(vkAllocateCommandBuffers(dev, &cbai, &cb)); + + VkImageAspectFlags aspect = useRgba + ? VK_IMAGE_ASPECT_COLOR_BIT + : (VK_IMAGE_ASPECT_PLANE_0_BIT | VK_IMAGE_ASPECT_PLANE_1_BIT); + // The library passes COLOR on a 2-plane image; mirror that unless asked not to. + if (!getenv("QFOT_PLANE_ASPECT")) aspect = VK_IMAGE_ASPECT_COLOR_BIT; + VkImageSubresourceRange rng = { aspect, 0, 1, 0, 1 }; + + int plain = useRgba || noVideoUsage || profileOnly; + VkImageLayout homeLayout = plain ? VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL + : VK_IMAGE_LAYOUT_VIDEO_ENCODE_SRC_KHR; + VkPipelineStageFlags2 homeStage = plain ? VK_PIPELINE_STAGE_2_TRANSFER_BIT + : VK_PIPELINE_STAGE_2_VIDEO_ENCODE_BIT_KHR; + VkAccessFlags2 homeAccess = plain ? VK_ACCESS_2_TRANSFER_WRITE_BIT + : VK_ACCESS_2_VIDEO_ENCODE_READ_BIT_KHR; + if (homeGeneral) { + // Video-encode resource, but handed over in GENERAL rather than + // VIDEO_ENCODE_SRC_KHR -- the shape the (green) filter-arm release uses. + homeLayout = VK_IMAGE_LAYOUT_GENERAL; + } + + VkCommandBufferBeginInfo bi = { VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO }; + bi.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + CHK(vkBeginCommandBuffer(cb, &bi)); + + #define BARRIER(b) do { VkDependencyInfo di = { VK_STRUCTURE_TYPE_DEPENDENCY_INFO }; \ + di.imageMemoryBarrierCount = 1; di.pImageMemoryBarriers = &(b); \ + vkCmdPipelineBarrier2(cb, &di); } while (0) + + // Control transition: UNDEFINED -> home layout, no ownership transfer. + VkImageMemoryBarrier2 init = { VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2 }; + init.srcStageMask = VK_PIPELINE_STAGE_2_NONE; init.srcAccessMask = 0; + init.dstStageMask = homeStage; init.dstAccessMask = homeAccess; + if (relStage) { init.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + init.dstAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT; } + init.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED; init.newLayout = homeLayout; + init.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + init.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED; + init.image = img; init.subresourceRange = rng; + BARRIER(init); + + if (!strcmp(mode, "acq") || !strcmp(mode, "pair")) { + // Put it in GENERAL first so the acquire has the library's exact pair. + VkImageMemoryBarrier2 g = init; + g.oldLayout = homeLayout; g.newLayout = VK_IMAGE_LAYOUT_GENERAL; + g.srcStageMask = homeStage; g.srcAccessMask = homeAccess; + g.dstStageMask = homeStage; g.dstAccessMask = homeAccess; + BARRIER(g); + // The library's Path-A acquire, verbatim: srcStage COMPUTE_SHADER, + // srcAccess SHADER_WRITE (both IGNORED for an acquire), FOREIGN -> enc. + VkImageMemoryBarrier2 a = { VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2 }; + a.srcStageMask = VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT; + a.srcAccessMask = VK_ACCESS_2_SHADER_WRITE_BIT; + a.dstStageMask = homeStage; a.dstAccessMask = homeAccess; + a.oldLayout = VK_IMAGE_LAYOUT_GENERAL; a.newLayout = homeLayout; + a.srcQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT; + a.dstQueueFamilyIndex = submitFam; + a.image = img; a.subresourceRange = rng; + BARRIER(a); + } + + if (!strcmp(mode, "rel") || !strcmp(mode, "rel-xition") || !strcmp(mode, "pair") || + !strcmp(mode, "rel-external") || !strcmp(mode, "rel-gfx")) { + VkImageMemoryBarrier2 r = { VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2 }; + r.srcStageMask = homeStage; r.srcAccessMask = homeAccess; + if (relStage == 1) { r.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT; + r.srcAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT; } + if (relStage == 2) { r.srcStageMask = VK_PIPELINE_STAGE_2_NONE; r.srcAccessMask = 0; } + r.dstStageMask = VK_PIPELINE_STAGE_2_NONE; r.dstAccessMask = 0; + r.oldLayout = homeLayout; + r.newLayout = (!strcmp(mode, "rel-xition") || !strcmp(mode, "pair")) + ? VK_IMAGE_LAYOUT_GENERAL : homeLayout; + r.srcQueueFamilyIndex = submitFam; + r.dstQueueFamilyIndex = !strcmp(mode, "rel-external") ? VK_QUEUE_FAMILY_EXTERNAL + : !strcmp(mode, "rel-gfx") ? gfxFam + : VK_QUEUE_FAMILY_FOREIGN_EXT; + r.image = img; r.subresourceRange = rng; + if (legacyBarrier) { + VkImageMemoryBarrier r1 = { VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER }; + r1.srcAccessMask = VK_ACCESS_MEMORY_READ_BIT; + r1.dstAccessMask = 0; + r1.oldLayout = r.oldLayout; r1.newLayout = r.newLayout; + r1.srcQueueFamilyIndex = r.srcQueueFamilyIndex; + r1.dstQueueFamilyIndex = r.dstQueueFamilyIndex; + r1.image = img; r1.subresourceRange = rng; + vkCmdPipelineBarrier(cb, + VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, + VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, + 0, 0, NULL, 0, NULL, 1, &r1); + } else { + BARRIER(r); + } + } + + CHK(vkEndCommandBuffer(cb)); + + VkFenceCreateInfo fci = { VK_STRUCTURE_TYPE_FENCE_CREATE_INFO }; + VkFence fence; CHK(vkCreateFence(dev, &fci, NULL, &fence)); + VkSubmitInfo si = { VK_STRUCTURE_TYPE_SUBMIT_INFO }; + si.commandBufferCount = 1; si.pCommandBuffers = &cb; + VkResult sr = vkQueueSubmit(queue, 1, &si, fence); + printf("mode=%s rgba=%d submitFamily=%u vkQueueSubmit -> %d (%s)\n", + mode, useRgba, submitFam, (int)sr, ResName(sr)); + VkResult wr = vkWaitForFences(dev, 1, &fence, VK_TRUE, 5000000000ull); + printf("mode=%s rgba=%d submitFamily=%u vkWaitForFences -> %d (%s)\n", + mode, useRgba, submitFam, (int)wr, ResName(wr)); + + if (wr == VK_ERROR_DEVICE_LOST && pfnFault) { + VkDeviceFaultCountsEXT counts = { VK_STRUCTURE_TYPE_DEVICE_FAULT_COUNTS_EXT }; + if (pfnFault(dev, &counts, NULL) == VK_SUCCESS) { + printf(" deviceFault: addressInfoCount=%u vendorInfoCount=%u vendorBinarySize=%llu\n", + counts.addressInfoCount, counts.vendorInfoCount, + (unsigned long long)counts.vendorBinarySize); + VkDeviceFaultInfoEXT info = { VK_STRUCTURE_TYPE_DEVICE_FAULT_INFO_EXT }; + info.pAddressInfos = counts.addressInfoCount + ? calloc(counts.addressInfoCount, sizeof(VkDeviceFaultAddressInfoEXT)) : NULL; + info.pVendorInfos = counts.vendorInfoCount + ? calloc(counts.vendorInfoCount, sizeof(VkDeviceFaultVendorInfoEXT)) : NULL; + if (pfnFault(dev, &counts, &info) == VK_SUCCESS) { + printf(" faultDescription: %s\n", info.description); + for (uint32_t i = 0; i < counts.vendorInfoCount; i++) + printf(" vendorFault[%u]: %s code=%llu data=%llu\n", i, + info.pVendorInfos[i].description, + (unsigned long long)info.pVendorInfos[i].vendorFaultCode, + (unsigned long long)info.pVendorInfos[i].vendorFaultData); + } + } + } + + printf("VERDICT mode=%s rgba=%d : %s\n", mode, useRgba, + (wr == VK_SUCCESS) ? "SURVIVED" : "LOST"); + return (wr == VK_SUCCESS) ? 0 : 1; +} diff --git a/vk_video_encoder/test/test_av1_quality_gate.py b/vk_video_encoder/test/test_av1_quality_gate.py new file mode 100644 index 00000000..3e18a894 --- /dev/null +++ b/vk_video_encoder/test/test_av1_quality_gate.py @@ -0,0 +1,202 @@ +"""Unit tests for the AV1 quality gate's skip and completeness decisions. + +These are the two places the gate can report something other than what +happened: it can call a failure a skip, and it can call a truncated stream a +pass. Both decisions used to be expressions inside main(), reachable only by +running a real encoder against real content on a GPU host. They are named +functions now, and these tests pin them down without hardware. + +The classifications asserted here mirror the encoder's own contract in +vk_video_encoder/test/vulkan-video-enc/Main.cpp: exit 69 means the device +cannot do this, and every other nonzero status is an ordinary failure. + +Copyright 2025 NVIDIA Corporation. + +Licensed under the Apache License, Version 2.0 (the "License"); +you may not use this file except in compliance with the License. +You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + +Unless required by applicable law or agreed to in writing, software +distributed under the License is distributed on an "AS IS" BASIS, +WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +See the License for the specific language governing permissions and +limitations under the License. +""" + +import importlib.util +import os +import sys +import tempfile +from unittest.mock import patch + +_HERE = os.path.dirname(os.path.abspath(__file__)) +_SCRIPT = os.path.join(_HERE, "av1_encoder_quality_test.py") + +_spec = importlib.util.spec_from_file_location("av1_quality_gate", _SCRIPT) +gate = importlib.util.module_from_spec(_spec) +_spec.loader.exec_module(gate) + + +def row(exit_code=0, encode_ok=True, decode_ok=True, frames_decoded=None, + codec="av1", gop=8): + """One result row, shaped like the ones main() accumulates.""" + return gate.EncodeResult( + codec=codec, gop=gop, file_size=1024, + encode_ok=encode_ok, decode_ok=decode_ok, + exit_code=exit_code, frames_decoded=frames_decoded) + + +def failed_row(exit_code, **kw): + """A row whose encode failed with the given process exit status.""" + return row(exit_code=exit_code, encode_ok=False, **kw) + + +# -------------------------------------------------------------------------- +# The skip decision. Only the encoder's unsupported status may skip. +# -------------------------------------------------------------------------- + +def test_all_rows_unsupported_is_a_skip(): + rows = [failed_row(69), failed_row(69, gop=16)] + assert gate.is_unsupported_run(rows) is True + + +def test_ordinary_failures_are_not_a_skip(): + # The encoder maps out-of-memory (-1), device-lost (-4), + # initialization-failed (-3) and any unknown VkResult to EXIT_FAILURE. It + # prints the same "Error creating the encoder instance" line for each, + # which is why the text cannot be the signal. + for status in (1, 2, 255): + rows = [failed_row(status), failed_row(status, gop=16)] + assert gate.is_unsupported_run(rows) is False, status + + +def test_mixed_rows_are_not_a_skip(): + # One unsupported row and one real failure is a failing run, not a skip: + # the failure still has to be reported. + rows = [failed_row(69), failed_row(1, gop=16)] + assert gate.is_unsupported_run(rows) is False + + +def test_a_judgeable_row_defeats_the_skip(): + # A host whose device merely encodes badly reaches the quality comparison + # with something to judge, so the run cannot be skipped away. + rows = [failed_row(69), row(frames_decoded=30, gop=16)] + assert gate.is_unsupported_run(rows) is False + + +def test_no_results_is_not_a_skip(): + assert gate.is_unsupported_run([]) is False + + +def test_a_decode_failure_is_not_a_skip(): + # encode_ok stays True, so this row was produced by a working device. + rows = [row(exit_code=0, decode_ok=False)] + assert gate.is_unsupported_run(rows) is False + + +# -------------------------------------------------------------------------- +# The completeness decision. A readable prefix is not a pass. +# -------------------------------------------------------------------------- + +def test_full_frame_count_is_complete(): + assert gate.short_rows([row(frames_decoded=30)], 30) == [] + + +def test_more_frames_than_requested_is_complete(): + assert gate.short_rows([row(frames_decoded=31)], 30) == [] + + +def test_partial_stream_is_short(): + rows = [row(frames_decoded=3)] + assert gate.short_rows(rows, 30) == rows + + +def test_unknown_frame_count_is_short(): + # Absence of a count is not evidence of completeness. + rows = [row(frames_decoded=None)] + assert gate.short_rows(rows, 30) == rows + + +def test_already_failed_rows_are_not_counted_twice(): + # These are reported by the encode/decode checks; counting them here + # would print a second, less informative failure for the same row. + rows = [failed_row(1), row(decode_ok=False, frames_decoded=None)] + assert gate.short_rows(rows, 30) == [] + + +# -------------------------------------------------------------------------- +# encode(): the exit status reaches the caller, and a zero exit that printed +# a fatal diagnostic is still a failure. +# -------------------------------------------------------------------------- + +def run_encode(rc, stderr, out_size=1024): + """Drive encode() with a canned process result and a real output file.""" + with tempfile.TemporaryDirectory() as d: + out_path = os.path.join(d, "out.ivf") + with open(out_path, "wb") as f: + f.write(b"\0" * out_size) + with patch.object(gate, "run_cmd", return_value=(rc, "", stderr)): + return gate.encode(os.path.join(d, "in.yuv"), out_path, "av1", + 176, 144, 30, 8, 30, + encoder_bin="/nonexistent/encoder") + + +def test_encode_reports_the_unsupported_status(): + ok, err, status = run_encode( + 69, "Error creating the encoder instance: -11\n") + assert ok is False + assert status == 69 + + +def test_encode_reports_an_ordinary_failure_status(): + # Same stderr line, different status: this one must not become a skip. + ok, err, status = run_encode( + 1, "Error creating the encoder instance: -4\n") + assert ok is False + assert status == 1 + + +def test_encode_succeeds_on_a_clean_zero_exit(): + ok, err, status = run_encode(0, "") + assert ok is True + assert err == "" + assert status == 0 + + +def test_zero_exit_with_a_frame_error_is_a_failure(): + # The C++ entry point propagates this into its status; if that regresses, + # the run must not become a quality pass over a truncated stream. + ok, err, status = run_encode( + 0, "Error encoding frame: 7, error: -4\n", out_size=64) + assert ok is False + assert "Error encoding frame:" in err + + +def test_zero_exit_with_a_bitstream_error_is_a_failure(): + ok, err, status = run_encode( + 0, "Error obtaining the encoded bitstream file: -2\n") + assert ok is False + assert "Error obtaining the encoded bitstream file:" in err + + +def test_zero_exit_with_an_empty_output_is_a_failure(): + ok, err, status = run_encode(0, "", out_size=0) + assert ok is False + assert "missing or empty" in err + + +if __name__ == "__main__": + failures = 0 + for name, fn in sorted(globals().items()): + if not name.startswith("test_") or not callable(fn): + continue + try: + fn() + print(f" ok {name}") + except AssertionError as exc: + failures += 1 + print(f" FAIL {name}: {exc}") + print(f"\n{failures} failure(s)") + sys.exit(1 if failures else 0) diff --git a/vk_video_encoder/test/vulkan-video-enc/Main.cpp b/vk_video_encoder/test/vulkan-video-enc/Main.cpp index 80f517cf..ea429ab0 100644 --- a/vk_video_encoder/test/vulkan-video-enc/Main.cpp +++ b/vk_video_encoder/test/vulkan-video-enc/Main.cpp @@ -14,8 +14,9 @@ * limitations under the License. */ +#include #include -#include "vulkan_video_encoder.h" +#include "vulkan_video_encoder_argv.h" #include "VkVSCommon.h" int main(int argc, const char** argv) @@ -27,26 +28,60 @@ int main(int argc, const char** argv) if (result != VK_SUCCESS) { std::cerr << "Error creating the encoder instance: " << result << std::endl; + // 69 is reserved for a device that cannot do this, and for nothing + // else. Every other creation failure -- out of memory, initialization, + // device loss, unknown -- is an ordinary failure, because a harness + // that reads 69 as "skip" would otherwise skip a real defect. return IsVideoUnsupportedResult(result) ? VVS_EXIT_UNSUPPORTED : EXIT_FAILURE; } - int64_t numFrames = vulkanVideoEncoder->GetNumberOfFrames(); + // A VK_SUCCESS that hands back nothing usable is still a failure, and one + // that would be invisible below: the frame loop simply would not run and + // the process would exit zero having encoded nothing. + if (!vulkanVideoEncoder) { + std::cerr << "Error: encoder creation reported success but produced no " + "encoder" << std::endl; + return EXIT_FAILURE; + } + + const int64_t numFrames = vulkanVideoEncoder->GetNumberOfFrames(); + if (numFrames < 0) { + std::cerr << "Error: invalid frame count " << numFrames << std::endl; + return EXIT_FAILURE; + } std::cout << "Number of frames to encode: " << numFrames << std::endl; + // THE FIRST ERROR IS KEPT. Nothing below may overwrite it -- not a later + // frame that happens to succeed, and not a completion that succeeds. + VkResult firstError = VK_SUCCESS; + for (int64_t frameNum = 0; frameNum < numFrames; frameNum++) { int64_t frameNumEncoded = -1; result = vulkanVideoEncoder->EncodeNextFrame(frameNumEncoded); if (result != VK_SUCCESS) { std::cerr << "Error encoding frame: " << frameNum << ", error: " << result << std::endl; + firstError = result; + // Stop asking. Continuing past a failed frame produces a stream + // with a hole in it and a longer log that says the same thing. + break; } } + // Called once for an initialized session even after a frame failed: the + // work already accepted has to be completed and the output closed. result = vulkanVideoEncoder->GetBitstream(); if (result != VK_SUCCESS) { std::cerr << "Error obtaining the encoded bitstream file: " << result << std::endl; + if (firstError == VK_SUCCESS) { + firstError = result; + } } std::cout << "Exit encoder test" << std::endl; + // EXPLICIT. Falling off the end of main returns zero, which is how a run + // that printed a failure for every frame was still read as a pass. A + // post-initialization failure is EXIT_FAILURE even when its VkResult looks + // like a capability answer -- the device was already known to be capable, + // or creation would have returned 69. + return (firstError == VK_SUCCESS) ? EXIT_SUCCESS : EXIT_FAILURE; } - -