diff --git a/app/src/main/cpp/CMakeLists.txt b/app/src/main/cpp/CMakeLists.txt index bc3d2085f..25b4d73c2 100644 --- a/app/src/main/cpp/CMakeLists.txt +++ b/app/src/main/cpp/CMakeLists.txt @@ -103,23 +103,6 @@ set(SHADER_LIST "effect_colorblind:frag:effect_colorblind_frag" "effect_pixelate:frag:effect_pixelate_frag" "sgsr1:frag:sgsr1_frag" - "dis_luma_r16:comp:dis_luma_r16_comp" - "dis_luma_r32:comp:dis_luma_r32_comp" - "dis_gradient:comp:dis_gradient_comp" - "dis_inverse_search:comp:dis_inverse_search_comp" - "dis_propagate:comp:dis_propagate_comp" - "dis_densify:comp:dis_densify_comp" - "dis_interpolate:comp:dis_interpolate_comp" - "dis_hist:comp:dis_hist_comp" - "dis_side:comp:dis_side_comp" - "dis_temporal:comp:dis_temporal_comp" - "dis_vr_prep:comp:dis_vr_prep_comp" - "dis_vr_d1:comp:dis_vr_d1_comp" - "dis_vr_d2:comp:dis_vr_d2_comp" - "dis_vr_w:comp:dis_vr_w_comp" - "dis_vr_coef:comp:dis_vr_coef_comp" - "dis_vr_sor:comp:dis_vr_sor_comp" - "dis_vr_add:comp:dis_vr_add_comp" ) set(SHADER_HEADERS "") @@ -149,6 +132,9 @@ endforeach() add_custom_target(winlator_shaders DEPENDS ${SHADER_HEADERS}) +# DIS frame generator (optical flow + interpolation), linked by libwinlator and libwnwayland. +add_subdirectory(dis) + # ---------------------------------------------------------------------------- # Winlator native library (X-server, AHB, Vulkan compositor, helpers) # ---------------------------------------------------------------------------- @@ -183,7 +169,6 @@ add_library(winlator SHARED winlator/vk/lsfg/lsfg_jni.c winlator/vk/framegen/fg_present.c winlator/vk/framegen/fg_jni.c - winlator/vk/dis/vkr_dis.c ) add_dependencies(winlator winlator_shaders) @@ -201,6 +186,7 @@ target_include_directories(winlator PRIVATE target_compile_features(winlator PRIVATE cxx_std_17) target_link_libraries(winlator + wndis log android mediandk diff --git a/app/src/main/cpp/dis/CMakeLists.txt b/app/src/main/cpp/dis/CMakeLists.txt new file mode 100644 index 000000000..d13b69cd2 --- /dev/null +++ b/app/src/main/cpp/dis/CMakeLists.txt @@ -0,0 +1,79 @@ +# DIS frame generator: Dense Inverse Search optical flow and the frame interpolator built on it, +# as one static library. libwinlator (X11 compositor, standalone presenter) and libwnwayland +# (Wayland compositor) both link it; each keeps its own VkDispatch, which the objects here +# resolve against at link time. +# +# Layout: +# include/vkr_dis.h public API +# include/shaders/*.comp GLSL compute shaders, compiled to SPIR-V headers at build time +# src/ implementation +# +# Uses GLSLC and BIN2C_SCRIPT from the parent CMakeLists. + +set(DIS_SHADER_SRC_DIR "${CMAKE_CURRENT_SOURCE_DIR}/include/shaders") +set(DIS_SHADER_OUT_DIR "${CMAKE_CURRENT_BINARY_DIR}/shaders") +file(MAKE_DIRECTORY "${DIS_SHADER_OUT_DIR}") + +set(DIS_SHADERS + dis_luma_r16 + dis_luma_r32 + dis_gradient + dis_inverse_search + dis_propagate + dis_densify + dis_interpolate + dis_hist + dis_side + dis_flow_pack + dis_vr_prep + dis_vr_d1 + dis_vr_d2 + dis_vr_w + dis_vr_coef + dis_vr_sor + dis_vr_add + dis_me_luma +) + +set(DIS_SHADER_HEADERS "") +foreach(base ${DIS_SHADERS}) + set(var "${base}_comp") + set(input "${DIS_SHADER_SRC_DIR}/${base}.comp") + set(spv "${DIS_SHADER_OUT_DIR}/${var}.spv") + set(hdr "${DIS_SHADER_OUT_DIR}/${var}.spv.h") + add_custom_command( + OUTPUT "${hdr}" + COMMAND "${GLSLC}" --target-env=vulkan1.1 -O "${input}" -o "${spv}" + COMMAND "${CMAKE_COMMAND}" + -DINPUT_FILE=${spv} + -DOUTPUT_FILE=${hdr} + -DVAR_NAME=${var} + -P "${BIN2C_SCRIPT}" + DEPENDS "${input}" "${BIN2C_SCRIPT}" + COMMENT "Compiling DIS shader ${base}.comp -> ${var}.spv.h" + VERBATIM + ) + list(APPEND DIS_SHADER_HEADERS "${hdr}") +endforeach() + +add_custom_target(wndis_shaders DEPENDS ${DIS_SHADER_HEADERS}) + +add_library(wndis STATIC + src/vkr_dis.c + src/dis_qcom_me.c +) +add_dependencies(wndis wndis_shaders) + +set_target_properties(wndis PROPERTIES POSITION_INDEPENDENT_CODE ON) +target_compile_options(wndis PRIVATE -Wall -Wextra -fvisibility=hidden) +target_compile_definitions(wndis PUBLIC VK_USE_PLATFORM_ANDROID_KHR) +target_include_directories(wndis + PUBLIC + ${CMAKE_CURRENT_SOURCE_DIR}/include + # vk_dispatch.h: the public header takes its Vulkan types from there. + ${CMAKE_CURRENT_SOURCE_DIR}/../winlator/vk + PRIVATE + ${CMAKE_CURRENT_BINARY_DIR} +) +# EGL/GLES are opened with dlopen by dis_qcom_me.c, so nothing links them here. +target_link_libraries(wndis PRIVATE log dl) diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_densify.comp b/app/src/main/cpp/dis/include/shaders/dis_densify.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_densify.comp rename to app/src/main/cpp/dis/include/shaders/dis_densify.comp diff --git a/app/src/main/cpp/dis/include/shaders/dis_flow_pack.comp b/app/src/main/cpp/dis/include/shaders/dis_flow_pack.comp new file mode 100644 index 000000000..3922790e3 --- /dev/null +++ b/app/src/main/cpp/dis/include/shaders/dis_flow_pack.comp @@ -0,0 +1,39 @@ +// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) +// SPDX-License-Identifier: GPL-3.0-or-later +// +// DIS frame generation: a Vulkan compute realisation of Dense Inverse Search +// optical flow. The algorithm and its reference implementation come from +// OpenCV's DISOpticalFlow, which adopted Till Kroeger's original OF_DIS. +// See CREDITS.md for the full attribution. + +#version 450 + +precision highp float; +precision highp int; + +// Hands the finished level-0 field to the interpolate pass as RG16F, once per source pair. +// +// The interpolate pass reads the field three times per output pixel (the lookup at the +// pixel plus the fixed-point steps towards its source) and it does so for every generated +// frame, at the full output resolution. RG32F is not filterable on every mobile driver, +// and without filtering each of those reads is four fetches and a manual lerp. RG16F is +// filterable everywhere and half the bandwidth; in uv units its precision is a few +// hundredths of a pixel at ordinary speeds and stays under a pixel even at the magnitude +// clamp in dis_vr_add, which is far below what a quarter-frame jump can show. + +layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in; + +layout(set = 0, binding = 0) uniform sampler2D flowIn; +layout(set = 0, binding = 5, rg16f) uniform image2D flowOut; + +void main() { + ivec2 p = ivec2(gl_GlobalInvocationID.xy); + ivec2 sz = imageSize(flowOut); + if (p.x >= sz.x || p.y >= sz.y) return; + + vec2 f = texelFetch(flowIn, p, 0).xy; + // A zero field degrades to repeating the real frame, which is the right answer + // wherever the flow has nothing finite to say. + if (any(isnan(f)) || any(isinf(f))) f = vec2(0.0); + imageStore(flowOut, p, vec4(f, 0.0, 0.0)); +} diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_gradient.comp b/app/src/main/cpp/dis/include/shaders/dis_gradient.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_gradient.comp rename to app/src/main/cpp/dis/include/shaders/dis_gradient.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_hist.comp b/app/src/main/cpp/dis/include/shaders/dis_hist.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_hist.comp rename to app/src/main/cpp/dis/include/shaders/dis_hist.comp diff --git a/app/src/main/cpp/dis/include/shaders/dis_interpolate.comp b/app/src/main/cpp/dis/include/shaders/dis_interpolate.comp new file mode 100644 index 000000000..f08a3e725 --- /dev/null +++ b/app/src/main/cpp/dis/include/shaders/dis_interpolate.comp @@ -0,0 +1,241 @@ +// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) +// SPDX-License-Identifier: GPL-3.0-or-later +// +// DIS frame generation: a Vulkan compute realisation of Dense Inverse Search +// optical flow. The algorithm and its reference implementation come from +// OpenCV's DISOpticalFlow, which adopted Till Kroeger's original OF_DIS. +// See CREDITS.md for the full attribution. + +#version 450 + +precision highp float; +precision highp int; + +layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in; + +layout(set = 0, binding = 0) uniform sampler2D prevColor; +layout(set = 0, binding = 1) uniform sampler2D nextColor; +layout(set = 0, binding = 2) uniform sampler2D flowTex; +layout(set = 0, binding = 3) uniform sampler2D sideTex; +layout(set = 0, binding = 4) uniform sampler2D histTex; +layout(set = 0, binding = 5, rgba8) uniform image2D outImage; + +layout(push_constant) uniform PC { + float t; + int debugMode; +} pc; + +layout(constant_id = 0) const int manualFlowFilter = 0; + +// Fixed-point steps that move the flow lookup from the output pixel to the point of the +// previous frame that actually lands on it. 0 samples the field at the output pixel. +#ifndef DIS_SOURCE_STEPS +#define DIS_SOURCE_STEPS 2 +#endif + +// Keep a static overlay in place where the unwarped pair explains the pixel better than +// the warped one. 0 turns it off. +#ifndef DIS_STATIC_SELECT +#define DIS_STATIC_SELECT 1 +#endif + +// Use the side map (sign of the flow divergence) for the frame an occlusion takes, +// instead of the nearer one in time. +#ifndef DIS_OCCL_SIDED +#define DIS_OCCL_SIDED 1 +#endif + +// Catmull-Rom (5 bilinear taps) for the warped fetches instead of one bilinear tap. +#ifndef DIS_WARP_CATMULL +#define DIS_WARP_CATMULL 1 +#endif + +// Take the stabilised overlay confidence from dis_hist on top of the per-pixel test. +#ifndef DIS_USE_HIST +#define DIS_USE_HIST 1 +#endif + +#ifndef SIDE_LO +#define SIDE_LO 4.0 +#endif +#ifndef SIDE_HI +#define SIDE_HI 12.0 +#endif + +#ifndef OCCL_LO +#define OCCL_LO 0.20 +#endif +#ifndef OCCL_HI +#define OCCL_HI 0.60 +#endif + +vec2 sampleFlow(vec2 uv) { + if (manualFlowFilter == 0) return textureLod(flowTex, uv, 0.0).xy; + + vec2 sz = vec2(textureSize(flowTex, 0)); + vec2 p = uv * sz - 0.5; + vec2 frac = fract(p); + ivec2 i0 = ivec2(floor(p)); + ivec2 mx = ivec2(sz) - 1; + + vec2 a = texelFetch(flowTex, clamp(i0, ivec2(0), mx), 0).xy; + vec2 b = texelFetch(flowTex, clamp(i0 + ivec2(1, 0), ivec2(0), mx), 0).xy; + vec2 c = texelFetch(flowTex, clamp(i0 + ivec2(0, 1), ivec2(0), mx), 0).xy; + vec2 e = texelFetch(flowTex, clamp(i0 + ivec2(1, 1), ivec2(0), mx), 0).xy; + return mix(mix(a, b, frac.x), mix(c, e, frac.x), frac.y); +} + +float maxChannel(vec3 v) { + return max(max(v.x, v.y), v.z); +} + +vec3 warpSample(sampler2D tex, vec2 uv, vec2 texSize) { +#if DIS_WARP_CATMULL == 0 + return textureLod(tex, clamp(uv, vec2(0.0), vec2(1.0)), 0.0).xyz; +#else + vec2 samplePos = uv * texSize; + vec2 texPos1 = floor(samplePos - 0.5) + 0.5; + vec2 fr = samplePos - texPos1; + vec2 w0 = fr * (-0.5 + fr * (1.0 - 0.5 * fr)); + vec2 w1 = 1.0 + fr * fr * (-2.5 + 1.5 * fr); + vec2 w2 = fr * (0.5 + fr * (2.0 - 1.5 * fr)); + vec2 w3 = fr * fr * (-0.5 + 0.5 * fr); + vec2 w12 = w1 + w2; + vec2 off12 = w2 / w12; + vec2 inv = 1.0 / texSize; + vec2 p0 = clamp((texPos1 - 1.0) * inv, vec2(0.0), vec2(1.0)); + vec2 p3 = clamp((texPos1 + 2.0) * inv, vec2(0.0), vec2(1.0)); + vec2 p12 = clamp((texPos1 + off12) * inv, vec2(0.0), vec2(1.0)); + vec3 r = vec3(0.0); + r += textureLod(tex, vec2(p12.x, p0.y), 0.0).xyz * (w12.x * w0.y); + r += textureLod(tex, vec2(p0.x, p12.y), 0.0).xyz * (w0.x * w12.y); + r += textureLod(tex, vec2(p12.x, p12.y), 0.0).xyz * (w12.x * w12.y); + r += textureLod(tex, vec2(p3.x, p12.y), 0.0).xyz * (w3.x * w12.y); + r += textureLod(tex, vec2(p12.x, p3.y), 0.0).xyz * (w12.x * w3.y); + r /= w12.x + w12.y * (1.0 - w12.x); + return clamp(r, vec3(0.0), vec3(1.0)); +#endif +} + +vec3 hsv2rgb(vec3 c) { + vec4 K = vec4(1.0, 2.0 / 3.0, 1.0 / 3.0, 3.0); + vec3 p = abs(fract(c.xxx + K.xyz) * 6.0 - K.www); + return c.z * mix(K.xxx, clamp(p - K.xxx, 0.0, 1.0), c.y); +} + +void main() { + ivec2 pix = ivec2(gl_GlobalInvocationID.xy); + ivec2 size = imageSize(outImage); + if (pix.x >= size.x || pix.y >= size.y) return; + + vec2 uv = (vec2(pix) + 0.5) / vec2(size); + vec2 texel = 1.0 / vec2(size); + + const float GUARD_PX = 20.0; + vec2 guard = GUARD_PX * texel; + vec2 dEdge = min(uv, 1.0 - uv); + vec2 ramp = clamp((dEdge - guard) / max(guard, texel), 0.0, 1.0); + vec2 eased = ramp * ramp * (3.0 - 2.0 * ramp); + float edgeMix = min(eased.x, eased.y); + + vec2 f = sampleFlow(uv); + if (edgeMix < 1.0) { + vec2 fInner = sampleFlow(clamp(uv, guard, 1.0 - guard)); + f = mix(fInner, f, edgeMix); + } + if (any(isnan(f)) || any(isinf(f))) f = vec2(0.0); + + if (pc.debugMode != 0) { + float m = length(f) * float(size.x) / 16.0; + float hue = atan(f.y, f.x) / 6.2831853 + 0.5; + vec3 fc = hsv2rgb(vec3(hue, clamp(m, 0.0, 1.0), min(1.0, 0.15 + m))); + imageStore(outImage, pix, vec4(fc, 1.0)); + return; + } + + // The field is defined on the previous frame's grid: f(x) carries the content at x in + // the previous frame to x + f(x) in the next. What reaches the output pixel at time t + // is the content whose x + t*f(x) equals uv, so the vector has to be read at x, not at + // uv. Reading it at uv is the halo: near a moving edge uv sits on the other layer in + // the previous frame. A couple of fixed-point steps find x wherever the field is + // smooth enough to have one. + vec2 fa = f; +#if DIS_SOURCE_STEPS > 0 + for (int i = 0; i < DIS_SOURCE_STEPS; i++) { + vec2 fn = sampleFlow(clamp(uv - pc.t * fa, vec2(0.0), vec2(1.0))); + if (any(isnan(fn)) || any(isinf(fn))) break; + fa = fn; + } + fa = mix(f, fa, edgeMix); +#endif + + // The real pair where it sits: the early-out below and the static test at the end both + // need it. + vec3 s0 = textureLod(prevColor, uv, 0.0).xyz; + vec3 s1 = textureLod(nextColor, uv, 0.0).xyz; + float rStatic = dot(abs(s0 - s1), vec3(1.0)); + + // Under a quarter of a pixel no warp can change an output pixel visibly, and where the + // pair also agrees nothing crossed the pixel either, so the ten warped fetches below + // would buy nothing. The agreement test matters: a still background just ahead of a + // moving object has a zero field too, and there the occlusion handling below is what + // brings the object in. Static scenery and HUD-heavy screens are mostly this case. + vec2 fpx = max(abs(f), abs(fa)) * vec2(size); + if (max(fpx.x, fpx.y) <= 0.25 && rStatic < 0.03) { + imageStore(outImage, pix, vec4(mix(s0, s1, pc.t), 1.0)); + return; + } + + vec2 uv0 = uv - pc.t * fa; + vec2 uv1 = uv + (1.0 - pc.t) * fa; + + vec3 c0 = warpSample(prevColor, uv0, vec2(size)); + vec3 c1 = warpSample(nextColor, uv1, vec2(size)); + + vec3 single = pc.t < 0.5 ? c0 : c1; +#if DIS_OCCL_SIDED + // Where the field opens up (divergence > 0) the content is new and only the next frame + // has it; where it closes, only the previous. The sign is only trusted across a real + // motion boundary - a few pixels of divergence is estimation noise, and there the + // nearer frame in time is the better guess. + float div = textureLod(sideTex, uv, 0.0).r; + float sure = smoothstep(SIDE_LO, SIDE_HI, abs(div)); + single = mix(single, div > 0.0 ? c1 : c0, sure); +#endif + + const float FEATHER_PX = 8.0; + vec2 feather = FEATHER_PX / vec2(size); + vec2 e0 = max(max(-uv0, uv0 - vec2(1.0)), vec2(0.0)) / feather; + vec2 e1 = max(max(-uv1, uv1 - vec2(1.0)), vec2(0.0)) / feather; + float out0 = clamp(max(e0.x, e0.y), 0.0, 1.0); + float out1 = clamp(max(e1.x, e1.y), 0.0, 1.0); + + float w0 = (1.0 - pc.t) * (1.0 - out0); + float w1 = pc.t * (1.0 - out1); + float wsum = w0 + w1; + + vec3 result = wsum > 1e-4 + ? (c0 * w0 + c1 * w1) / wsum + : (out0 <= out1 ? c0 : c1); + + float rMotion = dot(abs(c0 - c1), vec3(1.0)); + float occl = smoothstep(OCCL_LO, OCCL_HI, rMotion); + result = mix(result, single, occl); + +#if DIS_STATIC_SELECT + // Two hypotheses for this pixel, scored by how well each reconciles the real pair: + // "it moved with the field" (c0 vs c1) and "it stood still" (the pair where it sits). + // A HUD over a moving scene scores near zero standing still and badly moving; the + // scene itself scores the other way round. Only a clear win for standing still pins + // the pixel, so moving content keeps its motion even where it is flat. + float keep = (1.0 - smoothstep(0.03, 0.12, rStatic)) * + smoothstep(0.02, 0.10, rMotion - rStatic); +#if DIS_USE_HIST + float histStatic = textureLod(histTex, uv, 0.0).r; + keep = max(keep, histStatic * smoothstep(0.02, 0.10, rMotion - rStatic)); +#endif + result = mix(result, mix(s0, s1, pc.t), keep); +#endif + + imageStore(outImage, pix, vec4(result, 1.0)); +} diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_inverse_search.comp b/app/src/main/cpp/dis/include/shaders/dis_inverse_search.comp similarity index 82% rename from app/src/main/cpp/winlator/vk/shaders/dis_inverse_search.comp rename to app/src/main/cpp/dis/include/shaders/dis_inverse_search.comp index 84d70044e..46f9f9d61 100644 --- a/app/src/main/cpp/winlator/vk/shaders/dis_inverse_search.comp +++ b/app/src/main/cpp/dis/include/shaders/dis_inverse_search.comp @@ -61,14 +61,20 @@ layout(set = 0, binding = 0) uniform sampler2D lastLumaMap; layout(set = 0, binding = 1) uniform sampler2D nextLumaMap; layout(set = 0, binding = 2) uniform sampler2D lastGradientMap; layout(set = 0, binding = 3) uniform sampler2D flowMap; -layout(set = 0, binding = 4) uniform sampler2D lastFlowMap; +// Motion hint from a hardware estimator (GL_QCOM_motion_estimation), one vector per block, in +// normalised uv units; read only on the level named by hintLevel. A component beyond HINT_INVALID +// marks a block the estimator had nothing for. +layout(set = 0, binding = 4) uniform sampler2D hintFlowMap; layout(set = 0, binding = 5, rgba32f) uniform image2D sparseFlowMap; layout(push_constant) uniform PC { int level; int coarseLevel; + int hintLevel; } pc; +#define HINT_INVALID 1.0e6 + float uluminance(vec3 c) { return (0.299 * c.x + 0.587 * c.y + 0.114 * c.z) * 255.0; } @@ -154,11 +160,44 @@ void main() { flow = cf.xy * vec2(denseSize); if (any(isnan(flow)) || any(isinf(flow))) flow = vec2(0.0); } - vec2 initialFlow = flow; - vec2 invImageSize = 1.0 / vec2(denseSize); const float N = 64.0; + // A second starting point from the hardware estimator. Both candidates are scored on the + // patch with the same mean-normalised SSD the search minimises, and the search starts from + // the better one. The coarse level still covers what the estimator cannot: its search range + // is a few dozen pixels, and past that its vector is garbage that simply loses here. + if (pc.level == pc.hintLevel) { + vec2 hintSize = vec2(textureSize(hintFlowMap, 0)); + ivec2 hp = clamp(ivec2((vec2(pix) + 4.0) * hintSize / vec2(denseSize)), + ivec2(0), ivec2(hintSize) - 1); + vec2 hint = texelFetch(hintFlowMap, hp, 0).xy; + if (all(lessThan(abs(hint), vec2(HINT_INVALID * 0.5)))) { + hint *= vec2(denseSize); + vec2 cand[2] = vec2[2](flow, hint); + float best = 1e30; + for (int c = 0; c < 2; c++) { + vec2 o = clamp(vec2(pix) + cand[c], vec2(0.0), vec2(denseSize) - patchSize); + float s = 0.0; + float s2 = 0.0; + for (int i = 0; i < 8; i++) { + for (int j = 0; j < 8; j++) { + vec2 tc = (o + vec2(i, j) + 0.5) * invImageSize; + float diff = textureLod(nextLumaMap, tc, 0.0).x - lastImageData[i * 8 + j]; + s += diff; + s2 += diff * diff; + } + } + float ssd = s2 - s * s / N; + if (ssd < best) { + best = ssd; + flow = cand[c]; + } + } + } + } + vec2 initialFlow = flow; + const float N_INV = 1.0 / N; float prevSSD = 1e10; for (int iter = 0; iter < DIS_INVERSE_ITERS; iter++) { diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_luma.comp b/app/src/main/cpp/dis/include/shaders/dis_luma.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_luma.comp rename to app/src/main/cpp/dis/include/shaders/dis_luma.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_luma_r16.comp b/app/src/main/cpp/dis/include/shaders/dis_luma_r16.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_luma_r16.comp rename to app/src/main/cpp/dis/include/shaders/dis_luma_r16.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_luma_r32.comp b/app/src/main/cpp/dis/include/shaders/dis_luma_r32.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_luma_r32.comp rename to app/src/main/cpp/dis/include/shaders/dis_luma_r32.comp diff --git a/app/src/main/cpp/dis/include/shaders/dis_me_luma.comp b/app/src/main/cpp/dis/include/shaders/dis_me_luma.comp new file mode 100644 index 000000000..72d0dd1be --- /dev/null +++ b/app/src/main/cpp/dis/include/shaders/dis_me_luma.comp @@ -0,0 +1,53 @@ +// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) +// SPDX-License-Identifier: GPL-3.0-or-later +// +// DIS frame generation: a Vulkan compute realisation of Dense Inverse Search +// optical flow. The algorithm and its reference implementation come from +// OpenCV's DISOpticalFlow, which adopted Till Kroeger's original OF_DIS. +// See CREDITS.md for the full attribution. + +#version 450 + +precision highp float; +precision highp int; + +// Luminance of the newest real frame at the hardware motion estimator's input size, packed four +// pixels to a word into a host-visible buffer: GL_QCOM_motion_estimation takes an R8 texture, and +// the frame reaches GL as a plain upload. Each output pixel box-filters its footprint with four +// bilinear taps, which is exact for the usual 2:1 reduction and keeps larger ones from aliasing. + +layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in; + +layout(set = 0, binding = 0) uniform sampler2D colorTex; +layout(set = 0, binding = 1, std430) writeonly buffer LumaOut { + uint words[]; +}; + +layout(push_constant) uniform PC { + int width; // multiple of 4 + int height; +} pc; + +float lumaAt(vec2 uv, vec2 q) { + vec3 c = textureLod(colorTex, uv + vec2(-q.x, -q.y), 0.0).rgb + + textureLod(colorTex, uv + vec2( q.x, -q.y), 0.0).rgb + + textureLod(colorTex, uv + vec2(-q.x, q.y), 0.0).rgb + + textureLod(colorTex, uv + vec2( q.x, q.y), 0.0).rgb; + return dot(c * 0.25, vec3(0.299, 0.587, 0.114)); +} + +void main() { + ivec2 g = ivec2(gl_GlobalInvocationID.xy); + int wordsPerRow = pc.width / 4; + if (g.x >= wordsPerRow || g.y >= pc.height) return; + + vec2 texel = 1.0 / vec2(pc.width, pc.height); + vec2 q = 0.25 * texel; + uint packed = 0u; + for (int i = 0; i < 4; i++) { + vec2 uv = (vec2(g.x * 4 + i, g.y) + 0.5) * texel; + uint v = uint(clamp(lumaAt(uv, q), 0.0, 1.0) * 255.0 + 0.5); + packed |= v << (8u * uint(i)); + } + words[g.y * wordsPerRow + g.x] = packed; +} diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_propagate.comp b/app/src/main/cpp/dis/include/shaders/dis_propagate.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_propagate.comp rename to app/src/main/cpp/dis/include/shaders/dis_propagate.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_side.comp b/app/src/main/cpp/dis/include/shaders/dis_side.comp similarity index 82% rename from app/src/main/cpp/winlator/vk/shaders/dis_side.comp rename to app/src/main/cpp/dis/include/shaders/dis_side.comp index 4500e6031..455447ebd 100644 --- a/app/src/main/cpp/winlator/vk/shaders/dis_side.comp +++ b/app/src/main/cpp/dis/include/shaders/dis_side.comp @@ -11,7 +11,7 @@ precision highp float; precision highp int; -// One bit per level-0 flow texel: which real frame a true occlusion should take. +// Per level-0 flow texel: which real frame a true occlusion should take, and how sure. // The sign of the flow divergence over a wide window decides it - a diverging // field is area opening up, which only the next frame has, a converging one only // the previous. Computed once per pair at the flow resolution, it replaces four @@ -38,5 +38,8 @@ void main() { vec2 fU = textureLod(flowMap, clamp(uv - vec2(0.0, fo.y), vec2(0.0), vec2(1.0)), 0.0).xy; vec2 fD = textureLod(flowMap, clamp(uv + vec2(0.0, fo.y), vec2(0.0), vec2(1.0)), 0.0).xy; float divergence = (fR.x - fL.x) * uSize.x + (fD.y - fU.y) * uSize.y; - imageStore(sideOut, p, vec4(divergence > 0.0 ? 1.0 : 0.0, 0.0, 0.0, 0.0)); + // Signed, in flow-resolution pixels across the window: the interpolate pass needs the + // magnitude too, to tell a real occlusion boundary from estimation noise. + if (isnan(divergence) || isinf(divergence)) divergence = 0.0; + imageStore(sideOut, p, vec4(divergence, 0.0, 0.0, 0.0)); } diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_add.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_add.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_add.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_add.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_coef.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_coef.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_coef.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_coef.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_d1.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_d1.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_d1.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_d1.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_d2.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_d2.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_d2.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_d2.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_prep.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_prep.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_prep.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_prep.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_sor.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_sor.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_sor.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_sor.comp diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_vr_w.comp b/app/src/main/cpp/dis/include/shaders/dis_vr_w.comp similarity index 100% rename from app/src/main/cpp/winlator/vk/shaders/dis_vr_w.comp rename to app/src/main/cpp/dis/include/shaders/dis_vr_w.comp diff --git a/app/src/main/cpp/winlator/vk/dis/vkr_dis.h b/app/src/main/cpp/dis/include/vkr_dis.h similarity index 62% rename from app/src/main/cpp/winlator/vk/dis/vkr_dis.h rename to app/src/main/cpp/dis/include/vkr_dis.h index ee3eacc68..4f7728108 100644 --- a/app/src/main/cpp/winlator/vk/dis/vkr_dis.h +++ b/app/src/main/cpp/dis/include/vkr_dis.h @@ -11,7 +11,7 @@ #include #include -#include "../vk_dispatch.h" +#include "vk_dispatch.h" #ifdef __cplusplus extern "C" { @@ -47,6 +47,25 @@ uint32_t vkr_dis_plan(VkrDis* dis, uint32_t capacity, uint64_t source_frames); void vkr_dis_process(VkrDis* dis, VkCommandBuffer cmd, VkImage source, uint32_t width, uint32_t height, uint32_t generations); +// Submits `cmd` (ended by the callee, no semaphores), waits for it on the CPU, and returns a +// command buffer in the recording state for the rest of the frame - `cmd` itself reset and begun +// again is fine. VK_NULL_HANDLE on failure. +typedef VkCommandBuffer (*VkrDisFlushFn)(void* user, VkCommandBuffer cmd); + +// As vkr_dis_process, and when the GLES driver offers GL_QCOM_motion_estimation, seeds the flow +// with a hardware estimate. That needs this frame's pixels mid-frame, so DIS calls `flush` once +// and records the rest into the buffer it returns; the caller continues with - and finally +// submits - the returned buffer. With flush NULL, or without the extension, this is exactly +// vkr_dis_process and returns `cmd`. +VkCommandBuffer vkr_dis_process_ex(VkrDis* dis, VkCommandBuffer cmd, VkImage source, + uint32_t width, uint32_t height, uint32_t generations, + VkrDisFlushFn flush, void* flush_user); + +// Hardware motion hint on or off (default off, or debug.winnative.dis.hwme=1); takes effect at +// the next resource build. +void vkr_dis_set_hw_motion(VkrDis* dis, bool enabled); +bool vkr_dis_hw_motion_active(const VkrDis* dis); + void vkr_dis_generate_into(VkrDis* dis, VkCommandBuffer cmd, uint32_t generation, uint32_t target_index, VkImage target_image, VkImageView target_view, uint32_t width, uint32_t height, diff --git a/app/src/main/cpp/dis/src/dis_qcom_me.c b/app/src/main/cpp/dis/src/dis_qcom_me.c new file mode 100644 index 000000000..51d9bc803 --- /dev/null +++ b/app/src/main/cpp/dis/src/dis_qcom_me.c @@ -0,0 +1,373 @@ +// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) +// SPDX-License-Identifier: GPL-3.0-or-later +// +// Hardware motion estimation through GL_QCOM_motion_estimation. See dis_qcom_me.h. +// +// The estimator lives in the platform GLES driver, while DIS runs on whichever Vulkan driver the +// compositor loaded (often Turnip), so nothing is shared between the two APIs: the luminance goes +// in with a texture upload and the field comes back with a read, both a few hundred kilobytes at +// most. EGL and GLES are opened with dlopen so the libraries linking DIS take no hard dependency +// on them, and each instance owns a surfaceless context that is made current only for the +// duration of a call, restoring whatever the calling thread had current before. + +#include "dis_qcom_me.h" + +#ifdef __ANDROID__ + +#include +#include +#include + +#include +#include +#include +#include +#include + +#define ME_LOGI(...) __android_log_print(ANDROID_LOG_INFO, "VkrDis", __VA_ARGS__) +#define ME_LOGW(...) __android_log_print(ANDROID_LOG_WARN, "VkrDis", __VA_ARGS__) + +#define GL_MOTION_ESTIMATION_SEARCH_BLOCK_X_QCOM 0x8C90 +#define GL_MOTION_ESTIMATION_SEARCH_BLOCK_Y_QCOM 0x8C91 + +// The NDK's EGL headers only declare the core entry points as prototypes, so their pointer types +// are spelled out here for the dlsym'd table below. +typedef EGLDisplay (*MeGetDisplay)(EGLNativeDisplayType); +typedef EGLBoolean (*MeInitialize)(EGLDisplay, EGLint*, EGLint*); +typedef EGLBoolean (*MeChooseConfig)(EGLDisplay, const EGLint*, EGLConfig*, EGLint, EGLint*); +typedef EGLBoolean (*MeBindAPI)(EGLenum); +typedef EGLContext (*MeCreateContext)(EGLDisplay, EGLConfig, EGLContext, const EGLint*); +typedef EGLBoolean (*MeDestroyContext)(EGLDisplay, EGLContext); +typedef EGLSurface (*MeCreatePbufferSurface)(EGLDisplay, EGLConfig, const EGLint*); +typedef EGLBoolean (*MeDestroySurface)(EGLDisplay, EGLSurface); +typedef EGLBoolean (*MeMakeCurrent)(EGLDisplay, EGLSurface, EGLSurface, EGLContext); +typedef EGLContext (*MeGetCurrentContext)(void); +typedef EGLDisplay (*MeGetCurrentDisplay)(void); +typedef EGLSurface (*MeGetCurrentSurface)(EGLint); +typedef const char* (*MeQueryString)(EGLDisplay, EGLint); +typedef void (*(*MeGetProcAddress)(const char*))(void); +typedef struct { + bool loaded; + bool ok; + MeGetDisplay GetDisplay; + MeInitialize Initialize; + MeChooseConfig ChooseConfig; + MeBindAPI BindAPI; + MeCreateContext CreateContext; + MeDestroyContext DestroyContext; + MeCreatePbufferSurface CreatePbufferSurface; + MeDestroySurface DestroySurface; + MeMakeCurrent MakeCurrent; + MeGetCurrentContext GetCurrentContext; + MeGetCurrentDisplay GetCurrentDisplay; + MeGetCurrentSurface GetCurrentSurface; + MeQueryString QueryString; + MeGetProcAddress GetProcAddress; + + const GLubyte* (*GetString)(GLenum); + void (*GetIntegerv)(GLenum, GLint*); + GLenum (*GetError)(void); + void (*GenTextures)(GLsizei, GLuint*); + void (*DeleteTextures)(GLsizei, const GLuint*); + void (*BindTexture)(GLenum, GLuint); + void (*TexStorage2D)(GLenum, GLsizei, GLenum, GLsizei, GLsizei); + void (*TexSubImage2D)(GLenum, GLint, GLint, GLint, GLsizei, GLsizei, GLenum, GLenum, const void*); + void (*TexParameteri)(GLenum, GLenum, GLint); + void (*PixelStorei)(GLenum, GLint); + void (*GenFramebuffers)(GLsizei, GLuint*); + void (*DeleteFramebuffers)(GLsizei, const GLuint*); + void (*BindFramebuffer)(GLenum, GLuint); + void (*FramebufferTexture2D)(GLenum, GLenum, GLenum, GLuint, GLint); + GLenum (*CheckFramebufferStatus)(GLenum); + void (*ReadPixels)(GLint, GLint, GLsizei, GLsizei, GLenum, GLenum, void*); + void (*EstimateMotion)(GLuint, GLuint, GLuint); +} MeApi; + +static MeApi g_api; +static pthread_mutex_t g_api_lock = PTHREAD_MUTEX_INITIALIZER; +static int g_probe = -1; // -1 not probed, 0 unsupported, 1 supported +static uint32_t g_block_x, g_block_y; + +struct DisQcomMe { + EGLDisplay display; + EGLContext context; + EGLSurface surface; // EGL_NO_SURFACE when surfaceless + uint32_t width, height; + uint32_t field_w, field_h; + GLuint luma[2]; + GLuint field; + GLuint fbo; + uint32_t newest; // index of the texture holding the newest frame + bool have_prev; +}; + +static bool me_load_api(void) { + if (g_api.loaded) return g_api.ok; + g_api.loaded = true; + void* egl = dlopen("libEGL.so", RTLD_NOW | RTLD_LOCAL); + void* gles = dlopen("libGLESv3.so", RTLD_NOW | RTLD_LOCAL); + if (!egl || !gles) return false; +#define EGLFN(field, name) g_api.field = (void*)dlsym(egl, name); if (!g_api.field) return false +#define GLFN(field, name) g_api.field = (void*)dlsym(gles, name); if (!g_api.field) return false + EGLFN(GetDisplay, "eglGetDisplay"); + EGLFN(Initialize, "eglInitialize"); + EGLFN(ChooseConfig, "eglChooseConfig"); + EGLFN(BindAPI, "eglBindAPI"); + EGLFN(CreateContext, "eglCreateContext"); + EGLFN(DestroyContext, "eglDestroyContext"); + EGLFN(CreatePbufferSurface, "eglCreatePbufferSurface"); + EGLFN(DestroySurface, "eglDestroySurface"); + EGLFN(MakeCurrent, "eglMakeCurrent"); + EGLFN(GetCurrentContext, "eglGetCurrentContext"); + EGLFN(GetCurrentDisplay, "eglGetCurrentDisplay"); + EGLFN(GetCurrentSurface, "eglGetCurrentSurface"); + EGLFN(QueryString, "eglQueryString"); + EGLFN(GetProcAddress, "eglGetProcAddress"); + GLFN(GetString, "glGetString"); + GLFN(GetIntegerv, "glGetIntegerv"); + GLFN(GetError, "glGetError"); + GLFN(GenTextures, "glGenTextures"); + GLFN(DeleteTextures, "glDeleteTextures"); + GLFN(BindTexture, "glBindTexture"); + GLFN(TexStorage2D, "glTexStorage2D"); + GLFN(TexSubImage2D, "glTexSubImage2D"); + GLFN(TexParameteri, "glTexParameteri"); + GLFN(PixelStorei, "glPixelStorei"); + GLFN(GenFramebuffers, "glGenFramebuffers"); + GLFN(DeleteFramebuffers, "glDeleteFramebuffers"); + GLFN(BindFramebuffer, "glBindFramebuffer"); + GLFN(FramebufferTexture2D, "glFramebufferTexture2D"); + GLFN(CheckFramebufferStatus, "glCheckFramebufferStatus"); + GLFN(ReadPixels, "glReadPixels"); +#undef EGLFN +#undef GLFN + g_api.EstimateMotion = (void*)g_api.GetProcAddress("glTexEstimateMotionQCOM"); + g_api.ok = g_api.EstimateMotion != NULL; + return g_api.ok; +} + +typedef struct { + EGLDisplay display; + EGLContext context; + EGLSurface draw, read; +} MeSaved; + +static void me_save(MeSaved* s) { + s->display = g_api.GetCurrentDisplay(); + s->context = g_api.GetCurrentContext(); + s->draw = g_api.GetCurrentSurface(EGL_DRAW); + s->read = g_api.GetCurrentSurface(EGL_READ); +} + +static void me_restore(const MeSaved* s, EGLDisplay own) { + if (s->context != EGL_NO_CONTEXT && s->display != EGL_NO_DISPLAY) { + g_api.MakeCurrent(s->display, s->draw, s->read, s->context); + } else { + g_api.MakeCurrent(own, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT); + } +} + +// Display, context and (only without surfaceless support) a 1x1 pbuffer. +static bool me_context(EGLDisplay* out_dpy, EGLContext* out_ctx, EGLSurface* out_surf) { + EGLDisplay dpy = g_api.GetDisplay(EGL_DEFAULT_DISPLAY); + if (dpy == EGL_NO_DISPLAY || !g_api.Initialize(dpy, NULL, NULL)) return false; + const char* ext = g_api.QueryString(dpy, EGL_EXTENSIONS); + const bool surfaceless = ext && strstr(ext, "EGL_KHR_surfaceless_context") != NULL; + const EGLint cfg_attr[] = {EGL_RENDERABLE_TYPE, EGL_OPENGL_ES3_BIT_KHR, + EGL_SURFACE_TYPE, EGL_PBUFFER_BIT, EGL_NONE}; + EGLConfig cfg; + EGLint n = 0; + if (!g_api.ChooseConfig(dpy, cfg_attr, &cfg, 1, &n) || n < 1) return false; + g_api.BindAPI(EGL_OPENGL_ES_API); + const EGLint ctx_attr[] = {EGL_CONTEXT_CLIENT_VERSION, 3, EGL_NONE}; + EGLContext ctx = g_api.CreateContext(dpy, cfg, EGL_NO_CONTEXT, ctx_attr); + if (ctx == EGL_NO_CONTEXT) return false; + EGLSurface surf = EGL_NO_SURFACE; + if (!surfaceless) { + const EGLint pb_attr[] = {EGL_WIDTH, 1, EGL_HEIGHT, 1, EGL_NONE}; + surf = g_api.CreatePbufferSurface(dpy, cfg, pb_attr); + if (surf == EGL_NO_SURFACE) { + g_api.DestroyContext(dpy, ctx); + return false; + } + } + *out_dpy = dpy; + *out_ctx = ctx; + *out_surf = surf; + return true; +} + +bool dis_qcom_me_supported(uint32_t* block_x, uint32_t* block_y) { + pthread_mutex_lock(&g_api_lock); + if (g_probe < 0) { + g_probe = 0; + EGLDisplay dpy; + EGLContext ctx; + EGLSurface surf; + if (me_load_api() && me_context(&dpy, &ctx, &surf)) { + MeSaved saved; + me_save(&saved); + if (g_api.MakeCurrent(dpy, surf, surf, ctx)) { + const char* ext = (const char*)g_api.GetString(GL_EXTENSIONS); + GLint bx = 0, by = 0; + if (ext && strstr(ext, "GL_QCOM_motion_estimation")) { + g_api.GetIntegerv(GL_MOTION_ESTIMATION_SEARCH_BLOCK_X_QCOM, &bx); + g_api.GetIntegerv(GL_MOTION_ESTIMATION_SEARCH_BLOCK_Y_QCOM, &by); + } + if (bx > 0 && by > 0) { + g_block_x = (uint32_t)bx; + g_block_y = (uint32_t)by; + g_probe = 1; + } + ME_LOGI("GL_QCOM_motion_estimation: %s (block %dx%d, %s)", + g_probe ? "available" : "not available", bx, by, + (const char*)g_api.GetString(GL_RENDERER)); + } + me_restore(&saved, dpy); + if (surf != EGL_NO_SURFACE) g_api.DestroySurface(dpy, surf); + g_api.DestroyContext(dpy, ctx); + } + } + const bool ok = g_probe == 1; + if (ok) { + if (block_x) *block_x = g_block_x; + if (block_y) *block_y = g_block_y; + } + pthread_mutex_unlock(&g_api_lock); + return ok; +} + +static GLuint me_texture(GLenum format, uint32_t w, uint32_t h) { + GLuint t = 0; + g_api.GenTextures(1, &t); + g_api.BindTexture(GL_TEXTURE_2D, t); + g_api.TexStorage2D(GL_TEXTURE_2D, 1, format, (GLsizei)w, (GLsizei)h); + g_api.TexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_NEAREST); + g_api.TexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MAG_FILTER, GL_NEAREST); + return t; +} + +DisQcomMe* dis_qcom_me_create(uint32_t width, uint32_t height) { + uint32_t bx = 0, by = 0; + if (!dis_qcom_me_supported(&bx, &by)) return NULL; + if (width == 0 || height == 0 || width % bx || height % by) return NULL; + + DisQcomMe* me = (DisQcomMe*)calloc(1, sizeof(DisQcomMe)); + if (!me) return NULL; + me->width = width; + me->height = height; + me->field_w = width / bx; + me->field_h = height / by; + if (!me_context(&me->display, &me->context, &me->surface)) { + free(me); + return NULL; + } + + MeSaved saved; + me_save(&saved); + bool ok = g_api.MakeCurrent(me->display, me->surface, me->surface, me->context); + if (ok) { + me->luma[0] = me_texture(GL_R8, width, height); + me->luma[1] = me_texture(GL_R8, width, height); + me->field = me_texture(GL_RGBA16F, me->field_w, me->field_h); + g_api.GenFramebuffers(1, &me->fbo); + g_api.BindFramebuffer(GL_FRAMEBUFFER, me->fbo); + g_api.FramebufferTexture2D(GL_FRAMEBUFFER, GL_COLOR_ATTACHMENT0, GL_TEXTURE_2D, me->field, 0); + ok = g_api.CheckFramebufferStatus(GL_FRAMEBUFFER) == GL_FRAMEBUFFER_COMPLETE && + g_api.GetError() == GL_NO_ERROR; + } + me_restore(&saved, me->display); + if (!ok) { + ME_LOGW("GL_QCOM_motion_estimation: could not set up %ux%u; using DIS alone", width, height); + dis_qcom_me_destroy(me); + return NULL; + } + ME_LOGI("GL_QCOM_motion_estimation: %ux%u luminance -> %ux%u field", width, height, + me->field_w, me->field_h); + return me; +} + +void dis_qcom_me_destroy(DisQcomMe* me) { + if (!me) return; + if (me->context != EGL_NO_CONTEXT) { + MeSaved saved; + me_save(&saved); + if (g_api.MakeCurrent(me->display, me->surface, me->surface, me->context)) { + if (me->fbo) g_api.DeleteFramebuffers(1, &me->fbo); + GLuint tex[3] = {me->luma[0], me->luma[1], me->field}; + g_api.DeleteTextures(3, tex); + } + me_restore(&saved, me->display); + if (me->surface != EGL_NO_SURFACE) g_api.DestroySurface(me->display, me->surface); + g_api.DestroyContext(me->display, me->context); + } + free(me); +} + +void dis_qcom_me_invalidate(DisQcomMe* me) { + if (me) me->have_prev = false; +} + +bool dis_qcom_me_push(DisQcomMe* me, const uint8_t* luma, float* out_xy) { + if (!me || !luma) return false; + MeSaved saved; + me_save(&saved); + if (!g_api.MakeCurrent(me->display, me->surface, me->surface, me->context)) { + me_restore(&saved, me->display); + return false; + } + + const uint32_t cur = me->have_prev ? 1u - me->newest : me->newest; + g_api.BindTexture(GL_TEXTURE_2D, me->luma[cur]); + g_api.PixelStorei(GL_UNPACK_ALIGNMENT, 1); + g_api.TexSubImage2D(GL_TEXTURE_2D, 0, 0, 0, (GLsizei)me->width, (GLsizei)me->height, GL_RED, + GL_UNSIGNED_BYTE, luma); + + bool estimated = false; + if (me->have_prev && out_xy) { + g_api.EstimateMotion(me->luma[me->newest], me->luma[cur], me->field); + const uint32_t n = me->field_w * me->field_h; + float* rgba = (float*)malloc((size_t)n * 4 * sizeof(float)); + if (rgba) { + g_api.BindFramebuffer(GL_FRAMEBUFFER, me->fbo); + g_api.ReadPixels(0, 0, (GLsizei)me->field_w, (GLsizei)me->field_h, GL_RGBA, GL_FLOAT, + rgba); + if (g_api.GetError() == GL_NO_ERROR) { + for (uint32_t i = 0; i < n; i++) { + out_xy[i * 2] = rgba[i * 4]; + out_xy[i * 2 + 1] = rgba[i * 4 + 1]; + } + estimated = true; + } + free(rgba); + } + } + me->newest = cur; + me->have_prev = true; + + me_restore(&saved, me->display); + return estimated; +} + +#else // !__ANDROID__: no GLES driver to ask; DIS runs alone. + +bool dis_qcom_me_supported(uint32_t* block_x, uint32_t* block_y) { + (void)block_x; + (void)block_y; + return false; +} +DisQcomMe* dis_qcom_me_create(uint32_t width, uint32_t height) { + (void)width; + (void)height; + return NULL; +} +void dis_qcom_me_destroy(DisQcomMe* me) { (void)me; } +bool dis_qcom_me_push(DisQcomMe* me, const uint8_t* luma, float* out_xy) { + (void)me; + (void)luma; + (void)out_xy; + return false; +} +void dis_qcom_me_invalidate(DisQcomMe* me) { (void)me; } + +#endif diff --git a/app/src/main/cpp/dis/src/dis_qcom_me.h b/app/src/main/cpp/dis/src/dis_qcom_me.h new file mode 100644 index 000000000..3975cf596 --- /dev/null +++ b/app/src/main/cpp/dis/src/dis_qcom_me.h @@ -0,0 +1,30 @@ +// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) +// SPDX-License-Identifier: GPL-3.0-or-later +// +// Hardware motion estimation through GL_QCOM_motion_estimation (Adreno), used by DIS as a +// second starting point for its block search. Private to the DIS module. + +#pragma once + +#include +#include +#include + +typedef struct DisQcomMe DisQcomMe; + +// Whether the platform's GLES driver exposes the extension, and its search block size. +// Probed once per process; cheap after the first call. +bool dis_qcom_me_supported(uint32_t* block_x, uint32_t* block_y); + +// Input frames are width x height R8 luminance, both multiples of the block size; the field is +// (width / block_x) x (height / block_y) vectors. NULL when unsupported or on failure. +DisQcomMe* dis_qcom_me_create(uint32_t width, uint32_t height); +void dis_qcom_me_destroy(DisQcomMe* me); + +// Hands the newest frame over. When an earlier frame is held, estimates the motion from it to +// this one into out_xy (field_w * field_h pairs, pixels of the input size, previous -> newest) +// and returns true. Either way the newest frame becomes the reference for the next call. +bool dis_qcom_me_push(DisQcomMe* me, const uint8_t* luma, float* out_xy); + +// Drops the held frame, so the next push only primes the history. +void dis_qcom_me_invalidate(DisQcomMe* me); diff --git a/app/src/main/cpp/winlator/vk/dis/vkr_dis.c b/app/src/main/cpp/dis/src/vkr_dis.c similarity index 79% rename from app/src/main/cpp/winlator/vk/dis/vkr_dis.c rename to app/src/main/cpp/dis/src/vkr_dis.c index eff38c970..ebc29f5a1 100644 --- a/app/src/main/cpp/winlator/vk/dis/vkr_dis.c +++ b/app/src/main/cpp/dis/src/vkr_dis.c @@ -8,7 +8,8 @@ #include "vkr_dis.h" -#include "../vk_dispatch.h" +#include "dis_qcom_me.h" +#include "vk_dispatch.h" #include "shaders/dis_luma_r16_comp.spv.h" #include "shaders/dis_luma_r32_comp.spv.h" #include "shaders/dis_gradient_comp.spv.h" @@ -18,7 +19,8 @@ #include "shaders/dis_interpolate_comp.spv.h" #include "shaders/dis_hist_comp.spv.h" #include "shaders/dis_side_comp.spv.h" -#include "shaders/dis_temporal_comp.spv.h" +#include "shaders/dis_flow_pack_comp.spv.h" +#include "shaders/dis_me_luma_comp.spv.h" #include "shaders/dis_vr_prep_comp.spv.h" #include "shaders/dis_vr_d1_comp.spv.h" #include "shaders/dis_vr_d2_comp.spv.h" @@ -34,6 +36,9 @@ #include #include +#ifdef __ANDROID__ +#include +#endif #define DIS_LOGI(...) __android_log_print(ANDROID_LOG_INFO, "VkrDis", __VA_ARGS__) #define DIS_LOGW(...) __android_log_print(ANDROID_LOG_WARN, "VkrDis", __VA_ARGS__) @@ -50,6 +55,12 @@ #define DIS_SLOTS 3u +// How the hardware motion field enters the flow: as a second starting candidate for the search on +// its level (0), or as that level's result outright, skipping the search above it (1). +#ifndef DIS_ME_PRIMARY +#define DIS_ME_PRIMARY 0 +#endif + #define DIS_PROP_STEPS_MAX 4u #define DIS_SRC_SMOOTHING 0.08f @@ -154,7 +165,8 @@ struct VkrDis { DisImage flow_refined; DisImage hist[2]; DisImage side; - DisImage flow_smooth[2]; + DisImage flow_out; + DisImage me_field; VkImageView view_color[DIS_SLOTS]; VkImageView view_flow_color[DIS_SLOTS][DIS_MAX_LEVELS]; @@ -174,7 +186,8 @@ struct VkrDis { VkImageView view_flow_refined[DIS_MAX_LEVELS]; VkImageView view_hist[2]; VkImageView view_side; - VkImageView view_flow_smooth[2]; + VkImageView view_flow_out; + VkImageView view_me_field; VkSampler sampler; @@ -190,7 +203,7 @@ struct VkrDis { VkDescriptorSet interp_sets[DIS_SLOTS][2]; VkDescriptorSet hist_sets[DIS_SLOTS][2]; VkDescriptorSet side_sets[DIS_SLOTS]; - VkDescriptorSet temporal_sets[2]; + VkDescriptorSet pack_set; VkDescriptorSetLayout vr_set_layout; VkPipelineLayout vr_pipeline_layout; @@ -211,7 +224,8 @@ struct VkrDis { DisPass pass_interp; DisPass pass_hist; DisPass pass_side; - DisPass pass_temporal; + DisPass pass_pack; + DisPass pass_me_luma; DisPass pass_vr_prep; DisPass pass_vr_d1; DisPass pass_vr_d2; @@ -244,6 +258,31 @@ struct VkrDis { uint32_t hist_parity; bool hist_valid; + + // Hardware motion hint (GL_QCOM_motion_estimation). me is NULL whenever the hint is off: + // disabled, unsupported, or the pyramid too shallow for the level it seeds. + bool hw_motion; + DisQcomMe* me; + uint32_t me_level; + uint32_t me_w, me_h; + uint32_t me_field_w, me_field_h; + VkBuffer me_luma_buf; + VkDeviceMemory me_luma_mem; + void* me_luma_map; + bool me_luma_coherent; + VkBuffer me_field_buf; + VkDeviceMemory me_field_mem; + void* me_field_map; + bool me_field_coherent; + float* me_xy; + int hint_level; + bool me_primary_frame; + VkDescriptorSetLayout me_set_layout; + VkPipelineLayout me_pipeline_layout; + VkDescriptorPool me_pool; + VkDescriptorSet me_sets[DIS_SLOTS]; + uint64_t me_hinted; + uint64_t me_pairs; }; typedef struct { @@ -255,6 +294,7 @@ typedef struct { typedef struct { int level; int coarseLevel; + int hintLevel; } DisInversePC; typedef struct { @@ -283,6 +323,8 @@ typedef struct { int parity; } DisVrSorPC; +static void dis_destroy_me(VkrDis* d); + static uint64_t dis_now_ns(void) { struct timespec ts; clock_gettime(CLOCK_MONOTONIC, &ts); @@ -355,8 +397,8 @@ static uint32_t dis_collect_images(VkrDis* d, DisImage** out, uint32_t cap) { DIS_PUSH(&d->hist[0]); DIS_PUSH(&d->hist[1]); DIS_PUSH(&d->side); - DIS_PUSH(&d->flow_smooth[0]); - DIS_PUSH(&d->flow_smooth[1]); + DIS_PUSH(&d->flow_out); + DIS_PUSH(&d->me_field); #undef DIS_PUSH return n; } @@ -556,7 +598,7 @@ static bool dis_create_pipelines(VkrDis* d) { const uint32_t shared_sets = DIS_SLOTS * DIS_MAX_LEVELS * DIS_SHARED_SETS_PER_LEVEL + DIS_SLOTS * 2u // interpolation sets, one per history direction + DIS_SLOTS // side-map sets - + 2u; // temporal flow sets, one per direction + + 1u; // flow pack set // VR sets exist only for the levels the refinement actually runs on. const uint32_t vr_sets = (DIS_SLOTS + DIS_VR_SHARED_SETS) * DIS_VR_LEVELS; @@ -658,14 +700,62 @@ static bool dis_create_pipelines(VkrDis* d) { d->pass_vr_add.pipeline = dis_create_compute_pipeline_with_layout(d, dis_vr_add_comp, dis_vr_add_comp_size, d->vr_pipeline_layout, NULL); d->pass_hist.pipeline = dis_create_compute_pipeline_with_layout(d, dis_hist_comp, dis_hist_comp_size, d->vr_pipeline_layout, NULL); d->pass_side.pipeline = dis_create_compute_pipeline(d, dis_side_comp, dis_side_comp_size); - d->pass_temporal.pipeline = dis_create_compute_pipeline(d, dis_temporal_comp, dis_temporal_comp_size); + d->pass_pack.pipeline = dis_create_compute_pipeline(d, dis_flow_pack_comp, dis_flow_pack_comp_size); + + // The hardware-motion luminance pass writes a host-visible buffer, which neither shared + // layout has, so it gets a small layout and pool of its own: the colour frame and the buffer. + VkDescriptorSetLayoutBinding me_bindings[2]; + memset(me_bindings, 0, sizeof(me_bindings)); + me_bindings[0].binding = 0; + me_bindings[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + me_bindings[0].descriptorCount = 1; + me_bindings[0].stageFlags = VK_SHADER_STAGE_COMPUTE_BIT; + me_bindings[1].binding = 1; + me_bindings[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + me_bindings[1].descriptorCount = 1; + me_bindings[1].stageFlags = VK_SHADER_STAGE_COMPUTE_BIT; + VkDescriptorSetLayoutCreateInfo me_li; + memset(&me_li, 0, sizeof(me_li)); + me_li.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO; + me_li.bindingCount = 2; + me_li.pBindings = me_bindings; + if (vkd.CreateDescriptorSetLayout(d->device, &me_li, NULL, &d->me_set_layout) != VK_SUCCESS) { + return false; + } + VkPipelineLayoutCreateInfo me_pli; + memset(&me_pli, 0, sizeof(me_pli)); + me_pli.sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO; + me_pli.setLayoutCount = 1; + me_pli.pSetLayouts = &d->me_set_layout; + me_pli.pushConstantRangeCount = 1; + me_pli.pPushConstantRanges = &pcr; + if (vkd.CreatePipelineLayout(d->device, &me_pli, NULL, &d->me_pipeline_layout) != VK_SUCCESS) { + return false; + } + VkDescriptorPoolSize me_sizes[2]; + memset(me_sizes, 0, sizeof(me_sizes)); + me_sizes[0].type = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + me_sizes[0].descriptorCount = DIS_SLOTS; + me_sizes[1].type = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + me_sizes[1].descriptorCount = DIS_SLOTS; + VkDescriptorPoolCreateInfo me_pci; + memset(&me_pci, 0, sizeof(me_pci)); + me_pci.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO; + me_pci.maxSets = DIS_SLOTS; + me_pci.poolSizeCount = 2; + me_pci.pPoolSizes = me_sizes; + if (vkd.CreateDescriptorPool(d->device, &me_pci, NULL, &d->me_pool) != VK_SUCCESS) { + return false; + } + d->pass_me_luma.pipeline = dis_create_compute_pipeline_with_layout( + d, dis_me_luma_comp, dis_me_luma_comp_size, d->me_pipeline_layout, NULL); if (!d->pass_gradient.pipeline || !d->pass_inverse.pipeline || !d->pass_propagate.pipeline || !d->pass_densify.pipeline || !d->pass_interp.pipeline || !d->pass_vr_prep.pipeline || !d->pass_vr_d1.pipeline || !d->pass_vr_d2.pipeline || !d->pass_vr_w.pipeline || !d->pass_vr_coef.pipeline || !d->pass_vr_sor.pipeline || !d->pass_vr_add.pipeline || !d->pass_hist.pipeline || !d->pass_side.pipeline || - !d->pass_temporal.pipeline) { + !d->pass_pack.pipeline || !d->pass_me_luma.pipeline) { return false; } return true; @@ -814,7 +904,7 @@ static void dis_write_all_descriptors(VkrDis* d) { const VkImageView coarse_view = l + 1 < DIS_VR_LEVELS ? d->view_flow_refined[coarse_l] : d->view_dense[coarse_l]; dis_batch_sampled(d, &b, d->inverse_sets[s][l], 3, coarse_view, d->sampler); - dis_batch_sampled(d, &b, d->inverse_sets[s][l], 4, d->view_dense[coarse], d->sampler); + dis_batch_sampled(d, &b, d->inverse_sets[s][l], 4, d->view_me_field, d->sampler); dis_batch_storage(d, &b, d->inverse_sets[s][l], 5, d->view_sparse[l]); dis_batch_sampled(d, &b, d->prop_ab_sets[s][l], 0, d->view_flow_luma[prev][l], d->sampler); @@ -836,7 +926,7 @@ static void dis_write_all_descriptors(VkrDis* d) { for (uint32_t dir = 0; dir < 2u; dir++) { dis_batch_sampled(d, &b, d->interp_sets[s][dir], 0, d->view_color[prev], d->sampler); dis_batch_sampled(d, &b, d->interp_sets[s][dir], 1, d->view_color[next], d->sampler); - dis_batch_sampled(d, &b, d->interp_sets[s][dir], 2, d->view_flow_smooth[dir], d->sampler); + dis_batch_sampled(d, &b, d->interp_sets[s][dir], 2, d->view_flow_out, d->sampler); dis_batch_sampled(d, &b, d->interp_sets[s][dir], 3, d->view_side, d->sampler); dis_batch_sampled(d, &b, d->interp_sets[s][dir], 4, d->view_hist[dir], d->sampler); dis_batch_storage(d, &b, d->interp_sets[s][dir], 5, d->view_interp_out); @@ -896,13 +986,10 @@ static void dis_write_all_descriptors(VkrDis* d) { dis_batch_storage(d, &b, d->vr_add_set[l], DIS_VR_FIRST_STORAGE, d->view_flow_refined[l]); } - // Not per slot: both inputs and the output are single images, and these sets are - // written once here and never updated, so sharing them across frames is safe. - for (uint32_t dir = 0; dir < 2u; dir++) { - dis_batch_sampled(d, &b, d->temporal_sets[dir], 0, d->view_flow_refined[0], d->sampler); - dis_batch_sampled(d, &b, d->temporal_sets[dir], 1, d->view_flow_smooth[1u - dir], d->sampler); - dis_batch_storage(d, &b, d->temporal_sets[dir], 5, d->view_flow_smooth[dir]); - } + // Not per slot: the input and the output are single images, and this set is written + // once here and never updated, so sharing it across frames is safe. + dis_batch_sampled(d, &b, d->pack_set, 0, d->view_flow_refined[0], d->sampler); + dis_batch_storage(d, &b, d->pack_set, 5, d->view_flow_out); dis_batch_flush(d, &b); } @@ -913,8 +1000,8 @@ static void dis_destroy_views(VkrDis* d) { dis_destroy_view(d, &d->view_hist[0]); dis_destroy_view(d, &d->view_hist[1]); dis_destroy_view(d, &d->view_side); - dis_destroy_view(d, &d->view_flow_smooth[0]); - dis_destroy_view(d, &d->view_flow_smooth[1]); + dis_destroy_view(d, &d->view_flow_out); + dis_destroy_view(d, &d->view_me_field); for (uint32_t l = 0; l < DIS_MAX_LEVELS; l++) { dis_destroy_view(d, &d->view_vr_prep[l]); dis_destroy_view(d, &d->view_vr_d1[l]); @@ -962,8 +1049,131 @@ static void dis_destroy_images(VkrDis* d) { dis_destroy_image(d, &d->hist[0]); dis_destroy_image(d, &d->hist[1]); dis_destroy_image(d, &d->side); - dis_destroy_image(d, &d->flow_smooth[0]); - dis_destroy_image(d, &d->flow_smooth[1]); + dis_destroy_image(d, &d->flow_out); + dis_destroy_image(d, &d->me_field); + dis_destroy_me(d); +} + +static void dis_destroy_buffer(VkrDis* d, VkBuffer* buf, VkDeviceMemory* mem, void** map) { + if (*map) vkd.UnmapMemory(d->device, *mem); + if (*buf) vkd.DestroyBuffer(d->device, *buf, NULL); + if (*mem) vkd.FreeMemory(d->device, *mem, NULL); + *buf = VK_NULL_HANDLE; + *mem = VK_NULL_HANDLE; + *map = NULL; +} + +// Host-visible buffer, mapped for its whole life. Cached memory is preferred for a buffer the +// CPU reads back: uncached reads of a few hundred kilobytes cost far more than the invalidate. +static bool dis_create_host_buffer(VkrDis* d, VkDeviceSize size, VkBufferUsageFlags usage, + bool readback, VkBuffer* buf, VkDeviceMemory* mem, void** map, + bool* coherent) { + VkBufferCreateInfo bi; + memset(&bi, 0, sizeof(bi)); + bi.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO; + bi.size = size; + bi.usage = usage; + bi.sharingMode = VK_SHARING_MODE_EXCLUSIVE; + if (vkd.CreateBuffer(d->device, &bi, NULL, buf) != VK_SUCCESS) return false; + VkMemoryRequirements mr; + vkd.GetBufferMemoryRequirements(d->device, *buf, &mr); + const VkMemoryPropertyFlags HV = VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT; + const VkMemoryPropertyFlags HC = VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; + const VkMemoryPropertyFlags CA = VK_MEMORY_PROPERTY_HOST_CACHED_BIT; + uint32_t type = UINT32_MAX; + if (readback) { + type = dis_find_memory_type(d, mr.memoryTypeBits, HV | CA | HC); + if (type == UINT32_MAX) type = dis_find_memory_type(d, mr.memoryTypeBits, HV | CA); + } + if (type == UINT32_MAX) type = dis_find_memory_type(d, mr.memoryTypeBits, HV | HC); + if (type == UINT32_MAX) type = dis_find_memory_type(d, mr.memoryTypeBits, HV); + if (type == UINT32_MAX) return false; + *coherent = (d->mem_props.memoryTypes[type].propertyFlags & HC) != 0; + VkMemoryAllocateInfo ai; + memset(&ai, 0, sizeof(ai)); + ai.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO; + ai.allocationSize = mr.size; + ai.memoryTypeIndex = type; + if (vkd.AllocateMemory(d->device, &ai, NULL, mem) != VK_SUCCESS) return false; + if (vkd.BindBufferMemory(d->device, *buf, *mem, 0) != VK_SUCCESS) return false; + return vkd.MapMemory(d->device, *mem, 0, VK_WHOLE_SIZE, 0, map) == VK_SUCCESS; +} + +static void dis_destroy_me(VkrDis* d) { + dis_qcom_me_destroy(d->me); + d->me = NULL; + dis_destroy_buffer(d, &d->me_luma_buf, &d->me_luma_mem, &d->me_luma_map); + dis_destroy_buffer(d, &d->me_field_buf, &d->me_field_mem, &d->me_field_map); + free(d->me_xy); + d->me_xy = NULL; + d->hint_level = -1; +} + +// The hint seeds the search on level me_level of a w x h pyramid: the estimator's field is that +// level's size, and its input is the field times the block size, which for the usual 2:1 levels +// is twice the flow extent - so the estimator sees finer detail than the level it seeds, and its +// fixed search range covers twice the motion it would at the flow extent. +static void dis_create_me(VkrDis* d, uint32_t w, uint32_t h) { + dis_destroy_me(d); + if (!d->hw_motion || d->levels < 3) return; + uint32_t bx = 0, by = 0; + if (!dis_qcom_me_supported(&bx, &by)) return; + + d->me_level = 2; + d->me_field_w = w >> d->me_level; + d->me_field_h = h >> d->me_level; + d->me_w = d->me_field_w * bx; + d->me_h = d->me_field_h * by; + // Tiny inputs came back as NaN on Adreno 750 (320x176); stay well clear of that. + if (d->me_w < 256 || d->me_h < 144 || (d->me_w & 3u)) return; + + const VkDeviceSize luma_bytes = (VkDeviceSize)d->me_w * d->me_h; + const VkDeviceSize field_bytes = (VkDeviceSize)d->me_field_w * d->me_field_h * 2 * sizeof(float); + if (!dis_create_host_buffer(d, luma_bytes, VK_BUFFER_USAGE_STORAGE_BUFFER_BIT, true, + &d->me_luma_buf, &d->me_luma_mem, &d->me_luma_map, + &d->me_luma_coherent) || + !dis_create_host_buffer(d, field_bytes, VK_BUFFER_USAGE_TRANSFER_SRC_BIT, false, + &d->me_field_buf, &d->me_field_mem, &d->me_field_map, + &d->me_field_coherent)) { + DIS_LOGW("DIS hardware motion: host buffers unavailable; using DIS alone"); + dis_destroy_me(d); + return; + } + d->me_xy = (float*)malloc((size_t)d->me_field_w * d->me_field_h * 2 * sizeof(float)); + d->me = d->me_xy ? dis_qcom_me_create(d->me_w, d->me_h) : NULL; + if (!d->me) { + dis_destroy_me(d); + return; + } + + for (uint32_t s = 0; s < DIS_SLOTS; s++) { + VkDescriptorImageInfo ii; + memset(&ii, 0, sizeof(ii)); + ii.sampler = d->sampler; + ii.imageView = d->view_color[s]; + ii.imageLayout = VK_IMAGE_LAYOUT_GENERAL; + VkDescriptorBufferInfo bi; + memset(&bi, 0, sizeof(bi)); + bi.buffer = d->me_luma_buf; + bi.range = VK_WHOLE_SIZE; + VkWriteDescriptorSet w2[2]; + memset(w2, 0, sizeof(w2)); + w2[0].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + w2[0].dstSet = d->me_sets[s]; + w2[0].dstBinding = 0; + w2[0].descriptorCount = 1; + w2[0].descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER; + w2[0].pImageInfo = ⅈ + w2[1].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET; + w2[1].dstSet = d->me_sets[s]; + w2[1].dstBinding = 1; + w2[1].descriptorCount = 1; + w2[1].descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER; + w2[1].pBufferInfo = &bi; + vkd.UpdateDescriptorSets(d->device, 2, w2, 0, NULL); + } + DIS_LOGI("DIS hardware motion hint: GL_QCOM_motion_estimation on %ux%u seeds level %u (%ux%u)", + d->me_w, d->me_h, d->me_level, d->me_field_w, d->me_field_h); } static bool dis_create_resources(VkrDis* d, uint32_t w, uint32_t h, uint32_t full_w, @@ -1027,14 +1237,17 @@ static bool dis_create_resources(VkrDis* d, uint32_t w, uint32_t h, uint32_t ful VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_STORAGE_BIT)) return false; if (!dis_create_image(d, &d->side, w, h, VK_FORMAT_R32_SFLOAT, 1, VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_STORAGE_BIT)) return false; - // Ping-pong for the temporally smoothed field, level 0 only, at the flow extent. - // Four channels, not two: xy is this pair's chord and zw the previous pair's chord - // for the same content, so the interpolate pass gets both from the one fetch it was - // already making. The storage format is the same one the sparse flow maps use. - if (!dis_create_image(d, &d->flow_smooth[0], w, h, VK_FORMAT_R32G32B32A32_SFLOAT, 1, - VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_STORAGE_BIT)) return false; - if (!dis_create_image(d, &d->flow_smooth[1], w, h, VK_FORMAT_R32G32B32A32_SFLOAT, 1, + // The finished level-0 field as the interpolate pass reads it: RG16F is filterable on + // every device, so its three lookups per output pixel stay single bilinear taps. + if (!dis_create_image(d, &d->flow_out, w, h, VK_FORMAT_R16G16_SFLOAT, 1, VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_STORAGE_BIT)) return false; + // Hardware motion hint, one vector per block of the level it seeds. It always exists so the + // search's binding stays valid; the search reads it only on frames that uploaded one. + const uint32_t hint_w = L >= 3 ? (w >> 2) : 1u; + const uint32_t hint_h = L >= 3 ? (h >> 2) : 1u; + if (!dis_create_image(d, &d->me_field, hint_w ? hint_w : 1u, hint_h ? hint_h : 1u, + VK_FORMAT_R32G32_SFLOAT, 1, + VK_IMAGE_USAGE_SAMPLED_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT)) return false; for (uint32_t s = 0; s < DIS_SLOTS; s++) { if (!dis_create_view(d, d->color[s].image, format, 0, 1, &d->view_color[s])) return false; @@ -1064,11 +1277,12 @@ static bool dis_create_resources(VkrDis* d, uint32_t w, uint32_t h, uint32_t ful if (!dis_create_view(d, d->hist[0].image, VK_FORMAT_R32_SFLOAT, 0, 1, &d->view_hist[0])) return false; if (!dis_create_view(d, d->hist[1].image, VK_FORMAT_R32_SFLOAT, 0, 1, &d->view_hist[1])) return false; if (!dis_create_view(d, d->side.image, VK_FORMAT_R32_SFLOAT, 0, 1, &d->view_side)) return false; - if (!dis_create_view(d, d->flow_smooth[0].image, VK_FORMAT_R32G32B32A32_SFLOAT, 0, 1, &d->view_flow_smooth[0])) return false; - if (!dis_create_view(d, d->flow_smooth[1].image, VK_FORMAT_R32G32B32A32_SFLOAT, 0, 1, &d->view_flow_smooth[1])) return false; + if (!dis_create_view(d, d->flow_out.image, VK_FORMAT_R16G16_SFLOAT, 0, 1, &d->view_flow_out)) return false; + if (!dis_create_view(d, d->me_field.image, VK_FORMAT_R32G32_SFLOAT, 0, 1, &d->view_me_field)) return false; vkr_dis_reset(d); dis_write_all_descriptors(d); + dis_create_me(d, w, h); return true; } @@ -1113,8 +1327,17 @@ static bool dis_allocate_sets(VkrDis* d) { } } - if (!dis_alloc(d, d->set_layout, 1, &d->temporal_sets[0])) return false; - if (!dis_alloc(d, d->set_layout, 1, &d->temporal_sets[1])) return false; + if (!dis_alloc(d, d->set_layout, 1, &d->pack_set)) return false; + + for (uint32_t s = 0; s < DIS_SLOTS; s++) { + VkDescriptorSetAllocateInfo ai; + memset(&ai, 0, sizeof(ai)); + ai.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO; + ai.descriptorPool = d->me_pool; + ai.descriptorSetCount = 1; + ai.pSetLayouts = &d->me_set_layout; + if (vkd.AllocateDescriptorSets(d->device, &ai, &d->me_sets[s]) != VK_SUCCESS) return false; + } for (uint32_t l = 0; l < DIS_VR_LEVELS; l++) { VkDescriptorSet vr_sets[DIS_VR_SHARED_SETS]; @@ -1188,6 +1411,7 @@ static const char* dis_format_name(VkFormat f) { case VK_FORMAT_R16_SFLOAT: return "R16_SFLOAT"; case VK_FORMAT_R32_SFLOAT: return "R32_SFLOAT"; case VK_FORMAT_R32G32_SFLOAT: return "R32G32_SFLOAT"; + case VK_FORMAT_R16G16_SFLOAT: return "R16G16_SFLOAT"; case VK_FORMAT_R32G32B32A32_SFLOAT: return "R32G32B32A32_SFLOAT"; case VK_FORMAT_R8G8B8A8_UNORM: return "R8G8B8A8_UNORM"; case VK_FORMAT_UNDEFINED: return "none"; @@ -1229,6 +1453,7 @@ static bool dis_audit_formats(VkrDis* d) { const char* purpose; } reqs[] = { {VK_FORMAT_R32G32_SFLOAT, STORE | READ, "optical flow"}, + {VK_FORMAT_R16G16_SFLOAT, STORE | READ, "interpolation flow"}, {VK_FORMAT_R32G32B32A32_SFLOAT, STORE | READ, "sparse flow and refinement"}, {VK_FORMAT_R32_SFLOAT, STORE | READ, "refinement weights"}, {VK_FORMAT_R8G8B8A8_UNORM, STORE | VK_FORMAT_FEATURE_BLIT_SRC_BIT, @@ -1262,11 +1487,11 @@ static bool dis_audit_formats(VkrDis* d) { VkFormatProperties flow_fp; memset(&flow_fp, 0, sizeof(flow_fp)); - vkd.GetPhysicalDeviceFormatProperties(d->physical_device, VK_FORMAT_R32G32_SFLOAT, &flow_fp); + vkd.GetPhysicalDeviceFormatProperties(d->physical_device, VK_FORMAT_R16G16_SFLOAT, &flow_fp); d->manual_flow_filter = (flow_fp.optimalTilingFeatures & FILTER) == 0; DIS_LOGI("DIS format support: %s | flow filtering: %s", line, - d->manual_flow_filter ? "in shader (driver cannot filter R32G32_SFLOAT)" + d->manual_flow_filter ? "in shader (driver cannot filter R16G16_SFLOAT)" : "sampler"); return ok; } @@ -1281,6 +1506,18 @@ VkrDis* vkr_dis_create(VkDevice device, VkPhysicalDevice physical_device) { d->target_fps = 0; d->refresh_rate = 0.0f; d->plan_log_gen = -1; + // Off by default: on Adreno 750 the hint left quality unchanged and cost ~2.5 ms per real + // frame, most of it the mid-frame submit it needs. Opt in on a device without a rebuild: + // adb shell setprop debug.winnative.dis.hwme 1 + d->hw_motion = false; +#ifdef __ANDROID__ + char prop[PROP_VALUE_MAX] = {0}; + if (__system_property_get("debug.winnative.dis.hwme", prop) > 0 && prop[0] == '1') { + d->hw_motion = true; + DIS_LOGI("DIS hardware motion hint enabled by debug.winnative.dis.hwme"); + } +#endif + d->hint_level = -1; vkd.GetPhysicalDeviceMemoryProperties(physical_device, &d->mem_props); d->luma_format = dis_pick_luma_format(d); if (!dis_audit_formats(d)) { @@ -1321,7 +1558,12 @@ void vkr_dis_destroy(VkrDis* d) { if (d->pass_vr_sor.pipeline) vkd.DestroyPipeline(d->device, d->pass_vr_sor.pipeline, NULL); if (d->pass_vr_add.pipeline) vkd.DestroyPipeline(d->device, d->pass_vr_add.pipeline, NULL); if (d->pass_hist.pipeline) vkd.DestroyPipeline(d->device, d->pass_hist.pipeline, NULL); - if (d->pass_temporal.pipeline) vkd.DestroyPipeline(d->device, d->pass_temporal.pipeline, NULL); + if (d->pass_side.pipeline) vkd.DestroyPipeline(d->device, d->pass_side.pipeline, NULL); + if (d->pass_pack.pipeline) vkd.DestroyPipeline(d->device, d->pass_pack.pipeline, NULL); + if (d->pass_me_luma.pipeline) vkd.DestroyPipeline(d->device, d->pass_me_luma.pipeline, NULL); + if (d->me_pool) vkd.DestroyDescriptorPool(d->device, d->me_pool, NULL); + if (d->me_pipeline_layout) vkd.DestroyPipelineLayout(d->device, d->me_pipeline_layout, NULL); + if (d->me_set_layout) vkd.DestroyDescriptorSetLayout(d->device, d->me_set_layout, NULL); if (d->pool) vkd.DestroyDescriptorPool(d->device, d->pool, NULL); if (d->pipeline_layout) vkd.DestroyPipelineLayout(d->device, d->pipeline_layout, NULL); if (d->vr_pipeline_layout) vkd.DestroyPipelineLayout(d->device, d->vr_pipeline_layout, NULL); @@ -1670,11 +1912,156 @@ static void dis_vr_level(VkrDis* d, VkCommandBuffer cmd, uint32_t slot, uint32_t dis_compute_barrier(cmd); } -void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t width, - uint32_t height, uint32_t generations) { - if (!d || !d->built || d->unavailable) return; +// Hardware motion hint for the pair ending in slot: the newest frame's luminance goes to the +// GLES estimator and its field comes back as the starting candidate of the search on me_level. +// +// The estimator needs this frame's pixels, which the caller has only just recorded, so the +// command buffer is handed back through lush to be submitted and waited on first. What was +// recorded so far - the caller's composite and the copies above - then runs ahead of the rest +// of the frame; the caller submits the returned buffer for everything after. +static VkCommandBuffer dis_hardware_motion(VkrDis* d, VkCommandBuffer cmd, uint32_t slot, + VkrDisFlushFn flush, void* flush_user) { + const int32_t me_pc[2] = {(int32_t)d->me_w, (int32_t)d->me_h}; + vkd.CmdBindPipeline(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->pass_me_luma.pipeline); + vkd.CmdBindDescriptorSets(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->me_pipeline_layout, 0, 1, + &d->me_sets[slot], 0, NULL); + vkd.CmdPushConstants(cmd, d->me_pipeline_layout, VK_SHADER_STAGE_COMPUTE_BIT, 0, + sizeof(me_pc), me_pc); + vkd.CmdDispatch(cmd, (d->me_w / 4u + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, + (d->me_h + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, 1); + VkMemoryBarrier hb; + memset(&hb, 0, sizeof(hb)); + hb.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER; + hb.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT; + hb.dstAccessMask = VK_ACCESS_HOST_READ_BIT; + vkd.CmdPipelineBarrier(cmd, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_HOST_BIT, + 0, 1, &hb, 0, NULL, 0, NULL); + + cmd = flush(flush_user, cmd); + if (cmd == VK_NULL_HANDLE) return cmd; + + if (!d->me_luma_coherent) { + VkMappedMemoryRange r; + memset(&r, 0, sizeof(r)); + r.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE; + r.memory = d->me_luma_mem; + r.size = VK_WHOLE_SIZE; + vkd.InvalidateMappedMemoryRanges(d->device, 1, &r); + } + d->me_pairs++; + if (!dis_qcom_me_push(d->me, (const uint8_t*)d->me_luma_map, d->me_xy)) return cmd; + + // Pixels of the estimator's input -> normalised uv, which is what every flow image in the + // chain stores. Vectors the estimator could not have found - non-finite, or past half the + // frame - are marked invalid rather than clamped, so the search simply ignores them. + const uint32_t n = d->me_field_w * d->me_field_h; + float* dst = (float*)d->me_field_map; + const float inv_w = 1.0f / (float)d->me_w; + const float inv_h = 1.0f / (float)d->me_h; + for (uint32_t i = 0; i < n; i++) { + const float vx = d->me_xy[i * 2]; + const float vy = d->me_xy[i * 2 + 1]; + const bool ok = isfinite(vx) && isfinite(vy) && + fabsf(vx) < 0.5f * (float)d->me_w && fabsf(vy) < 0.5f * (float)d->me_h; + dst[i * 2] = ok ? vx * inv_w : 1.0e7f; + dst[i * 2 + 1] = ok ? vy * inv_h : 1.0e7f; + } + if (DIS_ME_PRIMARY) { + // As the level's result the field has no search behind it to reject a bad block, so + // outliers are taken out here: a 3x3 component median over the valid neighbours, and + // zero where there are none. + const int fw = (int)d->me_field_w, fh = (int)d->me_field_h; + float* med = d->me_xy; // reused: the raw pixels are no longer needed + for (int y = 0; y < fh; y++) { + for (int x = 0; x < fw; x++) { + for (int c = 0; c < 2; c++) { + float v[9]; + int k = 0; + for (int dy = -1; dy <= 1; dy++) { + for (int dx = -1; dx <= 1; dx++) { + const int xx = x + dx, yy = y + dy; + if (xx < 0 || yy < 0 || xx >= fw || yy >= fh) continue; + const float s = dst[(yy * fw + xx) * 2 + c]; + if (fabsf(s) < 1.0e6f) v[k++] = s; + } + } + for (int a = 1; a < k; a++) { + const float key = v[a]; + int b = a - 1; + while (b >= 0 && v[b] > key) { v[b + 1] = v[b]; b--; } + v[b + 1] = key; + } + med[(y * fw + x) * 2 + c] = k ? v[k / 2] : 0.0f; + } + } + } + memcpy(dst, med, (size_t)n * 2 * sizeof(float)); + } + if (!d->me_field_coherent) { + VkMappedMemoryRange r; + memset(&r, 0, sizeof(r)); + r.sType = VK_STRUCTURE_TYPE_MAPPED_MEMORY_RANGE; + r.memory = d->me_field_mem; + r.size = VK_WHOLE_SIZE; + vkd.FlushMappedMemoryRanges(d->device, 1, &r); + } + + dis_barrier(cmd, d->me_field.image, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_ACCESS_SHADER_READ_BIT, VK_ACCESS_TRANSFER_WRITE_BIT); + VkBufferImageCopy region; + memset(®ion, 0, sizeof(region)); + region.imageSubresource.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + region.imageSubresource.layerCount = 1; + region.imageExtent.width = d->me_field_w; + region.imageExtent.height = d->me_field_h; + region.imageExtent.depth = 1; + vkd.CmdCopyBufferToImage(cmd, d->me_field_buf, d->me_field.image, VK_IMAGE_LAYOUT_GENERAL, 1, + ®ion); + dis_barrier(cmd, d->me_field.image, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, + VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + VK_ACCESS_TRANSFER_WRITE_BIT, VK_ACCESS_SHADER_READ_BIT); + if (DIS_ME_PRIMARY) { + // The field is exactly the size of level me_level, so it drops into that mip of the + // refined flow, where the next finer level's search picks it up as its coarse estimate. + VkImageCopy ic; + memset(&ic, 0, sizeof(ic)); + ic.srcSubresource.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + ic.srcSubresource.layerCount = 1; + ic.dstSubresource.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT; + ic.dstSubresource.mipLevel = d->me_level; + ic.dstSubresource.layerCount = 1; + ic.extent.width = d->me_field_w; + ic.extent.height = d->me_field_h; + ic.extent.depth = 1; + dis_barrier(cmd, d->flow_refined.image, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, + VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, + VK_ACCESS_SHADER_READ_BIT, VK_ACCESS_TRANSFER_WRITE_BIT); + vkd.CmdCopyImage(cmd, d->me_field.image, VK_IMAGE_LAYOUT_GENERAL, d->flow_refined.image, + VK_IMAGE_LAYOUT_GENERAL, 1, &ic); + dis_barrier(cmd, d->flow_refined.image, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL, + VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, + VK_ACCESS_TRANSFER_WRITE_BIT, VK_ACCESS_SHADER_READ_BIT); + d->me_primary_frame = true; + } else { + d->hint_level = (int)d->me_level; + } + d->me_hinted++; + if ((d->me_hinted % 600u) == 1u) { + DIS_LOGI("DIS hardware motion hint: %llu of %llu pairs seeded", + (unsigned long long)d->me_hinted, (unsigned long long)d->me_pairs); + } + return cmd; +} + +VkCommandBuffer vkr_dis_process_ex(VkrDis* d, VkCommandBuffer cmd, VkImage source, + uint32_t width, uint32_t height, uint32_t generations, + VkrDisFlushFn flush, void* flush_user) { + if (!d || !d->built || d->unavailable) return cmd; d->last_generations = generations; + d->hint_level = -1; + d->me_primary_frame = false; dis_prime_layouts(d, cmd); @@ -1732,10 +2119,17 @@ void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t wi (lh + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, 1); } - dis_compute_barrier(cmd); + const bool wants_flow = generations > 0 || d->debug_flow; + if (d->me && flush && wants_flow) { + cmd = dis_hardware_motion(d, cmd, slot, flush, flush_user); + } else if (d->me) { + // Without this pair's estimate the held frame no longer precedes the next one. + dis_qcom_me_invalidate(d->me); + } - if (generations == 0 && !d->debug_flow) return; + dis_compute_barrier(cmd); + if (!wants_flow) return cmd; DisGradientPC gpc; gpc.lesser = 3.0f; gpc.upper = 10.0f; @@ -1756,6 +2150,8 @@ void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t wi for (uint32_t li = 0; li < L; li++) { const uint32_t l = coarse - li; + // The hardware field already stands in for this level and everything above it. + if (d->me_primary_frame && l >= d->me_level) continue; const uint32_t lw = w >> l; const uint32_t lh = h >> l; const uint32_t spw = dis_sparse_extent(lw); @@ -1764,6 +2160,7 @@ void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t wi DisInversePC ipc; ipc.level = (int)l; ipc.coarseLevel = (int)coarse; + ipc.hintLevel = d->hint_level; vkd.CmdBindPipeline(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->pass_inverse.pipeline); vkd.CmdBindDescriptorSets(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->pipeline_layout, 0, 1, &d->inverse_sets[slot][l], 0, NULL); @@ -1808,17 +2205,10 @@ void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t wi const uint32_t hist_dir = 1u - d->hist_parity; const int hist_reset = d->hist_valid ? 0 : 1; - // Temporal consistency of the field, before anything reads it. Shares the - // history parity with the overlay confidence below: both flip once per source - // pair, and hist_valid covers both. - vkd.CmdBindPipeline(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->pass_temporal.pipeline); - vkd.CmdBindDescriptorSets(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->pipeline_layout, 0, 1, - &d->temporal_sets[hist_dir], 0, NULL); - vkd.CmdPushConstants(cmd, d->pipeline_layout, VK_SHADER_STAGE_COMPUTE_BIT, 0, - sizeof(hist_reset), &hist_reset); - vkd.CmdDispatch(cmd, (w + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, - (h + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, 1); - dis_compute_barrier(cmd); + // The three per-pair products the interpolate pass reads are independent of each + // other - the packed field and the side map read the refined field, the overlay + // history reads the colour pair - so they share one barrier. + dis_dispatch(d, cmd, d->pass_pack.pipeline, d->pack_set, w, h); vkd.CmdBindPipeline(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->pass_hist.pipeline); vkd.CmdBindDescriptorSets(cmd, VK_PIPELINE_BIND_POINT_COMPUTE, d->vr_pipeline_layout, 0, 1, @@ -1827,15 +2217,19 @@ void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t wi sizeof(hist_reset), &hist_reset); vkd.CmdDispatch(cmd, (hist_w + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, (hist_h + DIS_LOCAL_SIZE - 1) / DIS_LOCAL_SIZE, 1); - dis_compute_barrier(cmd); d->hist_parity = hist_dir; d->hist_valid = true; - // Which real frame a true occlusion takes, one value per level-0 texel. + // Which real frame a true occlusion takes, and how sure, per level-0 texel. dis_dispatch(d, cmd, d->pass_side.pipeline, d->side_sets[slot], w, h); dis_compute_barrier(cmd); } + return cmd; +} +void vkr_dis_process(VkrDis* d, VkCommandBuffer cmd, VkImage source, uint32_t width, + uint32_t height, uint32_t generations) { + (void)vkr_dis_process_ex(d, cmd, source, width, height, generations, NULL, NULL); } static void dis_render_into(VkrDis* d, VkCommandBuffer cmd, float t, int debug_mode, @@ -1963,4 +2357,17 @@ void vkr_dis_reset(VkrDis* d) { d->plan_log_ns = 0; d->hist_parity = 0; d->hist_valid = false; + d->hint_level = -1; + if (d->me) dis_qcom_me_invalidate(d->me); +} + +void vkr_dis_set_hw_motion(VkrDis* d, bool enabled) { + if (!d || d->hw_motion == enabled) return; + d->hw_motion = enabled; + // Takes effect at the next resource build; force one. + d->built = false; +} + +bool vkr_dis_hw_motion_active(const VkrDis* d) { + return d && d->me != NULL; } diff --git a/app/src/main/cpp/waylandcomp/CMakeLists.txt b/app/src/main/cpp/waylandcomp/CMakeLists.txt index c15792839..e6f1d691c 100644 --- a/app/src/main/cpp/waylandcomp/CMakeLists.txt +++ b/app/src/main/cpp/waylandcomp/CMakeLists.txt @@ -43,14 +43,14 @@ target_include_directories(wnwayland PRIVATE ) # The X11 renderer's frame-generation engines run here too, on the compositor's own Turnip -# device. Their sources are compiled in rather than linked from libwinlator: both libraries keep -# a private VkDispatch, so the visibility below must stay for the two copies not to merge. +# device. LSFG is compiled in and DIS comes from the wndis static library rather than being linked +# from libwinlator: both libraries keep a private VkDispatch, so the visibility below must stay for +# the two copies not to merge. set(FRAMEGEN_DIR ${CMAKE_CURRENT_SOURCE_DIR}/../winlator/vk) -if (TARGET dxbc AND EXISTS ${FRAMEGEN_DIR}/dis/vkr_dis.c) +if (TARGET dxbc AND TARGET wndis) target_sources(wnwayland PRIVATE src/framegen_engine.c ${FRAMEGEN_DIR}/vk_dispatch.c - ${FRAMEGEN_DIR}/dis/vkr_dis.c ${FRAMEGEN_DIR}/lsfg/lsfg_dll.c ${FRAMEGEN_DIR}/lsfg/lsfg_dxbc.cpp ${FRAMEGEN_DIR}/lsfg/lsfg_common.cpp @@ -69,7 +69,7 @@ if (TARGET dxbc AND EXISTS ${FRAMEGEN_DIR}/dis/vkr_dis.c) target_compile_definitions(wnwayland PRIVATE VK_USE_PLATFORM_ANDROID_KHR) target_compile_options(wnwayland PRIVATE -fvisibility=hidden) target_compile_features(wnwayland PRIVATE cxx_std_17) - target_link_libraries(wnwayland dxbc) + target_link_libraries(wnwayland dxbc wndis) add_dependencies(wnwayland winlator_shaders) else() target_sources(wnwayland PRIVATE src/framegen_engine_stub.c) diff --git a/app/src/main/cpp/waylandcomp/src/framegen_engine.c b/app/src/main/cpp/waylandcomp/src/framegen_engine.c index 46af9101b..5573bf09c 100644 --- a/app/src/main/cpp/waylandcomp/src/framegen_engine.c +++ b/app/src/main/cpp/waylandcomp/src/framegen_engine.c @@ -3,7 +3,7 @@ #include "framegen_engine.h" #include "framegen_bridge.h" -#include "dis/vkr_dis.h" +#include "vkr_dis.h" #include "lsfg/vkr_lsfg.h" #include diff --git a/app/src/main/cpp/winlator/vk/framegen/fg_present.c b/app/src/main/cpp/winlator/vk/framegen/fg_present.c index 26c79a6d0..a563f00f5 100644 --- a/app/src/main/cpp/winlator/vk/framegen/fg_present.c +++ b/app/src/main/cpp/winlator/vk/framegen/fg_present.c @@ -15,7 +15,7 @@ #include "../vk_dispatch.h" #include "../vk_driver.h" -#include "../dis/vkr_dis.h" +#include "vkr_dis.h" #include "../lsfg/vkr_lsfg.h" #define LOG_TAG "FgPresent" @@ -87,6 +87,7 @@ struct FgPresenter { VkCommandPool command_pool; FgFrame frames[FG_FRAMES_IN_FLIGHT]; + VkFence flush_fence; uint32_t frame_index; FgTarget targets[FG_MAX_TARGETS]; @@ -540,6 +541,9 @@ static bool fg_create_frames(FgPresenter* fg) { cpi.queueFamilyIndex = fg->queue_family; if (vkCreateCommandPool(fg->device, &cpi, NULL, &fg->command_pool) != VK_SUCCESS) return false; + VkFenceCreateInfo ffi = {VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + if (vkCreateFence(fg->device, &ffi, NULL, &fg->flush_fence) != VK_SUCCESS) return false; + for (uint32_t i = 0; i < FG_FRAMES_IN_FLIGHT; i++) { FgFrame* f = &fg->frames[i]; @@ -804,6 +808,30 @@ static void fg_renew_semaphore(FgPresenter* fg, VkSemaphore* handle) { *handle = fresh; } +// VkrDisFlushFn: DIS needs this frame's pixels on the CPU mid-frame for the hardware motion +// estimator. Everything recorded so far - the source blit and DIS's copies - touches no +// swapchain image and waits on no semaphore, so it goes out on its own; the same buffer is then +// begun again for the rest of the frame, which keeps every semaphore on the final submit. +static VkCommandBuffer fg_dis_flush(void* user, VkCommandBuffer cmd) { + FgPresenter* fg = (FgPresenter*)user; + vkEndCommandBuffer(cmd); + VkSubmitInfo si = {VK_STRUCTURE_TYPE_SUBMIT_INFO}; + si.commandBufferCount = 1; + si.pCommandBuffers = &cmd; + vkResetFences(fg->device, 1, &fg->flush_fence); + if (vkQueueSubmit(fg->queue, 1, &si, fg->flush_fence) == VK_SUCCESS) { + vkWaitForFences(fg->device, 1, &fg->flush_fence, VK_TRUE, UINT64_MAX); + } else { + FG_LOGW("DIS mid-frame submit failed"); + vkDeviceWaitIdle(fg->device); + } + vkResetCommandBuffer(cmd, 0); + VkCommandBufferBeginInfo bi = {VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + bi.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + vkBeginCommandBuffer(cmd, &bi); + return cmd; +} + static void fg_record_and_present(FgPresenter* fg, FgImport* source, AImage* image) { FgFrame* f = &fg->frames[fg->frame_index]; @@ -879,8 +907,8 @@ static void fg_record_and_present(FgPresenter* fg, FgImport* source, AImage* ima VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT | VK_ACCESS_TRANSFER_READ_BIT); if (fg->active_engine == FG_ENGINE_DIS) { - vkr_dis_process(fg->dis, f->cmd, composite->image, fg->extent.width, fg->extent.height, - gen_count); + vkr_dis_process_ex(fg->dis, f->cmd, composite->image, fg->extent.width, + fg->extent.height, gen_count, fg_dis_flush, fg); } else if (fg->lsfg) { vkr_lsfg_process(fg->lsfg, f->cmd, composite->image, fg->extent.width, fg->extent.height, gen_count); @@ -1329,6 +1357,7 @@ void fg_destroy(FgPresenter* fg) { vkDestroySemaphore(fg->device, fg->retired[i], NULL); } fg->retired_count = 0; + if (fg->flush_fence) vkDestroyFence(fg->device, fg->flush_fence, NULL); if (fg->command_pool) vkDestroyCommandPool(fg->device, fg->command_pool, NULL); vkDestroyDevice(fg->device, NULL); } diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_interpolate.comp b/app/src/main/cpp/winlator/vk/shaders/dis_interpolate.comp deleted file mode 100644 index 465d211f0..000000000 --- a/app/src/main/cpp/winlator/vk/shaders/dis_interpolate.comp +++ /dev/null @@ -1,495 +0,0 @@ -// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) -// SPDX-License-Identifier: GPL-3.0-or-later -// -// DIS frame generation: a Vulkan compute realisation of Dense Inverse Search -// optical flow. The algorithm and its reference implementation come from -// OpenCV's DISOpticalFlow, which adopted Till Kroeger's original OF_DIS. -// See CREDITS.md for the full attribution. - -#version 450 - -precision highp float; -precision highp int; - -layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in; - -layout(set = 0, binding = 0) uniform sampler2D prevColor; -layout(set = 0, binding = 1) uniform sampler2D nextColor; -layout(set = 0, binding = 2) uniform sampler2D flowTex; -layout(set = 0, binding = 3) uniform sampler2D sideTex; -layout(set = 0, binding = 4) uniform sampler2D histTex; -layout(set = 0, binding = 5, rgba8) uniform image2D outImage; - -layout(push_constant) uniform PC { - float t; - int debugMode; -} pc; - -layout(constant_id = 0) const int manualFlowFilter = 0; - -// --------------------------------------------------------------------------- -// Static-overlay pass-through. -// -// A HUD, subtitles, a crosshair, a minimap or a letterbox bar does not move with the -// scene, but DIS has no notion of layers: under such an element the field carries the -// background's motion, so the warp drags the element around. Real frames show it in -// place, generated frames do not, and at 60 Hz that alternation is the flicker. -// -// The test needs no extra pass and no extra descriptor. One quantity decides it: -// -// rStatic = |prev(uv) - next(uv)| the two real frames, compared where they sit -// -// If the two real frames agree at a pixel, nothing happened there over the interval, so -// the value at any time between them is that same value and no warp can improve on it. -// The substitution is therefore correct by construction wherever it fires - including on -// a flat patch of moving scenery, where it simply writes the colour the warp would have -// produced anyway. That is why nothing gates it. -// -// An earlier revision multiplied this by "the field claims motion here" and by "the warp -// disagrees with itself". Both were wrong to include. The motion term ramped over 1..3 -// pixels, and under a wide HUD panel the field sags to a couple of pixels rather than to -// zero - so the mask opened halfway exactly where a two-pixel shift of crisp text reads -// as doubling. The disagreement term blocked the mask on low-contrast content, which is -// where the softness is worst. -// -// Cost on top of the existing shader: two full-res fetches, skipped only where the field -// cannot move anything at all, plus a handful of ALU. Nothing scales with the preset, so -// Fast pays the same as Quality. -// --------------------------------------------------------------------------- - -// 0.0 turns the whole block into dead code - the bisect switch. -const float UI_STRENGTH = 1.0; - -// Pure early-out: under a quarter of a pixel the warp cannot move anything, so the two -// fetches would be wasted. Deliberately far below the displacement at which doubling -// becomes visible - this is not a threshold on "is it moving". -const float UI_FLOW_SKIP_PX = 0.25; - -// Dilation. Inside an opaque panel the pair agrees and the pixel is pinned, but on an -// antialiased edge the pixel is alpha*glyph + (1-alpha)*background, the background under -// it moves, the pair disagrees and the mask shuts - so the outline keeps riding off with -// the field while the middle stands still. A pinned interior inside a twitching outline -// is what is left of the flicker. -// -// So a pixel also passes through when a neighbour's pair agrees: take the smallest pair -// difference over a small cross. The edge pixel is then frozen together with the glyph, -// which costs a thread of stale background one pixel wide - far less visible than an -// outline that moves every other frame. -// -// Measured on the model, error on the antialiased edge: 0.228 with no dilation, 0.175 at -// two taps, 0.149 at four taps over 1.5 px, 0.140 at 2.5 px. Background pays nothing up -// to 1.5 px (mask on the background hard against the panel 0.012) and starts to freeze at -// 2.5 px (0.112), so the cross stops at 1.5. -// -// 0, 2 or 4 taps; each tap is two full-res fetches. -// -// OFF, because dis_hist already does exactly this: it takes the minimum of the pair -// difference over a four-point cross and keeps it over time, and it runs ONCE PER SOURCE -// PAIR before the interpolate pass, at half resolution. The cross here recomputed the -// same thing at full resolution on every generated frame - three times per pair - and -// then handed the result to `max(uiMask, histStatic)`, where the history had it already. -// Eight of this pass's fetches for work that was done. The full-res cross could only add -// detail finer than half resolution, and hist's evidence is a max, so it reaches a newly -// appeared overlay in the same pair rather than a frame later. -const int UI_DILATE_TAPS = 0; -const float UI_DILATE_PX = 1.5; - -// Channel difference, 0..1. LO is rgb8 quantisation plus mild dither; HI is where a -// difference is unambiguously real content. -// -// Raise both if the game has visible grain or dithering - about three times sigma. -// -// A translucent HUD is the other reason to raise HI: there the pair never agrees, because -// the moving background shows through, and the difference is (1 - alpha) times the -// background's own. On the model with alpha = 0.7 the error inside such a panel goes -// 0.138 at HI = 0.05, 0.110 at 0.10, 0.081 at 0.20 - but background error goes 0.021, -// 0.031, 0.069 over the same steps. 0.10 is the point where the panel gains more than the -// scene loses; past that the scene starts freezing in earnest. -const float UI_LO = 0.012; -const float UI_HI = 0.050; - -// 0 off, 1 paints the mask green over a dimmed scene across the whole frame, 2 does it on -// the left half only so one build shows the mask and the finished picture side by side. -// The split is taken from the dispatch extent rather than from uv, because uv is -// normalised by imageSize and the two are not guaranteed to agree. -// -// What to expect now that the mask is ungated: solid green over the HUD and over every -// part of the scene that did not change between the two real frames, black only where -// something actually moved. A dark frame with green almost everywhere means the camera -// was still, which is correct - there the generated frame is meant to repeat the real one. -const int DEBUG_UI_MASK = 0; - -// Use the temporally stabilized overlay evidence from dis_hist.comp on top of -// the per-frame test above. 0 falls back to the per-frame test alone. -#define UI_HIST_ENABLE 1 - -// Pick the side for the blend from the sign of the flow divergence instead of -// from t: the side then stays the same for every generated frame of a pair, -// while t crosses 0.5 between them. -#define DIS_OCCL_SIDED 1 - -// Follow a parabola through three consecutive real frames instead of the straight -// chord of the current pair. 0 falls back to the chord. -#define DIS_QUADRATIC_PATH 1 - -// Fetch the warped samples through Catmull-Rom instead of one bilinear tap. 0 falls back -// to bilinear. This is the expensive switch in this file - see warpSample below. -#define DIS_WARP_CATMULL 1 - -// --------------------------------------------------------------------------- -// Averaging two warped samples that disagree is what softens the picture. Each sample is -// sharp on its own - one bilinear tap - but where the flow does not line them up the -// blend is a double image, and over a whole frame of wrong flow that reads as a global -// defocus. Where they disagree, take one sample instead of mixing. -// -// The old test summed the three channel differences and ramped over 0.10 .. 0.40, i.e. -// from a third of a step of grey - by then the blur is long since visible, and a purely -// coloured disagreement had to reach 0.10 in a single channel before it counted at all. -// Per-channel max over a lower band both fires earlier and treats colour properly. -// -// The cost of picking is judder: the chosen sample is the nearest real frame warped by a -// vector that was wrong, so the generated frame sits closer to a repeat of a real one. -// The band below is already back at the old behaviour; lower BLEND_LO if the picture -// reads too soft, raise it if the judder shows. -// --------------------------------------------------------------------------- -// Measured against the 68f678c1 blend (`smoothstep(0.10, 0.40, sum of channel diffs)`) -// at equal flow error, as the fraction of pixels that leave the average for a single -// frame. Soft content - sky, grass, fog, most of a frame - is the column that matters: -// -// band soft content: pick / sharpness vs reference -// 68f678c1 sum 0.10/0.40 0.207 - -// max 0.030/0.180 0.388 1.06 -// max 0.050/0.250 0.174 0.94 -// max 0.065/0.300 0.093 0.86 <- here -// max 0.080/0.350 0.048 0.81 -// max 0.100/0.450 0.016 0.77 -// -// Past 0.065/0.300 the picture starts paying real resolution for the smoothness, and by -// 0.100/0.450 mid-contrast content softens too (1.09 -> 0.89), not just the flat parts. -// -// Averaging two samples the flow failed to line up gives a soft frame, and softness -// reads as smoothness: there is nothing sharp for the eye to track, so the per-frame -// position error stops showing. On 68f678c1 that blur did two jobs at once - it hid the -// position error AND it hid the artifacts. The artifacts now have their own machinery -// (the overlay mask, dis_hist, dis_side, the flow clamp, dis_temporal, the parabola), so -// the blend no longer has to hide anything and can go back to the softer setting without -// bringing them back. The per-channel maximum is kept over the old sum: a purely coloured -// disagreement had to reach three times as far before the sum counted it at all. -const float BLEND_LO = 0.065; -const float BLEND_HI = 0.300; - -// --------------------------------------------------------------------------- -// Scene cut, and the same thing by another name: tracking that gave up. -// -// When the frame is replaced wholesale - a pause menu over gameplay - there is no -// correspondence to find. The search keeps whatever the residual happened to favour, the -// magnitude clamp in dis_vr_add bounds it to a quarter of the frame, and what is left is -// bounded garbage with the block structure of the search grid. Dragging a menu's own -// black panels and white text around with that field is what puts black and white squares -// on the screen. -// -// Two residuals decide it, and it works because it is TWO of them: -// -// |s0 - s1| the real pair, compared where it sits -// |c0 - c1| the same pair after the warp tried to reconcile it -// -// A fast pan pulls the first one far apart - median 0.290 in the runs below - but the warp -// closes it, so the second is 0.000. On a cut nothing closes it: 0.588 and 0.584. The -// minimum of the two separates the cases by an order of magnitude. Fraction of pixels -// above the threshold: -// -// case 0.25 0.35 0.45 -// static scene 0.000 0.000 0.000 -// pan, flow correct 0.035 0.009 0.001 -// pan, tracking lost 0.368 0.107 0.020 -// cut to a menu 0.992 0.936 0.745 -// -// This also retires the conclusion in dis-interpolation-glitches.md that a cut needs a -// global reduction to be told apart from lost tracking. It does - but the correct response -// to both is the same, show a real frame, so they never needed telling apart. -// -// The response is the nearest real frame UNWARPED. Warping it is what broke it. Over the -// three generated frames of a pair that reads as s0, s1, s1, i.e. a clean hold across the -// cut. Both residuals and both frames are already in registers here, so this costs no -// fetches at all. -const float CUT_LO = 0.30; -const float CUT_HI = 0.55; - -// xy: this pair's chord. zw: the previous pair's chord for the same content, advected -// here by dis_temporal - one fetch carries both, which is why the field is RGBA. -vec4 sampleFlow(vec2 uv) { - if (manualFlowFilter == 0) return textureLod(flowTex, uv, 0.0); - - vec2 sz = vec2(textureSize(flowTex, 0)); - vec2 p = uv * sz - 0.5; - vec2 frac = fract(p); - ivec2 i0 = ivec2(floor(p)); - ivec2 mx = ivec2(sz) - 1; - - vec4 a = texelFetch(flowTex, clamp(i0, ivec2(0), mx), 0); - vec4 b = texelFetch(flowTex, clamp(i0 + ivec2(1, 0), ivec2(0), mx), 0); - vec4 c = texelFetch(flowTex, clamp(i0 + ivec2(0, 1), ivec2(0), mx), 0); - vec4 e = texelFetch(flowTex, clamp(i0 + ivec2(1, 1), ivec2(0), mx), 0); - return mix(mix(a, b, frac.x), mix(c, e, frac.x), frac.y); -} - -float maxChannel(vec3 v) { - return max(max(v.x, v.y), v.z); -} - -// A real frame reaches the screen unresampled, one to one. A generated one is fetched at -// a fractional offset, and bilinear at half a texel throws away almost half the high -// frequencies. Measured against the unwarped frame, with a deliberately CORRECT flow so -// this is the filter's own cost and nothing else: -// -// fractional offset bilinear Catmull-Rom -// 0.000 1.000 1.000 -// 0.125 0.839 0.955 -// 0.250 0.699 0.844 -// 0.375 0.602 0.727 -// 0.500 0.564 0.675 -// -// So three of every four frames arrive softer than their neighbours by an amount that -// moves with the motion. That is a 30 Hz pulse of sharpness, and no amount of work on the -// flow touches it: the field can be exact and the filter still eats the detail. -// -// Catmull-Rom does not close the gap - no cheap filter invents what the sampling grid did -// not keep - but it moves the worst case from 0.564 to 0.675 and the common quarter-texel -// case from 0.699 to 0.844. -// -// Five bilinear taps, not nine: the four corners are products of two small negative lobes -// and carry nothing. Measured on a diagonal offset, sharpness against the unwarped frame -// and the mean difference between the two versions: -// -// fx fy bilinear 5 taps 9 taps |5-9| -// 0.500 0.500 0.453 0.584 0.577 0.0013 -// 0.250 0.250 0.584 0.780 0.774 0.0009 -// 0.125 0.125 0.756 0.933 0.930 0.0004 -// -// Five taps come out marginally AHEAD of nine once renormalised, and the difference -// between them is a third of an rgb8 step. DIS_WARP_CATMULL = 0 goes back to bilinear. -// -// The sampler is CLAMP_TO_EDGE, so the taps reaching one and a half texels past the -// border pick up the edge colour rather than black. -vec3 warpSample(sampler2D tex, vec2 uv, vec2 texSize) { -#if DIS_WARP_CATMULL == 0 - return textureLod(tex, clamp(uv, vec2(0.0), vec2(1.0)), 0.0).xyz; -#else - vec2 samplePos = uv * texSize; - vec2 texPos1 = floor(samplePos - 0.5) + 0.5; - vec2 fr = samplePos - texPos1; - - vec2 w0 = fr * (-0.5 + fr * (1.0 - 0.5 * fr)); - vec2 w1 = 1.0 + fr * fr * (-2.5 + 1.5 * fr); - vec2 w2 = fr * (0.5 + fr * (2.0 - 1.5 * fr)); - vec2 w3 = fr * fr * (-0.5 + 0.5 * fr); - - // The middle pair is taken as one bilinear tap placed between them - that is what - // turns sixteen taps into nine. w12 stays near 1 over the whole range, so the - // division is safe. - vec2 w12 = w1 + w2; - vec2 off12 = w2 / w12; - - vec2 inv = 1.0 / texSize; - vec2 p0 = clamp((texPos1 - 1.0) * inv, vec2(0.0), vec2(1.0)); - vec2 p3 = clamp((texPos1 + 2.0) * inv, vec2(0.0), vec2(1.0)); - vec2 p12 = clamp((texPos1 + off12) * inv, vec2(0.0), vec2(1.0)); - - vec3 r = vec3(0.0); - r += textureLod(tex, vec2(p12.x, p0.y), 0.0).xyz * (w12.x * w0.y); - r += textureLod(tex, vec2(p0.x, p12.y), 0.0).xyz * (w0.x * w12.y); - r += textureLod(tex, vec2(p12.x, p12.y), 0.0).xyz * (w12.x * w12.y); - r += textureLod(tex, vec2(p3.x, p12.y), 0.0).xyz * (w3.x * w12.y); - r += textureLod(tex, vec2(p12.x, p3.y), 0.0).xyz * (w12.x * w3.y); - - // Renormalise for the four dropped corners. Per axis w0 + w12 + w3 = 1 exactly, so - // the five remaining weights sum to this closed form - no extra adds needed. - r /= w12.x + w12.y * (1.0 - w12.x); - - // The negative lobes can ring past the source range. - return clamp(r, vec3(0.0), vec3(1.0)); -#endif -} - -// How far apart the two real frames are at one point, sampled where they sit. -float pairDiff(vec2 p) { - vec2 q = clamp(p, vec2(0.0), vec2(1.0)); - return maxChannel(abs(textureLod(prevColor, q, 0.0).xyz - textureLod(nextColor, q, 0.0).xyz)); -} - -vec3 hsv2rgb(vec3 c) { - vec4 K = vec4(1.0, 2.0 / 3.0, 1.0 / 3.0, 3.0); - vec3 p = abs(fract(c.xxx + K.xyz) * 6.0 - K.www); - return c.z * mix(K.xxx, clamp(p - K.xxx, 0.0, 1.0), c.y); -} - -void main() { - ivec2 pix = ivec2(gl_GlobalInvocationID.xy); - ivec2 size = imageSize(outImage); - if (pix.x >= size.x || pix.y >= size.y) return; - - vec2 uv = (vec2(pix) + 0.5) / vec2(size); - vec2 texel = 1.0 / vec2(size); - - const float GUARD_PX = 20.0; - vec2 guard = GUARD_PX * texel; - - vec2 dEdge = min(uv, 1.0 - uv); - vec2 ramp = clamp((dEdge - guard) / max(guard, texel), 0.0, 1.0); - vec2 eased = ramp * ramp * (3.0 - 2.0 * ramp); - float edgeMix = min(eased.x, eased.y); - - vec4 fs = sampleFlow(uv); - if (edgeMix < 1.0) { - vec4 fInner = sampleFlow(clamp(uv, guard, 1.0 - guard)); - fs = mix(fInner, fs, edgeMix); - } - // Last line of defence on the field. A non-finite value turns uv0/uv1 into NaN, and a - // NaN texture coordinate is an undefined fetch that reaches rgba8 as whatever the - // hardware felt like. The parabola can manufacture one from finite parts too: with - // fPrev infinite, (f + fPrev) and (f - fPrev) are +Inf and -Inf and their weighted sum - // is NaN. Upstream guards its own outputs; this is two instructions for the case where - // one of them is ever wrong again. - if (any(isnan(fs)) || any(isinf(fs))) fs = vec4(0.0); - - vec2 f = fs.xy; - // dis_temporal stores f itself in zw whenever it found no usable history, so the - // parabola below degenerates to the chord on its own in that case. - vec2 fPrev = DIS_QUADRATIC_PATH != 0 ? fs.zw : fs.xy; - - if (pc.debugMode != 0) { - float m = length(f) * float(size.x) / 16.0; - float hue = atan(f.y, f.x) / 6.2831853 + 0.5; - vec3 fc = hsv2rgb(vec3(hue, clamp(m, 0.0, 1.0), min(1.0, 0.15 + m))); - imageStore(outImage, pix, vec4(fc, 1.0)); - return; - } - - float flowPx = length(f * vec2(size)); - - // Under a quarter of a pixel the shift cannot change an output pixel visibly - - // the same premise the overlay pass-through uses when it skips its own fetches - // at this threshold. The warped pair, the side map and the edge dilation are - // then all dropped and the real pair is blended directly. Static scenery and - // HUD-heavy screens are largely made of such pixels, and this pass costs the - // most per output pixel in the chain. - if (flowPx <= UI_FLOW_SKIP_PX) { - vec3 e0 = textureLod(prevColor, uv, 0.0).xyz; - vec3 e1 = textureLod(nextColor, uv, 0.0).xyz; - vec3 still = mix(e0, e1, pc.t); - float disagree = maxChannel(abs(e0 - e1)); - float pick = smoothstep(BLEND_LO, BLEND_HI, disagree); - vec3 result = mix(still, pc.t < 0.5 ? e0 : e1, pick); - - float uiMask = UI_STRENGTH * (1.0 - smoothstep(UI_LO, UI_HI, disagree)); -#if UI_HIST_ENABLE - float histStatic = textureLod(histTex, uv, 0.0).r; - histStatic *= 1.0 - smoothstep(0.30, 0.60, disagree); - uiMask = max(uiMask, histStatic); -#endif - result = mix(result, still, uiMask); - - bool leftHalfE = uint(pix.x) * 2u < gl_NumWorkGroups.x * gl_WorkGroupSize.x; - if (DEBUG_UI_MASK == 1 || (DEBUG_UI_MASK == 2 && leftHalfE)) { - result = mix(result * 0.15, vec3(0.1, 1.0, 0.2), uiMask); - } - imageStore(outImage, pix, vec4(result, 1.0)); - return; - } - - // Displacement from the previous real frame to time t. A chord is straight, so its - // direction changes abruptly at every real frame and that break repeats at the source - // rate; 30 Hz of it is a large part of why 120 made from 30 does not read as 120. The - // parabola through three consecutive real frames removes it. Measured against a true - // path: position error 0.469 -> 0.000 under constant acceleration, 3.027 -> 0.450 on a - // smooth arc, 12.34 -> 6.07 on a mouse flick, with the velocity break falling the same - // way. On a uniform pan a chord is already exact and the parabola matches it. - // - // A reversal would make a parabola overshoot, but by then dis_temporal's gate has set - // fPrev = f, so those frames fall back to the chord by construction. - vec2 d = 0.5 * pc.t * (f + fPrev) + 0.5 * pc.t * pc.t * (f - fPrev); - vec2 uv0 = uv - d; - vec2 uv1 = uv + (f - d); - - vec3 c0 = warpSample(prevColor, uv0, vec2(size)); - vec3 c1 = warpSample(nextColor, uv1, vec2(size)); - - vec3 single = pc.t < 0.5 ? c0 : c1; -#if DIS_OCCL_SIDED - // The side choice is precomputed per flow texel by dis_side.comp, so this is - // one tap instead of four flow fetches and it does not change with t. - single = textureLod(sideTex, uv, 0.0).r > 0.5 ? c1 : c0; -#endif - - const float FEATHER_PX = 8.0; - vec2 feather = FEATHER_PX / vec2(size); - vec2 e0 = max(max(-uv0, uv0 - vec2(1.0)), vec2(0.0)) / feather; - vec2 e1 = max(max(-uv1, uv1 - vec2(1.0)), vec2(0.0)) / feather; - float out0 = clamp(max(e0.x, e0.y), 0.0, 1.0); - float out1 = clamp(max(e1.x, e1.y), 0.0, 1.0); - - float w0 = (1.0 - pc.t) * (1.0 - out0); - float w1 = pc.t * (1.0 - out1); - float wsum = w0 + w1; - - vec3 result = wsum > 1e-4 - ? (c0 * w0 + c1 * w1) / wsum - : (out0 <= out1 ? c0 : c1); - - float disagree = maxChannel(abs(c0 - c1)); - float pick = smoothstep(BLEND_LO, BLEND_HI, disagree); - result = mix(result, single, pick); - - // --- static-overlay pass-through --------------------------------------- - // The unwarped pair is fetched unconditionally now: the cut gate of the - // stabilized mask below needs its difference even where the flow early-out - // would have skipped the per-frame test. - vec3 s0 = textureLod(prevColor, uv, 0.0).xyz; - vec3 s1 = textureLod(nextColor, uv, 0.0).xyz; - float pairDiffHere = maxChannel(abs(s0 - s1)); - - float uiMask = 0.0; - vec3 uiColor = mix(s0, s1, pc.t); - - if (UI_STRENGTH > 0.0 && flowPx > UI_FLOW_SKIP_PX) { - float rStatic = pairDiffHere; - - if (UI_DILATE_TAPS > 0) { - vec2 d = UI_DILATE_PX * texel; - rStatic = min(rStatic, pairDiff(uv + vec2(d.x, 0.0))); - rStatic = min(rStatic, pairDiff(uv - vec2(d.x, 0.0))); - if (UI_DILATE_TAPS > 2) { - rStatic = min(rStatic, pairDiff(uv + vec2(0.0, d.y))); - rStatic = min(rStatic, pairDiff(uv - vec2(0.0, d.y))); - } - } - - uiMask = UI_STRENGTH * (1.0 - smoothstep(UI_LO, UI_HI, rStatic)); - } - -#if UI_HIST_ENABLE - // The half-resolution history keeps the same decision over several frames, so - // a per-frame test whose value hovers around the thresholds cannot flicker. - // A large current difference means a cut or a scene change, where the old - // element is gone: suppress the history there rather than holding it. - float histStatic = textureLod(histTex, uv, 0.0).r; - histStatic *= 1.0 - smoothstep(0.30, 0.60, pairDiffHere); - uiMask = max(uiMask, histStatic); -#endif - - result = mix(result, uiColor, uiMask); - - // --- scene cut / lost tracking ----------------------------------------- - // Last, so it overrides the warp, the pick and the overlay mask alike. On a cut the - // overlay mask is shut anyway - rStatic is huge - and dis_hist has its own cut gate, - // so there is nothing here to fight with. - float cut = smoothstep(CUT_LO, CUT_HI, min(pairDiffHere, disagree)); - result = mix(result, pc.t < 0.5 ? s0 : s1, cut); - - bool leftHalf = uint(pix.x) * 2u < gl_NumWorkGroups.x * gl_WorkGroupSize.x; - if (DEBUG_UI_MASK == 1 || (DEBUG_UI_MASK == 2 && leftHalf)) { - result = mix(result * 0.15, vec3(0.1, 1.0, 0.2), uiMask); - } - - imageStore(outImage, pix, vec4(result, 1.0)); -} diff --git a/app/src/main/cpp/winlator/vk/shaders/dis_temporal.comp b/app/src/main/cpp/winlator/vk/shaders/dis_temporal.comp deleted file mode 100644 index 5bf4a328d..000000000 --- a/app/src/main/cpp/winlator/vk/shaders/dis_temporal.comp +++ /dev/null @@ -1,128 +0,0 @@ -// SPDX-FileCopyrightText: Copyright 2026 qwertypower (DEVAR Entertainment LLC) -// SPDX-License-Identifier: GPL-3.0-or-later -// -// DIS frame generation: a Vulkan compute realisation of Dense Inverse Search -// optical flow. The algorithm and its reference implementation come from -// OpenCV's DISOpticalFlow, which adopted Till Kroeger's original OF_DIS. -// See CREDITS.md for the full attribution. - -#version 450 - -precision highp float; -precision highp int; - -// Temporal consistency of the final field, at the flow resolution, once per -// source pair. -// -// Every pair is estimated from scratch - the coarsest level starts from zero - -// so the estimation error is redrawn independently each time. A still picture -// never shows it: the field is about as accurate as before. What it costs is -// smoothness, because the generated frames sit at pos + t*f, and a field that -// jitters by a few pixels between pairs moves them back and forth around the -// true path. The eye reads that as a parasitic acceleration at the source rate. -// -// Modelled over long runs (displayed position vs the true path, jitter measured -// as the spread of the second difference): -// -// scenario sigma raw EMA 0.3 + gate -// uniform pan 6 px 3.88 1.82 -// uniform pan 15 px 9.36 4.48 -// hard ramp 40->160 over 20 pairs 15 px 9.34 5.29 -// camera reversal +-120 6 px 3.85 1.93 -// scene cut 6 px 3.95 1.89 -// -// Jitter roughly halves and the position error drops by a third. It wins on the -// ramp and the reversal too: the field's genuine change per pair is smaller than -// the estimation noise, so the lag costs less than the jitter. -// -// THE GATE IS NOT OPTIONAL. Without it the same runs get worse, not better: -// 3.85 -> 6.88 on the reversal and 3.95 -> 11.49 on the cut, because the history -// is then averaged into a field that has nothing to do with it any more. -// -// Advection: the history at this texel describes content that has since moved -// on, so it is read where that content came from, one dependent fetch. At a -// disocclusion there is no valid history at all, and that is the same gate. -// -// Cost: one dispatch at the flow resolution per source pair - 320x180 on Fast - -// and two flow-resolution images, RGBA32F because zw carries the previous chord for -// the interpolate pass's parabola. Nothing per output pixel beyond widening one fetch -// that pass already made, so the preset does not notice. - -// Weight of the new measurement. Higher keeps more of the new field. -// -// 0.3 measured best when this pass stood alone. It does not any more, and the reason is -// worth writing down: the interpolate pass now follows a parabola through three real -// frames, and that parabola lives on the SECOND DIFFERENCE of the field - how the chord -// changed between pairs. An EMA is exactly what erases a second difference. The two -// features want opposite things from the same signal. -// -// Jitter over long runs, re-measured with the parabola in place: -// -// scenario a=0.30 0.35 0.45 0.55 0.70 1.00 -// pan 2.28 2.45 2.78 3.10 3.60 4.69 -// arc 12.15 11.61 9.88 8.43 6.60 5.22 -// ramp 4.32 4.05 3.85 3.96 4.27 5.17 -// reversal 2.36 2.52 2.84 3.16 3.63 4.61 -// -// The arc is ordinary camera movement and its numbers dwarf the rest, so it decides the -// feel; heavy smoothing costs it more than it gains anywhere else. Summed, the minimum -// sits around 0.6-0.7. 0.55 is a step short of it on purpose: real straight-line motion -// still benefits from noise suppression, and the sigma those runs assume is an estimate, -// so the headroom is worth more on that side. -#define DIS_TEMPORAL_ALPHA 0.55 - -// Drop the history when the new field departs from it by more than this. The -// floor covers a still field; the relative term keeps the gate meaningful when -// the whole frame moves by a hundred pixels. -#define DIS_TEMPORAL_GATE_PX 8.0 -#define DIS_TEMPORAL_GATE_REL 0.35 - -layout(local_size_x = 8, local_size_y = 8, local_size_z = 1) in; - -layout(set = 0, binding = 0) uniform sampler2D flowNew; -layout(set = 0, binding = 1) uniform sampler2D flowHist; -layout(set = 0, binding = 5, rgba32f) uniform image2D flowOut; - -layout(push_constant) uniform PC { - int reset; -} pc; - -void main() { - ivec2 p = ivec2(gl_GlobalInvocationID.xy); - ivec2 sz = imageSize(flowOut); - if (p.x >= sz.x || p.y >= sz.y) return; - - vec2 uSize = vec2(sz); - vec2 uv = (vec2(p) + 0.5) / uSize; - - // Stored in normalised uv units, as the rest of the chain expects. - vec2 fNew = texelFetch(flowNew, p, 0).xy; - - if (pc.reset != 0) { - imageStore(flowOut, p, vec4(fNew, fNew)); - return; - } - - vec2 back = clamp(uv - fNew, vec2(0.0), vec2(1.0)); - vec2 fOld = textureLod(flowHist, back, 0.0).xy; - - float gate = DIS_TEMPORAL_GATE_PX + DIS_TEMPORAL_GATE_REL * length(fNew * uSize); - bool usable = length((fNew - fOld) * uSize) <= gate; - - vec2 fOut = usable ? mix(fOld, fNew, DIS_TEMPORAL_ALPHA) : fNew; - - // zw is the previous pair's chord for this same content, which the interpolate pass - // needs for its parabola. Writing fNew where there is no usable history is what makes - // that parabola degenerate to the chord instead of inventing curvature - and it is the - // same condition, so a reversal or a cut is covered once, here. - vec2 fPrev = usable ? fOld : fNew; - - // Zero, not fNew: fNew is the value that may BE the non-finite one, and falling back - // to it makes the guard a no-op. That is exactly the mistake the old vr_add guard made - // when it fell back to W. A zero field degrades to repeating the real frame, which is - // the right answer whenever the flow has nothing to say. - if (any(isnan(fOut)) || any(isinf(fOut))) fOut = vec2(0.0); - if (any(isnan(fPrev)) || any(isinf(fPrev))) fPrev = vec2(0.0); - - imageStore(flowOut, p, vec4(fOut, fPrev)); -} diff --git a/app/src/main/cpp/winlator/vk/vk_dispatch.c b/app/src/main/cpp/winlator/vk/vk_dispatch.c index 77335b211..8983daf9a 100644 --- a/app/src/main/cpp/winlator/vk/vk_dispatch.c +++ b/app/src/main/cpp/winlator/vk/vk_dispatch.c @@ -81,6 +81,7 @@ bool vkd_load_instance(VkInstance instance) { LOAD(MapMemory); LOAD(UnmapMemory); LOAD(FlushMappedMemoryRanges); + LOAD(InvalidateMappedMemoryRanges); LOAD(GetAndroidHardwareBufferPropertiesANDROID); // Buffer diff --git a/app/src/main/cpp/winlator/vk/vk_dispatch.h b/app/src/main/cpp/winlator/vk/vk_dispatch.h index 4e7072c2f..ed5a1de11 100644 --- a/app/src/main/cpp/winlator/vk/vk_dispatch.h +++ b/app/src/main/cpp/winlator/vk/vk_dispatch.h @@ -57,6 +57,7 @@ typedef struct VkDispatch { PFN_vkMapMemory MapMemory; PFN_vkUnmapMemory UnmapMemory; PFN_vkFlushMappedMemoryRanges FlushMappedMemoryRanges; + PFN_vkInvalidateMappedMemoryRanges InvalidateMappedMemoryRanges; PFN_vkGetAndroidHardwareBufferPropertiesANDROID GetAndroidHardwareBufferPropertiesANDROID; // Buffer @@ -206,6 +207,7 @@ bool vkd_bind_proc(PFN_vkGetInstanceProcAddr gipa, VkInstance instance); #define vkMapMemory vkd.MapMemory #define vkUnmapMemory vkd.UnmapMemory #define vkFlushMappedMemoryRanges vkd.FlushMappedMemoryRanges +#define vkInvalidateMappedMemoryRanges vkd.InvalidateMappedMemoryRanges #define vkGetAndroidHardwareBufferPropertiesANDROID vkd.GetAndroidHardwareBufferPropertiesANDROID #define vkCreateBuffer vkd.CreateBuffer diff --git a/app/src/main/cpp/winlator/vk/vk_renderer.c b/app/src/main/cpp/winlator/vk/vk_renderer.c index 9755849b5..1e5a8138a 100644 --- a/app/src/main/cpp/winlator/vk/vk_renderer.c +++ b/app/src/main/cpp/winlator/vk/vk_renderer.c @@ -1923,6 +1923,10 @@ static void destroy_dis(VkRenderer* r) { if (!r->dis) return; vkr_dis_destroy(r->dis); r->dis = NULL; + if (r->dis_flush_fence) { + vkDestroyFence(r->device, r->dis_flush_fence, NULL); + r->dis_flush_fence = VK_NULL_HANDLE; + } r->framegen_real_frames = 0; r->framegen_made_frames = 0; r->framegen_draw_ns = 0; @@ -1931,6 +1935,39 @@ static void destroy_dis(VkRenderer* r) { r->framegen_timed_frames = 0; } +// VkrDisFlushFn: DIS needs this frame's pixels on the CPU mid-frame for the hardware motion +// estimator. What has been recorded by then is the scene pass into the composite target and +// DIS's own copies - no swapchain image, no semaphore - so it is submitted on its own and the +// same buffer is begun again; acquire waits and present signals all stay on the frame's submit. +static VkCommandBuffer dis_flush_frame(void* user, VkCommandBuffer cmd) { + VkRenderer* r = (VkRenderer*)user; + if (!r->dis_flush_fence) { + VkFenceCreateInfo fci = {VK_STRUCTURE_TYPE_FENCE_CREATE_INFO}; + if (vkCreateFence(r->device, &fci, NULL, &r->dis_flush_fence) != VK_SUCCESS) { + r->dis_flush_fence = VK_NULL_HANDLE; + } + } + vkEndCommandBuffer(cmd); + VkSubmitInfo si = {VK_STRUCTURE_TYPE_SUBMIT_INFO}; + si.commandBufferCount = 1; + si.pCommandBuffers = &cmd; + if (r->dis_flush_fence) vkResetFences(r->device, 1, &r->dis_flush_fence); + pthread_mutex_lock(&r->queue_mutex); + VkResult sr = vkQueueSubmit(r->graphics_queue, 1, &si, r->dis_flush_fence); + if (sr != VK_SUCCESS || !r->dis_flush_fence) vkQueueWaitIdle(r->graphics_queue); + pthread_mutex_unlock(&r->queue_mutex); + if (sr == VK_SUCCESS && r->dis_flush_fence) { + vkWaitForFences(r->device, 1, &r->dis_flush_fence, VK_TRUE, UINT64_MAX); + } else if (sr != VK_SUCCESS) { + VK_LOGW("DIS mid-frame submit -> %d", (int)sr); + } + vkResetCommandBuffer(cmd, 0); + VkCommandBufferBeginInfo bi = {VK_STRUCTURE_TYPE_COMMAND_BUFFER_BEGIN_INFO}; + bi.flags = VK_COMMAND_BUFFER_USAGE_ONE_TIME_SUBMIT_BIT; + vkBeginCommandBuffer(cmd, &bi); + return cmd; +} + static void create_dis(VkRenderer* r) { if (r->dis || !r->device || !r->physical_device) return; @@ -2885,8 +2922,9 @@ static bool record_and_submit_frame(VkRenderer* r) { if (composite) { if ((use_dis || r->lsfg) && framegen_capacity > 0) { if (use_dis) { - vkr_dis_process(r->dis, f->cmd, composite->image, - composite->width, composite->height, gen_count); + vkr_dis_process_ex(r->dis, f->cmd, composite->image, + composite->width, composite->height, gen_count, + dis_flush_frame, r); } else { vkr_lsfg_process(r->lsfg, f->cmd, composite->image, r->swapchain_extent.width, r->swapchain_extent.height, gen_count); diff --git a/app/src/main/cpp/winlator/vk/vk_state.h b/app/src/main/cpp/winlator/vk/vk_state.h index cf8f70cce..7bcba786e 100644 --- a/app/src/main/cpp/winlator/vk/vk_state.h +++ b/app/src/main/cpp/winlator/vk/vk_state.h @@ -15,7 +15,7 @@ // this translation unit (do not include directly). #include "vk_dispatch.h" #include "lsfg/vkr_lsfg.h" -#include "dis/vkr_dis.h" +#include "vkr_dis.h" #define VK_LOG_TAG "VkRenderer" #define VK_LOGI(...) __android_log_print(ANDROID_LOG_INFO, VK_LOG_TAG, __VA_ARGS__) @@ -426,6 +426,7 @@ typedef struct VkRenderer { struct VkrDis* dis; bool dis_requested; uint32_t dis_scale; + VkFence dis_flush_fence; // mid-frame submit for DIS's hardware motion hint uint32_t dis_target_fps; bool dis_debug_flow; uint64_t sgsr1_dbg_sig; diff --git a/docs/FRAME-GENERATION.md b/docs/FRAME-GENERATION.md index 1bb568623..8f375b5e4 100644 --- a/docs/FRAME-GENERATION.md +++ b/docs/FRAME-GENERATION.md @@ -29,7 +29,7 @@ the shaders import successfully. ## DIS **Nothing to buy, nothing to import.** DIS is a complete open-source Dense Inverse Search frame -generator built into WinNative — fourteen compute shaders that ship with the APK — so it works on +generator built into WinNative — sixteen compute shaders that ship with the APK — so it works on a fresh install with no Steam account and no `Lossless.dll`. Turn it on in the **FG** tab and it runs.