Files
L'électron rare ee7c290b1c feat: add CUDA engine for GPU remap, gain blur, nvenc
Adds gpu_engine.py behind --engine cuda: uploads remap maps, runs
remap and the spatial-gain blur on the device via cv2.cuda, and adds
a piped nvenc encoder. Falls back to the CPU path when the OpenCV
wheel has no CUDA kernels (CudaUnavailable). Seam/DIS paths and final
blend stay on CPU for parity
2026-08-07 20:20:38 +02:00

232 lines
9.5 KiB
Python

"""CUDA-accelerated core for the X5 stitching pipeline.
Replaces the hottest CPU stages of x5_pipeline.py with cv2.cuda kernels
while keeping the exact same numpy semantics at the boundary, so the CPU
and GPU paths are interchangeable at the call site (PSNR gate valid):
- remap_fisheye -> CudaEngine.remap (cv2.cuda.remap)
- compute_spatial_gain -> CudaEngine.spatial_gain
(sigma=80 blurs on GPU)
The seam-band logic (DIS flow, find_seam_dp, multiband blend) deliberately
stays on the CPU: those paths are narrow (~30 deg) and DIS has no CUDA
kernel in OpenCV.
CUDA availability: the stock pip 'opencv-contrib-python' wheel ships the
cv2.cuda *framework* (GpuMat/Stream) but NO kernels, so importing this
module never fails; the engine refuses to start on such a build with a
precise diagnostic (the Phase-1 gate of the GPU plan).
"""
from __future__ import annotations
import numpy as np
import cv2
class CudaUnavailable(RuntimeError):
"""The engine was asked for but this OpenCV build has no CUDA kernels."""
def _missing_cuda_kernels() -> list[str]:
"""cv2.cuda symbols that the loaded OpenCV build does not expose."""
if not hasattr(cv2, "cuda"):
return ["cuda"]
want = ["GpuMat", "remap", "GaussianBlur", "Stream"]
missing = [n for n in want if getattr(cv2.cuda, n, None) is None]
if hasattr(cv2, "cuda"):
if not hasattr(cv2.cuda.GpuMat, "convertTo"):
missing.append("GpuMat.convertTo")
if not hasattr(cv2.cuda.GpuMat, "copyTo"):
missing.append("GpuMat.copyTo")
return missing
class CudaEngine:
"""Persistent GPU working set + faithful wrappers over CUDA kernels.
One engine per pipeline instance. Remap tables are uploaded once per
map build; frame sources are uploaded per frame and produced as a
(u8, f32) pair to feed both the blend stage and the CPU seam code.
GPU buffers are reused across frames (BufferPool); the default (null)
stream is used, and every download synchronises, matching the
existing sequential CPU decode/band model.
"""
def __init__(self, eq_w: int, eq_h: int) -> None:
self.eq_w, self.eq_h = eq_w, eq_h
missing = _missing_cuda_kernels()
if missing:
raise CudaUnavailable(
"CUDA kernels missing from this OpenCV build: "
+ ", ".join(missing)
+ "\nInstall an OpenCV built with CUDA_ENABLED=ON and "
"re-run (see Phase 2 of the GPU plan).")
cv2.cuda.setBufferPoolUsage(True)
self.stream = cv2.cuda.Stream()
# Smoke-test remap on a tiny buffer so a shadow-glob-gated build
# fails here, not 30 000 frames into a take. Also pick the best
# remap interpolation this build accepts (CUDA builds differ on
# cubic support): cubic first, linear as a safe fallback.
try:
probe = np.zeros((16, 16, 3), dtype=np.uint8)
mx = np.zeros((16, 16), dtype=np.float32)
gsrc = CudaEngine._upload(probe, np.uint8)
gm = CudaEngine._upload(mx)
self._interp = None
for interp in (cv2.INTER_CUBIC, cv2.INTER_LINEAR):
out = cv2.cuda.GpuMat()
try:
cv2.cuda.remap(gsrc, gm, gm, interp, out)
_ = out.download()
self._interp = interp
break
except cv2.error:
continue
if self._interp is None:
raise CudaUnavailable(
"cv2.cuda.remap rejected both INTER_CUBIC and "
"INTER_LINEAR; no usable remap kernel in this build.")
except CudaUnavailable:
raise
except Exception as exc: # noqa: BLE001
raise CudaUnavailable(
"cv2.cuda.remap smoke test failed although kernels are "
"present: " + str(exc)) from exc
# Persistent buffers (alloc'd once, reused across frames).
self._eq_u8 = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
self._eq_f32 = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
self._mapsx = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
self._mapsy = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
self._validm = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
self._scratch = cv2.cuda.GpuMat()
self._maps_uploaded = False
# ------------------------------------------------------------------
# upload / download helpers
# ------------------------------------------------------------------
@staticmethod
def _upload(arr, dtype=None):
if dtype is None:
dtype = arr.dtype
m = cv2.cuda.GpuMat()
m.upload(np.ascontiguousarray(arr, dtype=dtype))
return m
@staticmethod
def _from(m):
return m.download()
# ------------------------------------------------------------------
# maps / valid-mask upload (call once per map build)
# ------------------------------------------------------------------
def upload_maps(self, maps):
"""Upload remap tables + valid zeroing-masks for the current maps.
maps: list[(map_x, map_y, valid)] for front and back lens.
valid is True where the remap output is legitimate; the mask is
stored as 0/255 u8 so it doubles as a device multipler.
"""
for i, (mx, my, valid) in enumerate(maps):
self._mapsx[i] = CudaEngine._upload(mx, np.float32)
self._mapsy[i] = CudaEngine._upload(my, np.float32)
# u8 valid mask for device zeroing via copyTo (0 = drop).
self._validm[i] = CudaEngine._upload(
valid.astype(np.uint8) * 255, np.uint8)
self._maps_uploaded = True
# ------------------------------------------------------------------
# remap
# ------------------------------------------------------------------
def remap(self, src_bgr_u8, lens_idx):
"""Mirror of remap_fisheye().
Returns (u8, f32) eq frames: the u8 output pixel-exactly matches
remap_fisheye() (invalid pixels zeroed), and the f32 copy avoids
an extra CPU conversion for the gain stage.
"""
if not self._maps_uploaded:
raise RuntimeError("CudaEngine.upload_maps() must run first")
g_src = CudaEngine._upload(src_bgr_u8, np.uint8)
g_dst = self._eq_u8[lens_idx]
self._scratch.create(g_dst.size(), g_dst.type())
cv2.cuda.remap(g_src, self._mapsx[lens_idx], self._mapsy[lens_idx],
self._interp, self._scratch, stream=self.stream)
# Zero invalid pixels on device: copyTo(dst, mask) keeps only the
# valid-masked pixels — same semantics as result[~valid] = 0.
g_dst.create(self._scratch.size(), self._scratch.type())
self._scratch.copyTo(g_dst, self._validm[lens_idx])
out_u8 = CudaEngine._from(g_dst)
g_f = self._eq_f32[lens_idx]
g_dst.convertTo(g_f, cv2.CV_32FC3, stream=self.stream)
out_f32 = CudaEngine._from(g_f)
return out_u8, out_f32
# ------------------------------------------------------------------
# gaussian blur on device
# ------------------------------------------------------------------
def blur32(self, cpu_rw, sigma):
"""cv2.GaussianBlur(..., (0, 0), sigma) on a (H, W) float32 array.
The heavy blurred pair of spatial_gain runs here; callers keep the
mask/ratio work on numpy so the algorithm is bit-compatible.
cv2.GaussianBlur with ksize=(0, 0) resolves its kernel from sigma
as ((round(sigma*6))|1); we pass that kernel explicitly so this
matches the CPU pathway down to the last neighbor.
"""
ksize = int(round(sigma * 6)) | 1
g = CudaEngine._upload(cpu_rw, np.float32)
b = cv2.cuda.GpuMat(g.size(), g.type())
cv2.cuda.GaussianBlur(g, (ksize, ksize), float(sigma), b,
sigmaY=float(sigma),
borderType=cv2.BORDER_REFLECT101,
stream=self.stream)
return CudaEngine._from(b)
# ------------------------------------------------------------------
# spatial gain (sigma=80 blurs on device)
# ------------------------------------------------------------------
def spatial_gain(self, eq_front, eq_back, overlap_mask, blur_sigma=80.0):
"""Mirror of compute_spatial_gain().
eq_front/eq_back: (H, W, 3) float32 arrays in 0..255.
Returns (H, W, 3) float32 gain field clipped to 0.5..2.0. The
per-pixel ratio/weight logic runs exactly like the CPU version;
the two heavy blur operations run on the GPU.
"""
eq_w, eq_h = self.eq_w, self.eq_h
gain = np.empty((eq_h, eq_w, 3), dtype=np.float32)
for c in range(3):
f = eq_front[:, :, c]
b = eq_back[:, :, c]
valid = overlap_mask & (b > 10) & (f > 5)
ratio = np.ones((eq_h, eq_w), dtype=np.float32)
if valid.any():
ratio_safe = ratio.copy()
ratio_safe[valid] = np.divide(f[valid], b[valid])
ratio = ratio_safe
ratio = np.clip(ratio, 0.3, 3.0)
weight = valid.astype(np.float32)
ratio_w = ratio * weight
ratio_blurred = self.blur32(ratio_w, blur_sigma)
weight_blurred = self.blur32(weight, blur_sigma)
weight_blurred = np.maximum(weight_blurred, 1e-6)
gain[:, :, c] = ratio_blurred / weight_blurred
return np.clip(gain, 0.5, 2.0)