Adds gpu_engine.py behind --engine cuda: uploads remap maps, runs remap and the spatial-gain blur on the device via cv2.cuda, and adds a piped nvenc encoder. Falls back to the CPU path when the OpenCV wheel has no CUDA kernels (CudaUnavailable). Seam/DIS paths and final blend stay on CPU for parity
232 lines
9.5 KiB
Python
232 lines
9.5 KiB
Python
"""CUDA-accelerated core for the X5 stitching pipeline.
|
|
|
|
Replaces the hottest CPU stages of x5_pipeline.py with cv2.cuda kernels
|
|
while keeping the exact same numpy semantics at the boundary, so the CPU
|
|
and GPU paths are interchangeable at the call site (PSNR gate valid):
|
|
|
|
- remap_fisheye -> CudaEngine.remap (cv2.cuda.remap)
|
|
- compute_spatial_gain -> CudaEngine.spatial_gain
|
|
(sigma=80 blurs on GPU)
|
|
|
|
The seam-band logic (DIS flow, find_seam_dp, multiband blend) deliberately
|
|
stays on the CPU: those paths are narrow (~30 deg) and DIS has no CUDA
|
|
kernel in OpenCV.
|
|
|
|
CUDA availability: the stock pip 'opencv-contrib-python' wheel ships the
|
|
cv2.cuda *framework* (GpuMat/Stream) but NO kernels, so importing this
|
|
module never fails; the engine refuses to start on such a build with a
|
|
precise diagnostic (the Phase-1 gate of the GPU plan).
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import numpy as np
|
|
import cv2
|
|
|
|
|
|
class CudaUnavailable(RuntimeError):
|
|
"""The engine was asked for but this OpenCV build has no CUDA kernels."""
|
|
|
|
|
|
def _missing_cuda_kernels() -> list[str]:
|
|
"""cv2.cuda symbols that the loaded OpenCV build does not expose."""
|
|
if not hasattr(cv2, "cuda"):
|
|
return ["cuda"]
|
|
want = ["GpuMat", "remap", "GaussianBlur", "Stream"]
|
|
missing = [n for n in want if getattr(cv2.cuda, n, None) is None]
|
|
if hasattr(cv2, "cuda"):
|
|
if not hasattr(cv2.cuda.GpuMat, "convertTo"):
|
|
missing.append("GpuMat.convertTo")
|
|
if not hasattr(cv2.cuda.GpuMat, "copyTo"):
|
|
missing.append("GpuMat.copyTo")
|
|
return missing
|
|
|
|
|
|
class CudaEngine:
|
|
"""Persistent GPU working set + faithful wrappers over CUDA kernels.
|
|
|
|
One engine per pipeline instance. Remap tables are uploaded once per
|
|
map build; frame sources are uploaded per frame and produced as a
|
|
(u8, f32) pair to feed both the blend stage and the CPU seam code.
|
|
|
|
GPU buffers are reused across frames (BufferPool); the default (null)
|
|
stream is used, and every download synchronises, matching the
|
|
existing sequential CPU decode/band model.
|
|
"""
|
|
|
|
def __init__(self, eq_w: int, eq_h: int) -> None:
|
|
self.eq_w, self.eq_h = eq_w, eq_h
|
|
|
|
missing = _missing_cuda_kernels()
|
|
if missing:
|
|
raise CudaUnavailable(
|
|
"CUDA kernels missing from this OpenCV build: "
|
|
+ ", ".join(missing)
|
|
+ "\nInstall an OpenCV built with CUDA_ENABLED=ON and "
|
|
"re-run (see Phase 2 of the GPU plan).")
|
|
|
|
cv2.cuda.setBufferPoolUsage(True)
|
|
self.stream = cv2.cuda.Stream()
|
|
|
|
# Smoke-test remap on a tiny buffer so a shadow-glob-gated build
|
|
# fails here, not 30 000 frames into a take. Also pick the best
|
|
# remap interpolation this build accepts (CUDA builds differ on
|
|
# cubic support): cubic first, linear as a safe fallback.
|
|
try:
|
|
probe = np.zeros((16, 16, 3), dtype=np.uint8)
|
|
mx = np.zeros((16, 16), dtype=np.float32)
|
|
gsrc = CudaEngine._upload(probe, np.uint8)
|
|
gm = CudaEngine._upload(mx)
|
|
self._interp = None
|
|
for interp in (cv2.INTER_CUBIC, cv2.INTER_LINEAR):
|
|
out = cv2.cuda.GpuMat()
|
|
try:
|
|
cv2.cuda.remap(gsrc, gm, gm, interp, out)
|
|
_ = out.download()
|
|
self._interp = interp
|
|
break
|
|
except cv2.error:
|
|
continue
|
|
if self._interp is None:
|
|
raise CudaUnavailable(
|
|
"cv2.cuda.remap rejected both INTER_CUBIC and "
|
|
"INTER_LINEAR; no usable remap kernel in this build.")
|
|
except CudaUnavailable:
|
|
raise
|
|
except Exception as exc: # noqa: BLE001
|
|
raise CudaUnavailable(
|
|
"cv2.cuda.remap smoke test failed although kernels are "
|
|
"present: " + str(exc)) from exc
|
|
|
|
# Persistent buffers (alloc'd once, reused across frames).
|
|
self._eq_u8 = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
|
|
self._eq_f32 = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
|
|
self._mapsx = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
|
|
self._mapsy = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
|
|
self._validm = [cv2.cuda.GpuMat(), cv2.cuda.GpuMat()]
|
|
self._scratch = cv2.cuda.GpuMat()
|
|
self._maps_uploaded = False
|
|
|
|
# ------------------------------------------------------------------
|
|
# upload / download helpers
|
|
# ------------------------------------------------------------------
|
|
|
|
@staticmethod
|
|
def _upload(arr, dtype=None):
|
|
if dtype is None:
|
|
dtype = arr.dtype
|
|
m = cv2.cuda.GpuMat()
|
|
m.upload(np.ascontiguousarray(arr, dtype=dtype))
|
|
return m
|
|
|
|
@staticmethod
|
|
def _from(m):
|
|
return m.download()
|
|
|
|
# ------------------------------------------------------------------
|
|
# maps / valid-mask upload (call once per map build)
|
|
# ------------------------------------------------------------------
|
|
|
|
def upload_maps(self, maps):
|
|
"""Upload remap tables + valid zeroing-masks for the current maps.
|
|
|
|
maps: list[(map_x, map_y, valid)] for front and back lens.
|
|
valid is True where the remap output is legitimate; the mask is
|
|
stored as 0/255 u8 so it doubles as a device multipler.
|
|
"""
|
|
for i, (mx, my, valid) in enumerate(maps):
|
|
self._mapsx[i] = CudaEngine._upload(mx, np.float32)
|
|
self._mapsy[i] = CudaEngine._upload(my, np.float32)
|
|
# u8 valid mask for device zeroing via copyTo (0 = drop).
|
|
self._validm[i] = CudaEngine._upload(
|
|
valid.astype(np.uint8) * 255, np.uint8)
|
|
self._maps_uploaded = True
|
|
|
|
# ------------------------------------------------------------------
|
|
# remap
|
|
# ------------------------------------------------------------------
|
|
|
|
def remap(self, src_bgr_u8, lens_idx):
|
|
"""Mirror of remap_fisheye().
|
|
|
|
Returns (u8, f32) eq frames: the u8 output pixel-exactly matches
|
|
remap_fisheye() (invalid pixels zeroed), and the f32 copy avoids
|
|
an extra CPU conversion for the gain stage.
|
|
"""
|
|
if not self._maps_uploaded:
|
|
raise RuntimeError("CudaEngine.upload_maps() must run first")
|
|
|
|
g_src = CudaEngine._upload(src_bgr_u8, np.uint8)
|
|
g_dst = self._eq_u8[lens_idx]
|
|
self._scratch.create(g_dst.size(), g_dst.type())
|
|
cv2.cuda.remap(g_src, self._mapsx[lens_idx], self._mapsy[lens_idx],
|
|
self._interp, self._scratch, stream=self.stream)
|
|
# Zero invalid pixels on device: copyTo(dst, mask) keeps only the
|
|
# valid-masked pixels — same semantics as result[~valid] = 0.
|
|
g_dst.create(self._scratch.size(), self._scratch.type())
|
|
self._scratch.copyTo(g_dst, self._validm[lens_idx])
|
|
out_u8 = CudaEngine._from(g_dst)
|
|
g_f = self._eq_f32[lens_idx]
|
|
g_dst.convertTo(g_f, cv2.CV_32FC3, stream=self.stream)
|
|
out_f32 = CudaEngine._from(g_f)
|
|
return out_u8, out_f32
|
|
|
|
# ------------------------------------------------------------------
|
|
# gaussian blur on device
|
|
# ------------------------------------------------------------------
|
|
|
|
def blur32(self, cpu_rw, sigma):
|
|
"""cv2.GaussianBlur(..., (0, 0), sigma) on a (H, W) float32 array.
|
|
|
|
The heavy blurred pair of spatial_gain runs here; callers keep the
|
|
mask/ratio work on numpy so the algorithm is bit-compatible.
|
|
|
|
cv2.GaussianBlur with ksize=(0, 0) resolves its kernel from sigma
|
|
as ((round(sigma*6))|1); we pass that kernel explicitly so this
|
|
matches the CPU pathway down to the last neighbor.
|
|
"""
|
|
ksize = int(round(sigma * 6)) | 1
|
|
g = CudaEngine._upload(cpu_rw, np.float32)
|
|
b = cv2.cuda.GpuMat(g.size(), g.type())
|
|
cv2.cuda.GaussianBlur(g, (ksize, ksize), float(sigma), b,
|
|
sigmaY=float(sigma),
|
|
borderType=cv2.BORDER_REFLECT101,
|
|
stream=self.stream)
|
|
return CudaEngine._from(b)
|
|
|
|
# ------------------------------------------------------------------
|
|
# spatial gain (sigma=80 blurs on device)
|
|
# ------------------------------------------------------------------
|
|
|
|
def spatial_gain(self, eq_front, eq_back, overlap_mask, blur_sigma=80.0):
|
|
"""Mirror of compute_spatial_gain().
|
|
|
|
eq_front/eq_back: (H, W, 3) float32 arrays in 0..255.
|
|
Returns (H, W, 3) float32 gain field clipped to 0.5..2.0. The
|
|
per-pixel ratio/weight logic runs exactly like the CPU version;
|
|
the two heavy blur operations run on the GPU.
|
|
"""
|
|
eq_w, eq_h = self.eq_w, self.eq_h
|
|
gain = np.empty((eq_h, eq_w, 3), dtype=np.float32)
|
|
|
|
for c in range(3):
|
|
f = eq_front[:, :, c]
|
|
b = eq_back[:, :, c]
|
|
|
|
valid = overlap_mask & (b > 10) & (f > 5)
|
|
ratio = np.ones((eq_h, eq_w), dtype=np.float32)
|
|
if valid.any():
|
|
ratio_safe = ratio.copy()
|
|
ratio_safe[valid] = np.divide(f[valid], b[valid])
|
|
ratio = ratio_safe
|
|
ratio = np.clip(ratio, 0.3, 3.0)
|
|
|
|
weight = valid.astype(np.float32)
|
|
ratio_w = ratio * weight
|
|
|
|
ratio_blurred = self.blur32(ratio_w, blur_sigma)
|
|
weight_blurred = self.blur32(weight, blur_sigma)
|
|
weight_blurred = np.maximum(weight_blurred, 1e-6)
|
|
|
|
gain[:, :, c] = ratio_blurred / weight_blurred
|
|
|
|
return np.clip(gain, 0.5, 2.0) |