perf(av-live): pyobjc + GPU MediaPipe + queues

Three perf optimizations stacked on top of the distributed pipeline:

1. multihmr_server.py ported to pyobjc CoreML.framework direct
   (drops the ~30 ms coremltools.MLModel.predict overhead). Compiles
   the mlpackage to .mlmodelc on load, then predicts via
   MLDictionaryFeatureProvider + MLMultiArray ctypes memcpy. Toggle
   MULTIHMR_SERVER_BACKEND=coremltools for fallback. setup script
   installs pyobjc-framework-CoreML on the macm1 venv.

2. MediaPipe Holistic now uses GPU Metal delegate on M5. Required
   wrapping camera frames as SRGBA (4-channel) instead of SRGB --
   the Metal CVPixelBuffer path rejects 3-channel formats. Bench
   M5 standalone : pose 6.7 -> 2.9 ms, face 4.0 -> 1.0 ms, hand
   6.1 -> 3.2 ms. Frees ~10 ms CPU per frame for OSC + rigger.

3. Remote client queues bumped maxsize 1 -> 2/3 (in/out) to absorb
   jitter without stalling capture.
This commit is contained in:
L'électron rare
2026-05-14 02:17:29 +02:00
parent 9838da3137
commit 67302e7ca2
5 changed files with 150 additions and 12 deletions
+1
View File
@@ -20,6 +20,7 @@ Python **3.11+** requis. `pyproject.toml` est la source de vérité — ne jamai
| Backend | Fichier | Statut |
|---------|---------|--------|
| MediaPipe Holistic | `holistic.py` | stable |
| MediaPipe multi (Pose+Face+Hand) | `multi.py` | stable ; `MEDIAPIPE_DELEGATE=gpu` (défaut) ou `cpu`. **GPU Metal exige SRGBA 4-ch** (3-ch SRGB crashe `gpu_buffer_storage_cv_pixel_buffer.cc`) — multi.py route auto vers `cv2.COLOR_BGR2RGBA` + `mp.ImageFormat.SRGBA` quand delegate=GPU. Bench M5 image-mode SRGBA : pose 2.9 vs 6.7 ms (GPU/CPU), face 1.0 vs 4.1, hand 3.2 vs 6.1 |
| Ultralytics YOLOv8-pose | `pose.py` | stable, modèle `yolov8n-pose.pt` à la racine repo |
| Apple Vision (Core ML) | `apple_vision_pose.py`, `coreml_pose.py` | macOS uniquement |
| DETRPose | `detrpose.py` | clone manuel + checkpoint, voir docstring |
+12 -2
View File
@@ -358,8 +358,18 @@ class MultiWorker:
time.sleep(self.period)
continue
h, w = frame_bgr.shape[:2]
frame_rgb = cv2.cvtColor(frame_bgr, cv2.COLOR_BGR2RGB)
mp_img = mp.Image(image_format=mp.ImageFormat.SRGB, data=frame_rgb)
# MediaPipe GPU delegate on macOS uploads via CVPixelBuffer
# which only accepts 4-channel formats. SRGB (3ch) crashes
# in gpu_buffer_storage_cv_pixel_buffer.cc with
# "unsupported ImageFrame format: 1". Use SRGBA when on GPU.
if _deleg == BaseOptions.Delegate.GPU:
frame_rgba = cv2.cvtColor(frame_bgr, cv2.COLOR_BGR2RGBA)
mp_img = mp.Image(image_format=mp.ImageFormat.SRGBA,
data=frame_rgba)
else:
frame_rgb = cv2.cvtColor(frame_bgr, cv2.COLOR_BGR2RGB)
mp_img = mp.Image(image_format=mp.ImageFormat.SRGB,
data=frame_rgb)
ts = int(time.monotonic() * 1000) - t0_ms
try:
pose_res = pose.detect_for_video(mp_img, ts)
+4 -2
View File
@@ -235,11 +235,13 @@ class MultiHMRRemoteBackend:
self.use_async = _env_flag("MULTIHMR_REMOTE_ASYNC", True)
# Async pipeline state.
# Multi-buffer queues (2 in / 3 out) absorb jitter without
# stalling capture. Drop-oldest semantics on overflow.
self._in_q: queue.Queue[tuple[bytes, float, float]] = queue.Queue(
maxsize=1)
maxsize=2)
self._out_q: queue.Queue[
tuple[list[dict[str, Any]], dict[str, float]]
] = queue.Queue(maxsize=1)
] = queue.Queue(maxsize=3)
self._stop = threading.Event()
self._async_det_thresh = 0.3
self._worker_thread: threading.Thread | None = None
+131 -7
View File
@@ -150,17 +150,109 @@ def decode_request(payload: bytes) -> tuple[np.ndarray, np.ndarray, float]:
return img, K, decode_ms
ML_DTYPE_FLOAT16 = 65552
ML_DTYPE_FLOAT32 = 65568
ML_DTYPE_DOUBLE = 65600
def _np_to_mlarray(arr: np.ndarray, MLMultiArray):
"""Create a contiguous float32 MLMultiArray from a numpy array."""
import ctypes
arr = np.ascontiguousarray(arr, dtype=np.float32)
shape = [int(s) for s in arr.shape]
ml = MLMultiArray.alloc().initWithShape_dataType_error_(
shape, ML_DTYPE_FLOAT32, None)
if ml is None:
raise RuntimeError("MLMultiArray alloc failed")
ptr = ml.dataPointer()
addr = int(ptr) if isinstance(ptr, int) else ctypes.cast(
ptr, ctypes.c_void_p).value
if addr is None:
raise RuntimeError("MLMultiArray dataPointer null")
ctypes.memmove(addr, arr.ctypes.data, arr.nbytes)
return ml
def _mlarray_to_np(ml) -> np.ndarray:
"""Copy an MLMultiArray (FLOAT16/32/64) to numpy float32."""
import ctypes
shape = tuple(int(s) for s in ml.shape())
dtype_id = int(ml.dataType())
count = 1
for s in shape:
count *= s
ptr = ml.dataPointer()
addr = int(ptr) if isinstance(ptr, int) else ctypes.cast(
ptr, ctypes.c_void_p).value
if addr is None:
raise RuntimeError("MLMultiArray dataPointer null")
if dtype_id == ML_DTYPE_FLOAT16:
raw = (ctypes.c_uint16 * count).from_address(addr)
arr = np.ctypeslib.as_array(raw).view(np.float16).astype(np.float32)
elif dtype_id == ML_DTYPE_FLOAT32:
raw = (ctypes.c_float * count).from_address(addr)
arr = np.ctypeslib.as_array(raw).copy()
elif dtype_id == ML_DTYPE_DOUBLE:
raw = (ctypes.c_double * count).from_address(addr)
arr = np.ctypeslib.as_array(raw).astype(np.float32)
else:
raise RuntimeError(f"unsupported MLMultiArray dtype {dtype_id}")
return arr.reshape(shape)
class CoreMLModel:
"""Thin wrapper around coremltools.MLModel for server-side use."""
"""pyobjc-direct CoreML wrapper. Drops the ~30 ms coremltools.MLModel.predict
overhead by using CoreML.framework directly (MLDictionaryFeatureProvider
+ MLMultiArray ctypes memcpy). Fallback to coremltools if pyobjc missing,
via MULTIHMR_SERVER_BACKEND=coremltools env."""
def __init__(self, mlpackage_path: Path) -> None:
import coremltools as ct
from coremltools.models import MLModel
self.path = Path(mlpackage_path)
if not self.path.exists():
raise FileNotFoundError(f"mlpackage missing: {self.path}")
backend = os.environ.get(
"MULTIHMR_SERVER_BACKEND", "pyobjc").strip().lower()
cu_env = os.environ.get(
"COREML_COMPUTE_UNITS", "cpu_and_gpu").strip().lower()
if backend == "pyobjc":
self._use_pyobjc = True
self._init_pyobjc(cu_env)
else:
self._use_pyobjc = False
self._init_coremltools(cu_env)
def _init_pyobjc(self, cu_env: str) -> None:
import objc
from Foundation import NSURL
ns: dict = {}
objc.loadBundle("CoreML", ns,
"/System/Library/Frameworks/CoreML.framework")
cu_map = {"cpu_only": 0, "cpu_and_gpu": 1, "all": 2,
"cpu_and_ne": 3}
cu = cu_map.get(cu_env, 1)
MLModel = ns["MLModel"]
MLModelConfiguration = ns["MLModelConfiguration"]
cfg = MLModelConfiguration.alloc().init()
try:
cfg.setComputeUnits_(cu)
except Exception: # noqa: BLE001
pass
url = NSURL.fileURLWithPath_(str(self.path))
compiled_url = MLModel.compileModelAtURL_error_(url, None)
if compiled_url is None:
raise RuntimeError(f"compileModelAtURL failed for {self.path}")
model = MLModel.modelWithContentsOfURL_configuration_error_(
compiled_url, cfg, None)
if model is None:
raise RuntimeError(f"MLModel load failed for {compiled_url}")
self._model = model
self._ns = ns
LOG.info("loading mlpackage %s via pyobjc (computeUnit=%s)",
self.path.name, cu_env)
def _init_coremltools(self, cu_env: str) -> None:
import coremltools as ct
from coremltools.models import MLModel as CTMLModel
cu_map = {
"cpu_only": ct.ComputeUnit.CPU_ONLY,
"cpu_and_gpu": ct.ComputeUnit.CPU_AND_GPU,
@@ -168,9 +260,9 @@ class CoreMLModel:
"cpu_and_ne": ct.ComputeUnit.CPU_AND_NE,
}
cu = cu_map.get(cu_env, ct.ComputeUnit.CPU_AND_GPU)
LOG.info("loading mlpackage %s (computeUnit=%s)",
LOG.info("loading mlpackage %s via coremltools (computeUnit=%s)",
self.path.name, cu_env)
self.model = MLModel(str(self.path), compute_units=cu)
self.model = CTMLModel(str(self.path), compute_units=cu)
def predict(self, image_uint8_hwc: np.ndarray, K_33: np.ndarray
) -> dict[str, np.ndarray]:
@@ -179,8 +271,40 @@ class CoreMLModel:
K = K_33.astype(np.float32)
if K.ndim == 2:
K = K[np.newaxis, ...]
feats = {"image": img4, "cam_K": K}
return self.model.predict(feats)
if self._use_pyobjc:
return self._predict_pyobjc(img4, K)
return self.model.predict({"image": img4, "cam_K": K})
def _predict_pyobjc(self, image_4d: np.ndarray, K_33: np.ndarray
) -> dict[str, np.ndarray]:
ns = self._ns
MLMultiArray = ns["MLMultiArray"]
MLDictionaryFeatureProvider = ns["MLDictionaryFeatureProvider"]
MLFeatureValue = ns["MLFeatureValue"]
img_ml = _np_to_mlarray(image_4d, MLMultiArray)
k_ml = _np_to_mlarray(K_33, MLMultiArray)
feats = {
"image": MLFeatureValue.featureValueWithMultiArray_(img_ml),
"cam_K": MLFeatureValue.featureValueWithMultiArray_(k_ml),
}
provider = MLDictionaryFeatureProvider.alloc(
).initWithDictionary_error_(feats, None)
if provider is None:
raise RuntimeError("MLDictionaryFeatureProvider alloc failed")
out = self._model.predictionFromFeatures_error_(provider, None)
if out is None:
raise RuntimeError("MLModel predict failed")
names = [str(n) for n in out.featureNames()]
result: dict[str, np.ndarray] = {}
for n in names:
fv = out.featureValueForName_(n)
if fv is None:
continue
ml = fv.multiArrayValue()
if ml is None:
continue
result[n] = _mlarray_to_np(ml)
return result
def _zero_outputs() -> tuple[np.ndarray, ...]:
+2 -1
View File
@@ -46,7 +46,8 @@ ssh "$HOST" "bash -lc 'set -e
uv venv --python 3.12 $REMOTE_VENV --quiet
fi
uv pip install --python $REMOTE_VENV/bin/python --quiet \
coremltools numpy opencv-python-headless
coremltools numpy opencv-python-headless \
pyobjc-core pyobjc-framework-Cocoa pyobjc-framework-CoreML
'"
echo "==> Killing any stale server on :$PORT"