Files
L'électron rare e6eabc0526 feat: add content-prep tooling scripts
Replace the unusable prepare-content.py draft (untracked, never
committed) with two composable, argument-driven CLIs and no
hardcoded absolute paths:

- sync-offset.py cross-correlates the H3-VR W channel against a
  camera reference track (extracted from video via ffmpeg, or a
  pre-extracted WAV) to recover the audio/video sync offset. Ported
  from a validated standalone script; keeps the full-mode
  correlate()+correlation_lags() approach and its reasoning, since
  'valid'-mode index arithmetic was a real source of sign bugs.
- prepare-scene.py builds one scene into public/content/scenes/<id>
  (trim/pad all 4 channels identically, transcode video to a web
  bitrate, extract a thumbnail, verify channel count/rate/duration)
  and upserts it into the manifest without touching other entries.

Ignore scripts/__pycache__/ generated by running these.
2026-08-07 14:20:17 +02:00

262 lines
10 KiB
Python
Executable File

#!/usr/bin/env python3
"""Measure the sync offset between a Zoom H3-VR ambiX WAV and camera audio.
Usage:
uv run --with numpy,scipy python scripts/sync-offset.py \\
--h3vr H3-VR.WAV --video stitched-equirect.mp4
uv run --with numpy,scipy python scripts/sync-offset.py \\
--h3vr H3-VR.WAV --camera-audio camera-embedded.wav
Requires numpy and scipy. This is a standalone content-prep tool, not
a dependency of the shipped player — run it with `uv run --with
numpy,scipy python scripts/sync-offset.py ...` so uv resolves an
ephemeral environment; no project venv needed.
Cross-correlates the H3-VR's W channel (channel 0 of the ambiX WAV —
the recorder outputs ACN order W,Y,Z,X, confirmed via bext/iXML
metadata `Rec Mode=AmbiX`, TRACK_LIST=W,Y,Z,X) against a camera
reference audio track. If --video is given, that reference is
extracted automatically via ffmpeg from the video's embedded audio
track. On Insta360 cameras that embedded track (`Ambarella AAC`,
4-channel, balanced per-channel RMS) is a discrete mic array, NOT
encoded ambisonics — it is usable only as a sync witness here, never
as spatial audio for the player (see README "Preparing real
content").
Sign convention (validated against a synthetic ground-truth signal
with a known +3.0s offset — exact recovery):
offset_s = the position IN the H3-VR file that corresponds to
video/camera time 0.
offset_s > 0 => H3-VR started recording BEFORE the camera
(H3 has a head start); to align, TRIM offset_s
seconds off the FRONT of the H3-VR wav.
offset_s < 0 => H3-VR started AFTER the camera; to align, PAD
|offset_s| seconds of silence onto the FRONT of
the H3-VR wav.
This is exactly the convention `prepare-scene.py --offset` expects.
Implementation note: this only ever uses scipy.signal.correlate(...,
mode='full') together with scipy.signal.correlation_lags(). An
earlier version used 'valid'-mode correlation with manual index
arithmetic to recover the lag, which has a non-obvious sign/offset
convention that depends on which signal is longer — that was a real
source of bugs (silently wrong offsets that still "looked plausible").
'full' mode + correlation_lags() gives the lag directly from scipy,
with no index-juggling left to get wrong.
"""
import argparse
import subprocess
import sys
import tempfile
from pathlib import Path
import numpy as np
from scipy.io import wavfile
from scipy.signal import correlate, correlation_lags
def load_mono(path):
sr, data = wavfile.read(path)
if data.ndim > 1:
data = data[:, 0]
data = data.astype(np.float64)
maxv = np.max(np.abs(data)) + 1e-9
data = data / maxv
return sr, data
def envelope(x, sr, target_sr=200):
block = max(1, sr // target_sr)
n = len(x) // block
x = x[: n * block].reshape(n, block)
env = np.sqrt(np.mean(x**2, axis=1) + 1e-12)
return env, sr / block
def full_corr_offset_samples(a, b):
"""Return d (samples) such that b[n+d] ~= a[n], via full+lags method."""
a0 = a - a.mean()
b0 = b - b.mean()
corr = correlate(a0, b0, mode="full")
lags = correlation_lags(len(a0), len(b0), mode="full")
lag = lags[np.argmax(corr)]
return int(-lag)
def pearson_at_offset(cam, h3, start_cam, window_n, d_samples):
"""Pearson r between cam[start_cam:start_cam+window_n] and the
corresponding h3 segment shifted by d_samples."""
end_cam = start_cam + window_n
h3_0 = start_cam + d_samples
h3_1 = h3_0 + window_n
if start_cam < 0 or end_cam > len(cam) or h3_0 < 0 or h3_1 > len(h3):
return None
a = cam[start_cam:end_cam]
b = h3[h3_0:h3_1]
return np.corrcoef(a, b)[0, 1]
def coarse_offset(cam, sr, h3):
env_cam, esr = envelope(cam, sr)
env_h3, _ = envelope(h3, sr)
d_env = full_corr_offset_samples(env_cam, env_h3)
return d_env / esr, esr
def refine_offset_local(cam, sr, h3, d_guess_samples, center_pos_s,
window_s=40.0, search_s=1.0):
"""Refine using a big window centered at center_pos_s (cam-time),
full corr against a widened h3 segment (window + search margin on
each side), then read back the exact offset."""
win_n = int(window_s * sr)
search_n = int(search_s * sr)
start_cam = max(0, int(center_pos_s * sr) - win_n // 2)
end_cam = min(len(cam), start_cam + win_n)
seg_cam = cam[start_cam:end_cam]
h3_start = max(0, start_cam + d_guess_samples - search_n)
h3_end = min(len(h3), end_cam + d_guess_samples + search_n)
seg_h3 = h3[h3_start:h3_end]
if len(seg_h3) <= len(seg_cam) or len(seg_cam) == 0:
return None
d_local = full_corr_offset_samples(seg_cam, seg_h3)
d_abs = (h3_start + d_local) - start_cam
r = pearson_at_offset(cam, h3, start_cam, len(seg_cam), d_abs)
return d_abs, r, start_cam, len(seg_cam)
def windowed_check(cam, sr, h3, d_guess_samples, positions_s,
window_s=15.0, search_s=0.5):
results = []
for pos in positions_s:
res = refine_offset_local(cam, sr, h3, d_guess_samples,
pos + window_s / 2,
window_s=window_s, search_s=search_s)
if res is None:
results.append(None)
continue
d_abs, r, _start_cam, _n = res
if r is None:
results.append(None)
continue
results.append((pos, d_abs / sr, r))
return results
def extract_camera_audio(video_path, out_dir):
"""ffmpeg-extract the video's embedded audio track to a WAV in
out_dir (a caller-managed temp directory — never a hardcoded path)."""
out_wav = out_dir / "camera-audio.wav"
subprocess.run(
["ffmpeg", "-y", "-v", "error", "-i", str(video_path),
"-vn", "-acodec", "pcm_s16le", "-ar", "48000", str(out_wav)],
check=True,
)
return out_wav
def analyze(cam_path, h3_path, window_s, search_s):
sr_cam, cam = load_mono(cam_path)
sr_h3, h3 = load_mono(h3_path)
if sr_cam != sr_h3:
sys.exit(f"error: sample rate mismatch (camera {sr_cam} Hz vs "
f"H3-VR {sr_h3} Hz)")
sr = sr_cam
d_coarse_s, esr = coarse_offset(cam, sr, h3)
print(f"coarse offset (envelope @ ~{esr:.1f} Hz): {d_coarse_s:.4f} s")
d_guess_samples = int(round(d_coarse_s * sr))
dur_cam_s = len(cam) / sr
center = dur_cam_s / 2
res = refine_offset_local(cam, sr, h3, d_guess_samples, center,
window_s=min(40.0, dur_cam_s * 0.6),
search_s=1.0)
if res is None:
sys.exit("error: refine failed (out of range) — files may not "
"overlap at all")
d_abs, r, _start_cam, _n = res
offset_s = d_abs / sr
print(f"offset_s = {offset_s:+.4f} s (pearson r={r:.4f})")
if offset_s > 0:
print(f" H3-VR started first -> TRIM {offset_s:.4f}s off the "
f"front of the H3-VR wav")
else:
print(f" H3-VR started later -> PAD {abs(offset_s):.4f}s of "
f"silence onto the front of the H3-VR wav")
d_guess_samples = d_abs
overlap_start_s = max(0.0, -offset_s) + 1.0
overlap_end_s = min(dur_cam_s, len(h3) / sr - offset_s)
positions = [
overlap_start_s,
max(overlap_start_s, dur_cam_s / 2 - window_s / 2),
max(overlap_start_s, overlap_end_s - window_s - 2.0),
]
checks = windowed_check(cam, sr, h3, d_guess_samples, positions,
window_s=window_s, search_s=search_s)
print(f"windowed confidence checks (pos_s, local_offset_s, pearson_r):")
for c in checks:
if c is None:
print(" None (out of range)")
else:
pos, doff, r2 = c
print(f" pos={pos:7.2f}s offset={doff:+.5f}s r={r2:.4f}")
valid = [c for c in checks if c is not None]
if valid:
mean_r = sum(c[2] for c in valid) / len(valid)
print(f"mean confidence over {len(valid)} window(s): r={mean_r:.4f}")
if len(valid) >= 2:
first, last = valid[0], valid[-1]
dt_pos = last[0] - first[0]
dt_off = last[1] - first[1]
drift_ms = dt_off * 1000
drift_ms_per_min = drift_ms / (dt_pos / 60) if dt_pos > 0 else float("nan")
print(f"drift first->last window: {drift_ms:.2f} ms over "
f"{dt_pos:.1f}s span ({drift_ms_per_min:.3f} ms/min)")
return offset_s, (mean_r if valid else None)
def parse_args():
ap = argparse.ArgumentParser(
description="Measure H3-VR/camera sync offset via cross-correlation.")
ap.add_argument("--h3vr", required=True, type=Path,
help="H3-VR ambiX WAV (4ch, ACN order W,Y,Z,X)")
src = ap.add_mutually_exclusive_group(required=True)
src.add_argument("--video", type=Path,
help="Video file; embedded audio is extracted via "
"ffmpeg and used as the sync reference")
src.add_argument("--camera-audio", type=Path,
help="Pre-extracted camera reference WAV")
ap.add_argument("--window-s", type=float, default=15.0,
help="Window length for confidence checks (default: 15)")
ap.add_argument("--search-s", type=float, default=0.5,
help="Search radius for windowed refinement (default: 0.5)")
return ap.parse_args()
def main():
args = parse_args()
if not args.h3vr.exists():
sys.exit(f"error: H3-VR file not found: {args.h3vr}")
if args.video:
if not args.video.exists():
sys.exit(f"error: video file not found: {args.video}")
with tempfile.TemporaryDirectory(prefix="sync-offset-") as tmp:
cam_path = extract_camera_audio(args.video, Path(tmp))
analyze(cam_path, args.h3vr, args.window_s, args.search_s)
else:
if not args.camera_audio.exists():
sys.exit(f"error: camera audio file not found: {args.camera_audio}")
analyze(args.camera_audio, args.h3vr, args.window_s, args.search_s)
if __name__ == "__main__":
main()