Replace the unusable prepare-content.py draft (untracked, never committed) with two composable, argument-driven CLIs and no hardcoded absolute paths: - sync-offset.py cross-correlates the H3-VR W channel against a camera reference track (extracted from video via ffmpeg, or a pre-extracted WAV) to recover the audio/video sync offset. Ported from a validated standalone script; keeps the full-mode correlate()+correlation_lags() approach and its reasoning, since 'valid'-mode index arithmetic was a real source of sign bugs. - prepare-scene.py builds one scene into public/content/scenes/<id> (trim/pad all 4 channels identically, transcode video to a web bitrate, extract a thumbnail, verify channel count/rate/duration) and upserts it into the manifest without touching other entries. Ignore scripts/__pycache__/ generated by running these.
262 lines
10 KiB
Python
Executable File
262 lines
10 KiB
Python
Executable File
#!/usr/bin/env python3
|
|
"""Measure the sync offset between a Zoom H3-VR ambiX WAV and camera audio.
|
|
|
|
Usage:
|
|
uv run --with numpy,scipy python scripts/sync-offset.py \\
|
|
--h3vr H3-VR.WAV --video stitched-equirect.mp4
|
|
|
|
uv run --with numpy,scipy python scripts/sync-offset.py \\
|
|
--h3vr H3-VR.WAV --camera-audio camera-embedded.wav
|
|
|
|
Requires numpy and scipy. This is a standalone content-prep tool, not
|
|
a dependency of the shipped player — run it with `uv run --with
|
|
numpy,scipy python scripts/sync-offset.py ...` so uv resolves an
|
|
ephemeral environment; no project venv needed.
|
|
|
|
Cross-correlates the H3-VR's W channel (channel 0 of the ambiX WAV —
|
|
the recorder outputs ACN order W,Y,Z,X, confirmed via bext/iXML
|
|
metadata `Rec Mode=AmbiX`, TRACK_LIST=W,Y,Z,X) against a camera
|
|
reference audio track. If --video is given, that reference is
|
|
extracted automatically via ffmpeg from the video's embedded audio
|
|
track. On Insta360 cameras that embedded track (`Ambarella AAC`,
|
|
4-channel, balanced per-channel RMS) is a discrete mic array, NOT
|
|
encoded ambisonics — it is usable only as a sync witness here, never
|
|
as spatial audio for the player (see README "Preparing real
|
|
content").
|
|
|
|
Sign convention (validated against a synthetic ground-truth signal
|
|
with a known +3.0s offset — exact recovery):
|
|
offset_s = the position IN the H3-VR file that corresponds to
|
|
video/camera time 0.
|
|
offset_s > 0 => H3-VR started recording BEFORE the camera
|
|
(H3 has a head start); to align, TRIM offset_s
|
|
seconds off the FRONT of the H3-VR wav.
|
|
offset_s < 0 => H3-VR started AFTER the camera; to align, PAD
|
|
|offset_s| seconds of silence onto the FRONT of
|
|
the H3-VR wav.
|
|
This is exactly the convention `prepare-scene.py --offset` expects.
|
|
|
|
Implementation note: this only ever uses scipy.signal.correlate(...,
|
|
mode='full') together with scipy.signal.correlation_lags(). An
|
|
earlier version used 'valid'-mode correlation with manual index
|
|
arithmetic to recover the lag, which has a non-obvious sign/offset
|
|
convention that depends on which signal is longer — that was a real
|
|
source of bugs (silently wrong offsets that still "looked plausible").
|
|
'full' mode + correlation_lags() gives the lag directly from scipy,
|
|
with no index-juggling left to get wrong.
|
|
"""
|
|
|
|
import argparse
|
|
import subprocess
|
|
import sys
|
|
import tempfile
|
|
from pathlib import Path
|
|
|
|
import numpy as np
|
|
from scipy.io import wavfile
|
|
from scipy.signal import correlate, correlation_lags
|
|
|
|
|
|
def load_mono(path):
|
|
sr, data = wavfile.read(path)
|
|
if data.ndim > 1:
|
|
data = data[:, 0]
|
|
data = data.astype(np.float64)
|
|
maxv = np.max(np.abs(data)) + 1e-9
|
|
data = data / maxv
|
|
return sr, data
|
|
|
|
|
|
def envelope(x, sr, target_sr=200):
|
|
block = max(1, sr // target_sr)
|
|
n = len(x) // block
|
|
x = x[: n * block].reshape(n, block)
|
|
env = np.sqrt(np.mean(x**2, axis=1) + 1e-12)
|
|
return env, sr / block
|
|
|
|
|
|
def full_corr_offset_samples(a, b):
|
|
"""Return d (samples) such that b[n+d] ~= a[n], via full+lags method."""
|
|
a0 = a - a.mean()
|
|
b0 = b - b.mean()
|
|
corr = correlate(a0, b0, mode="full")
|
|
lags = correlation_lags(len(a0), len(b0), mode="full")
|
|
lag = lags[np.argmax(corr)]
|
|
return int(-lag)
|
|
|
|
|
|
def pearson_at_offset(cam, h3, start_cam, window_n, d_samples):
|
|
"""Pearson r between cam[start_cam:start_cam+window_n] and the
|
|
corresponding h3 segment shifted by d_samples."""
|
|
end_cam = start_cam + window_n
|
|
h3_0 = start_cam + d_samples
|
|
h3_1 = h3_0 + window_n
|
|
if start_cam < 0 or end_cam > len(cam) or h3_0 < 0 or h3_1 > len(h3):
|
|
return None
|
|
a = cam[start_cam:end_cam]
|
|
b = h3[h3_0:h3_1]
|
|
return np.corrcoef(a, b)[0, 1]
|
|
|
|
|
|
def coarse_offset(cam, sr, h3):
|
|
env_cam, esr = envelope(cam, sr)
|
|
env_h3, _ = envelope(h3, sr)
|
|
d_env = full_corr_offset_samples(env_cam, env_h3)
|
|
return d_env / esr, esr
|
|
|
|
|
|
def refine_offset_local(cam, sr, h3, d_guess_samples, center_pos_s,
|
|
window_s=40.0, search_s=1.0):
|
|
"""Refine using a big window centered at center_pos_s (cam-time),
|
|
full corr against a widened h3 segment (window + search margin on
|
|
each side), then read back the exact offset."""
|
|
win_n = int(window_s * sr)
|
|
search_n = int(search_s * sr)
|
|
start_cam = max(0, int(center_pos_s * sr) - win_n // 2)
|
|
end_cam = min(len(cam), start_cam + win_n)
|
|
seg_cam = cam[start_cam:end_cam]
|
|
|
|
h3_start = max(0, start_cam + d_guess_samples - search_n)
|
|
h3_end = min(len(h3), end_cam + d_guess_samples + search_n)
|
|
seg_h3 = h3[h3_start:h3_end]
|
|
if len(seg_h3) <= len(seg_cam) or len(seg_cam) == 0:
|
|
return None
|
|
|
|
d_local = full_corr_offset_samples(seg_cam, seg_h3)
|
|
d_abs = (h3_start + d_local) - start_cam
|
|
r = pearson_at_offset(cam, h3, start_cam, len(seg_cam), d_abs)
|
|
return d_abs, r, start_cam, len(seg_cam)
|
|
|
|
|
|
def windowed_check(cam, sr, h3, d_guess_samples, positions_s,
|
|
window_s=15.0, search_s=0.5):
|
|
results = []
|
|
for pos in positions_s:
|
|
res = refine_offset_local(cam, sr, h3, d_guess_samples,
|
|
pos + window_s / 2,
|
|
window_s=window_s, search_s=search_s)
|
|
if res is None:
|
|
results.append(None)
|
|
continue
|
|
d_abs, r, _start_cam, _n = res
|
|
if r is None:
|
|
results.append(None)
|
|
continue
|
|
results.append((pos, d_abs / sr, r))
|
|
return results
|
|
|
|
|
|
def extract_camera_audio(video_path, out_dir):
|
|
"""ffmpeg-extract the video's embedded audio track to a WAV in
|
|
out_dir (a caller-managed temp directory — never a hardcoded path)."""
|
|
out_wav = out_dir / "camera-audio.wav"
|
|
subprocess.run(
|
|
["ffmpeg", "-y", "-v", "error", "-i", str(video_path),
|
|
"-vn", "-acodec", "pcm_s16le", "-ar", "48000", str(out_wav)],
|
|
check=True,
|
|
)
|
|
return out_wav
|
|
|
|
|
|
def analyze(cam_path, h3_path, window_s, search_s):
|
|
sr_cam, cam = load_mono(cam_path)
|
|
sr_h3, h3 = load_mono(h3_path)
|
|
if sr_cam != sr_h3:
|
|
sys.exit(f"error: sample rate mismatch (camera {sr_cam} Hz vs "
|
|
f"H3-VR {sr_h3} Hz)")
|
|
sr = sr_cam
|
|
|
|
d_coarse_s, esr = coarse_offset(cam, sr, h3)
|
|
print(f"coarse offset (envelope @ ~{esr:.1f} Hz): {d_coarse_s:.4f} s")
|
|
d_guess_samples = int(round(d_coarse_s * sr))
|
|
|
|
dur_cam_s = len(cam) / sr
|
|
center = dur_cam_s / 2
|
|
res = refine_offset_local(cam, sr, h3, d_guess_samples, center,
|
|
window_s=min(40.0, dur_cam_s * 0.6),
|
|
search_s=1.0)
|
|
if res is None:
|
|
sys.exit("error: refine failed (out of range) — files may not "
|
|
"overlap at all")
|
|
d_abs, r, _start_cam, _n = res
|
|
offset_s = d_abs / sr
|
|
print(f"offset_s = {offset_s:+.4f} s (pearson r={r:.4f})")
|
|
if offset_s > 0:
|
|
print(f" H3-VR started first -> TRIM {offset_s:.4f}s off the "
|
|
f"front of the H3-VR wav")
|
|
else:
|
|
print(f" H3-VR started later -> PAD {abs(offset_s):.4f}s of "
|
|
f"silence onto the front of the H3-VR wav")
|
|
|
|
d_guess_samples = d_abs
|
|
overlap_start_s = max(0.0, -offset_s) + 1.0
|
|
overlap_end_s = min(dur_cam_s, len(h3) / sr - offset_s)
|
|
positions = [
|
|
overlap_start_s,
|
|
max(overlap_start_s, dur_cam_s / 2 - window_s / 2),
|
|
max(overlap_start_s, overlap_end_s - window_s - 2.0),
|
|
]
|
|
checks = windowed_check(cam, sr, h3, d_guess_samples, positions,
|
|
window_s=window_s, search_s=search_s)
|
|
print(f"windowed confidence checks (pos_s, local_offset_s, pearson_r):")
|
|
for c in checks:
|
|
if c is None:
|
|
print(" None (out of range)")
|
|
else:
|
|
pos, doff, r2 = c
|
|
print(f" pos={pos:7.2f}s offset={doff:+.5f}s r={r2:.4f}")
|
|
|
|
valid = [c for c in checks if c is not None]
|
|
if valid:
|
|
mean_r = sum(c[2] for c in valid) / len(valid)
|
|
print(f"mean confidence over {len(valid)} window(s): r={mean_r:.4f}")
|
|
if len(valid) >= 2:
|
|
first, last = valid[0], valid[-1]
|
|
dt_pos = last[0] - first[0]
|
|
dt_off = last[1] - first[1]
|
|
drift_ms = dt_off * 1000
|
|
drift_ms_per_min = drift_ms / (dt_pos / 60) if dt_pos > 0 else float("nan")
|
|
print(f"drift first->last window: {drift_ms:.2f} ms over "
|
|
f"{dt_pos:.1f}s span ({drift_ms_per_min:.3f} ms/min)")
|
|
|
|
return offset_s, (mean_r if valid else None)
|
|
|
|
|
|
def parse_args():
|
|
ap = argparse.ArgumentParser(
|
|
description="Measure H3-VR/camera sync offset via cross-correlation.")
|
|
ap.add_argument("--h3vr", required=True, type=Path,
|
|
help="H3-VR ambiX WAV (4ch, ACN order W,Y,Z,X)")
|
|
src = ap.add_mutually_exclusive_group(required=True)
|
|
src.add_argument("--video", type=Path,
|
|
help="Video file; embedded audio is extracted via "
|
|
"ffmpeg and used as the sync reference")
|
|
src.add_argument("--camera-audio", type=Path,
|
|
help="Pre-extracted camera reference WAV")
|
|
ap.add_argument("--window-s", type=float, default=15.0,
|
|
help="Window length for confidence checks (default: 15)")
|
|
ap.add_argument("--search-s", type=float, default=0.5,
|
|
help="Search radius for windowed refinement (default: 0.5)")
|
|
return ap.parse_args()
|
|
|
|
|
|
def main():
|
|
args = parse_args()
|
|
if not args.h3vr.exists():
|
|
sys.exit(f"error: H3-VR file not found: {args.h3vr}")
|
|
|
|
if args.video:
|
|
if not args.video.exists():
|
|
sys.exit(f"error: video file not found: {args.video}")
|
|
with tempfile.TemporaryDirectory(prefix="sync-offset-") as tmp:
|
|
cam_path = extract_camera_audio(args.video, Path(tmp))
|
|
analyze(cam_path, args.h3vr, args.window_s, args.search_s)
|
|
else:
|
|
if not args.camera_audio.exists():
|
|
sys.exit(f"error: camera audio file not found: {args.camera_audio}")
|
|
analyze(args.camera_audio, args.h3vr, args.window_s, args.search_s)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|