From 87475f3ccdbf3f04dae29bd2c064ffb403fc58ee Mon Sep 17 00:00:00 2001 From: romleiaj Date: Tue, 8 Sep 2026 13:33:59 -0400 Subject: [PATCH 01/39] Keep DIVE's venv out of the environment handed to VIAME The worker image sets VIRTUAL_ENV to DIVE's own Python 3.11 venv, and `uv run` re-exports it to the Celery process, so every pipeline subprocess inherited it. The VIAME image's kwiver ships a second Python plugin loader (plugins_from_python) that reads VIRTUAL_ENV and edits sys.path via PySys_GetObject without holding the GIL after modules_python has already started the interpreter. The result is a SIGSEGV at plugin load for every `viame runner` invocation, before any pipeline process runs, with nothing useful on stderr. get_gpu_environment already strips the venv from PATH for the same reason; now it also drops VIRTUAL_ENV and the UV_PYTHON* variables. The variable has to be absent rather than empty because kwiver only checks getenv for null. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ --- server/dive_tasks/viame_config.py | 6 ++++-- server/tests/test_viame_config_env.py | 24 ++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 2 deletions(-) create mode 100644 server/tests/test_viame_config_env.py diff --git a/server/dive_tasks/viame_config.py b/server/dive_tasks/viame_config.py index a09489bcf..f3ea4602a 100644 --- a/server/dive_tasks/viame_config.py +++ b/server/dive_tasks/viame_config.py @@ -32,8 +32,10 @@ def get_gpu_environment() -> Dict[str, str]: env["PATH"] = env.get("PATH").replace("/opt/dive/local/venv/bin", "") # VIAME ships its own python, and kwiver prepends $VIRTUAL_ENV's # site-packages to the embedded interpreter's sys.path, so leaving this set - # points VIAME at our venv (a different python version). - env.pop("VIRTUAL_ENV", None) + # points VIAME at our venv (a different python version). `uv run` also + # re-exports its UV_PYTHON* selection; drop those for the same reason. + for key in ("VIRTUAL_ENV", "UV_PYTHON", "UV_PYTHON_INSTALL_DIR"): + env.pop(key, None) return env diff --git a/server/tests/test_viame_config_env.py b/server/tests/test_viame_config_env.py new file mode 100644 index 000000000..c9a26ede3 --- /dev/null +++ b/server/tests/test_viame_config_env.py @@ -0,0 +1,24 @@ +from unittest import mock + +from dive_tasks.viame_config import get_gpu_environment + + +def test_gpu_environment_drops_dive_venv(): + """VIAME must not inherit DIVE's venv: kwiver segfaults with VIRTUAL_ENV set.""" + worker_env = { + "PATH": "/opt/dive/local/venv/bin:/usr/local/bin:/usr/bin", + "VIRTUAL_ENV": "/opt/dive/local/venv", + "UV_PYTHON": "3.11", + "UV_PYTHON_INSTALL_DIR": "/opt/dive/local/uv-python", + "KEEP_ME": "1", + } + with mock.patch.dict("os.environ", worker_env, clear=True), mock.patch( + "dive_tasks.viame_config.getGPUs", return_value=[] + ): + env = get_gpu_environment() + + assert "VIRTUAL_ENV" not in env + assert "UV_PYTHON" not in env + assert "UV_PYTHON_INSTALL_DIR" not in env + assert "/opt/dive/local/venv/bin" not in env["PATH"] + assert env["KEEP_ME"] == "1" From cbcb3373dff5c649a7687ba11a6c2ac968a6eb9f Mon Sep 17 00:00:00 2001 From: romleiaj Date: Tue, 8 Sep 2026 15:32:35 -0400 Subject: [PATCH 02/39] Ask VIAME for the homography_json transform reader VIAME renamed its DIVE registration reader on 2026-08-25 (b540a933b, "Rename dive_transform_io to homography_json_io"): the transform_2d_io implementation is registered as "homography_json" now, not "dive". Every VIAME built since then, including the web worker image, refuses the warp pipelines with Could not find implementation "dive" for "kwiver::vital::algo::transform_2d_io" from key "transform_reader:type" viame: Caught unhandled std::exception: Unable to create transform_reader Send the new name from both the web task and the Desktop backend. kwiver has no fallback for an algorithm type, so a VIAME older than the rename needs rebuilding before its 2-cam/3-cam warp pipes run again. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ --- .../backend/native/cameraRegistration.spec.ts | 18 +++++++++--------- .../backend/native/cameraRegistration.ts | 6 +++--- server/dive_tasks/multicam_pipeline.py | 8 ++++---- server/tests/test_multicam_pipeline.py | 6 +++--- 4 files changed, 19 insertions(+), 19 deletions(-) diff --git a/client/platform/desktop/backend/native/cameraRegistration.spec.ts b/client/platform/desktop/backend/native/cameraRegistration.spec.ts index 2e589e790..986488b41 100644 --- a/client/platform/desktop/backend/native/cameraRegistration.spec.ts +++ b/client/platform/desktop/backend/native/cameraRegistration.spec.ts @@ -133,13 +133,13 @@ describe('buildRegistrationPipelineArgs', () => { const uvPath = npath.join(jobWorkDir, 'uv_to_rgb_registration.json'); expect(args).toStrictEqual({ 'warp2:transformation_file': uvPath, - 'warp2:transform_reader:type': 'dive', - 'warp2:transform_reader:dive:from_camera': 'uv', - 'warp2:transform_reader:dive:to_camera': 'rgb', + 'warp2:transform_reader:type': 'homography_json', + 'warp2:transform_reader:homography_json:from_camera': 'uv', + 'warp2:transform_reader:homography_json:to_camera': 'rgb', 'warp3:transformation_file': irPath, - 'warp3:transform_reader:type': 'dive', - 'warp3:transform_reader:dive:from_camera': 'ir', - 'warp3:transform_reader:dive:to_camera': 'rgb', + 'warp3:transform_reader:type': 'homography_json', + 'warp3:transform_reader:homography_json:from_camera': 'ir', + 'warp3:transform_reader:homography_json:to_camera': 'rgb', }); const written = await fs.readJSON(irPath); expect(written.type).toBe('dive-camera-registration'); @@ -159,7 +159,7 @@ describe('buildRegistrationPipelineArgs', () => { // (no warps declared, so that is not an error); ir (warp3) has a // reference pair. expect(Object.keys(args).some((key) => key.startsWith('warp2'))).toBe(false); - expect(args['warp3:transform_reader:dive:from_camera']).toBe('ir'); + expect(args['warp3:transform_reader:homography_json:from_camera']).toBe('ir'); const irPath = npath.join(jobWorkDir, 'ir_to_rgb_registration.json'); expect(args['warp3:transformation_file']).toBe(irPath); // The uv-to-ir pair is dropped from ir's job file too. @@ -197,7 +197,7 @@ describe('buildRegistrationPipelineArgs', () => { const jobWorkDir = '/home/user/job/seeded'; await fs.ensureDir(jobWorkDir); const args = await buildRegistrationPipelineArgs(settings, meta, jobWorkDir, ['rgb', 'ir']); - expect(args['warp2:transform_reader:dive:from_camera']).toBe('ir'); - expect(args['warp2:transform_reader:dive:to_camera']).toBe('rgb'); + expect(args['warp2:transform_reader:homography_json:from_camera']).toBe('ir'); + expect(args['warp2:transform_reader:homography_json:to_camera']).toBe('rgb'); }); }); diff --git a/client/platform/desktop/backend/native/cameraRegistration.ts b/client/platform/desktop/backend/native/cameraRegistration.ts index f12c07f30..2fb03c579 100644 --- a/client/platform/desktop/backend/native/cameraRegistration.ts +++ b/client/platform/desktop/backend/native/cameraRegistration.ts @@ -279,9 +279,9 @@ export async function buildRegistrationPipelineArgs( const registrationPath = npath.join(jobWorkDir, registrationFileName(camera, reference)); writes.push(writeJsonFile(registrationPath, { ...file.body, pairs: [pair] })); args[`warp${input}:transformation_file`] = registrationPath; - args[`warp${input}:transform_reader:type`] = 'dive'; - args[`warp${input}:transform_reader:dive:from_camera`] = camera; - args[`warp${input}:transform_reader:dive:to_camera`] = reference; + args[`warp${input}:transform_reader:type`] = 'homography_json'; + args[`warp${input}:transform_reader:homography_json:from_camera`] = camera; + args[`warp${input}:transform_reader:homography_json:to_camera`] = reference; }); await Promise.all(writes); return args; diff --git a/server/dive_tasks/multicam_pipeline.py b/server/dive_tasks/multicam_pipeline.py index 72d870c5e..8630738f0 100644 --- a/server/dive_tasks/multicam_pipeline.py +++ b/server/dive_tasks/multicam_pipeline.py @@ -255,7 +255,7 @@ def build_registration_pairs(folder_meta: dict) -> List[dict]: become the file's imageLeft/imageRight, and each point's a/b pair becomes one `leftX leftY rightX rightY` row. - VIAME's dive transform reader only consumes the matrices; the + VIAME's homography_json transform reader only consumes the matrices; the observations travel for provenance and so a file round-trips back into DIVE without losing which frame contributed what. """ @@ -387,9 +387,9 @@ def build_registration_kwiver_settings( ) warp = f'warp{index + 1}' settings[f'{warp}:transformation_file'] = str(registration_path) - settings[f'{warp}:transform_reader:type'] = 'dive' - settings[f'{warp}:transform_reader:dive:from_camera'] = name - settings[f'{warp}:transform_reader:dive:to_camera'] = reference + settings[f'{warp}:transform_reader:type'] = 'homography_json' + settings[f'{warp}:transform_reader:homography_json:from_camera'] = name + settings[f'{warp}:transform_reader:homography_json:to_camera'] = reference return settings diff --git a/server/tests/test_multicam_pipeline.py b/server/tests/test_multicam_pipeline.py index 581ce8bf0..31f090498 100644 --- a/server/tests/test_multicam_pipeline.py +++ b/server/tests/test_multicam_pipeline.py @@ -284,9 +284,9 @@ def test_build_registration_kwiver_settings(tmp_path: Path): registration_path = str(tmp_path / 'ir_to_rgb_registration.json') assert settings == { 'warp2:transformation_file': registration_path, - 'warp2:transform_reader:type': 'dive', - 'warp2:transform_reader:dive:from_camera': 'ir', - 'warp2:transform_reader:dive:to_camera': 'rgb', + 'warp2:transform_reader:type': 'homography_json', + 'warp2:transform_reader:homography_json:from_camera': 'ir', + 'warp2:transform_reader:homography_json:to_camera': 'rgb', } written = json.loads((tmp_path / 'ir_to_rgb_registration.json').read_text(encoding='utf-8')) assert written['type'] == 'dive-camera-registration' From cd84af8f1e3648e22360958778b742f30bfccfde Mon Sep 17 00:00:00 2001 From: romleiaj Date: Tue, 8 Sep 2026 15:49:25 -0400 Subject: [PATCH 03/39] Cut multicam pipeline inputs to the shortest camera DIVE lets the cameras of a rig differ in frame count, but the 2-cam and 3-cam pipes run in lockstep and kwiver's detected_object_output grabs its optional image_file_name port unconditionally. When one camera ran out first the writer for the other tried to read the "complete" marker as a string: RuntimeError: Failed to cast datum of type 'complete' into std::string: bad any_cast. Before launching, take the native frame count of each video camera from its folder's ffprobe metadata and the list length of each image camera, and when they disagree cap every camera at the minimum: image lists are truncated, video readers get vidl_ffmpeg:stop_after_frame (1-based, so exactly that many frames). A registration frame subset already pairs row for row and is left alone. The job log says when the cap is applied. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ --- server/dive_tasks/multicam_pipeline.py | 56 ++++++++++++++++++++++++++ server/dive_tasks/run_pipeline.py | 17 ++++++++ server/tests/test_multicam_pipeline.py | 42 +++++++++++++++++++ 3 files changed, 115 insertions(+) diff --git a/server/dive_tasks/multicam_pipeline.py b/server/dive_tasks/multicam_pipeline.py index 8630738f0..0bd380bda 100644 --- a/server/dive_tasks/multicam_pipeline.py +++ b/server/dive_tasks/multicam_pipeline.py @@ -393,6 +393,51 @@ def build_registration_kwiver_settings( return settings +def video_frame_count(folder_meta: dict) -> Optional[int]: + """Native frame count of a video camera from its folder metadata, or None.""" + info = folder_meta.get('ffprobe_info') or {} + try: + nb_frames = int(info.get('nb_frames') or 0) + except (TypeError, ValueError): + nb_frames = 0 + if nb_frames > 0: + return nb_frames + try: + duration = float(info.get('duration') or 0) + fps = float(folder_meta.get(constants.OriginalFPSMarker) or 0) + except (TypeError, ValueError): + return None + if duration > 0 and fps > 0: + return int(round(duration * fps)) + return None + + +def common_frame_bound( + camera_media: Dict[str, Tuple[List[str], str]], + image_pairs: Optional[Dict[str, List[str]]], + frame_counts: Dict[str, int], +) -> Optional[int]: + """ + Frame count every camera must be cut to so the lockstep pipe ends together. + + Returns None when the cameras already agree or no count is known. Video + cameras without a frame subset take their count from frame_counts. + """ + counts: List[int] = [] + for name, (media_list, media_type) in camera_media.items(): + subset = (image_pairs or {}).get(name) + if subset is not None: + counts.append(len(subset)) + elif media_type == constants.VideoType: + if name in frame_counts: + counts.append(frame_counts[name]) + else: + counts.append(len(media_list)) + if len(counts) < 2 or min(counts) == max(counts): + return None + return min(counts) + + def build_multicam_kwiver_settings( work_dir: Path, cameras: List[MulticamCameraJob], @@ -402,6 +447,7 @@ def build_multicam_kwiver_settings( image_pairs: Optional[Dict[str, List[str]]] = None, fps: Optional[float] = None, on_progress: Optional[Callable[[str], None]] = None, + frame_bound: Optional[int] = None, ) -> Tuple[Dict[str, str], Dict[str, str]]: """ Build KWIVER -s key/value pairs for per-camera inputs/outputs. @@ -415,6 +461,9 @@ def build_multicam_kwiver_settings( must then drop any video reader type it would otherwise set (see video_subset_cameras). + frame_bound caps every camera at that many frames (see common_frame_bound): + image lists are truncated and video readers get stop_after_frame. + Returns (arg_file_pair, out_files) where out_files maps camera name -> output csv basename. """ arg_file_pair: Dict[str, str] = {} @@ -481,6 +530,8 @@ def camera_progress(done: int, total: int, name: str = key) -> None: arg_file_pair['track_writer:file_name'] = output_file_name if media_type in constants.ImageListTypes or extracted_subset: + if frame_bound is not None: + media_list = media_list[:frame_bound] input_file_name = str(work_dir / f'input{i + 1}_images.txt') with open(input_file_name, 'w', encoding='utf-8') as img_list_file: img_list_file.write('\n'.join(media_list)) @@ -490,6 +541,11 @@ def camera_progress(done: int, total: int, name: str = key) -> None: elif media_type == constants.VideoType: assert len(media_list) == 1, 'Expected exactly one video per camera' arg_file_pair[f'input{i + 1}:video_reader:type'] = 'vidl_ffmpeg' + if frame_bound is not None: + # vidl_ffmpeg frames are 1-based, so this reads exactly frame_bound frames. + arg_file_pair[f'input{i + 1}:video_reader:vidl_ffmpeg:stop_after_frame'] = str( + frame_bound + ) arg_file_pair[input_arg] = media_list[0] if i == 0: arg_file_pair['input:video_filename'] = media_list[0] diff --git a/server/dive_tasks/run_pipeline.py b/server/dive_tasks/run_pipeline.py index c9d98946b..a2552dc90 100644 --- a/server/dive_tasks/run_pipeline.py +++ b/server/dive_tasks/run_pipeline.py @@ -18,8 +18,10 @@ append_stereo_calibration_kwiver_settings, build_multicam_kwiver_settings, build_registration_kwiver_settings, + common_frame_bound, find_downloaded_calibration_file, is_stereo_measurement_pipeline, + video_frame_count, video_subset_cameras, ) from dive_tasks.pipeline_creates_dataset import ( @@ -328,6 +330,20 @@ def report_extraction(message: str) -> None: raise utils.CanceledError('Job was canceled') manager.write(f'{message}\n') + # Cut every camera to the shortest so the lockstep pipe ends together. + frame_counts: Dict[str, int] = {} + for camera in multicam_cameras: + if camera_media[camera['name']][1] != constants.VideoType: + continue + count = video_frame_count(gc.getFolder(camera['folder_id']).get('meta') or {}) + if count is not None: + frame_counts[camera['name']] = count + frame_bound = common_frame_bound(camera_media, image_pairs, frame_counts) + if frame_bound is not None: + manager.write( + f'Cameras differ in length; running the first {frame_bound} frames of each\n' + ) + arg_file_pair, out_files = build_multicam_kwiver_settings( _working_directory_path, multicam_cameras, @@ -336,6 +352,7 @@ def report_extraction(message: str) -> None: image_pairs=image_pairs, fps=input_fps, on_progress=report_extraction if extracted_cameras else None, + frame_bound=frame_bound, ) command = [ diff --git a/server/tests/test_multicam_pipeline.py b/server/tests/test_multicam_pipeline.py index 31f090498..29d1e91b9 100644 --- a/server/tests/test_multicam_pipeline.py +++ b/server/tests/test_multicam_pipeline.py @@ -10,6 +10,7 @@ build_multicam_kwiver_settings, build_registration_kwiver_settings, build_registration_pairs, + common_frame_bound, find_downloaded_calibration_file, infer_camera_role, infer_camera_roles, @@ -19,6 +20,7 @@ pipeline_requires_input, pseudo_frame_number, stereo_calibration_keys, + video_frame_count, video_subset_cameras, ) from dive_utils import constants @@ -442,3 +444,43 @@ def test_build_multicam_kwiver_settings_large_image_subset(tmp_path: Path): assert (tmp_path / 'input1_images.txt').read_text(encoding='utf-8') == '/tmp/ir/001.tif' assert arg_pair['input:video_filename'] == str(tmp_path / 'input1_images.txt') + + +def test_video_frame_count(): + assert video_frame_count({'ffprobe_info': {'nb_frames': '300'}}) == 300 + assert video_frame_count({'ffprobe_info': {'duration': '10.0'}, 'originalFps': 29.97}) == 300 + assert video_frame_count({'ffprobe_info': {'duration': '10.0'}}) is None + assert video_frame_count({}) is None + + +def test_common_frame_bound(): + seq = constants.ImageSequenceType + equal = {'a': (['1', '2'], seq), 'b': (['1', '2'], seq)} + assert common_frame_bound(equal, None, {}) is None + uneven = {'a': (['1', '2', '3'], seq), 'b': (['1', '2'], seq)} + assert common_frame_bound(uneven, None, {}) == 2 + videos = {'a': (['a.mp4'], constants.VideoType), 'b': (['b.mp4'], constants.VideoType)} + assert common_frame_bound(videos, None, {'a': 900, 'b': 850}) == 850 + assert common_frame_bound(videos, None, {'a': 900}) is None + # A frame subset already pairs row for row, so its length is the count. + assert common_frame_bound(videos, {'a': ['frame://1'], 'b': ['frame://1']}, {'a': 9}) is None + mixed = {'a': (['1', '2', '3'], seq), 'b': (['b.mp4'], constants.VideoType)} + assert common_frame_bound(mixed, None, {'b': 2}) == 2 + + +def test_build_multicam_kwiver_settings_frame_bound(tmp_path: Path): + cameras = [ + {'name': 'ir', 'folder_id': 'i', 'media_type': constants.ImageSequenceType}, + {'name': 'eo', 'folder_id': 'e', 'media_type': constants.VideoType}, + ] + camera_media = { + 'ir': (['/tmp/ir/0.png', '/tmp/ir/1.png', '/tmp/ir/2.png'], constants.ImageSequenceType), + 'eo': (['/tmp/eo.mp4'], constants.VideoType), + } + arg_pair, _ = build_multicam_kwiver_settings(tmp_path, cameras, camera_media, frame_bound=2) + assert (tmp_path / 'input1_images.txt').read_text(encoding='utf-8') == ( + '/tmp/ir/0.png\n/tmp/ir/1.png' + ) + assert arg_pair['input2:video_reader:vidl_ffmpeg:stop_after_frame'] == '2' + unbounded, _ = build_multicam_kwiver_settings(tmp_path, cameras, camera_media) + assert 'input2:video_reader:vidl_ffmpeg:stop_after_frame' not in unbounded From 2d9dd03750a01e8cf091bb6a52b0b3c6e83d879c Mon Sep 17 00:00:00 2001 From: romleiaj Date: Fri, 21 Aug 2026 10:43:17 -0400 Subject: [PATCH 04/39] Add an offset-derived aligned timeline for fixed-rig camera pairs buildAlignedTimeline needs a timestamp on every frame, which only image sequences carry (parsed from filenames). Video panes never qualify, so they scrub in raw-index lockstep and a recording start offset between two cameras is baked into the review -- a fixed EO/IR rig whose encoders start a fraction of a second apart has no way to express that today. On a fixed rig the offset is one constant per camera, so buildOffsetTimeline emits the same slot structure buildAlignedTimeline does from just that number. Everything downstream (pane seek, gap indication, cross-camera frame translation) then behaves exactly as it does for a timestamp-aligned dataset. Slots span the union of coverage, so the non-overlapping ends blank through the existing gap handling rather than being trimmed away. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit 821a3469561e4180d612a136ef22e1d16d48885f) --- client/dive-common/alignedTimeline.spec.ts | 64 +++++++++++++++++++- client/dive-common/alignedTimeline.ts | 68 ++++++++++++++++++++++ 2 files changed, 131 insertions(+), 1 deletion(-) diff --git a/client/dive-common/alignedTimeline.spec.ts b/client/dive-common/alignedTimeline.spec.ts index d0d0520fb..821c9b09e 100644 --- a/client/dive-common/alignedTimeline.spec.ts +++ b/client/dive-common/alignedTimeline.spec.ts @@ -1,6 +1,7 @@ import type { FrameImage } from './apispec'; import { - buildAlignedTimeline, buildInverseAlignedIndex, canAlign, computeGapGradient, computeGapSlots, + buildAlignedTimeline, buildInverseAlignedIndex, buildOffsetTimeline, canAlign, + computeGapGradient, computeGapSlots, } from './alignedTimeline'; function frame(timestamp?: number): FrameImage { @@ -209,3 +210,64 @@ describe('alignedTimeline', () => { }); }); }); + +/** + * Fixed-rig start offsets: the case timestamps can't cover, because video + * frames carry none. EO/IR pairs from the same fixed rig are recorded by + * independent encoders, so one can start a fraction of a second after the + * other; that lag is constant for the whole recording. + */ +describe('buildOffsetTimeline', () => { + it('pairs a later-starting camera with the reference instant', () => { + // B starts 2 frames later: B's frame 2 is the same instant as A's 0. + const result = buildOffsetTimeline({ A: 5, B: 5 }, { A: 0, B: 2 }); + if (!result.aligned) throw new Error('expected aligned'); + // Slot 0 is the earliest instant ANY camera saw: B's frame 0, before A began. + expect(result.slots[0]).toEqual({ A: undefined, B: 0 }); + expect(result.slots[2]).toEqual({ A: 0, B: 2 }); + expect(result.slots[4]).toEqual({ A: 2, B: 4 }); + }); + + it('keeps the union, blanking each camera outside its own coverage', () => { + const result = buildOffsetTimeline({ A: 3, B: 3 }, { A: 0, B: 2 }); + if (!result.aligned) throw new Error('expected aligned'); + // 2 leading slots before A starts + 3 shared + 0 trailing. + expect(result.slots).toHaveLength(5); + expect(computeGapSlots(result.slots)).toEqual([0, 1, 3, 4]); + // The overlap in the middle has both cameras. + expect(result.slots[2]).toEqual({ A: 0, B: 2 }); + }); + + it('round-trips through the inverse index the resolver uses', () => { + const result = buildOffsetTimeline({ A: 4, B: 4 }, { A: 0, B: 1 }); + if (!result.aligned) throw new Error('expected aligned'); + const inverse = buildInverseAlignedIndex(result.slots); + // Whatever slot holds A's frame 2 must hold B's frame 3 -- the same instant. + const slotForA2 = inverse.A.get(2) as number; + expect(result.slots[slotForA2].B).toBe(3); + expect(inverse.B.get(3)).toBe(slotForA2); + }); + + it('handles a negative offset (reference is the later camera)', () => { + const result = buildOffsetTimeline({ A: 4, B: 4 }, { A: 0, B: -1 }); + if (!result.aligned) throw new Error('expected aligned'); + expect(result.slots[1]).toEqual({ A: 1, B: 0 }); + }); + + it('declines when nothing needs correcting or there is no pair', () => { + // All-zero offsets: the positional path already does this, more cheaply. + expect(buildOffsetTimeline({ A: 5, B: 5 }, { A: 0, B: 0 })).toEqual({ aligned: false }); + // A camera with no frames loaded can't be aligned against. + expect(buildOffsetTimeline({ A: 5, B: 0 }, { A: 0, B: 2 })).toEqual({ aligned: false }); + expect(buildOffsetTimeline({ A: 5 }, { A: 3 })).toEqual({ aligned: false }); + }); + + it('treats a missing camera entry as no offset', () => { + // A is absent from the offsets map, so it behaves as A: 0 -- identical to + // { A: 0, B: 1 }: one leading slot for B's frame 0, then the pairs. + const result = buildOffsetTimeline({ A: 3, B: 3 }, { B: 1 }); + if (!result.aligned) throw new Error('expected aligned'); + expect(result.slots[0]).toEqual({ A: undefined, B: 0 }); + expect(result.slots[1]).toEqual({ A: 0, B: 1 }); + }); +}); diff --git a/client/dive-common/alignedTimeline.ts b/client/dive-common/alignedTimeline.ts index 750566003..08f3d4934 100644 --- a/client/dive-common/alignedTimeline.ts +++ b/client/dive-common/alignedTimeline.ts @@ -210,3 +210,71 @@ export function computeGapSlots(slots: AlignedSlot[]): number[] { }); return gaps; } + +/** + * A camera's constant start offset, in its own frames: local frame + * `slot + offset` shows the same instant as the reference camera's frame + * `slot`. A positive offset means this camera starts LATER -- its frame 0 + * happens before the reference's frame 0, so it must be read further in. + */ +export type CameraFrameOffsets = Record; + +/** + * Build a timeline from fixed per-camera start offsets rather than per-frame + * timestamps. + * + * buildAlignedTimeline needs a timestamp on every frame, which only image + * sequences carry (parsed from filenames) -- video panes never qualify, so + * they scrub in raw-index lockstep and any recording start offset between + * two cameras is baked into the review. On a fixed rig that offset is a + * single constant, so one number per camera is enough to line them up, and + * emitting it as slots means everything downstream (pane seek, gap + * indication, cross-camera frame translation) behaves exactly as it does + * for a timestamp-aligned dataset. + * + * Slots span the UNION of the cameras' coverage: where one camera has run + * out (or has not started), its entry is undefined, which the existing gap + * handling already renders and blanks correctly, rather than silently + * trimming footage off the ends. + * + * Returns { aligned: false } when fewer than two cameras have frames, or + * when every offset is zero -- there is nothing to correct then, so the + * caller should stay on the cheaper positional path. + */ +export function buildOffsetTimeline( + cameraFrameCounts: Record, + offsets: CameraFrameOffsets, +): TimelineResult { + const cameras = Object.keys(cameraFrameCounts) + .filter((camera) => cameraFrameCounts[camera] > 0); + if (cameras.length < 2) { + return { aligned: false }; + } + if (cameras.every((camera) => (offsets[camera] ?? 0) === 0)) { + return { aligned: false }; + } + // Slot s shows camera c's local frame s + offset[c]; that frame exists for + // s in [-offset[c], count[c] - offset[c]). Take the union across cameras, + // then rebase so the emitted slot array is 0-based. + const starts = cameras.map((camera) => -(offsets[camera] ?? 0)); + const ends = cameras.map( + (camera) => cameraFrameCounts[camera] - (offsets[camera] ?? 0), + ); + const base = Math.min(...starts); + const total = Math.max(...ends) - base; + if (total <= 0) { + return { aligned: false }; + } + const slots: AlignedSlot[] = new Array(total); + for (let index = 0; index < total; index += 1) { + const slot: AlignedSlot = {}; + cameras.forEach((camera) => { + const local = index + base + (offsets[camera] ?? 0); + slot[camera] = local >= 0 && local < cameraFrameCounts[camera] + ? local + : undefined; + }); + slots[index] = slot; + } + return { aligned: true, slots }; +} From dd4b0ef81ea00c840bbd0cf628eb0f336272ecc8 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Fri, 21 Aug 2026 10:46:01 -0400 Subject: [PATCH 05/39] Drive the aligned timeline from per-camera frame offsets Holds the offsets on CameraRegistrationStore, beside the spatial transform -- they are the other half of how two cameras line up, and persist with the same dataset save. The viewer now falls back to buildOffsetTimeline when no frame carries a timestamp, which is always the case for video panes, so a fixed rig whose encoders started a fraction of a second apart can finally express that. An all-zero offset still yields { aligned: false }, keeping uncorrected datasets on the existing positional path. Also extracts the per-camera frame count the auto-register service was computing inline, since the timeline needs the same video-aware lookup, and moves the registration store's construction above the timeline watch that now reads it. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit 149942d72d69044789d7ace263e2d7f33a574cba) --- client/dive-common/components/Viewer.vue | 51 ++++++++++++++----- .../alignedView/CameraRegistrationStore.ts | 18 +++++++ 2 files changed, 55 insertions(+), 14 deletions(-) diff --git a/client/dive-common/components/Viewer.vue b/client/dive-common/components/Viewer.vue index 030b969d8..10a2133bd 100644 --- a/client/dive-common/components/Viewer.vue +++ b/client/dive-common/components/Viewer.vue @@ -77,7 +77,8 @@ import { } from 'dive-common/apispec'; import { orderedMultiCamCameraNames } from 'dive-common/multicamDisplay'; import { - buildAlignedTimeline, buildInverseAlignedIndex, computeGapSlots, TimelineResult, + buildAlignedTimeline, buildInverseAlignedIndex, buildOffsetTimeline, computeGapSlots, + TimelineResult, } from 'dive-common/alignedTimeline'; import { computeOutputs, @@ -274,6 +275,26 @@ export default defineComponent({ * always for singleCam datasets -- playback falls back to today's exact * positional (broadcast-same-index) behavior via useMediaController.ts. */ + // Constructed here rather than beside the other aligned-view state below + // because the aligned timeline reads its frame offsets, and that watch + // runs immediately. + const cameraRegistration = new CameraRegistrationStore(); + /** + * How many frames this camera has: its image list when it is an image + * sequence, otherwise the annotator's own maxFrame (video panes carry no + * imageData). Returns 0 before the pane has loaded. + */ + function cameraFrameCount(camera: string): number { + const images = imageData.value[camera]; + if (images && images.length) { + return images.length; + } + try { + return aggregateController.value.getController(camera).maxFrame.value + 1; + } catch { + return 0; + } + } const alignedTimeline = computed(() => { if (!progress.loaded || multiCamList.value.length < 2) { return { aligned: false }; @@ -286,7 +307,20 @@ export default defineComponent({ multiCamList.value.forEach((camera) => { camerasFrames[camera] = imageData.value[camera] ?? []; }); - return buildAlignedTimeline(camerasFrames); + const byTimestamp = buildAlignedTimeline(camerasFrames); + if (byTimestamp.aligned) { + return byTimestamp; + } + // No usable timestamps (always the case for video panes). Fall back to + // the reviewer's fixed per-camera start offsets, which is the only way + // a video rig can express that one camera started late. Returns + // { aligned: false } when every offset is zero, so an uncorrected + // dataset stays on the cheaper positional path exactly as before. + const frameCounts: Record = {}; + multiCamList.value.forEach((camera) => { + frameCounts[camera] = cameraFrameCount(camera); + }); + return buildOffsetTimeline(frameCounts, cameraRegistration.frameOffsets.value); }); // Serialized shape of the currently installed timeline. The computed // re-evaluates whenever any camera's imageData array identity changes -- @@ -625,7 +659,6 @@ export default defineComponent({ * loadData resolves), but watches, aligned navigation, and metadata * hydration only run for multicamera datasets. */ - const cameraRegistration = new CameraRegistrationStore(); const alignedView = new AlignedViewStore(); const referenceCamera = computed(() => { const cams = multiCamList.value; @@ -805,17 +838,7 @@ export default defineComponent({ const autoRegisterJob = createAutoRegisterJobService({ datasetId, cameras: multiCamList, - frameCount: (camera: string) => { - const images = imageData.value[camera]; - if (images && images.length) { - return images.length; - } - try { - return aggregateController.value.getController(camera).maxFrame.value + 1; - } catch { - return 0; - } - }, + frameCount: cameraFrameCount, timestampsFor: (camera: string) => { const images = imageData.value[camera]; if (!images || !images.length diff --git a/client/src/alignedView/CameraRegistrationStore.ts b/client/src/alignedView/CameraRegistrationStore.ts index dd033f80b..092b923d2 100644 --- a/client/src/alignedView/CameraRegistrationStore.ts +++ b/client/src/alignedView/CameraRegistrationStore.ts @@ -306,6 +306,20 @@ export default class CameraRegistrationStore { */ source: Ref; + /** + * Per-camera constant start offset, in that camera's own frames: local + * frame `slot + offset` shows the same instant as the reference camera's + * `slot` (see dive-common/alignedTimeline.ts's buildOffsetTimeline). + * + * Lives here, beside the spatial transform, because it is the other half of + * "how these two cameras line up" and persists with the same dataset save. + * It is deliberately NOT derived from the transform: on a fixed rig the + * static background carries the homography fit, so the registration is + * insensitive to the temporal offset and cannot be used to recover or + * validate it -- only reviewing moving subjects can. + */ + frameOffsets: Ref>; + /** True when the calibration has unsaved changes since the last save or load. */ dirty: ComputedRef; @@ -337,6 +351,7 @@ export default class CameraRegistrationStore { this.recenterRequest = ref(null); this.fitError = ref(null); this.source = ref(null); + this.frameOffsets = ref({}); this.nextId = 1; this.nextRecenterId = 1; this.homographySources = {}; @@ -352,6 +367,7 @@ export default class CameraRegistrationStore { observations: this.observations.value, transformTypes: this.transformTypes.value, source: this.source.value, + frameOffsets: this.frameOffsets.value, }); } @@ -1378,11 +1394,13 @@ export default class CameraRegistrationStore { observations?: CameraObservations, transformTypes?: CameraTransformTypes, source?: RegistrationSource | null, + frameOffsets?: Record | null, ) { this.homographies.value = homographies ? { ...homographies } : {}; this.observations.value = observations ? { ...observations } : {}; this.transformTypes.value = transformTypes ? { ...transformTypes } : {}; this.source.value = source ?? null; + this.frameOffsets.value = frameOffsets ? { ...frameOffsets } : {}; this.markHomographySources(); this.activePair.value = null; this.pendingPoint.value = null; From aa3812c18b6d930cac3d990f9edec465997a329f Mon Sep 17 00:00:00 2001 From: romleiaj Date: Fri, 21 Aug 2026 10:48:18 -0400 Subject: [PATCH 06/39] Add a Time Offset slider to the registration panel Sits under Overlay Warp because the warp is how the value gets judged: scrub with the overlay on and only the correct offset makes moving subjects line up. The fit is no guide -- a static background matches at every offset, so nothing in the registration stats can confirm it. Slider plus single-frame nudge buttons, bounded at +/-3s of the dataset's frame rate, with a frames-and-seconds readout. Only the right camera moves; the left stays the time reference. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit e7de7c1d888d1cfd1e482ce85faef0f16f8c6391) --- .../CameraRegistration/RegistrationTools.vue | 98 +++++++++++++++++++ 1 file changed, 98 insertions(+) diff --git a/client/dive-common/components/CameraRegistration/RegistrationTools.vue b/client/dive-common/components/CameraRegistration/RegistrationTools.vue index 619426bce..cf86b9899 100644 --- a/client/dive-common/components/CameraRegistration/RegistrationTools.vue +++ b/client/dive-common/components/CameraRegistration/RegistrationTools.vue @@ -7,6 +7,7 @@ import { useCameraStore, useCameraRegistration, useDatasetId, + useTime, } from 'vue-media-annotator/provides'; import { TransformType, TRANSFORM_TYPES, DEFAULT_TRANSFORM_TYPE, minPointsForTransform, @@ -561,6 +562,56 @@ export default defineComponent({ return cam; } + /** + * Start-offset control for the right camera, in its own frames. + * + * Two independently-started recorders on a fixed rig can be a fraction of + * a second apart for the whole clip. The homography can't reveal that -- + * the static background fits equally well at any offset -- so this is set + * by eye: turn Overlay Warp on and scrub, and moving subjects line up + * only at the right value. + * + * Only the right camera moves; the left stays the time reference. On a + * 3-camera rig each non-reference camera keeps its own offset, so + * switching the pair selector edits whichever camera is on the right. + */ + const OFFSET_LIMIT_SECONDS = 3; + const { frameRate } = useTime(); + const offsetLimit = computed( + () => Math.max(1, Math.round((frameRate.value || 30) * OFFSET_LIMIT_SECONDS)), + ); + const rightFrameOffset = computed({ + get: () => (camRight.value + ? registration.frameOffsets.value[camRight.value] ?? 0 + : 0), + set: (value: number) => { + const camera = camRight.value; + if (!camera) { + return; + } + // Replace the map rather than mutating it: the aligned timeline is a + // computed over this ref and only re-runs on identity change. + registration.frameOffsets.value = { + ...registration.frameOffsets.value, + [camera]: Math.round(value), + }; + }, + }); + /** e.g. "+12 frames (+0.400s)"; the seconds half needs a known frame rate. */ + const offsetReadout = computed(() => { + const frames = rightFrameOffset.value; + const sign = frames > 0 ? '+' : ''; + const plural = Math.abs(frames) === 1 ? '' : 's'; + const rate = frameRate.value; + const seconds = rate + ? ` (${sign}${(frames / rate).toFixed(3)}s)` + : ''; + return `${sign}${frames} frame${plural}${seconds}`; + }); + function nudgeOffset(delta: number) { + rightFrameOffset.value += delta; + } + /** Live cursor readout text: this camera's coord, and its linked point in the other camera. */ const cursorReadout = computed(() => { const cursor = registration.cursorCoord.value; @@ -766,6 +817,10 @@ export default defineComponent({ selectedCorrespondenceId, deleteSelectedCorrespondence, cursorReadout, + rightFrameOffset, + offsetLimit, + offsetReadout, + nudgeOffset, correspondences, pairStats, currentPairFrame, @@ -1394,6 +1449,49 @@ export default defineComponent({ + +

Time Offset

+ + Frames to shift R so it plays in step with L. + +
+ + + +
+
+ {{ offsetReadout }} +
+ + + Date: Fri, 21 Aug 2026 10:51:06 -0400 Subject: [PATCH 07/39] Persist per-camera frame offsets with the dataset Adds cameraFrameOffsets to the mutable dataset config (client type, allowlisted mutable keys, and the server metadata model) and threads it through the two save paths and all three hydrate paths, so an offset dialed in during review survives a reload and travels with the dataset the same way the registration does. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit ee4d66b8b36e81012d9649a5d8949bf97d2f41c2) --- client/dive-common/apispec.ts | 4 +++- .../components/CameraRegistration/RegistrationTools.vue | 1 + client/dive-common/components/ImportAnnotations.vue | 1 + client/dive-common/components/Viewer.vue | 2 ++ client/dive-common/use/useAutoRegisterJob.ts | 1 + server/dive_utils/models.py | 2 ++ 6 files changed, 10 insertions(+), 1 deletion(-) diff --git a/client/dive-common/apispec.ts b/client/dive-common/apispec.ts index 977006bdd..226171b34 100644 --- a/client/dive-common/apispec.ts +++ b/client/dive-common/apispec.ts @@ -332,9 +332,11 @@ interface DatasetConfigMutable { * role are absent. */ cameraRoles?: Record; + /** Per-camera start offset in its own frames, for recorders that started at different times. */ + cameraFrameOffsets?: Record; error?: string; } -const DatasetConfigMutableKeys = ['attributes', 'confidenceFilters', 'timeFilters', 'imageEnhancements', 'customTypeStyling', 'customGroupStyling', 'attributeTrackFilters', 'datasetInfo', 'cameraHomographies', 'cameraCorrespondences', 'cameraTransformTypes', 'cameraRegistrationSource', 'typeHierarchy', 'taxonomySources', 'cameraRoles']; +const DatasetConfigMutableKeys = ['attributes', 'confidenceFilters', 'timeFilters', 'imageEnhancements', 'customTypeStyling', 'customGroupStyling', 'attributeTrackFilters', 'datasetInfo', 'cameraHomographies', 'cameraCorrespondences', 'cameraTransformTypes', 'cameraRegistrationSource', 'cameraFrameOffsets', 'typeHierarchy', 'taxonomySources', 'cameraRoles']; /** * Cross-dataset color/style overrides, reused across every dataset when the * "shared" color scope is enabled (see clientSettings.typeSettings.colorScope). diff --git a/client/dive-common/components/CameraRegistration/RegistrationTools.vue b/client/dive-common/components/CameraRegistration/RegistrationTools.vue index cf86b9899..881a91c8c 100644 --- a/client/dive-common/components/CameraRegistration/RegistrationTools.vue +++ b/client/dive-common/components/CameraRegistration/RegistrationTools.vue @@ -699,6 +699,7 @@ export default defineComponent({ cameraCorrespondences: registration.observations.value, cameraTransformTypes: registration.transformTypes.value, cameraRegistrationSource: registration.source.value, + cameraFrameOffsets: registration.frameOffsets.value, }); registration.markSaved(); } finally { diff --git a/client/dive-common/components/ImportAnnotations.vue b/client/dive-common/components/ImportAnnotations.vue index 2b34a4082..4f8c9bf2f 100644 --- a/client/dive-common/components/ImportAnnotations.vue +++ b/client/dive-common/components/ImportAnnotations.vue @@ -340,6 +340,7 @@ export default defineComponent({ meta.cameraCorrespondences, meta.cameraTransformTypes, meta.cameraRegistrationSource, + meta.cameraFrameOffsets, ); if (priorPair) { // The panel is open: re-select the imported pair (falling back to diff --git a/client/dive-common/components/Viewer.vue b/client/dive-common/components/Viewer.vue index 10a2133bd..ed1554ed3 100644 --- a/client/dive-common/components/Viewer.vue +++ b/client/dive-common/components/Viewer.vue @@ -1534,6 +1534,7 @@ export default defineComponent({ cameraCorrespondences: cameraRegistration.observations.value, cameraTransformTypes: cameraRegistration.transformTypes.value, cameraRegistrationSource: cameraRegistration.source.value, + cameraFrameOffsets: cameraRegistration.frameOffsets.value, }); cameraRegistration.markSaved(); } @@ -2077,6 +2078,7 @@ export default defineComponent({ meta.cameraCorrespondences, meta.cameraTransformTypes, meta.cameraRegistrationSource, + meta.cameraFrameOffsets, ); // Media is loaded at this point: resolve observation frames from // their image names against this dataset's own frame ordering. diff --git a/client/dive-common/use/useAutoRegisterJob.ts b/client/dive-common/use/useAutoRegisterJob.ts index e3b4f091e..e6248edd1 100644 --- a/client/dive-common/use/useAutoRegisterJob.ts +++ b/client/dive-common/use/useAutoRegisterJob.ts @@ -241,6 +241,7 @@ export function createAutoRegisterJobService(deps: AutoRegisterJobDeps): AutoReg meta.cameraCorrespondences, meta.cameraTransformTypes, meta.cameraRegistrationSource, + meta.cameraFrameOffsets, ); if (priorPair) { registration.setActivePair(priorPair.camA, priorPair.camB); diff --git a/server/dive_utils/models.py b/server/dive_utils/models.py index d0139c7b7..288017b8b 100644 --- a/server/dive_utils/models.py +++ b/server/dive_utils/models.py @@ -305,6 +305,8 @@ class MetadataMutable(BaseModel): # and image names and editable afterwards; used to place cameras onto a # pipeline's declared camera slots. Cameras with no known role are absent. cameraRoles: Optional[Dict[str, CameraRole]] + # Per-camera start offset in its own frames, for recorders that started at different times. + cameraFrameOffsets: Optional[Dict[str, int]] fps: Optional[float] @staticmethod From bd7bd6b9cbe882a153eb79bbdde3da7ab90bae39 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Fri, 21 Aug 2026 11:00:58 -0400 Subject: [PATCH 08/39] Carry the frame offset in registration files Adds an optional per-pair frameOffset to the portable registration JSON: the right camera's start offset in its own frames, relative to the left. A producer that measured it (the batch registration driver) now hands it over with the transform instead of leaving the reviewer to rediscover it, and a correction made in the panel travels back out on export. Read on file load and through the multicam import seed, merged per camera so a later file wins one camera's offset without clearing another's, and omitted from written files when a camera is in step. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit 8b8191ed0ee3c3dd3b6c719234989a22e632ea0f) --- .../web-girder/multicamRegistrationSeed.ts | 3 ++ .../alignedView/CameraRegistrationStore.ts | 20 ++++++++ .../cameraRegistrationFiles.spec.ts | 51 +++++++++++++++++++ .../alignedView/cameraRegistrationFiles.ts | 22 +++++++- 4 files changed, 95 insertions(+), 1 deletion(-) diff --git a/client/platform/web-girder/multicamRegistrationSeed.ts b/client/platform/web-girder/multicamRegistrationSeed.ts index 27e3fcebb..57cb4ea03 100644 --- a/client/platform/web-girder/multicamRegistrationSeed.ts +++ b/client/platform/web-girder/multicamRegistrationSeed.ts @@ -32,6 +32,7 @@ export async function parseRegistrationSeed( const homographies: CameraRegistrationValues['homographies'] = {}; const observations: CameraRegistrationValues['observations'] = {}; const transformTypes: CameraRegistrationValues['transformTypes'] = {}; + const frameOffsets: Record = {}; const stamps: { file: string; source: RegistrationSource | null }[] = []; const warnings: string[] = []; // eslint-disable-next-line no-restricted-syntax @@ -58,6 +59,7 @@ export async function parseRegistrationSeed( Object.assign(homographies, store.homographies.value); Object.assign(observations, store.observations.value); Object.assign(transformTypes, store.transformTypes.value); + Object.assign(frameOffsets, store.frameOffsets.value); stamps.push({ file: file.name, source: store.source.value }); } const seeded = Object.keys(homographies).length || Object.keys(observations).length; @@ -67,6 +69,7 @@ export async function parseRegistrationSeed( observations, transformTypes, source: mergeRegistrationSources(stamps), + ...(Object.keys(frameOffsets).length ? { frameOffsets } : {}), } : null, warnings, }; diff --git a/client/src/alignedView/CameraRegistrationStore.ts b/client/src/alignedView/CameraRegistrationStore.ts index 092b923d2..e8e27875e 100644 --- a/client/src/alignedView/CameraRegistrationStore.ts +++ b/client/src/alignedView/CameraRegistrationStore.ts @@ -139,6 +139,15 @@ export interface RegistrationFilePair { leftToRight?: Matrix3 | null; rightToLeft?: Matrix3 | null; transformType?: TransformType; + /** + * The right camera's constant start offset in its own frames, relative to + * the left: right frame `n + frameOffset` is the same instant as left frame + * `n`. Carried here so a producer that measured it (see the batch + * registration driver) hands it over with the transform, and a reviewer's + * correction travels back out on export. Absent means zero -- no producer + * before this wrote one. + */ + frameOffset?: number; } /** Portable calibration file: everything needed to restore all pairs. */ @@ -1218,6 +1227,7 @@ export default class CameraRegistrationStore { const observations: CameraObservations = {}; const homographies: CameraHomographies = {}; const transformTypes: CameraTransformTypes = {}; + const frameOffsets: Record = {}; const cameras = new Set(); file.pairs.forEach((pair, i) => { const context = `Pair ${i + 1}`; @@ -1228,6 +1238,15 @@ export default class CameraRegistrationStore { const key = this.pairKey(pair.left, pair.right); cameras.add(pair.left); cameras.add(pair.right); + if (pair.frameOffset !== undefined) { + if (!Number.isInteger(pair.frameOffset)) { + throw new Error(`${context}: "frameOffset" must be a whole number of frames`); + } + // Recorded against the left camera, and stored per camera against the + // rig reference -- the same thing whenever left is the reference, + // which is how producers write these files. + frameOffsets[pair.right] = pair.frameOffset; + } if (pair.transformType !== undefined) { if (!TRANSFORM_TYPES.some((t) => t.value === pair.transformType)) { throw new Error( @@ -1260,6 +1279,7 @@ export default class CameraRegistrationStore { this.homographies.value = homographies; this.transformTypes.value = transformTypes; this.source.value = source; + this.frameOffsets.value = frameOffsets; this.markHomographySources(); this.renumberPoints(); this.pendingPoint.value = null; diff --git a/client/src/alignedView/cameraRegistrationFiles.spec.ts b/client/src/alignedView/cameraRegistrationFiles.spec.ts index 82c62e790..05fb1587e 100644 --- a/client/src/alignedView/cameraRegistrationFiles.spec.ts +++ b/client/src/alignedView/cameraRegistrationFiles.spec.ts @@ -234,3 +234,54 @@ describe('mergeRegistrationValues', () => { }); }); }); + +/** + * A fixed rig's recorders can start a fraction of a second apart. That lag + * rides with the transform so a producer that measured it hands it over, and + * a reviewer's correction travels back out. + */ +describe('frame offsets in registration files', () => { + it('writes the right camera\'s offset onto its pair', () => { + const files = buildPerCameraRegistrationFiles({ + homographies: {}, + observations: {}, + transformTypes: { 'EO::IR': 'homography' }, + source: null, + frameOffsets: { IR: 12 }, + }, 'EO'); + expect(files).toHaveLength(1); + expect(files[0].body.pairs[0].frameOffset).toBe(12); + }); + + it('omits the field entirely when a camera is in step', () => { + const files = buildPerCameraRegistrationFiles({ + homographies: {}, + observations: {}, + transformTypes: { 'EO::IR': 'homography' }, + source: null, + frameOffsets: { IR: 0 }, + }, 'EO'); + expect(files[0].body.pairs[0]).not.toHaveProperty('frameOffset'); + }); + + it('merges per camera, letting a later file win without clearing others', () => { + const merged = mergeRegistrationValues( + { + homographies: {}, + observations: {}, + transformTypes: {}, + source: null, + frameOffsets: { IR: 12, UV: -3 }, + }, + { + homographies: {}, + observations: {}, + transformTypes: {}, + source: null, + frameOffsets: { IR: 9 }, + }, + 'ir_to_eo_registration.json', + ); + expect(merged.frameOffsets).toStrictEqual({ IR: 9, UV: -3 }); + }); +}); diff --git a/client/src/alignedView/cameraRegistrationFiles.ts b/client/src/alignedView/cameraRegistrationFiles.ts index 6e00a1c74..f2a1dedd4 100644 --- a/client/src/alignedView/cameraRegistrationFiles.ts +++ b/client/src/alignedView/cameraRegistrationFiles.ts @@ -25,6 +25,12 @@ export interface CameraRegistrationValues { observations: CameraObservations; transformTypes: CameraTransformTypes; source: RegistrationSource | null; + /** + * Per-camera constant start offset in that camera's own frames. Optional: + * files written before this existed carry none, and the in-app default is + * an empty map (every camera in step). + */ + frameOffsets?: Record; } /** @@ -109,6 +115,12 @@ function toRegistrationFilePairs(values: CameraRegistrationValues): Registration leftToRight: homography ? homography.AtoB : null, rightToLeft: homography ? homography.BtoA : null, transformType: values.transformTypes[key] || DEFAULT_TRANSFORM_TYPE, + // Offsets are stored per camera against the rig reference; a pair + // carries the right camera's, which is what "right relative to left" + // means whenever the left side IS the reference (the normal case). + ...(values.frameOffsets?.[right] + ? { frameOffset: values.frameOffsets[right] } + : {}), }; }); } @@ -291,7 +303,15 @@ export function mergeRegistrationValues( files: { previous: existing.source, [incomingLabel]: incoming.source }, }; } + // Later files win a camera's offset, matching how pairs merge above. An + // incoming file that carries none leaves any existing value alone rather + // than resetting it to zero. + const frameOffsets = { ...existing.frameOffsets, ...incoming.frameOffsets }; return { - homographies, observations, transformTypes, source, + homographies, + observations, + transformTypes, + source, + ...(Object.keys(frameOffsets).length ? { frameOffsets } : {}), }; } From 1be058c2245031994f587b948e397323744cf430 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Tue, 25 Aug 2026 13:26:31 -0400 Subject: [PATCH 09/39] Pair multicam pipeline inputs through the camera time offsets Pipeline inputs pair positionally -- row i of one camera's list with row i of every other's -- so a rig whose recorders started at different times fed a 2-cam/3-cam detector mismatched instants, and every cross-camera association it made was quietly wrong. Only the auto-register path was safe, because it builds its imagePairs subset from aligned-timeline slots; every ordinary run took the raw-index path. Image-sequence cameras now drop the frames before their first paired instant and stop where the shortest camera does. Video cameras get the same correction as a seek (video_input's start_at_frame) rather than being extracted. Datasets with no offset set are untouched, taking exactly the previous path, and offsets that leave no overlapping span raise a clear error instead of writing empty input lists. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit 5b544c2568ed7f73599e23b3b2cc73643b489f7d) --- .../native/multiCamFrameOffset.spec.ts | 53 +++++++++++++++++++ .../desktop/backend/native/multiCamUtils.ts | 53 +++++++++++++++++++ 2 files changed, 106 insertions(+) create mode 100644 client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts diff --git a/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts b/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts new file mode 100644 index 000000000..081138d3c --- /dev/null +++ b/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts @@ -0,0 +1,53 @@ +import { pairedStartFrames } from './multiCamUtils'; + +/** A camera entry with just the field the pairing needs. */ +function cam(name: string, frames: number): [string, { originalImageFiles: string[] }] { + return [name, { originalImageFiles: new Array(frames).fill('x') }]; +} + +/** + * Pipeline inputs pair positionally -- row i of one camera's list with row i + * of every other's -- so a rig whose recorders started at different times + * hands a two-camera detector mismatched instants unless the lists are + * aligned first. Nothing downstream reports that; the associations are just + * wrong. These cover the arithmetic that prevents it. + */ +describe('pairedStartFrames', () => { + it('skips the late camera past its unpaired lead', () => { + // IR frame 9 is the same instant as EO frame 0. + const result = pairedStartFrames([cam('EO', 100), cam('IR', 100)], { IR: 9 }); + expect(result?.start).toStrictEqual({ EO: 0, IR: 9 }); + // IR runs out 9 frames earlier, so both stop there. + expect(result?.length).toBe(91); + }); + + it('skips the reference instead when the other camera leads', () => { + const result = pairedStartFrames([cam('EO', 100), cam('IR', 100)], { IR: -4 }); + expect(result?.start).toStrictEqual({ EO: 4, IR: 0 }); + expect(result?.length).toBe(96); + }); + + it('ends where the shortest camera does', () => { + const result = pairedStartFrames([cam('EO', 100), cam('IR', 50)], { IR: 9 }); + expect(result?.start).toStrictEqual({ EO: 0, IR: 9 }); + expect(result?.length).toBe(41); + }); + + it('handles three cameras with independent offsets', () => { + const rig = [cam('EO', 100), cam('IR', 100), cam('UV', 100)]; + const result = pairedStartFrames(rig, { IR: 9, UV: -4 }); + // UV is earliest, so every camera skips forward to UV's first paired slot. + expect(result?.start).toStrictEqual({ EO: 4, IR: 13, UV: 0 }); + expect(result?.length).toBe(87); + }); + + it('declines when nothing is offset, keeping the untouched path', () => { + expect(pairedStartFrames([cam('EO', 100), cam('IR', 100)], { EO: 0, IR: 0 })).toBeNull(); + expect(pairedStartFrames([cam('EO', 100), cam('IR', 100)], undefined)).toBeNull(); + }); + + it('reports no paired span when an offset outruns a camera', () => { + const result = pairedStartFrames([cam('EO', 100), cam('IR', 5)], { IR: 9 }); + expect(result?.length).toBeLessThanOrEqual(0); + }); +}); diff --git a/client/platform/desktop/backend/native/multiCamUtils.ts b/client/platform/desktop/backend/native/multiCamUtils.ts index 2b7830b12..55e57afe5 100644 --- a/client/platform/desktop/backend/native/multiCamUtils.ts +++ b/client/platform/desktop/backend/native/multiCamUtils.ts @@ -98,6 +98,41 @@ async function extractVideoFrames( return results; } +/** + * Where each camera must start reading so row i of every camera's input is + * the same instant. + * + * Cameras on a fixed rig can start recording fractions of a second apart + * (see the dataset's cameraFrameOffsets and dive-common/alignedTimeline.ts). + * Pipeline inputs pair positionally -- row i with row i -- so without this + * a two-camera detector is handed mismatched instants and every cross-camera + * association it makes is wrong, silently. + * + * Camera c's local frame `slot + offset[c]` is slot `slot`; the first slot + * every camera can cover is the largest -offset. Returns null when nothing + * is offset, so an uncorrected dataset takes exactly the previous path. + */ +export function pairedStartFrames( + cameras: [string, { originalImageFiles: string[] }][], + offsets: Record | undefined, +): { start: Record; length: number } | null { + const names = cameras.map(([name]) => name); + if (!offsets || !names.some((name) => (offsets[name] ?? 0) !== 0)) { + return null; + } + const firstSlot = Math.max(...names.map((name) => -(offsets[name] ?? 0))); + const start = Object.fromEntries( + names.map((name) => [name, firstSlot + (offsets[name] ?? 0)]), + ); + // Every list must also END together, or the trailing rows of the longest + // pair against nothing. Only meaningful for image sequences, where the + // counts are known here. + const lengths = cameras + .filter(([, list]) => list.originalImageFiles.length) + .map(([name, list]) => list.originalImageFiles.length - start[name]); + return { start, length: lengths.length ? Math.min(...lengths) : 0 }; +} + /** frame://N pseudo-name to frame number, or null for real image names. */ function pseudoFrameNumber(entry: string): number | null { const match = /^frame:\/\/(\d+)$/.exec(entry); @@ -226,6 +261,15 @@ async function writeMultiCamStereoPipelineArgs( ? cameraOrder.filter((name) => name in cameras) : orderedMultiCamCameraNames(meta.multiCam); const cameraList = cameraNames.map((name) => [name, cameras[name]] as const); + const startFrames = pairedStartFrames(cameraList, meta.cameraFrameOffsets); + if (startFrames && startFrames.length <= 0) { + // Empty input lists would fail deep inside the pipeline; say why here. + throw new Error( + 'The dataset\'s camera time offsets leave no overlapping frames between ' + + 'cameras, so there is nothing to run. Check the Time Offset in the ' + + 'Camera Registration panel.', + ); + } for (let i = 0; i < cameraList.length; i += 1) { const [key, list] = cameraList[i]; const { originalBasePath } = list; @@ -260,6 +304,10 @@ async function writeMultiCamStereoPipelineArgs( throw new Error(`Image file not found: ${image}`); } } + } else if (startFrames) { + // Keep this camera's paired span: from its first paired instant, as long as the shortest. + const from = startFrames.start[key]; + images = images.slice(from, from + startFrames.length); } else if (runtime.frameRange) { // The single-camera path filters image lists by frameRange; // multicam silently ignored it (a pre-existing no-op) -- apply it @@ -311,6 +359,11 @@ async function writeMultiCamStereoPipelineArgs( const vidTypeArg = `input${i + 1}:video_reader:type`; const vidType = 'vidl_ffmpeg'; argFilePair[vidTypeArg] = vidType; + if (startFrames) { + // Same correction as the image-list slice, expressed as a seek: + // start_at_frame counts from 1, and 0 would mean "the beginning". + argFilePair[`input${i + 1}:start_at_frame`] = String(startFrames.start[key] + 1); + } const videoFileName = npath.join(originalBasePath, vidFile); argFilePair[inputArg] = videoFileName; if (i === 0) { From 499ca3ea13e3728043c223fe3728df593767b3e0 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Wed, 26 Aug 2026 10:54:31 -0400 Subject: [PATCH 10/39] Bound multicam video inputs to the cameras' common span The offset correction seeked each video to its first paired frame but left the tail open, so unequal-length recordings still ran on past each other -- which for video is not merely wasted work: one input completing while another still has frames desynchronizes the pipeline, and a writer takes a 'complete' datum where it expects data. Each video input now also carries stop_after_frame. Bounding the tail needs a frame count per camera, which a video only knows after a probe, so pairedStartFrames now takes counts rather than digging them out of the image list. That also fixes an all-video rig computing a zero-length span from image counts that were never there -- the case the previous commit would have wrongly refused to run. The probe reads the container header rather than decoding: counting frames for real would add tens of seconds per camera to every launch, and it only runs when an offset is actually set. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01K9imLktmzCkAR3wmUybaEs (cherry picked from commit fb9600090241b4809df1d386caa914ea55e4a260) --- .../native/multiCamFrameOffset.spec.ts | 42 ++++++++---- .../desktop/backend/native/multiCamUtils.ts | 68 ++++++++++++++++--- 2 files changed, 85 insertions(+), 25 deletions(-) diff --git a/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts b/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts index 081138d3c..d31def7fb 100644 --- a/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts +++ b/client/platform/desktop/backend/native/multiCamFrameOffset.spec.ts @@ -1,53 +1,67 @@ import { pairedStartFrames } from './multiCamUtils'; -/** A camera entry with just the field the pairing needs. */ -function cam(name: string, frames: number): [string, { originalImageFiles: string[] }] { - return [name, { originalImageFiles: new Array(frames).fill('x') }]; -} - /** * Pipeline inputs pair positionally -- row i of one camera's list with row i * of every other's -- so a rig whose recorders started at different times - * hands a two-camera detector mismatched instants unless the lists are + * hands a two-camera detector mismatched instants unless the inputs are * aligned first. Nothing downstream reports that; the associations are just * wrong. These cover the arithmetic that prevents it. + * + * Frame counts arrive already resolved (an image sequence knows its own, a + * video is probed), so this works the same for either medium -- which is the + * point: an all-video rig is exactly the case that needs it. */ describe('pairedStartFrames', () => { it('skips the late camera past its unpaired lead', () => { // IR frame 9 is the same instant as EO frame 0. - const result = pairedStartFrames([cam('EO', 100), cam('IR', 100)], { IR: 9 }); + const result = pairedStartFrames({ EO: 100, IR: 100 }, { IR: 9 }); expect(result?.start).toStrictEqual({ EO: 0, IR: 9 }); // IR runs out 9 frames earlier, so both stop there. expect(result?.length).toBe(91); }); it('skips the reference instead when the other camera leads', () => { - const result = pairedStartFrames([cam('EO', 100), cam('IR', 100)], { IR: -4 }); + const result = pairedStartFrames({ EO: 100, IR: 100 }, { IR: -4 }); expect(result?.start).toStrictEqual({ EO: 4, IR: 0 }); expect(result?.length).toBe(96); }); it('ends where the shortest camera does', () => { - const result = pairedStartFrames([cam('EO', 100), cam('IR', 50)], { IR: 9 }); + const result = pairedStartFrames({ EO: 100, IR: 50 }, { IR: 9 }); expect(result?.start).toStrictEqual({ EO: 0, IR: 9 }); expect(result?.length).toBe(41); }); + it('bounds a pair of unequal-length videos to their common span', () => { + // The case that desynchronized the warp pipeline at teardown: one input + // ran on after the other had completed. + const result = pairedStartFrames({ EO: 9000, IR: 9008 }, { IR: 9 }); + expect(result?.start).toStrictEqual({ EO: 0, IR: 9 }); + // IR contributes 8999 frames from index 9; EO has 9000 from index 0. + expect(result?.length).toBe(8999); + }); + it('handles three cameras with independent offsets', () => { - const rig = [cam('EO', 100), cam('IR', 100), cam('UV', 100)]; - const result = pairedStartFrames(rig, { IR: 9, UV: -4 }); + const result = pairedStartFrames({ EO: 100, IR: 100, UV: 100 }, { IR: 9, UV: -4 }); // UV is earliest, so every camera skips forward to UV's first paired slot. expect(result?.start).toStrictEqual({ EO: 4, IR: 13, UV: 0 }); expect(result?.length).toBe(87); }); it('declines when nothing is offset, keeping the untouched path', () => { - expect(pairedStartFrames([cam('EO', 100), cam('IR', 100)], { EO: 0, IR: 0 })).toBeNull(); - expect(pairedStartFrames([cam('EO', 100), cam('IR', 100)], undefined)).toBeNull(); + expect(pairedStartFrames({ EO: 100, IR: 100 }, { EO: 0, IR: 0 })).toBeNull(); + expect(pairedStartFrames({ EO: 100, IR: 100 }, undefined)).toBeNull(); }); it('reports no paired span when an offset outruns a camera', () => { - const result = pairedStartFrames([cam('EO', 100), cam('IR', 5)], { IR: 9 }); + const result = pairedStartFrames({ EO: 100, IR: 5 }, { IR: 9 }); expect(result?.length).toBeLessThanOrEqual(0); }); + + it('ignores a camera whose count could not be resolved', () => { + // A probe that fails yields 0; that camera must not drag the span to + // nothing and abort a run the other cameras could still support. + const result = pairedStartFrames({ EO: 100, IR: 0 }, { IR: 9 }); + expect(result?.length).toBe(100); + }); }); diff --git a/client/platform/desktop/backend/native/multiCamUtils.ts b/client/platform/desktop/backend/native/multiCamUtils.ts index 55e57afe5..0ec85d36a 100644 --- a/client/platform/desktop/backend/native/multiCamUtils.ts +++ b/client/platform/desktop/backend/native/multiCamUtils.ts @@ -14,6 +14,7 @@ import { orderedMultiCamCameraNames } from 'dive-common/multicamDisplay'; import { getBinaryPath, spawnResult } from './utils'; const ffmpegPath = getBinaryPath('ffmpeg-ffprobe-static/ffmpeg'); +const ffprobePath = getBinaryPath('ffmpeg-ffprobe-static/ffprobe'); /** Frame subset / range inputs for multicam pipeline arg writing. */ export interface MultiCamRuntimeSubset { @@ -113,10 +114,10 @@ async function extractVideoFrames( * is offset, so an uncorrected dataset takes exactly the previous path. */ export function pairedStartFrames( - cameras: [string, { originalImageFiles: string[] }][], + frameCounts: Record, offsets: Record | undefined, ): { start: Record; length: number } | null { - const names = cameras.map(([name]) => name); + const names = Object.keys(frameCounts); if (!offsets || !names.some((name) => (offsets[name] ?? 0) !== 0)) { return null; } @@ -124,13 +125,35 @@ export function pairedStartFrames( const start = Object.fromEntries( names.map((name) => [name, firstSlot + (offsets[name] ?? 0)]), ); - // Every list must also END together, or the trailing rows of the longest - // pair against nothing. Only meaningful for image sequences, where the - // counts are known here. - const lengths = cameras - .filter(([, list]) => list.originalImageFiles.length) - .map(([name, list]) => list.originalImageFiles.length - start[name]); - return { start, length: lengths.length ? Math.min(...lengths) : 0 }; + // Every camera must also STOP together, or the trailing frames of the + // longest pair against nothing. For video that is not merely wasted work: + // one input completing while another still has frames desynchronizes the + // pipeline, and a writer takes a 'complete' datum where it expects data. + const spans = names + .filter((name) => frameCounts[name] > 0) + .map((name) => frameCounts[name] - start[name]); + return { start, length: spans.length ? Math.min(...spans) : 0 }; +} + +/** + * Frames in a video, from the container header. + * + * Deliberately the declared count rather than a decode: counting frames for + * real means decoding the whole file, which would add tens of seconds per + * camera to every pipeline launch. On intact media the two agree. A damaged + * stream can overstate here, leaving the bound slightly generous -- no worse + * than the unbounded behaviour this replaces, and one more reason to + * re-encode damaged captures before processing them. + */ +async function videoFrameCount(videoPath: string): Promise { + const result = await spawnResult(ffprobePath, [ + '-v', 'error', '-select_streams', 'v:0', + '-show_entries', 'stream=nb_frames', + '-of', 'default=nokey=1:noprint_wrappers=1', + videoPath, + ]); + const parsed = Number.parseInt((result.output ?? '').trim(), 10); + return Number.isFinite(parsed) && parsed > 0 ? parsed : 0; } /** frame://N pseudo-name to frame number, or null for real image names. */ @@ -261,7 +284,26 @@ async function writeMultiCamStereoPipelineArgs( ? cameraOrder.filter((name) => name in cameras) : orderedMultiCamCameraNames(meta.multiCam); const cameraList = cameraNames.map((name) => [name, cameras[name]] as const); - const startFrames = pairedStartFrames(cameraList, meta.cameraFrameOffsets); + // Videos are probed for a frame count only when an offset is set. + const frameCounts: Record = {}; + if (meta.cameraFrameOffsets + && cameraList.some(([name]) => (meta.cameraFrameOffsets?.[name] ?? 0) !== 0)) { + // eslint-disable-next-line no-restricted-syntax + for (const [name, list] of cameraList) { + if (list.originalImageFiles.length) { + frameCounts[name] = list.originalImageFiles.length; + } else if (list.originalVideoFile) { + const vidFile = (list.transcodedVideoFile && forceTranscoded) || list.transcodedMisalign + ? list.transcodedVideoFile : list.originalVideoFile; + frameCounts[name] = await videoFrameCount( + npath.join(list.originalBasePath, vidFile), + ); + } else { + frameCounts[name] = 0; + } + } + } + const startFrames = pairedStartFrames(frameCounts, meta.cameraFrameOffsets); if (startFrames && startFrames.length <= 0) { // Empty input lists would fail deep inside the pipeline; say why here. throw new Error( @@ -362,7 +404,11 @@ async function writeMultiCamStereoPipelineArgs( if (startFrames) { // Same correction as the image-list slice, expressed as a seek: // start_at_frame counts from 1, and 0 would mean "the beginning". - argFilePair[`input${i + 1}:start_at_frame`] = String(startFrames.start[key] + 1); + const from = startFrames.start[key]; + argFilePair[`input${i + 1}:start_at_frame`] = String(from + 1); + // ...and stop where the shortest camera does, so the inputs end + // together too. Also 1-based, and inclusive of the named frame. + argFilePair[`input${i + 1}:stop_after_frame`] = String(from + startFrames.length); } const videoFileName = npath.join(originalBasePath, vidFile); argFilePair[inputArg] = videoFileName; From d3c0b8529d1508539b25c8ca3d7029387b0124a3 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Tue, 8 Sep 2026 16:18:29 -0400 Subject: [PATCH 11/39] Pair web multicam pipeline inputs through the camera time offsets The time-offset branch corrected Desktop pipeline inputs but never the web task, so a web run of a 2-cam/3-cam pipe still paired frame i of each camera regardless of the stored cameraFrameOffsets. Port the same arithmetic: each camera skips to its first paired frame and every camera stops where the shortest does, sliced for image lists and set as start_at_frame/stop_after_frame on the video reader. Registration frame subsets are left alone since they already pair row for row. A dataset with no offset keeps the plain shortest-camera cap. The Desktop side set start_at_frame and stop_after_frame on the video_input process, which has no such keys, so the seek was silently ignored. Address the reader (video_reader:vidl_ffmpeg) instead. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ --- .../desktop/backend/native/multiCamUtils.ts | 10 ++--- server/dive_tasks/multicam_pipeline.py | 41 +++++++++++++++---- server/dive_tasks/run_pipeline.py | 30 ++++++++++++-- server/tests/test_multicam_pipeline.py | 35 ++++++++++++++++ 4 files changed, 98 insertions(+), 18 deletions(-) diff --git a/client/platform/desktop/backend/native/multiCamUtils.ts b/client/platform/desktop/backend/native/multiCamUtils.ts index 0ec85d36a..569fe5b1e 100644 --- a/client/platform/desktop/backend/native/multiCamUtils.ts +++ b/client/platform/desktop/backend/native/multiCamUtils.ts @@ -402,13 +402,11 @@ async function writeMultiCamStereoPipelineArgs( const vidType = 'vidl_ffmpeg'; argFilePair[vidTypeArg] = vidType; if (startFrames) { - // Same correction as the image-list slice, expressed as a seek: - // start_at_frame counts from 1, and 0 would mean "the beginning". + // vidl_ffmpeg frames are 1-based and stop_after_frame is inclusive. const from = startFrames.start[key]; - argFilePair[`input${i + 1}:start_at_frame`] = String(from + 1); - // ...and stop where the shortest camera does, so the inputs end - // together too. Also 1-based, and inclusive of the named frame. - argFilePair[`input${i + 1}:stop_after_frame`] = String(from + startFrames.length); + const reader = `input${i + 1}:video_reader:${vidType}`; + argFilePair[`${reader}:start_at_frame`] = String(from + 1); + argFilePair[`${reader}:stop_after_frame`] = String(from + startFrames.length); } const videoFileName = npath.join(originalBasePath, vidFile); argFilePair[inputArg] = videoFileName; diff --git a/server/dive_tasks/multicam_pipeline.py b/server/dive_tasks/multicam_pipeline.py index 0bd380bda..75d7a07af 100644 --- a/server/dive_tasks/multicam_pipeline.py +++ b/server/dive_tasks/multicam_pipeline.py @@ -438,6 +438,26 @@ def common_frame_bound( return min(counts) +def paired_start_frames( + frame_counts: Dict[str, int], + offsets: Optional[Dict[str, int]], +) -> Optional[Tuple[Dict[str, int], int]]: + """ + Per-camera start index and common length from the dataset's camera time offsets. + + offsets[name] is the frame of that camera that coincides with frame 0 of a + camera at offset 0. Returns None when no camera is offset; the length is + <= 0 when the offsets leave no overlapping span. + """ + names = list(frame_counts) + if not offsets or not any(offsets.get(name, 0) for name in names): + return None + first_slot = max(-offsets.get(name, 0) for name in names) + start = {name: first_slot + offsets.get(name, 0) for name in names} + spans = [frame_counts[name] - start[name] for name in names if frame_counts[name] > 0] + return start, (min(spans) if spans else 0) + + def build_multicam_kwiver_settings( work_dir: Path, cameras: List[MulticamCameraJob], @@ -448,6 +468,7 @@ def build_multicam_kwiver_settings( fps: Optional[float] = None, on_progress: Optional[Callable[[str], None]] = None, frame_bound: Optional[int] = None, + frame_starts: Optional[Dict[str, int]] = None, ) -> Tuple[Dict[str, str], Dict[str, str]]: """ Build KWIVER -s key/value pairs for per-camera inputs/outputs. @@ -461,8 +482,9 @@ def build_multicam_kwiver_settings( must then drop any video reader type it would otherwise set (see video_subset_cameras). - frame_bound caps every camera at that many frames (see common_frame_bound): - image lists are truncated and video readers get stop_after_frame. + frame_starts skips each camera to its first paired frame (see + paired_start_frames) and frame_bound caps every camera at that many frames + from there: image lists are sliced, video readers get start/stop frames. Returns (arg_file_pair, out_files) where out_files maps camera name -> output csv basename. """ @@ -529,9 +551,11 @@ def camera_progress(done: int, total: int, name: str = key) -> None: arg_file_pair['detector_writer:file_name'] = output_file_name arg_file_pair['track_writer:file_name'] = output_file_name + first = (frame_starts or {}).get(key, 0) if not extracted_subset else 0 if media_type in constants.ImageListTypes or extracted_subset: - if frame_bound is not None: - media_list = media_list[:frame_bound] + if first or frame_bound is not None: + stop = None if frame_bound is None else first + frame_bound + media_list = media_list[first:stop] input_file_name = str(work_dir / f'input{i + 1}_images.txt') with open(input_file_name, 'w', encoding='utf-8') as img_list_file: img_list_file.write('\n'.join(media_list)) @@ -541,11 +565,12 @@ def camera_progress(done: int, total: int, name: str = key) -> None: elif media_type == constants.VideoType: assert len(media_list) == 1, 'Expected exactly one video per camera' arg_file_pair[f'input{i + 1}:video_reader:type'] = 'vidl_ffmpeg' + reader = f'input{i + 1}:video_reader:vidl_ffmpeg' + if first: + # vidl_ffmpeg frames are 1-based; 0 means unset. + arg_file_pair[f'{reader}:start_at_frame'] = str(first + 1) if frame_bound is not None: - # vidl_ffmpeg frames are 1-based, so this reads exactly frame_bound frames. - arg_file_pair[f'input{i + 1}:video_reader:vidl_ffmpeg:stop_after_frame'] = str( - frame_bound - ) + arg_file_pair[f'{reader}:stop_after_frame'] = str(first + frame_bound) arg_file_pair[input_arg] = media_list[0] if i == 0: arg_file_pair['input:video_filename'] = media_list[0] diff --git a/server/dive_tasks/run_pipeline.py b/server/dive_tasks/run_pipeline.py index a2552dc90..0bc0aa92d 100644 --- a/server/dive_tasks/run_pipeline.py +++ b/server/dive_tasks/run_pipeline.py @@ -21,6 +21,7 @@ common_frame_bound, find_downloaded_calibration_file, is_stereo_measurement_pipeline, + paired_start_frames, video_frame_count, video_subset_cameras, ) @@ -333,16 +334,36 @@ def report_extraction(message: str) -> None: # Cut every camera to the shortest so the lockstep pipe ends together. frame_counts: Dict[str, int] = {} for camera in multicam_cameras: - if camera_media[camera['name']][1] != constants.VideoType: + media_list, media_type = camera_media[camera['name']] + if media_type != constants.VideoType: + frame_counts[camera['name']] = len(media_list) continue count = video_frame_count(gc.getFolder(camera['folder_id']).get('meta') or {}) if count is not None: frame_counts[camera['name']] = count - frame_bound = common_frame_bound(camera_media, image_pairs, frame_counts) - if frame_bound is not None: + frame_starts: Optional[Dict[str, int]] = None + frame_bound: Optional[int] = None + # A frame subset already pairs row for row, so offsets apply only to full runs. + offsets = None if image_pairs else fromMeta(input_folder, 'cameraFrameOffsets', None) + span = paired_start_frames(frame_counts, offsets) + if span is not None: + frame_starts, frame_bound = span + if frame_bound <= 0: + raise ValueError( + 'The camera time offsets leave no overlapping frames between cameras; ' + 'check the Time Offset in the Camera Registration panel.' + ) + skipped = ', '.join(f'{k} from frame {v}' for k, v in frame_starts.items() if v) manager.write( - f'Cameras differ in length; running the first {frame_bound} frames of each\n' + f'Applying camera time offsets ({skipped}); running {frame_bound} frames\n' ) + else: + frame_bound = common_frame_bound(camera_media, image_pairs, frame_counts) + if frame_bound is not None: + manager.write( + 'Cameras differ in length; ' + f'running the first {frame_bound} frames of each\n' + ) arg_file_pair, out_files = build_multicam_kwiver_settings( _working_directory_path, @@ -353,6 +374,7 @@ def report_extraction(message: str) -> None: fps=input_fps, on_progress=report_extraction if extracted_cameras else None, frame_bound=frame_bound, + frame_starts=frame_starts, ) command = [ diff --git a/server/tests/test_multicam_pipeline.py b/server/tests/test_multicam_pipeline.py index 29d1e91b9..e72e94e16 100644 --- a/server/tests/test_multicam_pipeline.py +++ b/server/tests/test_multicam_pipeline.py @@ -11,6 +11,7 @@ build_registration_kwiver_settings, build_registration_pairs, common_frame_bound, + paired_start_frames, find_downloaded_calibration_file, infer_camera_role, infer_camera_roles, @@ -484,3 +485,37 @@ def test_build_multicam_kwiver_settings_frame_bound(tmp_path: Path): assert arg_pair['input2:video_reader:vidl_ffmpeg:stop_after_frame'] == '2' unbounded, _ = build_multicam_kwiver_settings(tmp_path, cameras, camera_media) assert 'input2:video_reader:vidl_ffmpeg:stop_after_frame' not in unbounded + + +def test_paired_start_frames(): + assert paired_start_frames({'EO': 100, 'IR': 100}, None) is None + assert paired_start_frames({'EO': 100, 'IR': 100}, {'EO': 0, 'IR': 0}) is None + # IR frame 9 is the same instant as EO frame 0, so IR skips ahead and both stop together. + assert paired_start_frames({'EO': 100, 'IR': 100}, {'IR': 9}) == ({'EO': 0, 'IR': 9}, 91) + assert paired_start_frames({'EO': 100, 'IR': 100}, {'IR': -4}) == ({'EO': 4, 'IR': 0}, 96) + assert paired_start_frames({'EO': 9000, 'IR': 9008}, {'IR': 9}) == ({'EO': 0, 'IR': 9}, 8999) + assert paired_start_frames({'EO': 100, 'IR': 100, 'UV': 100}, {'IR': 9, 'UV': -4}) == ( + {'EO': 4, 'IR': 13, 'UV': 0}, + 87, + ) + start, length = paired_start_frames({'EO': 100, 'IR': 5}, {'IR': 9}) + assert length <= 0 + + +def test_build_multicam_kwiver_settings_frame_starts(tmp_path: Path): + cameras = [ + {'name': 'ir', 'folder_id': 'i', 'media_type': constants.ImageSequenceType}, + {'name': 'eo', 'folder_id': 'e', 'media_type': constants.VideoType}, + ] + camera_media = { + 'ir': (['/tmp/ir/0.png', '/tmp/ir/1.png', '/tmp/ir/2.png'], constants.ImageSequenceType), + 'eo': (['/tmp/eo.mp4'], constants.VideoType), + } + arg_pair, _ = build_multicam_kwiver_settings( + tmp_path, cameras, camera_media, frame_bound=2, frame_starts={'ir': 1, 'eo': 3} + ) + assert (tmp_path / 'input1_images.txt').read_text(encoding='utf-8') == ( + '/tmp/ir/1.png\n/tmp/ir/2.png' + ) + assert arg_pair['input2:video_reader:vidl_ffmpeg:start_at_frame'] == '4' + assert arg_pair['input2:video_reader:vidl_ffmpeg:stop_after_frame'] == '5' From 74be74f7a7f1dbb7449e0d623b14c54804e7e932 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Tue, 8 Sep 2026 16:54:12 -0400 Subject: [PATCH 12/39] Keep the displayed instant when the aligned timeline is rebuilt Installing an aligned-timeline resolver always seeked to slot 0. With a camera time offset the timeline starts with the earlier camera's lead, where the other camera has no frame yet, and every Time Offset nudge rebuilds the timeline -- so each nudge jumped the viewer to the start and blanked a pane with "No frame at this instant", which made the slider unusable for the very judgement it exists for. When a resolver replaces another, take the first camera's displayed local frame and seek to its slot in the new timeline instead. A fresh install still starts at slot 0. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ (cherry picked from commit 38e5a499554beefca1c959d21825199cdeae5efe) --- .../annotators/useMediaController.spec.ts | 24 +++++++++++++++++ .../annotators/useMediaController.ts | 26 ++++++++++++++----- 2 files changed, 44 insertions(+), 6 deletions(-) diff --git a/client/src/components/annotators/useMediaController.spec.ts b/client/src/components/annotators/useMediaController.spec.ts index 4798e5b9a..187845296 100644 --- a/client/src/components/annotators/useMediaController.spec.ts +++ b/client/src/components/annotators/useMediaController.spec.ts @@ -275,6 +275,30 @@ describe('useMediaController', () => { expect(composable.aggregateController.value.frame.value).toBe(0); }); + it('replacing a resolver keeps the first camera\'s displayed frame instead of jumping to slot 0', () => { + const { composable, mocks } = mountMediaController(); + composable.setAlignedFrameResolver(makeGappedResolver(10)); + composable.aggregateController.value.seek(5); + // A Time Offset nudge: B now runs three frames ahead of A, so the timeline + // grows a lead of three slots where A has no frame. + const shifted: AlignedFrameResolver = { + slotCount: ref(13), + frameRate: ref(2), + resolveSlot: (f: number) => ({ + A: f < 3 ? undefined : f - 3, + B: f < 10 ? f : undefined, + }), + resolveGlobalSlot: (camera: string, localFrame: number) => ( + camera === 'A' ? localFrame + 3 : localFrame), + gapSlots: ref([0, 1, 2, 10, 11, 12]), + }; + composable.setAlignedFrameResolver(shifted); + // A stays on local frame 5, now global slot 8; B follows to its frame there. + expect(composable.aggregateController.value.frame.value).toBe(8); + expect(mocks.seekA).toHaveBeenLastCalledWith(5); + expect(mocks.seekB).toHaveBeenLastCalledWith(8); + }); + it('re-applies the current aligned slot when a camera registers after the resolver is installed', async () => { const { composable, wrapper } = mountMediaController(); const resolver: AlignedFrameResolver = { diff --git a/client/src/components/annotators/useMediaController.ts b/client/src/components/annotators/useMediaController.ts index c904e4e04..bc6c58815 100644 --- a/client/src/components/annotators/useMediaController.ts +++ b/client/src/components/annotators/useMediaController.ts @@ -195,15 +195,29 @@ export function useMediaController(options?: { */ function setAlignedFrameResolver(resolver: AlignedFrameResolver | null) { stopAlignedPlaybackTimer(); + const previous = alignedFrameResolver.value; + const previousSlot = alignedCurrentFrame.value; alignedFrameResolver.value = resolver; - if (resolver) { - // Immediately perform an aligned seek to slot 0 so that any camera with - // no frame at that slot blanks right away, rather than continuing to - // show its own local frame 0 until the first user-driven seek. - alignedSeek(resolver, 0); - } else { + if (!resolver) { alignedCurrentFrame.value = 0; + return; + } + // A rebuilt timeline (e.g. a Time Offset nudge) keeps the instant on screen; a fresh one starts at 0. + let target = 0; + if (previous) { + const shown = previous.resolveSlot(previousSlot); + let kept: number | undefined; + subControllers.forEach((mc) => { + const local = shown[mc.cameraName.value]; + if (kept === undefined && local !== undefined) { + kept = resolver.resolveGlobalSlot(mc.cameraName.value, local); + } + }); + if (kept !== undefined) { + target = kept; + } } + alignedSeek(resolver, target); } /** From 909f9d63b98dc5138375684cd1570b6e4c6b2d6e Mon Sep 17 00:00:00 2001 From: romleiaj Date: Thu, 10 Sep 2026 11:19:32 -0400 Subject: [PATCH 13/39] Move the Time Offset to Multi Camera Tools and let it shift annotations The offset slider lived in the Camera Registration panel, tied to that panel's left/right pair selection. It now sits in Multi Camera Tools with one control per non-reference camera (the first camera in display order is the reference, as in registration), so it no longer depends on which pair is picked. A "Save annotations" button moves every annotation on a shifted camera by its offset so the boxes line up with that camera's own video, then saves. The part of each offset already applied is recorded as cameraFrameOffsetsApplied, so a later nudge only shifts by the difference and the button disables when nothing is pending. Features that would land before frame 0 are dropped. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ --- client/dive-common/apispec.ts | 4 +- .../CameraRegistration/RegistrationTools.vue | 99 +------------ .../dive-common/components/MultiCamTools.vue | 130 +++++++++++++++++- client/dive-common/components/Viewer.vue | 2 + .../frameOffsetAnnotations.spec.ts | 48 +++++++ client/dive-common/frameOffsetAnnotations.ts | 42 ++++++ .../alignedView/CameraRegistrationStore.ts | 7 + server/dive_utils/models.py | 2 + 8 files changed, 234 insertions(+), 100 deletions(-) create mode 100644 client/dive-common/frameOffsetAnnotations.spec.ts create mode 100644 client/dive-common/frameOffsetAnnotations.ts diff --git a/client/dive-common/apispec.ts b/client/dive-common/apispec.ts index 226171b34..537ac718d 100644 --- a/client/dive-common/apispec.ts +++ b/client/dive-common/apispec.ts @@ -334,9 +334,11 @@ interface DatasetConfigMutable { cameraRoles?: Record; /** Per-camera start offset in its own frames, for recorders that started at different times. */ cameraFrameOffsets?: Record; + /** The part of cameraFrameOffsets already applied to each camera's annotations. */ + cameraFrameOffsetsApplied?: Record; error?: string; } -const DatasetConfigMutableKeys = ['attributes', 'confidenceFilters', 'timeFilters', 'imageEnhancements', 'customTypeStyling', 'customGroupStyling', 'attributeTrackFilters', 'datasetInfo', 'cameraHomographies', 'cameraCorrespondences', 'cameraTransformTypes', 'cameraRegistrationSource', 'cameraFrameOffsets', 'typeHierarchy', 'taxonomySources', 'cameraRoles']; +const DatasetConfigMutableKeys = ['attributes', 'confidenceFilters', 'timeFilters', 'imageEnhancements', 'customTypeStyling', 'customGroupStyling', 'attributeTrackFilters', 'datasetInfo', 'cameraHomographies', 'cameraCorrespondences', 'cameraTransformTypes', 'cameraRegistrationSource', 'cameraFrameOffsets', 'cameraFrameOffsetsApplied', 'typeHierarchy', 'taxonomySources', 'cameraRoles']; /** * Cross-dataset color/style overrides, reused across every dataset when the * "shared" color scope is enabled (see clientSettings.typeSettings.colorScope). diff --git a/client/dive-common/components/CameraRegistration/RegistrationTools.vue b/client/dive-common/components/CameraRegistration/RegistrationTools.vue index 881a91c8c..60a89546c 100644 --- a/client/dive-common/components/CameraRegistration/RegistrationTools.vue +++ b/client/dive-common/components/CameraRegistration/RegistrationTools.vue @@ -7,7 +7,6 @@ import { useCameraStore, useCameraRegistration, useDatasetId, - useTime, } from 'vue-media-annotator/provides'; import { TransformType, TRANSFORM_TYPES, DEFAULT_TRANSFORM_TYPE, minPointsForTransform, @@ -562,56 +561,6 @@ export default defineComponent({ return cam; } - /** - * Start-offset control for the right camera, in its own frames. - * - * Two independently-started recorders on a fixed rig can be a fraction of - * a second apart for the whole clip. The homography can't reveal that -- - * the static background fits equally well at any offset -- so this is set - * by eye: turn Overlay Warp on and scrub, and moving subjects line up - * only at the right value. - * - * Only the right camera moves; the left stays the time reference. On a - * 3-camera rig each non-reference camera keeps its own offset, so - * switching the pair selector edits whichever camera is on the right. - */ - const OFFSET_LIMIT_SECONDS = 3; - const { frameRate } = useTime(); - const offsetLimit = computed( - () => Math.max(1, Math.round((frameRate.value || 30) * OFFSET_LIMIT_SECONDS)), - ); - const rightFrameOffset = computed({ - get: () => (camRight.value - ? registration.frameOffsets.value[camRight.value] ?? 0 - : 0), - set: (value: number) => { - const camera = camRight.value; - if (!camera) { - return; - } - // Replace the map rather than mutating it: the aligned timeline is a - // computed over this ref and only re-runs on identity change. - registration.frameOffsets.value = { - ...registration.frameOffsets.value, - [camera]: Math.round(value), - }; - }, - }); - /** e.g. "+12 frames (+0.400s)"; the seconds half needs a known frame rate. */ - const offsetReadout = computed(() => { - const frames = rightFrameOffset.value; - const sign = frames > 0 ? '+' : ''; - const plural = Math.abs(frames) === 1 ? '' : 's'; - const rate = frameRate.value; - const seconds = rate - ? ` (${sign}${(frames / rate).toFixed(3)}s)` - : ''; - return `${sign}${frames} frame${plural}${seconds}`; - }); - function nudgeOffset(delta: number) { - rightFrameOffset.value += delta; - } - /** Live cursor readout text: this camera's coord, and its linked point in the other camera. */ const cursorReadout = computed(() => { const cursor = registration.cursorCoord.value; @@ -700,6 +649,7 @@ export default defineComponent({ cameraTransformTypes: registration.transformTypes.value, cameraRegistrationSource: registration.source.value, cameraFrameOffsets: registration.frameOffsets.value, + cameraFrameOffsetsApplied: registration.appliedFrameOffsets.value, }); registration.markSaved(); } finally { @@ -818,10 +768,6 @@ export default defineComponent({ selectedCorrespondenceId, deleteSelectedCorrespondence, cursorReadout, - rightFrameOffset, - offsetLimit, - offsetReadout, - nudgeOffset, correspondences, pairStats, currentPairFrame, @@ -1450,49 +1396,6 @@ export default defineComponent({ - -

Time Offset

- - Frames to shift R so it plays in step with L. - -
- - - -
-
- {{ offsetReadout }} -
- - - import { computed, defineComponent, ref } from 'vue'; import { + useCameraRegistration, useCameraStore, + useDatasetId, useEditingMode, useHandler, useSelectedCamera, @@ -11,6 +13,9 @@ import { } from 'vue-media-annotator/provides'; import TooltipBtn from 'vue-media-annotator/components/TooltipButton.vue'; import { AnnotationId } from 'vue-media-annotator/BaseAnnotation'; +import Track from 'vue-media-annotator/track'; +import { useApi } from 'dive-common/apispec'; +import { pendingFrameShifts, shiftTrackData } from 'dive-common/frameOffsetAnnotations'; interface CameraTrackData { trackExists: boolean; @@ -25,10 +30,76 @@ export default defineComponent({ const inEditingMode = useEditingMode(); const enabledTracksRef = useTrackFilters().enabledAnnotations; const handler = useHandler(); - const { frame } = useTime(); + const { frame, frameRate } = useTime(); const selectedTrackId = useSelectedTrackId(); const cameraStore = useCameraStore(); const cameras = computed(() => cameraStore.orderedCameraNames()); + const registration = useCameraRegistration(); + const datasetId = useDatasetId(); + const { saveConfig } = useApi(); + + // Time offset: the first camera is the reference; every other camera is shifted onto it. + const OFFSET_LIMIT_SECONDS = 3; + const referenceCamera = computed(() => cameras.value[0] ?? null); + const offsetCameras = computed(() => cameras.value.slice(1)); + const offsetLimit = computed( + () => Math.max(1, Math.round((frameRate.value || 30) * OFFSET_LIMIT_SECONDS)), + ); + function cameraOffset(camera: string): number { + return registration.frameOffsets.value[camera] ?? 0; + } + function setCameraOffset(camera: string, value: number) { + // Replace the map: the aligned timeline is a computed over this ref. + registration.frameOffsets.value = { + ...registration.frameOffsets.value, + [camera]: Math.round(value), + }; + } + function nudgeOffset(camera: string, delta: number) { + setCameraOffset(camera, cameraOffset(camera) + delta); + } + function offsetReadout(camera: string): string { + const frames = cameraOffset(camera); + const sign = frames > 0 ? '+' : ''; + const plural = Math.abs(frames) === 1 ? '' : 's'; + const seconds = frameRate.value ? ` (${sign}${(frames / frameRate.value).toFixed(3)}s)` : ''; + return `${sign}${frames} frame${plural}${seconds}`; + } + const pendingShifts = computed(() => pendingFrameShifts( + registration.frameOffsets.value, + registration.appliedFrameOffsets.value, + offsetCameras.value, + )); + const savingOffsets = ref(false); + /** Move each shifted camera's annotations by its unapplied offset, then save everything. */ + async function saveAnnotationOffsets() { + savingOffsets.value = true; + try { + Object.entries(pendingShifts.value).forEach(([camera, delta]) => { + const store = cameraStore.camMap.value.get(camera)?.trackStore; + if (!store) { + return; + } + const tracks = Array.from(store.annotationMap.values()) as Track[]; + tracks.forEach((track) => { + const shifted = shiftTrackData(track.serialize(), delta); + store.remove(track.id, shifted !== null); + if (shifted !== null) { + store.insert(Track.fromJSON(shifted, track.set)); + } + }); + }); + registration.appliedFrameOffsets.value = { ...registration.frameOffsets.value }; + await saveConfig(datasetId.value, { + cameraFrameOffsets: registration.frameOffsets.value, + cameraFrameOffsetsApplied: registration.appliedFrameOffsets.value, + }); + registration.markSaved(); + await handler.save(); + } finally { + savingOffsets.value = false; + } + } const canary = ref(false); function _depend(): boolean { return canary.value; @@ -119,6 +190,16 @@ export default defineComponent({ deleteTrack, startLinking, handler, + referenceCamera, + offsetCameras, + offsetLimit, + cameraOffset, + setCameraOffset, + nudgeOffset, + offsetReadout, + pendingShifts, + savingOffsets, + saveAnnotationOffsets, }; }, }); @@ -130,6 +211,53 @@ export default defineComponent({ Multi Camera Tools for creating tracks, linking and unlinking tracks +
+

Time Offset

+ + Frames to shift each camera so it plays in step with {{ referenceCamera }}. + +
+ {{ camera }} +
+ + + +
+
+ {{ offsetReadout(camera) }} +
+
+ + Save annotations + + + Moves every annotation on a shifted camera by its offset, then saves. + +
Selected Track: {{ selectedTrackId }} Frame: {{ frame }} diff --git a/client/dive-common/components/Viewer.vue b/client/dive-common/components/Viewer.vue index ed1554ed3..db7d3906d 100644 --- a/client/dive-common/components/Viewer.vue +++ b/client/dive-common/components/Viewer.vue @@ -1535,6 +1535,7 @@ export default defineComponent({ cameraTransformTypes: cameraRegistration.transformTypes.value, cameraRegistrationSource: cameraRegistration.source.value, cameraFrameOffsets: cameraRegistration.frameOffsets.value, + cameraFrameOffsetsApplied: cameraRegistration.appliedFrameOffsets.value, }); cameraRegistration.markSaved(); } @@ -2079,6 +2080,7 @@ export default defineComponent({ meta.cameraTransformTypes, meta.cameraRegistrationSource, meta.cameraFrameOffsets, + meta.cameraFrameOffsetsApplied, ); // Media is loaded at this point: resolve observation frames from // their image names against this dataset's own frame ordering. diff --git a/client/dive-common/frameOffsetAnnotations.spec.ts b/client/dive-common/frameOffsetAnnotations.spec.ts new file mode 100644 index 000000000..f2e3b4380 --- /dev/null +++ b/client/dive-common/frameOffsetAnnotations.spec.ts @@ -0,0 +1,48 @@ +import { describe, expect, it } from 'vitest'; +import type { TrackData } from 'vue-media-annotator/track'; +import { pendingFrameShifts, shiftTrackData } from './frameOffsetAnnotations'; + +function track(frames: number[]): TrackData { + return { + id: 1, + attributes: {}, + confidencePairs: [['fish', 1]], + begin: Math.min(...frames), + end: Math.max(...frames), + features: frames.map((frame) => ({ frame, bounds: [0, 0, 1, 1] })), + }; +} + +describe('shiftTrackData', () => { + it('moves every feature and the bounds by the delta', () => { + const shifted = shiftTrackData(track([3, 4, 5]), 9); + expect(shifted?.begin).toBe(12); + expect(shifted?.end).toBe(14); + expect(shifted?.features.map((f) => f.frame)).toStrictEqual([12, 13, 14]); + }); + + it('drops features that would land before frame 0', () => { + const shifted = shiftTrackData(track([3, 4, 5]), -4); + expect(shifted?.features.map((f) => f.frame)).toStrictEqual([0, 1]); + expect(shifted?.begin).toBe(0); + expect(shifted?.end).toBe(1); + }); + + it('returns null when nothing survives', () => { + expect(shiftTrackData(track([0, 1]), -5)).toBeNull(); + }); + + it('is the identity at delta 0', () => { + const data = track([1]); + expect(shiftTrackData(data, 0)).toBe(data); + }); +}); + +describe('pendingFrameShifts', () => { + it('reports only the part of an offset not yet applied', () => { + expect(pendingFrameShifts({ IR: 9 }, {}, ['EO', 'IR'])).toStrictEqual({ IR: 9 }); + expect(pendingFrameShifts({ IR: 9 }, { IR: 9 }, ['EO', 'IR'])).toStrictEqual({}); + expect(pendingFrameShifts({ IR: 7 }, { IR: 9 }, ['EO', 'IR'])).toStrictEqual({ IR: -2 }); + expect(pendingFrameShifts({}, { IR: 9 }, ['EO', 'IR'])).toStrictEqual({ IR: -9 }); + }); +}); diff --git a/client/dive-common/frameOffsetAnnotations.ts b/client/dive-common/frameOffsetAnnotations.ts new file mode 100644 index 000000000..88cf65155 --- /dev/null +++ b/client/dive-common/frameOffsetAnnotations.ts @@ -0,0 +1,42 @@ +import type { TrackData } from 'vue-media-annotator/track'; + +/** + * Shift every frame of a serialized track by `delta`, dropping features that + * would land before frame 0. Returns null when nothing survives. + */ +export function shiftTrackData(track: TrackData, delta: number): TrackData | null { + if (delta === 0) { + return track; + } + const features = track.features + .filter((feature) => feature && feature.frame + delta >= 0) + .map((feature) => ({ ...feature, frame: feature.frame + delta })); + if (!features.length) { + return null; + } + return { + ...track, + features, + begin: Math.max(0, track.begin + delta), + end: Math.max(0, track.end + delta), + }; +} + +/** + * Frames each camera's annotations still need to move: its stored time + * offset minus what was already applied to them. Cameras in step are absent. + */ +export function pendingFrameShifts( + offsets: Record, + applied: Record, + cameras: string[], +): Record { + const shifts: Record = {}; + cameras.forEach((camera) => { + const delta = (offsets[camera] ?? 0) - (applied[camera] ?? 0); + if (delta !== 0) { + shifts[camera] = delta; + } + }); + return shifts; +} diff --git a/client/src/alignedView/CameraRegistrationStore.ts b/client/src/alignedView/CameraRegistrationStore.ts index e8e27875e..1018d0b75 100644 --- a/client/src/alignedView/CameraRegistrationStore.ts +++ b/client/src/alignedView/CameraRegistrationStore.ts @@ -329,6 +329,9 @@ export default class CameraRegistrationStore { */ frameOffsets: Ref>; + /** Per-camera frame offset already baked into that camera's annotations. */ + appliedFrameOffsets: Ref>; + /** True when the calibration has unsaved changes since the last save or load. */ dirty: ComputedRef; @@ -361,6 +364,7 @@ export default class CameraRegistrationStore { this.fitError = ref(null); this.source = ref(null); this.frameOffsets = ref({}); + this.appliedFrameOffsets = ref({}); this.nextId = 1; this.nextRecenterId = 1; this.homographySources = {}; @@ -377,6 +381,7 @@ export default class CameraRegistrationStore { transformTypes: this.transformTypes.value, source: this.source.value, frameOffsets: this.frameOffsets.value, + appliedFrameOffsets: this.appliedFrameOffsets.value, }); } @@ -1415,12 +1420,14 @@ export default class CameraRegistrationStore { transformTypes?: CameraTransformTypes, source?: RegistrationSource | null, frameOffsets?: Record | null, + appliedFrameOffsets?: Record | null, ) { this.homographies.value = homographies ? { ...homographies } : {}; this.observations.value = observations ? { ...observations } : {}; this.transformTypes.value = transformTypes ? { ...transformTypes } : {}; this.source.value = source ?? null; this.frameOffsets.value = frameOffsets ? { ...frameOffsets } : {}; + this.appliedFrameOffsets.value = appliedFrameOffsets ? { ...appliedFrameOffsets } : {}; this.markHomographySources(); this.activePair.value = null; this.pendingPoint.value = null; diff --git a/server/dive_utils/models.py b/server/dive_utils/models.py index 288017b8b..fea62c721 100644 --- a/server/dive_utils/models.py +++ b/server/dive_utils/models.py @@ -307,6 +307,8 @@ class MetadataMutable(BaseModel): cameraRoles: Optional[Dict[str, CameraRole]] # Per-camera start offset in its own frames, for recorders that started at different times. cameraFrameOffsets: Optional[Dict[str, int]] + # The part of cameraFrameOffsets already applied to each camera's annotations. + cameraFrameOffsetsApplied: Optional[Dict[str, int]] fps: Optional[float] @staticmethod From 6cfb9ff7b24d779835d25a3a3301435daa22107c Mon Sep 17 00:00:00 2001 From: romleiaj Date: Thu, 10 Sep 2026 11:52:53 -0400 Subject: [PATCH 14/39] Hold annotations at their instant while the time offset shifts the video Nudging the Time Offset moved a camera's video and its annotations together, so the boxes never disagreed with the frame under them and there was nothing to line up. Each camera pane now reads its annotations at `frame` minus that camera's unapplied offset, so while the reviewer dials the offset the boxes stay at the instant the reference camera shows and only the video underneath moves; Save annotations then bakes the offset in and the two coincide again. The first nudge on an uncorrected dataset also still jumped to frame 0: the earlier keep-the-instant fix only covered replacing one timeline with another, and a dataset with no offset has none, so a fresh install started at slot 0. Take the frame positional playback was showing in that case. Widen the slider to +/-10 s. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_01EKg5t3obzg2v7yrjCR47VZ --- .../dive-common/components/MultiCamTools.vue | 2 +- client/dive-common/components/Viewer.vue | 12 +++++ client/src/components/LayerManager.spec.ts | 2 + client/src/components/LayerManager.vue | 2 +- .../annotators/mediaControllerType.ts | 2 + .../annotators/useMediaController.spec.ts | 49 ++++++++++++++++--- .../annotators/useMediaController.ts | 35 +++++++------ 7 files changed, 80 insertions(+), 24 deletions(-) diff --git a/client/dive-common/components/MultiCamTools.vue b/client/dive-common/components/MultiCamTools.vue index f5f078ba8..533f1569f 100644 --- a/client/dive-common/components/MultiCamTools.vue +++ b/client/dive-common/components/MultiCamTools.vue @@ -39,7 +39,7 @@ export default defineComponent({ const { saveConfig } = useApi(); // Time offset: the first camera is the reference; every other camera is shifted onto it. - const OFFSET_LIMIT_SECONDS = 3; + const OFFSET_LIMIT_SECONDS = 10; const referenceCamera = computed(() => cameras.value[0] ?? null); const offsetCameras = computed(() => cameras.value.slice(1)); const offsetLimit = computed( diff --git a/client/dive-common/components/Viewer.vue b/client/dive-common/components/Viewer.vue index db7d3906d..ca8d3522d 100644 --- a/client/dive-common/components/Viewer.vue +++ b/client/dive-common/components/Viewer.vue @@ -95,6 +95,7 @@ import { import { usePrompt } from 'dive-common/vue-utilities/prompt-service'; import context from 'dive-common/store/context'; import { MarkChangesPendingFilter } from 'vue-media-annotator/BaseFilterControls'; +import { pendingFrameShifts } from 'dive-common/frameOffsetAnnotations'; import GroupSidebarVue from './GroupSidebar.vue'; import MultiCamToolsVue from './MultiCamTools.vue'; import RegistrationToolsVue from './CameraRegistration/RegistrationTools.vue'; @@ -242,6 +243,7 @@ export default defineComponent({ onResize, clear: mediaControllerClear, setAlignedFrameResolver, + setAnnotationFrameShifts, setResetZoomOverride, } = useMediaController({ segmentationCursorLoading }); const { time, updateTime, initialize: initTime } = useTimeObserver(); @@ -322,6 +324,16 @@ export default defineComponent({ }); return buildOffsetTimeline(frameCounts, cameraRegistration.frameOffsets.value); }); + // Until an offset is applied to a camera's annotations, they stay put while its video shifts. + watch( + () => pendingFrameShifts( + cameraRegistration.frameOffsets.value, + cameraRegistration.appliedFrameOffsets.value, + multiCamList.value, + ), + (shifts) => setAnnotationFrameShifts(shifts), + { immediate: true }, + ); // Serialized shape of the currently installed timeline. The computed // re-evaluates whenever any camera's imageData array identity changes -- // including pure display-URL swaps (e.g. the percentile-stretch remap) diff --git a/client/src/components/LayerManager.spec.ts b/client/src/components/LayerManager.spec.ts index dd5d3700c..f20a61d76 100644 --- a/client/src/components/LayerManager.spec.ts +++ b/client/src/components/LayerManager.spec.ts @@ -165,6 +165,7 @@ describe('LayerManager hierarchy frame data', () => { }; const annotator = { frame: ref(0), + annotationFrame: ref(0), flick: ref(0), hasFrame: ref(true), imageRevision: ref(0), @@ -283,6 +284,7 @@ function renderCamera( ) { const annotator = { frame: ref(0), + annotationFrame: ref(0), flick: ref(0), hasFrame: ref(true), imageRevision: ref(0), diff --git a/client/src/components/LayerManager.vue b/client/src/components/LayerManager.vue index c56b57b5f..c94887607 100644 --- a/client/src/components/LayerManager.vue +++ b/client/src/components/LayerManager.vue @@ -131,7 +131,7 @@ export default defineComponent({ const aggregateController = injectAggregateController(); const annotator = aggregateController.value.getController(props.camera); - const frameNumberRef = annotator.frame; + const frameNumberRef = annotator.annotationFrame; const flickNumberRef = annotator.flick; const hasFrameRef = annotator.hasFrame; diff --git a/client/src/components/annotators/mediaControllerType.ts b/client/src/components/annotators/mediaControllerType.ts index 002225814..e3f29bbc5 100644 --- a/client/src/components/annotators/mediaControllerType.ts +++ b/client/src/components/annotators/mediaControllerType.ts @@ -144,6 +144,8 @@ export interface MediaController extends AggregateMediaController { * draw annotations for the stale `frame` value left over from before. */ hasFrame: Readonly>; + /** Frame annotations are read at: `frame` minus this camera's unapplied time offset. */ + annotationFrame: Readonly>; /** * Bumped whenever the annotator redraws its media quad: the async * swap after a seek finishes loading, an image-enhancement change (the diff --git a/client/src/components/annotators/useMediaController.spec.ts b/client/src/components/annotators/useMediaController.spec.ts index 187845296..1dbab9975 100644 --- a/client/src/components/annotators/useMediaController.spec.ts +++ b/client/src/components/annotators/useMediaController.spec.ts @@ -1,4 +1,4 @@ -import { defineComponent, ref } from 'vue'; +import { defineComponent, ref, type Ref } from 'vue'; import { mount } from '@vue/test-utils'; import { useMediaController } from './useMediaController'; import type { AlignedFrameResolver } from './mediaControllerType'; @@ -254,10 +254,9 @@ describe('useMediaController', () => { expect(composable.aggregateController.value.alignedGapSlots.value).toEqual([]); }); - it('installing a resolver immediately performs an aligned seek to slot 0', () => { + it('installing a first resolver seeks to the slot showing the current frame', () => { const { composable, mocks } = mountMediaController(); - // Gap at slot 0: A has no frame there and must blank right away rather - // than continuing to show its local frame 0 until the first user seek. + // Both cameras sit at local frame 0; A's frame 0 lives at slot 1 here. const resolver: AlignedFrameResolver = { slotCount: ref(3), frameRate: ref(2), @@ -270,9 +269,9 @@ describe('useMediaController', () => { gapSlots: ref([0]), }; composable.setAlignedFrameResolver(resolver); - expect(mocks.seekA).toHaveBeenCalledWith(undefined); - expect(mocks.seekB).toHaveBeenCalledWith(0); - expect(composable.aggregateController.value.frame.value).toBe(0); + expect(mocks.seekA).toHaveBeenLastCalledWith(0); + expect(mocks.seekB).toHaveBeenLastCalledWith(1); + expect(composable.aggregateController.value.frame.value).toBe(1); }); it('replacing a resolver keeps the first camera\'s displayed frame instead of jumping to slot 0', () => { @@ -299,6 +298,42 @@ describe('useMediaController', () => { expect(mocks.seekB).toHaveBeenLastCalledWith(8); }); + it('a first resolver keeps the frame positional playback was showing', () => { + const { composable, mocks } = mountMediaController(); + // The mock annotators never report back, so place the frames directly. + (composable.aggregateController.value.getController('A').frame as Ref).value = 4; + (composable.aggregateController.value.getController('B').frame as Ref).value = 4; + // B now runs three frames ahead of A: the timeline gains a three-slot lead. + const shifted: AlignedFrameResolver = { + slotCount: ref(13), + frameRate: ref(2), + resolveSlot: (f: number) => ({ + A: f < 3 ? undefined : f - 3, + B: f < 10 ? f : undefined, + }), + resolveGlobalSlot: (camera: string, localFrame: number) => ( + camera === 'A' ? localFrame + 3 : localFrame), + gapSlots: ref([0, 1, 2, 10, 11, 12]), + }; + composable.setAlignedFrameResolver(shifted); + expect(composable.aggregateController.value.frame.value).toBe(7); + expect(mocks.seekA).toHaveBeenLastCalledWith(4); + expect(mocks.seekB).toHaveBeenLastCalledWith(7); + }); + + it('annotationFrame trails the video frame by the camera\'s unapplied shift', () => { + const { composable } = mountMediaController(); + const a = composable.aggregateController.value.getController('A'); + (a.frame as Ref).value = 12; + (composable.aggregateController.value.getController('B').frame as Ref).value = 12; + expect(a.annotationFrame.value).toBe(12); + composable.setAnnotationFrameShifts({ A: 9 }); + expect(a.annotationFrame.value).toBe(3); + expect(composable.aggregateController.value.getController('B').annotationFrame.value).toBe(12); + composable.setAnnotationFrameShifts({}); + expect(a.annotationFrame.value).toBe(12); + }); + it('re-applies the current aligned slot when a camera registers after the resolver is installed', async () => { const { composable, wrapper } = mountMediaController(); const resolver: AlignedFrameResolver = { diff --git a/client/src/components/annotators/useMediaController.ts b/client/src/components/annotators/useMediaController.ts index bc6c58815..b1af27848 100644 --- a/client/src/components/annotators/useMediaController.ts +++ b/client/src/components/annotators/useMediaController.ts @@ -144,6 +144,8 @@ export function useMediaController(options?: { // that must NOT be deep-reactive-converted/auto-unwrapped by a plain ref(). const alignedFrameResolver: Ref = shallowRef(null); const alignedCurrentFrame: Ref = ref(0); + // Per camera: frames its annotations lag its video by until a time offset is applied to them. + const annotationFrameShifts: Ref> = ref({}); const externallyDriven = computed(() => alignedFrameResolver.value !== null); const alignedGapSlots = computed(() => alignedFrameResolver.value?.gapSlots.value ?? []); let alignedPlaybackTimer: ReturnType | undefined; @@ -202,22 +204,21 @@ export function useMediaController(options?: { alignedCurrentFrame.value = 0; return; } - // A rebuilt timeline (e.g. a Time Offset nudge) keeps the instant on screen; a fresh one starts at 0. - let target = 0; - if (previous) { - const shown = previous.resolveSlot(previousSlot); - let kept: number | undefined; - subControllers.forEach((mc) => { - const local = shown[mc.cameraName.value]; - if (kept === undefined && local !== undefined) { - kept = resolver.resolveGlobalSlot(mc.cameraName.value, local); - } - }); - if (kept !== undefined) { - target = kept; + // Keep the instant on screen: the first camera's displayed frame, from the old timeline or positional playback. + const shown = previous ? previous.resolveSlot(previousSlot) : null; + let kept: number | undefined; + subControllers.forEach((mc) => { + const local = shown ? shown[mc.cameraName.value] : mc.frame.value; + if (kept === undefined && local !== undefined) { + kept = resolver.resolveGlobalSlot(mc.cameraName.value, local); } - } - alignedSeek(resolver, target); + }); + alignedSeek(resolver, kept ?? 0); + } + + /** Set how far each camera's annotations trail its video (see MediaController.annotationFrame). */ + function setAnnotationFrameShifts(shifts: Record) { + annotationFrameShifts.value = { ...shifts }; } /** @@ -696,6 +697,9 @@ export function useMediaController(options?: { speed: toRef(state[camera], 'speed'), syncedFrame: toRef(state[camera], 'syncedFrame'), hasFrame: toRef(state[camera], 'hasFrame'), + annotationFrame: computed( + () => state[camera].frame - (annotationFrameShifts.value[cameraName] ?? 0), + ), imageRevision: toRef(state[camera], 'imageRevision'), frameTexture: toRef(state[camera], 'frameTexture'), originalBounds: toRef(state[camera], 'originalBounds'), @@ -905,6 +909,7 @@ export function useMediaController(options?: { onResize, clear, setAlignedFrameResolver, + setAnnotationFrameShifts, setResetZoomOverride, }; } From 1abce4206d1f491f3e7e84162681a6dcfae9de2f Mon Sep 17 00:00:00 2001 From: romleiaj Date: Thu, 10 Sep 2026 13:12:55 -0400 Subject: [PATCH 15/39] Play offset video rigs natively and render video panes mid-seek Two playback problems with a camera time offset on a video rig. Playing fought itself. play() started every
diff --git a/client/dive-common/components/Viewer.vue b/client/dive-common/components/Viewer.vue index ca8d3522d..a5784ba32 100644 --- a/client/dive-common/components/Viewer.vue +++ b/client/dive-common/components/Viewer.vue @@ -2250,6 +2250,37 @@ export default defineComponent({ ? props.comparisonSets.slice(0, 1) : props.comparisonSets; }; + /** + * Replace one camera's in-memory annotations with what persistence now + * holds, e.g. after the server shifted them onto a time offset. Same + * bulk-insert path as loadData, so a large camera reloads without + * blocking the page; a comparison view falls back to a full reload + * because its extra sets are only rebuilt there. + */ + const reloadCameraAnnotations = async (camera: string) => { + const stores = cameraStore.camMap.value.get(camera); + if (!stores || props.comparisonSets.length) { + await reloadAnnotations(); + return; + } + handler.trackSelect(null, false); + const parentId = baseMulticamDatasetId.value ?? datasetId.value; + const cameraId = multiCamList.value.length > 1 ? `${parentId}/${camera}` : parentId; + const { tracks, groups } = await loadDetections(cameraId, props.revision, props.currentSet); + stores.trackStore.clearAll(); + stores.groupStore.clearAll(); + for (let j = 0; j < tracks.length; j += 1) { + if (j % 4000 === 0 && j > 0) { + // eslint-disable-next-line no-await-in-loop + await new Promise((resolve) => window.setTimeout(resolve, 0)); + } + stores.trackStore.insert(Track.fromJSON(tracks[j]), { imported: true }); + } + groups.forEach((group) => { + stores.groupStore.insert(Group.fromJSON(group), { imported: true }); + }); + }; + watch(datasetId, reloadAnnotations); watch(readonlyState, () => handler.trackSelect(null, false)); // Update segmentation recipe when frame changes to show only current frame's points @@ -2297,6 +2328,7 @@ export default defineComponent({ setAttribute, deleteAttribute, reloadAnnotations, + reloadCameraAnnotations, setSVGFilters, selectCamera, linkCameraTrack, diff --git a/client/dive-common/frameOffsetAnnotations.spec.ts b/client/dive-common/frameOffsetAnnotations.spec.ts index f2e3b4380..525e152a9 100644 --- a/client/dive-common/frameOffsetAnnotations.spec.ts +++ b/client/dive-common/frameOffsetAnnotations.spec.ts @@ -1,10 +1,13 @@ import { describe, expect, it } from 'vitest'; import type { TrackData } from 'vue-media-annotator/track'; -import { pendingFrameShifts, shiftTrackData } from './frameOffsetAnnotations'; +import type { GroupData } from 'vue-media-annotator/Group'; +import { + pendingFrameShifts, shiftAnnotationRecords, shiftGroupData, shiftTrackData, +} from './frameOffsetAnnotations'; -function track(frames: number[]): TrackData { +function track(frames: number[], id = 1): TrackData { return { - id: 1, + id, attributes: {}, confidencePairs: [['fish', 1]], begin: Math.min(...frames), @@ -13,6 +16,18 @@ function track(frames: number[]): TrackData { }; } +function group(members: GroupData['members'], id = 1): GroupData { + const ranges = Object.values(members).flatMap((m) => m.ranges); + return { + id, + attributes: {}, + confidencePairs: [['school', 1]], + begin: Math.min(...ranges.map(([b]) => b)), + end: Math.max(...ranges.map(([, e]) => e)), + members, + }; +} + describe('shiftTrackData', () => { it('moves every feature and the bounds by the delta', () => { const shifted = shiftTrackData(track([3, 4, 5]), 9); @@ -28,6 +43,13 @@ describe('shiftTrackData', () => { expect(shifted?.end).toBe(1); }); + it('rebounds to the first surviving feature across a gap', () => { + const shifted = shiftTrackData(track([3, 10]), -5); + expect(shifted?.features.map((f) => f.frame)).toStrictEqual([5]); + expect(shifted?.begin).toBe(5); + expect(shifted?.end).toBe(5); + }); + it('returns null when nothing survives', () => { expect(shiftTrackData(track([0, 1]), -5)).toBeNull(); }); @@ -38,6 +60,36 @@ describe('shiftTrackData', () => { }); }); +describe('shiftGroupData', () => { + it('moves and clips member ranges, dropping emptied members', () => { + const shifted = shiftGroupData(group({ + 1: { ranges: [[2, 6], [10, 20]] }, + 2: { ranges: [[0, 1]] }, + }), -4); + expect(shifted?.members).toStrictEqual({ 1: { ranges: [[0, 2], [6, 16]] } }); + expect(shifted?.begin).toBe(0); + expect(shifted?.end).toBe(16); + }); + + it('returns null when no member survives', () => { + expect(shiftGroupData(group({ 1: { ranges: [[0, 1]] } }), -2)).toBeNull(); + }); +}); + +describe('shiftAnnotationRecords', () => { + it('shifts a keyed annotation file and counts what fell off', () => { + const result = shiftAnnotationRecords( + { 1: track([3, 4], 1), 2: track([0], 2) }, + { 1: group({ 2: { ranges: [[0, 0]] } }) }, + -2, + ); + expect(Object.keys(result.tracks)).toStrictEqual(['1']); + expect(result.tracks[1].begin).toBe(1); + expect(result.groups).toStrictEqual({}); + expect(result.dropped).toBe(2); + }); +}); + describe('pendingFrameShifts', () => { it('reports only the part of an offset not yet applied', () => { expect(pendingFrameShifts({ IR: 9 }, {}, ['EO', 'IR'])).toStrictEqual({ IR: 9 }); diff --git a/client/dive-common/frameOffsetAnnotations.ts b/client/dive-common/frameOffsetAnnotations.ts index 88cf65155..d5c950233 100644 --- a/client/dive-common/frameOffsetAnnotations.ts +++ b/client/dive-common/frameOffsetAnnotations.ts @@ -1,8 +1,11 @@ import type { TrackData } from 'vue-media-annotator/track'; +import type { GroupData } from 'vue-media-annotator/Group'; /** * Shift every frame of a serialized track by `delta`, dropping features that - * would land before frame 0. Returns null when nothing survives. + * would land before frame 0. begin/end follow the surviving features, which + * is what the server's Track validator requires. Returns null when nothing + * survives. */ export function shiftTrackData(track: TrackData, delta: number): TrackData | null { if (delta === 0) { @@ -17,11 +20,73 @@ export function shiftTrackData(track: TrackData, delta: number): TrackData | nul return { ...track, features, - begin: Math.max(0, track.begin + delta), - end: Math.max(0, track.end + delta), + begin: features[0].frame, + end: features[features.length - 1].frame, }; } +/** + * Shift every member range of a serialized group by `delta`. Ranges ending + * before frame 0 are dropped, ranges straddling it are clipped, and members + * left with no range are removed. Returns null when no member survives. + */ +export function shiftGroupData(group: GroupData, delta: number): GroupData | null { + if (delta === 0) { + return group; + } + const members: GroupData['members'] = {}; + Object.entries(group.members).forEach(([memberId, member]) => { + const ranges = member.ranges + .filter(([, end]) => end + delta >= 0) + .map(([begin, end]) => [Math.max(0, begin + delta), end + delta] as [number, number]); + if (ranges.length) { + members[Number(memberId)] = { ...member, ranges }; + } + }); + const allRanges = Object.values(members).flatMap((member) => member.ranges); + if (!allRanges.length) { + return null; + } + return { + ...group, + members, + begin: Math.min(...allRanges.map(([begin]) => begin)), + end: Math.max(...allRanges.map(([, end]) => end)), + }; +} + +/** + * Shift a whole annotation file's tracks and groups by `delta`, keyed by id + * as the desktop track file stores them. `dropped` counts annotations that + * had nothing left after the shift. + */ +export function shiftAnnotationRecords( + tracks: Record, + groups: Record, + delta: number, +): { tracks: Record; groups: Record; dropped: number } { + let dropped = 0; + const shiftedTracks: Record = {}; + Object.entries(tracks).forEach(([id, track]) => { + const shifted = shiftTrackData(track, delta) as T | null; + if (shifted === null) { + dropped += 1; + } else { + shiftedTracks[id] = shifted; + } + }); + const shiftedGroups: Record = {}; + Object.entries(groups).forEach(([id, group]) => { + const shifted = shiftGroupData(group, delta) as G | null; + if (shifted === null) { + dropped += 1; + } else { + shiftedGroups[id] = shifted; + } + }); + return { tracks: shiftedTracks, groups: shiftedGroups, dropped }; +} + /** * Frames each camera's annotations still need to move: its stored time * offset minus what was already applied to them. Cameras in step are absent. diff --git a/client/platform/desktop/backend/native/common.ts b/client/platform/desktop/backend/native/common.ts index 02d13c896..5250631d8 100644 --- a/client/platform/desktop/backend/native/common.ts +++ b/client/platform/desktop/backend/native/common.ts @@ -30,6 +30,7 @@ import { PipelineParamType, FrameMetadataAttachmentText, FrameMetadataSourcesResponse, + CameraFrameOffsetResult, } from 'dive-common/apispec'; import { orderedMultiCamCameraNames } from 'dive-common/multicamDisplay'; import { parseCameraOrderHeader } from 'dive-common/pipelineCameraOrder'; @@ -44,6 +45,7 @@ import type { ScoringSource, ScoringSourceOptions, } from 'dive-common/scoring/types'; import { summarizeResult } from 'dive-common/scoring/metrics'; +import { shiftAnnotationRecords } from 'dive-common/frameOffsetAnnotations'; import * as viameSerializers from 'platform/desktop/backend/serializers/viame'; import * as nistSerializers from 'platform/desktop/backend/serializers/nist'; import * as dive from 'platform/desktop/backend/serializers/dive'; @@ -1404,6 +1406,13 @@ async function saveConfig(settings: Settings, datasetId: string, args: DatasetCo if (cameraRolesPresent && !cameraName) { existing.cameraRoles = args.cameraRoles; } + // Time offsets describe the rig, so they live on the multicamera parent only. + if (args.cameraFrameOffsets && !cameraName) { + existing.cameraFrameOffsets = args.cameraFrameOffsets; + } + if (args.cameraFrameOffsetsApplied && !cameraName) { + existing.cameraFrameOffsetsApplied = args.cameraFrameOffsetsApplied; + } // Registration files remain separate so each camera pair has one persisted owner. if (args.cameraHomographies || args.cameraCorrespondences || args.cameraTransformTypes @@ -1421,6 +1430,48 @@ async function saveConfig(settings: Settings, datasetId: string, args: DatasetCo } } +/** + * Bring one camera's stored annotations onto its time offset (the desktop + * twin of the server's apply_camera_frame_offset). Only the part not yet + * applied moves, and the offset plus its applied record are written together + * afterwards, so a failed shift never leaves the record ahead of the data. + */ +async function applyCameraFrameOffset( + settings: Settings, + datasetId: string, + camera: string, + offset: number, +): Promise { + const { parentId, cameraName } = parseCompositeDatasetId(datasetId); + if (cameraName) { + throw new Error('Time offsets are applied through the multicamera dataset, not a camera'); + } + const projectDirInfo = await getValidatedProjectDir(settings, parentId); + const config = await loadJsonConfig(projectDirInfo.datasetFileAbsPath); + if (!config.multiCam?.cameras[camera]) { + throw new Error(`Unknown camera "${camera}"`); + } + const delta = offset - (config.cameraFrameOffsetsApplied?.[camera] ?? 0); + const counts = { tracks: 0, groups: 0, dropped: 0 }; + if (delta !== 0) { + const cameraId = `${parentId}/${camera}`; + const cameraDir = await getValidatedProjectDir(settings, cameraId); + const existing = await loadAnnotationFile(cameraDir.trackFileAbsPath); + const { tracks, groups, dropped } = shiftAnnotationRecords(existing.tracks, existing.groups, delta); + await _saveSerialized(settings, cameraId, { ...existing, tracks, groups }); + counts.tracks = Object.keys(tracks).length; + counts.groups = Object.keys(groups).length; + counts.dropped = dropped; + } + await saveConfig(settings, parentId, { + cameraFrameOffsets: { ...config.cameraFrameOffsets, [camera]: offset }, + cameraFrameOffsetsApplied: { ...config.cameraFrameOffsetsApplied, [camera]: offset }, + }); + return { + camera, offset, delta, ...counts, + }; +} + async function saveAttributes(settings: Settings, datasetId: string, args: SaveAttributeArgs) { const projectDirData = await getValidatedProjectDir(settings, datasetId); const projectMetaData = await loadJsonConfig(projectDirData.datasetFileAbsPath); @@ -3130,6 +3181,7 @@ export { ingestDataFiles, saveDetections, saveConfig, + applyCameraFrameOffset, saveProjectConfig, processTrainedPipeline, findResumableTrainingJobs, diff --git a/client/platform/desktop/backend/server.ts b/client/platform/desktop/backend/server.ts index 0137321b5..3980387f4 100644 --- a/client/platform/desktop/backend/server.ts +++ b/client/platform/desktop/backend/server.ts @@ -117,6 +117,18 @@ apirouter.post('/dataset/:id{/:camera}/attribute_track_filters', async (req, res return null; }); +/* Shift one camera's annotations onto its time offset */ +apirouter.post('/dataset/:id/camera_frame_offset', async (req, res, next) => { + try { + const { camera, offset } = req.body as { camera: string; offset: number }; + const result = await common.applyCameraFrameOffset(settings.get(), req.params.id, camera, offset); + res.json(result); + } catch (err) { + (err as { status?: number }).status = 500; + next(err); + } +}); + /* SAVE detections */ apirouter.post('/dataset/:id{/:camera}/detections', async (req, res, next) => { try { diff --git a/client/platform/desktop/frontend/api.ts b/client/platform/desktop/frontend/api.ts index 74d87a1aa..d763a0a29 100644 --- a/client/platform/desktop/frontend/api.ts +++ b/client/platform/desktop/frontend/api.ts @@ -12,6 +12,7 @@ import type { PipelineJobResult, ScoringDatasetSummary, ScoringJobArgs, ScoringResult, ScoringResultSummary, ScoringSourceOptions, VideoSearchIndexStatus, VideoSearchIndexMethod, VideoSearchQueryResponse, VideoSearchIndexInfo, + CameraFrameOffsetResult, } from 'dive-common/apispec'; import axios, { AxiosInstance } from 'axios'; import { watch } from 'vue'; @@ -976,6 +977,15 @@ async function saveConfig(id: string, args: DatasetConfigMutable) { return client.post(`dataset/${id}/meta`, args); } +async function applyCameraFrameOffset(id: string, camera: string, offset: number) { + const client = await getClient(); + const { data } = await client.post( + `dataset/${id}/camera_frame_offset`, + { camera, offset }, + ); + return data; +} + async function saveDetections(id: string, args: SaveDetectionsArgs) { const client = await getClient(); return client.post(`dataset/${id}/detections`, args); @@ -1081,6 +1091,7 @@ export { loadScoringResult, deleteScoringResult, saveConfig, + applyCameraFrameOffset, saveDetections, saveAttributes, saveAttributeTrackFilters, diff --git a/client/platform/web-girder/App.vue b/client/platform/web-girder/App.vue index a2874863c..2bc24d958 100644 --- a/client/platform/web-girder/App.vue +++ b/client/platform/web-girder/App.vue @@ -26,6 +26,7 @@ import { getTrainingConfigurations, runTraining, saveConfig, + applyCameraFrameOffset, loadGlobalStyleSettings, saveGlobalStyleSettings, saveAttributes, @@ -105,6 +106,7 @@ export default defineComponent({ loadFrameMetadata, saveDetections: unwrap(saveDetections), saveConfig: unwrap(saveConfig), + applyCameraFrameOffset: unwrap(applyCameraFrameOffset), loadGlobalStyleSettings, saveGlobalStyleSettings, saveAttributes: unwrap(saveAttributes), diff --git a/client/platform/web-girder/api/dataset.service.ts b/client/platform/web-girder/api/dataset.service.ts index c1c4390c1..698b10c3c 100644 --- a/client/platform/web-girder/api/dataset.service.ts +++ b/client/platform/web-girder/api/dataset.service.ts @@ -5,7 +5,7 @@ import { registrationValuesSummary, filterRegistrationValues, mergeRegistrationValues, } from 'vue-media-annotator/alignedView/cameraRegistrationFiles'; import { - DatasetConfigMutable, DatasetType, FrameImage, GlobalStyleSettings, + CameraFrameOffsetResult, DatasetConfigMutable, DatasetType, FrameImage, GlobalStyleSettings, SaveAttributeArgs, SaveAttributeTrackFilterArgs, } from 'dive-common/apispec'; import { @@ -261,6 +261,15 @@ async function saveConfig(datasetId: string, config: DatasetConfigMutable) { return girderRest.patch(`/dive_dataset/${folderId}`, config); } +/** Shift one camera's stored annotations onto its time offset; the server does the work. */ +async function applyCameraFrameOffset(datasetId: string, camera: string, offset: number) { + const { folderId } = await resolveDatasetFolderId(datasetId); + return girderRest.patch( + `/dive_dataset/${folderId}/camera_frame_offset`, + { camera, offset }, + ); +} + // Cross-dataset "shared" color/style overrides. Persisted in localStorage, // consistent with how the web client already stores user preferences // (clientSettings). This scopes shared colors to the current user/browser. @@ -555,6 +564,7 @@ export { saveAttributes, saveAttributeTrackFilters, saveConfig, + applyCameraFrameOffset, loadGlobalStyleSettings, saveGlobalStyleSettings, uploadCalibrationItem, diff --git a/client/src/alignedView/CameraRegistrationStore.ts b/client/src/alignedView/CameraRegistrationStore.ts index 1018d0b75..fe5466854 100644 --- a/client/src/alignedView/CameraRegistrationStore.ts +++ b/client/src/alignedView/CameraRegistrationStore.ts @@ -390,6 +390,17 @@ export default class CameraRegistrationStore { this.savedSnapshot.value = this.registrationSnapshot(); } + /** + * Record that `camera`'s time offset was persisted and applied, leaving + * the rest of the baseline untouched so other unsaved edits stay dirty. + */ + markFrameOffsetSaved(camera: string, offset: number) { + const saved = JSON.parse(this.savedSnapshot.value); + saved.frameOffsets = { ...saved.frameOffsets, [camera]: offset }; + saved.appliedFrameOffsets = { ...saved.appliedFrameOffsets, [camera]: offset }; + this.savedSnapshot.value = JSON.stringify(saved); + } + /** * The saved-baseline calibration (what the persisted registration -- the * per-camera registration files on desktop, the dataset meta on web -- diff --git a/client/src/provides.ts b/client/src/provides.ts index 449894b9b..d1b07c6c3 100644 --- a/client/src/provides.ts +++ b/client/src/provides.ts @@ -217,6 +217,8 @@ export interface Handler { unstageFromMerge(ids: AnnotationId[]): void; /* Reload Annotation File */ reloadAnnotations(): Promise; + /* Replace one camera's in-memory annotations with what persistence now holds */ + reloadCameraAnnotations(camera: string): Promise; setSVGFilters({ brightness, contrast, saturation, sharpen, percentileStretch, }: { @@ -281,6 +283,7 @@ function dummyHandler(handle: (name: string, args: unknown[]) => void): Handler groupEdit(...args) { handle('groupEdit', args); }, unstageFromMerge(...args) { handle('unstageFromMerge', args); }, reloadAnnotations(...args) { handle('reloadTracks', args); return Promise.resolve(); }, + reloadCameraAnnotations(...args) { handle('reloadCameraAnnotations', args); return Promise.resolve(); }, setSVGFilters(...args) { handle('setSVGFilter', args); }, unlinkCameraTrack(...args) { handle('unlinkCameraTrack', args); }, linkCameraTrack(...args) { handle('linkCameraTrack', args); }, diff --git a/server/dive_server/crud_annotation.py b/server/dive_server/crud_annotation.py index a4a12f482..ba5c303cc 100644 --- a/server/dive_server/crud_annotation.py +++ b/server/dive_server/crud_annotation.py @@ -482,3 +482,109 @@ def get_labels(user: types.GirderUserModel, published=False, shared=False): {'$sort': {'_id': 1}}, ] return Folder().collection.aggregate(pipeline) + + +def shift_track_frames(track: dict, delta: int) -> Optional[dict]: + """Move every feature of a serialized track by ``delta`` frames. + + Features that would land before frame 0 are dropped, and begin/end are + recomputed from what survives so the Track validator's bounds check still + holds. Returns None when nothing survives. + """ + if delta == 0: + return track + features = [ + {**feature, 'frame': feature['frame'] + delta} + for feature in track.get('features', []) + if feature['frame'] + delta >= 0 + ] + if not features: + return None + return { + **track, + 'features': features, + 'begin': features[0]['frame'], + 'end': features[-1]['frame'], + } + + +def shift_group_frames(group: dict, delta: int) -> Optional[dict]: + """Move every member range of a serialized group by ``delta`` frames. + + Ranges ending before frame 0 are dropped, ranges straddling it are + clipped, and members left with no range are removed. Returns None when + no member survives. + """ + if delta == 0: + return group + members = {} + for member_id, member in group.get('members', {}).items(): + ranges = [ + [max(0, begin + delta), end + delta] + for begin, end in member.get('ranges', []) + if end + delta >= 0 + ] + if ranges: + members[member_id] = {**member, 'ranges': ranges} + if not members: + return None + all_ranges = [r for member in members.values() for r in member['ranges']] + return { + **group, + 'members': members, + 'begin': min(r[0] for r in all_ranges), + 'end': max(r[1] for r in all_ranges), + } + + +def shift_annotation_frames( + dsFolder: types.GirderModel, + user: types.GirderUserModel, + delta: int, +) -> dict: + """Move every annotation in a dataset, across all of its sets, by ``delta`` frames. + + Written as one new revision per set through save_annotations, so the + move is undoable with rollback like any other edit. Every set is shifted + because they all annotate the same media: leaving one behind would put + it out of step with the video it describes. + """ + counts = {'tracks': 0, 'groups': 0, 'dropped': 0} + if delta == 0: + return counts + sets: List[Optional[str]] = [None] + sets.extend(s for s in RevisionLogItem().sets(dsFolder) if s) + description = f'shift frames by {delta:+d}' + for annotation_set in sets: + upsert_tracks: List[dict] = [] + delete_tracks: List[int] = [] + for track in TrackItem().list(dsFolder, set=annotation_set): + shifted = shift_track_frames(track, delta) + if shifted is None: + delete_tracks.append(track[IDENTIFIER]) + else: + upsert_tracks.append(shifted) + upsert_groups: List[dict] = [] + delete_groups: List[int] = [] + for group in GroupItem().list(dsFolder, set=annotation_set): + shifted = shift_group_frames(group, delta) + if shifted is None: + delete_groups.append(group[IDENTIFIER]) + else: + upsert_groups.append(shifted) + if not (upsert_tracks or delete_tracks or upsert_groups or delete_groups): + continue + save_annotations( + dsFolder, + user, + upsert_tracks=upsert_tracks, + delete_tracks=delete_tracks, + upsert_groups=upsert_groups, + delete_groups=delete_groups, + description=description, + set=annotation_set or '', + ) + counts['tracks'] += len(upsert_tracks) + counts['groups'] += len(upsert_groups) + counts['dropped'] += len(delete_tracks) + len(delete_groups) + return counts diff --git a/server/dive_server/crud_dataset.py b/server/dive_server/crud_dataset.py index 99484cf2c..aee42ad7e 100644 --- a/server/dive_server/crud_dataset.py +++ b/server/dive_server/crud_dataset.py @@ -2163,3 +2163,41 @@ def set_metadata_file( 'metadataFileItemId': str(md_item['_id']), 'metadataFileOriginalName': md_item['name'], } + + +def apply_camera_frame_offset( + parent: types.GirderModel, + user: types.GirderUserModel, + camera: str, + offset: int, +) -> dict: + """Bring one camera's stored annotations onto its time offset. + + The offset is the camera's start offset in its own frames (see the + client's alignedTimeline.buildOffsetTimeline). Only the part not yet + applied to the annotations is shifted, and the offset plus its applied + record are written together afterwards, so a failed shift never leaves + the record ahead of the data. + """ + crud.verify_dataset(parent) + if fromMeta(parent, constants.TypeMarker) != constants.MultiType: + raise RestException('Time offsets apply to multicamera datasets only', code=400) + multi_cam = fromMeta(parent, constants.MultiCamMarker) or {} + cam_info = (multi_cam.get('cameras') or {}).get(camera) + if not cam_info: + raise RestException(f'Unknown camera "{camera}"', code=400) + child = Folder().load(cam_info['folderId'], level=AccessType.WRITE, user=user) + if child is None: + raise RestException(f'Camera folder for "{camera}" was not found', code=404) + offsets = dict(fromMeta(parent, 'cameraFrameOffsets', {}) or {}) + applied = dict(fromMeta(parent, 'cameraFrameOffsetsApplied', {}) or {}) + delta = offset - applied.get(camera, 0) + counts = crud_annotation.shift_annotation_frames(child, user, delta) + offsets[camera] = offset + applied[camera] = offset + update_metadata( + parent, + {'cameraFrameOffsets': offsets, 'cameraFrameOffsetsApplied': applied}, + verify=False, + ) + return {'camera': camera, 'offset': offset, 'delta': delta, **counts} diff --git a/server/dive_server/views_dataset.py b/server/dive_server/views_dataset.py index e71e558d9..db857a9b3 100644 --- a/server/dive_server/views_dataset.py +++ b/server/dive_server/views_dataset.py @@ -61,6 +61,7 @@ def __init__(self, resourceName): self.route("POST", ("validate_files",), self.validate_files) self.route("PATCH", (":id",), self.patch_metadata) + self.route("PATCH", (":id", "camera_frame_offset"), self.apply_camera_frame_offset) # do we make this another resource in girder? self.route("PATCH", (":id", "attributes"), self.patch_attributes) @@ -469,6 +470,26 @@ def patch_metadata(self, folder, data): crud_dataset.remove_camera_type_hierarchy(folder) return result + @access.user + @autoDescribeRoute( + Description("Shift one camera's annotations onto its time offset") + .modelParam("id", level=AccessType.WRITE, **DatasetModelParam) + .jsonParam( + "data", + description="{camera: string, offset: integer frames}", + requireObject=True, + paramType="body", + ) + ) + def apply_camera_frame_offset(self, folder, data): + camera = data.get('camera') + offset = data.get('offset') + if not isinstance(camera, str) or not camera: + raise RestException('"camera" must be a camera name', code=400) + if isinstance(offset, bool) or not isinstance(offset, int): + raise RestException('"offset" must be a whole number of frames', code=400) + return crud_dataset.apply_camera_frame_offset(folder, self.getCurrentUser(), camera, offset) + @access.user @autoDescribeRoute( Description("Update set of possible attributes") diff --git a/server/tests/test_camera_frame_offset.py b/server/tests/test_camera_frame_offset.py new file mode 100644 index 000000000..e8b1abf48 --- /dev/null +++ b/server/tests/test_camera_frame_offset.py @@ -0,0 +1,161 @@ +from unittest.mock import MagicMock, patch + +from girder.exceptions import RestException +import pytest + +from dive_server import crud_annotation, crud_dataset +from dive_server.crud_annotation import IDENTIFIER +from dive_utils import constants + + +def _track(track_id, frames): + return { + 'id': track_id, + 'begin': frames[0], + 'end': frames[-1], + 'confidencePairs': [['fish', 1.0]], + 'attributes': {}, + 'features': [{'frame': frame, 'bounds': [0, 0, 1, 1]} for frame in frames], + } + + +class TestShiftTrackFrames: + def test_moves_features_and_bounds(self): + shifted = crud_annotation.shift_track_frames(_track(1, [3, 4, 5]), 9) + assert [f['frame'] for f in shifted['features']] == [12, 13, 14] + assert (shifted['begin'], shifted['end']) == (12, 14) + + def test_drops_features_before_frame_zero(self): + shifted = crud_annotation.shift_track_frames(_track(1, [3, 4, 5]), -4) + assert [f['frame'] for f in shifted['features']] == [0, 1] + assert (shifted['begin'], shifted['end']) == (0, 1) + + def test_rebounds_to_the_first_surviving_feature(self): + # A gap after the dropped feature: begin follows the data, not a clamp to 0. + shifted = crud_annotation.shift_track_frames(_track(1, [3, 10]), -5) + assert [f['frame'] for f in shifted['features']] == [5] + assert (shifted['begin'], shifted['end']) == (5, 5) + + def test_none_when_nothing_survives(self): + assert crud_annotation.shift_track_frames(_track(1, [0, 1]), -5) is None + + def test_identity_at_zero(self): + track = _track(1, [1]) + assert crud_annotation.shift_track_frames(track, 0) is track + + +class TestShiftGroupFrames: + def test_moves_and_clips_member_ranges(self): + group = { + 'id': 1, + 'begin': 2, + 'end': 20, + 'members': {'1': {'ranges': [[2, 6], [10, 20]]}, '2': {'ranges': [[0, 1]]}}, + } + shifted = crud_annotation.shift_group_frames(group, -4) + assert shifted['members'] == {'1': {'ranges': [[0, 2], [6, 16]]}} + assert (shifted['begin'], shifted['end']) == (0, 16) + + def test_none_when_no_member_survives(self): + group = {'id': 1, 'begin': 0, 'end': 1, 'members': {'1': {'ranges': [[0, 1]]}}} + assert crud_annotation.shift_group_frames(group, -2) is None + + +@patch('dive_server.crud_annotation.save_annotations') +@patch('dive_server.crud_annotation.GroupItem') +@patch('dive_server.crud_annotation.TrackItem') +@patch('dive_server.crud_annotation.RevisionLogItem') +def test_shift_annotation_frames_writes_every_set(revision_log, track_item, group_item, save): + folder = {'_id': 'camera-id'} + revision_log.return_value.sets.return_value = [None, 'review'] + track_item.return_value.list.side_effect = lambda _f, set=None: [ + _track(1, [3, 4]), + _track(2, [0]), + ] + group_item.return_value.list.return_value = [] + + counts = crud_annotation.shift_annotation_frames(folder, {'login': 'u'}, -2) + + assert counts == {'tracks': 2, 'groups': 0, 'dropped': 2} + assert save.call_count == 2 + assert [call.kwargs['set'] for call in save.call_args_list] == ['', 'review'] + first = save.call_args_list[0].kwargs + assert [t[IDENTIFIER] for t in first['upsert_tracks']] == [1] + assert first['upsert_tracks'][0]['begin'] == 1 + assert first['delete_tracks'] == [2] + assert first['description'] == 'shift frames by -2' + + +@patch('dive_server.crud_annotation.save_annotations') +@patch('dive_server.crud_annotation.TrackItem') +def test_shift_annotation_frames_is_a_no_op_at_zero(track_item, save): + counts = crud_annotation.shift_annotation_frames({'_id': 'x'}, {'login': 'u'}, 0) + assert counts == {'tracks': 0, 'groups': 0, 'dropped': 0} + track_item.return_value.list.assert_not_called() + save.assert_not_called() + + +def _multi_parent(applied=None): + return { + '_id': 'parent-id', + 'meta': { + 'annotate': True, + 'type': constants.MultiType, + 'fps': 5, + 'multiCam': { + 'defaultDisplay': 'EO', + 'cameras': { + 'EO': {'folderId': 'eo-id', 'type': 'video'}, + 'IR': {'folderId': 'ir-id', 'type': 'video'}, + }, + }, + **({'cameraFrameOffsetsApplied': applied} if applied else {}), + }, + } + + +@patch('dive_server.crud_dataset.update_metadata') +@patch('dive_server.crud_dataset.crud_annotation.shift_annotation_frames') +@patch('dive_server.crud_dataset.Folder') +def test_apply_camera_frame_offset_shifts_only_the_unapplied_part(folder_cls, shift, update): + parent = _multi_parent(applied={'IR': 4}) + child = {'_id': 'ir-id'} + folder_cls.return_value.load.return_value = child + shift.return_value = {'tracks': 3, 'groups': 0, 'dropped': 1} + user = {'login': 'u'} + + result = crud_dataset.apply_camera_frame_offset(parent, user, 'IR', 9) + + shift.assert_called_once_with(child, user, 5) + update.assert_called_once_with( + parent, + {'cameraFrameOffsets': {'IR': 9}, 'cameraFrameOffsetsApplied': {'IR': 9}}, + verify=False, + ) + assert result == { + 'camera': 'IR', + 'offset': 9, + 'delta': 5, + 'tracks': 3, + 'groups': 0, + 'dropped': 1, + } + + +@patch('dive_server.crud_dataset.update_metadata') +@patch('dive_server.crud_dataset.crud_annotation.shift_annotation_frames') +@patch('dive_server.crud_dataset.Folder') +def test_apply_camera_frame_offset_rejects_unknown_camera(folder_cls, shift, update): + with pytest.raises(RestException): + crud_dataset.apply_camera_frame_offset(_multi_parent(), {'login': 'u'}, 'UV', 1) + shift.assert_not_called() + update.assert_not_called() + + +@patch('dive_server.crud_dataset.update_metadata') +@patch('dive_server.crud_dataset.crud_annotation.shift_annotation_frames') +def test_apply_camera_frame_offset_rejects_single_camera_dataset(shift, update): + folder = {'_id': 'x', 'meta': {'annotate': True, 'type': constants.VideoType, 'fps': 5}} + with pytest.raises(RestException): + crud_dataset.apply_camera_frame_offset(folder, {'login': 'u'}, 'EO', 1) + shift.assert_not_called() From 47adb618ca143dc94d50eecf861246449d82c9c5 Mon Sep 17 00:00:00 2001 From: romleiaj Date: Thu, 17 Sep 2026 11:39:18 -0400 Subject: [PATCH 21/39] Load cameras in parallel and mount panes before annotations insert A multicamera dataset loaded one camera after another: config, then annotations, then track insertion with a hard 500 ms sleep every 4000 tracks, and only once every camera was through did the annotator panes mount and the videos begin to download. Two large cameras spent seconds sleeping and seconds waiting on serialized requests before either video had fetched a byte. Every camera's config and annotations are now requested at once, the media is applied for all cameras first, and a new progress.mediaLoaded flag mounts the panes at that point so the videos fetch their metadata while the annotations are still being inserted. The insertion loop yields with a zero-delay timeout instead of sleeping; the yield was only ever there to let the page paint. A progress ring overlays the panes until the annotations are in, and the pane keybindings stay off until then. Both transcodes also write the MP4 index up front (faststart), so a browser can read the video's metadata from the first bytes instead of reaching to the end of the file first. Co-Authored-By: Claude Fable 5.1 --- client/dive-common/components/Viewer.vue | 106 +++++++++++++----- .../desktop/backend/native/mediaJobs.ts | 2 + server/dive_tasks/convert_video.py | 3 + 3 files changed, 83 insertions(+), 28 deletions(-) diff --git a/client/dive-common/components/Viewer.vue b/client/dive-common/components/Viewer.vue index a5784ba32..a62bac251 100644 --- a/client/dive-common/components/Viewer.vue +++ b/client/dive-common/components/Viewer.vue @@ -265,6 +265,9 @@ export default defineComponent({ // with stale data from props, for example if a persistent store // like vuex is used to drive them. loaded: false, + // Every camera's media is known, so the panes can mount and start fetching + // video while the annotations are still being inserted. + mediaLoaded: false, // Tracks loaded progress: 0, // Total tracks @@ -1774,6 +1777,8 @@ export default defineComponent({ emit('change-camera', camera); }; /** Trigger data load */ + /** Let the browser paint and run other tasks between batches of track inserts. */ + const yieldToBrowser = () => new Promise((resolve) => { window.setTimeout(resolve, 0); }); const loadData = async () => { annotationUndo.reset(); try { @@ -1936,14 +1941,22 @@ export default defineComponent({ imageData.value = Object.fromEntries( multiCamList.value.map((camera) => [camera, [] as FrameImage[]]), ); - for (let i = 0; i < multiCamList.value.length; i += 1) { - const camera = multiCamList.value[i]; - let cameraId = baseMulticamDatasetId.value; - if (multiCamList.value.length > 1) { - cameraId = `${baseMulticamDatasetId.value}/${camera}`; - } - // eslint-disable-next-line no-await-in-loop - const subCameraMeta = await loadConfig(cameraId); + // Fetch every camera's config and annotations at once instead of one + // camera after another; the per-camera work below only applies them. + const cameraLoads = await Promise.all(multiCamList.value.map(async (camera) => { + const cameraId = multiCamList.value.length > 1 + ? `${datasetId.value}/${camera}` + : datasetId.value; + const [subCameraMeta, detections] = await Promise.all([ + loadConfig(cameraId), + loadDetections(cameraId, props.revision, props.currentSet), + ]); + return { + camera, cameraId, subCameraMeta, detections, + }; + })); + for (let i = 0; i < cameraLoads.length; i += 1) { + const { camera, subCameraMeta } = cameraLoads[i]; VueSet(cameraTypesByCamera.value, camera, subCameraMeta.type as DatasetType); if (multiCamList.value.length <= 1) { datasetType.value = subCameraMeta.type as DatasetType; @@ -1968,12 +1981,13 @@ export default defineComponent({ } cameraStore.addCamera(camera); addSaveCamera(camera); - const { - tracks, - groups, - sets: foundSets, - // eslint-disable-next-line no-await-in-loop - } = await loadDetections(cameraId, props.revision, props.currentSet); + } + // Media is known for every camera: mount the panes now so the videos + // fetch their metadata while the annotations are still being inserted. + progress.mediaLoaded = true; + for (let i = 0; i < cameraLoads.length; i += 1) { + const { camera, cameraId, detections } = cameraLoads[i]; + const { tracks, groups, sets: foundSets } = detections; sets.value = foundSets.filter((item) => item); if (props.currentSet !== '' || sets.value.length > 0) { sets.value.push('default'); @@ -1998,7 +2012,7 @@ export default defineComponent({ /* Every N tracks, yeild some cycles for other scheduled tasks */ progress.progress = j; // eslint-disable-next-line no-await-in-loop - await new Promise((resolve) => window.setTimeout(resolve, 500)); + await yieldToBrowser(); } trackStore.insert(Track.fromJSON(tracks[j], baseSet), { imported: true }); } @@ -2007,7 +2021,7 @@ export default defineComponent({ /* Every N tracks, yeild some cycles for other scheduled tasks */ progress.progress = tracks.length + j; // eslint-disable-next-line no-await-in-loop - await new Promise((resolve) => window.setTimeout(resolve, 500)); + await yieldToBrowser(); } groupStore.insert(Group.fromJSON(groups[j]), { imported: true }); } @@ -2035,7 +2049,7 @@ export default defineComponent({ /* Every N tracks, yeild some cycles for other scheduled tasks */ progress.progress = j; // eslint-disable-next-line no-await-in-loop - await new Promise((resolve) => window.setTimeout(resolve, 500)); + await yieldToBrowser(); } // We need to increment the trackIds for the new comparison sets setTracks[j].id = trackStore.getNewId(); @@ -2189,6 +2203,7 @@ export default defineComponent({ } } catch (err) { progress.loaded = false; + progress.mediaLoaded = false; console.error(err); const errorEl = document.createElement('div'); errorEl.innerHTML = getResponseError(err); @@ -2231,6 +2246,7 @@ export default defineComponent({ const reloadAnnotations = async () => { progress.loaded = false; + progress.mediaLoaded = false; discardChanges(); Object.values(debouncedSaves).forEach((fn) => fn.cancel()); Object.keys(debouncedSaves).forEach((k) => delete debouncedSaves[k]); @@ -2272,7 +2288,7 @@ export default defineComponent({ for (let j = 0; j < tracks.length; j += 1) { if (j % 4000 === 0 && j > 0) { // eslint-disable-next-line no-await-in-loop - await new Promise((resolve) => window.setTimeout(resolve, 0)); + await yieldToBrowser(); } stores.trackStore.insert(Track.fromJSON(tracks[j]), { imported: true }); } @@ -3041,13 +3057,13 @@ export default defineComponent({ dense >
@@ -3068,7 +3084,7 @@ export default defineComponent({ >
+ + Loading + {{ progressValue }}% +
@@ -3167,7 +3196,7 @@ export default defineComponent({ >
+ + Loading + {{ progressValue }}% +
@@ -3280,6 +3322,14 @@ export default defineComponent({