diff --git a/THIRD-PARTY-NOTICES.md b/THIRD-PARTY-NOTICES.md index f80ec2361..b2611d75b 100644 --- a/THIRD-PARTY-NOTICES.md +++ b/THIRD-PARTY-NOTICES.md @@ -6,7 +6,9 @@ application resources and satisfies the attribution and source-offer obligations that come with them. npm dependencies are not listed here: they are resolved from `package.json` and -distributed by their own registries, not redistributed inside our binaries. +distributed by their own registries, not redistributed inside our binaries. The +exception is code bundled into the renderer whose licence asks for its notice in +the distributed form (js-aruco2 below): the bundler strips the source headers. --- @@ -172,6 +174,49 @@ distributed by their own registries, not redistributed inside our binaries. notice as OpenScreen's own [LICENSE](LICENSE). Published by Recordly under MIT, before its relicensing in March 2026. +## js-aruco2 — ArUco marker detection + +- **Components**: `src/aruco.js` and `src/cv.js` of the npm package `js-aruco2` + 2.0.0, compiled into the renderer bundle (`dist/assets/*.js`). +- **Used by**: the camera calibration dialog, which finds the four printed corner + markers in a camera still. +- **License**: MIT — Copyright (c) 2020 Damiano Falcioni, Copyright (c) 2011 Juan + Mellado, under the same permission notice as OpenScreen's own + [LICENSE](LICENSE). +- **Upstream**: , release 2.0.0. + +## OpenCV — ArUco 4×4 dictionary codes + +- **Component**: the marker codes in js-aruco2's + `src/dictionaries/aruco_4x4_1000.js`, compiled into the renderer bundle. They + are taken from OpenCV's `predefined_dictionaries.hpp`; the marker sheet the + editor prints draws markers 0–3 of this dictionary. +- **License**: BSD 3-Clause — Copyright (C) 2013, OpenCV Foundation, all rights + reserved. Third party copyrights are property of their respective owners. + + > Redistribution and use in source and binary forms, with or without + > modification, are permitted provided that the following conditions are met: + > + > - Redistributions of source code must retain the above copyright notice, + > this list of conditions and the following disclaimer. + > - Redistributions in binary form must reproduce the above copyright notice, + > this list of conditions and the following disclaimer in the documentation + > and/or other materials provided with the distribution. + > - Neither the names of the copyright holders nor the names of the + > contributors may be used to endorse or promote products derived from this + > software without specific prior written permission. + > + > This software is provided by the copyright holders and contributors "as is" + > and any express or implied warranties, including, but not limited to, the + > implied warranties of merchantability and fitness for a particular purpose + > are disclaimed. In no event shall copyright holders or contributors be liable + > for any direct, indirect, incidental, special, exemplary, or consequential + > damages (including, but not limited to, procurement of substitute goods or + > services; loss of use, data, or profits; or business interruption) however + > caused and on any theory of liability, whether in contract, strict liability, + > or tort (including negligence or otherwise) arising in any way out of the use + > of this software, even if advised of the possibility of such damage. + ## Rust crates — compiled into the compositor addon - **Component**: `compositor_view.node`, under diff --git a/crates/compositor-view-napi/src/lib.rs b/crates/compositor-view-napi/src/lib.rs index 905a6a5e5..5a1ded0e1 100644 --- a/crates/compositor-view-napi/src/lib.rs +++ b/crates/compositor-view-napi/src/lib.rs @@ -15,7 +15,7 @@ use openscreen_compositor::frame_geometry::FootageQuad; use openscreen_compositor::gif_export::{GifExportParams, GifStats}; use openscreen_compositor::gif_export_control::{GifExportCancelled, GifExportControl}; use openscreen_compositor::live::{LiveView, PausedPreviews}; -use openscreen_compositor::scene::Scene; +use openscreen_compositor::scene::{Scene, SceneClipCamera}; use openscreen_compositor::{config, pipeline}; use std::collections::HashMap; use std::path::PathBuf; @@ -368,6 +368,9 @@ pub fn present_time(id: i32, seconds: f64) { /// Remplace les sources du clip actif sans recréer la vue ni son thread de rendu. L'identité /// timeline et le playhead source sont atomiques avec le switch : deux clips partageant les /// mêmes fichiers restent distincts, et les deux décodeurs ouvrent directement la bonne frame. +/// +/// `additional_cameras`: cameras 2-4 of the clip (index k-1 = camera k, an empty `path` = no +/// camera in that slot). Absent = none; the view decodes only those its layout regions show. #[napi] pub fn set_active_clip( id: i32, @@ -376,18 +379,33 @@ pub fn set_active_clip( webcam_offset_sec: f64, clip_index: u32, source_time_sec: f64, + additional_cameras: Option>, ) { if let Some(v) = registry().lock().unwrap().get(&id) { + let cameras = additional_cameras + .unwrap_or_default() + .into_iter() + .map(|c| SceneClipCamera { path: c.path, offset_sec: c.offset_sec }) + .collect(); v.set_active_clip( &screen_path, &webcam_path, webcam_offset_sec, + cameras, clip_index as usize, source_time_sec, ); } } +/// An additional camera of a clip (= TS `CompositorClipCamera`). +#[napi(object)] +pub struct ClipCameraInput { + pub path: String, + /// Camera source time = screen source time - this. + pub offset_sec: f64, +} + /// Installe la scène de l'app (JSON `SceneDescription`) sur la vue : layout preset piloté par /// l'app au lieu de la fixture. JSON invalide → ignoré côté natif. #[napi] @@ -504,6 +522,26 @@ pub struct ClipInput { pub webcam_offset_sec: f64, /// `false` évite une ouverture ffmpeg vouée à échouer et réserve du silence à ce clip. pub has_audio: bool, + /// Cameras 2-4 of the clip (index k-1 = camera k, an empty `path` = none). Absent = none. + pub additional_cameras: Option>, +} + +/// `ClipInput` → `pipeline::ClipSource`, shared by the MP4 and the GIF export. +fn clip_source(c: ClipInput) -> pipeline::ClipSource { + pipeline::ClipSource { + screen: c.screen_path, + webcam: c.webcam_path, + source_start_sec: c.source_start_sec, + source_end_sec: c.source_end_sec, + webcam_offset_sec: c.webcam_offset_sec, + has_audio: c.has_audio, + additional_cameras: c + .additional_cameras + .unwrap_or_default() + .into_iter() + .map(|cam| pipeline::ClipCamera { path: cam.path, offset_sec: cam.offset_sec }) + .collect(), + } } /// The export's pixel size. The SHAPE belongs to the scene: `scene.output` is what the live @@ -701,17 +739,7 @@ pub fn export_multi( params: Option, on_progress: Option, ) -> Result> { - let clips = clips - .into_iter() - .map(|c| pipeline::ClipSource { - screen: c.screen_path, - webcam: c.webcam_path, - source_start_sec: c.source_start_sec, - source_end_sec: c.source_end_sec, - webcam_offset_sec: c.webcam_offset_sec, - has_audio: c.has_audio, - }) - .collect(); + let clips = clips.into_iter().map(clip_source).collect(); Ok(AsyncTask::new(ExportMultiTask { out_path, clips, @@ -882,17 +910,7 @@ pub fn export_gif( // Deliberately the same argument shape as `export_multi`: the caller builds // one clip list and one scene, and picks the container. Cursor comes from // the scene like every other effect — there is no GIF-specific input left. - let clips = clips - .into_iter() - .map(|c| pipeline::ClipSource { - screen: c.screen_path, - webcam: c.webcam_path, - source_start_sec: c.source_start_sec, - source_end_sec: c.source_end_sec, - webcam_offset_sec: c.webcam_offset_sec, - has_audio: c.has_audio, - }) - .collect(); + let clips = clips.into_iter().map(clip_source).collect(); let gif_params = params .map(|p| GifExportParams { width: p.width, diff --git a/crates/compositor/src/camera_layers.rs b/crates/compositor/src/camera_layers.rs new file mode 100644 index 000000000..ed5447b97 --- /dev/null +++ b/crates/compositor/src/camera_layers.rs @@ -0,0 +1,476 @@ +//! Which camera layers a frame draws, and where, inside camera layout regions. +//! +//! Pure planning: `plan_frame` calls `camera_layers_at` and the backends draw the result. A +//! region's layers glide in from whatever was on screen before it (the default camera-1 PiP, +//! or the layers of a region that ends exactly where it starts) and glide back out to the +//! default, on the Full Camera envelope (`regions::camera_fullscreen_region_phase`): the same +//! window lengths, the same `ease_out_screen_studio` curve, measured on the screen clock. A +//! camera-1 Full Camera region that meets a layout region at a seam is a neighbour like any +//! other: camera 0 frame-filling, gliding straight into or out of the layout region's layers. + +use crate::frame_geometry::webcam_shape_code; +use crate::regions::{ + ease_out_screen_studio, ScreenClock, FULLSCREEN_LEAD_OUT_WINDOW_S, REGION_SEAM_S, + TRANSITION_WINDOW_S, +}; +use crate::scene::{SceneCameraFullscreenRegion, SceneCameraLayer, SceneCameraLayoutRegion}; + +/// Two regions closer than this (source seconds) hand over directly, without the default. +const ADJACENT_S: f64 = REGION_SEAM_S; +/// Layers fainter than this are not drawn at all. +const MIN_OPACITY: f32 = 1e-3; +/// Cameras beyond camera 0 a compositor takes frames for (`set_extra_camera_frames`). +pub const MAX_EXTRA_CAMERAS: usize = 3; + +/// One camera to draw this frame. +#[derive(Debug, Clone, Copy, PartialEq)] +pub struct CameraLayerPlan { + pub camera: usize, + /// x, y, w, h in output fractions. + pub dst: [f32; 4], + /// Corner radius as a fraction of min(dst w, h) in pixels. + pub radius_frac: f32, + /// `webcam_shape_code` vocabulary. + pub shape: u32, + /// 0..1. + pub opacity: f32, + pub fills_frame: bool, +} + +fn plan_of(layer: &SceneCameraLayer) -> CameraLayerPlan { + let r = layer.rect; + CameraLayerPlan { + camera: layer.camera, + dst: [r.x, r.y, r.width, r.height], + radius_frac: layer.radius_frac, + shape: webcam_shape_code(&layer.shape), + opacity: 1.0, + fills_frame: layer.fills_frame, + } +} + +fn lerp(a: f32, b: f32, k: f32) -> f32 { + a + (b - a) * k +} + +/// `a` turning into `b` at `k` (0 = all `a`, 1 = all `b`). A camera in both moves; a camera in +/// only one of them fades. Draw order: frame-filling layers first, then the rest, each group in +/// `b`'s order followed by what only `a` still shows. +fn blend(a: &[CameraLayerPlan], b: &[CameraLayerPlan], k: f32) -> Vec { + let k = k.clamp(0.0, 1.0); + let mut out = Vec::with_capacity(a.len() + b.len()); + for to in b { + out.push(match a.iter().find(|from| from.camera == to.camera) { + Some(from) => { + // Discrete properties switch half-way through the move. + let flags = if k >= 0.5 { to } else { from }; + CameraLayerPlan { + camera: to.camera, + dst: std::array::from_fn(|i| lerp(from.dst[i], to.dst[i], k)), + radius_frac: lerp(from.radius_frac, to.radius_frac, k), + shape: flags.shape, + opacity: 1.0, + fills_frame: flags.fills_frame, + } + } + None => CameraLayerPlan { opacity: k, ..*to }, + }); + } + for from in a { + if !b.iter().any(|to| to.camera == from.camera) { + out.push(CameraLayerPlan { opacity: 1.0 - k, ..*from }); + } + } + out.retain(|layer| layer.opacity > MIN_OPACITY); + // Stable: each group keeps the order built above. + out.sort_by_key(|layer| !layer.fills_frame); + out +} + +/// Camera 0 as a Full Camera region shows it: the whole frame, square corners. Same discrete +/// shape as the default layer, so the hand-over to the default inside the region is seamless. +fn full_camera_layer(default: &CameraLayerPlan) -> CameraLayerPlan { + CameraLayerPlan { + camera: 0, + dst: [0.0, 0.0, 1.0, 1.0], + radius_frac: 0.0, + shape: default.shape, + opacity: 1.0, + fills_frame: true, + } +} + +/// The layers to draw at source time `t`, in draw order. +/// +/// `default_cam0`: camera 0's layer as today's plan places it (None = camera 0 not drawn). It +/// is what shows outside every region, and what a region glides from and back to. +/// +/// `fullscreen`: the camera-1 Full Camera regions. One that meets a layout region at a seam +/// is that region's neighbour, with camera 0 frame-filling: the layout region glides in from +/// it, or holds to its end and the Full Camera region's lead-in glides on from its layers +/// (the Full Camera phase holds 1 on that side, `regions::camera_fullscreen_region_phase`). +/// +/// Time base: copied from `camera_fullscreen_region_phase` — the region is found on SOURCE +/// time (its bounds compared to `t` as `f32`), then the windows are measured on the SCREEN +/// clock (`clock.at`), so a speed region does not stretch or squash the glide. +pub fn camera_layers_at( + regions: &[SceneCameraLayoutRegion], + fullscreen: &[SceneCameraFullscreenRegion], + t: f32, + clock: &ScreenClock, + default_cam0: Option, +) -> Vec { + let default: Vec = default_cam0.into_iter().collect(); + let full_cam0: Vec = default_cam0.iter().map(full_camera_layer).collect(); + // An empty (or reversed) region would match `t` at a single instant and draw its layers + // at full strength for that sample; it is neither drawn nor anyone's neighbour. + let non_empty = || regions.iter().enumerate().filter(|(_, r)| r.end_sec > r.start_sec); + let full_cameras = || fullscreen.iter().filter(|r| r.end_sec > r.start_sec); + let layers_of = |r: &SceneCameraLayoutRegion| r.layers.iter().map(plan_of).collect::>(); + let Some(index) = non_empty() + .find(|(_, r)| r.start_sec as f32 <= t && t <= r.end_sec as f32) + .map(|(i, _)| i) + else { + return full_camera_lead_in(regions, fullscreen, t, clock, &full_cam0).unwrap_or(default); + }; + let region = ®ions[index]; + let others = || non_empty().filter(move |(i, _)| *i != index).map(|(_, r)| r); + let prev = others() + .find(|r| (r.end_sec - region.start_sec).abs() <= ADJACENT_S) + .map(layers_of) + .or_else(|| { + full_cameras() + .any(|f| (f.end_sec - region.start_sec).abs() <= ADJACENT_S) + .then(|| full_cam0.clone()) + }); + let has_next = others().any(|r| (r.start_sec - region.end_sec).abs() <= ADJACENT_S) + || full_cameras().any(|f| (f.start_sec - region.end_sec).abs() <= ADJACENT_S); + let current = layers_of(region); + + let (start, end, t) = ( + clock.at(region.start_sec as f32), + clock.at(region.end_sec as f32), + clock.at(t), + ); + let half = (end - start) * 0.5; + let win_in = TRANSITION_WINDOW_S.min(half); + let win_out = FULLSCREEN_LEAD_OUT_WINDOW_S.min(half); + if t - start < win_in { + let from = prev.as_deref().unwrap_or(&default); + blend(from, ¤t, ease_out_screen_studio((t - start) / win_in)) + } else if !has_next && end - t < win_out { + // With a region right after this one, the hand-over is that region's lead-in instead. + blend(¤t, &default, 1.0 - ease_out_screen_studio((end - t) / win_out)) + } else { + blend(&[], ¤t, 1.0) + } +} + +/// Inside a Full Camera region that starts where a layout region ends, during its lead-in: +/// the layout region's layers gliding into camera 0 frame-filling (`full_cam0`), on the same +/// window and curve as a layout region's lead-in. None anywhere else. +fn full_camera_lead_in( + regions: &[SceneCameraLayoutRegion], + fullscreen: &[SceneCameraFullscreenRegion], + t: f32, + clock: &ScreenClock, + full_cam0: &[CameraLayerPlan], +) -> Option> { + let full = fullscreen.iter().find(|f| { + f.end_sec > f.start_sec && f.start_sec as f32 <= t && t <= f.end_sec as f32 + })?; + let before = regions.iter().find(|r| { + r.end_sec > r.start_sec && (r.end_sec - full.start_sec).abs() <= ADJACENT_S + })?; + let (start, end) = (clock.at(full.start_sec as f32), clock.at(full.end_sec as f32)); + let t = clock.at(t); + let win_in = TRANSITION_WINDOW_S.min((end - start) * 0.5); + if t - start >= win_in { + return None; + } + let from: Vec = before.layers.iter().map(plan_of).collect(); + Some(blend(&from, full_cam0, ease_out_screen_studio((t - start) / win_in))) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::scene::SceneRect; + + const PIP: [f32; 4] = [0.75, 0.7, 0.22, 0.22]; + const CORNER: [f32; 4] = [0.05, 0.05, 0.2, 0.2]; + const FULL: [f32; 4] = [0.0, 0.0, 1.0, 1.0]; + + fn default_pip() -> Option { + Some(CameraLayerPlan { + camera: 0, + dst: PIP, + radius_frac: 0.5, + shape: 1, + opacity: 1.0, + fills_frame: false, + }) + } + + fn region(start: f64, end: f64, layers: &[(usize, [f32; 4], bool)]) -> SceneCameraLayoutRegion { + SceneCameraLayoutRegion { + clip_index: Some(0), + start_sec: start, + end_sec: end, + layers: layers + .iter() + .map(|&(camera, r, fills_frame)| SceneCameraLayer { + camera, + rect: SceneRect { x: r[0], y: r[1], width: r[2], height: r[3] }, + radius_frac: if fills_frame { 0.0 } else { 0.3 }, + shape: if fills_frame { "rectangle" } else { "rounded" }.to_string(), + fills_frame, + }) + .collect(), + } + } + + fn at(regions: &[SceneCameraLayoutRegion], t: f32) -> Vec { + camera_layers_at(regions, &[], t, &ScreenClock::default(), default_pip()) + } + + fn layer(layers: &[CameraLayerPlan], camera: usize) -> Option { + layers.iter().copied().find(|l| l.camera == camera) + } + + fn strictly_between(v: f32, a: f32, b: f32) -> bool { + v > a.min(b) && v < a.max(b) + } + + #[test] + fn outside_every_region_only_the_default_camera_is_drawn() { + let regions = [region(2.0, 6.0, &[(1, FULL, true)])]; + for t in [0.0, 1.9, 6.1, 9.0] { + assert_eq!(at(®ions, t), vec![default_pip().unwrap()], "t = {t}"); + } + assert!(camera_layers_at(®ions, &[], 1.0, &ScreenClock::default(), None).is_empty()); + } + + #[test] + fn inside_a_region_after_the_lead_in_its_layers_are_drawn_fully() { + let regions = [region(2.0, 6.0, &[(1, FULL, true), (0, CORNER, false)])]; + let layers = at(®ions, 4.0); + assert_eq!(layers.len(), 2); + assert_eq!(layers[0].camera, 1); + assert_eq!(layers[0].dst, FULL); + assert_eq!(layers[0].opacity, 1.0); + assert!(layers[0].fills_frame); + assert_eq!(layers[1].camera, 0); + assert_eq!(layers[1].dst, CORNER); + assert_eq!(layers[1].opacity, 1.0); + assert_eq!(layers[1].shape, webcam_shape_code("rounded")); + assert_eq!(layers[1].radius_frac, 0.3); + } + + #[test] + fn entering_a_region_glides_camera_0_and_fades_in_a_new_camera() { + let regions = [region(2.0, 6.0, &[(1, FULL, true), (0, CORNER, false)])]; + let layers = at(®ions, 2.0 + TRANSITION_WINDOW_S / 2.0); + let cam0 = layer(&layers, 0).expect("camera 0"); + for i in 0..4 { + assert!(strictly_between(cam0.dst[i], PIP[i], CORNER[i]), "dst[{i}] = {}", cam0.dst[i]); + } + assert_eq!(cam0.opacity, 1.0); + let cam1 = layer(&layers, 1).expect("camera 1"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0), "opacity {}", cam1.opacity); + assert_eq!(cam1.dst, FULL); + // On the region's start the glide has not begun. + assert_eq!(at(®ions, 2.0), vec![default_pip().unwrap()]); + } + + #[test] + fn leaving_a_region_returns_to_the_default() { + let regions = [region(2.0, 6.0, &[(1, FULL, true), (0, CORNER, false)])]; + let win_out = FULLSCREEN_LEAD_OUT_WINDOW_S.min(2.0); + let layers = at(®ions, 6.0 - win_out / 2.0); + let cam0 = layer(&layers, 0).expect("camera 0"); + for i in 0..4 { + assert!(strictly_between(cam0.dst[i], PIP[i], CORNER[i]), "dst[{i}] = {}", cam0.dst[i]); + } + let cam1 = layer(&layers, 1).expect("camera 1"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0), "opacity {}", cam1.opacity); + // On the region's end it is back to the default, exactly. + let end = at(®ions, 6.0); + assert_eq!(end.len(), 1); + assert_eq!(end[0].camera, 0); + for i in 0..4 { + assert!((end[0].dst[i] - PIP[i]).abs() < 1e-6); + } + } + + #[test] + fn adjacent_regions_glide_directly() { + let regions = [region(2.0, 5.0, &[(1, FULL, true)]), region(5.0, 8.0, &[(2, FULL, true)])]; + // The end of A keeps A: no lead-out to the default PiP. + let before = at(®ions, 5.0 - 0.01); + assert_eq!(before.len(), 1); + assert_eq!(before[0].camera, 1); + assert_eq!(before[0].opacity, 1.0); + // B's lead-in hands camera 1 over to camera 2. + let during = at(®ions, 5.0 + TRANSITION_WINDOW_S / 2.0); + let cam1 = layer(&during, 1).expect("camera 1 fading out"); + let cam2 = layer(&during, 2).expect("camera 2 fading in"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0)); + assert!(strictly_between(cam2.opacity, 0.0, 1.0)); + // Neither region has camera 0, so the default PiP never shows around the seam. + let mut t = 4.8f32; + while t <= 5.2 { + assert!(layer(&at(®ions, t), 0).is_none(), "camera 0 at t = {t}"); + t += 0.01; + } + } + + fn full_camera(start: f64, end: f64) -> SceneCameraFullscreenRegion { + SceneCameraFullscreenRegion { + clip_index: Some(0), + start_sec: start, + end_sec: end, + rotation: 0, + mirror: None, + full_frame: false, + } + } + + /// The plan with Full Camera regions, camera 0's default placed as `plan_frame` places it: + /// grown from the PiP to the frame by the Full Camera progress. + fn at_with_full( + regions: &[SceneCameraLayoutRegion], + fulls: &[SceneCameraFullscreenRegion], + t: f32, + ) -> Vec { + let clock = ScreenClock::default(); + let p = crate::regions::camera_fullscreen_progress_at(fulls, t, &clock, regions); + let default = CameraLayerPlan { + dst: std::array::from_fn(|i| lerp(PIP[i], FULL[i], p)), + fills_frame: p >= 1.0, + ..default_pip().unwrap() + }; + camera_layers_at(regions, fulls, t, &clock, Some(default)) + } + + /// Camera 0 on the straight path between two rects: never the default PiP's geometry. + fn on_the_path(dst: [f32; 4], a: [f32; 4], b: [f32; 4]) -> bool { + (0..4).all(|i| dst[i] >= a[i].min(b[i]) - 1e-5 && dst[i] <= a[i].max(b[i]) + 1e-5) + } + + #[test] + fn full_camera_then_layout_region_glides_directly() { + let regions = [region(5.0, 8.0, &[(1, FULL, true), (0, CORNER, false)])]; + let fulls = [full_camera(2.0, 5.0)]; + // The end of the Full Camera region keeps camera 0 on the frame: no shrink. + let before = at_with_full(®ions, &fulls, 5.0 - 0.01); + assert_eq!(before.len(), 1); + assert_eq!(before[0].camera, 0); + assert_eq!(before[0].dst, FULL); + assert!(before[0].fills_frame); + // The layout region's lead-in moves camera 0 from the frame to its corner. + let during = at_with_full(®ions, &fulls, 5.0 + TRANSITION_WINDOW_S / 2.0); + let cam0 = layer(&during, 0).expect("camera 0"); + for i in 2..4 { + assert!(strictly_between(cam0.dst[i], FULL[i], CORNER[i]), "dst[{i}] = {}", cam0.dst[i]); + } + let cam1 = layer(&during, 1).expect("camera 1 fading in"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0)); + // The default PiP never shows around the seam. + let mut t = 3.5f32; + while t <= 5.0 + TRANSITION_WINDOW_S + 0.1 { + let cam0 = layer(&at_with_full(®ions, &fulls, t), 0).expect("camera 0"); + assert!(on_the_path(cam0.dst, FULL, CORNER), "t = {t}: {:?}", cam0.dst); + assert_eq!(cam0.opacity, 1.0, "t = {t}"); + t += 0.005; + } + } + + #[test] + fn layout_region_then_full_camera_glides_directly() { + let regions = [region(2.0, 5.0, &[(1, FULL, true), (0, CORNER, false)])]; + let fulls = [full_camera(5.0, 8.0)]; + // The end of the layout region keeps its layers: no lead-out to the default PiP. + let before = at_with_full(®ions, &fulls, 5.0 - 0.01); + assert_eq!(layer(&before, 0).expect("camera 0").dst, CORNER); + assert_eq!(layer(&before, 1).expect("camera 1").opacity, 1.0); + // The Full Camera lead-in takes camera 0 from its corner to the frame; camera 1 fades. + let during = at_with_full(®ions, &fulls, 5.0 + TRANSITION_WINDOW_S / 2.0); + let cam0 = layer(&during, 0).expect("camera 0"); + for i in 2..4 { + assert!(strictly_between(cam0.dst[i], CORNER[i], FULL[i]), "dst[{i}] = {}", cam0.dst[i]); + } + let cam1 = layer(&during, 1).expect("camera 1 fading out"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0)); + // After the lead-in camera 0 is the default again, which the held phase keeps full. + let after = at_with_full(®ions, &fulls, 5.0 + TRANSITION_WINDOW_S + 0.01); + assert_eq!(after.len(), 1); + assert_eq!(after[0].dst, FULL); + let mut t = 3.5f32; + while t <= 7.0 { + let cam0 = layer(&at_with_full(®ions, &fulls, t), 0).expect("camera 0"); + assert!(on_the_path(cam0.dst, CORNER, FULL), "t = {t}: {:?}", cam0.dst); + assert_eq!(cam0.opacity, 1.0, "t = {t}"); + t += 0.005; + } + } + + #[test] + fn a_full_camera_region_off_the_seam_is_no_neighbour() { + // 10 ms apart: the layout region glides from the default as before. + let regions = [region(5.01, 8.0, &[(1, FULL, true), (0, CORNER, false)])]; + let fulls = [full_camera(2.0, 5.0)]; + assert_eq!( + at_with_full(®ions, &fulls, 5.01 + TRANSITION_WINDOW_S / 2.0), + at(®ions, 5.01 + TRANSITION_WINDOW_S / 2.0) + ); + let shrinking = layer(&at_with_full(®ions, &fulls, 4.9), 0).expect("camera 0"); + assert!(shrinking.dst[2] < 1.0, "the Full Camera region leads out: {:?}", shrinking.dst); + } + + #[test] + fn a_short_region_halves_its_windows() { + let regions = [region(2.0, 3.0, &[(1, FULL, true)])]; + // Both windows are 0.5 s: the middle is the only frame at full strength. + let mid = at(®ions, 2.5); + assert_eq!(mid.len(), 1); + assert_eq!(mid[0].camera, 1); + assert_eq!(mid[0].opacity, 1.0); + for t in [2.25, 2.75] { + let cam1 = layer(&at(®ions, t), 1).expect("camera 1"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0), "t = {t}"); + } + } + + #[test] + fn an_empty_region_draws_nothing_of_its_own() { + // start == end (and a reversed region) would otherwise match `t` at its one instant + // and draw its layers at full strength for that sample. + for regions in [ + [region(3.0, 3.0, &[(1, FULL, true)])], + [region(3.0, 2.0, &[(1, FULL, true)])], + ] { + for t in [2.0, 2.5, 3.0] { + let layers = at(®ions, t); + assert!(layer(&layers, 1).is_none(), "t = {t}"); + assert_eq!(layers.len(), 1, "t = {t}"); + assert_eq!(layers[0].camera, 0, "t = {t}"); + } + } + // Nor does it count as a neighbour: the region before it still leads out. + let regions = [region(2.0, 6.0, &[(1, FULL, true)]), region(6.0, 6.0, &[(1, FULL, true)])]; + let cam1 = layer(&at(®ions, 5.9), 1).expect("camera 1"); + assert!(strictly_between(cam1.opacity, 0.0, 1.0)); + } + + #[test] + fn fills_frame_layers_are_drawn_first() { + let regions = [region(2.0, 6.0, &[(0, CORNER, false), (1, FULL, true)])]; + let order: Vec = at(®ions, 4.0).iter().map(|l| l.camera).collect(); + assert_eq!(order, vec![1, 0]); + // During the lead-in too. + let order: Vec = + at(®ions, 2.0 + TRANSITION_WINDOW_S / 2.0).iter().map(|l| l.camera).collect(); + assert_eq!(order, vec![1, 0]); + } +} diff --git a/crates/compositor/src/compositor_linux.rs b/crates/compositor/src/compositor_linux.rs index f7c5204b6..e9505224c 100644 --- a/crates/compositor/src/compositor_linux.rs +++ b/crates/compositor/src/compositor_linux.rs @@ -67,12 +67,12 @@ fn layer_source(models: bool) -> String { /// en parcourant les 18 wallpapers livres) en laissant le jeu actif resident. const IMG_CACHE_BUDGET_BYTES: u64 = 512 * 1024 * 1024; -/// Taille du buffer uniforme d'un calque : `LayerCB` entier (176 octets), le `struct Layer` de +/// Taille du buffer uniforme d'un calque : `LayerCB` entier (256 octets), le `struct Layer` de /// `layer.wgsl`. `blur.wgsl` n'en lit que les 128 premiers. const LAYER_BYTES: u64 = std::mem::size_of::() as u64; /// `&LayerCB` -> ses octets. `LayerCB` est `#[repr(C, align(16))]`, son layout EST le buffer -/// uniforme WGSL (dix vec4 et un vec2 + 2 f32 = 176 octets). +/// uniforme WGSL (seize vec4 = 256 octets). fn layer_bytes(cb: &LayerCB) -> &[u8] { unsafe { std::slice::from_raw_parts(cb as *const LayerCB as *const u8, LAYER_BYTES as usize) } } @@ -349,6 +349,9 @@ pub struct Compositor { readback_yuv: RefCell, // Etat pilote par live.rs (interior mutability : les methodes sont `&self`). + /// The frames of cameras 1..=3 (`set_extra_camera_frames`), as addresses (0 = none) so + /// the compositor's auto traits stay what they were without the raw pointers. + extra_camera_frames: std::cell::Cell<[usize; crate::camera_layers::MAX_EXTRA_CAMERAS]>, live_params: RefCell, scene: RefCell>, cursor: RefCell>, @@ -745,6 +748,7 @@ impl Compositor { readback, yuv: RefCell::new(None), readback_yuv, + extra_camera_frames: std::cell::Cell::new([0; crate::camera_layers::MAX_EXTRA_CAMERAS]), live_params: RefCell::new(LiveParams::default()), scene: RefCell::new(None), cursor: RefCell::new(None), @@ -1166,6 +1170,9 @@ impl Compositor { pub fn set_scene(&self, s: Option) { *self.scene.borrow_mut() = s; + // A new scene can drop the regions that drew an extra camera: forget its frames so a + // stale pointer is never drawn before the next `set_extra_camera_frames`. + self.extra_camera_frames.set([0; crate::camera_layers::MAX_EXTRA_CAMERAS]); } pub fn set_cursor(&self, track: crate::cursor::CursorTrack) { @@ -1200,8 +1207,29 @@ impl Compositor { } /// Pas de cache de SRV cote wgpu (les `TextureView`s sont recreees a chaque - /// draw depuis le carrier) -- no-op conserve pour la symetrie d'API. - pub fn clear_srv_cache(&self) {} + /// draw depuis le carrier). Only the extra cameras' frame slots are reset. + pub fn clear_srv_cache(&self) { + // The extra cameras' frames belong to the decoders just closed: forget them too, so + // a stale pointer is never read before the next `set_extra_camera_frames`. + self.extra_camera_frames.set([0; crate::camera_layers::MAX_EXTRA_CAMERAS]); + } + + /// Frames for cameras 1..=3 (index 0 = scene camera 1). A null or missing entry means that + /// camera is not drawn this frame. The compositor keeps the pointers and reads them in every + /// `compose_frame` until the next call, so they must stay valid until then. + pub unsafe fn set_extra_camera_frames(&self, frames: &[*const AVFrame]) { + let mut slots = [0usize; crate::camera_layers::MAX_EXTRA_CAMERAS]; + for (slot, frame) in slots.iter_mut().zip(frames) { + *slot = *frame as usize; + } + self.extra_camera_frames.set(slots); + } + + /// The frame set for scene camera `camera` (1..=3), null when there is none. + fn extra_camera_frame(&self, camera: usize) -> *const AVFrame { + let slots = self.extra_camera_frames.get(); + camera.checked_sub(1).and_then(|k| slots.get(k).copied()).unwrap_or(0) as *const AVFrame + } // -- seam frame (lit le carrier `data[0]`) -- @@ -1568,7 +1596,8 @@ impl Compositor { /// son propre repli, et un echec silencieux redonnerait le noir qu'on corrige. /// /// `motion` anime l'image au temps programme `programme_t` ; la bulle webcam - /// passe `WallpaperMotion::None`. + /// passe `WallpaperMotion::None`. `transparency` = `layer_fx.x` (0 = opaque): a camera + /// layer's background box fades with the layer. #[allow(clippy::too_many_arguments)] fn image_bg_draw( &self, @@ -1579,6 +1608,7 @@ impl Compositor { aspect: f32, motion: WallpaperMotion, programme_t: f32, + transparency: f32, dummy: &wgpu::TextureView, ) -> Result { let (tex, iw, ih) = self.cached_image(path)?; @@ -1600,6 +1630,7 @@ impl Compositor { mode: 6.0, fx: [0.0, 0.0, anim[0], anim[1]], mb, + layer_fx: [transparency, 0.0, 0.0, 0.0], ..Default::default() }; let view = tex.create_view(&wgpu::TextureViewDescriptor::default()); @@ -1620,15 +1651,19 @@ impl Compositor { /// /// `quad_px` / `radius_px` sont ceux de la bulle : le fond doit epouser ses /// coins arrondis, sinon un rectangle deborde derriere la camera. + /// + /// `transparency` = the camera layer's `layer_fx.x`: the box fades with the camera. fn webcam_bg_draw( &self, bg: Option<&SceneBackground>, dst: [f32; 4], quad_px: [f32; 2], radius_px: f32, + transparency: f32, dummy: &wgpu::TextureView, ) -> BgDraw { const BLACK: [f32; 4] = [0.0, 0.0, 0.0, 1.0]; + let layer_fx = [transparency, 0.0, 0.0, 0.0]; let flat = |cb: LayerCB| BgDraw { layer: self.make_bind(&cb, None, dummy), _tex: None, _view: None }; let solid = |color: [f32; 4]| LayerCB { dst, @@ -1636,6 +1671,7 @@ impl Compositor { radius_px, mode: 1.0, color, + layer_fx, ..Default::default() }; match bg { @@ -1653,6 +1689,7 @@ impl Compositor { quad_px, radius_px, fx: [a.sin(), -a.cos(), 0.0, 0.0], + layer_fx, ..crate::frame_geometry::gradient_layer(stops, offsets, BLACK) }) } @@ -1661,7 +1698,10 @@ impl Compositor { // elle que l'image doit remplir sans etirement. let aspect = if quad_px[1] > 0.0 { quad_px[0] / quad_px[1] } else { 1.0 }; let still = WallpaperMotion::None; - match self.image_bg_draw(path, dst, quad_px, radius_px, aspect, still, 0.0, dummy) { + let drawn = self.image_bg_draw( + path, dst, quad_px, radius_px, aspect, still, 0.0, transparency, dummy, + ); + match drawn { Ok(d) => d, Err(e) => { // Meme contrat que le fond d'ecran : un chemin casse est @@ -2397,7 +2437,10 @@ impl Compositor { BgLayer::Image(path, motion) => { let full = [0.0, 0.0, 1.0, 1.0]; let t = g.programme_t; - match self.image_bg_draw(&path, full, [0.0, 0.0], 0.0, rw / rh, motion, t, &dummy) { + let drawn = self.image_bg_draw( + &path, full, [0.0, 0.0], 0.0, rw / rh, motion, t, 0.0, &dummy, + ); + match drawn { Ok(d) => Some(d), Err(e) => { eprintln!("[fond image] \"{path}\" : {e:#}"); @@ -2429,7 +2472,10 @@ impl Compositor { // lancement gracieux, le temps que l'inference rende son premier masque. // // Calcule ICI, avant le draw comme avant l'ombre : les deux en dependent. - let (effect_code, blur_intensity, webcam_bg) = { + // Without layout regions camera 0 is drawn from `w_dst` (`webcam_bg`, `webcam_draw`, + // `webcam_shadow` below); with them every planned layer is (`camera_layer_draws`). + let unplanned = g.camera_layers.is_empty(); + let (effect_code, blur_intensity, custom_bg) = { let has_mask = self.webcam_mask.borrow().is_some(); let effect = scene_ref .as_ref() @@ -2445,24 +2491,16 @@ impl Compositor { // fond. Le shader ne sait peindre qu'une couleur plate sous le // masque ; degrades et images y tombaient sur du noir, et le defaut // EST une image. - Some((code, e)) if code > 2.5 => { - // Sans piste webcam le fond peindrait un rectangle seul dans le - // cadre : il ne se prepare que si la camera se dessine. - let bg = webcam_planes.is_some().then(|| { - self.webcam_bg_draw( - e.background.as_ref(), - g.w_dst, - g.w_px, - g.w_radius, - &dummy, - ) - }); - (1.0, 0.0, bg) - } + Some((code, e)) if code > 2.5 => (1.0, 0.0, Some(e.background.as_ref())), Some((code, e)) => (code, e.blur_intensity.clamp(0.0, 1.0), None), None => (0.0, 0.0, None), } }; + // Sans piste webcam le fond peindrait un rectangle seul dans le cadre : il ne se + // prepare que si la camera se dessine. + let webcam_bg = custom_bg + .filter(|_| unplanned && webcam_planes.is_some()) + .map(|bg| self.webcam_bg_draw(bg, g.w_dst, g.w_px, g.w_radius, 0.0, &dummy)); // L'ombre se juge sur le mode DE LA SCENE, pas sur `effect_code` : le fond // personnalise se compose desormais en detourage (code 1) tout en gardant // sa bulle, et tester le code compose la lui retirerait. Meme lecture que @@ -2471,7 +2509,10 @@ impl Compositor { scene_ref.as_ref().and_then(|s| s.webcam_effect.as_ref()), Some(e) if e.shader_code() == 1.0 ) && self.webcam_mask.borrow().is_some(); - let webcam_draw = webcam_planes.as_ref().map(|(wy, wu, wv)| { + // Camera 0's crop, mirror and desk turn as camera settings, plus its homography + // (`Scene::camera(0)`): with one, `src` is not read and the crop is dropped. + let cam0 = g.camera0_settings(scene_ref.as_ref()); + let webcam_draw = webcam_planes.as_ref().filter(|_| unplanned).map(|(wy, wu, wv)| { // COVER-CROP. `src` etait cable a [0,0,1,1], donc la texture entiere // etait etiree sur la boite quelle que soit sa forme : le facteur de // deformation valait exactement `box_ar / cam_ar`. Invisible en PiP @@ -2481,45 +2522,27 @@ impl Compositor { // // `cover_crop_uv` est la primitive partagee que macOS et Windows // utilisent ; elle rend le rect inchange quand il a deja le bon - // ratio, donc aucun placement correct ne bouge. - let [cu0, cv0, cu1, cv1] = crate::frame_geometry::webcam_source_rect( + // ratio, donc aucun placement correct ne bouge. Mirror and the desk-shot + // turn are both bound swaps: u for horizontal, v for vertical. + let src = crate::frame_geometry::camera_source_rect( + Some(&cam0), [wcw, wch], [wtw as f32, wth as f32], - scene_ref - .as_ref() - .and_then(|scene| scene.layout.webcam_crop), g.w_px[0] / g.w_px[1].max(0.0001), ); - // MIROIR : on inverse l'intervalle u. Le VS interpole `src` - // lineairement et `fs_main` ne re-clampe pas `i.uv`, donc un - // intervalle a l'envers suffit -- aucune retouche du WGSL. Apres le - // cover-crop les deux bornes sont strictement a l'interieur de la - // texture, donc le sampler ClampToEdge ne bave pas sur les bords. - let (u0, u1) = if lp.webcam_mirror { (cu1, cu0) } else { (cu0, cu1) }; // `src_prev` doit valoir EXACTEMENT le `src` de ce draw, miroir // compris : le shader s'en sert pour reconstruire l'UV de la frame // precedente, et un rect source qui ne correspond pas au calque // dessine ferait diverger la trainee vers une zone de la texture qui // n'a jamais ete affichee. Seul `dst_prev` porte le mouvement. - let cb = LayerCB { - dst: g.w_dst, - src: [u0, cv0, u1, cv1], - quad_px: g.w_px, - radius_px: g.w_radius, - mode: 0.0, - // `color.a` porte l'alpha du decoupage (`color.a * personne`) ; le - // RGB n'est plus lu, le fond ayant deja ete peint sous la camera. - color: [0.0, 0.0, 0.0, 1.0], - // `fx.xy` = etendue valide de la texture webcam, par quoi le - // shader divise `uv` pour retomber dans l'espace du masque ; - // `fx.z` = mode, `fx.w` = intensite du flou. Contrat commun aux - // trois back-ends, cf. `layer.wgsl` et `webcam-segmentation.md`. - fx: [w_valid[0], w_valid[1], effect_code, blur_intensity], - src_prev: [u0, cv0, u1, cv1], - dst_prev: g.w_dst_prev, - mb: [g.mb_taps, g.mb_amount, 1.0, 0.0], - ..Default::default() - }; + // `fx.xy` = etendue valide de la texture webcam, par quoi le + // shader divise `uv` pour retomber dans l'espace du masque ; + // `fx.z` = mode, `fx.w` = intensite du flou. Contrat commun aux + // trois back-ends, cf. `layer.wgsl` et `webcam-segmentation.md`. + let cb = crate::frame_geometry::with_camera_homography( + g.webcam_video_cb(src, w_valid, effect_code, blur_intensity), + Some(&cam0), + ); // Le masque est lie par `make_bind` sur tous les draws, pas seulement // celui-ci : le layout l'exige (cf. `tex_entry(4)`). self.make_bind(&cb, Some((wy, wu, wv)), &dummy) @@ -2553,6 +2576,71 @@ impl Compositor { self.make_bind(&cb, None, &dummy) }); + // Every planned camera layer, in order, each as shadow, then (camera 0) the custom + // background, then the video: one uniform buffer and bind group per draw, drawn in + // the "webcam-pass" below. Camera 0 keeps its frame, effects and mask; the extra + // cameras draw from `set_extra_camera_frames` without effects, and a camera without a + // frame is skipped. `extra_planes` keeps their views alive until the pass is encoded. + let mut camera_layer_draws: Vec = Vec::new(); + let mut extra_planes = Vec::new(); + let render = [rw, rh]; + let cam0_base = g.webcam_video_cb([0.0; 4], w_valid, effect_code, blur_intensity); + let plain = |layer: LayerBind| BgDraw { layer, _tex: None, _view: None }; + for plan in &g.camera_layers { + if plan.camera == 0 { + // `webcam_planes` is already gated on `lp.has_webcam`. + let Some((wy, wu, wv)) = webcam_planes.as_ref() else { continue }; + let video = crate::frame_geometry::camera_layer_cb( + plan, + Some(&cam0), + [wcw, wch], + [wtw as f32, wth as f32], + render, + &cam0_base, + ); + if let Some(shadow) = + g.camera_layer_shadow(plan, &video, render, cfg.shadow, is_cutout) + { + camera_layer_draws.push(plain(self.make_bind(&shadow, None, &dummy))); + } + if let Some(bg) = custom_bg { + let (dst, quad_px, radius_px) = (video.dst, video.quad_px, video.radius_px); + let t = video.layer_fx[0]; + let bg = self.webcam_bg_draw(bg, dst, quad_px, radius_px, t, &dummy); + camera_layer_draws.push(bg); + } + let bind = self.make_bind(&video, Some((wy, wu, wv)), &dummy); + camera_layer_draws.push(plain(bind)); + } else { + let frame = self.extra_camera_frame(plan.camera); + if Self::pixel_buffer_of(frame).is_none() { + continue; + } + let Ok(planes) = self.nv12_srvs(frame) else { continue }; + let (tw, th) = self.tex_dims(frame); + let visible = [(*frame).width as f32, (*frame).height as f32]; + let tex = [tw as f32, th as f32]; + let valid = [visible[0] / tex[0].max(1.0), visible[1] / tex[1].max(1.0)]; + let video = crate::frame_geometry::camera_layer_cb( + plan, + scene_ref.as_ref().and_then(|s| s.camera(plan.camera)), + visible, + tex, + render, + &crate::frame_geometry::extra_camera_base_cb(valid), + ); + if let Some(shadow) = + g.camera_layer_shadow(plan, &video, render, cfg.shadow, false) + { + camera_layer_draws.push(plain(self.make_bind(&shadow, None, &dummy))); + } + let (ey, eu, ev) = &planes; + let bind = self.make_bind(&video, Some((ey, eu, ev)), &dummy); + camera_layer_draws.push(plain(bind)); + extra_planes.push(planes); + } + } + // ANNOTATIONS -- calque le plus haut, place relativement au rect ecran // (les coords x/y/w/h de l'annotation sont des fractions de ce rect, cf. // `scene.rs`). Le rect est `g.s_ann`, l'ecran SANS ZOOM, et surtout pas @@ -2753,6 +2841,13 @@ impl Compositor { if text.content.trim().is_empty() { continue; } + // The desk label shows only while the camera is covered: skip the rest of + // its section before rasterizing anything. + if crate::text_anim::is_desk_cover(text.animation.as_deref()) + && g.webcam_cover <= 0.0 + { + continue; + } let color = parse_hex(&text.color).unwrap_or([1.0, 1.0, 1.0, 1.0]); let background = parse_hex(&text.background_color).unwrap_or([0.0, 0.0, 0.0, 0.0]); @@ -2787,10 +2882,11 @@ impl Compositor { // et remis a l'echelle de la sortie, comme la taille de // police : en px absolus la meme animation sauterait deux // fois plus haut dans un rendu 4K que dans l'apercu. - let anim = crate::text_anim::text_animation_state( + let anim = crate::text_anim::annotation_text_state( text.animation.as_deref(), (g.source_t - a.start_sec as f32) * 1000.0, ((a.end_sec - a.start_sec) * 1000.0) as f32, + g.webcam_cover, ); let anim_px = rh / crate::text_anim::ANIMATION_REFERENCE_HEIGHT; let (mut ax, mut ay, mut aw, mut ah) = ( @@ -3409,6 +3505,9 @@ impl Compositor { if let Some(layer) = &webcam_draw { self.draw_layer(&mut rpass, layer, false); } + for draw in &camera_layer_draws { + self.draw_layer(&mut rpass, &draw.layer, false); + } } // Passe 3 : les autres annotations, par-dessus tout le reste (les flous sont passes avant // le curseur). Elle existe meme sans annotation : deux passes consecutives sur la MEME @@ -4768,12 +4867,37 @@ mod tests { /// `compose_pip` sur une scene deja construite, pour les tests qui la retouchent. fn compose_pip_scene(comp: &Compositor, gpu: &Gpu, scene: Scene, shadow: bool) -> Vec { + let screen = FakeFrame::new(gpu, 128, 128, |_, _| 126); + let webcam = FakeFrame::new(gpu, 64, 64, |_, _| Y_WHITE); + compose_pip_frames(comp, scene, shadow, &screen, &webcam) + } + + /// `compose_pip_scene` with the caller's own screen and camera frames. + fn compose_pip_frames( + comp: &Compositor, + scene: Scene, + shadow: bool, + screen: &FakeFrame, + webcam: &FakeFrame, + ) -> Vec { + compose_pip_frames_with_extras(comp, scene, shadow, screen, webcam, &[]) + } + + /// `compose_pip_frames` with extra camera frames. They are handed over AFTER `set_scene`, + /// which forgets the previous ones, and before the compose, as the product code does. + fn compose_pip_frames_with_extras( + comp: &Compositor, + scene: Scene, + shadow: bool, + screen: &FakeFrame, + webcam: &FakeFrame, + extra: &[*const AVFrame], + ) -> Vec { comp.set_live_params(live_params_from_scene(&scene)); comp.set_has_webcam(true); comp.set_scene(Some(scene)); + unsafe { comp.set_extra_camera_frames(extra) }; - let screen = FakeFrame::new(gpu, 128, 128, |_, _| 126); - let webcam = FakeFrame::new(gpu, 64, 64, |_, _| Y_WHITE); let mut cfg = Cfg::c8(); cfg.bg_blur = 0.0; cfg.zoom = false; @@ -4888,6 +5012,9 @@ mod tests { clip_index: None, start_sec: 2.0, end_sec: 8.0, + rotation: 0, + mirror: None, + full_frame: false, }); // 5 s : la montee (~1 s) est finie, le retour (~1,5 s) n'a pas commence. comp.set_timeline_time(Some(5.0)); @@ -4899,6 +5026,241 @@ mod tests { ); } + /// `pip_scene_json` on a black background: with a black screen too, the camera is drawn over + /// black wherever it lands, so its colour can be read without knowing where the PiP is. + fn black_pip_scene() -> Scene { + let json = pip_scene_json(NO_EFFECT).replace("#0080ff", "#000000"); + Scene::from_json(&json).expect("scene json") + } + + /// The x of every pixel of a `width`-wide RGBA image that `pred` accepts. + fn xs_where(rgba: &[u8], width: usize, pred: impl Fn(&[u8]) -> bool) -> Vec { + rgba.chunks_exact(4) + .enumerate() + .filter(|(_, px)| pred(px)) + .map(|(i, _)| i % width) + .collect() + } + + /// The RGBA of pixel (x, y) of a 320-wide readback. + fn pixel_at(rgba: &[u8], x: usize, y: usize) -> [u8; 4] { + let i = (y * 320 + x) * 4; + [rgba[i], rgba[i + 1], rgba[i + 2], rgba[i + 3]] + } + + /// A 64x64 camera frame of one RGB colour (BT.709 limited, `rgb_to_nv12`). + fn solid_camera(gpu: &Gpu, rgb: [u8; 3]) -> FakeFrame { + let (w, h) = (64u32, 64u32); + let pixels: Vec = (0..w * h).flat_map(|_| rgb).collect(); + let (y, uv) = rgb_to_nv12(&pixels, w, h); + FakeFrame::from_planes(gpu, w, h, &y, &uv) + } + + /// Camera 1 over the whole frame, and camera 0 as a PiP in the bottom-right corner + /// (x 224..304, y 126..171 of a 320x180 frame). + const CAMERA_1_FULL: &str = r#"{"camera":1,"rect":{"x":0,"y":0,"width":1,"height":1},"shape":"rectangle","fillsFrame":true}"#; + const CAMERA_0_PIP: &str = r#"{"camera":0,"rect":{"x":0.7,"y":0.7,"width":0.25,"height":0.25},"shape":"rectangle"}"#; + + /// A camera layout region from 2 s to 8 s with these layers (JSON objects, comma-separated). + fn layout_region(layers: &str) -> crate::scene::SceneCameraLayoutRegion { + serde_json::from_str(&format!(r#"{{"startSec":2.0,"endSec":8.0,"layers":[{layers}]}}"#)) + .expect("region json") + } + + /// The time in the lead-in of `layout_region` where a layer only the region shows is drawn + /// at opacity 0.5 (and one only the default shows, at 0.5 too). Checked against the plan. + fn half_way_in(region: &crate::scene::SceneCameraLayoutRegion) -> f32 { + let (mut lo, mut hi) = (0.0f32, 1.0f32); + for _ in 0..40 { + let mid = 0.5 * (lo + hi); + if crate::regions::ease_out_screen_studio(mid) < 0.5 { + lo = mid; + } else { + hi = mid; + } + } + let t = 2.0 + lo * crate::regions::TRANSITION_WINDOW_S; + let default_cam0 = crate::camera_layers::CameraLayerPlan { + camera: 0, + dst: [0.7, 0.7, 0.25, 0.25], + radius_frac: 0.0, + shape: 0, + opacity: 1.0, + fills_frame: false, + }; + let clock = crate::regions::ScreenClock::new(&[], 0); + let layers = crate::camera_layers::camera_layers_at( + std::slice::from_ref(region), + &[], + t, + &clock, + Some(default_cam0), + ); + assert!( + layers.iter().any(|l| (l.opacity - 0.5).abs() < 1e-3), + "no layer at half opacity at {t}: {layers:?}" + ); + t + } + + /// `compose_pip_frames` at source time `t`, with these extra camera frames set (and cleared + /// again afterwards). + fn compose_layers( + comp: &Compositor, + scene: Scene, + t: f32, + screen: &FakeFrame, + camera_0: &FakeFrame, + extra: &[*const AVFrame], + ) -> Vec { + comp.set_timeline_time(Some(t)); + let rgba = compose_pip_frames_with_extras(comp, scene, false, screen, camera_0, extra); + unsafe { comp.set_extra_camera_frames(&[]) }; + rgba + } + + fn is_green(px: [u8; 4]) -> bool { + px[1] > 200 && px[0] < 60 && px[2] < 60 + } + + /// A `camera-full-pip` region: camera 1 (green) fills the frame, camera 0 (white) is the PiP + /// over it. The centre shows camera 1, the PiP rect camera 0. + #[test] + fn two_cameras_draw_in_their_planned_rects() { + let Some(gpu) = gpu() else { return }; + let comp = Compositor::new_sized(&gpu, 320, 180).expect("Compositor::new_sized"); + let screen = FakeFrame::new(&gpu, 128, 128, |_, _| Y_BLACK); + let white = FakeFrame::new(&gpu, 64, 64, |_, _| Y_WHITE); + let green = solid_camera(&gpu, [0, 255, 0]); + let mut scene = black_pip_scene(); + scene.camera_layout_regions.push(layout_region(&format!("{CAMERA_1_FULL},{CAMERA_0_PIP}"))); + // 5 s: past the lead-in, before the lead-out. + let rgba = compose_layers(&comp, scene, 5.0, &screen, &white, &[green.as_ptr()]); + let centre = pixel_at(&rgba, 160, 90); + assert!(is_green(centre), "the centre is not camera 1: {centre:?}"); + let pip = pixel_at(&rgba, 264, 148); + assert!(pip[..3].iter().all(|&c| c > 240), "the PiP rect is not camera 0: {pip:?}"); + let corner = pixel_at(&rgba, 10, 10); + assert!(is_green(corner), "camera 1 does not fill the frame: {corner:?}"); + } + + /// Without a frame for camera 1 its layer is skipped, empty slice or null pointer alike: + /// the centre shows the screen, as it does without any region. + #[test] + fn a_missing_extra_camera_skips_its_layer() { + let Some(gpu) = gpu() else { return }; + let comp = Compositor::new_sized(&gpu, 320, 180).expect("Compositor::new_sized"); + let screen = FakeFrame::new(&gpu, 128, 128, |_, _| 126); + let white = FakeFrame::new(&gpu, 64, 64, |_, _| Y_WHITE); + let control = compose_layers(&comp, black_pip_scene(), 5.0, &screen, &white, &[]); + let screen_px = pixel_at(&control, 160, 90); + assert!(screen_px[1] > 100 && screen_px[1] < 160, "control: the centre is not the screen"); + + let mut scene = black_pip_scene(); + scene.camera_layout_regions.push(layout_region(&format!("{CAMERA_1_FULL},{CAMERA_0_PIP}"))); + for extra in [&[][..], &[std::ptr::null()][..]] { + let rgba = compose_layers(&comp, scene.clone(), 5.0, &screen, &white, extra); + assert_eq!(pixel_at(&rgba, 160, 90), screen_px, "{} frames set", extra.len()); + let pip = pixel_at(&rgba, 264, 148); + assert!(pip[..3].iter().all(|&c| c > 240), "camera 0 is no longer drawn: {pip:?}"); + } + } + + /// Half-way into the lead-in camera 1 fades in at opacity 0.5: over the black screen the + /// centre is half as green as in the hold. + #[test] + fn a_fading_extra_camera_is_half_transparent() { + let Some(gpu) = gpu() else { return }; + let comp = Compositor::new_sized(&gpu, 320, 180).expect("Compositor::new_sized"); + let screen = FakeFrame::new(&gpu, 128, 128, |_, _| Y_BLACK); + let white = FakeFrame::new(&gpu, 64, 64, |_, _| Y_WHITE); + let green = solid_camera(&gpu, [0, 255, 0]); + let region = layout_region(&format!("{CAMERA_1_FULL},{CAMERA_0_PIP}")); + let t = half_way_in(®ion); + let mut scene = black_pip_scene(); + scene.camera_layout_regions.push(region); + + let hold = compose_layers(&comp, scene.clone(), 5.0, &screen, &white, &[green.as_ptr()]); + let full = pixel_at(&hold, 160, 90); + assert!(is_green(full), "control: the centre is not camera 1 in the hold: {full:?}"); + let half = compose_layers(&comp, scene, t, &screen, &white, &[green.as_ptr()]); + let px = pixel_at(&half, 160, 90); + let expected = full[1] as i32 / 2; + assert!( + (px[1] as i32 - expected).abs() <= 8 && px[0] < 30 && px[2] < 30, + "half-way in: {px:?}, expected green near {expected}" + ); + } + + /// Camera 0 fading out at opacity 0.5 (a region that shows only camera 1, which has no + /// frame) draws the (white) camera at half strength over the (black) screen: every camera + /// pixel is mid grey, none of them is white any more. + #[test] + fn transparency_half_halves_the_camera_over_the_screen() { + let Some(gpu) = gpu() else { return }; + let comp = Compositor::new_sized(&gpu, 320, 180).expect("Compositor::new_sized"); + let screen = FakeFrame::new(&gpu, 128, 128, |_, _| Y_BLACK); + let webcam = FakeFrame::new(&gpu, 64, 64, |_, _| Y_WHITE); + let opaque = compose_pip_frames(&comp, black_pip_scene(), false, &screen, &webcam); + let whole = camera_pixels(&opaque); + assert!(whole > 200, "the camera is not on screen, the test proves nothing"); + + let region = layout_region(CAMERA_1_FULL); + let t = half_way_in(®ion); + let mut scene = black_pip_scene(); + scene.camera_layout_regions.push(region); + let half = compose_layers(&comp, scene, t, &screen, &webcam, &[]); + assert_eq!(camera_pixels(&half), 0, "a half transparent camera still has white pixels"); + let grey = half + .chunks_exact(4) + .filter(|px| px[..3].iter().all(|&c| (126..=130).contains(&c))) + .count(); + assert!( + grey as f32 >= whole as f32 * 0.9, + "{grey} mid-grey pixels for {whole} camera pixels drawn opaque" + ); + } + + /// Camera 0's perspective applies without any layout region. A homography that flips the + /// camera horizontally draws its right half on the left of the layer: a camera red on the + /// left and blue on the right shows blue left of red. The control (no perspective) shows it + /// the other way round, so the flip is the homography's doing. + #[test] + fn camera_0_perspective_applies_without_layout_regions() { + let Some(gpu) = gpu() else { return }; + let comp = Compositor::new_sized(&gpu, 320, 180).expect("Compositor::new_sized"); + let screen = FakeFrame::new(&gpu, 128, 128, |_, _| Y_BLACK); + // BT.709 limited: red = (Y 63, Cb 102, Cr 240), blue = (Y 32, Cb 240, Cr 118). + let (w, h) = (64u32, 64u32); + let y: Vec = (0..w * h).map(|i| if i % w < w / 2 { 63 } else { 32 }).collect(); + let uv: Vec = (0..h / 2) + .flat_map(|_| (0..w / 2).flat_map(|j| if j < w / 4 { [102, 240] } else { [240, 118] })) + .collect(); + let webcam = FakeFrame::from_planes(&gpu, w, h, &y, &uv); + let red = |px: &[u8]| px[0] > 200 && px[1] < 60 && px[2] < 60; + let blue = |px: &[u8]| px[2] > 200 && px[0] < 60 && px[1] < 60; + + let plain = compose_pip_frames(&comp, black_pip_scene(), false, &screen, &webcam); + let (r, b) = (xs_where(&plain, 320, red), xs_where(&plain, 320, blue)); + assert!(r.len() > 50 && b.len() > 50, "control: {} red, {} blue pixels", r.len(), b.len()); + assert!(r.iter().max() < b.iter().min(), "control: the camera's red half is on the left"); + + let mut scene = black_pip_scene(); + assert!(scene.camera_layout_regions.is_empty()); + scene.cameras.push(crate::scene::SceneCamera { + index: 0, + rotation: 0, + mirror: None, + crop: None, + homography: Some([-1.0, 0.0, 1.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0]), + aspect: None, + }); + let flipped = compose_pip_frames(&comp, scene, false, &screen, &webcam); + let (r, b) = (xs_where(&flipped, 320, red), xs_where(&flipped, 320, blue)); + assert!(r.len() > 50 && b.len() > 50, "flipped: {} red, {} blue pixels", r.len(), b.len()); + assert!(b.iter().max() < r.iter().min(), "flipped: the left half of the layer is blue"); + } + /// Le tour complet, celui qui a besoin d'ONNX Runtime : capture -> inference /// -> masque -> composite, entraine par `compose_frame` seul. Se saute /// proprement sans la bibliotheque, ce que fait la CI — cf. diff --git a/crates/compositor/src/compositor_macos.rs b/crates/compositor/src/compositor_macos.rs index 5440cfb51..a9a571e98 100644 --- a/crates/compositor/src/compositor_macos.rs +++ b/crates/compositor/src/compositor_macos.rs @@ -266,6 +266,9 @@ pub struct Compositor { timeline_time: RefCell>, /// Temps programme (secondes de sortie) — cf. `FrameGeometryInput::programme_time`. programme_time: RefCell>, + /// The frames of cameras 1..=3 (`set_extra_camera_frames`), as addresses (0 = none) so + /// the compositor's auto traits stay what they were without the raw pointers. + extra_camera_frames: std::cell::Cell<[usize; crate::camera_layers::MAX_EXTRA_CAMERAS]>, live_params: RefCell, metal_texture_cache: CVMetalTextureCache, /// Dernier command buffer soumis, gardé pour pouvoir l'attendre AU MOMENT où le CPU lit @@ -729,6 +732,7 @@ impl Compositor { footage: std::cell::Cell::new(None), timeline_time: RefCell::new(None), programme_time: RefCell::new(None), + extra_camera_frames: std::cell::Cell::new([0; crate::camera_layers::MAX_EXTRA_CAMERAS]), live_params: RefCell::new(LiveParams::default()), metal_texture_cache: cache, last_cmd: RefCell::new(None), @@ -836,6 +840,9 @@ impl Compositor { pub fn set_scene(&self, s: Option) { *self.scene.borrow_mut() = s; + // A new scene can drop the regions that drew an extra camera: forget its frames so a + // stale pointer is never drawn before the next `set_extra_camera_frames`. + self.extra_camera_frames.set([0; crate::camera_layers::MAX_EXTRA_CAMERAS]); } pub fn set_cursor(&self, track: crate::cursor::CursorTrack) { @@ -937,6 +944,26 @@ impl Compositor { /// IOSurface, pas par pointeur Rust. `flush()` est donc la vidange elle-même. pub fn clear_srv_cache(&self) { self.metal_texture_cache.flush(); + // The extra cameras' frames belong to the decoders just closed: forget them too, so + // a stale pointer is never read before the next `set_extra_camera_frames`. + self.extra_camera_frames.set([0; crate::camera_layers::MAX_EXTRA_CAMERAS]); + } + + /// Frames for cameras 1..=3 (index 0 = scene camera 1). A null or missing entry means that + /// camera is not drawn this frame. The compositor keeps the pointers and reads them in every + /// `compose_frame` until the next call, so they must stay valid until then. + pub unsafe fn set_extra_camera_frames(&self, frames: &[*const AVFrame]) { + let mut slots = [0usize; crate::camera_layers::MAX_EXTRA_CAMERAS]; + for (slot, frame) in slots.iter_mut().zip(frames) { + *slot = *frame as usize; + } + self.extra_camera_frames.set(slots); + } + + /// The frame set for scene camera `camera` (1..=3), null when there is none. + fn extra_camera_frame(&self, camera: usize) -> *const AVFrame { + let slots = self.extra_camera_frames.get(); + camera.checked_sub(1).and_then(|k| slots.get(k).copied()).unwrap_or(0) as *const AVFrame } /// Les verbes de dessin, côté Metal. Mêmes noms et mêmes paramètres que leurs @@ -1150,13 +1177,15 @@ impl Compositor { programme_t: f32, ) -> Result<()> { let full = [0.0, 0.0, 1.0, 1.0]; - self.draw_image_in(enc, path, full, [0.0, 0.0], 0.0, output_aspect, motion, programme_t) + let (aspect, t) = (output_aspect, programme_t); + self.draw_image_in(enc, path, full, [0.0, 0.0], 0.0, aspect, motion, t, 0.0) } /// `draw_image_bg` pour un rect quelconque — la bulle webcam s'en sert avec ses coins /// arrondis. `output_aspect` est le ratio du RECT visé, pas celui de la sortie : le crop /// « cover » se calcule contre la zone qu'on remplit. `motion` anime l'image au temps - /// programme `programme_t` ; la bulle passe `WallpaperMotion::None`. + /// programme `programme_t` ; la bulle passe `WallpaperMotion::None`. `transparency` = + /// `layer_fx.x` (0 = opaque): a camera layer's background box fades with the layer. #[allow(clippy::too_many_arguments)] unsafe fn draw_image_in( &self, @@ -1168,6 +1197,7 @@ impl Compositor { output_aspect: f32, motion: WallpaperMotion, programme_t: f32, + transparency: f32, ) -> Result<()> { let (tex, iw, ih) = self.cached_image(path)?; let ai = iw as f32 / ih.max(1) as f32; @@ -1191,6 +1221,7 @@ impl Compositor { mode: 6.0, fx: [0.0, 0.0, anim[0], anim[1]], mb, + layer_fx: [transparency, 0.0, 0.0, 0.0], ..Default::default() }, ); @@ -1209,6 +1240,8 @@ impl Compositor { /// /// `quad_px` / `radius_px` sont ceux de la bulle : le fond doit épouser ses coins arrondis, /// sinon un rectangle déborde derrière la caméra. + /// + /// `transparency` = the camera layer's `layer_fx.x`: the box fades with the camera. unsafe fn draw_webcam_bg( &self, enc: &metal::RenderCommandEncoderRef, @@ -1216,14 +1249,17 @@ impl Compositor { dst: [f32; 4], quad_px: [f32; 2], radius_px: f32, + transparency: f32, ) { const BLACK: [f32; 4] = [0.0, 0.0, 0.0, 1.0]; + let layer_fx = [transparency, 0.0, 0.0, 0.0]; let solid = |color: [f32; 4]| LayerCB { dst, quad_px, radius_px, mode: 1.0, color, + layer_fx, ..Default::default() }; match bg { @@ -1241,6 +1277,7 @@ impl Compositor { quad_px, radius_px, fx: [a.sin(), -a.cos(), 0.0, 0.0], + layer_fx, ..crate::frame_geometry::gradient_layer(stops, offsets, BLACK) }, ); @@ -1251,7 +1288,9 @@ impl Compositor { let aspect = if quad_px[1] > 0.0 { quad_px[0] / quad_px[1] } else { 1.0 }; let still = WallpaperMotion::None; if let Err(e) = - self.draw_image_in(enc, path, dst, quad_px, radius_px, aspect, still, 0.0) + self.draw_image_in( + enc, path, dst, quad_px, radius_px, aspect, still, 0.0, transparency, + ) { eprintln!("[compositor] fond webcam \"{path}\" : {e:#}"); self.draw_solid(enc, &solid(BLACK)); @@ -1673,6 +1712,13 @@ impl Compositor { if text.content.trim().is_empty() { continue; } + // The desk label shows only while the camera is covered: skip the rest of + // its section before rasterizing anything. + if crate::text_anim::is_desk_cover(text.animation.as_deref()) + && g.webcam_cover <= 0.0 + { + continue; + } let spec = crate::text::TextSpec { content: text.content.clone(), color: parse_hex(&text.color).unwrap_or([1.0, 1.0, 1.0, 1.0]), @@ -1706,10 +1752,11 @@ impl Compositor { }) else { continue; }; - let anim = crate::text_anim::text_animation_state( + let anim = crate::text_anim::annotation_text_state( text.animation.as_deref(), (t - a.start_sec as f32) * 1000.0, ((a.end_sec - a.start_sec) * 1000.0) as f32, + g.webcam_cover, ); let anim_px = rh / crate::text_anim::ANIMATION_REFERENCE_HEIGHT; let (mut ax, mut ay, mut aw, mut ah) = ( @@ -2659,97 +2706,141 @@ impl Compositor { // --- caméra : ombre PiP puis vidéo --- let enc = self.begin_pass(cmd_buf, &self.rt, None, &self.pipeline_main)?; - if let (true, Some((wy, wuv))) = (lp.has_webcam, webcam_tex.as_ref()) { - let [cu0, cv0, cu1, cv1] = crate::frame_geometry::webcam_source_rect( - [wcw, wch], - [wtw as f32, wth as f32], - scene_ref.as_ref().and_then(|scene| scene.layout.webcam_crop), - g.w_px[0] / g.w_px[1].max(0.0001), - ); - let (u0, u1) = if lp.webcam_mirror { (cu1, cu0) } else { (cu0, cu1) }; - let webcam_is_block = matches!( - g.scene_preset.as_deref(), - Some("dual-frame") | Some("vertical-stack") - ); - // Effet d'arrière-plan : le mode vient de la scène, le masque par pixel de - // l'inférence. Les DEUX sont requis — un mode sans masque rendrait la webcam - // invisible en détourage, donc tant que rien n'a été segmenté on dessine la - // piste telle quelle. C'est aussi ce qui rend le premier lancement gracieux. - let mask = self.webcam_mask.borrow(); - let effect = scene_ref - .as_ref() - .and_then(|s| s.webcam_effect.as_ref()) - .filter(|_| mask.is_some()) - .map(|e| (e.shader_code(), e)) - .filter(|(code, _)| *code > 0.0); - - // L'ombre appartient à la bulle PiP. En détourage il n'y a plus de bulle — une - // ombre portée par un rectangle invisible se lit comme un artefact. Le test porte - // sur le code de la SCÈNE et non sur celui envoyé au shader : le fond personnalisé - // part lui aussi en détourage ci-dessous, mais sa bulle, elle, est bien peinte et - // garde donc son ombre. - let is_cutout = matches!(effect, Some((code, _)) if code == 1.0); - if cfg.shadow && !webcam_is_block && !is_cutout && g.shape_fade > 0.0 { - self.draw_shadow( - enc, - g.w_dst, - g.w_px, - g.w_radius, - WEBCAM_SHADOW_SPREAD_FRAC * g.frame_min_px, - [0.0, WEBCAM_SHADOW_OFFSET_FRAC * g.frame_min_px], - WEBCAM_SHADOW_OPACITY * g.shape_fade, + // Camera 0's crop, mirror and desk turn as camera settings, plus its homography + // (`Scene::camera(0)`): with one, `src` is not read and the crop is dropped. + let cam0 = g.camera0_settings(scene_ref.as_ref()); + let webcam_is_block = matches!( + g.scene_preset.as_deref(), + Some("dual-frame") | Some("vertical-stack") + ); + // Effet d'arrière-plan : le mode vient de la scène, le masque par pixel de + // l'inférence. Les DEUX sont requis — un mode sans masque rendrait la webcam + // invisible en détourage, donc tant que rien n'a été segmenté on dessine la + // piste telle quelle. C'est aussi ce qui rend le premier lancement gracieux. + let mask = self.webcam_mask.borrow(); + let effect = scene_ref + .as_ref() + .and_then(|s| s.webcam_effect.as_ref()) + .filter(|_| mask.is_some()) + .map(|e| (e.shader_code(), e)) + .filter(|(code, _)| *code > 0.0); + // L'ombre appartient à la bulle PiP. En détourage il n'y a plus de bulle — une + // ombre portée par un rectangle invisible se lit comme un artefact. Le test porte + // sur le code de la SCÈNE et non sur celui envoyé au shader : le fond personnalisé + // part lui aussi en détourage ci-dessous, mais sa bulle, elle, est bien peinte et + // garde donc son ombre. + let is_cutout = matches!(effect, Some((code, _)) if code == 1.0); + // Fond personnalisé : on PEINT le fond dans la bulle, puis on y découpe la caméra + // par-dessus — le mélange alpha donne `lerp(fond, caméra, personne)`, soit exactement + // ce que la branche « mode 3 » du shader calculait, mais pour les TROIS sortes de + // fond. Le shader ne sait peindre qu'une couleur plate sous le masque ; dégradés et + // images y tombaient sur du noir, et le défaut EST une image. L'ordre est imposé : + // ombre, puis fond, puis caméra. `custom_bg` = Some(background) in that mode. + let (effect_code, blur_intensity, custom_bg) = match effect { + Some((code, e)) if code > 2.5 => (1.0, 0.0, Some(e.background.as_ref())), + Some((code, e)) => (code, e.blur_intensity.clamp(0.0, 1.0), None), + None => (0.0, 0.0, None), + }; + // Metal tolère l'index 3 non lié tant que `fx.z` reste à 0 : la branche n'est + // pas prise, la texture n'est pas échantillonnée. Dès qu'il monte, elle doit + // l'être sur TOUT draw capable de la prendre — ici seul celui de la caméra 0. L'état + // d'un encodeur est rémanent, donc lier avant le draw suffit, et l'ombre puis le + // fond qui précèdent sont en modes 1/2/5/6, que `ps_main` garde hors de la branche + // (`mode < 0.5`). The extra cameras draw with `fx.z = 0` and never read it. + // + // Pas de déliaison après coup, contrairement au chemin Windows qui remet le slot + // t3 à `None` : cet état meurt avec l'encodeur, et les annotations en ouvrent un + // autre. Il n'y a rien sur quoi fuir. + if let Some(m) = mask.as_ref() { + enc.set_fragment_texture(3, Some(&m.tex)); + } + // The extra cameras' textures, held to the end of `compose_frame` like `webcam_tex`. + let mut extra_textures = Vec::new(); + if g.camera_layers.is_empty() { + // Without layout regions: camera 0 from `w_dst`, as before the layers. + if let (true, Some((wy, wuv))) = (lp.has_webcam, webcam_tex.as_ref()) { + // Mirror and the desk-shot turn are both bound swaps: u for horizontal, v for + // vertical. + let src = crate::frame_geometry::camera_source_rect( + Some(&cam0), + [wcw, wch], + [wtw as f32, wth as f32], + g.w_px[0] / g.w_px[1].max(0.0001), ); + if cfg.shadow && !webcam_is_block && !is_cutout && g.shape_fade > 0.0 { + self.draw_shadow( + enc, + g.w_dst, + g.w_px, + g.w_radius, + WEBCAM_SHADOW_SPREAD_FRAC * g.frame_min_px, + [0.0, WEBCAM_SHADOW_OFFSET_FRAC * g.frame_min_px], + WEBCAM_SHADOW_OPACITY * g.shape_fade, + ); + } + if let Some(bg) = custom_bg { + self.draw_webcam_bg(enc, bg, g.w_dst, g.w_px, g.w_radius, 0.0); + } + let cb = g.webcam_video_cb(src, w_valid, effect_code, blur_intensity); + let cb = crate::frame_geometry::with_camera_homography(cb, Some(&cam0)); + self.draw_video(enc, &cb, wy, wuv); } - - // Fond personnalisé : on PEINT le fond dans la bulle, puis on y découpe la caméra - // par-dessus — le mélange alpha donne `lerp(fond, caméra, personne)`, soit exactement - // ce que la branche « mode 3 » du shader calculait, mais pour les TROIS sortes de - // fond. Le shader ne sait peindre qu'une couleur plate sous le masque ; dégradés et - // images y tombaient sur du noir, et le défaut EST une image. L'ordre est imposé : - // ombre, puis fond, puis caméra. - let (effect_code, blur_intensity) = match effect { - Some((code, e)) if code > 2.5 => { - self.draw_webcam_bg(enc, e.background.as_ref(), g.w_dst, g.w_px, g.w_radius); - (1.0, 0.0) + } else { + // Every planned layer, in order: shadow, then (camera 0) the custom background, + // then the video. Camera 0 keeps its frame, effects and mask; the extra cameras + // draw from `set_extra_camera_frames` without effects, and a missing one is skipped. + let render = [rw, rh]; + let cam0_base = g.webcam_video_cb([0.0; 4], w_valid, effect_code, blur_intensity); + for plan in &g.camera_layers { + if plan.camera == 0 { + let (true, Some((wy, wuv))) = (lp.has_webcam, webcam_tex.as_ref()) else { + continue; + }; + let video = crate::frame_geometry::camera_layer_cb( + plan, + Some(&cam0), + [wcw, wch], + [wtw as f32, wth as f32], + render, + &cam0_base, + ); + if let Some(shadow) = + g.camera_layer_shadow(plan, &video, render, cfg.shadow, is_cutout) + { + self.draw_solid(enc, &shadow); + } + if let Some(bg) = custom_bg { + let (dst, quad_px, radius_px) = (video.dst, video.quad_px, video.radius_px); + self.draw_webcam_bg(enc, bg, dst, quad_px, radius_px, video.layer_fx[0]); + } + self.draw_video(enc, &video, wy, wuv); + } else { + // Tolerant like camera 0: no frame, or no pixel buffer in it, skips the layer. + let frame = self.extra_camera_frame(plan.camera); + let Ok((ey, euv)) = self.nv12_srvs(frame) else { continue }; + let (tw, th) = self.tex_dims(frame); + let visible = [(*frame).width as f32, (*frame).height as f32]; + let tex = [tw as f32, th as f32]; + let valid = [visible[0] / tex[0].max(1.0), visible[1] / tex[1].max(1.0)]; + let video = crate::frame_geometry::camera_layer_cb( + plan, + scene_ref.as_ref().and_then(|s| s.camera(plan.camera)), + visible, + tex, + render, + &crate::frame_geometry::extra_camera_base_cb(valid), + ); + if let Some(shadow) = + g.camera_layer_shadow(plan, &video, render, cfg.shadow, false) + { + self.draw_solid(enc, &shadow); + } + self.draw_video(enc, &video, &ey, &euv); + extra_textures.push((ey, euv)); } - Some((code, e)) => (code, e.blur_intensity.clamp(0.0, 1.0)), - None => (0.0, 0.0), - }; - - // Metal tolère l'index 3 non lié tant que `fx.z` reste à 0 : la branche n'est - // pas prise, la texture n'est pas échantillonnée. Dès qu'il monte, elle doit - // l'être sur TOUT draw capable de la prendre — ici il n'y en a qu'un. L'état - // d'un encodeur est rémanent, donc lier avant le draw suffit, et l'ombre puis le - // fond qui précèdent sont en modes 1/2/5/6, que `ps_main` garde hors de la branche - // (`mode < 0.5`). - // - // Pas de déliaison après coup, contrairement au chemin Windows qui remet le slot - // t3 à `None` : cet état meurt avec l'encodeur, et les annotations en ouvrent un - // autre. Il n'y a rien sur quoi fuir. - if let Some(m) = mask.as_ref() { - enc.set_fragment_texture(3, Some(&m.tex)); } - self.draw_video( - enc, - &LayerCB { - dst: g.w_dst, - src: [u0, cv0, u1, cv1], - quad_px: g.w_px, - radius_px: g.w_radius, - mode: 0.0, - // `color.a` porte l'alpha du découpage (`color.a * personne`) ; le RGB n'est - // plus lu, le fond ayant déjà été peint sous la caméra. - color: [0.0, 0.0, 0.0, 1.0], - fx: [w_valid[0], w_valid[1], effect_code, blur_intensity], - src_prev: [u0, cv0, u1, cv1], - dst_prev: g.w_dst_prev, - mb: [g.mb_taps, g.mb_amount, 1.0, 0.0], - ..Default::default() - }, - wy, - wuv, - ); } + drop(mask); enc.end_encoding(); diff --git a/crates/compositor/src/compositor_windows.rs b/crates/compositor/src/compositor_windows.rs index 11952713d..74c60b07f 100644 --- a/crates/compositor/src/compositor_windows.rs +++ b/crates/compositor/src/compositor_windows.rs @@ -92,6 +92,15 @@ struct DofPyramid { height: u32, } +/// An extra camera's frame for this `compose_frame`: its private copy's plane views, and the +/// visible and texture sizes `camera_layer_cb` crops against. +struct ExtraCamera { + y: ID3D11ShaderResourceView, + uv: ID3D11ShaderResourceView, + visible: [f32; 2], + tex: [f32; 2], +} + pub struct Compositor { dev: ID3D11Device, ctx: ID3D11DeviceContext, @@ -171,6 +180,9 @@ pub struct Compositor { /// the copy's size and format on every hit, so a new texture landing on an old address /// is never handed a copy sized for the old one. `clear_srv_cache` frees stale entries. srv_cache: RefCell>, + /// The frames of cameras 1..=3 (`set_extra_camera_frames`), as addresses (0 = none) so + /// the compositor's auto traits stay what they were without the raw pointers. + extra_camera_frames: Cell<[usize; crate::camera_layers::MAX_EXTRA_CAMERAS]>, live_params: RefCell, /// Scène pilotée par l'app (contrat) : quand présente, remplace le layout fixture de /// `timeline()`. Voir `scene.rs` / `SceneDescription` (TS). @@ -762,6 +774,7 @@ impl Compositor { timeline_t_override: RefCell::new(None), programme_time: RefCell::new(None), srv_cache: RefCell::new(HashMap::new()), + extra_camera_frames: Cell::new([0; crate::camera_layers::MAX_EXTRA_CAMERAS]), live_params: RefCell::new(LiveParams::default()), scene: RefCell::new(None), text_raster: match crate::text::TextRasterizer::new() { @@ -840,6 +853,53 @@ impl Compositor { /// depuis le layout preset au lieu du planning fixture. pub fn set_scene(&self, s: Option) { *self.scene.borrow_mut() = s; + // A new scene can drop the regions that drew an extra camera: forget its frames so a + // stale pointer is never drawn before the next `set_extra_camera_frames`. + self.extra_camera_frames.set([0; crate::camera_layers::MAX_EXTRA_CAMERAS]); + } + + /// Frames for cameras 1..=3 (index 0 = scene camera 1). A null or missing entry means that + /// camera is not drawn this frame. The compositor keeps the pointers and reads them in every + /// `compose_frame` until the next call, so they must stay valid until then. They go through + /// `nv12_srvs` and its cache like camera 0's frame (`clear_srv_cache` clears them alike). + pub unsafe fn set_extra_camera_frames(&self, frames: &[*const AVFrame]) { + let mut slots = [0usize; crate::camera_layers::MAX_EXTRA_CAMERAS]; + for (slot, frame) in slots.iter_mut().zip(frames) { + *slot = *frame as usize; + } + self.extra_camera_frames.set(slots); + } + + /// The frame set for scene camera `camera` (1..=3), null when there is none. + fn extra_camera_frame(&self, camera: usize) -> *const AVFrame { + let slots = self.extra_camera_frames.get(); + camera.checked_sub(1).and_then(|k| slots.get(k).copied()).unwrap_or(0) as *const AVFrame + } + + /// The copies (`nv12_srvs`) of every extra camera `layers` draws, with their visible and + /// texture sizes. Tolerant: a camera without a frame, or whose frame has no D3D11 texture, + /// is `None` and simply not drawn. + unsafe fn extra_camera_srvs( + &self, + layers: &[crate::camera_layers::CameraLayerPlan], + ) -> [Option; crate::camera_layers::MAX_EXTRA_CAMERAS] { + let mut out: [Option; crate::camera_layers::MAX_EXTRA_CAMERAS] = + Default::default(); + for (k, slot) in out.iter_mut().enumerate() { + let frame = self.extra_camera_frame(k + 1); + if frame.is_null() || !layers.iter().any(|l| l.camera == k + 1) { + continue; + } + let Ok((y, uv)) = self.nv12_srvs(frame) else { continue }; + let (tw, th) = self.tex_dims(frame); + *slot = Some(ExtraCamera { + y, + uv, + visible: [(*frame).width as f32, (*frame).height as f32], + tex: [tw as f32, th as f32], + }); + } + out } /// The Y (R8) and UV (R8G8) views of the decoder frame — over a private COPY, never @@ -1227,13 +1287,14 @@ impl Compositor { programme_t: f32, ) -> Result<()> { let full = [0.0, 0.0, 1.0, 1.0]; - self.draw_image_in(path, full, [0.0, 0.0], 0.0, output_aspect, motion, programme_t) + self.draw_image_in(path, full, [0.0, 0.0], 0.0, output_aspect, motion, programme_t, 0.0) } /// `draw_image_bg` pour un rect quelconque — la bulle webcam s'en sert avec ses coins /// arrondis. `output_aspect` est le ratio du RECT visé, pas celui de la sortie : le crop /// « cover » se calcule contre la zone qu'on remplit. `motion` anime l'image au temps - /// programme `programme_t` ; la bulle passe `WallpaperMotion::None`. + /// programme `programme_t` ; la bulle passe `WallpaperMotion::None`. `transparency` = + /// `layer_fx.x` (0 = opaque): a camera layer's background box fades with the layer. #[allow(clippy::too_many_arguments)] unsafe fn draw_image_in( &self, @@ -1244,6 +1305,7 @@ impl Compositor { output_aspect: f32, motion: WallpaperMotion, programme_t: f32, + transparency: f32, ) -> Result<()> { let (srv, iw, ih) = self.cached_image(path)?; let ai = iw as f32 / ih as f32; @@ -1268,6 +1330,7 @@ impl Compositor { mode: 6.0, fx: [0.0, 0.0, anim[0], anim[1]], mb, + layer_fx: [transparency, 0.0, 0.0, 0.0], ..Default::default() }); Ok(()) @@ -1285,20 +1348,25 @@ impl Compositor { /// /// `quad_px` / `radius_px` sont ceux de la bulle : le fond doit épouser ses coins arrondis, /// sinon un rectangle déborde derrière la caméra. + /// + /// `transparency` = the camera layer's `layer_fx.x`: the box fades with the camera. unsafe fn draw_webcam_bg( &self, bg: Option<&SceneBackground>, dst: [f32; 4], quad_px: [f32; 2], radius_px: f32, + transparency: f32, ) { const BLACK: [f32; 4] = [0.0, 0.0, 0.0, 1.0]; + let layer_fx = [transparency, 0.0, 0.0, 0.0]; let solid = |color: [f32; 4]| LayerCB { dst, quad_px, radius_px, mode: 1.0, color, + layer_fx, ..Default::default() }; match bg { @@ -1315,6 +1383,7 @@ impl Compositor { quad_px, radius_px, fx: [dir[0], dir[1], 0.0, 0.0], + layer_fx, ..crate::frame_geometry::gradient_layer(stops, offsets, BLACK) }); } @@ -1324,7 +1393,9 @@ impl Compositor { let aspect = if quad_px[1] > 0.0 { quad_px[0] / quad_px[1] } else { 1.0 }; let still = WallpaperMotion::None; if let Err(e) = - self.draw_image_in(path, dst, quad_px, radius_px, aspect, still, 0.0) + self.draw_image_in( + path, dst, quad_px, radius_px, aspect, still, 0.0, transparency, + ) { eprintln!("[compositor] fond webcam \"{}\" : {:#}", path, e); self.draw_solid(&solid(BLACK)); @@ -2022,8 +2093,9 @@ impl Compositor { programme_time: *self.programme_time.borrow(), }); self.footage.set(Some(g.footage_quad([self.rw(), self.rh()]))); + // The extra cameras' copies, made before anything is drawn like the two above. + let extra_cams = self.extra_camera_srvs(&g.camera_layers); let scene_preset = g.scene_preset.clone(); - let mb_taps = g.mb_taps; let mb_amount = g.mb_amount; let source_t = g.source_t; let _padding_scale = g.padding_scale; @@ -2034,7 +2106,6 @@ impl Compositor { let s_radius = g.s_radius; let frame_min_px = g.frame_min_px; let w_dst = g.w_dst; - let w_dst_prev = g.w_dst_prev; let w_px = g.w_px; let w_radius = g.w_radius; let shape_fade = g.shape_fade; @@ -2435,99 +2506,144 @@ impl Compositor { // // Le center-crop carré de square/circle en est un cas particulier (boîte 1:1) — il n'a // plus besoin d'être traité à part. - let [su0, sv0, su1, sv1] = crate::frame_geometry::webcam_source_rect( - [wcw, wch], - [wtw as f32, wth as f32], - scene_ref - .as_ref() - .and_then(|scene| scene.layout.webcam_crop), - w_px[0] / w_px[1].max(0.0001), + // Camera 0's crop, mirror and desk turn as camera settings, plus its homography + // (`Scene::camera(0)`): with one, `src` is not read and the crop is dropped. + let cam0 = g.camera0_settings(scene_ref.as_ref()); + // L'ombre portée appartient à la bulle flottante PiP : elle se retire avec elle + // (`shape_fade`), pour qu'au plein écran plus rien n'encadre la caméra. C'est une + // ombre légère NON paramétrable — indépendante du slider Shadow, qui ne pilote plus + // que l'écran (`WEBCAM_SHADOW_OPACITY`, pas `shadow_scale`) — et propre au PiP : les + // blocs side-by-side / top-bottom soudent la caméra à l'écran et n'en portent aucune + // (parité `preset.shadow` web, `null` hors PiP dans `compositeLayout.ts`). + let webcam_is_block = matches!( + scene_preset.as_deref(), + Some("dual-frame") | Some("vertical-stack"), ); - // miroir = échanger les bornes u du rect source (flip horizontal). - let (u0, u1) = if lp.webcam_mirror { (su1, su0) } else { (su0, su1) }; - if lp.has_webcam { - // L'ombre portée appartient à la bulle flottante PiP : elle se retire avec elle - // (`shape_fade`), pour qu'au plein écran plus rien n'encadre la caméra. C'est une - // ombre légère NON paramétrable — indépendante du slider Shadow, qui ne pilote plus - // que l'écran (`WEBCAM_SHADOW_OPACITY`, pas `shadow_scale`) — et propre au PiP : les - // blocs side-by-side / top-bottom soudent la caméra à l'écran et n'en portent aucune - // (parité `preset.shadow` web, `null` hors PiP dans `compositeLayout.ts`). - let webcam_is_block = matches!( - scene_preset.as_deref(), - Some("dual-frame") | Some("vertical-stack"), - ); - // L'ombre appartient à la bulle PiP. En détourage il n'y a plus de bulle — une - // ombre portée par un rectangle invisible se lit comme un artefact. - let is_cutout = matches!( - scene_ref.as_ref().and_then(|s| s.webcam_effect.as_ref()), - Some(e) if e.shader_code() == 1.0 - ) && self.webcam_mask.borrow().is_some(); - if cfg.shadow && !webcam_is_block && !is_cutout && shape_fade > 0.0 { - let strength = WEBCAM_SHADOW_OPACITY * shape_fade; - self.draw_shadow( - w_dst, - w_px, - w_radius, - WEBCAM_SHADOW_SPREAD_FRAC * frame_min_px, - [0.0, WEBCAM_SHADOW_OFFSET_FRAC * frame_min_px], - strength, - ); - } - // Effet d'arrière-plan : le mode vient de la scène, le masque par pixel de - // l'inférence. Les DEUX sont requis — un mode sans masque rendrait la webcam - // invisible en détourage, donc tant que rien n'a été segmenté on dessine la piste - // telle quelle. C'est aussi ce qui rend le premier lancement gracieux. - let mask = self.webcam_mask.borrow(); - let effect = scene_ref - .as_ref() - .and_then(|s| s.webcam_effect.as_ref()) - .filter(|_| mask.is_some()) - .map(|e| (e.shader_code(), e)) - .filter(|(code, _)| *code > 0.0); - - // Fond personnalisé : on PEINT le fond dans la bulle, puis on y découpe la caméra - // par-dessus — le mélange alpha donne `lerp(fond, caméra, personne)`, soit exactement - // ce que la branche « mode 3 » du shader calculait, mais pour les TROIS sortes de - // fond. Le shader ne sait peindre qu'une couleur plate sous le masque ; dégradés et - // images y tombaient sur du noir, et le défaut EST une image. - let (effect_code, blur_intensity) = match effect { - Some((code, e)) if code > 2.5 => { - self.draw_webcam_bg(e.background.as_ref(), w_dst, w_px, w_radius); - (1.0, 0.0) - } - Some((code, e)) => (code, e.blur_intensity.clamp(0.0, 1.0)), - None => (0.0, 0.0), - }; - + // L'ombre appartient à la bulle PiP. En détourage il n'y a plus de bulle — une + // ombre portée par un rectangle invisible se lit comme un artefact. + let is_cutout = matches!( + scene_ref.as_ref().and_then(|s| s.webcam_effect.as_ref()), + Some(e) if e.shader_code() == 1.0 + ) && self.webcam_mask.borrow().is_some(); + // Effet d'arrière-plan : le mode vient de la scène, le masque par pixel de + // l'inférence. Les DEUX sont requis — un mode sans masque rendrait la webcam + // invisible en détourage, donc tant que rien n'a été segmenté on dessine la piste + // telle quelle. C'est aussi ce qui rend le premier lancement gracieux. + let mask = self.webcam_mask.borrow(); + let effect = scene_ref + .as_ref() + .and_then(|s| s.webcam_effect.as_ref()) + .filter(|_| mask.is_some()) + .map(|e| (e.shader_code(), e)) + .filter(|(code, _)| *code > 0.0); + // Fond personnalisé : on PEINT le fond dans la bulle, puis on y découpe la caméra + // par-dessus — le mélange alpha donne `lerp(fond, caméra, personne)`, soit exactement + // ce que la branche « mode 3 » du shader calculait, mais pour les TROIS sortes de + // fond. Le shader ne sait peindre qu'une couleur plate sous le masque ; dégradés et + // images y tombaient sur du noir, et le défaut EST une image. + // `custom_bg` = Some(background) in that mode; it is painted between shadow and camera. + let (effect_code, blur_intensity, custom_bg) = match effect { + Some((code, e)) if code > 2.5 => (1.0, 0.0, Some(e.background.as_ref())), + Some((code, e)) => (code, e.blur_intensity.clamp(0.0, 1.0), None), + None => (0.0, 0.0, None), + }; + // Camera 0's video draw: the mask, bound to slot 3 for this draw only. `draw_video` ne + // lie que les slots 0-1, donc le masque posé ici tient pour l'appel qui suit. Il est + // délié juste après pour ne pas fuir sur les calques d'annotation, qui utilisent eux + // aussi le slot 2 et au-delà. + let draw_camera0 = |cb: &LayerCB| { if let Some(m) = mask.as_ref() { - // `draw_video` ne lie que les slots 0-1, donc le masque posé ici tient pour - // l'appel qui suit. Il est délié juste après pour ne pas fuir sur les calques - // d'annotation, qui utilisent eux aussi le slot 2 et au-delà. self.ctx.PSSetShaderResources(3, Some(&[Some(m.srv.clone())])); } - self.draw_video( - &LayerCB { - dst: w_dst, - src: [u0, sv0, u1, sv1], - quad_px: w_px, - radius_px: w_radius, - mode: 0.0, - // `color.a` porte l'alpha du découpage (`color.a * personne`) ; le RGB n'est - // plus lu, le fond ayant déjà été peint sous la caméra. - color: [0.0, 0.0, 0.0, 1.0], - fx: [w_valid[0], w_valid[1], effect_code, blur_intensity], - src_prev: [u0, sv0, u1, sv1], // src fixe (pas de zoom webcam) - dst_prev: w_dst_prev, - mb: [mb_taps, mb_amount, 1.0, 0.0], - ..Default::default() - }, - &wy, - &wuv, - ); + self.draw_video(cb, &wy, &wuv); if mask.is_some() { self.ctx.PSSetShaderResources(3, Some(&[None])); } + }; + + if g.camera_layers.is_empty() { + // Without layout regions: camera 0 from `w_dst`, as before the layers. + let [u0, v0, u1, v1] = crate::frame_geometry::camera_source_rect( + Some(&cam0), + [wcw, wch], + [wtw as f32, wth as f32], + w_px[0] / w_px[1].max(0.0001), + ); + if lp.has_webcam { + if cfg.shadow && !webcam_is_block && !is_cutout && shape_fade > 0.0 { + let strength = WEBCAM_SHADOW_OPACITY * shape_fade; + self.draw_shadow( + w_dst, + w_px, + w_radius, + WEBCAM_SHADOW_SPREAD_FRAC * frame_min_px, + [0.0, WEBCAM_SHADOW_OFFSET_FRAC * frame_min_px], + strength, + ); + } + if let Some(bg) = custom_bg { + self.draw_webcam_bg(bg, w_dst, w_px, w_radius, 0.0); + } + let cb = g.webcam_video_cb([u0, v0, u1, v1], w_valid, effect_code, blur_intensity); + draw_camera0(&crate::frame_geometry::with_camera_homography(cb, Some(&cam0))); + } + } else { + // Every planned layer, in order: shadow, then (camera 0) the custom background, + // then the video. Camera 0 keeps its frame, effects and mask; the extra cameras + // draw from `set_extra_camera_frames` without effects, and a missing one is skipped. + let render = [self.rw(), self.rh()]; + let cam0_base = g.webcam_video_cb([0.0; 4], w_valid, effect_code, blur_intensity); + for plan in &g.camera_layers { + if plan.camera == 0 { + if !lp.has_webcam { + continue; + } + let video = crate::frame_geometry::camera_layer_cb( + plan, + Some(&cam0), + [wcw, wch], + [wtw as f32, wth as f32], + render, + &cam0_base, + ); + if let Some(shadow) = + g.camera_layer_shadow(plan, &video, render, cfg.shadow, is_cutout) + { + self.draw_solid(&shadow); + } + if let Some(bg) = custom_bg { + let t = video.layer_fx[0]; + self.draw_webcam_bg(bg, video.dst, video.quad_px, video.radius_px, t); + } + draw_camera0(&video); + } else { + let Some(cam) = extra_cams.get(plan.camera - 1).and_then(|c| c.as_ref()) + else { + continue; + }; + let settings = scene_ref.as_ref().and_then(|s| s.camera(plan.camera)); + let valid = [ + cam.visible[0] / cam.tex[0].max(1.0), + cam.visible[1] / cam.tex[1].max(1.0), + ]; + let video = crate::frame_geometry::camera_layer_cb( + plan, + settings, + cam.visible, + cam.tex, + render, + &crate::frame_geometry::extra_camera_base_cb(valid), + ); + if let Some(shadow) = + g.camera_layer_shadow(plan, &video, render, cfg.shadow, false) + { + self.draw_solid(&shadow); + } + self.draw_video(&video, &cam.y, &cam.uv); + } + } } + drop(mask); // --- annotations : calque le plus haut, comme dans le DOM de la preview (le calque y est // monté après la vidéo). Ancrées sur `s_ann`, le rect ÉCRAN SANS ZOOM — c'est le conteneur @@ -2741,6 +2857,13 @@ impl Compositor { if text.content.trim().is_empty() { continue; } + // The desk label shows only while the camera is covered: skip the rest of + // its section before rasterizing anything. + if crate::text_anim::is_desk_cover(text.animation.as_deref()) + && g.webcam_cover <= 0.0 + { + continue; + } // `font_size_rel` est une fraction de la HAUTEUR DE LA BOÎTE D'ANCRAGE — rect // écran, ou cadre de sortie pour un sous-titre (cf. le contrat et // `annotationScale.ts`) : on la ramène en pixels de sortie ici, avec le même @@ -2786,10 +2909,13 @@ impl Compositor { // dispose ici) : dans une région accélérée, elle défile donc au rythme du // clip. À vitesse 1 — le cas de toutes les annotations existantes — c'est // exactement le timing de l'aperçu DOM. - let anim = crate::text_anim::text_animation_state( + // The desk label is the exception: its opacity is the camera cover, which + // runs on the screen clock (`annotation_text_state`). + let anim = crate::text_anim::annotation_text_state( text.animation.as_deref(), (t - annotation.start_sec as f32) * 1000.0, ((annotation.end_sec - annotation.start_sec) * 1000.0) as f32, + g.webcam_cover, ); // Les décalages sont donnés à la hauteur de référence : on les ramène à la // sortie, comme la taille de police, pour que l'animation ait la même @@ -3375,6 +3501,9 @@ impl Compositor { /// (p.ex. après un export) pour ne pas retenir indéfiniment des textures de pool. pub fn clear_srv_cache(&self) { self.srv_cache.borrow_mut().clear(); + // The extra cameras' frames belong to the decoders just closed: forget them too, so + // a stale pointer is never read before the next `set_extra_camera_frames`. + self.extra_camera_frames.set([0; crate::camera_layers::MAX_EXTRA_CAMERAS]); } } diff --git a/crates/compositor/src/extra_cameras.rs b/crates/compositor/src/extra_cameras.rs new file mode 100644 index 000000000..0b2833c84 --- /dev/null +++ b/crates/compositor/src/extra_cameras.rs @@ -0,0 +1,241 @@ +//! The extra cameras (k >= 1) a camera layout region draws, shared by the live preview +//! (`live.rs`) and the export walk (`timeline_walk.rs`). +//! +//! Index convention: camera 0 is the clip's `webcam`; camera k >= 1 is +//! `additional_cameras[k - 1]`. Only the cameras a non-empty layout region shows are opened, +//! and each is decoded only near those regions (`extra_camera_active`). Unlike camera 0, an +//! extra camera never shortens a clip and never holds its last picture: past its end its +//! layer disappears. + +use crate::camera_layers::MAX_EXTRA_CAMERAS; +use crate::ffi::AVFrame; +use crate::live::PREFETCH_LEAD_SEC; +use crate::pipeline::Decoder; +use crate::scene::{SceneCameraLayoutRegion, SceneClipCamera}; +use crate::timeline_walk::{frame_step, FrameStep, NextFrameTime}; +use anyhow::Result; +use std::collections::{HashMap, HashSet}; + +/// An extra camera of an export clip: its file and its offset +/// (camera source time = screen source time - `offset_sec`). +pub type ClipCamera = SceneClipCamera; + +/// What a clip opens for its extra cameras: index k-1 = camera k, `None` = not opened (no +/// layout region of the clip shows it, or the clip has no file for it). No extra camera at all +/// is the empty list. +pub(crate) type ExtraCameraKeys = Vec>; + +/// Re-seek an extra camera instead of decoding forward when its target is further ahead than +/// this (it sat idle between two regions). +pub(crate) const EXTRA_RESEEK_SEC: f64 = 1.0; + +/// Camera indices >= 1 that the non-empty `regions` draw, sorted, without duplicates. +pub(crate) fn cameras_in_regions(regions: &[SceneCameraLayoutRegion]) -> Vec { + let mut cameras: Vec = regions + .iter() + .filter(|r| r.end_sec > r.start_sec) + .flat_map(|r| r.layers.iter().map(|l| l.camera)) + .filter(|camera| (1..=MAX_EXTRA_CAMERAS).contains(camera)) + .collect(); + cameras.sort_unstable(); + cameras.dedup(); + cameras +} + +pub(crate) fn has_camera_file(camera: Option<&SceneClipCamera>) -> bool { + camera.is_some_and(|c| !c.path.trim().is_empty()) +} + +/// The slots to open for `cameras` from a clip's camera files (`sources[k-1]` = camera k). +pub(crate) fn extra_camera_keys(cameras: &[usize], sources: &[SceneClipCamera]) -> ExtraCameraKeys { + let mut keys: ExtraCameraKeys = (1..=MAX_EXTRA_CAMERAS) + .map(|k| { + let source = sources.get(k - 1); + if cameras.contains(&k) && has_camera_file(source) { + source.cloned() + } else { + None + } + }) + .collect(); + while keys.last().is_some_and(|k| k.is_none()) { + keys.pop(); + } + keys +} + +/// Is extra camera `camera` near one of its layout regions at screen source time `t` — from +/// `PREFETCH_LEAD_SEC` before the region to its end? Outside, its decoder stays idle. +pub(crate) fn extra_camera_active(regions: &[SceneCameraLayoutRegion], camera: usize, t: f64) -> bool { + regions.iter().any(|r| { + r.end_sec > r.start_sec + && r.layers.iter().any(|l| l.camera == camera) + && (r.start_sec - PREFETCH_LEAD_SEC..=r.end_sec).contains(&t) + }) +} + +/// A camera's source time at screen source time `screen_t` (`camera = screen - offset`), +/// never before the file's start. Camera 0 (`live.rs`) and the extra cameras share it. +pub(crate) fn camera_source_time(screen_t: f64, offset_sec: f64) -> f64 { + (screen_t - offset_sec).max(0.0) +} + +/// An extra camera's open result: a file that will not open is `None` with one warning line, +/// and its layer is skipped — never drawn from another source. +pub(crate) fn opened_or_skipped(path: &str, opened: Result) -> Option { + match opened { + Ok(dec) => Some(dec), + Err(e) => { + eprintln!("WARNING: extra camera unreadable ({path}): {e:#}. Its layer will not be drawn."); + None + } + } +} + +/// Opens `path` into `decs` once. A file that will not open is remembered in `unreadable` +/// and never tried again, so its layer is skipped for the whole export. `true` when `decs` +/// holds a decoder for `path`. +pub(crate) fn open_once( + decs: &mut HashMap, + unreadable: &mut HashSet, + path: &str, + open: impl FnOnce(&str) -> Result, +) -> bool { + if decs.contains_key(path) { + return true; + } + if unreadable.contains(path) { + return false; + } + match opened_or_skipped(path, open(path)) { + Some(dec) => { + decs.insert(path.to_string(), dec); + true + } + None => { + unreadable.insert(path.to_string()); + false + } + } +} + +/// The list `set_extra_camera_frames` takes: null for an empty slot, else `frame` of the camera +/// (itself null when it has nothing to show). +pub(crate) fn extra_frame_list( + extra: &[Option], + frame: impl Fn(&T) -> *const AVFrame, +) -> Vec<*const AVFrame> { + extra.iter().map(|slot| slot.as_ref().map_or(std::ptr::null(), &frame)).collect() +} + +/// Advances `dec` to `target` (camera source time) with the webcam's hold semantics +/// (`frame_step`); re-seeks after an idle stretch or a jump back. Null past the camera's last +/// frame. `ended_at` is the camera source time past which the file has no frame left: the +/// decoder is not asked again until the target moves back before it. +pub(crate) unsafe fn step_extra_camera( + dec: &mut Decoder, + ended_at: &mut Option, + target: f64, +) -> Result<*const AVFrame> { + if ended_at.is_some_and(|end| target >= end) { + return Ok(std::ptr::null()); + } + let frame_dur = 1.0 / dec.fps().max(1.0); + let cur = dec.cur_frame(); + let far = cur.is_null() || { + let t = dec.cur_time_sec(); + target < t - frame_dur * 0.5 || target > t + EXTRA_RESEEK_SEC + }; + let frame = if far { + dec.seek_to(target)? + } else { + let mut frame = cur; + let mut guard = 0u32; + loop { + let next = dec.peek_next_time_sec()?; + // Unlike camera 0, an extra camera does not hold its last picture: past its + // end (one frame's worth of slack) its layer disappears. + if matches!(next, NextFrameTime::Eof) && target > dec.cur_time_sec() + frame_dur { + frame = std::ptr::null_mut(); + break; + } + match frame_step(next, 0.0, target) { + FrameStep::Commit => frame = dec.commit_peek()?, + FrameStep::CommitAndStop => { + frame = dec.commit_peek()?; + break; + } + FrameStep::Hold => break, + } + guard += 1; + if guard > 1000 { + break; + } + } + frame + }; + Ok(settle(ended_at, frame, target)) +} + +/// Seeks `dec` to `target` (camera source time). Null past the camera's last frame. +pub(crate) unsafe fn seek_extra_camera( + dec: &mut Decoder, + ended_at: &mut Option, + target: f64, +) -> Result<*const AVFrame> { + if ended_at.is_some_and(|end| target >= end) { + return Ok(std::ptr::null()); + } + let frame = dec.seek_to(target)?; + Ok(settle(ended_at, frame, target)) +} + +fn settle(ended_at: &mut Option, frame: *mut AVFrame, target: f64) -> *const AVFrame { + *ended_at = if frame.is_null() { Some(target) } else { None }; + frame +} + +#[cfg(test)] +mod tests { + use super::*; + use anyhow::anyhow; + + #[test] + fn export_skips_an_extra_camera_that_will_not_open() { + let mut decs: HashMap = HashMap::new(); + let mut unreadable = HashSet::new(); + let mut attempts = 0; + let mut refuse = |_: &str| -> Result { + attempts += 1; + Err(anyhow!("no such file")) + }; + assert!(!open_once(&mut decs, &mut unreadable, "/cam2.mp4", &mut refuse)); + // A later clip of the same export does not try the file again. + assert!(!open_once(&mut decs, &mut unreadable, "/cam2.mp4", &mut refuse)); + assert_eq!(attempts, 1); + assert!(decs.is_empty()); + + // A readable camera opens once and is reused by the next clip. + let mut opens = 0; + let mut accept = |_: &str| -> Result { + opens += 1; + Ok(7) + }; + assert!(open_once(&mut decs, &mut unreadable, "/cam3.mp4", &mut accept)); + assert!(open_once(&mut decs, &mut unreadable, "/cam3.mp4", &mut accept)); + assert_eq!(opens, 1); + assert_eq!(decs.get("/cam3.mp4"), Some(&7)); + } + + #[test] + fn extra_offsets_move_into_the_screen_clock() { + // camera + offset = screen: at offset 2 s, a camera frame at 0.5 s is due at 2.5 s + // of screen time, not before. + let held = camera_source_time(2.4, 2.0); + let due = camera_source_time(2.5, 2.0); + assert_eq!(frame_step(NextFrameTime::At(0.5), 0.0, held), FrameStep::Hold); + assert_eq!(frame_step(NextFrameTime::At(0.5), 0.0, due), FrameStep::Commit); + // Before the camera started, its source time clamps to its first frame. + assert_eq!(camera_source_time(1.0, 2.0), 0.0); + } +} diff --git a/crates/compositor/src/frame_geometry.rs b/crates/compositor/src/frame_geometry.rs index a8ef080e4..9cba80d82 100644 --- a/crates/compositor/src/frame_geometry.rs +++ b/crates/compositor/src/frame_geometry.rs @@ -27,7 +27,7 @@ use crate::config::Cfg; use crate::scene::{Scene, SceneCrop}; -/// Constant buffer d'un calque : **176 octets**, un par draw. +/// Constant buffer d'un calque : **256 octets**, un par draw. /// /// C'est le contrat partagé par les trois côtés — `cbuffer Layer` dans `shaders.hlsl`, /// `struct Layer` dans `shaders.metal` et `vk_shaders/layer.wgsl`, et ce struct. Ils doivent @@ -35,13 +35,15 @@ use crate::scene::{Scene, SceneCrop}; /// produit un shader qui lit `color` là où on a écrit `fx`. /// /// `align(16)` vient de la version macOS ; sous `repr(C)` seul, les offsets sont déjà -/// 0/16/32/40/44/48/64/80/96/112/128/144/160 des deux côtés — l'alignement Rust ne change que +/// 0/16/32/40/44/48/64/80/96/112/128/144/160/176 des deux côtés — l'alignement Rust ne change que /// l'adresse du struct, pas son contenu, et Windows le `copy_nonoverlapping` dans un /// constant buffer mappé où l'alignement source est sans effet. Les deux formes étaient /// donc compatibles ; les unifier évite qu'elles cessent de l'être. /// /// (Le commentaire d'origine annonçait « 64 octets ». Il n'a jamais été juste : dix champs, -/// trente-deux `f32`. Les trois derniers, le flou de mouvement de l'écran incliné, en font 176.) +/// trente-deux `f32`. Les trois suivants, le flou de mouvement de l'écran incliné, en font 176, et `cover`, le voile de la vue bureau, 192.) +/// (The camera layers' homography `persp` and transparency `layer_fx` make it 256: offsets 192 +/// and 240.) #[repr(C, align(16))] #[derive(Clone, Copy, Default)] pub struct LayerCB { @@ -62,6 +64,17 @@ pub struct LayerCB { pub trail_a: [f32; 4], pub trail_b: [f32; 4], pub trail_mb: [f32; 4], + /// Desk-view cover of the webcam layer: x = strength 0..1, y = blur radius (quad px), + /// z = dim factor, w unused. Zero everywhere else. + pub cover: [f32; 4], + /// Camera (mode 0) only: the rows of the homography H that maps a point of the layer's + /// `dst` quad (0..1, top-left origin) to the camera frame (0..1 of its valid part): + /// `[h0, h1, h2, 0], [h3, h4, h5, 0], [h6, h7, h8, 0]`. Read only when `layer_fx.y` is 1; + /// `src` is then ignored. Every other mode ignores it. + pub persp: [[f32; 4]; 3], + /// x = transparency 0..1 (0 = opaque), applied to the output of every mode; y = 1 when + /// `persp` applies; z, w = 0. All zero = the layer draws exactly as before the lanes. + pub layer_fx: [f32; 4], } impl LayerCB { @@ -330,6 +343,33 @@ pub(crate) fn cover_crop_uv(visible: [f32; 2], tex: [f32; 2], box_ar: f32) -> (f (u0, v0, u1, v1) } +/// How the webcam texture is laid onto its quad this frame. The backends swap the source +/// rect's u bounds for `flip_u` and its v bounds for `flip_v` — 180° is both — and drop the +/// webcam crop for `full_frame`. Derived per frame because a Full Camera region can turn +/// the camera for a desk shot. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct WebcamOrientation { + pub flip_u: bool, + pub flip_v: bool, + pub full_frame: bool, +} + +pub fn webcam_orientation( + region: Option<&crate::scene::SceneCameraFullscreenRegion>, + layout_mirror: bool, +) -> WebcamOrientation { + let Some(r) = region else { + return WebcamOrientation { flip_u: layout_mirror, flip_v: false, full_frame: false }; + }; + let turned = r.rotation == 180; + let mirror = r.mirror.unwrap_or(layout_mirror); + WebcamOrientation { + flip_u: mirror ^ turned, + flip_v: turned, + full_frame: turned && r.full_frame, + } +} + /// Camera equivalent of the screen crop pipeline: apply the user crop first, then a centred /// cover-crop inside that authored window so arbitrary layout slots never stretch the image. pub(crate) fn webcam_source_rect( @@ -346,6 +386,140 @@ pub(crate) fn webcam_source_rect( box_ar, ) } + +/// The camera's perspective correction, when it has one whose nine values are all finite. A +/// matrix with a NaN or an infinity would draw garbage (or nothing), so it counts as none. +pub(crate) fn camera_homography(camera: Option<&crate::scene::SceneCamera>) -> Option<[f32; 9]> { + camera + .and_then(|c| c.homography) + .filter(|h| h.iter().all(|v| v.is_finite())) +} + +/// The source rect a camera is drawn from: its crop, then a cover-crop to `box_ar`, then its +/// rotation (180° = both bounds swapped) and mirror (u bounds swapped), `flip_u = mirror ^ +/// turned` as `webcam_orientation` does. A camera without settings is drawn upright and +/// unmirrored. With a homography, crop, rotation and mirror are all ignored: the corner order +/// already defines the picture (the shader then does not read `src` at all). +pub(crate) fn camera_source_rect( + camera: Option<&crate::scene::SceneCamera>, + visible: [f32; 2], + tex: [f32; 2], + box_ar: f32, +) -> [f32; 4] { + let corrected = camera_homography(camera).is_some(); + let settings = camera.filter(|_| !corrected); + let [u0, v0, u1, v1] = webcam_source_rect(visible, tex, settings.and_then(|c| c.crop), box_ar); + let turned = settings.is_some_and(|c| c.rotation == 180); + let mirror = settings.and_then(|c| c.mirror).unwrap_or(false); + let (u0, u1) = if mirror ^ turned { (u1, u0) } else { (u0, u1) }; + let (v0, v1) = if turned { (v1, v0) } else { (v0, v1) }; + [u0, v0, u1, v1] +} + +/// The window of a corrected picture (`aspect` = its width/height) a box of ratio `box_ar` +/// shows, as `[u0, v0, du, dv]` of the picture's unit square: centred and cover-fitted, so the +/// picture is cropped, never stretched — what `cover_uv_rect` does for an uncorrected camera, +/// whose `src` the shader stops reading once a homography is set. The whole picture without a +/// usable aspect or box. +pub(crate) fn corrected_cover_window(aspect: Option, box_ar: f32) -> [f32; 4] { + let Some(aspect) = aspect.filter(|a| a.is_finite() && *a > 0.0) else { + return [0.0, 0.0, 1.0, 1.0]; + }; + if !(box_ar.is_finite() && box_ar > 0.0) { + return [0.0, 0.0, 1.0, 1.0]; + } + if box_ar > aspect { + let dv = aspect / box_ar; + [0.0, (1.0 - dv) * 0.5, 1.0, dv] + } else { + let du = box_ar / aspect; + [(1.0 - du) * 0.5, 0.0, du, 1.0] + } +} + +/// `cb` with the camera's homography in `persp` and `layer_fx.y = 1`; unchanged without one. +/// The matrix is composed with `corrected_cover_window` for the box `cb.quad_px`, so a box +/// whose ratio is not the camera's `aspect` (a frame-filling layer, a half of side-by-side) +/// crops the corrected picture instead of stretching it. +pub(crate) fn with_camera_homography( + mut cb: LayerCB, + camera: Option<&crate::scene::SceneCamera>, +) -> LayerCB { + if let Some(h) = camera_homography(camera) { + let box_ar = cb.quad_px[0] / cb.quad_px[1]; + let [u0, v0, du, dv] = corrected_cover_window(camera.and_then(|c| c.aspect), box_ar); + // H * C, with C = [[du, 0, u0], [0, dv, v0], [0, 0, 1]] taking box uv to picture uv. + let h: [f32; 9] = std::array::from_fn(|i| { + let (r, c) = (i / 3, i % 3); + match c { + 0 => h[r * 3] * du, + 1 => h[r * 3 + 1] * dv, + _ => h[r * 3] * u0 + h[r * 3 + 1] * v0 + h[r * 3 + 2], + } + }); + cb.persp = [ + [h[0], h[1], h[2], 0.0], + [h[3], h[4], h[5], 0.0], + [h[6], h[7], h[8], 0.0], + ]; + cb.layer_fx[1] = 1.0; + } + cb +} + +/// The video draw of an extra camera before `camera_layer_cb` places it: mode 0, no background +/// effect (`fx.zw = 0`), `valid` = the valid fraction of its decoder texture, no motion trail. +pub(crate) fn extra_camera_base_cb(valid: [f32; 2]) -> LayerCB { + LayerCB { + mode: 0.0, + color: [0.0, 0.0, 0.0, 1.0], + fx: [valid[0], valid[1], 0.0, 0.0], + mb: [1.0, 0.0, 1.0, 0.0], + ..Default::default() + } +} + +/// The video draw of one planned camera layer: `base` (camera 0's `webcam_video_cb`, or +/// `extra_camera_base_cb` for the others) moved to the plan's rect, with its corners, its +/// source rect (`camera_source_rect`), its homography and its transparency (`1 - opacity`). +/// Planned layers have no motion trail: `dst_prev = dst` and one tap. The exception is camera +/// 0's untouched default layer (the plan's rect is `base.dst`, fully opaque — what a frame +/// outside every layout region plans): it keeps `base`'s trail, as on today's path. +pub(crate) fn camera_layer_cb( + plan: &crate::camera_layers::CameraLayerPlan, + camera: Option<&crate::scene::SceneCamera>, + visible_px: [f32; 2], + tex_px: [f32; 2], + render: [f32; 2], + base: &LayerCB, +) -> LayerCB { + let quad_px = [plan.dst[2] * render[0], plan.dst[3] * render[1]]; + let min_px = quad_px[0].min(quad_px[1]); + let src = camera_source_rect(camera, visible_px, tex_px, quad_px[0] / quad_px[1].max(0.0001)); + let cover = base.cover[0]; + let untouched = plan.dst == base.dst && plan.opacity >= 1.0; + let (dst_prev, mb) = if untouched { + (base.dst_prev, base.mb) + } else { + (plan.dst, [1.0, 0.0, base.mb[2], base.mb[3]]) + }; + with_camera_homography( + LayerCB { + dst: plan.dst, + src, + quad_px, + radius_px: plan.radius_frac * min_px, + src_prev: src, + dst_prev, + mb, + cover: [cover, 0.04 * min_px * cover, base.cover[2], base.cover[3]], + persp: [[0.0; 4]; 3], + layer_fx: [1.0 - plan.opacity.clamp(0.0, 1.0), 0.0, 0.0, 0.0], + ..*base + }, + camera, + ) +} /// Rétrécit un rect SOURCE déjà exprimé en UV (`[u0, v0, u1, v1]`) autour de son /// centre pour qu'il porte le ratio `box_ar` une fois rapporté aux pixels de la /// texture. C'est la forme générale de `object-fit: cover`, et LA primitive qui @@ -1908,6 +2082,14 @@ pub struct FrameGeometry { pub w_px: [f32; 2], pub w_radius: f32, pub shape_fade: f32, + /// Orientation of the webcam this frame (mirror, desk-shot turn, crop bypass). + pub webcam: WebcamOrientation, + /// Desk-view cover strength of the webcam this frame, 0..1 (`camera_fullscreen_cover_at`). + pub webcam_cover: f32, + /// Every camera layer to draw this frame, in draw order (`camera_layers::camera_layers_at`). + /// Empty when the scene has no camera layout regions: the backends then draw camera 0 from + /// `w_dst` as before. Otherwise camera 0 is in here too, and its `dst`/opacity come from it. + pub camera_layers: Vec, /// Cadre autour de l'écran : chrome de fenêtre plat (mode 14) ou appareil modelé (mode 17). /// `None` : aucun, et le rendu est celui d'avant le cadre, à l'octet. `Some` : `s_dst` est /// déjà la boîte rétrécie, et `s_radius` le rayon des coins de l'écran — des seuls coins BAS @@ -2146,6 +2328,109 @@ impl FrameGeometry { self.tilt_trail(render_px).filter(|_| !self.screen_trail(render_px)) } + /// The camera-0 video draw (mode 0), the same on every backend. `src` = the source rect with + /// the mirror / desk turn already applied as swapped bounds (it is also `src_prev`: only + /// `dst_prev` carries the motion), `valid` = the valid fraction of the decoder texture, + /// `effect_code` / `blur_intensity` = the background effect sent to the shader. + pub fn webcam_video_cb( + &self, + src: [f32; 4], + valid: [f32; 2], + effect_code: f32, + blur_intensity: f32, + ) -> LayerCB { + LayerCB { + dst: self.w_dst, + src, + quad_px: self.w_px, + radius_px: self.w_radius, + mode: 0.0, + // `color.a` carries the cutout alpha (`color.a * person`); the RGB is not read, the + // background has already been painted under the camera. + color: [0.0, 0.0, 0.0, 1.0], + fx: [valid[0], valid[1], effect_code, blur_intensity], + src_prev: src, + dst_prev: self.w_dst_prev, + mb: [self.mb_taps, self.mb_amount, 1.0, 0.0], + cover: [ + self.webcam_cover, + 0.04 * self.w_px[0].min(self.w_px[1]) * self.webcam_cover, + 0.35, + 0.0, + ], + ..Default::default() + } + } + + /// Camera 0's settings as a `SceneCamera`, for `camera_source_rect` / `camera_layer_cb`: + /// its orientation is the project mirror and the desk-view turn (`self.webcam`, expressed so + /// that `mirror ^ turned` gives back `flip_u`), its crop is the layout's (none for a + /// full-frame desk shot), and only its homography and the corrected picture's aspect (so a + /// box of another ratio cover-crops it, `with_camera_homography`) come from + /// `Scene::camera(0)`. + pub(crate) fn camera0_settings(&self, scene: Option<&Scene>) -> crate::scene::SceneCamera { + let turned = self.webcam.flip_v; + let settings = scene.and_then(|s| s.camera(0)); + crate::scene::SceneCamera { + index: 0, + rotation: if turned { 180 } else { 0 }, + mirror: Some(self.webcam.flip_u ^ turned), + crop: if self.webcam.full_frame { + None + } else { + scene.and_then(|s| s.layout.webcam_crop) + }, + homography: settings.and_then(|c| c.homography), + aspect: settings.and_then(|c| c.aspect), + } + } + + /// The drop shadow under a planned camera layer, `video` being its `camera_layer_cb`. Same + /// constants as camera 0's PiP shadow, and it fades with the layer (`video.layer_fx.x`). + /// None when `shadows_on` is off or the layer fills the frame. Camera 0 also keeps today's + /// rules: no shadow in the block presets nor in cutout (`camera0_cutout`), and it leaves + /// with the bubble (`shape_fade`). + pub(crate) fn camera_layer_shadow( + &self, + plan: &crate::camera_layers::CameraLayerPlan, + video: &LayerCB, + render: [f32; 2], + shadows_on: bool, + camera0_cutout: bool, + ) -> Option { + let strength = if plan.camera == 0 { + let block = matches!( + self.scene_preset.as_deref(), + Some("dual-frame") | Some("vertical-stack") + ); + if block || camera0_cutout { + return None; + } + WEBCAM_SHADOW_OPACITY * self.shape_fade + } else { + WEBCAM_SHADOW_OPACITY + }; + if !shadows_on || plan.fills_frame || strength <= 0.0 { + return None; + } + let spread = WEBCAM_SHADOW_SPREAD_FRAC * self.frame_min_px; + let offset = WEBCAM_SHADOW_OFFSET_FRAC * self.frame_min_px; + let (sx, sy) = (spread / render[0].max(1.0), spread / render[1].max(1.0)); + let oy = offset / render[1].max(1.0); + let dst = video.dst; + Some(LayerCB { + dst: [dst[0] - sx, dst[1] - sy + oy, dst[2] + 2.0 * sx, dst[3] + 2.0 * sy], + quad_px: [video.quad_px[0] + 2.0 * spread, video.quad_px[1] + 2.0 * spread], + radius_px: video.radius_px, + mode: 2.0, + color: [0.0, 0.0, 0.0, strength], + fx: [spread, 0.0, 0.0, 0.0], + mb: [0.0, 1.0, 1.0, 0.0], + layer_fx: [video.layer_fx[0], 0.0, 0.0, 0.0], + ..Default::default() + }) + } + /// Le calque du mode 18 : le rendu isolé de l'écran cadré (t2), recomposé le long de sa /// trajectoire. `src` = la coupe (le métrage relu directement là où le rendu isolé s'arrête /// au bord de la sortie), `quad_px` et `radius_px` = l'écran et ses coins : ce repli ne lit le @@ -2985,6 +3270,9 @@ pub fn plan_frame(input: &FrameGeometryInput) -> FrameGeometry { let zoom_regions = scene.map(|s| &s.zoom_regions).unwrap_or(&empty_zoom); let cam_regions = scene.map(|s| &s.camera_fullscreen_regions).unwrap_or(&empty_cam); + // The layout regions, for the seams a Full Camera region shares with them. + let empty_layouts: Vec = Vec::new(); + let layouts = scene.map(|s| &s.camera_layout_regions).unwrap_or(&empty_layouts); let webcam_reactive = scene.map(|s| s.layout.webcam_reactive_zoom).unwrap_or(false); let source_t = input.timeline_t_override.unwrap_or(frame / FPS); // Les transitions se mesurent à l'écran (`ScreenClock`), la frame précédente aussi : une @@ -3056,11 +3344,22 @@ pub fn plan_frame(input: &FrameGeometryInput) -> FrameGeometry { // Full Camera ignore le rétrécissement réactif de la webcam (design web : mélanger // "rétrécit pour le zoom" et "grandit en plein cadre" dans la même frame n'a pas de sens). let cam_progress = - crate::regions::camera_fullscreen_progress_at(cam_regions, source_t, &clock); + crate::regions::camera_fullscreen_progress_at(cam_regions, source_t, &clock, layouts); let cam_progress_prev = - crate::regions::camera_fullscreen_progress_at(cam_regions, source_t_prev, &clock); + crate::regions::camera_fullscreen_progress_at( + cam_regions, + source_t_prev, + &clock, + layouts, + ); let shape_fade = - crate::regions::camera_fullscreen_shape_at(cam_regions, source_t, &clock); + crate::regions::camera_fullscreen_shape_at(cam_regions, source_t, &clock, layouts); + let webcam = webcam_orientation( + crate::regions::camera_fullscreen_region_at(cam_regions, source_t, &clock, layouts), + lp.webcam_mirror, + ); + let webcam_cover = + crate::regions::camera_fullscreen_cover_at(cam_regions, source_t, &clock, layouts); // rétrécissement réactif : la webcam garde 70 % de sa taille pendant un zoom actif, quel // que soit son niveau (elle suivait 1/zoom, et rétrécissait donc d'autant plus que le zoom // était profond : ×0,6 au zoom maximal). L'enveloppe est celle de la région : elle descend @@ -3473,6 +3772,30 @@ pub fn plan_frame(input: &FrameGeometryInput) -> FrameGeometry { _ => 0.12, }, }; + // Camera layout regions: every camera layer of this frame, camera 0 included, on the + // same time and clock as Full Camera. Camera 0's default is the layer just planned + // above, so a region glides from and back to exactly where `w_dst` puts it. Without + // regions nothing is planned, and the backends keep drawing camera 0 from `w_dst`. + let camera_layers = match scene { + Some(s) if !s.camera_layout_regions.is_empty() => { + let default_cam0 = lp.has_webcam.then(|| crate::camera_layers::CameraLayerPlan { + camera: 0, + dst: w_dst, + radius_frac: w_radius / w_px[0].min(w_px[1]).max(1.0), + shape: lp.webcam_shape, + opacity: 1.0, + fills_frame: cam_progress >= 1.0, + }); + crate::camera_layers::camera_layers_at( + &s.camera_layout_regions, + &s.camera_fullscreen_regions, + source_t, + &clock, + default_cam0, + ) + } + _ => Vec::new(), + }; FrameGeometry { scene_preset, @@ -3504,6 +3827,9 @@ pub fn plan_frame(input: &FrameGeometryInput) -> FrameGeometry { w_px, w_radius, shape_fade, + webcam, + webcam_cover, + camera_layers, window_frame, screen_mask, } @@ -6305,6 +6631,7 @@ mod tests { .into_iter() .chain(c.quad_px) .chain([c.radius_px, c.mode]) + .chain(c.cover) .map(f32::to_bits) .collect() }) @@ -7302,7 +7629,7 @@ mod tests { #[test] fn layer_cb_matches_the_shader_constant_buffer() { use std::mem::{align_of, offset_of, size_of}; - assert_eq!(size_of::(), 176); + assert_eq!(size_of::(), 256); assert_eq!(align_of::(), 16); for (name, got, want) in [ ("dst", offset_of!(LayerCB, dst), 0), @@ -7318,11 +7645,527 @@ mod tests { ("trail_a", offset_of!(LayerCB, trail_a), 128), ("trail_b", offset_of!(LayerCB, trail_b), 144), ("trail_mb", offset_of!(LayerCB, trail_mb), 160), + ("cover", offset_of!(LayerCB, cover), 176), + ("persp", offset_of!(LayerCB, persp), 192), + ("layer_fx", offset_of!(LayerCB, layer_fx), 240), ] { assert_eq!(got, want, "offset de `{name}`"); } } + /// The camera-0 draw as every backend builds it today leaves both new lanes at zero: no + /// homography (`layer_fx.y = 0`) and opaque (`layer_fx.x = 0`), so the shaders take exactly + /// the path they took before the lanes existed. The other fields are the ones the three + /// backends used to spell out inline. + #[test] + fn the_webcam_draw_leaves_the_homography_and_transparency_lanes_at_zero() { + let cfg = crate::config::all().pop().expect("cfg"); + let scene = golden_scene(); + let g = plan_frame(&golden_input(&scene, &cfg)); + let src = [0.1, 0.2, 0.9, 0.8]; + let cb = g.webcam_video_cb(src, [1.0, 0.75], 2.0, 0.4); + assert_eq!(cb.persp, [[0.0; 4]; 3]); + assert_eq!(cb.layer_fx, [0.0; 4]); + assert_eq!(cb.dst, g.w_dst); + assert_eq!(cb.src, src); + assert_eq!(cb.src_prev, src); + assert_eq!(cb.quad_px, g.w_px); + assert_eq!(cb.radius_px, g.w_radius); + assert_eq!(cb.mode, 0.0); + assert_eq!(cb.color, [0.0, 0.0, 0.0, 1.0]); + assert_eq!(cb.fx, [1.0, 0.75, 2.0, 0.4]); + assert_eq!(cb.dst_prev, g.w_dst_prev); + assert_eq!(cb.mb, [g.mb_taps, g.mb_amount, 1.0, 0.0]); + let min_px = g.w_px[0].min(g.w_px[1]); + assert_eq!(cb.cover, [g.webcam_cover, 0.04 * min_px * g.webcam_cover, 0.35, 0.0]); + } + + fn extra_camera(index: usize) -> crate::scene::SceneCamera { + crate::scene::SceneCamera { + index, + rotation: 0, + mirror: None, + crop: None, + homography: None, + aspect: None, + } + } + + fn planned(camera: usize, opacity: f32) -> crate::camera_layers::CameraLayerPlan { + crate::camera_layers::CameraLayerPlan { + camera, + dst: [0.5, 0.25, 0.25, 0.5], + radius_frac: 0.2, + shape: 0, + opacity, + fills_frame: false, + } + } + + /// One planned layer's video draw: placed at the plan's rect with its corners, no trail, + /// the transparency is `1 - opacity`, and the camera's homography replaces its crop. + #[test] + fn camera_layer_cb_places_the_layer_and_carries_its_lanes() { + let render = [1920.0, 1080.0]; + let (visible, tex) = ([1280.0, 720.0], [1280.0, 736.0]); + let base = extra_camera_base_cb([1.0, 720.0 / 736.0]); + let plan = planned(1, 0.25); + let plain = camera_layer_cb(&plan, None, visible, tex, render, &base); + assert_eq!(plain.dst, plan.dst); + assert_eq!(plain.dst_prev, plan.dst); + assert_eq!(plain.quad_px, [480.0, 540.0]); + assert_eq!(plain.radius_px, 0.2 * 480.0); + assert_eq!(plain.layer_fx, [0.75, 0.0, 0.0, 0.0]); + assert_eq!(plain.persp, [[0.0; 4]; 3]); + assert_eq!(plain.src_prev, plain.src); + assert_eq!(plain.mb[0], 1.0); + assert_eq!(plain.fx, base.fx); + + // A crop moves the source rect; with a homography as well, the crop is ignored. + let crop = SceneCrop { x: 0.5, y: 0.0, width: 0.5, height: 1.0 }; + let cropped = extra_camera(1); + let cropped = crate::scene::SceneCamera { crop: Some(crop), ..cropped }; + let with_crop = camera_layer_cb(&plan, Some(&cropped), visible, tex, render, &base); + assert_ne!(with_crop.src, plain.src); + let h = [-1.0, 0.0, 1.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0]; + let corrected = crate::scene::SceneCamera { homography: Some(h), ..cropped.clone() }; + let warped = camera_layer_cb(&plan, Some(&corrected), visible, tex, render, &base); + assert_eq!(warped.layer_fx[1], 1.0); + assert_eq!(warped.layer_fx[0], 0.75); + let rows = [[-1.0, 0.0, 1.0, 0.0], [0.0, 1.0, 0.0, 0.0], [0.0, 0.0, 1.0, 0.0]]; + assert_eq!(warped.persp, rows); + assert_eq!(warped.src, plain.src); + + // Mirror swaps the u bounds only; 180° swaps both, and mirror on top of it un-swaps u. + let mirrored = crate::scene::SceneCamera { mirror: Some(true), ..extra_camera(1) }; + let m = camera_layer_cb(&plan, Some(&mirrored), visible, tex, render, &base).src; + assert_eq!(m, [plain.src[2], plain.src[1], plain.src[0], plain.src[3]]); + let turned = crate::scene::SceneCamera { rotation: 180, ..extra_camera(1) }; + let t = camera_layer_cb(&plan, Some(&turned), visible, tex, render, &base).src; + assert_eq!(t, [plain.src[2], plain.src[3], plain.src[0], plain.src[1]]); + let both = crate::scene::SceneCamera { mirror: Some(true), ..turned.clone() }; + let b = camera_layer_cb(&plan, Some(&both), visible, tex, render, &base).src; + assert_eq!(b, [plain.src[0], plain.src[3], plain.src[2], plain.src[1]]); + } + + /// Camera 0's default layer, planned outside every layout region at its own rect and fully + /// opaque, keeps its motion trail; moved or faded, it draws without one (R8). + #[test] + fn camera_0s_untouched_layer_keeps_its_trail() { + let render = [1920.0, 1080.0]; + let base = LayerCB { + dst: [0.7, 0.7, 0.2, 0.2], + dst_prev: [0.65, 0.7, 0.2, 0.2], + mb: [8.0, 0.5, 1.0, 0.0], + ..Default::default() + }; + let plan = crate::camera_layers::CameraLayerPlan { + camera: 0, + dst: base.dst, + radius_frac: 0.2, + shape: 0, + opacity: 1.0, + fills_frame: false, + }; + let kept = camera_layer_cb(&plan, None, [64.0; 2], [64.0; 2], render, &base); + assert_eq!(kept.dst_prev, base.dst_prev); + assert_eq!(kept.mb, base.mb); + + let moved = crate::camera_layers::CameraLayerPlan { dst: [0.1, 0.1, 0.2, 0.2], ..plan }; + let m = camera_layer_cb(&moved, None, [64.0; 2], [64.0; 2], render, &base); + assert_eq!(m.dst_prev, moved.dst); + assert_eq!(m.mb, [1.0, 0.0, 1.0, 0.0]); + + let faded = crate::camera_layers::CameraLayerPlan { opacity: 0.5, ..plan }; + let f = camera_layer_cb(&faded, None, [64.0; 2], [64.0; 2], render, &base); + assert_eq!(f.dst_prev, faded.dst); + assert_eq!(f.mb, [1.0, 0.0, 1.0, 0.0]); + } + + /// A corrected camera in a box of another ratio is cover-cropped, not stretched: the box's + /// local uv lands on a centred window of the corrected picture with the box's own ratio. + #[test] + fn a_corrected_camera_is_cover_fitted_into_its_box() { + let identity = [1.0, 0.0, 0.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0]; + let cam = crate::scene::SceneCamera { + homography: Some(identity), + aspect: Some(16.0 / 9.0), + ..extra_camera(2) + }; + let row = |cb: &LayerCB, r: usize| [cb.persp[r][0], cb.persp[r][1], cb.persp[r][2]]; + let near = |a: [f32; 3], b: [f32; 3]| a.iter().zip(b).all(|(x, y)| (x - y).abs() < 1e-5); + + // A box narrower than 16:9 (1752x1080, a 1.62 export): the full height, a centred + // slice of the width with the box's ratio. + let narrow = LayerCB { quad_px: [1752.0, 1080.0], ..Default::default() }; + let cb = with_camera_homography(narrow, Some(&cam)); + let du = (1752.0 / 1080.0) / (16.0 / 9.0); + assert!(near(row(&cb, 0), [du, 0.0, (1.0 - du) * 0.5]), "{:?}", cb.persp); + assert!(near(row(&cb, 1), [0.0, 1.0, 0.0]), "{:?}", cb.persp); + assert!(near(row(&cb, 2), [0.0, 0.0, 1.0]), "{:?}", cb.persp); + + // A wider box (a 16:9 picture in a 21:9 box): full width, a centred band of height. + let wide = LayerCB { quad_px: [2100.0, 900.0], ..Default::default() }; + let cb = with_camera_homography(wide, Some(&cam)); + let dv = (16.0 / 9.0) / (2100.0 / 900.0); + assert!(near(row(&cb, 0), [1.0, 0.0, 0.0]), "{:?}", cb.persp); + assert!(near(row(&cb, 1), [0.0, dv, (1.0 - dv) * 0.5]), "{:?}", cb.persp); + + // The box the template gives a PiP of this camera has its aspect: nothing is cropped. + let pip = LayerCB { quad_px: [422.4, 237.6], ..Default::default() }; + let cb = with_camera_homography(pip, Some(&cam)); + assert!(near(row(&cb, 0), [1.0, 0.0, 0.0]) && near(row(&cb, 1), [0.0, 1.0, 0.0])); + + // A real (projective) matrix is composed on the right: H * C, so H's third row picks + // up the window too. + let h = [0.8, 0.1, 0.05, -0.02, 0.9, 0.04, 0.1, -0.2, 1.0]; + let cam = crate::scene::SceneCamera { homography: Some(h), ..cam }; + let cb = with_camera_homography(narrow, Some(&cam)); + let u0 = (1.0 - du) * 0.5; + assert!(near(row(&cb, 2), [0.1 * du, -0.2, 0.1 * u0 + 1.0]), "{:?}", cb.persp); + assert_eq!(cb.layer_fx[1], 1.0); + } + + /// Without an aspect, or with a degenerate box, the whole corrected picture is shown. + #[test] + fn the_cover_window_falls_back_to_the_whole_picture() { + let whole = [0.0, 0.0, 1.0, 1.0]; + assert_eq!(corrected_cover_window(None, 1.5), whole); + assert_eq!(corrected_cover_window(Some(0.0), 1.5), whole); + assert_eq!(corrected_cover_window(Some(1.5), f32::NAN), whole); + assert_eq!(corrected_cover_window(Some(1.5), f32::INFINITY), whole); + assert_eq!(corrected_cover_window(Some(1.5), 0.0), whole); + assert_eq!(corrected_cover_window(Some(1.5), 1.5), whole); + } + + /// A homography with a NaN or an infinity in it counts as none. + #[test] + fn a_non_finite_homography_is_ignored() { + let mut h = [1.0, 0.0, 0.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0]; + let ok = crate::scene::SceneCamera { homography: Some(h), ..extra_camera(1) }; + assert_eq!(camera_homography(Some(&ok)), Some(h)); + for bad in [f32::NAN, f32::INFINITY, f32::NEG_INFINITY] { + h[7] = bad; + let cam = crate::scene::SceneCamera { homography: Some(h), ..extra_camera(1) }; + assert_eq!(camera_homography(Some(&cam)), None); + let cb = with_camera_homography(LayerCB::default(), Some(&cam)); + assert_eq!(cb.layer_fx, [0.0; 4]); + assert_eq!(cb.persp, [[0.0; 4]; 3]); + } + } + + /// Camera 0's settings give back exactly today's source rect: layout crop, project mirror + /// and desk turn, through `camera_source_rect`. + #[test] + fn camera_0_settings_reproduce_todays_source_rect() { + let crop = SceneCrop { x: 0.1, y: 0.2, width: 0.6, height: 0.7 }; + let (visible, tex, box_ar) = ([1280.0, 720.0], [1280.0, 736.0], 1.3); + for (flip_u, flip_v, full_frame) in [ + (false, false, false), + (true, false, false), + (false, true, false), + (true, true, false), + (false, true, true), + ] { + let mut g = plan_frame(&golden_input(&golden_scene(), &crate::config::all()[0])); + g.webcam = WebcamOrientation { flip_u, flip_v, full_frame }; + let mut scene = golden_scene(); + scene.layout.webcam_crop = Some(crop); + let [cu0, cv0, cu1, cv1] = webcam_source_rect( + visible, + tex, + if full_frame { None } else { Some(crop) }, + box_ar, + ); + let (u0, u1) = if flip_u { (cu1, cu0) } else { (cu0, cu1) }; + let (v0, v1) = if flip_v { (cv1, cv0) } else { (cv0, cv1) }; + let cam0 = g.camera0_settings(Some(&scene)); + assert!(cam0.homography.is_none()); + assert_eq!(camera_source_rect(Some(&cam0), visible, tex, box_ar), [u0, v0, u1, v1]); + } + } + + /// Camera 0 with a perspective is cover-fitted into a box of another ratio, on both paths: + /// its default draw in a Full Camera section (no layout region, the box fills the frame) + /// and a planned frame-filling layer. Its PiP box, which has the corrected aspect, shows + /// the whole corrected picture. + #[test] + fn camera_0_with_a_perspective_is_cover_fitted_on_both_paths() { + let cfg = crate::config::all().pop().expect("cfg"); + let identity = "[1,0,0,0,1,0,0,0,1]"; + let aspect = 4.0_f32 / 3.0; + let cameras = format!(r#""cameras":[{{"index":0,"homography":{identity},"aspect":{aspect}}}]"#); + let zoom = r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#; + let near = |a: f32, b: f32| (a - b).abs() < 1e-4; + let render = [1170.0_f32, 658.0]; + let box_ar = render[0] / render[1]; + // A 4:3 picture in a wider box: the full width, a centred band of the box's ratio. + let dv = aspect / box_ar; + let check = |cb: &LayerCB| { + assert_eq!(cb.layer_fx[1], 1.0); + assert!(near(cb.persp[0][0], 1.0) && near(cb.persp[0][2], 0.0), "{:?}", cb.persp); + assert!(near(cb.persp[1][1], dv), "{:?}", cb.persp); + assert!(near(cb.persp[1][2], (1.0 - dv) * 0.5), "{:?}", cb.persp); + }; + + // No layout region, inside a Full Camera section: the default box fills the frame. + let json = zoomed_golden_scene_json().replace( + zoom, + &format!( + r#""zoomRegions":[],{cameras},"cameraFullscreenRegions":[{{"clipIndex":0,"startSec":0.0,"endSec":9.0}}]"# + ), + ); + let scene = Scene::from_json(&json).expect("scene"); + let g = plan_frame(&FrameGeometryInput { + timeline_t_override: Some(4.0), + ..golden_input(&scene, &cfg) + }); + assert!(g.camera_layers.is_empty()); + assert!(near(g.w_px[0], render[0]) && near(g.w_px[1], render[1]), "{:?}", g.w_px); + let cam0 = g.camera0_settings(Some(&scene)); + assert_eq!(cam0.aspect, Some(aspect)); + check(&with_camera_homography( + g.webcam_video_cb([0.0, 0.0, 1.0, 1.0], [1.0, 1.0], 0.0, 0.0), + Some(&cam0), + )); + + // A planned layer that fills the frame. + let json = zoomed_golden_scene_json().replace( + zoom, + &format!( + r#""zoomRegions":[],{cameras},"cameraLayoutRegions":[{{"clipIndex":0,"startSec":0.0,"endSec":4.0,"layers":[ + {{"camera":0,"rect":{{"x":0,"y":0,"width":1,"height":1}},"fillsFrame":true}}]}}]"# + ), + ); + let scene = Scene::from_json(&json).expect("scene"); + let g = plan_frame(&golden_input(&scene, &cfg)); + assert_eq!(g.camera_layers.len(), 1); + let cam0 = g.camera0_settings(Some(&scene)); + let base = g.webcam_video_cb([0.0; 4], [1.0, 1.0], 0.0, 0.0); + let plan = g.camera_layers[0]; + check(&camera_layer_cb(&plan, Some(&cam0), [1280.0, 720.0], [1280.0, 720.0], render, &base)); + + // The PiP box the app gives camera 0 has the corrected aspect: nothing is cropped. + let pip = LayerCB { quad_px: [400.0, 300.0], ..Default::default() }; + let cb = with_camera_homography(pip, Some(&cam0)); + assert!(near(cb.persp[0][0], 1.0) && near(cb.persp[1][1], 1.0), "{:?}", cb.persp); + } + + /// The shadow of a planned layer fades with it, and a frame-filling layer has none. + #[test] + fn a_planned_layers_shadow_fades_with_it() { + let cfg = crate::config::all().pop().expect("cfg"); + let g = plan_frame(&golden_input(&golden_scene(), &cfg)); + let render = [1920.0, 1080.0]; + let plan = planned(2, 0.4); + let base = extra_camera_base_cb([1.0; 2]); + let video = camera_layer_cb(&plan, None, [64.0; 2], [64.0; 2], render, &base); + let shadow = g.camera_layer_shadow(&plan, &video, render, true, false).expect("shadow"); + assert_eq!(shadow.mode, 2.0); + assert_eq!(shadow.layer_fx[0], video.layer_fx[0]); + assert_eq!(shadow.color[3], WEBCAM_SHADOW_OPACITY); + assert!(g.camera_layer_shadow(&plan, &video, render, false, false).is_none()); + let full = crate::camera_layers::CameraLayerPlan { fills_frame: true, ..plan }; + assert!(g.camera_layer_shadow(&full, &video, render, true, false).is_none()); + // Cutout removes camera 0's bubble, not the others'. + assert!(g.camera_layer_shadow(&plan, &video, render, true, true).is_some()); + let cam0 = crate::camera_layers::CameraLayerPlan { camera: 0, ..plan }; + assert!(g.camera_layer_shadow(&cam0, &video, render, true, true).is_none()); + } + + /// A turned Full Camera section from 1 s to 9 s: the plan carries the cover strength, full in + /// the hold after the start and zero in the steady part. + #[test] + fn the_frame_plan_carries_the_cover() { + let cfg = crate::config::all().pop().expect("cfg"); + let json = zoomed_golden_scene_json().replace( + r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#, + r#""zoomRegions":[],"cameraFullscreenRegions":[{"clipIndex":0,"startSec":1.0,"endSec":9.0,"rotation":180,"fullFrame":true}]"#, + ); + let scene = Scene::from_json(&json).expect("scene"); + let cover_at = |t: f32| { + plan_frame(&FrameGeometryInput { timeline_t_override: Some(t), ..golden_input(&scene, &cfg) }) + .webcam_cover + }; + assert_eq!(cover_at(1.5), 1.0); + assert_eq!(cover_at(5.0), 0.0); + } + + /// A scene without layout regions plans no camera layers, and camera 0 lands exactly where + /// it did: the new (empty) keys change nothing. + #[test] + fn without_layout_regions_the_plan_has_no_camera_layers() { + let cfg = crate::config::all().pop().expect("cfg"); + let before = plan_frame(&golden_input(&golden_scene(), &cfg)); + let json = zoomed_golden_scene_json().replace( + r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#, + r#""zoomRegions":[],"cameras":[],"cameraLayoutRegions":[]"#, + ); + let scene = Scene::from_json(&json).expect("scene"); + let after = plan_frame(&golden_input(&scene, &cfg)); + assert!(before.camera_layers.is_empty()); + assert!(after.camera_layers.is_empty()); + assert_eq!(after.w_dst, before.w_dst); + assert_eq!(after.w_dst_prev, before.w_dst_prev); + assert_eq!(after.w_px, before.w_px); + assert_eq!(after.w_radius, before.w_radius); + assert_eq!(after.shape_fade, before.shape_fade); + assert_eq!(after.webcam, before.webcam); + assert_eq!(after.webcam_cover, before.webcam_cover); + } + + /// Inside a layout region camera 0 is one of the planned layers, at the region's rect. + #[test] + fn with_a_layout_region_camera_0_comes_from_the_plan() { + let cfg = crate::config::all().pop().expect("cfg"); + let json = zoomed_golden_scene_json().replace( + r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#, + r#""zoomRegions":[],"cameraLayoutRegions":[{"clipIndex":0,"startSec":0.0,"endSec":4.0,"layers":[ + {"camera":0,"rect":{"x":0.1,"y":0.2,"width":0.3,"height":0.4},"radiusFrac":0.1,"shape":"rectangle"}]}]"#, + ); + let scene = Scene::from_json(&json).expect("scene"); + // `golden_input` samples t = 1.5 s: past the lead-in, well before the lead-out. + let g = plan_frame(&golden_input(&scene, &cfg)); + assert_eq!(g.camera_layers.len(), 1); + let cam0 = g.camera_layers[0]; + assert_eq!(cam0.camera, 0); + assert_eq!(cam0.dst, [0.1, 0.2, 0.3, 0.4]); + assert_eq!(cam0.opacity, 1.0); + assert_eq!(cam0.shape, webcam_shape_code("rectangle")); + assert!(!cam0.fills_frame); + // Outside the region camera 0 is the default PiP, where `w_dst` puts it. + let outside = plan_frame(&FrameGeometryInput { + timeline_t_override: Some(6.0), + ..golden_input(&scene, &cfg) + }); + assert_eq!(outside.camera_layers.len(), 1); + assert_eq!(outside.camera_layers[0].dst, outside.w_dst); + let r = outside.w_radius / outside.w_px[0].min(outside.w_px[1]).max(1.0); + assert_eq!(outside.camera_layers[0].radius_frac, r); + } + + /// The desk label's opacity at the compositor caller boundary, against the cover of the same + /// frame plan: equal at every sampled frame, and the label (spanned over its whole section by + /// the app) is on screen wherever the cover is not 0. Three sections: the review's 2.6 s one + /// at 1×, where the label's old fade (from its own 1.15 s window) was half gone while the + /// cover still held 1; a long one; and a section inside a 2× speed region, whose cover runs + /// on the screen clock. + #[test] + fn the_desk_label_fades_with_the_cover() { + use crate::text_anim::{annotation_text_state, DESK_COVER_ANIMATION}; + let cfg = crate::config::all().pop().expect("cfg"); + let zoom = r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#; + let section = |start: f32, end: f32, speed: Option<&str>| { + let regions = format!( + r#""zoomRegions":[],"cameraFullscreenRegions":[{{"clipIndex":0,"startSec":{start},"endSec":{end},"rotation":180,"fullFrame":true}}]{}"#, + speed.map(|s| format!(",{s}")).unwrap_or_default() + ); + Scene::from_json(&zoomed_golden_scene_json().replace(zoom, ®ions)).expect("scene") + }; + // (label, cover) at source time `t` of a label spanning [start, end]. + let sample = |scene: &Scene, start: f32, end: f32, t: f32| { + let g = plan_frame(&FrameGeometryInput { + timeline_t_override: Some(t), + ..golden_input(scene, &cfg) + }); + let label = annotation_text_state( + Some(DESK_COVER_ANIMATION), + (t - start) * 1000.0, + (end - start) * 1000.0, + g.webcam_cover, + ); + (label.opacity, g.webcam_cover) + }; + let speed = r#""speedRegions":[{"clipIndex":0,"startSec":1.0,"endSec":9.0,"speed":2.0}]"#; + let cases = [ + ("2.6 s at 1x", section(1.0, 3.6, None), 1.0f32, 3.6f32), + ("long at 1x", section(1.0, 9.0, None), 1.0, 9.0), + ("2.6 s on screen inside 2x", section(2.0, 7.2, Some(speed)), 2.0, 7.2), + ]; + for (name, scene, start, end) in &cases { + let mut t = start - 0.1; + while t < end + 0.1 { + let (label, cover) = sample(scene, *start, *end, t); + assert_eq!(label, cover, "{name}: t = {t}"); + if cover > 0.0 { + assert!(t >= *start && t < *end, "{name}: cover {cover} outside the label at {t}"); + } + t += 1.0 / 120.0; + } + } + // The review's frames: the cover still holds 1 at 0.9 s and 1.0 s (the old label was at + // ~0.5 and below). The 0.28 s left between the holds is two 0.14 s fades meeting at + // +1.1575 s, where label and cover touch 0 together before rising into the end hold. + let (_, short, s0, s1) = &cases[0]; + for at in [0.5, 0.9, 1.0] { + assert_eq!(sample(short, *s0, *s1, s0 + at), (1.0, 1.0), "at +{at} s"); + } + let (mid, _) = sample(short, *s0, *s1, s0 + 1.09); + assert!(mid > 0.0 && mid < 1.0, "mid-fade {mid}"); + let (low, low_cover) = sample(short, *s0, *s1, s0 + 1.1575); + assert!(low < 1e-3 && low == low_cover, "junction {low} vs {low_cover}"); + assert_eq!(sample(short, *s0, *s1, s1 - 1.25), (1.0, 1.0), "end hold"); + // Inside 2x the same 2.6 s of screen time takes 5.2 s of source: every screen instant + // matches the 1x section's, which the source-time label could not do. + let (_, fast, f0, f1) = &cases[2]; + for k in 0..=26 { + let screen = k as f32 * 0.1; + let slow = sample(short, *s0, *s1, s0 + screen); + let quick = sample(fast, *f0, *f1, f0 + 2.0 * screen); + assert!((slow.0 - quick.0).abs() < 1e-3, "screen +{screen} s: {slow:?} vs {quick:?}"); + } + } + + /// A turned Full Camera section that meets a layout region: the cover follows the seam rule + /// (no hold before the layout region, only its own fade), and the label, spanned over the + /// whole section by the app, follows the cover there frame for frame without any seam + /// logic of its own. + #[test] + fn the_desk_label_follows_the_cover_across_a_seam_with_a_layout_region() { + use crate::regions::DESK_COVER_FADE_S; + use crate::text_anim::{annotation_text_state, DESK_COVER_ANIMATION}; + let cfg = crate::config::all().pop().expect("cfg"); + let zoom = r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#; + let full = r#""zoomRegions":[],"cameraFullscreenRegions":[{"clipIndex":0,"startSec":1.0,"endSec":6.0,"rotation":180,"fullFrame":true}]"#; + let layout = r#","cameraLayoutRegions":[{"clipIndex":0,"startSec":6.0,"endSec":9.0,"layers":[ + {"camera":0,"rect":{"x":0.1,"y":0.2,"width":0.3,"height":0.4},"radiusFrac":0.1,"shape":"rectangle"}]}]"#; + let json = zoomed_golden_scene_json(); + let seam = + Scene::from_json(&json.replace(zoom, &format!("{full}{layout}"))).expect("scene"); + let lone = Scene::from_json(&json.replace(zoom, full)).expect("scene"); + let (start, end) = (1.0f32, 6.0f32); + let sample = |scene: &Scene, t: f32| { + let g = plan_frame(&FrameGeometryInput { + timeline_t_override: Some(t), + ..golden_input(scene, &cfg) + }); + let label = annotation_text_state( + Some(DESK_COVER_ANIMATION), + (t - start) * 1000.0, + (end - start) * 1000.0, + g.webcam_cover, + ); + (label.opacity, g.webcam_cover) + }; + let mut t = start - 0.1; + while t < 9.1 { + let (label, cover) = sample(&seam, t); + assert_eq!(label, cover, "t = {t}"); + if cover > 0.0 { + assert!(t >= start && t < end, "cover {cover} outside the label at {t}"); + } + t += 1.0 / 120.0; + } + // Before the seam the cover (and so the label) is only its fade: half way up at + // end - fade/2 and sharp just before the fade, where a lone section still holds 1. + let (mid, _) = sample(&seam, end - DESK_COVER_FADE_S / 2.0); + assert!((mid - 0.5).abs() < 1e-3, "half way up at the seam: {mid}"); + let before_fade = end - DESK_COVER_FADE_S - 0.05; + assert_eq!(sample(&seam, before_fade), (0.0, 0.0), "sharp before the seam's fade"); + assert_eq!(sample(&lone, before_fade), (1.0, 1.0), "a lone section still holds"); + assert!(sample(&seam, end - 0.001).0 > 0.99, "covered at the seam"); + } + /// Le pivot doit rester collé à `center` quand le sprite grandit — c'est exactement ce qui /// était cassé (ancrage centré en dur : la pointe s'éloignait proportionnellement à la /// taille). On dessine la même flèche à deux tailles et on vérifie que le point désigné @@ -7730,6 +8573,9 @@ mod tests { w_px: [0.0, 0.0], w_radius: 0.0, shape_fade: 0.0, + webcam: WebcamOrientation::default(), + webcam_cover: 0.0, + camera_layers: Vec::new(), window_frame: None, screen_mask: None, } @@ -8794,4 +9640,117 @@ mod tests { println!("left ×2 : {:.3} côté proche, {:.3} côté lointain", near / at_focus, far / at_focus); assert!(near > 1.04 * at_focus && far < 0.96 * at_focus, "{near} {at_focus} {far}"); } + + use crate::scene::SceneCameraFullscreenRegion; + + fn cam_region(rotation: u16, mirror: Option) -> SceneCameraFullscreenRegion { + SceneCameraFullscreenRegion { + clip_index: None, + start_sec: 0.0, + end_sec: 1.0, + rotation, + mirror, + full_frame: rotation == 180, + } + } + + #[test] + fn without_a_region_the_webcam_is_laid_as_before() { + for m in [false, true] { + assert_eq!( + webcam_orientation(None, m), + WebcamOrientation { flip_u: m, flip_v: false, full_frame: false } + ); + } + } + + /// 180° is both axes swapped; a mirror on top cancels the horizontal one. + #[test] + fn a_turned_region_swaps_both_axes_and_a_mirror_cancels_one() { + let turned = cam_region(180, Some(false)); + assert_eq!( + webcam_orientation(Some(&turned), true), + WebcamOrientation { flip_u: true, flip_v: true, full_frame: true } + ); + let turned_mirrored = cam_region(180, Some(true)); + assert_eq!( + webcam_orientation(Some(&turned_mirrored), false), + WebcamOrientation { flip_u: false, flip_v: true, full_frame: true } + ); + } + + #[test] + fn a_plain_region_keeps_the_layout_mirror_and_an_unknown_rotation_is_none() { + assert!(webcam_orientation(Some(&cam_region(0, None)), true).flip_u); + let odd = cam_region(90, None); + assert_eq!( + webcam_orientation(Some(&odd), false), + WebcamOrientation { flip_u: false, flip_v: false, full_frame: false } + ); + } + + /// `layer.wgsl` is only compiled on Linux, in CI, where a name or syntax error breaks every + /// draw. Parse and validate it on every host through wgpu's own naga, and pin the desk-view + /// cover: the fragment entry point must call the radius-taking blur kernel, and the kernel + /// must keep its taps inside the picture's valid area. + /// + /// The file does not declare `LAYER_MODELS`: `compositor_linux::layer_source` prefixes it, + /// once per pipeline. Both variants are checked here the same way, the one without the 3D + /// models first, since that is the pipeline the webcam (mode 0) is drawn with. + #[test] + fn layer_wgsl_validates_and_its_fragment_applies_the_desk_view_cover() { + use wgpu::naga; + fn calls(block: &naga::Block, target: naga::Handle) -> usize { + block + .iter() + .map(|st| match st { + naga::Statement::Call { function, .. } => usize::from(*function == target), + naga::Statement::Block(b) => calls(b, target), + naga::Statement::If { accept, reject, .. } => calls(accept, target) + calls(reject, target), + naga::Statement::Loop { body, continuing, .. } => calls(body, target) + calls(continuing, target), + naga::Statement::Switch { cases, .. } => cases.iter().map(|c| calls(&c.body, target)).sum(), + _ => 0, + }) + .sum() + } + + for models in [false, true] { + // The same prefix as `compositor_linux::layer_source`, which only builds on Linux. + let source = + format!("const LAYER_MODELS: bool = {models};\n{}", include_str!("vk_shaders/layer.wgsl")); + let module = naga::front::wgsl::parse_str(&source) + .unwrap_or_else(|e| panic!("layer.wgsl (LAYER_MODELS = {models}) parses: {e:?}")); + naga::valid::Validator::new(naga::valid::ValidationFlags::all(), naga::valid::Capabilities::all()) + .validate(&module) + .unwrap_or_else(|e| panic!("layer.wgsl (LAYER_MODELS = {models}) validates: {e:?}")); + let (kernel, f) = module + .functions + .iter() + .find(|(_, f)| f.name.as_deref() == Some("blur_webcam_radius")) + .expect("blur_webcam_radius exists"); + assert_eq!(f.arguments.len(), 5, "uv, max_r_px, qpx, local_px, valid"); + // The taps are clamped to the valid part of an aligned decoder texture, half a texel + // in: the kernel reads the texture size to know what half a texel is. + assert!( + f.expressions.iter().any(|(_, e)| matches!( + e, + naga::Expression::ImageQuery { query: naga::ImageQuery::Size { .. }, .. } + )), + "blur_webcam_radius reads the texture size for its half-texel clamp" + ); + // `fs_main` only applies the layer's transparency to `fs_layer`, which draws it. + let fs = module.entry_points.iter().find(|e| e.name == "fs_main").expect("fs_main"); + let (body, f) = module + .functions + .iter() + .find(|(_, f)| f.name.as_deref() == Some("fs_layer")) + .expect("fs_layer exists"); + assert_eq!(calls(&fs.function.body, body), 1, "fs_main draws through fs_layer"); + assert_eq!( + calls(&f.body, kernel), + 1, + "fs_layer blurs the covered camera once (LAYER_MODELS = {models})" + ); + } + } } diff --git a/crates/compositor/src/gif_export.rs b/crates/compositor/src/gif_export.rs index 8215704c2..7ab0c1b6c 100644 --- a/crates/compositor/src/gif_export.rs +++ b/crates/compositor/src/gif_export.rs @@ -256,6 +256,7 @@ fn export_gif_inner( let mut screen_decs: HashMap = HashMap::new(); let mut webcam_decs: HashMap = HashMap::new(); + let mut extra_decs: HashMap = HashMap::new(); screen_decs.insert(clips[0].screen.clone(), unsafe { Decoder::open_for_export(&clips[0].screen, gpu)? }); @@ -270,6 +271,7 @@ fn export_gif_inner( &scene, &mut screen_decs, &mut webcam_decs, + &mut extra_decs, &mut |frame_index| { control.check()?; // CPU readback of the staged RT (RGBA8 tightly-packed, diff --git a/crates/compositor/src/lib.rs b/crates/compositor/src/lib.rs index 767abac8a..67c233e4b 100644 --- a/crates/compositor/src/lib.rs +++ b/crates/compositor/src/lib.rs @@ -30,6 +30,7 @@ pub mod audio; pub mod audio_jobs; pub mod camera; +pub mod camera_layers; pub mod config; pub mod cursor; pub mod cursor_sdf; @@ -56,6 +57,7 @@ pub mod shared_frames; pub mod text_anim; pub mod text_fonts; pub mod text_plate; +pub(crate) mod extra_cameras; pub(crate) mod timeline_walk; // GPU backend : Windows → d3d_windows, macOS → d3d_macos. Ré-exporté sous le nom `d3d` diff --git a/crates/compositor/src/live.rs b/crates/compositor/src/live.rs index 8f56f22a2..d8e2c4e87 100644 --- a/crates/compositor/src/live.rs +++ b/crates/compositor/src/live.rs @@ -23,8 +23,9 @@ //! le napi — le `gen` est l'identité de la frame (cf. `LatestFrame`). use crate::compositor::{Compositor, LiveParams}; +use crate::ffi::AVFrame; use crate::regions::{speed_at, ProgrammeClock}; -use crate::scene::Scene; +use crate::scene::{Scene, SceneCameraLayoutRegion, SceneClipCamera}; use crate::config::{self, Cfg}; use crate::cursor::CursorTrack; use crate::d3d::{Backend, Gpu}; @@ -33,6 +34,10 @@ use crate::pipeline::Decoder; #[cfg(windows)] use crate::shared_frames::SharedRing; use crate::shared_frames::{SharedFrame, SlotBook}; +use crate::extra_cameras::{ + camera_source_time, cameras_in_regions, extra_camera_active, extra_camera_keys, extra_frame_list, + has_camera_file, opened_or_skipped, seek_extra_camera, step_extra_camera, ExtraCameraKeys, +}; use crate::timeline_walk::{frame_step, FrameStep, NextFrameTime}; use anyhow::Result; use std::sync::atomic::{AtomicBool, AtomicU64, Ordering}; @@ -52,10 +57,6 @@ fn parse_hex_color(s: &str) -> Option<[f32; 4]> { Some([r, g, b, 1.0]) } -fn webcam_seek_time(screen_source_time_sec: f64, webcam_offset_sec: f64) -> f64 { - (screen_source_time_sec - webcam_offset_sec).max(0.0) -} - /// Décodeurs déjà ouverts ET positionnés au bon playhead pour un clip à venir — le résultat /// d'un préchargement en tâche de fond (voir `open_and_seek_clip`/`maybe_start_prefetch` /// dans `render_thread`). Appliquer ceci à un `Player` (`apply_prefetched`) ne fait plus @@ -72,6 +73,9 @@ struct PrefetchedClip { /// des paires ouvertes plusieurs bascules plus tôt. Cf. `open_webcam_or_stand_in`. webcam_decoder_is_real: bool, webcam_offset_sec: f64, + /// The clip's extra cameras (index k-1 = camera k) and what they were opened for. + extra: Vec>, + extra_keys: ExtraCameraKeys, idx: u32, /// Piste curseur du clip à venir, préchargée ici pour la même raison que les décodeurs : /// sans ça, la bascule à la frontière restait synchrone sur CE point précis (lecture + @@ -149,18 +153,122 @@ fn should_draw_webcam(webcam_path: &str, screen_path: &str, decoder_is_real: boo webcam_is_real(webcam_path, screen_path) && decoder_is_real } +/// An extra camera (k >= 1) of the active clip, decoded only near the layout regions that show +/// it (`extra_camera_active`). +struct ExtraCamera { + dec: Decoder, + offset_sec: f64, + /// The last step or seek left a frame to draw. Read by `Player::recompose`. + shown: bool, + /// Camera source time past which the file has no frame left (`step_extra_camera`). + ended_at: Option, +} + +impl ExtraCamera { + /// `step_extra_camera` toward `target` (camera source time). + unsafe fn step_to(&mut self, target: f64) -> Result<*const AVFrame> { + let frame = step_extra_camera(&mut self.dec, &mut self.ended_at, target)?; + self.shown = !frame.is_null(); + Ok(frame) + } + + /// `seek_extra_camera` to `target` (camera source time). + unsafe fn seek(&mut self, target: f64) -> Result<*const AVFrame> { + let frame = seek_extra_camera(&mut self.dec, &mut self.ended_at, target)?; + self.shown = !frame.is_null(); + Ok(frame) + } +} + +/// The extra cameras clip `clip_index` opens: those its own layout regions +/// (`Scene::for_clip_window`) draw and that the clip has a file for. +fn extra_cameras_for_clip(scene: &Scene, clip_index: usize) -> Vec { + let Some(clip) = scene.clips.get(clip_index) else { + return Vec::new(); + }; + // The common case — no layout region anywhere — skips the windowing copy of the scene. + if scene.camera_layout_regions.is_empty() || clip.additional_cameras.is_empty() { + return Vec::new(); + } + cameras_in_regions(&scene_for_clip(scene, clip_index).camera_layout_regions) + .into_iter() + .filter(|&k| has_camera_file(clip.additional_cameras.get(k - 1))) + .collect() +} + +/// `extra_camera_keys` for clip `clip_index` of `scene`, from the clip's own camera files. +fn scene_extra_camera_keys(scene: &Scene, clip_index: usize) -> ExtraCameraKeys { + let sources = scene.clips.get(clip_index).map(|c| c.additional_cameras.as_slice()).unwrap_or(&[]); + extra_camera_keys(&extra_cameras_for_clip(scene, clip_index), sources) +} + +/// The windowed layout regions of clip `clip_index` (what `camera_layers_at` plans from). +fn clip_layout_regions(scene: &Scene, clip_index: usize) -> Vec { + if scene.camera_layout_regions.is_empty() { + return Vec::new(); + } + scene_for_clip(scene, clip_index).camera_layout_regions +} + +fn same_extra_key(a: &Option, b: &Option) -> bool { + match (a, b) { + (None, None) => true, + (Some(a), Some(b)) => a.path == b.path && (a.offset_sec - b.offset_sec).abs() < 1e-9, + _ => false, + } +} + +fn same_extra_keys(a: &[Option], b: &[Option]) -> bool { + a.len() == b.len() && a.iter().zip(b).all(|(a, b)| same_extra_key(a, b)) +} + +/// Same spans and same cameras in each: what decides when an extra camera is active +/// (`extra_camera_active`). Rects and shapes do not matter to the decoders. +fn same_region_timing(a: &[SceneCameraLayoutRegion], b: &[SceneCameraLayoutRegion]) -> bool { + a.len() == b.len() + && a.iter().zip(b).all(|(a, b)| { + a.clip_index == b.clip_index + && a.start_sec == b.start_sec + && a.end_sec == b.end_sec + && a.layers.len() == b.layers.len() + && a.layers.iter().zip(&b.layers).all(|(x, y)| x.camera == y.camera) + }) +} + +/// Whether `set_extra_cameras` seeks the extra cameras right away. Playing on (`steady`: +/// playing, and the screen position did not jump) with the same files and the same region +/// timing, the next playback step (`advance_extras(.., false)`) moves them on: seeking every +/// camera on each scene push would stall the preview while it plays. +fn extras_need_seek(keys_changed: bool, timing_changed: bool, steady: bool) -> bool { + keys_changed || timing_changed || !steady +} + +/// Opens the extra cameras `keys` asks for, without seeking them: they are positioned when +/// they become active (`Player::advance_extras`). +unsafe fn open_extra_cameras(keys: &[Option], gpu: &Gpu) -> Vec> { + keys.iter() + .map(|key| { + let key = key.as_ref()?; + let dec = opened_or_skipped(&key.path, Decoder::open(&key.path, gpu))?; + Some(ExtraCamera { dec, offset_sec: key.offset_sec, shown: false, ended_at: None }) + }) + .collect() +} + unsafe fn open_and_seek_clip( screen_path: &str, webcam_path: &str, webcam_offset_sec: f64, + extra_keys: &[Option], source_time_sec: f64, gpu: &Gpu, ) -> Result { let source_time_sec = source_time_sec.max(0.0); let mut sdec = Decoder::open(screen_path, gpu)?; let (mut wdec, webcam_decoder_is_real) = open_webcam_or_stand_in(screen_path, webcam_path, gpu)?; + let extra = open_extra_cameras(extra_keys, gpu); let sf = sdec.seek_to_or_last(source_time_sec)?; - if webcam_decoder_is_real && wdec.seek_to(webcam_seek_time(source_time_sec, webcam_offset_sec))?.is_null() { + if webcam_decoder_is_real && wdec.seek_to(camera_source_time(source_time_sec, webcam_offset_sec))?.is_null() { wdec.seek_to(0.0)?; } if sf.is_null() { @@ -168,7 +276,16 @@ unsafe fn open_and_seek_clip( } let idx = (source_time_sec * sdec.fps()).round().max(0.0) as u32; let cursor_track = CursorTrack::load(&format!("{screen_path}.cursor.json"), 0.0, 24.0 * 3600.0).ok(); - Ok(PrefetchedClip { sdec, wdec, webcam_decoder_is_real, webcam_offset_sec, idx, cursor_track }) + Ok(PrefetchedClip { + sdec, + wdec, + webcam_decoder_is_real, + webcam_offset_sec, + extra, + extra_keys: extra_keys.to_vec(), + idx, + cursor_track, + }) } /// Nombre de paires de décodeurs INACTIVES gardées ouvertes en plus de la paire active. @@ -180,11 +297,29 @@ unsafe fn open_and_seek_clip( /// les timelines 2-4 clips, à baisser si la VRAM serre. const DECODER_POOL_CAP: usize = 3; -/// Une paire de décodeurs mise de côté, prête à être réactivée sans réouverture. -struct PooledClip { +/// What a set of open decoders was opened for: the screen, camera 0 and the extra cameras. +/// A pooled set is reused only for the same key, extra cameras included — a pooled pair +/// opened without camera 2 must not stand in for a clip whose layout shows it. +#[derive(Clone)] +struct ClipKey { screen_path: String, webcam_path: String, webcam_offset_sec: f64, + extras: ExtraCameraKeys, +} + +impl ClipKey { + fn matches(&self, other: &ClipKey) -> bool { + self.screen_path == other.screen_path + && self.webcam_path == other.webcam_path + && (self.webcam_offset_sec - other.webcam_offset_sec).abs() < 1e-9 + && same_extra_keys(&self.extras, &other.extras) + } +} + +/// Une paire de décodeurs mise de côté, prête à être réactivée sans réouverture. +struct PooledClip { + key: ClipKey, clip: PrefetchedClip, } @@ -207,7 +342,7 @@ unsafe fn seek_pair( if !webcam_decoder_is_real { return Ok(true); } - let mut wf = wdec.seek_to(webcam_seek_time(source_time_sec, webcam_offset_sec))?; + let mut wf = wdec.seek_to(camera_source_time(source_time_sec, webcam_offset_sec))?; if wf.is_null() { wf = wdec.seek_to(0.0)?; } @@ -226,21 +361,15 @@ unsafe fn seek_pair( unsafe fn swap_clip_pooled( player: &mut Player, pool: &mut Vec, - request: &ActiveClipRequest, - active_screen: &str, - active_webcam: &str, - active_webcam_offset_sec: f64, + request: &ClipKey, + source_time_sec: f64, + active: &ClipKey, ) -> Result<()> { let timing = std::env::var("OPENSCREEN_CLIPSWITCH_TIMING").is_ok(); let t0 = std::time::Instant::now(); - let t = request.source_time_sec.max(0.0); - let matches = |p: &PooledClip| { - p.screen_path == request.screen_path - && p.webcam_path == request.webcam_path - && (p.webcam_offset_sec - request.webcam_offset_sec).abs() < 1e-9 - }; + let t = source_time_sec.max(0.0); let mut hit = false; - let incoming: PrefetchedClip = match pool.iter().position(&matches) { + let incoming: PrefetchedClip = match pool.iter().position(|p| p.key.matches(request)) { Some(i) => { let mut pooled = pool.remove(i); // Reseek les décodeurs poolés AVANT de les installer. Échec → on les jette et on @@ -257,25 +386,16 @@ unsafe fn swap_clip_pooled( pooled.clip } else { drop(pooled); - player.open_clip(&request.screen_path, &request.webcam_path, request.webcam_offset_sec, t)? + player.open_clip(request, t)? } } - None => player.open_clip(&request.screen_path, &request.webcam_path, request.webcam_offset_sec, t)?, + None => player.open_clip(request, t)?, }; let outgoing = player.swap_active(incoming); // Met la paire quittée en pool : dédup par clé (jamais deux entrées d'un même média), puis // éviction LRU (le plus ancien, en tête, part en premier). - pool.retain(|p| { - !(p.screen_path == active_screen - && p.webcam_path == active_webcam - && (p.webcam_offset_sec - active_webcam_offset_sec).abs() < 1e-9) - }); - pool.push(PooledClip { - screen_path: active_screen.to_string(), - webcam_path: active_webcam.to_string(), - webcam_offset_sec: active_webcam_offset_sec, - clip: outgoing, - }); + pool.retain(|p| !p.key.matches(active)); + pool.push(PooledClip { key: active.clone(), clip: outgoing }); while pool.len() > DECODER_POOL_CAP { pool.remove(0); } @@ -302,6 +422,12 @@ pub struct Player { /// (`swap_active`), lu par la boucle de rendu pour décider de dessiner la vignette. webcam_decoder_is_real: bool, webcam_offset_sec: f64, + /// The active clip's extra cameras (index k-1 = camera k), what they were opened for, and + /// the clip's windowed layout regions that decide when each one is decoded. All empty + /// without layout regions: no decoder, nothing to advance. + extra: Vec>, + extra_keys: ExtraCameraKeys, + camera_regions: Vec, has_current_frame: bool, use_current_on_next_step: bool, idx: u32, @@ -324,6 +450,9 @@ impl Player { }, webcam_decoder_is_real, webcam_offset_sec: 0.0, + extra: Vec::new(), + extra_keys: Vec::new(), + camera_regions: Vec::new(), has_current_frame: false, use_current_on_next_step: false, idx: 0, @@ -342,10 +471,17 @@ impl Player { screen_path: &str, webcam_path: &str, webcam_offset_sec: f64, + extra_keys: &[Option], source_time_sec: f64, ) -> Result<()> { - let prefetched = - open_and_seek_clip(screen_path, webcam_path, webcam_offset_sec, source_time_sec, &self.gpu)?; + let prefetched = open_and_seek_clip( + screen_path, + webcam_path, + webcam_offset_sec, + extra_keys, + source_time_sec, + &self.gpu, + )?; self.apply_prefetched(prefetched); Ok(()) } @@ -412,6 +548,8 @@ impl Player { // recalculer), l'entrante impose la sienne au player. webcam_decoder_is_real: self.webcam_decoder_is_real, webcam_offset_sec: self.webcam_offset_sec, + extra: std::mem::replace(&mut self.extra, incoming.extra), + extra_keys: std::mem::replace(&mut self.extra_keys, incoming.extra_keys), idx: self.idx, // Le curseur est re-dérivé du chemin à la réactivation ; inutile de le trimballer. cursor_track: None, @@ -426,14 +564,107 @@ impl Player { /// Ouvre une nouvelle paire de décodeurs positionnée à `source_time_sec`, SANS l'installer /// (l'appelant l'échange via `swap_active`). Réutilise le device D3D11 du player. - unsafe fn open_clip( - &self, - screen: &str, - webcam: &str, - webcam_offset_sec: f64, - source_time_sec: f64, - ) -> Result { - open_and_seek_clip(screen, webcam, webcam_offset_sec, source_time_sec, &self.gpu) + unsafe fn open_clip(&self, key: &ClipKey, source_time_sec: f64) -> Result { + open_and_seek_clip( + &key.screen_path, + &key.webcam_path, + key.webcam_offset_sec, + &key.extras, + source_time_sec, + &self.gpu, + ) + } + + /// What the active extra cameras were opened for (the extra half of the pool key). + fn extra_keys(&self) -> &[Option] { + &self.extra_keys + } + + /// Installs the active clip's windowed layout regions and the extra cameras they need + /// (`keys`): slots whose key is unchanged keep their decoder, the others are closed and + /// (re)opened — a file that will not open leaves its slot empty, once. Then positions every + /// camera that is near one of its regions at the current screen time, so a paused + /// recompose shows it at once — unless nothing that matters to the decoders changed while + /// playback goes on (`steady`, see `extras_need_seek`). `true` when a decoder was closed: + /// the caller must then clear the compositor's SRV cache (which also forgets the extra + /// frame pointers). + pub(crate) unsafe fn set_extra_cameras( + &mut self, + regions: Vec, + keys: ExtraCameraKeys, + steady: bool, + ) -> bool { + let timing_changed = !same_region_timing(&self.camera_regions, ®ions); + let keys_changed = !same_extra_keys(&self.extra_keys, &keys); + self.camera_regions = regions; + let mut closed = false; + if keys_changed { + let mut old = std::mem::take(&mut self.extra); + let mut extra = Vec::with_capacity(keys.len()); + for (k, key) in keys.iter().enumerate() { + let old_cam = old.get_mut(k).and_then(Option::take); + let old_key = self.extra_keys.get(k).cloned().flatten(); + if same_extra_key(&old_key, key) { + extra.push(old_cam); + continue; + } + closed |= old_cam.is_some(); + drop(old_cam); + extra.extend(open_extra_cameras(std::slice::from_ref(key), &self.gpu)); + } + closed |= old.iter().any(Option::is_some); + self.extra = extra; + self.extra_keys = keys; + } + if !extras_need_seek(keys_changed, timing_changed, steady) { + return closed; + } + let t = self.sdec.cur_time_sec(); + closed | self.advance_extras(t, true).1 + } + + /// The extra cameras' frames at screen source time `screen_t`, each camera stepped + /// (`seek == false`) or sought toward its own source time only while it is near one of + /// its regions; an idle camera is left alone and gives null. A camera whose decoder fails + /// is closed with one warning — its layer disappears, the preview goes on. The `bool` says + /// a decoder was closed (see `set_extra_cameras`). + unsafe fn advance_extras(&mut self, screen_t: f64, seek: bool) -> (Vec<*const AVFrame>, bool) { + let mut closed = false; + let mut frames = Vec::with_capacity(self.extra.len()); + for (k, slot) in self.extra.iter_mut().enumerate() { + let camera = k + 1; + let frame = match slot { + Some(cam) if extra_camera_active(&self.camera_regions, camera, screen_t) => { + let target = camera_source_time(screen_t, cam.offset_sec); + let result = if seek { cam.seek(target) } else { cam.step_to(target) }; + match result { + Ok(frame) => frame, + Err(e) => { + eprintln!("WARNING: extra camera {camera} stopped decoding: {e:#}. Its layer will not be drawn."); + *slot = None; + closed = true; + std::ptr::null() + } + } + } + Some(cam) => { + cam.shown = false; + std::ptr::null() + } + None => std::ptr::null(), + }; + frames.push(frame); + } + (frames, closed) + } + + /// Hands the extra cameras' frames to `comp` for the next `compose_frame`. Called before + /// every compose: the compositor keeps the pointers until the next call. + unsafe fn hand_extra_frames(comp: &Compositor, frames: &[*const AVFrame], closed: bool) { + if closed { + comp.clear_srv_cache(); + } + comp.set_extra_camera_frames(frames); } /// Le décodeur webcam ACTIF est-il la vraie caméra ? `false` quand c'est le remplaçant @@ -600,6 +831,8 @@ impl Player { self.has_current_frame = true; self.sync_time(comp); + let (extra, closed) = self.advance_extras(self.sdec.cur_time_sec(), false); + Self::hand_extra_frames(comp, &extra, closed); comp.compose_frame(sf, wf, self.idx as f32, cfg)?; self.idx = self.idx.wrapping_add(1); Ok(true) @@ -630,6 +863,10 @@ impl Player { return Ok(false); } self.sync_time(comp); + let extra = extra_frame_list(&self.extra, |cam| { + if cam.shown { cam.dec.cur_frame() as *const AVFrame } else { std::ptr::null() } + }); + comp.set_extra_camera_frames(&extra); let f = self.idx.saturating_sub(1); comp.compose_frame(sf, wf, f as f32, cfg)?; Ok(true) @@ -646,7 +883,7 @@ impl Player { // ne serait pas composée. Sans caméra, la frame écran tient sa place. let wf = if self.webcam_decoder_is_real { self.wdec - .seek_to_or_last(webcam_seek_time(target_sec, self.webcam_offset_sec))? + .seek_to_or_last(camera_source_time(target_sec, self.webcam_offset_sec))? } else { sf }; @@ -660,6 +897,8 @@ impl Player { // "idx" ne sert plus qu'au fallback fixture (jamais lu si une scène est posée) — dérivé // du temps réel pour rester cohérent si jamais consulté. self.idx = (target_sec * self.sdec.fps()).round().max(0.0) as u32; + let (extra, closed) = self.advance_extras(target_sec, true); + Self::hand_extra_frames(comp, &extra, closed); comp.compose_frame(sf, wf, self.idx as f32, cfg)?; Ok(true) } @@ -738,6 +977,9 @@ struct ActiveClipRequest { screen_path: String, webcam_path: String, webcam_offset_sec: f64, + /// Cameras 2-4 of the clip (index k-1 = camera k); which of them open is decided by the + /// clip's layout regions (`extra_cameras_for_clip`). + additional_cameras: Vec, /// Identité dans le flux `Scene.clips` trié (les chemins ne suffisent pas pour un asset partagé). clip_index: usize, /// Playhead exprimé sur l'horloge source écran du nouveau clip. @@ -1134,6 +1376,7 @@ impl LiveView { screen_path: &str, webcam_path: &str, webcam_offset_sec: f64, + additional_cameras: Vec, clip_index: usize, source_time_sec: f64, ) { @@ -1142,6 +1385,7 @@ impl LiveView { screen_path: screen_path.to_string(), webcam_path: webcam_path.to_string(), webcam_offset_sec, + additional_cameras, clip_index, source_time_sec: source_time_sec.max(0.0), }); @@ -1240,7 +1484,7 @@ type PendingPrefetch = (usize, std::sync::mpsc::Receiver> /// tâche de fond. Assez large pour couvrir un `Decoder::open` typique (ouverture fichier + /// `avformat_find_stream_info` + init D3D11VA), assez court pour ne pas garder deux paires de /// décodeurs ouvertes plus longtemps que nécessaire. -const PREFETCH_LEAD_SEC: f64 = 0.75; +pub(crate) const PREFETCH_LEAD_SEC: f64 = 0.75; /// Durée pendant laquelle la boucle continue de recomposer après un changement en pause, le /// temps qu'un effet asynchrone (segmentation webcam) livre son résultat. Généreuse : à @@ -1294,6 +1538,7 @@ unsafe fn maybe_start_prefetch( 0 }; let next_clip = scene.clips[next_index].clone(); + let extra_keys = scene_extra_camera_keys(scene, next_index); // Copie légère (COM refcount, pas de nouveau device) — même motif que `Player::open`. let gpu_clone = Gpu { device: gpu.device.clone(), @@ -1308,6 +1553,7 @@ unsafe fn maybe_start_prefetch( &next_clip.screen_path, &next_clip.webcam_path, next_clip.webcam_offset_sec, + &extra_keys, next_clip.source_start_sec, &gpu_clone, ) @@ -1345,6 +1591,7 @@ unsafe fn advance_to_next_scene_clip( active_screen_path: &mut String, active_webcam_path: &mut String, active_webcam_offset_sec: &mut f64, + active_additional_cameras: &mut Vec, active_clip_index: &mut usize, raw_cursor: &mut Option, loaded_cursor_path: &mut String, @@ -1359,6 +1606,7 @@ unsafe fn advance_to_next_scene_clip( 0 }; let next_clip = &scene.clips[next_index]; + let extra_keys = scene_extra_camera_keys(scene, next_index); // N'importe quel préchargement en cours ne concerne plus que CETTE frontière (on vient // de la franchir, bien ou mal ciblée) — on le consomme s'il correspond, on l'abandonne @@ -1386,6 +1634,7 @@ unsafe fn advance_to_next_scene_clip( &next_clip.screen_path, &next_clip.webcam_path, next_clip.webcam_offset_sec, + &extra_keys, next_clip.source_start_sec, ) } @@ -1393,6 +1642,7 @@ unsafe fn advance_to_next_scene_clip( &next_clip.screen_path, &next_clip.webcam_path, next_clip.webcam_offset_sec, + &extra_keys, next_clip.source_start_sec, ), }; @@ -1404,12 +1654,18 @@ unsafe fn advance_to_next_scene_clip( // sur l'adresse de la texture, et garder des entrées d'un décodeur fermé fait // fuir de la VRAM puis, en cas de réutilisation d'adresse, rendre l'image du clip // précédent. + let windowed = scene_for_clip(scene, next_index); + // The prefetch opened exactly these keys, so this only installs the regions and + // positions the cameras; the cache is cleared right below in any case. + // The screen position jumped to the next clip: always position the cameras. + player.set_extra_cameras(windowed.camera_layout_regions.clone(), extra_keys, false); comp.clear_srv_cache(); *active_screen_path = next_clip.screen_path.clone(); *active_webcam_path = next_clip.webcam_path.clone(); *active_webcam_offset_sec = next_clip.webcam_offset_sec; + *active_additional_cameras = next_clip.additional_cameras.clone(); *active_clip_index = next_index; - comp.set_scene(Some(scene_for_clip(scene, *active_clip_index))); + comp.set_scene(Some(windowed)); player.set_programme_clock(Some(scene), *active_clip_index); // Réutilise le curseur préchargé s'il est disponible (voir plus haut) — sinon // (préchargement pas encore prêt / raté) on retombe sur la lecture synchrone @@ -1487,6 +1743,9 @@ unsafe fn render_thread( let mut active_screen_path = screen.to_string(); let mut active_webcam_path = webcam.to_string(); let mut active_webcam_offset_sec = 0.0f64; + // Cameras 2-4 of the active clip, as the app (`set_active_clip`) or the scene (auto-advance) + // gave them. The view starts without: `create_view` carries camera 0 only. + let mut active_additional_cameras: Vec = Vec::new(); let mut active_clip_index = 0usize; // Copie de la Scene complète (tous les clips), tenue à jour à chaque push de l'app — // permet à la boucle de lecture libre de connaître la fenêtre source @@ -1567,6 +1826,26 @@ unsafe fn render_thread( // JAMAIS fatal : on retombe sur l'ouverture complète, chemin connu comme sûr. // L'optimisation ne s'applique donc que là où elle fonctionne démontrablement. let repositioned = same_media && matches!(player.seek_active(request.source_time_sec), Ok(true)); + // The extra cameras this clip's layout regions show, from the files the request + // carries. Resolved against the scene before the switch so that the pool and the + // fresh open both see the full key. + let request_scene = shared.scene.lock().unwrap().clone(); + let request_clip_index = request_scene.as_ref().and_then(|scene| { + resolve_scene_clip_index( + scene, + request.clip_index, + &request.screen_path, + &request.webcam_path, + request.webcam_offset_sec, + ) + }); + let request_extras = match (&request_scene, request_clip_index) { + (Some(scene), Some(index)) => extra_camera_keys( + &extra_cameras_for_clip(scene, index), + &request.additional_cameras, + ), + _ => Vec::new(), + }; let switch_result = if repositioned { Ok(()) } else { @@ -1574,13 +1853,24 @@ unsafe fn render_thread( // ouverte si possible (reseek au lieu de rouvrir) et met en pool celle qu'on // quitte, au lieu du couple ouvrir-puis-fermer. Voir `swap_clip_pooled` et la // mesure de ~120 ms/franchissement qui l'a motivé. + let incoming = ClipKey { + screen_path: request.screen_path.clone(), + webcam_path: request.webcam_path.clone(), + webcam_offset_sec: request.webcam_offset_sec, + extras: request_extras.clone(), + }; + let active = ClipKey { + screen_path: active_screen_path.clone(), + webcam_path: active_webcam_path.clone(), + webcam_offset_sec: active_webcam_offset_sec, + extras: player.extra_keys().to_vec(), + }; swap_clip_pooled( &mut player, &mut decoder_pool, - &request, - &active_screen_path, - &active_webcam_path, - active_webcam_offset_sec, + &incoming, + request.source_time_sec, + &active, ) }; match switch_result { @@ -1602,8 +1892,18 @@ unsafe fn render_thread( active_screen_path = request.screen_path; active_webcam_path = request.webcam_path; active_webcam_offset_sec = request.webcam_offset_sec; - let scene = shared.scene.lock().unwrap().clone(); + active_additional_cameras = request.additional_cameras; + let scene = request_scene; full_scene = scene.clone(); + // No scene clip for the request → no layout regions (and no extras). + let regions = match (&scene, request_clip_index) { + (Some(s), Some(index)) => clip_layout_regions(s, index), + _ => Vec::new(), + }; + // A clip request moves the screen position: position the cameras. + if player.set_extra_cameras(regions, request_extras, false) { + comp.clear_srv_cache(); + } if let Some(base_scene) = scene { if let Some(index) = resolve_scene_clip_index( &base_scene, @@ -1709,6 +2009,7 @@ unsafe fn render_thread( prefetch = None; let scene = shared.scene.lock().unwrap().clone(); full_scene = scene.clone(); + let mut clip_resolved = false; let scene = scene.map(|base_scene| { scene_applied = true; if let Some(index) = resolve_scene_clip_index( @@ -1719,9 +2020,35 @@ unsafe fn render_thread( active_webcam_offset_sec, ) { active_clip_index = index; + clip_resolved = true; } scene_for_clip(&base_scene, active_clip_index) }); + // The resolved scene clip's own camera files win over the last clip request's: an + // offset or file edited mid-clip arrives with the scene, not with a new request. + if clip_resolved { + if let Some(clip) = full_scene.as_ref().and_then(|s| s.clips.get(active_clip_index)) { + active_additional_cameras = clip.additional_cameras.clone(); + } + } + // A new scene can add, move or drop the layout regions that show an extra camera: + // open what the active clip now shows, close what it no longer does. + let (regions, keys) = match (&full_scene, &scene) { + (Some(full), Some(windowed)) => ( + windowed.camera_layout_regions.clone(), + extra_camera_keys( + &extra_cameras_for_clip(full, active_clip_index), + &active_additional_cameras, + ), + ), + _ => (Vec::new(), Vec::new()), + }; + // A scene push leaves the screen position where it is: while playing, unchanged + // cameras are stepped by playback instead of sought again. + let steady = shared.playing.load(Ordering::Relaxed); + if player.set_extra_cameras(regions, keys, steady) { + comp.clear_srv_cache(); + } comp.set_scene(scene); } // Le temps programme dépend du clip actif ET de la scène entière (durées des clips @@ -1836,6 +2163,7 @@ unsafe fn render_thread( &mut active_screen_path, &mut active_webcam_path, &mut active_webcam_offset_sec, + &mut active_additional_cameras, &mut active_clip_index, &mut raw_cursor, &mut loaded_cursor_path, @@ -1864,6 +2192,7 @@ unsafe fn render_thread( &mut active_screen_path, &mut active_webcam_path, &mut active_webcam_offset_sec, + &mut active_additional_cameras, &mut active_clip_index, &mut raw_cursor, &mut loaded_cursor_path, @@ -2323,8 +2652,147 @@ mod tests { #[test] fn webcam_seek_uses_screen_source_time_and_offset() { - assert_eq!(webcam_seek_time(22.5, 1.25), 21.25); - assert_eq!(webcam_seek_time(0.5, 1.25), 0.0); + assert_eq!(camera_source_time(22.5, 1.25), 21.25); + assert_eq!(camera_source_time(0.5, 1.25), 0.0); + } + + // --- extra cameras (2-4) ---------------------------------------------------- + + /// Two clips: clip 0 has files for cameras 1 and 3 (camera 2's slot is empty), clip 1 for + /// camera 1 only. Layout regions: clip 0 shows cameras 1 and 2 on 2-5 s and camera 3 only + /// in an empty region; clip 1 shows camera 1 on 21-22 s. + fn layout_scene() -> Scene { + let layer = |camera: usize| { + format!(r#"{{"camera":{camera},"rect":{{"x":0,"y":0,"width":0.5,"height":0.5}},"radiusFrac":0,"shape":"rectangle","fillsFrame":false}}"#) + }; + let json = format!( + r##"{{ + "clips": [ + {{"screenPath":"/s0.mp4","webcamPath":"/w0.mp4","sourceStartSec":0,"sourceEndSec":10,"webcamOffsetSec":0,"hasAudio":true, + "additionalCameras":[{{"path":"/c1.mp4","offsetSec":0.5}},{{"path":"","offsetSec":0}},{{"path":"/c3.mp4","offsetSec":0}}]}}, + {{"screenPath":"/s1.mp4","webcamPath":"/w1.mp4","sourceStartSec":20,"sourceEndSec":30,"webcamOffsetSec":0,"hasAudio":true, + "additionalCameras":[{{"path":"/d1.mp4","offsetSec":0}}]}} + ], + "layout":{{"preset":"picture-in-picture","webcamSize":1,"webcamShape":"rectangle","webcamMirror":false,"webcamPosition":null,"webcamReactiveZoom":false}}, + "effects":{{"padding":0,"blur":false,"shadow":0,"roundnessFrac":0,"motionBlur":0}}, + "background":{{"kind":"color","color":"#000000"}}, + "zoomRegions":[], + "cameraLayoutRegions":[ + {{"clipIndex":0,"startSec":2,"endSec":5,"layers":[{l0},{l1},{l2}]}}, + {{"clipIndex":0,"startSec":7,"endSec":7,"layers":[{l3}]}}, + {{"clipIndex":1,"startSec":21,"endSec":22,"layers":[{l1}]}} + ], + "cursor":{{"show":false,"size":1,"smoothing":0,"motionBlur":0,"clickBounce":0,"clipToBounds":false,"theme":"default"}}, + "cropByClip":[null,null], + "output":{{"width":1920,"height":1080,"fps":30}} + }}"##, + l0 = layer(0), + l1 = layer(1), + l2 = layer(2), + l3 = layer(3), + ); + Scene::from_json(&json).expect("layout scene") + } + + fn camera(path: &str, offset_sec: f64) -> SceneClipCamera { + SceneClipCamera { path: path.to_string(), offset_sec } + } + + #[test] + fn extra_cameras_to_open_are_those_the_clips_regions_reference() { + let scene = layout_scene(); + // Camera 2 is shown but clip 0 has no file for it; camera 3 has a file but only an + // empty region; camera 0 is not an extra camera. + assert_eq!(extra_cameras_for_clip(&scene, 0), vec![1]); + // Clip 1 only sees its own region. + assert_eq!(extra_cameras_for_clip(&scene, 1), vec![1]); + assert!(extra_cameras_for_clip(&scene, 9).is_empty()); + // No layout region at all: nothing to open, whatever files the clip has. + assert!(extra_cameras_for_clip(&multiclip_scene(), 0).is_empty()); + + let keys = scene_extra_camera_keys(&scene, 0); + assert_eq!(keys.len(), 1, "trailing empty slots are dropped"); + let first = keys[0].as_ref().expect("camera 1"); + assert_eq!((first.path.as_str(), first.offset_sec), ("/c1.mp4", 0.5)); + // A shown camera after an unshown one keeps its index; nothing shown is an empty list. + let sources = [camera("/a.mp4", 0.0), camera("/b.mp4", 0.0), camera("/c.mp4", 0.0)]; + let keys = extra_camera_keys(&[3], &sources); + assert_eq!(keys.len(), 3); + assert!(keys[0].is_none() && keys[1].is_none()); + assert_eq!(keys[2].as_ref().map(|k| k.path.as_str()), Some("/c.mp4")); + assert!(extra_camera_keys(&[], &sources).is_empty()); + assert!(extra_camera_keys(&[2], &[camera("/a.mp4", 0.0)]).is_empty()); + } + + #[test] + fn an_extra_camera_is_decoded_only_near_its_regions() { + let regions = scene_for_clip(&layout_scene(), 0).camera_layout_regions; + // From PREFETCH_LEAD_SEC before the region to its end. + assert!(!extra_camera_active(®ions, 1, 2.0 - PREFETCH_LEAD_SEC - 0.01)); + assert!(extra_camera_active(®ions, 1, 2.0 - PREFETCH_LEAD_SEC)); + assert!(extra_camera_active(®ions, 1, 3.0)); + assert!(extra_camera_active(®ions, 1, 5.0)); + assert!(!extra_camera_active(®ions, 1, 5.01)); + // A camera the region does not show, and one only an empty region names. + assert!(!extra_camera_active(®ions, 3, 7.0)); + assert!(!extra_camera_active(&[], 1, 3.0)); + } + + /// Playing on with the same files and region timing, a scene push does not seek the extra + /// cameras again; any change that matters, a jump, or a pause does. + #[test] + fn extra_cameras_are_sought_again_only_when_it_matters() { + assert!(!extras_need_seek(false, false, true)); + assert!(extras_need_seek(true, false, true), "new files"); + assert!(extras_need_seek(false, true, true), "regions moved"); + assert!(extras_need_seek(false, false, false), "paused, or the position jumped"); + + let regions = scene_for_clip(&layout_scene(), 0).camera_layout_regions; + assert!(same_region_timing(®ions, ®ions.clone())); + // A moved rect is the same timing; a moved bound or another camera is not. + let mut moved_rect = regions.clone(); + moved_rect[0].layers[0].rect.x += 0.1; + assert!(same_region_timing(®ions, &moved_rect)); + let mut moved_end = regions.clone(); + moved_end[0].end_sec += 0.5; + assert!(!same_region_timing(®ions, &moved_end)); + let mut other_camera = regions.clone(); + other_camera[0].layers[0].camera = 3; + assert!(!same_region_timing(®ions, &other_camera)); + assert!(!same_region_timing(®ions, ®ions[..1])); + } + + #[test] + fn a_missing_extra_camera_skips_its_layer() { + let failed: Result = Err(anyhow::anyhow!("0-byte file")); + assert_eq!(opened_or_skipped("/c1.mp4", failed), None); + assert_eq!(opened_or_skipped("/c1.mp4", Ok(7u8)), Some(7)); + + // Slot 0 failed to open, slot 1 is open with a frame, slot 2 is open with nothing to + // show (idle or past its end): only slot 1 hands a frame to the compositor. + let frame = 0x10usize as *const AVFrame; + let slots = [None, Some(frame), Some(std::ptr::null())]; + let list = extra_frame_list(&slots, |f| *f); + assert_eq!(list, vec![std::ptr::null(), frame, std::ptr::null()]); + assert!(extra_frame_list::<*const AVFrame>(&[], |f| *f).is_empty()); + } + + #[test] + fn the_pool_key_includes_the_extra_cameras() { + let key = |extras: ExtraCameraKeys| ClipKey { + screen_path: "/s0.mp4".into(), + webcam_path: "/w0.mp4".into(), + webcam_offset_sec: 0.0, + extras, + }; + let with_camera_1 = key(vec![Some(camera("/c1.mp4", 0.5))]); + assert!(with_camera_1.matches(&key(vec![Some(camera("/c1.mp4", 0.5))]))); + // Same screen and camera 0, but camera 1 missing, another file or another offset. + assert!(!with_camera_1.matches(&key(Vec::new()))); + assert!(!with_camera_1.matches(&key(vec![Some(camera("/other.mp4", 0.5))]))); + assert!(!with_camera_1.matches(&key(vec![Some(camera("/c1.mp4", 0.75))]))); + assert!(!with_camera_1.matches(&key(vec![None, Some(camera("/c1.mp4", 0.5))]))); + assert!(key(Vec::new()).matches(&key(Vec::new()))); } #[test] diff --git a/crates/compositor/src/pipeline_linux.rs b/crates/compositor/src/pipeline_linux.rs index 1c47e0b0f..c31eb494f 100644 --- a/crates/compositor/src/pipeline_linux.rs +++ b/crates/compositor/src/pipeline_linux.rs @@ -54,8 +54,13 @@ pub struct ClipSource { pub source_end_sec: f64, pub webcam_offset_sec: f64, pub has_audio: bool, + /// Cameras 2-4 of the clip (index k-1 = camera k, an empty `path` = none). Only those a + /// layout region of the clip shows are decoded. + pub additional_cameras: Vec, } +pub use crate::extra_cameras::ClipCamera; + /// Codec cible. Memes variantes que `pipeline_macos::ExportCodec`. #[derive(Clone, Copy, Debug)] pub enum ExportCodec { @@ -945,6 +950,7 @@ pub fn run_composited_multi( let mut screen_decs: HashMap = HashMap::new(); let mut webcam_decs: HashMap = HashMap::new(); + let mut extra_decs: HashMap = HashMap::new(); // ---- muxer MP4 (flux video + flux AAC) ---- let outc = CString::new(out)?; @@ -1038,6 +1044,7 @@ pub fn run_composited_multi( &scene, &mut screen_decs, &mut webcam_decs, + &mut extra_decs, &mut |n| { // Soumet la copie de la frame n SANS l'attendre et recolte la // precedente : c'est tout le pipelining GPU. L'encodage, lui, diff --git a/crates/compositor/src/pipeline_macos.rs b/crates/compositor/src/pipeline_macos.rs index ad43084a0..6fc75f01a 100644 --- a/crates/compositor/src/pipeline_macos.rs +++ b/crates/compositor/src/pipeline_macos.rs @@ -663,8 +663,13 @@ pub struct ClipSource { pub source_end_sec: f64, pub webcam_offset_sec: f64, pub has_audio: bool, + /// Cameras 2-4 of the clip (index k-1 = camera k, an empty `path` = none). Only those a + /// layout region of the clip shows are decoded. + pub additional_cameras: Vec, } +pub use crate::extra_cameras::ClipCamera; + /// Codec cible pour l'export. Identique à `pipeline_windows::ExportCodec`. pub enum ExportCodec { H264, @@ -1160,6 +1165,8 @@ pub fn run_composited_multi( std::collections::HashMap::new(); let mut webcam_decs: std::collections::HashMap = std::collections::HashMap::new(); + let mut extra_decs: std::collections::HashMap = + std::collections::HashMap::new(); // ---- encodeur (candidat VT ou software) ---- let mut enc = @@ -1241,6 +1248,7 @@ pub fn run_composited_multi( &scene, &mut screen_decs, &mut webcam_decs, + &mut extra_decs, &mut |n| { enc.send_composited(comp, out_w, out_h, n as i64)?; { diff --git a/crates/compositor/src/pipeline_windows.rs b/crates/compositor/src/pipeline_windows.rs index a232f06b7..8094fd44b 100644 --- a/crates/compositor/src/pipeline_windows.rs +++ b/crates/compositor/src/pipeline_windows.rs @@ -1284,8 +1284,13 @@ pub struct ClipSource { pub source_end_sec: f64, pub webcam_offset_sec: f64, pub has_audio: bool, + /// Cameras 2-4 of the clip (index k-1 = camera k, an empty `path` = none). Only those a + /// layout region of the clip shows are decoded. + pub additional_cameras: Vec, } +pub use crate::extra_cameras::ClipCamera; + /// Export **multiclip** : rend la timeline (clips ordonnés, avec trims) en un seul MP4. /// Perf (contrainte §multiclip) : décodeurs ouverts une fois par source (cache) et réutilisés /// entre clips du même asset ; **un seul seek keyframe par frontière de clip** ; décodage @@ -1650,6 +1655,7 @@ unsafe fn run_multi_inner( // pour deux &mut indépendants). let mut screen_decs: HashMap = HashMap::new(); let mut webcam_decs: HashMap = HashMap::new(); + let mut extra_decs: HashMap = HashMap::new(); // fps de sortie : choix explicite de l'app si fourni, sinon dérivé du 1er clip (recordings // uniformes) — comportement historique. @@ -1730,6 +1736,7 @@ unsafe fn run_multi_inner( &scene, &mut screen_decs, &mut webcam_decs, + &mut extra_decs, &mut |frame_index| { // Backend CPU (WARP) : la frame composée descend en mémoire système via // `send_composited` (le compositeur relit son NV12 interne vers un AVFrame diff --git a/crates/compositor/src/regions.rs b/crates/compositor/src/regions.rs index d14c37773..971525b53 100644 --- a/crates/compositor/src/regions.rs +++ b/crates/compositor/src/regions.rs @@ -8,7 +8,9 @@ use crate::camera::Follow; use crate::cursor::CursorTrack; -use crate::scene::{SceneCameraFullscreenRegion, SceneSpeedRegion, SceneZoomRegion}; +use crate::scene::{ + SceneCameraFullscreenRegion, SceneCameraLayoutRegion, SceneSpeedRegion, SceneZoomRegion, +}; /// Quantification commune vidéo/audio : le web retranche exactement 1 ms avant `ceil`. pub const SPEED_FRAME_EPSILON_SEC: f64 = 0.001; @@ -271,8 +273,11 @@ fn push_speed_segment( } // mêmes fenêtres de transition que le web (TRANSITION_WINDOW_MS etc., converties en secondes). -const TRANSITION_WINDOW_S: f32 = 1.01505; -const FULLSCREEN_LEAD_OUT_WINDOW_S: f32 = TRANSITION_WINDOW_S * 1.5; +pub(crate) const TRANSITION_WINDOW_S: f32 = 1.01505; +pub(crate) const FULLSCREEN_LEAD_OUT_WINDOW_S: f32 = TRANSITION_WINDOW_S * 1.5; +/// Two camera regions (layout or Full Camera) closer than this (source seconds) meet at a +/// seam: they hand over directly, without the default PiP in between. +pub(crate) const REGION_SEAM_S: f64 = 0.001; // Durée d'un zoom à l'écran selon l'échelle visée, cf. `zoom_transition_s` ; miroir de // `ZOOM_TRANSITION_BASE_MS` / `ZOOM_TRANSITION_PER_LN_MS` (TS). const ZOOM_TRANSITION_BASE_S: f32 = 0.6; @@ -328,7 +333,7 @@ fn cubic_bezier(x1: f32, y1: f32, x2: f32, y2: f32, t: f32) -> f32 { } /// Port de `easeOutScreenStudio` (TS) : cubic-bezier(0.16, 1, 0.3, 1). -fn ease_out_screen_studio(t: f32) -> f32 { +pub(crate) fn ease_out_screen_studio(t: f32) -> f32 { cubic_bezier(0.16, 1.0, 0.3, 1.0, t) } @@ -973,16 +978,20 @@ fn camera_fullscreen_region_phase( region: &SceneCameraFullscreenRegion, t: f32, clock: &ScreenClock, + layouts: &[SceneCameraLayoutRegion], ) -> f32 { let start = region.start_sec as f32; let end = region.end_sec as f32; if t <= start || t >= end { return 0.0; } + // At a seam with a layout region the camera is already frame-filling on that side: the + // layout region's own glide (`camera_layers_at`) moves it, so the phase holds 1 there. + let seams = full_camera_seams(region, layouts); let (start, end, t) = (clock.at(start), clock.at(end), clock.at(t)); let half = (end - start) * 0.5; - let lead_in = TRANSITION_WINDOW_S.min(half); - let lead_out = FULLSCREEN_LEAD_OUT_WINDOW_S.min(half); + let lead_in = if seams.start { 0.0 } else { TRANSITION_WINDOW_S.min(half) }; + let lead_out = if seams.end { 0.0 } else { FULLSCREEN_LEAD_OUT_WINDOW_S.min(half) }; let lead_in_end = start + lead_in; let lead_out_start = end - lead_out; if t < lead_in_end { @@ -1002,16 +1011,38 @@ fn camera_fullscreen_region_phase( } } +/// Which ends of a Full Camera region meet a (non-empty) camera layout region at a seam +/// (`REGION_SEAM_S`): `start` = a layout region ends where it starts, `end` = one starts where +/// it ends. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub(crate) struct FullCameraSeams { + pub start: bool, + pub end: bool, +} + +pub(crate) fn full_camera_seams( + region: &SceneCameraFullscreenRegion, + layouts: &[SceneCameraLayoutRegion], +) -> FullCameraSeams { + let mut seams = FullCameraSeams::default(); + for l in layouts.iter().filter(|l| l.end_sec > l.start_sec) { + seams.start |= (l.end_sec - region.start_sec).abs() <= REGION_SEAM_S; + seams.end |= (l.start_sec - region.end_sec).abs() <= REGION_SEAM_S; + } + seams +} + /// Phase Full Camera (0..1) au temps `t`. Régions superposées (ne devrait pas arriver, gardé /// défensif comme le web) → la plus forte gagne. fn camera_fullscreen_phase_at( regions: &[SceneCameraFullscreenRegion], t: f32, clock: &ScreenClock, + layouts: &[SceneCameraLayoutRegion], ) -> f32 { let mut strongest = 0.0f32; for r in regions { - let s = camera_fullscreen_region_phase(r, t, clock); + let s = camera_fullscreen_region_phase(r, t, clock, layouts); if s > strongest { strongest = s; } @@ -1024,8 +1055,9 @@ pub fn camera_fullscreen_progress_at( regions: &[SceneCameraFullscreenRegion], t: f32, clock: &ScreenClock, + layouts: &[SceneCameraLayoutRegion], ) -> f32 { - ease_out_screen_studio(camera_fullscreen_phase_at(regions, t, clock)) + ease_out_screen_studio(camera_fullscreen_phase_at(regions, t, clock, layouts)) } /// Ce qui reste de la bulle PiP (coins arrondis, ombre) au temps `t` : 1 = bulle entière, @@ -1038,8 +1070,72 @@ pub fn camera_fullscreen_shape_at( regions: &[SceneCameraFullscreenRegion], t: f32, clock: &ScreenClock, + layouts: &[SceneCameraLayoutRegion], ) -> f32 { - 1.0 - smoothstep(0.5, 1.0, camera_fullscreen_phase_at(regions, t, clock)) + 1.0 - smoothstep(0.5, 1.0, camera_fullscreen_phase_at(regions, t, clock, layouts)) +} + +/// The Full Camera region in effect at `t`: the one with the strongest phase, so a region's +/// orientation and its grow/shrink always refer to the same region. `None` wherever the +/// envelope is 0 — outside every region and on their bounds. +pub fn camera_fullscreen_region_at<'a>( + regions: &'a [SceneCameraFullscreenRegion], + t: f32, + clock: &ScreenClock, + layouts: &[SceneCameraLayoutRegion], +) -> Option<&'a SceneCameraFullscreenRegion> { + let mut best: Option<(&'a SceneCameraFullscreenRegion, f32)> = None; + for r in regions { + let phase = camera_fullscreen_region_phase(r, t, clock, layouts); + if phase > 0.0 && best.map_or(true, |(_, b)| phase > b) { + best = Some((r, phase)); + } + } + best.map(|(r, _)| r) +} + +/// How long a turned section's camera takes to sharpen after the hold, and to blur again +/// before the shrink. See `camera_fullscreen_cover_at`. +pub const DESK_COVER_FADE_S: f32 = 0.5; + +/// Cover strength of the webcam at `t` (0 = sharp, 1 = fully blurred and dimmed). Only a +/// turned section (a camera tilted onto the desk) is covered: the camera is moving at both +/// ends of such a section, so the picture is hidden for exactly the grow and the shrink, plus a +/// short fade into and out of the steady part. Measured on screen time like the grow, and 0 +/// on and outside the bounds like the grow. At a seam with a layout region (`layouts`) that +/// side has no grow or shrink to hold over: the cover is only its own fade there. +pub fn camera_fullscreen_cover_at( + regions: &[SceneCameraFullscreenRegion], + t: f32, + clock: &ScreenClock, + layouts: &[SceneCameraLayoutRegion], +) -> f32 { + let Some(r) = camera_fullscreen_region_at(regions, t, clock, layouts) else { + return 0.0; + }; + if r.rotation != 180 { + return 0.0; + } + let seams = full_camera_seams(r, layouts); + let (start, end, t) = (clock.at(r.start_sec as f32), clock.at(r.end_sec as f32), clock.at(t)); + // Each hold gets at most half the section; the two fades share what is left, so the start + // fade ends at or before the point where the end fade begins and the cover never jumps. + let len = end - start; + let half = len * 0.5; + let hold_in = if seams.start { 0.0 } else { TRANSITION_WINDOW_S.min(half) }; + let hold_out = if seams.end { 0.0 } else { FULLSCREEN_LEAD_OUT_WINDOW_S.min(half) }; + let steady = (len - hold_in - hold_out).max(0.0); + let fade = DESK_COVER_FADE_S.min(steady * 0.5); + let side = |since: f32, hold: f32, fade: f32| -> f32 { + if since <= hold { + 1.0 + } else if fade > 0.0 && since < hold + fade { + 1.0 - smoothstep(0.0, fade, since - hold) + } else { + 0.0 + } + }; + side(t - start, hold_in, fade).max(side(end - t, hold_out, fade)) } // ============ Rotation 3D (tilt perspective, présets iso/left/right) ================ @@ -1977,6 +2073,9 @@ mod zoom_focus_tests { clip_index: None, start_sec: 18.0 * k, end_sec: 22.0 * k, + rotation: 0, + mirror: None, + full_frame: false, }] }; let (sped, plain) = (zooms(4.0), zooms(1.0)); @@ -1996,21 +2095,223 @@ mod zoom_focus_tests { b.scale ); assert!((a.focus[0] - b.focus[0]).abs() < 1e-4, "t={t}"); - let a = camera_fullscreen_progress_at(&full_camera(4.0), t, &clock); + let a = camera_fullscreen_progress_at(&full_camera(4.0), t, &clock, &[]); let b = - camera_fullscreen_progress_at(&full_camera(1.0), t / 4.0, &ScreenClock::default()); + camera_fullscreen_progress_at(&full_camera(1.0), t / 4.0, &ScreenClock::default(), &[]); assert!((a - b).abs() < 1e-4, "Full Camera t={t} : {a} vs {b}"); } } + fn cam(start: f64, end: f64, rotation: u16) -> SceneCameraFullscreenRegion { + SceneCameraFullscreenRegion { + clip_index: None, + start_sec: start, + end_sec: end, + rotation, + mirror: None, + full_frame: rotation != 0, + } + } + + /// Orientation and transition must name the same region: inside it the region, at its + /// bounds and outside nothing, exactly where the progress envelope is 0. + #[test] + fn the_full_camera_region_at_t_is_the_one_the_envelope_uses() { + let r = [cam(10.0, 20.0, 180), cam(30.0, 40.0, 0)]; + let clock = ScreenClock::default(); + assert_eq!(camera_fullscreen_region_at(&r, 15.0, &clock, &[]).map(|c| c.rotation), Some(180)); + assert_eq!(camera_fullscreen_region_at(&r, 35.0, &clock, &[]).map(|c| c.rotation), Some(0)); + for t in [5.0, 10.0, 20.0, 25.0] { + assert!(camera_fullscreen_region_at(&r, t, &clock, &[]).is_none(), "t={t}"); + assert_eq!(camera_fullscreen_progress_at(&r, t, &clock, &[]), 0.0, "t={t}"); + } + } + + + /// The hold covers exactly the grow; then the picture sharpens over the fade. + #[test] + fn a_turned_section_is_covered_while_the_camera_moves() { + let r = [cam(10.0, 30.0, 180)]; + let clock = ScreenClock::default(); + let c = |t: f32| camera_fullscreen_cover_at(&r, t, &clock, &[]); + assert_eq!(c(10.0 + 0.01), 1.0, "start of the hold"); + assert_eq!(c(10.0 + TRANSITION_WINDOW_S - 0.01), 1.0, "end of the hold"); + let mid_fade = c(10.0 + TRANSITION_WINDOW_S + DESK_COVER_FADE_S / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-3, "half way through the fade: {mid_fade}"); + assert_eq!(c(20.0), 0.0, "steady part is sharp"); + // Mirrored at the end: fade up, then hold for the whole shrink. + let mid_rise = c(30.0 - FULLSCREEN_LEAD_OUT_WINDOW_S - DESK_COVER_FADE_S / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-3, "half way up: {mid_rise}"); + assert_eq!(c(30.0 - FULLSCREEN_LEAD_OUT_WINDOW_S + 0.01), 1.0); + assert_eq!(c(30.0 - 0.01), 1.0); + } + + fn layout_region(start: f64, end: f64) -> SceneCameraLayoutRegion { + SceneCameraLayoutRegion { clip_index: None, start_sec: start, end_sec: end, layers: Vec::new() } + } + + /// A layout region meeting a Full Camera region at a seam: the phase holds 1 on that side + /// (no lead-out before it, no lead-in after it); the other side is unchanged. + #[test] + fn the_full_camera_phase_holds_at_a_seam_with_a_layout_region() { + let clock = ScreenClock::default(); + let r = [cam(10.0, 20.0, 0)]; + let p = |t: f32, layouts: &[SceneCameraLayoutRegion]| { + camera_fullscreen_progress_at(&r, t, &clock, layouts) + }; + let after = [layout_region(20.0, 25.0)]; + let before = [layout_region(5.0, 10.0)]; + for t in [19.0, 19.5, 19.99] { + assert_eq!(p(t, &after), 1.0, "t={t}"); + assert!(p(t, &[]) < 1.0, "a lone region leads out (t={t})"); + } + assert!(p(10.3, &after) < 1.0, "the far side keeps its lead-in"); + for t in [10.01, 10.3, 10.9] { + assert_eq!(p(t, &before), 1.0, "t={t}"); + } + assert!(p(19.5, &before) < 1.0, "the far side keeps its lead-out"); + // On the bounds the phase is 0 as always: the layout region draws the seam. + assert_eq!(p(20.0, &after), 0.0); + assert_eq!(p(10.0, &before), 0.0); + // Off the seam (10 ms), or an empty layout region, changes nothing. + for layouts in [[layout_region(20.01, 25.0)], [layout_region(20.0, 20.0)]] { + assert_eq!(p(19.5, &layouts), p(19.5, &[])); + } + } + + /// A turned section at a seam: no hold on that side, the cover is its own fade there. + #[test] + fn at_a_seam_the_cover_is_only_its_fade() { + let clock = ScreenClock::default(); + let r = [cam(10.0, 30.0, 180)]; + let layouts = [layout_region(5.0, 10.0), layout_region(30.0, 35.0)]; + let c = |t: f32| camera_fullscreen_cover_at(&r, t, &clock, &layouts); + assert!(c(10.001) > 0.99, "covered at the seam: {}", c(10.001)); + let mid_fade = c(10.0 + DESK_COVER_FADE_S / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-3, "half way through the fade: {mid_fade}"); + assert_eq!(c(10.0 + DESK_COVER_FADE_S + 0.01), 0.0, "sharp after the fade"); + let mid_rise = c(30.0 - DESK_COVER_FADE_S / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-3, "half way up: {mid_rise}"); + assert_eq!(c(30.0 - DESK_COVER_FADE_S - 0.01), 0.0); + assert!(c(29.999) > 0.99); + // Without the seams the holds are back. + let lone = |t: f32| camera_fullscreen_cover_at(&r, t, &clock, &[]); + assert_eq!(lone(10.0 + DESK_COVER_FADE_S + 0.01), 1.0); + } + + #[test] + fn the_cover_is_zero_on_and_outside_the_bounds() { + let r = [cam(10.0, 30.0, 180)]; + let clock = ScreenClock::default(); + for t in [0.0, 9.99, 10.0, 30.0, 30.01, 40.0] { + assert_eq!(camera_fullscreen_cover_at(&r, t, &clock, &[]), 0.0, "t={t}"); + } + } + + #[test] + fn a_plain_section_is_never_covered() { + let r = [cam(10.0, 30.0, 0)]; + let clock = ScreenClock::default(); + for step in 0..=400 { + let t = step as f32 * 0.1; + assert_eq!(camera_fullscreen_cover_at(&r, t, &clock, &[]), 0.0, "t={t}"); + } + } + + /// Each hold gets at most half the section and the fades share what is left, so a section + /// shorter than both holds is covered throughout and a longer one keeps its full fades. + #[test] + fn short_sections_shrink_the_fades_first() { + let clock = ScreenClock::default(); + let short = [cam(10.0, 11.5, 180)]; // half = 0.75 < hold-in: no fade at all + for step in 1..150 { + let t = 10.0 + step as f32 * 0.01; + let c = camera_fullscreen_cover_at(&short, t, &clock, &[]); + assert!((0.0..=1.0).contains(&c), "t={t}: {c}"); + assert_eq!(c, 1.0, "a section shorter than both holds is covered throughout (t={t})"); + } + // 3.6 s: steady = 3.6 - 1.015 - 1.523 = 1.06, so both ends keep their full 0.5 s fade. + let medium = [cam(10.0, 13.6, 180)]; + let c = |t: f32| camera_fullscreen_cover_at(&medium, t, &clock, &[]); + let mid_fade = c(10.0 + TRANSITION_WINDOW_S + DESK_COVER_FADE_S / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-3, "start fade half way: {mid_fade}"); + let mid_rise = c(13.6 - FULLSCREEN_LEAD_OUT_WINDOW_S - DESK_COVER_FADE_S / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-3, "end fade half way: {mid_rise}"); + } + + /// 2.6 s: hold-in 1.015, hold-out 1.3 (half), steady 0.285 — the two fades share it, + /// 0.1425 s each, so the start fade ends exactly where the end fade begins. + #[test] + fn medium_sections_share_the_steady_part_between_the_fades() { + let clock = ScreenClock::default(); + let r = [cam(10.0, 12.6, 180)]; + let c = |t: f32| camera_fullscreen_cover_at(&r, t, &clock, &[]); + let fade = (2.6 - TRANSITION_WINDOW_S - 1.3) / 2.0; + let meet = 10.0 + TRANSITION_WINDOW_S + fade; + let mid_fade = c(10.0 + TRANSITION_WINDOW_S + fade / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-2, "start fade half way: {mid_fade}"); + assert!(c(meet) < 1e-2, "both fades are near zero where they meet: {}", c(meet)); + let mid_rise = c(12.6 - 1.3 - fade / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-2, "end fade half way: {mid_rise}"); + assert_eq!(c(12.6 - 1.3 + 0.01), 1.0, "the end hold is half the section"); + } + + /// No section length makes the cover jump: sampled every 1 ms, no step exceeds 0.1. (At + /// 10 ms the shortest legitimate fade — 42 ms in a 2.2 s section — already steps ~0.3 per + /// sample, so the finer grid is what tells a ramp from a jump.) + #[test] + fn the_cover_is_continuous_for_every_section_length() { + let clock = ScreenClock::default(); + for len in [2.0_f32, 2.2, 2.6, 3.0, 3.5, 4.0, 4.5] { + let r = [cam(10.0, 10.0 + len as f64, 180)]; + let steps = (len * 1000.0).round() as i32; + let mut prev = camera_fullscreen_cover_at(&r, 10.0 + 0.0005, &clock, &[]); + for step in 1..steps { + let t = 10.0 + step as f32 * 0.001; + let c = camera_fullscreen_cover_at(&r, t, &clock, &[]); + assert!((0.0..=1.0).contains(&c), "len={len} t={t}: {c}"); + assert!((c - prev).abs() <= 0.1, "len={len} t={t}: {prev} -> {c}"); + prev = c; + } + } + } + + /// Under a speed change the windows are measured on screen time, like the grow. + #[test] + fn the_cover_runs_on_screen_time() { + let clock = ScreenClock::new(&[speed(None, 0.0, 1000.0, 2.0)], 0); + let r = [cam(10.0, 40.0, 180)]; + // At 2x, the 1.015 s hold spans 2.03 s of source time. + assert_eq!(camera_fullscreen_cover_at(&r, 10.0 + 1.9, &clock, &[]), 1.0); + assert!(camera_fullscreen_cover_at(&r, 10.0 + 3.6, &clock, &[]) < 1.0); + } + #[test] + fn the_region_fields_parse_and_default() { + let plain: SceneCameraFullscreenRegion = + serde_json::from_str(r#"{"startSec":1,"endSec":2}"#).unwrap(); + assert_eq!((plain.rotation, plain.mirror, plain.full_frame), (0, None, false)); + let desk: SceneCameraFullscreenRegion = serde_json::from_str( + r#"{"startSec":1,"endSec":2,"rotation":180,"mirror":false,"fullFrame":true}"#, + ) + .unwrap(); + assert_eq!((desk.rotation, desk.mirror, desk.full_frame), (180, Some(false), true)); + } + /// Les coins suivent la phase, pas le rect. Avec `1 - progrès`, ils étaient carrés presque /// tout du long de la montée et ne revenaient qu'à la toute fin du retour. #[test] fn full_camera_corners_dissolve_late_and_come_back_early() { - let r = [SceneCameraFullscreenRegion { clip_index: None, start_sec: 10.0, end_sec: 20.0 }]; + let r = [SceneCameraFullscreenRegion { + clip_index: None, + start_sec: 10.0, + end_sec: 20.0, + rotation: 0, + mirror: None, + full_frame: false, + }]; let clock = ScreenClock::default(); let at = |t: f32| { - (camera_fullscreen_progress_at(&r, t, &clock), camera_fullscreen_shape_at(&r, t, &clock)) + (camera_fullscreen_progress_at(&r, t, &clock, &[]), camera_fullscreen_shape_at(&r, t, &clock, &[])) }; // Premier tiers de la montée : la caméra a fait l'essentiel du chemin, coins intacts. let (progress, shape) = at(10.0 + TRANSITION_WINDOW_S / 3.0); diff --git a/crates/compositor/src/scene.rs b/crates/compositor/src/scene.rs index bddd7d55c..98a5957c4 100644 --- a/crates/compositor/src/scene.rs +++ b/crates/compositor/src/scene.rs @@ -19,6 +19,67 @@ pub struct SceneClip { /// Une source sans piste audio décodable garde sa durée via du silence natif. #[serde(default)] pub has_audio: bool, + /// Cameras 2-4: index k-1 = camera k. An empty `path` means "no camera in this slot". + #[serde(default)] + pub additional_cameras: Vec, +} + +/// An additional camera (2-4) of a clip. = `CompositorClipCamera` (TS). +#[derive(Debug, Clone, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SceneClipCamera { + pub path: String, + /// Camera source time = screen source time - this. + pub offset_sec: f64, +} + +/// Settings of one camera (index 0 = camera 1). With a `homography` the camera's rotation, +/// mirror and crop are not sent (the corner order already defines the picture). +#[derive(Debug, Clone, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SceneCamera { + pub index: usize, + /// 0 or 180. + #[serde(default)] + pub rotation: u16, + #[serde(default)] + pub mirror: Option, + #[serde(default)] + pub crop: Option, + /// Row-major 3x3, target uv -> camera uv. + #[serde(default)] + pub homography: Option<[f32; 9]>, + /// Width/height of the corrected picture. + #[serde(default)] + pub aspect: Option, +} + +/// One camera placed by a layout region, in draw order. +#[derive(Debug, Clone, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SceneCameraLayer { + pub camera: usize, + pub rect: SceneRect, + #[serde(default)] + pub radius_frac: f32, + /// Same vocabulary as `SceneLayout::webcam_shape` (`webcam_shape_code`). + #[serde(default)] + pub shape: String, + /// Covers the screen: drawn without a shadow. + #[serde(default)] + pub fills_frame: bool, +} + +/// A camera layout (anything but a plain Full Camera) on a span of a clip's source time. +#[derive(Debug, Clone, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct SceneCameraLayoutRegion { + #[serde(default)] + pub clip_index: Option, + pub start_sec: f64, + pub end_sec: f64, + #[serde(default)] + pub layers: Vec, } #[derive(Debug, Clone, Copy, Deserialize)] @@ -528,7 +589,8 @@ pub struct SceneSpeedRegion { /// Une zone "Full Camera" de la timeline (temps en secondes) : la caméra PREND tout le cadre /// pendant cette fenêtre (plein écran net — ni marge, ni arrondi, ni masque, ni fond derrière). -/// Pas de champs au-delà des bornes temporelles (miroir de `CameraFullscreenRegion`, TS). +/// The orientation fields are resolved by the app (`sceneDescription.ts`): a desk shot +/// arrives as `rotation: 180`, `mirror: false`, `fullFrame: true`. #[derive(Debug, Clone, Copy, Deserialize)] #[serde(rename_all = "camelCase")] pub struct SceneCameraFullscreenRegion { @@ -537,6 +599,15 @@ pub struct SceneCameraFullscreenRegion { pub clip_index: Option, pub start_sec: f64, pub end_sec: f64, + /// 0 or 180. Anything else is treated as 0 by `frame_geometry::webcam_orientation`. + #[serde(default)] + pub rotation: u16, + /// Mirror inside this region; `None` = the layout's `webcam_mirror`. + #[serde(default)] + pub mirror: Option, + /// Ignore `layout.webcam_crop` inside this region. + #[serde(default)] + pub full_frame: bool, } /// Rendu du curseur. @@ -730,6 +801,12 @@ pub struct Scene { /// `#[serde(default)]` : champ ajouté après coup, absent des JSON de test existants. #[serde(default)] pub camera_fullscreen_regions: Vec, + /// Settings of the cameras that have any (see `Scene::camera`). + #[serde(default)] + pub cameras: Vec, + /// Layouts that put several cameras on screen. Absent = none. + #[serde(default)] + pub camera_layout_regions: Vec, pub cursor: SceneCursor, /// Global audio finishing. Default keeps old scene payloads bit-for-bit compatible. #[serde(default)] @@ -751,6 +828,11 @@ pub struct Scene { } impl Scene { + /// The settings of camera `index` (0 = camera 1), if the scene carries any. + pub fn camera(&self, index: usize) -> Option<&SceneCamera> { + self.cameras.iter().find(|c| c.index == index) + } + /// Parse le JSON produit par `buildSceneDescription` (TS). pub fn from_json(json: &str) -> anyhow::Result { Ok(serde_json::from_str(json)?) @@ -788,6 +870,9 @@ impl Scene { scene.camera_fullscreen_regions.retain(|region| { belongs(region.clip_index, region.start_sec, region.end_sec) }); + scene.camera_layout_regions.retain(|region| { + belongs(region.clip_index, region.start_sec, region.end_sec) + }); scene.annotations.retain(|annotation| { belongs(annotation.clip_index, annotation.start_sec, annotation.end_sec) }); @@ -943,6 +1028,83 @@ mod tests { } } + /// A scene JSON with one clip; the arguments are spliced in as extra clip / top-level fields. + fn scene_json_with(clip_extra: &str, top_extra: &str) -> String { + format!( + r##"{{"clips":[{{"screenPath":"/s.mp4","webcamPath":"/w.mp4","sourceStartSec":0,"sourceEndSec":10,"webcamOffsetSec":0{clip_extra}}}],"layout":{{"preset":"picture-in-picture","webcamSize":1,"webcamShape":"rectangle","webcamMirror":false,"webcamPosition":null,"webcamReactiveZoom":false}},"effects":{{"padding":0,"blur":false,"shadow":0,"roundnessFrac":0,"motionBlur":0}},"background":{{"kind":"color","color":"#000000"}},"zoomRegions":[],"cursor":{{"show":false,"size":1,"smoothing":0,"motionBlur":0,"clickBounce":0,"clipToBounds":false,"theme":"default"}},"cropByClip":[],"output":{{"width":1920,"height":1080,"fps":null}}{top_extra}}}"## + ) + } + + #[test] + fn parses_extra_cameras_settings_and_layout_regions() { + let json = scene_json_with( + r#","additionalCameras":[{"path":"/w-2.mp4","offsetSec":0.12},{"path":"","offsetSec":0}]"#, + r#","cameras":[{"index":1,"rotation":180,"mirror":true,"crop":{"x":0,"y":0.1,"width":0.5,"height":0.8}},{"index":2,"homography":[1,0,0,0,1,0,0,0,1],"aspect":1.5}], + "cameraLayoutRegions":[{"clipIndex":0,"startSec":1,"endSec":4,"underTrim":true,"layers":[ + {"camera":1,"rect":{"x":0,"y":0,"width":1,"height":1},"radiusFrac":0,"shape":"rectangle","fillsFrame":true}, + {"camera":0,"rect":{"x":0.7,"y":0.7,"width":0.22,"height":0.2},"radiusFrac":0.12,"shape":"rounded","fillsFrame":false}]}]"#, + ); + let s = Scene::from_json(&json).expect("parse"); + assert_eq!(s.clips[0].additional_cameras.len(), 2); + assert_eq!(s.clips[0].additional_cameras[0].path, "/w-2.mp4"); + assert!((s.clips[0].additional_cameras[0].offset_sec - 0.12).abs() < 1e-9); + assert_eq!(s.clips[0].additional_cameras[1].path, ""); + assert!(s.camera(0).is_none()); + let c1 = s.camera(1).expect("camera 1"); + assert_eq!(c1.rotation, 180); + assert_eq!(c1.mirror, Some(true)); + assert_eq!(c1.crop.expect("crop").height, 0.8); + assert!(c1.homography.is_none() && c1.aspect.is_none()); + let c2 = s.camera(2).expect("camera 2"); + assert_eq!(c2.rotation, 0); + assert_eq!(c2.homography, Some([1.0, 0.0, 0.0, 0.0, 1.0, 0.0, 0.0, 0.0, 1.0])); + assert_eq!(c2.aspect, Some(1.5)); + assert_eq!(s.camera_layout_regions.len(), 1); + let r = &s.camera_layout_regions[0]; + assert_eq!((r.clip_index, r.start_sec, r.end_sec), (Some(0), 1.0, 4.0)); + assert_eq!(r.layers.len(), 2); + assert_eq!(r.layers[0].camera, 1); + assert!(r.layers[0].fills_frame); + assert_eq!(r.layers[1].shape, "rounded"); + assert!((r.layers[1].radius_frac - 0.12).abs() < 1e-6); + assert!((r.layers[1].rect.width - 0.22).abs() < 1e-6); + } + + #[test] + fn old_scene_json_parses_without_the_new_fields() { + let s = Scene::from_json(&scene_json_with("", "")).expect("parse"); + assert!(s.clips[0].additional_cameras.is_empty()); + assert!(s.cameras.is_empty()); + assert!(s.camera_layout_regions.is_empty()); + assert!(s.camera(0).is_none()); + } + + #[test] + fn for_clip_window_keeps_only_this_clips_layout_regions() { + let layers = r#""layers":[{"camera":0,"rect":{"x":0,"y":0,"width":1,"height":1},"radiusFrac":0,"shape":"rectangle","fillsFrame":true}]"#; + let json = scene_json_with( + "", + &format!( + r#","cameraLayoutRegions":[ + {{"clipIndex":0,"startSec":1,"endSec":2,{layers}}}, + {{"clipIndex":1,"startSec":1,"endSec":2,{layers}}}, + {{"startSec":8,"endSec":9,{layers}}}, + {{"startSec":20,"endSec":30,{layers}}}]"# + ), + ); + let s = Scene::from_json(&json).expect("parse"); + let w = s.for_clip_window(0, 0.0, 10.0); + // Clip 0's own region and the unaddressed one that overlaps the window; not clip 1's, + // not the unaddressed one outside the window. + let kept: Vec<(Option, f64)> = w + .camera_layout_regions + .iter() + .map(|r| (r.clip_index, r.start_sec)) + .collect(); + assert_eq!(kept, vec![(Some(0), 1.0), (None, 8.0)]); + assert_eq!(s.for_clip_window(1, 0.0, 10.0).camera_layout_regions.len(), 2); + } + #[test] fn parses_webcam_rect_payload() { // webcamRect est une fraction 0..1 du cadre de sortie ; sa présence doit désactiver diff --git a/crates/compositor/src/shaders.hlsl b/crates/compositor/src/shaders.hlsl index c17d207d9..e5777929e 100644 --- a/crates/compositor/src/shaders.hlsl +++ b/crates/compositor/src/shaders.hlsl @@ -23,6 +23,9 @@ cbuffer Layer : register(b0) float4 trail_a; // mode 8 : coins TL, TR du plan à la frame précédente (px locaux, comme fx) ; mode 18 incliné : en fractions de sortie float4 trail_b; // mode 8 : coins BR, BL du plan à la frame précédente (comme src_prev) ; mode 18 incliné : en fractions de sortie float4 trail_mb; // mode 8 : x = taps, y = force du flou de mouvement (ceux du mode 0) ; mode 18 incliné : le `mb` du mode 8 (profondeur de champ), et `color.xy` sa lampe ; 0 ailleurs + float4 cover; // x = desk-view cover 0..1, y = blur radius (quad px), z = dim, w unused + float4 persp[3]; // mode 0: rows of the homography quad point (0..1 in dst) -> camera uv (0..1 of the valid frame), xyz; read when layer_fx.y = 1 + float4 layer_fx; // x = transparency 0..1 (0 = opaque), every mode; y = 1 when persp applies; z, w unused }; // Mode 15 (curseur modélisé) : le détail des emplacements est dans `frame_geometry.rs`, en tête // de la section « Curseur modélisé » (`cursor_model_cb`). Mode 17 (appareil modelé) : en tête de @@ -475,6 +478,19 @@ float band_cov(float x, float half_w) return saturate(half_w + 0.5 - abs(x)); } +// Camera homography (`persp`, read when `layer_fx.y` = 1): the camera point (0..1 of its valid +// frame) seen at `local` (0..1 in the layer's dst quad), and in z whether there is one -- 0 when +// the point is behind the projection (q.z <= 0) or outside the camera frame, where the layer is +// transparent. +float3 persp_camera(float2 local) +{ + float3 p = float3(local, 1.0); + float3 q = float3(dot(persp[0].xyz, p), dot(persp[1].xyz, p), dot(persp[2].xyz, p)); + float2 cam = q.xy / max(q.z, 1e-6); + bool inside = q.z > 0.0 && all(cam >= 0.0) && all(cam <= 1.0); + return float3(cam, inside ? 1.0 : 0.0); +} + // Fond flouté pour le mode "blur" de la webcam. // Disque de Vogel (spirale à angle d'or) à 21 échantillons avec pondération gaussienne et // rotation par pixel via Interleaved Gradient Noise (IGN) pour un bokeh photographique doux, isotrope et rapide. @@ -502,10 +518,15 @@ static const float3 VOGEL_TAPS[21] = { float3(-0.633036, -0.758588, 0.087119) }; -float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px) +// Vogel-disc blur of the camera texture at a radius in quad pixels. `valid` is the part of the +// texture the picture fills (fx.xy): decoders allocate aligned textures (a 1080-line camera in a +// 1088-line texture), so each tap is clamped half a chroma texel inside it, never into padding. +float3 blur_webcam_radius(float2 uv, float max_r_px, float2 qpx, float2 local_px, float2 valid) { - float max_r_px = max(intensity, 0.0) * 22.0 + 1.5; float2 step = max_r_px / max(qpx, 1.0); + float cw, ch; + texUV.GetDimensions(cw, ch); + float2 hi = max(valid - 0.5 / max(float2(cw, ch), 1.0), 0.0); // Interleaved Gradient Noise pour rotation aléatoire par pixel float noise = frac(52.9829189 * frac(0.06711056 * local_px.x + 0.00583715 * local_px.y)); float angle = noise * 6.2831853; @@ -518,12 +539,17 @@ float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px) float2 p = VOGEL_TAPS[k].xy; float w = VOGEL_TAPS[k].z; float2 rot_p = float2(p.x * c - p.y * s, p.x * s + p.y * c); - sum += sample_yuv(saturate(uv + rot_p * step)) * w; + sum += sample_yuv(clamp(uv + rot_p * step, 0.0, hi)) * w; total += w; } return sum / max(total, 1e-4); } +float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px, float2 valid) +{ + return blur_webcam_radius(uv, max(intensity, 0.0) * 22.0 + 1.5, qpx, local_px, valid); +} + // ============ Curseur MODÉLISÉ (mode 15) ============ // L'état courant en objet 3D, lancé de rayons par pixel. Deux sortes d'objets : // - un curseur SCULPTÉ (`trail_a.x` > 0) : la flèche ou la main d'un des cinq thèmes d'origine, @@ -3468,7 +3494,8 @@ float4 device_frame(float2 local) return float4(rgb * a, a); // prémultiplié } -float4 ps_main(VSOut i) : SV_Target +// The body of `ps_main`, which only adds the layer's transparency on top of it. +float4 ps_layer(VSOut i) { // mode 18 : l'écran CADRÉ (ombre, cadre, métrage, appareil) flouté comme UN objet rigide // (`FrameGeometry::screen_trail`). t2 = son rendu isolé, prémultiplié, à la taille de la @@ -3980,6 +4007,8 @@ float4 ps_main(VSOut i) : SV_Target float3 rgb; // 1 sauf en mode detourage, ou il porte le masque du sujet (cf. la branche fx.z ci-dessous). float alpha_mask = 1.0; + // 0 where a camera homography finds no camera point (`persp_camera`), 1 everywhere else. + float persp_keep = 1.0; if (mode < 0.5) { // flou de mouvement par vélocité (§8) : pour CE pixel sortie, uv à la frame @@ -3988,6 +4017,17 @@ float4 ps_main(VSOut i) : SV_Target float2 uv_now = i.uv; float2 localp = (i.pout - dst_prev.xy) / dst_prev.zw; float2 uv_prev = src_prev.xy + localp * (src_prev.zw - src_prev.xy); + if (layer_fx.y > 0.5) + { + // Camera homography: the texture uv comes from the fragment's place in the quad + // (`src` is not used), scaled to the valid part of the decoder texture. The previous + // frame's uv is the same map at the quad's previous place. + float3 cam = persp_camera((i.pout - dst.xy) / dst.zw); + persp_keep = cam.z; + uv_now = cam.xy * fx.xy; + float3 cam_prev = persp_camera(localp); + uv_prev = (cam_prev.z > 0.5) ? cam_prev.xy * fx.xy : uv_now; + } float2 duv = uv_now - uv_prev; float mb_scale = saturate(mb.y); float2 duv_blur = duv * mb_scale; @@ -4028,20 +4068,28 @@ float4 ps_main(VSOut i) : SV_Target } else if (effect > 1.5) { - rgb = lerp(blur_webcam_bg(uv_now, fx.w, quad_px, i.local), rgb, person); + rgb = lerp(blur_webcam_bg(uv_now, fx.w, quad_px, i.local, fx.xy), rgb, person); } else { alpha_mask = person; } } + + // Desk-view cover: the camera is being tilted, so the whole picture is blurred and + // dimmed (cover.x = strength, cover.y = radius in quad px, cover.z = dim at full cover). + if (cover.x > 0.001) + { + float3 hidden = blur_webcam_radius(uv_now, cover.y, quad_px, i.local, fx.xy); + rgb = lerp(rgb, hidden, cover.x) * (1.0 - cover.z * cover.x); + } } else { rgb = color.rgb; } - float alpha = color.a * alpha_mask; + float alpha = color.a * alpha_mask * persp_keep; if (radius_px > 0.0) { // `quad_px` est en px de SORTIE (le render target porte la géométrie de sortie) et @@ -4065,6 +4113,12 @@ float4 ps_main(VSOut i) : SV_Target return float4(rgb * alpha, alpha); // prémultiplié } +float4 ps_main(VSOut i) : SV_Target +{ + // Premultiplied, so the transparency scales all four channels; 0 leaves them as they are. + return ps_layer(i) * (1.0 - layer_fx.x); +} + // ============ RGB -> NV12 (§5) : deux passes vers les plans d'une texture NV12 ============ // VS plein écran (triangle unique) qui expose l'UV. struct FSOut { float4 pos : SV_Position; float2 uv : TEXCOORD0; }; diff --git a/crates/compositor/src/shaders.metal b/crates/compositor/src/shaders.metal index 7ca87d3f2..b0e2cb40e 100644 --- a/crates/compositor/src/shaders.metal +++ b/crates/compositor/src/shaders.metal @@ -48,7 +48,7 @@ using namespace metal; // ================================================================================= // // Le moteur côté CPU upload ce buffer via `setVertexBytes` (vertex stage) et -// `setFragmentBytes` (fragment stage) avant chaque draw — la copie est de 176 octets, +// `setFragmentBytes` (fragment stage) avant chaque draw — la copie est de 256 octets, // ce qui est sous le seuil d'alignement 4K de Metal pour le mode « immediate ». struct Layer @@ -66,6 +66,9 @@ struct Layer float4 trail_a; // mode 8 : coins TL, TR du plan à la frame précédente (px locaux, comme fx) ; mode 18 incliné : en fractions de sortie float4 trail_b; // mode 8 : coins BR, BL du plan à la frame précédente (comme src_prev) ; mode 18 incliné : en fractions de sortie float4 trail_mb; // mode 8 : x = taps, y = force du flou de mouvement (ceux du mode 0) ; 0 ailleurs + float4 cover; // x = desk-view cover 0..1, y = blur radius (quad px), z = dim, w unused + float4 persp[3]; // mode 0: rows of the homography quad point (0..1 in dst) -> camera uv (0..1 of the valid frame), xyz; read when layer_fx.y = 1 + float4 layer_fx; // x = transparency 0..1 (0 = opaque), every mode; y = 1 when persp applies; z, w unused }; // Mode 15 (curseur modélisé) : le détail des emplacements est dans `frame_geometry.rs`, en tête // de la section « Curseur modélisé » (`cursor_model_cb`). Mode 17 (appareil modelé) : en tête de @@ -424,12 +427,17 @@ constant float3 VOGEL_TAPS[21] = { float3(-0.633036, -0.758588, 0.087119) }; -inline float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px, - texture2d texY, - texture2d texUV) +// Vogel-disc blur of the camera texture at a radius in quad pixels. `valid` is the part of the +// texture the picture fills (fx.xy): decoders allocate aligned textures (a 1080-line camera in a +// 1088-line texture), so each tap is clamped half a chroma texel inside it, never into padding. +inline float3 blur_webcam_radius(float2 uv, float max_r_px, float2 qpx, float2 local_px, + float2 valid, + texture2d texY, + texture2d texUV) { - float max_r_px = max(intensity, 0.0) * 22.0 + 1.5; float2 step = max_r_px / max(qpx, float2(1.0)); + float2 chroma = max(float2(float(texUV.get_width()), float(texUV.get_height())), float2(1.0)); + float2 hi = max(valid - 0.5 / chroma, float2(0.0)); float noise = fract(52.9829189 * fract(0.06711056 * local_px.x + 0.00583715 * local_px.y)); float angle = noise * 6.2831853; float s = sin(angle); @@ -441,12 +449,34 @@ inline float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 loca float2 p = VOGEL_TAPS[k].xy; float w = VOGEL_TAPS[k].z; float2 rot_p = float2(p.x * c - p.y * s, p.x * s + p.y * c); - sum += sample_yuv(saturate(uv + rot_p * step), texY, texUV) * w; + sum += sample_yuv(clamp(uv + rot_p * step, float2(0.0), hi), texY, texUV) * w; total += w; } return sum / max(total, 1e-4); } +// Camera homography (`persp`, read when `layer_fx.y` = 1): the camera point (0..1 of its valid +// frame) seen at `local` (0..1 in the layer's dst quad), and in z whether there is one -- 0 when +// the point is behind the projection (q.z <= 0) or outside the camera frame, where the layer is +// transparent. Mirror of the HLSL `persp_camera`. +inline float3 persp_camera(constant Layer &layer, float2 local) +{ + float3 p = float3(local, 1.0); + float3 q = float3(dot(layer.persp[0].xyz, p), dot(layer.persp[1].xyz, p), dot(layer.persp[2].xyz, p)); + float2 cam = q.xy / max(q.z, 1e-6); + bool inside = q.z > 0.0 && all(cam >= 0.0) && all(cam <= 1.0); + return float3(cam, inside ? 1.0 : 0.0); +} + +inline float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px, + float2 valid, + texture2d texY, + texture2d texUV) +{ + return blur_webcam_radius(uv, max(intensity, 0.0) * 22.0 + 1.5, qpx, local_px, valid, texY, + texUV); +} + // Hash 2D -> [0,1) sans sin(). Miroir de `hash12` côté HLSL. inline float hash12(float2 p) { @@ -3226,22 +3256,16 @@ static float4 device_frame(float2 local, constant Layer &layer) return float4(rgb * a, a); // prémultiplié } -fragment float4 ps_main(VSOut i [[stage_in]], - constant Layer &layer [[buffer(0)]], - texture2d texY [[texture(0)]], - texture2d texUV [[texture(1)]], - texture2d texImg [[texture(2)]], - // Masque de segmentation du sujet webcam. Non lie tant qu'aucun - // masque n'existe : Metal rend alors 0, ce qui est sans effet - // puisque la branche n'est prise que si layer.fx.z > 0.5. - texture2d texMask [[texture(3)]], - // Champ de distance du sprite de curseur (mode 15 seulement), R16F, cf. - // `cursor_sdf.rs`. Le sprite lui-même est en texture(2), comme aux - // modes 7 et 13. - texture2d texSdf [[texture(4)]], - // Pyramide de profondeur de champ (`tilted_sample`), lue par le mode 8 et - // par le repli du mode 18, qui garde texture(2) pour son rendu isolé. - texture2d texDof [[texture(5)]]) +// The body of `ps_main`, which only adds the layer's transparency on top of it. Same resources +// as `ps_main`, without their bindings (MSL only allows those on an entry point). +inline float4 ps_layer(VSOut i, + constant Layer &layer, + texture2d texY, + texture2d texUV, + texture2d texImg, + texture2d texMask, + texture2d texSdf, + texture2d texDof) { // mode 18 : l'écran CADRÉ (ombre, cadre, métrage, appareil) flouté comme UN objet rigide // (`FrameGeometry::screen_trail`), port 1:1 du HLSL. texImg = son rendu isolé, prémultiplié, @@ -3732,12 +3756,25 @@ fragment float4 ps_main(VSOut i [[stage_in]], float3 rgb; // 1 sauf en detourage, ou il porte le masque du sujet. Cf. la branche fx.z plus bas. float alpha_mask = 1.0; + // 0 where a camera homography finds no camera point (`persp_camera`), 1 everywhere else. + float persp_keep = 1.0; if (layer.mode < 0.5) { // flou de mouvement par vélocité (§8) float2 uv_now = i.uv; float2 localp = (i.pout - layer.dst_prev.xy) / layer.dst_prev.zw; float2 uv_prev = layer.src_prev.xy + localp * (layer.src_prev.zw - layer.src_prev.xy); + if (layer.layer_fx.y > 0.5) + { + // Camera homography: the texture uv comes from the fragment's place in the quad + // (`src` is not used), scaled to the valid part of the decoder texture. The previous + // frame's uv is the same map at the quad's previous place. + float3 cam = persp_camera(layer, (i.pout - layer.dst.xy) / layer.dst.zw); + persp_keep = cam.z; + uv_now = cam.xy * layer.fx.xy; + float3 cam_prev = persp_camera(layer, localp); + uv_prev = (cam_prev.z > 0.5) ? cam_prev.xy * layer.fx.xy : uv_now; + } float2 duv = uv_now - uv_prev; float mb_scale = saturate(layer.mb.y); float2 duv_blur = duv * mb_scale; @@ -3772,20 +3809,29 @@ fragment float4 ps_main(VSOut i [[stage_in]], } else if (effect > 1.5) { - rgb = mix(blur_webcam_bg(uv_now, layer.fx.w, layer.quad_px, i.local, texY, texUV), rgb, person); + rgb = mix(blur_webcam_bg(uv_now, layer.fx.w, layer.quad_px, i.local, layer.fx.xy, texY, texUV), rgb, person); } else { alpha_mask = person; } } + + // Desk-view cover: the camera is being tilted, so the whole picture is blurred and + // dimmed (cover.x = strength, cover.y = radius in quad px, cover.z = dim at full cover). + if (layer.cover.x > 0.001) + { + float3 hidden = blur_webcam_radius(uv_now, layer.cover.y, layer.quad_px, i.local, layer.fx.xy, + texY, texUV); + rgb = mix(rgb, hidden, layer.cover.x) * (1.0 - layer.cover.z * layer.cover.x); + } } else { rgb = layer.color.rgb; } - float alpha = layer.color.a * alpha_mask; + float alpha = layer.color.a * alpha_mask * persp_keep; if (layer.radius_px > 0.0) { // mb.w = 1 : écran sous le chrome de fenêtre, coins HAUTS carrés et rognés par l'arc du @@ -3804,6 +3850,27 @@ fragment float4 ps_main(VSOut i [[stage_in]], return float4(rgb * alpha, alpha); } +fragment float4 ps_main(VSOut i [[stage_in]], + constant Layer &layer [[buffer(0)]], + texture2d texY [[texture(0)]], + texture2d texUV [[texture(1)]], + texture2d texImg [[texture(2)]], + // Masque de segmentation du sujet webcam. Non lie tant qu'aucun + // masque n'existe : Metal rend alors 0, ce qui est sans effet + // puisque la branche n'est prise que si layer.fx.z > 0.5. + texture2d texMask [[texture(3)]], + // Champ de distance du sprite de curseur (mode 15 seulement), R16F, cf. + // `cursor_sdf.rs`. Le sprite lui-même est en texture(2), comme aux + // modes 7 et 13. + texture2d texSdf [[texture(4)]], + // Pyramide de profondeur de champ (`tilted_sample`), lue par le mode 8 et + // par le repli du mode 18, qui garde texture(2) pour son rendu isolé. + texture2d texDof [[texture(5)]]) +{ + // Premultiplied, so the transparency scales all four channels; 0 leaves them as they are. + return ps_layer(i, layer, texY, texUV, texImg, texMask, texSdf, texDof) * (1.0 - layer.layer_fx.x); +} + // ================================================================================= // Fullscreen pass : RGB -> NV12. Mêmes shaders que la passe équivalente HLSL. // ================================================================================= diff --git a/crates/compositor/src/text_anim.rs b/crates/compositor/src/text_anim.rs index d6d3bade1..ae3887f98 100644 --- a/crates/compositor/src/text_anim.rs +++ b/crates/compositor/src/text_anim.rs @@ -4,7 +4,8 @@ //! amplitudes. Les sept animations étaient déjà nommées dans le schéma, traduites dans les treize //! langues et transportées jusqu'ici par la scène — mais rien ne les jouait. Reprendre les //! constantes du TS plutôt que d'en réinventer garantit qu'un projet fait à l'époque de l'aperçu -//! DOM s'anime toujours pareil. +//! DOM s'anime toujours pareil. Exception : l'étiquette de la vue bureau (`deskCover`), que seul +//! le compositeur joue et qui n'a pas de courbe propre (cf. `annotation_text_state`). /// Les décalages ci-dessous sont exprimés en px À CETTE HAUTEUR : l'appelant les met à l'échelle /// de la sortie, exactement comme la taille de police (cf. `annotationScale.ts`). En pixels @@ -17,6 +18,10 @@ pub const TEXT_ANIMATION_DURATION_MS: f32 = 700.0; /// fin de l'annotation est une coupe sèche alors que son arrivée est animée. pub const TEXT_EXIT_DURATION_MS: f32 = 300.0; +/// The desk-view label's animation. The app generates it for a turned Full Camera section and +/// never stores it in a document, so it exists only here, not in `annotationTextAnimation.ts`. +pub const DESK_COVER_ANIMATION: &str = "deskCover"; + #[derive(Debug, Clone, Copy, PartialEq)] pub struct TextAnimationState { pub opacity: f32, @@ -37,6 +42,11 @@ fn clamp01(v: f32) -> f32 { v.clamp(0.0, 1.0) } +fn smoothstep01(v: f32) -> f32 { + let t = clamp01(v); + t * t * (3.0 - 2.0 * t) +} + fn ease_out_cubic(v: f32) -> f32 { let t = clamp01(v); 1.0 - (1.0 - t).powi(3) @@ -106,6 +116,30 @@ pub fn text_animation_state( } } +/// `true` for the desk-view label, whose opacity is the camera cover (`annotation_text_state`). +pub fn is_desk_cover(animation: Option<&str>) -> bool { + animation == Some(DESK_COVER_ANIMATION) +} + +/// What the compositors draw a text annotation with. Ordinary animations run on +/// `text_animation_state` (source time, unchanged). The desk-view label has no timing of its +/// own: its opacity IS `cover`, the webcam's cover strength this frame +/// (`camera_fullscreen_cover_at`, already on the screen clock). A second fade derived from the +/// label's own window drifted from the cover on short sections and inside speed regions; one +/// value read twice cannot. The app spans the label over its whole section, so the cover alone +/// decides when it shows. +pub fn annotation_text_state( + animation: Option<&str>, + elapsed_ms: f32, + region_ms: f32, + cover: f32, +) -> TextAnimationState { + if is_desk_cover(animation) { + return TextAnimationState { opacity: clamp01(cover), ..TextAnimationState::IDLE }; + } + text_animation_state(animation, elapsed_ms, region_ms) +} + #[cfg(test)] mod tests { use super::*; @@ -245,4 +279,28 @@ mod tests { fn an_empty_region_does_not_divide_by_zero() { assert!(text_animation_state(Some("fade"), 0.0, 0.0).opacity.is_finite()); } + + #[test] + fn the_desk_label_takes_the_cover_and_nothing_else() { + // Whatever the label's own window says, only the cover counts; the motion fields stay idle. + for cover in [0.0, 0.21, 0.45, 1.0] { + for (elapsed, region) in [(0.0, 2600.0), (900.0, 1150.0), (5000.0, LONG)] { + let s = annotation_text_state(Some(DESK_COVER_ANIMATION), elapsed, region, cover); + assert_eq!(s, TextAnimationState { opacity: cover, ..TextAnimationState::IDLE }); + } + } + } + + #[test] + fn ordinary_animations_ignore_the_cover() { + for name in [None, Some("fade"), Some("rise"), Some("pop"), Some("typewriter"), Some("pulse")] { + for at in [0.0, 200.0, 650.0, 9_900.0] { + assert_eq!( + annotation_text_state(name, at, LONG, 0.37), + text_animation_state(name, at, LONG), + "{name:?} at {at}" + ); + } + } + } } diff --git a/crates/compositor/src/timeline_walk.rs b/crates/compositor/src/timeline_walk.rs index a16271c8b..5a1205ae0 100644 --- a/crates/compositor/src/timeline_walk.rs +++ b/crates/compositor/src/timeline_walk.rs @@ -17,13 +17,17 @@ use crate::compositor::Compositor; use crate::config::Cfg; use crate::cursor::CursorTrack; use crate::d3d::Gpu; +use crate::extra_cameras::{ + camera_source_time, cameras_in_regions, extra_camera_active, extra_camera_keys, open_once, + step_extra_camera, +}; use crate::ffi::AVFrame; use crate::frame_geometry::webcam_is_real; use crate::pipeline::{ClipSource, Decoder}; use crate::regions::{speed_segments_for_window, SpeedSegment}; -use crate::scene::Scene; +use crate::scene::{Scene, SceneCameraLayoutRegion}; use anyhow::Result; -use std::collections::HashMap; +use std::collections::{HashMap, HashSet}; /// Ce que le décodeur sait de la PROCHAINE frame, sans l'adopter. /// @@ -142,9 +146,24 @@ pub(crate) unsafe fn advance_decoder_to( /// N" defined exactly once: a GIF driven by its own loop is how the slow-motion /// truncation bug happened. /// +/// The extra cameras (k >= 1) a clip's layout regions show are opened once per path into +/// `extra_decs`, decoded only near those regions, and handed to the compositor with +/// `set_extra_camera_frames`. Unlike camera 0 they never shorten a clip, and a file that will +/// not open (or stops decoding) is skipped with a warning for the rest of the export. +/// /// `on_frame` runs after `compose_frame` with the running output index; /// `on_clip_end` runs once per clip with its clamped source window, the frames /// it produced, and the speed segments used (MP4 needs those for audio). +/// Runs its closure when dropped, so on every exit path of a scope, early `?` returns +/// included. +struct OnExit(F); + +impl Drop for OnExit { + fn drop(&mut self) { + (self.0)(); + } +} + #[allow(clippy::too_many_arguments)] pub(crate) unsafe fn walk_composited_timeline( clips: &[ClipSource], @@ -155,6 +174,7 @@ pub(crate) unsafe fn walk_composited_timeline( scene: &Option, screen_decs: &mut HashMap, webcam_decs: &mut HashMap, + extra_decs: &mut HashMap, on_frame: &mut dyn FnMut(u64) -> Result<()>, on_clip_end: &mut dyn FnMut(usize, f64, u64, &[SpeedSegment]) -> Result<()>, ) -> Result { @@ -168,6 +188,14 @@ pub(crate) unsafe fn walk_composited_timeline( let mut cursor_active_path: Option = None; let mut frames: u64 = 0; + let mut unreadable_extras: HashSet = HashSet::new(); + // The extra decoders outlive this call in the caller's map, but the compositor must not + // keep pointers into their frames once the walk is over — also when it fails half-way + // (a decode or encode error returns early through `?`). + let _forget_extra_frames = OnExit(|| { + // SAFETY: an empty list only resets the compositor's slots; no frame is read. + unsafe { comp.set_extra_camera_frames(&[]) } + }); // L'export doit être reproductible : deux rendus du même projet, les mêmes pixels. Cette // boucle avance aussi vite que la machine décode, sans rapport avec le temps réel, alors que @@ -230,17 +258,12 @@ pub(crate) unsafe fn walk_composited_timeline( clip.source_end_sec, ); } - // Les bornes de clip sont en temps écran. La disponibilité webcam est donc translatée - // par le même offset que le seek (`webcam_time = screen_time - offset`). - let webcam_available_screen_end = - webcam_available_duration.map(|duration| duration + clip.webcam_offset_sec); - let mut source_end_sec = clip.source_end_sec; - if let Some(duration) = screen_available_duration { - source_end_sec = source_end_sec.min(duration); - } - if let Some(duration) = webcam_available_screen_end { - source_end_sec = source_end_sec.min(duration); - } + let source_end_sec = available_clip_end( + clip.source_end_sec, + screen_available_duration, + webcam_available_duration, + clip.webcam_offset_sec, + ); if source_end_sec + 1e-6 < clip.source_end_sec { eprintln!( "[pipeline] warning: clip #{} raccourci de {:.3}s (fin demandée {:.3}s, fin disponible {:.3}s; screen=\"{}\", webcam=\"{}\")", @@ -268,6 +291,27 @@ pub(crate) unsafe fn walk_composited_timeline( source_end_sec, out_fps as f64, ); + // Only the extra cameras this clip's own regions show, and only those it has a file + // for. `None` slots stay null in `set_extra_camera_frames`. + let camera_regions = clip_scene + .as_ref() + .map(|s| s.camera_layout_regions.clone()) + .unwrap_or_default(); + let mut extras: Vec> = + extra_camera_keys(&cameras_in_regions(&camera_regions), &clip.additional_cameras) + .into_iter() + .map(|key| { + let key = key?; + open_once(extra_decs, &mut unreadable_extras, &key.path, |path| { + Decoder::open_for_export(path, gpu) + }) + .then_some(ExportExtraCamera { + path: key.path, + offset_sec: key.offset_sec, + ended_at: None, + }) + }) + .collect(); if clip_scene.is_some() { comp.set_scene(clip_scene); } @@ -328,6 +372,24 @@ pub(crate) unsafe fn walk_composited_timeline( break 'clip_frames; } + let (extra_frames, closed) = { + let _p = crate::export_probe::scope(crate::export_probe::Stage::DecodeWebcam); + step_extra_cameras( + &mut extras, + extra_decs, + &mut unreadable_extras, + &camera_regions, + target_source_time, + ) + }; + if closed { + // A dropped decoder's textures may still sit in the SRV cache; a later + // decoder could reuse their addresses (same contract as the preview's + // `hand_extra_frames`). + comp.clear_srv_cache(); + } + comp.set_extra_camera_frames(&extra_frames); + comp.set_timeline_time(Some(target_source_time as f32)); // Le temps de SORTIE, lui, ne saute ni aux coupes ni aux clips. Même // arithmétique (f64 puis f32) que `ProgrammeClock::at` côté preview. @@ -361,9 +423,104 @@ pub(crate) unsafe fn walk_composited_timeline( Ok(frames) } +/// The screen-time end of a clip: the requested end, clamped to what the screen and camera 0 +/// can deliver. Clip bounds are in screen time, so camera 0's duration is moved by the same +/// offset as its seek (`webcam_time = screen_time - offset`). The extra cameras do not enter +/// here: one that ends early only loses its layer, it never shortens the clip. +fn available_clip_end( + requested_end_sec: f64, + screen_duration: Option, + webcam_duration: Option, + webcam_offset_sec: f64, +) -> f64 { + let mut end = requested_end_sec; + if let Some(duration) = screen_duration { + end = end.min(duration); + } + if let Some(duration) = webcam_duration { + end = end.min(duration + webcam_offset_sec); + } + end +} + +/// An extra camera of the clip being exported; its decoder lives in `extra_decs` under `path`. +struct ExportExtraCamera { + path: String, + offset_sec: f64, + /// See `step_extra_camera`. + ended_at: Option, +} + +/// The frames for `set_extra_camera_frames` at screen source time `t`: each camera that is +/// near one of its regions steps toward its own source time, the others give null. A camera +/// that stops decoding is dropped with a warning and not reopened for the rest of the export; +/// the `bool` says one was dropped, so the caller clears the compositor's texture cache. +unsafe fn step_extra_cameras( + extras: &mut [Option], + extra_decs: &mut HashMap, + unreadable: &mut HashSet, + regions: &[SceneCameraLayoutRegion], + t: f64, +) -> (Vec<*const AVFrame>, bool) { + let mut closed = false; + let mut frames = Vec::with_capacity(extras.len()); + for (k, slot) in extras.iter_mut().enumerate() { + let camera = k + 1; + let mut frame: *const AVFrame = std::ptr::null(); + if let Some(cam) = slot.as_mut().filter(|_| extra_camera_active(regions, camera, t)) { + if let Some(dec) = extra_decs.get_mut(&cam.path) { + let target = camera_source_time(t, cam.offset_sec); + match step_extra_camera(dec, &mut cam.ended_at, target) { + Ok(f) => frame = f, + Err(e) => { + eprintln!( + "WARNING: extra camera {camera} stopped decoding ({}): {e:#}. Its layer will not be drawn.", + cam.path + ); + extra_decs.remove(&cam.path); + unreadable.insert(cam.path.clone()); + *slot = None; + closed = true; + } + } + } + } + frames.push(frame); + } + (frames, closed) +} + #[cfg(test)] mod tests { - use super::{frame_step, FrameStep, NextFrameTime}; + use super::{available_clip_end, frame_step, FrameStep, NextFrameTime, OnExit}; + + /// The guard that forgets the extra camera frames runs on an early `?` return as well as + /// at the end of the walk. + #[test] + fn the_exit_guard_runs_on_every_exit_path() { + fn walk(fail: bool, ran: &std::cell::Cell) -> anyhow::Result<()> { + let _guard = OnExit(|| ran.set(ran.get() + 1)); + if fail { + Err(anyhow::anyhow!("decode error"))?; + } + Ok(()) + } + let ran = std::cell::Cell::new(0); + assert!(walk(true, &ran).is_err()); + assert_eq!(ran.get(), 1, "early return"); + assert!(walk(false, &ran).is_ok()); + assert_eq!(ran.get(), 2, "normal end"); + } + + #[test] + fn a_short_extra_camera_does_not_shorten_the_clip() { + // Screen 60 s, camera 0 50 s at offset 2 s: the clip ends where camera 0 does. + assert_eq!(available_clip_end(60.0, Some(60.0), Some(50.0), 2.0), 52.0); + // A 10 s extra camera in the same clip changes nothing: only the screen and camera 0 + // bound the clip, the extra camera's layer just disappears past its end. + assert_eq!(available_clip_end(30.0, Some(60.0), Some(50.0), 2.0), 30.0); + assert_eq!(available_clip_end(30.0, None, None, 0.0), 30.0); + } /// La cadence de lecture, en une phrase : à 24 fps, une seconde réelle doit adopter 24 /// frames et pas une de plus. Le bug d'origine (un pas fixe de 1/60 s, une frame par diff --git a/crates/compositor/src/vk_shaders/layer.wgsl b/crates/compositor/src/vk_shaders/layer.wgsl index ecf4960a1..6d745cd13 100644 --- a/crates/compositor/src/vk_shaders/layer.wgsl +++ b/crates/compositor/src/vk_shaders/layer.wgsl @@ -40,6 +40,9 @@ struct Layer { trail_a: vec4, // mode 8 : coins TL, TR du plan a la frame precedente (px locaux, comme fx) ; mode 18 incline : en fractions de sortie trail_b: vec4, // mode 8 : coins BR, BL du plan a la frame precedente (comme src_prev) ; mode 18 incline : en fractions de sortie trail_mb: vec4, // mode 8 : x = taps, y = force du flou de mouvement (ceux du mode 0) ; mode 18 incline : le `mb` du mode 8 (profondeur de champ), et `color.xy` sa lampe ; 0 ailleurs + cover: vec4, // x = desk-view cover 0..1, y = blur radius (quad px), z = dim, w unused + persp: array, 3>, // mode 0: rows of the homography quad point (0..1 in dst) -> camera uv (0..1 of the valid frame), xyz; read when layer_fx.y = 1 + layer_fx: vec4, // x = transparency 0..1 (0 = opaque), every mode; y = 1 when persp applies; z, w unused } @group(0) @binding(0) var layer: Layer; @@ -560,9 +563,13 @@ const VOGEL_TAPS = array, 21>( vec3(-0.633036, -0.758588, 0.087119) ); -fn blur_webcam_bg(uv: vec2, intensity: f32, qpx: vec2, local_px: vec2) -> vec3 { - let max_r_px = max(intensity, 0.0) * 22.0 + 1.5; +// Vogel-disc blur of the camera texture at a radius in quad pixels. `valid` is the part of the +// texture the picture fills (fx.xy): decoders allocate aligned textures (a 1080-line camera in a +// 1088-line texture), so each tap is clamped half a chroma texel inside it, never into padding. +fn blur_webcam_radius(uv: vec2, max_r_px: f32, qpx: vec2, local_px: vec2, valid: vec2) -> vec3 { let step = max_r_px / max(qpx, vec2(1.0)); + let chroma = max(vec2(textureDimensions(texU)), vec2(1.0)); + let hi = max(valid - 0.5 / chroma, vec2(0.0)); let noise = fract(52.9829189 * fract(0.06711056 * local_px.x + 0.00583715 * local_px.y)); let angle = noise * 6.2831853; let s = sin(angle); @@ -573,12 +580,28 @@ fn blur_webcam_bg(uv: vec2, intensity: f32, qpx: vec2, local_px: vec2< let p = VOGEL_TAPS[k].xy; let w = VOGEL_TAPS[k].z; let rot_p = vec2(p.x * c - p.y * s, p.x * s + p.y * c); - sum = sum + sample_yuv(clamp(uv + rot_p * step, vec2(0.0), vec2(1.0))) * w; + sum = sum + sample_yuv(clamp(uv + rot_p * step, vec2(0.0), hi)) * w; total = total + w; } return sum / max(total, 1e-4); } +fn blur_webcam_bg(uv: vec2, intensity: f32, qpx: vec2, local_px: vec2, valid: vec2) -> vec3 { + return blur_webcam_radius(uv, max(intensity, 0.0) * 22.0 + 1.5, qpx, local_px, valid); +} + +// Camera homography (`persp`, read when `layer_fx.y` = 1): the camera point (0..1 of its valid +// frame) seen at `local` (0..1 in the layer's dst quad), and in z whether there is one -- 0 when +// the point is behind the projection (q.z <= 0) or outside the camera frame, where the layer is +// transparent. Mirror of the HLSL and MSL `persp_camera`. +fn persp_camera(local: vec2) -> vec3 { + let p = vec3(local, 1.0); + let q = vec3(dot(layer.persp[0].xyz, p), dot(layer.persp[1].xyz, p), dot(layer.persp[2].xyz, p)); + let cam = q.xy / max(q.z, 1e-6); + let inside = q.z > 0.0 && all(cam >= vec2(0.0)) && all(cam <= vec2(1.0)); + return vec3(cam, select(0.0, 1.0, inside)); +} + // ---- Curseur MODELISE (mode 15) ---- // Port ligne pour ligne de `cursor_model` (HLSL), dont les commentaires font foi : un curseur // SCULPTE (`trail_a.x` > 0, la fleche ou la main d'un des cinq themes d'origine modelee en @@ -2256,12 +2279,14 @@ fn device_frame(local: vec2) -> vec4 { return vec4(rgb * a, a); // premultiplie } -@fragment -fn fs_main(i: VsOut) -> @location(0) vec4 { +// The body of `fs_main`, which only adds the layer's transparency on top of it. +fn fs_layer(i: VsOut) -> vec4 { var rgb: vec3; var alpha: f32; // 1 sauf en detourage, ou il porte le masque du sujet. Cf. la branche fx.z plus bas. var alpha_mask = 1.0; + // 0 where a camera homography finds no camera point (`persp_camera`), 1 everywhere else. + var persp_keep = 1.0; if layer.mode < 0.5 { // Mode 0 — vidéo NV12 + flou de mouvement par vélocité (§8), port 1:1 du @@ -2271,18 +2296,31 @@ fn fs_main(i: VsOut) -> @location(0) vec4 { // du calque sans avoir à transporter un champ de vitesse. let taps = i32(layer.mb.x); let mb_scale = clamp(layer.mb.y, 0.0, 1.0); + var uv_now = i.uv; + if layer.layer_fx.y > 0.5 { + // Camera homography: the texture uv comes from the fragment's place in the quad + // (`src` is not used), scaled to the valid part of the decoder texture. + let cam = persp_camera((i.pout - layer.dst.xy) / layer.dst.zw); + persp_keep = cam.z; + uv_now = cam.xy * layer.fx.xy; + } // `taps` d'abord : un draw qui a oublié `dst_prev` le laisse à zéro, et // la division par `dst_prev.zw` produirait des UV infinis. Dégrader vers // le chemin net est le seul échec acceptable pour un effet cosmétique. if taps <= 1 || mb_scale <= 0.001 || layer.dst_prev.z <= 0.0 || layer.dst_prev.w <= 0.0 { - rgb = sample_yuv(i.uv); + rgb = sample_yuv(uv_now); } else { let localp = (i.pout - layer.dst_prev.xy) / layer.dst_prev.zw; - let uv_prev = layer.src_prev.xy + localp * (layer.src_prev.zw - layer.src_prev.xy); - let duv = i.uv - uv_prev; + var uv_prev = layer.src_prev.xy + localp * (layer.src_prev.zw - layer.src_prev.xy); + if layer.layer_fx.y > 0.5 { + // The previous frame's uv is the same map at the quad's previous place. + let cam_prev = persp_camera(localp); + uv_prev = select(uv_now, cam_prev.xy * layer.fx.xy, cam_prev.z > 0.5); + } + let duv = uv_now - uv_prev; let duv_blur = duv * mb_scale; if dot(duv_blur, duv_blur) < 1e-9 { - rgb = sample_yuv(i.uv); + rgb = sample_yuv(uv_now); } else { // Borne 16 en dur, identique au HLSL et au MSL : `taps` vient d'un // uniform et une boucle sans borne statique ne se déroule pas. @@ -2292,7 +2330,7 @@ fn fs_main(i: VsOut) -> @location(0) vec4 { let step = 1.0 / f32(taps - 1); for (var k: i32 = 0; k < 16; k = k + 1) { if k >= taps { break; } - acc = acc + sample_yuv(i.uv - duv_blur * (1.0 - f32(k) * step)); + acc = acc + sample_yuv(uv_now - duv_blur * (1.0 - f32(k) * step)); } rgb = acc / f32(taps); } @@ -2303,16 +2341,23 @@ fn fs_main(i: VsOut) -> @location(0) vec4 { // l'etendue valide de la texture webcam pour ramener uv dans l'espace du masque. let effect = layer.fx.z; if effect > 0.5 { - let mask_uv = i.uv / max(layer.fx.xy, vec2(1e-6)); + let mask_uv = uv_now / max(layer.fx.xy, vec2(1e-6)); let person = clamp(textureSample(texMask, samp, mask_uv).r, 0.0, 1.0); if effect > 2.5 { rgb = mix(layer.color.rgb, rgb, person); } else if effect > 1.5 { - rgb = mix(blur_webcam_bg(i.uv, layer.fx.w, layer.quad_px, i.local), rgb, person); + rgb = mix(blur_webcam_bg(uv_now, layer.fx.w, layer.quad_px, i.local, layer.fx.xy), rgb, person); } else { alpha_mask = person; } } + + // Desk-view cover: the camera is being tilted, so the whole picture is blurred and + // dimmed (cover.x = strength, cover.y = radius in quad px, cover.z = dim at full cover). + if layer.cover.x > 0.001 { + let hidden = blur_webcam_radius(uv_now, layer.cover.y, layer.quad_px, i.local, layer.fx.xy); + rgb = mix(rgb, hidden, layer.cover.x) * (1.0 - layer.cover.z * layer.cover.x); + } } else if layer.mode < 1.5 { // Mode 1 — couleur pleine. rgb = layer.color.rgb; @@ -2685,7 +2730,7 @@ fn fs_main(i: VsOut) -> @location(0) vec4 { if layer.mode > 4.5 && layer.mode < 5.5 { base_alpha = 1.0; } - alpha = base_alpha * alpha_mask; + alpha = base_alpha * alpha_mask * persp_keep; if layer.radius_px > 0.0 { // Feather ~1.5 px sur le bord du quad — parité exacte avec le HLSL @@ -2708,3 +2753,9 @@ fn fs_main(i: VsOut) -> @location(0) vec4 { return vec4(rgb * alpha, alpha); // alpha prémultiplié } + +@fragment +fn fs_main(i: VsOut) -> @location(0) vec4 { + // Premultiplied, so the transparency scales all four channels; 0 leaves them as they are. + return fs_layer(i) * (1.0 - layer.layer_fx.x); +} diff --git a/crates/compositor/tests/compose_linux.rs b/crates/compositor/tests/compose_linux.rs index c8bc45936..1c74e011e 100644 --- a/crates/compositor/tests/compose_linux.rs +++ b/crates/compositor/tests/compose_linux.rs @@ -799,6 +799,7 @@ fn export_linux_mp4() { source_end_sec: 1.0, webcam_offset_sec: 0.0, has_audio: true, + additional_cameras: Vec::new(), }]; let params = ExportParams { width: 640, diff --git a/crates/compositor/tests/export_timing.rs b/crates/compositor/tests/export_timing.rs index 34d5e8de4..adb90bb43 100644 --- a/crates/compositor/tests/export_timing.rs +++ b/crates/compositor/tests/export_timing.rs @@ -117,6 +117,7 @@ fn whole_clip(dir: &PathBuf) -> ClipSource { source_end_sec: SOURCE_SEC, webcam_offset_sec: 0.0, has_audio: false, + additional_cameras: Vec::new(), } } diff --git a/crates/poc-d3d/src/bench.rs b/crates/poc-d3d/src/bench.rs index 3da91883b..2b657b91a 100644 --- a/crates/poc-d3d/src/bench.rs +++ b/crates/poc-d3d/src/bench.rs @@ -154,6 +154,7 @@ fn run_bench(args: &[String]) -> Result<()> { source_end_sec: 6.0, // la fixture entière (§ fixture.json : 6 s, 360 frames) webcam_offset_sec: 0.0, has_audio: false, + additional_cameras: Vec::new(), }; let path = format!("{out}/{}_{:?}.mp4", cfg.name, backend).to_lowercase(); let s = pipeline::run_composited_multi( @@ -296,6 +297,7 @@ fn run_gif_bench( source_end_sec: f64::MAX, webcam_offset_sec: 0.0, has_audio: false, + additional_cameras: Vec::new(), }]; for r in 0..repeat { // Each run writes to the same path — the last frame wins. The diff --git a/electron/ai-edition/agent-tools.test.ts b/electron/ai-edition/agent-tools.test.ts index fc8ba4085..f612c92eb 100644 --- a/electron/ai-edition/agent-tools.test.ts +++ b/electron/ai-edition/agent-tools.test.ts @@ -1211,6 +1211,96 @@ describe("a full-camera region needs a camera", () => { }); }); +describe("full-camera regions and layout sections share one lane", () => { + /** A side-by-side section of cameras 1 and 2 over 10–15 s of clip_1. */ + function withLayoutSection(document: AxcutDocument): AxcutDocument { + return documentSchema.parse({ + ...document, + legacyEditor: { + ...((document.legacyEditor as Record) ?? {}), + cameraLayoutRegions: [ + { + id: "camlayout_1", + startMs: 10_000, + endMs: 15_000, + clipId: "clip_1", + assetId: "asset_1", + sourceStartSec: 10, + sourceEndSec: 15, + template: "side-by-side", + slots: [{ camera: 0 }, { camera: 1 }], + }, + ], + }, + }); + } + + it("exposes layout sections read-only in the snapshot", () => { + const document = withLayoutSection(withCameraTrack(fixtureDocument())); + const snapshot = JSON.parse(executeAgentTool(document, "getCurrentDocument", "").resultJson); + expect(snapshot.cameraLayoutRegions).toEqual([ + { id: "camlayout_1", startSec: 10, endSec: 15, template: "side-by-side", cameras: [1, 2] }, + ]); + expect(snapshot.cameraLayoutNote).toMatch(/read-only/); + expect(snapshot.timeBaseNote).toMatch(/cameraLayoutRegions/); + }); + + it("an empty project reports no layout sections", () => { + const snapshot = JSON.parse( + executeAgentTool(fixtureDocument(), "getCurrentDocument", "").resultJson, + ); + expect(snapshot.cameraLayoutRegions).toEqual([]); + }); + + it("refuses to add a full-camera region over a layout section", () => { + const document = withLayoutSection(withCameraTrack(fixtureDocument())); + const result = executeAgentTool( + document, + "addCameraFullscreen", + JSON.stringify({ startSec: 12, endSec: 18 }), + ); + expect(result.ok).toBe(false); + expect(result.document).toBeUndefined(); + const error = JSON.parse(result.resultJson).error as string; + expect(error).toMatch(/camlayout_1/); + expect(error).toMatch(/side-by-side, 10–15 s/); + expect(error).toMatch(/may not overlap/); + }); + + it("allows a full-camera region that only touches a layout section", () => { + const document = withLayoutSection(withCameraTrack(fixtureDocument())); + const result = executeAgentTool( + document, + "addCameraFullscreen", + JSON.stringify({ startSec: 5, endSec: 10 }), + ); + expect(result.ok).toBe(true); + }); + + it("refuses to move a full-camera region onto a layout section", () => { + const document = withLayoutSection(withCameraTrack(fixtureDocument())); + const added = executeAgentTool( + document, + "addCameraFullscreen", + JSON.stringify({ startSec: 0, endSec: 5 }), + ); + expect(added.ok).toBe(true); + const regionId = ( + (added.document?.legacyEditor as Record).cameraFullscreenRegions as Array<{ + id: string; + }> + )[0].id; + const moved = executeAgentTool( + added.document as AxcutDocument, + "setCameraFullscreen", + JSON.stringify({ cameraFullscreenId: regionId, startSec: 8, endSec: 11 }), + ); + expect(moved.ok).toBe(false); + expect(moved.document).toBeUndefined(); + expect(JSON.parse(moved.resultJson).error).toMatch(/camlayout_1/); + }); +}); + // ── D-DESTRUCT ────────────────────────────────────────────────────────────── // // "Swap the two clips: put the demo first." There was no tool for it — while diff --git a/electron/ai-edition/agent-tools.ts b/electron/ai-edition/agent-tools.ts index 3ffc9b215..efc2444d8 100644 --- a/electron/ai-edition/agent-tools.ts +++ b/electron/ai-edition/agent-tools.ts @@ -57,6 +57,10 @@ import { effectiveZoomScale, ZOOM_DEPTH_LEGEND, } from "../../src/lib/ai-edition/timeline/zoom-scale"; +import { + cameraSectionsOverlapping, + normalizeCameraLayoutRegions, +} from "../../src/lib/cameraLayouts"; import { SETTING_BOUNDS } from "../../src/lib/projectDefaults"; export interface AgentToolExecution { @@ -97,6 +101,10 @@ function modifierIds(document: AxcutDocument, kind: ModifierKind): string[] { return ((legacy.cameraFullscreenRegions as Array<{ id: string }> | undefined) ?? []).map( (region) => region.id, ); + case "cameraLayout": + return ((legacy.cameraLayoutRegions as Array<{ id: string }> | undefined) ?? []).map( + (region) => region.id, + ); case "audio": // Fragments of one user-visible track share `trackId` and render as a single pill, // so the id that disappears is the group key once, not one per fragment. @@ -265,6 +273,39 @@ function noCameraUnderSpan( ); } +/** The editor's file the agent reads layout sections from, coalesced to the pills the ruler draws. */ +function cameraLayoutPillsForAgent(document: AxcutDocument) { + const legacy = document.legacyEditor as Record | null; + return coalesceForAgent(normalizeCameraLayoutRegions(legacy?.cameraLayoutRegions)); +} + +/** + * Refuse a full-camera region that would land on a camera layout section. Both share one + * lane on the timeline and may never overlap (the editor refuses the same add as + * "occupied"); two overlapping sections would leave the scene to drop one of them. + * Layout sections are not editable from here, so the message names the one in the way. + */ +function cameraLayoutInTheWay( + document: AxcutDocument, + startSec: number, + endSec: number, +): AgentToolExecution | null { + const blocking = cameraSectionsOverlapping( + cameraLayoutPillsForAgent(document), + toMs(startSec), + toMs(endSec), + )[0]; + if (!blocking) return null; + return failure( + `The span ${startSec.toFixed(1)}–${endSec.toFixed(1)} s overlaps the camera layout ` + + `section ${blocking.id} (${blocking.template}, ${roundSec(blocking.startMs)}–` + + `${roundSec(blocking.endMs)} s), so no full-camera region was written. Full-camera ` + + "regions and layout sections share one camera lane and may not overlap. Pick a span " + + "outside every entry of cameraLayoutRegions in getCurrentDocument, or ask the user " + + "to change the layout section in the editor.", + ); +} + /** The span the edited timeline actually occupies, for an actionable refusal. */ function editedExtentSec(document: AxcutDocument): { startSec: number; endSec: number } { const clips = document.timeline.clips; @@ -335,18 +376,21 @@ function landingSuffix( return parts.length ? ` (${parts.join(", ")})` : ""; } -/** Ids of every modifier in the document, all four families at once — the basis +/** Ids of every modifier in the document, every family at once — the basis * for naming what a destructive edit took with it. */ function modifierIdsOf(document: AxcutDocument): string[] { const legacy = (document.legacyEditor as Record) ?? {}; const speedRegions = (legacy.speedRegions as Array<{ id: string }> | undefined) ?? []; const cameraFullscreenRegions = (legacy.cameraFullscreenRegions as Array<{ id: string }> | undefined) ?? []; + const cameraLayoutRegions = + (legacy.cameraLayoutRegions as Array<{ id: string }> | undefined) ?? []; return [ ...document.zoomRanges.map((r) => r.id), ...document.annotations.map((r) => r.id), ...speedRegions.map((r) => r.id), ...cameraFullscreenRegions.map((r) => r.id), + ...cameraLayoutRegions.map((r) => r.id), ]; } @@ -758,7 +802,11 @@ export function documentSnapshotForModel( const autoFocusAll = legacy?.autoFocusAll === true; return { timeBaseNote: - "clips and trims are in source-time seconds; zooms, speedRegions, annotations, cameraFullscreenRegions and audioTracks are in virtual (edited-timeline) seconds.", + "clips and trims are in source-time seconds; zooms, speedRegions, annotations, cameraFullscreenRegions, cameraLayoutRegions and audioTracks are in virtual (edited-timeline) seconds.", + cameraLayoutNote: + "cameraLayoutRegions are multi-camera layout sections the user placed in the editor; they are read-only here. " + + "They share one camera lane with cameraFullscreenRegions and the two may never overlap, so addCameraFullscreen / setCameraFullscreen refuse a span that lands on one. " + + "cameras lists the cameras shown, in place order, numbered as the user sees them (1 = the recording's main webcam).", audioNote: "audioTracks are imported voiceover / music files laid over the recording. They are clip-anchored like every other region, so they travel with their clip through reorder and trim, and they play at 1x whatever a speed region does to the picture under them. addAudio places an EXISTING asset of kind 'audio'; nothing here can import a file from disk or record one, so if the project has no audio asset, say so rather than inventing an id.", zoomNote: @@ -854,6 +902,13 @@ export function documentSnapshotForModel( startSec: roundSec(c.startMs), endSec: roundSec(c.endMs), })), + cameraLayoutRegions: cameraLayoutPillsForAgent(document).map((l) => ({ + id: l.id, + startSec: roundSec(l.startMs), + endSec: roundSec(l.endMs), + template: l.template, + cameras: l.slots.map((slot) => slot.camera + 1), + })), // Imported audio, collapsed to the pills the ruler draws — a track ventilated // across a clip boundary is several fragments the user sees as one thing, and // the model has to name what the user sees. @@ -2034,6 +2089,8 @@ export function executeAgentTool( } const blind = noCameraUnderSpan(document, landing.startSec, landing.endSec); if (blind) return blind; + const occupied = cameraLayoutInTheWay(document, landing.startSec, landing.endSec); + if (occupied) return occupied; const next: AxcutDocument = { ...document, legacyEditor: { ...legacy, cameraFullscreenRegions: [...prev, ...placed] }, @@ -2078,6 +2135,8 @@ export function executeAgentTool( } const blindMove = noCameraUnderSpan(document, landing.startSec, landing.endSec); if (blindMove) return blindMove; + const occupiedMove = cameraLayoutInTheWay(document, landing.startSec, landing.endSec); + if (occupiedMove) return occupiedMove; const next: AxcutDocument = { ...document, legacyEditor: { ...legacy, cameraFullscreenRegions: rebuiltCamera }, diff --git a/electron/ai-edition/deep-agent/service.ts b/electron/ai-edition/deep-agent/service.ts index 21f5b79a5..90a1b815f 100644 --- a/electron/ai-edition/deep-agent/service.ts +++ b/electron/ai-edition/deep-agent/service.ts @@ -177,9 +177,9 @@ export const TOOL_DESCRIPTIONS: Record = { setAnnotation: "Move, resize, or edit the text of an existing annotation by id (virtual-timeline seconds). Only the fields you pass are changed.", addCameraFullscreen: - "Add a camera-fullscreen region over a span of the edited timeline (virtual seconds): the webcam fills the frame for that span. This only does something when the footage under that span comes from an asset with a linked webcam — check assets[].hasCameraTrack (or hasAnyCamera) in getCurrentDocument first. On footage with no camera the call is refused rather than storing a region that would render nothing; say so instead of retrying.", + "Add a camera-fullscreen region over a span of the edited timeline (virtual seconds): the webcam fills the frame for that span. This only does something when the footage under that span comes from an asset with a linked webcam — check assets[].hasCameraTrack (or hasAnyCamera) in getCurrentDocument first. On footage with no camera the call is refused rather than storing a region that would render nothing; say so instead of retrying. It is also refused over a multi-camera layout section (cameraLayoutRegions): the two share one lane and may not overlap.", setCameraFullscreen: - "Move or resize an existing camera-fullscreen region by id (virtual-timeline seconds). Only the fields you pass are changed. Refused if the new span lands on footage with no linked webcam.", + "Move or resize an existing camera-fullscreen region by id (virtual-timeline seconds). Only the fields you pass are changed. Refused if the new span lands on footage with no linked webcam or on a layout section (cameraLayoutRegions).", addAudio: "Lay an ALREADY-IMPORTED audio file over the recording across a span of the edited timeline (virtual seconds): a voiceover, or a music bed. assetId must name an asset whose kind is 'audio' — getCurrentDocument lists them; nothing here can import a file from disk or record one, so if there is none, say so instead of guessing an id. Omit endSec to play the whole file from offsetSec. kind picks the lane ('voiceover' or 'music'). offsetSec is where in the FILE playback starts, gainDb its level (0 unchanged, negative ducks it). A voiceover-lane track is also what gets transcribed, so the lane is not only cosmetic.", setAudio: diff --git a/electron/app-settings.test.ts b/electron/app-settings.test.ts index f3781734f..f7590fdb6 100644 --- a/electron/app-settings.test.ts +++ b/electron/app-settings.test.ts @@ -165,4 +165,59 @@ describe("app settings store", () => { expect(new AppSettingsStore(dir).getSnapshot().recording.camQuality).toBe("2160p"); }); + + it("reads a missing additional camera list as empty", () => { + const dir = temp(); + writeFileSync( + path.join(dir, "recording-settings.json"), + JSON.stringify({ camEnabled: true }), + "utf8", + ); + + expect(new AppSettingsStore(dir).getSnapshot().recording.camAdditionalDevices).toEqual([]); + }); + + it("drops additional camera entries without a name and keeps at most three", () => { + const store = new AppSettingsStore(temp()); + store.setRecordingPreferences({ + camAdditionalDevices: [ + { id: "a", name: "A" }, + { id: null, name: "B" }, + { id: "x", name: "" }, + { id: "c", name: "C" }, + { id: "d", name: "D" }, + { id: "e", name: "E" }, + ], + }); + + expect(store.getSnapshot().recording.camAdditionalDevices).toEqual([ + { id: "a", name: "A" }, + { id: null, name: "B" }, + { id: "c", name: "C" }, + ]); + }); + + it("ignores junk in a stored additional camera list", () => { + const dir = temp(); + writeFileSync( + path.join(dir, "recording-settings.json"), + JSON.stringify({ + camAdditionalDevices: [null, 3, "x", { name: 5 }, { id: 7, name: "Kept" }, { name: "Ok" }], + }), + "utf8", + ); + + expect(new AppSettingsStore(dir).getSnapshot().recording.camAdditionalDevices).toEqual([ + { id: null, name: "Kept" }, + { id: null, name: "Ok" }, + ]); + }); + + it("rejects an additional camera list that is not a list", () => { + const store = new AppSettingsStore(temp()); + + expect(() => store.setRecordingPreferences({ camAdditionalDevices: "x" as never })).toThrow( + TypeError, + ); + }); }); diff --git a/electron/app-settings.ts b/electron/app-settings.ts index bef025bf3..d1efb9aae 100644 --- a/electron/app-settings.ts +++ b/electron/app-settings.ts @@ -15,6 +15,8 @@ export interface RecordingPreferences { camEnabled: boolean; camDeviceId: string | null; camDeviceName: string | null; + /** Cameras 2-4 of a native Windows recording, in the order they were picked. At most three. */ + camAdditionalDevices: Array<{ id: string | null; name: string }>; /** Capture resolution for the camera. See WEBCAM_QUALITY_PRESETS. */ camQuality: WebcamQualityId; systemAudioEnabled: boolean; @@ -38,6 +40,7 @@ export const DEFAULT_RECORDING_PREFERENCES: RecordingPreferences = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: DEFAULT_WEBCAM_QUALITY, systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", @@ -101,6 +104,22 @@ const bool = (value: unknown, fallback: boolean) => (typeof value === "boolean" const nullableString = (value: unknown, fallback: string | null) => value === null || typeof value === "string" ? value : fallback; +const MAX_ADDITIONAL_CAMERAS = 3; + +/** Keeps the entries that name a camera, in order, up to the cap; everything else is junk. */ +function additionalCameras(value: unknown): RecordingPreferences["camAdditionalDevices"] { + if (!Array.isArray(value)) return []; + const kept: RecordingPreferences["camAdditionalDevices"] = []; + for (const entry of value) { + if (kept.length >= MAX_ADDITIONAL_CAMERAS) break; + if (!entry || typeof entry !== "object") continue; + const { id, name } = entry as Record; + if (typeof name !== "string" || name.length === 0) continue; + kept.push({ id: typeof id === "string" ? id : null, name }); + } + return kept; +} + function parseRecording(raw: RawSettings): RecordingPreferences { return { micEnabled: bool(raw.micEnabled, DEFAULT_RECORDING_PREFERENCES.micEnabled), @@ -109,6 +128,7 @@ function parseRecording(raw: RawSettings): RecordingPreferences { camEnabled: bool(raw.camEnabled, DEFAULT_RECORDING_PREFERENCES.camEnabled), camDeviceId: nullableString(raw.camDeviceId, DEFAULT_RECORDING_PREFERENCES.camDeviceId), camDeviceName: nullableString(raw.camDeviceName, DEFAULT_RECORDING_PREFERENCES.camDeviceName), + camAdditionalDevices: additionalCameras(raw.camAdditionalDevices), // Unset in every settings file written before the camera had a quality // setting, and `webcamQualityFrom` answers those with the default. camQuality: webcamQualityFrom(raw.camQuality), @@ -169,6 +189,10 @@ function validateRecordingPatch(patch: Partial): void { if ((key.endsWith("Enabled") || key === "hideDesktopIcons") && typeof value !== "boolean") { throw new TypeError(`${key} must be a boolean`); } + if (key === "camAdditionalDevices") { + if (!Array.isArray(value)) throw new TypeError("camAdditionalDevices must be a list"); + continue; + } if ( (key.endsWith("DeviceId") || key.endsWith("DeviceName")) && value !== null && @@ -207,6 +231,9 @@ export class AppSettingsStore { const next = Object.fromEntries( Object.entries(patch).filter(([, value]) => value !== undefined), ) as Partial; + if (next.camAdditionalDevices) { + next.camAdditionalDevices = additionalCameras(next.camAdditionalDevices); + } atomicWrite(this.userData, { ...raw, ...current, ...next }); return this.getSnapshot(); } diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index 26f2f5308..1c28c8b4f 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -177,6 +177,17 @@ interface Window { * saved without it. Still a success — the screen video is intact. */ webcamDropped?: boolean; + /** + * Labels (device name, else `Camera `) of additional cameras (2-4) that + * produced nothing usable and were left out of the session. Camera 1 is + * `webcamDropped`. + */ + droppedWebcams?: string[]; + /** + * Labels of cameras (camera 1 included) that the helper disabled mid-take. + * Their partial files are kept in the session; the user is told which. + */ + webcamsStoppedEarly?: string[]; }>; pauseNativeWindowsRecording: () => Promise<{ success: boolean; @@ -353,12 +364,9 @@ interface Window { /** Why this recording ended before it was stopped, when it did. */ warning?: string; }>; - findRecordingCamera: (videoPath: string) => Promise<{ - success: boolean; - webcamVideoPath?: string; - offsetMs?: number; - error?: string; - }>; + findRecordingCamera: ( + videoPath: string, + ) => Promise; readBinaryFile: (filePath: string) => Promise<{ success: boolean; data?: ArrayBuffer; diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 81e706c08..8270c69da 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -34,7 +34,9 @@ import { } from "../../src/lib/nativeMacRecording"; import type { NativeWindowsRecordingRequest } from "../../src/lib/nativeWindowsRecording"; import { + type AdditionalWebcam, type CursorCaptureMode, + type FindRecordingCameraResult, normalizeCursorCaptureMode, normalizeProjectMedia, normalizeRecordingSession, @@ -121,12 +123,25 @@ import { NATIVE_WINDOWS_SALVAGEABLE_OUTPUT_BYTES, readMicrophoneDefaulted, readMicrophoneUnavailable, + readReportedWebcamPaths, readSecondaryWindowsApplied, - readWebcamFormat, - readWebcamUnavailable, + readStoppedWebcamPaths, + readUnavailableWebcamIndices, + readWebcamFormatAt, terminateNativeWindowsCapture, waitForNativeWindowsCaptureStop, } from "../recording/nativeWindowsCaptureStop"; +import { + additionalWebcamLabels, + buildHelperWebcamConfig, + collectStoppedWebcams, + dedupeAdditionalWebcams, + isWebcamSidecarFile, + labelsOfUnavailableAdditionalWebcams, + labelsOfWebcamsStoppedEarly, + stripWebcamSuffix, + webcamOutputPath, +} from "../recording/nativeWindowsWebcams"; import { patchWebmDurationOnDisk } from "../recording/webm-duration"; import { reindexRecordingOnDisk } from "../recording/webm-seek-index"; import { @@ -598,9 +613,25 @@ async function getApprovedProjectSession( throw new Error("Project references an invalid or unsupported webcam video path"); } - return webcamVideoPath - ? { screenVideoPath, webcamVideoPath, createdAt: Date.now() } - : { screenVideoPath, createdAt: Date.now() }; + // Additional cameras go through the same approval as camera 1; one that no + // longer resolves is dropped rather than failing the whole project. + const additionalWebcams: AdditionalWebcam[] = []; + for (const extra of media.additionalWebcams ?? []) { + const approved = await approveReadableVideoPath( + await resolveWithSiblingFallback(extra.path), + trustedDirs, + ); + if (approved) { + additionalWebcams.push({ path: approved, label: extra.label }); + } + } + + return { + screenVideoPath, + ...(webcamVideoPath ? { webcamVideoPath } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), + createdAt: Date.now(), + }; } type SelectedSource = { @@ -657,6 +688,8 @@ export interface RecordingPrefs { camDeviceId: string | null; /** Camera label paired with the preferred id for restart-safe resolution. */ camDeviceName: string | null; + /** Cameras 2-4 of a native Windows recording, in pick order. At most three. */ + camAdditionalDevices: Array<{ id: string | null; name: string }>; /** Capture resolution for the camera. See WEBCAM_QUALITY_PRESETS. */ camQuality: WebcamQualityId; systemAudioEnabled: boolean; @@ -672,6 +705,7 @@ const defaultRecordingPrefs: RecordingPrefs = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: DEFAULT_WEBCAM_QUALITY, systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", @@ -726,6 +760,18 @@ let nativeWindowsCaptureProcess: ChildProcessWithoutNullStreams | null = null; let nativeWindowsCaptureOutput = ""; let nativeWindowsCaptureTargetPath: string | null = null; let nativeWindowsCaptureWebcamTargetPath: string | null = null; +/** + * Cameras 2-4 of the running take, in helper order after camera 1, each with + * the label it is reported under. Only paths generated here ever land in it. + */ +let nativeWindowsCaptureAdditionalWebcamTargets: Array<{ path: string; label: string }> = []; +/** Camera 1's label for notices: its device name, else "Camera 1". */ +let nativeWindowsCaptureWebcamLabel = "Camera 1"; +/** + * Files of cameras the helper dropped at start. The helper deletes them, but a + * delete that failed leaves a 0-byte stub; stop and discard remove it if empty. + */ +let nativeWindowsCaptureDroppedWebcamPaths: string[] = []; let nativeWindowsCaptureRecordingId: number | null = null; let nativeWindowsCursorOffsetMs = 0; let nativeWindowsCursorCaptureMode: CursorCaptureMode = "editable-overlay"; @@ -752,6 +798,9 @@ function resetNativeWindowsCaptureState() { nativeWindowsCaptureProcess = null; nativeWindowsCaptureTargetPath = null; nativeWindowsCaptureWebcamTargetPath = null; + nativeWindowsCaptureAdditionalWebcamTargets = []; + nativeWindowsCaptureWebcamLabel = "Camera 1"; + nativeWindowsCaptureDroppedWebcamPaths = []; nativeWindowsCaptureRecordingId = null; nativeWindowsCursorOffsetMs = 0; nativeWindowsCursorCaptureMode = "editable-overlay"; @@ -779,12 +828,12 @@ async function salvageNativeWindowsFragmentedCapture(screenVideoPath: string | n */ async function removeNativeWindowsCaptureOutputs( screenVideoPath: string | null, - webcamVideoPath: string | null, + webcamVideoPaths: Array, options: { onlyIfUnusable?: boolean } = {}, ) { const targets = [ screenVideoPath, - webcamVideoPath, + ...webcamVideoPaths, screenVideoPath ? `${screenVideoPath}.cursor.json` : null, ]; @@ -810,6 +859,22 @@ async function removeNativeWindowsCaptureOutputs( } } } + +/** Removes the 0-byte stubs of cameras dropped at start; anything with data stays. */ +async function removeEmptyNativeWindowsWebcamFiles(paths: string[]) { + for (const target of paths) { + if (!isPathWithinDir(target, RECORDINGS_DIR)) { + continue; + } + const stats = await fs.stat(target).catch(() => null); + if (stats?.size !== 0) { + continue; + } + await fs.rm(target, { force: true }).catch((error) => { + console.warn("[native-wgc] could not remove an empty camera file:", target, error); + }); + } +} let nativeMacCaptureProcess: ChildProcessWithoutNullStreams | null = null; /** * Apple's system picker session (macOS 15.2+), started on first use and kept for the app's @@ -1341,6 +1406,7 @@ async function registerRecordingMediaLinks( options: { webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorCaptureMode?: CursorCaptureMode; }, ) { @@ -1355,6 +1421,9 @@ async function registerRecordingMediaLinks( ...(options.webcamVideoPath && Number.isFinite(options.webcamOffsetMs) ? { webcamOffsetMs: options.webcamOffsetMs } : {}), + ...(options.additionalWebcams?.length + ? { additionalWebcams: options.additionalWebcams } + : {}), ...(hasCursorTelemetry ? { cursorTelemetryPath } : {}), ...(options.cursorCaptureMode ? { cursorCaptureMode: options.cursorCaptureMode } : {}), }); @@ -1736,9 +1805,7 @@ function setCurrentRecordingSessionState(session: RecordingSession | null) { function getSessionManifestPathForVideo(videoPath: string) { const parsedPath = path.parse(videoPath); - const baseName = parsedPath.name.endsWith("-webcam") - ? parsedPath.name.slice(0, -"-webcam".length) - : parsedPath.name; + const baseName = stripWebcamSuffix(parsedPath.name); return path.join(parsedPath.dir, `${baseName}${RECORDING_SESSION_SUFFIX}`); } @@ -1792,10 +1859,34 @@ async function loadRecordedSessionForVideoPath( } } + if (session.additionalWebcams) { + const approvedExtras: AdditionalWebcam[] = []; + for (const extra of session.additionalWebcams) { + let extraPath: string | null = extra.path; + if (!isPathAllowed(extraPath)) { + extraPath = await approveReadableVideoPath(extraPath, [ + path.dirname(manifestPath), + RECORDINGS_DIR, + ]); + } + if (extraPath) { + approvedExtras.push({ path: extraPath, label: extra.label }); + } + } + if (approvedExtras.length > 0) { + session.additionalWebcams = approvedExtras; + } else { + delete session.additionalWebcams; + } + } + approveFilePath(session.screenVideoPath); if (session.webcamVideoPath) { approveFilePath(session.webcamVideoPath); } + for (const extra of session.additionalWebcams ?? []) { + approveFilePath(extra.path); + } return session; } catch (error) { const nodeError = error as NodeJS.ErrnoException; @@ -1815,6 +1906,7 @@ async function loadRecordedSessionForVideoPath( async function resolveMediaLinksForVideo(videoPath: string): Promise<{ webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; resolvedVia: "sidecar" | "fingerprint" | "none"; }> { @@ -1825,6 +1917,7 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ .then(() => true) .catch(() => false); + const sessionAdditionalWebcams = session?.additionalWebcams ?? []; if (session?.webcamVideoPath || hasCursorTelemetry) { // Opportunistic backfill so the link survives a later move even if this // recording predates the registry, or if its sidecar doesn't travel with it. @@ -1833,6 +1926,9 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ ...(session?.webcamVideoPath && Number.isFinite(session.webcamOffsetMs) ? { webcamOffsetMs: session.webcamOffsetMs } : {}), + ...(sessionAdditionalWebcams.length > 0 + ? { additionalWebcams: sessionAdditionalWebcams } + : {}), ...(hasCursorTelemetry ? { cursorTelemetryPath } : {}), }).catch((error) => console.warn("[media-links] backfill failed:", error)); @@ -1843,6 +1939,9 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ webcamOffsetMs: session.webcamOffsetMs ?? 0, } : {}), + ...(sessionAdditionalWebcams.length > 0 + ? { additionalWebcams: sessionAdditionalWebcams } + : {}), ...(hasCursorTelemetry ? { cursorTelemetryPath } : {}), resolvedVia: "sidecar", }; @@ -1856,8 +1955,19 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ webcamVideoPath = (await approveReadableVideoPath(webcamVideoPath, [RECORDINGS_DIR])) ?? undefined; } + const additionalWebcams: AdditionalWebcam[] = []; + for (const extra of links.additionalWebcams ?? []) { + let extraPath: string | null = extra.path; + if (!isPathAllowed(extraPath)) { + extraPath = await approveReadableVideoPath(extraPath, [RECORDINGS_DIR]); + } + if (extraPath) { + additionalWebcams.push({ path: extraPath, label: extra.label }); + } + } return { ...(webcamVideoPath ? { webcamVideoPath, webcamOffsetMs: links.webcamOffsetMs ?? 0 } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), ...(links.cursorTelemetryPath ? { cursorTelemetryPath: links.cursorTelemetryPath } : {}), resolvedVia: "fingerprint", }; @@ -2791,10 +2901,7 @@ export function registerIpcHandlers( ? request.recordingId : Date.now(); const outputPath = path.join(RECORDINGS_DIR, `${RECORDING_FILE_PREFIX}${recordingId}.mp4`); - const webcamOutputPath = path.join( - RECORDINGS_DIR, - `${RECORDING_FILE_PREFIX}${recordingId}-webcam.mp4`, - ); + const webcamPath = webcamOutputPath(RECORDINGS_DIR, RECORDING_FILE_PREFIX, recordingId, 1); const sourceDisplay = request.source.type === "display" && typeof request.source.displayId === "number" ? (screen.getAllDisplays().find((display) => display.id === request.source.displayId) ?? @@ -2813,6 +2920,24 @@ export function registerIpcHandlers( const webcamDirectShowClsid = request.webcam.enabled ? await resolveDirectShowWebcamClsid(request.webcam.deviceName) : null; + // Cameras 2-4 only while camera 1 is on: its toggle governs every + // camera. Files are numbered from 2 in the order they are sent. + const keptExtras = request.webcam.enabled + ? dedupeAdditionalWebcams( + request.webcam, + Array.isArray(request.additionalWebcams) ? request.additionalWebcams : [], + ) + : []; + const extraLabels = additionalWebcamLabels(request.webcam.deviceName, keptExtras); + const additionalWebcams = await Promise.all( + keptExtras.map(async (extra, i) => ({ + deviceId: extra.deviceId, + deviceName: extra.deviceName, + label: extraLabels[i], + clsid: await resolveDirectShowWebcamClsid(extra.deviceName), + path: webcamOutputPath(RECORDINGS_DIR, RECORDING_FILE_PREFIX, recordingId, i + 2), + })), + ); const cursorCaptureMode = normalizeCursorCaptureMode(request.cursor?.mode) ?? "editable-overlay"; const envPreferSoftwareEncoder = (process.env.OPENSCREEN_WGC_PREFER_SOFTWARE_ENCODER ?? "") @@ -2844,13 +2969,12 @@ export function registerIpcHandlers( microphoneDeviceId: request.audio.microphone.deviceId ?? null, microphoneDeviceName: request.audio.microphone.deviceName ?? null, microphoneGain: request.audio.microphone.gain, - webcamEnabled: request.webcam.enabled, - webcamDeviceId: request.webcam.deviceId ?? null, - webcamDeviceName: request.webcam.deviceName ?? null, - webcamDirectShowClsid, - webcamWidth: request.webcam.width, - webcamHeight: request.webcam.height, - webcamFps: request.webcam.fps, + ...buildHelperWebcamConfig({ + camera1: request.webcam, + camera1Clsid: webcamDirectShowClsid, + camera1Path: webcamPath, + extras: additionalWebcams, + }), captureCursor: cursorCaptureMode === "system", cursorCaptureMode, hideDesktopIcons: @@ -2858,7 +2982,7 @@ export function registerIpcHandlers( appSettings.getSnapshot().recording.hideDesktopIcons, outputs: { screenPath: outputPath, - webcamPath: webcamOutputPath, + webcamPath, }, source: { type: request.source.type, @@ -2880,6 +3004,10 @@ export function registerIpcHandlers( source: request.source, audio: request.audio, webcam: request.webcam, + additionalWebcams: additionalWebcams.map(({ label, path: cameraPath }) => ({ + label, + path: cameraPath, + })), encoder: { preferSoftwareEncoder }, cursor: { mode: cursorCaptureMode }, // Both spaces, deliberately: the helper's own errors quote the physical @@ -2894,7 +3022,12 @@ export function registerIpcHandlers( await fs.mkdir(RECORDINGS_DIR, { recursive: true }); nativeWindowsCaptureOutput = ""; nativeWindowsCaptureTargetPath = outputPath; - nativeWindowsCaptureWebcamTargetPath = request.webcam.enabled ? webcamOutputPath : null; + nativeWindowsCaptureWebcamTargetPath = request.webcam.enabled ? webcamPath : null; + nativeWindowsCaptureAdditionalWebcamTargets = additionalWebcams.map( + ({ label, path: cameraPath }) => ({ label, path: cameraPath }), + ); + nativeWindowsCaptureWebcamLabel = request.webcam.deviceName?.trim() || "Camera 1"; + nativeWindowsCaptureDroppedWebcamPaths = []; nativeWindowsCaptureRecordingId = recordingId; nativeWindowsCursorOffsetMs = 0; nativeWindowsCursorCaptureMode = cursorCaptureMode; @@ -2930,7 +3063,13 @@ export function registerIpcHandlers( cursorCaptureMode === "editable-overlay" ? Math.max(0, captureStartedAtMs - cursorStartTimeMs) : 0; - const webcamFormat = readWebcamFormat(nativeWindowsCaptureOutput); + // Index-aware readers: with several cameras the helper prints one + // format line and possibly one unavailable warning per camera, and + // only index 0 (or no index, from an old helper) is camera 1. + const webcamFormat = readWebcamFormatAt(nativeWindowsCaptureOutput, 0); + const unavailableWebcamIndices = new Set( + readUnavailableWebcamIndices(nativeWindowsCaptureOutput), + ); const encoderSelection = readNativeWindowsEncoderSelection(nativeWindowsCaptureOutput); // Captured now because stop may have no helper left to ask. A helper // killed mid-recording is exactly the case where this matters most. @@ -2939,6 +3078,9 @@ export function registerIpcHandlers( captureStartedAtMs, cursorOffsetMs: nativeWindowsCursorOffsetMs, webcamFormat, + additionalWebcamFormats: nativeWindowsCaptureAdditionalWebcamTargets.map((_, i) => + readWebcamFormatAt(nativeWindowsCaptureOutput, i + 1), + ), encoderSelection, // Logged only: menus missing from a window take on Windows before 11 // 24H2 are a platform limit, not something to put in front of the user. @@ -2958,8 +3100,7 @@ export function registerIpcHandlers( // missing `webcamFormat`: absence of the format line also means "the // line could not be parsed", which would put a red toast on a recording // whose camera is working perfectly. - const webcamUnavailable = - request.webcam.enabled && readWebcamUnavailable(nativeWindowsCaptureOutput); + const webcamUnavailable = request.webcam.enabled && unavailableWebcamIndices.has(0); // Same shape as the camera notice: the helper records the Windows // default input rather than failing, so this take is usable but is // almost certainly the wrong microphone. @@ -2977,6 +3118,35 @@ export function registerIpcHandlers( deviceName: request.webcam.deviceName, }); } + // Helper indices follow the `webcams` list: camera 1 at 0, extras after. + const startedWebcams = [ + { path: webcamPath, label: request.webcam.deviceName ?? "" }, + ...nativeWindowsCaptureAdditionalWebcamTargets, + ]; + const unavailableWebcams = request.webcam.enabled + ? labelsOfUnavailableAdditionalWebcams(startedWebcams, [...unavailableWebcamIndices]) + : []; + // The helper deletes a dropped camera's file but may fail to; a stub + // left behind is removed at stop or discard if it is still empty. + nativeWindowsCaptureDroppedWebcamPaths = request.webcam.enabled + ? startedWebcams + .filter((_, i) => unavailableWebcamIndices.has(i)) + .map((camera) => camera.path) + : []; + if (unavailableWebcams.length > 0) { + console.warn( + "[native-wgc] recording without additional cameras the helper could not open", + { + unavailableWebcams, + }, + ); + // Already reported now; the helper deleted their files, so leaving + // them in the targets would report them a second time at stop. + nativeWindowsCaptureAdditionalWebcamTargets = + nativeWindowsCaptureAdditionalWebcamTargets.filter( + (_, i) => !unavailableWebcamIndices.has(i + 1), + ); + } return { success: true, @@ -2986,6 +3156,7 @@ export function registerIpcHandlers( videoEncoderSelection: encoderSelection?.video ?? null, videoEncoderRuntime: encoderSelection?.videoEncoderRuntime ?? null, webcamUnavailable, + ...(unavailableWebcams.length > 0 ? { unavailableWebcams } : {}), microphoneDefaulted, }; } catch (error) { @@ -3321,6 +3492,16 @@ export function registerIpcHandlers( const proc = nativeWindowsCaptureProcess; const preferredPath = nativeWindowsCaptureTargetPath; const preferredWebcamPath = nativeWindowsCaptureWebcamTargetPath; + const additionalWebcamTargets = nativeWindowsCaptureAdditionalWebcamTargets; + const camera1Label = nativeWindowsCaptureWebcamLabel; + const droppedWebcamPaths = nativeWindowsCaptureDroppedWebcamPaths; + // Start-dropped cameras ride along so a discard or a failed stop also + // removes a stub the helper could not delete (both are empty). + const allWebcamPaths = [ + preferredWebcamPath, + ...additionalWebcamTargets.map((target) => target.path), + ...droppedWebcamPaths, + ]; const recordingId = nativeWindowsCaptureRecordingId ?? Date.now(); const cursorCaptureMode = nativeWindowsCursorCaptureMode; @@ -3343,7 +3524,7 @@ export function registerIpcHandlers( if (!exited) { detachNativeWindowsCaptureOutputDrain(); } - await removeNativeWindowsCaptureOutputs(preferredPath, preferredWebcamPath); + await removeNativeWindowsCaptureOutputs(preferredPath, allWebcamPaths); return { success: true, discarded: true }; } finally { // Unconditional. Killing a wedged helper can itself throw, and @@ -3417,7 +3598,7 @@ export function registerIpcHandlers( // explain. Size-gate it anyway: throwing away a recording to tidy // up after a failed stop is the worse mistake of the two, and the // gate is the same one the salvage check above uses. - await removeNativeWindowsCaptureOutputs(preferredPath, preferredWebcamPath, { + await removeNativeWindowsCaptureOutputs(preferredPath, allWebcamPaths, { onlyIfUnusable: true, }); // The helper log goes to console/diagnostics above, not into this @@ -3453,31 +3634,70 @@ export function registerIpcHandlers( shiftPendingCursorTelemetry(nativeWindowsCursorOffsetMs); await writePendingCursorTelemetry(screenVideoPath); } - let webcamVideoPath: string | undefined; - if (preferredWebcamPath) { - try { - // Size, not just existence. A camera that opened but delivered no - // frame still gets a file created for it, and its `Finalize()` then - // fails, leaving nought bytes on disk. Admitting that file put a - // camera track in the document pointing at something no demuxer can - // read, and the preview compositor answers an unreadable camera by - // drawing the SCREEN recording inside the little camera rectangle — - // which is how a webcam that never recorded showed up as the desktop - // duplicated into its own corner (getopenscreen/openscreen#387). - const webcamStat = await fs.stat(preferredWebcamPath); - webcamVideoPath = webcamStat.size > 0 ? preferredWebcamPath : undefined; - if (!webcamVideoPath) { - console.warn("[native-wgc] the webcam file is empty; saving without a camera", { - path: preferredWebcamPath, - }); - } - } catch { - webcamVideoPath = undefined; + // Size, not just existence. A camera that opened but delivered no frame + // still gets a file created for it, and its `Finalize()` then fails, + // leaving nought bytes on disk. Admitting that file put a camera track in + // the document pointing at something no demuxer can read, and the preview + // compositor answers an unreadable camera by drawing the SCREEN recording + // inside the little camera rectangle — which is how a webcam that never + // recorded showed up as the desktop duplicated into its own corner + // (getopenscreen/openscreen#387). Every camera is judged by its own file, + // not by the helper's list at stop (see `collectStoppedWebcams`). + const requestedWebcams = [ + ...(preferredWebcamPath ? [{ path: preferredWebcamPath, label: camera1Label }] : []), + ...additionalWebcamTargets, + ]; + const webcamSizes = new Map(); + for (const camera of requestedWebcams) { + const stat = await fs.stat(camera.path).catch(() => null); + if (stat) { + webcamSizes.set(camera.path, stat.size); } } - const session: RecordingSession = webcamVideoPath - ? { screenVideoPath, webcamVideoPath, createdAt: recordingId, cursorCaptureMode } - : { screenVideoPath, createdAt: recordingId, cursorCaptureMode }; + const stoppedWebcams = collectStoppedWebcams({ + camera1Enabled: Boolean(preferredWebcamPath), + requested: requestedWebcams, + sizes: webcamSizes, + }); + const webcamVideoPath = stoppedWebcams.camera1; + const additionalWebcams = stoppedWebcams.additional; + if (preferredWebcamPath && !webcamVideoPath && webcamSizes.has(preferredWebcamPath)) { + console.warn("[native-wgc] the webcam file is empty; saving without a camera", { + path: preferredWebcamPath, + }); + } + if (stoppedWebcams.dropped.length > 0) { + console.warn("[native-wgc] additional cameras produced nothing usable", { + dropped: stoppedWebcams.dropped, + helperWebcamPaths: readStoppedWebcamPaths(nativeWindowsCaptureOutput), + }); + } + // A camera the helper disabled mid-take keeps its partial file (it is + // in the take) but is named, so the user knows why it ends early. Only + // a helper that sends `webcamPaths` can tell; otherwise nothing is said. + const webcamsStoppedEarly = labelsOfWebcamsStoppedEarly({ + requested: requestedWebcams, + sizes: webcamSizes, + helperWebcamPaths: readReportedWebcamPaths(nativeWindowsCaptureOutput), + }); + if (webcamsStoppedEarly.length > 0) { + console.warn("[native-wgc] cameras stopped before the end of the take", { + webcamsStoppedEarly, + }); + } + await removeEmptyNativeWindowsWebcamFiles(droppedWebcamPaths); + // Generated by the start handler, never taken from the renderer or the + // helper; approved like the session's other media. + for (const extra of additionalWebcams) { + approveFilePath(extra.path); + } + const session: RecordingSession = { + screenVideoPath, + ...(webcamVideoPath ? { webcamVideoPath } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), + createdAt: recordingId, + cursorCaptureMode, + }; setCurrentRecordingSessionState(session); currentProjectPath = null; @@ -3486,7 +3706,11 @@ export function registerIpcHandlers( `${path.parse(screenVideoPath).name}${RECORDING_SESSION_SUFFIX}`, ); await fs.writeFile(sessionManifestPath, JSON.stringify(session, null, 2), "utf-8"); - await registerRecordingMediaLinks(screenVideoPath, { webcamVideoPath, cursorCaptureMode }); + await registerRecordingMediaLinks(screenVideoPath, { + webcamVideoPath, + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), + cursorCaptureMode, + }); return { success: true, @@ -3501,6 +3725,8 @@ export function registerIpcHandlers( // unreported, the user would find out in the editor — which is exactly // the silence this change exists to end. webcamDropped: Boolean(preferredWebcamPath) && !webcamVideoPath, + ...(stoppedWebcams.dropped.length > 0 ? { droppedWebcams: stoppedWebcams.dropped } : {}), + ...(webcamsStoppedEarly.length > 0 ? { webcamsStoppedEarly } : {}), message: recovered ? "Native Windows recording recovered from a failed stop" : "Native Windows recording session stored successfully", @@ -3969,7 +4195,7 @@ export function registerIpcHandlers( const files = await fs.readdir(RECORDINGS_DIR); const videoFiles = files.filter( - (file) => file.endsWith(".webm") && !file.endsWith("-webcam.webm"), + (file) => file.endsWith(".webm") && !isWebcamSidecarFile(file), ); if (videoFiles.length === 0) { @@ -4727,21 +4953,16 @@ export function registerIpcHandlers( // `addAsset` in the new editor's project store. ipcMain.handle( "find-recording-camera", - async ( - _event, - videoPath: string, - ): Promise<{ - success: boolean; - webcamVideoPath?: string; - offsetMs?: number; - error?: string; - }> => { + async (_event, videoPath: string): Promise => { try { const normalized = normalizeVideoSourcePath(videoPath); if (!normalized || !isPathAllowed(normalized)) { return { success: false, error: "Video path has not been approved" }; } const resolution = await resolveMediaLinksForVideo(normalized); + // Additional cameras are only returned alongside camera 1. That relies + // on R6 (extras are recorded only while camera 1 is on), so extras + // without camera 1 means camera 1's file came out empty. if (!resolution.webcamVideoPath) { return { success: false, error: "No camera attached to this recording" }; } @@ -4749,6 +4970,9 @@ export function registerIpcHandlers( success: true, webcamVideoPath: resolution.webcamVideoPath, offsetMs: resolution.webcamOffsetMs ?? 0, + ...(resolution.additionalWebcams?.length + ? { additionalWebcams: resolution.additionalWebcams } + : {}), }; } catch (err) { return { diff --git a/electron/ipc/nativeBridge.ts b/electron/ipc/nativeBridge.ts index 61ed8ef86..82e7c15a1 100644 --- a/electron/ipc/nativeBridge.ts +++ b/electron/ipc/nativeBridge.ts @@ -425,6 +425,7 @@ export function registerNativeBridgeHandlers(context: NativeBridgeContext) { request.payload.webcamOffsetSec, request.payload.clipIndex, request.payload.sourceTimeSec, + request.payload.additionalCameras ?? [], ); return createSuccessResponse(requestId, { ok: true }); case "destroyView": diff --git a/electron/ipc/recordingPrefs.test.ts b/electron/ipc/recordingPrefs.test.ts index 2c3317af9..7d2d056a1 100644 --- a/electron/ipc/recordingPrefs.test.ts +++ b/electron/ipc/recordingPrefs.test.ts @@ -19,6 +19,7 @@ const defaults: RecordingPrefs = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", diff --git a/electron/media/mediaLinksRegistry.test.ts b/electron/media/mediaLinksRegistry.test.ts index 7c76a7580..0aae45269 100644 --- a/electron/media/mediaLinksRegistry.test.ts +++ b/electron/media/mediaLinksRegistry.test.ts @@ -187,6 +187,29 @@ describe("mediaLinksRegistry", () => { } }); + it("round-trips additional cameras and resolves old records without them", async () => { + const screenPath = path.join(tempDir, "multi.webm"); + const oldScreenPath = path.join(tempDir, "old.webm"); + await writeFileOfSize(screenPath, 5000, "m"); + await writeFileOfSize(oldScreenPath, 5000, "o"); + const additionalWebcams = [ + { path: path.join(tempDir, "multi-webcam-2.mp4"), label: "Desk" }, + { path: path.join(tempDir, "multi-webcam-3.mp4"), label: "" }, + ]; + await registerMediaLinks(tempDir, screenPath, { + webcamVideoPath: path.join(tempDir, "multi-webcam.mp4"), + additionalWebcams, + }); + await registerMediaLinks(tempDir, oldScreenPath, { + webcamVideoPath: path.join(tempDir, "old-webcam.mp4"), + }); + + const resolved = await findMediaLinksByFingerprint(tempDir, screenPath); + expect(resolved?.additionalWebcams).toEqual(additionalWebcams); + const old = await findMediaLinksByFingerprint(tempDir, oldScreenPath); + expect(old).not.toHaveProperty("additionalWebcams"); + }); + it("returns null when there is no matching fingerprint", async () => { const unknownPath = path.join(tempDir, "unknown.webm"); await writeFileOfSize(unknownPath, 1000, "z"); diff --git a/electron/media/mediaLinksRegistry.ts b/electron/media/mediaLinksRegistry.ts index ec0ad76de..56bd7d57f 100644 --- a/electron/media/mediaLinksRegistry.ts +++ b/electron/media/mediaLinksRegistry.ts @@ -19,7 +19,11 @@ import fs from "node:fs/promises"; import path from "node:path"; -import type { CursorCaptureMode } from "../../src/lib/recordingSession"; +import { + type AdditionalWebcam, + type CursorCaptureMode, + normalizeAdditionalWebcams, +} from "../../src/lib/recordingSession"; // ponytail: `baseDir` is passed in by every caller (RECORDINGS_DIR in // electron/ipc/handlers.ts) rather than imported here, so this module has no @@ -40,6 +44,7 @@ export interface MediaLinkEntry { fingerprint: MediaFingerprint; webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; cursorCaptureMode?: CursorCaptureMode; updatedAt: string; @@ -98,6 +103,7 @@ function normalizeEntry(candidate: unknown): MediaLinkEntry | null { if (!candidate || typeof candidate !== "object") return null; const raw = candidate as Partial; const fp = raw.fingerprint; + const additionalWebcams = normalizeAdditionalWebcams(raw.additionalWebcams); if ( typeof raw.lastKnownPath !== "string" || !fp || @@ -116,6 +122,7 @@ function normalizeEntry(candidate: unknown): MediaLinkEntry | null { }, ...(typeof raw.webcamVideoPath === "string" ? { webcamVideoPath: raw.webcamVideoPath } : {}), ...(typeof raw.webcamOffsetMs === "number" ? { webcamOffsetMs: raw.webcamOffsetMs } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), ...(typeof raw.cursorTelemetryPath === "string" ? { cursorTelemetryPath: raw.cursorTelemetryPath } : {}), @@ -220,6 +227,7 @@ async function updateRegistry( export interface MediaLinksToRegister { webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; cursorCaptureMode?: CursorCaptureMode; } @@ -235,7 +243,12 @@ export async function registerMediaLinks( videoPath: string, links: MediaLinksToRegister, ): Promise { + // Extras alone register nothing. That relies on R6: additional cameras are + // recorded only while camera 1 is on, so a take with extras but no camera 1 + // is one whose camera 1 file came out empty — a rare loss accepted here. if (!links.webcamVideoPath && !links.cursorTelemetryPath) return; + const { additionalWebcams: rawAdditionalWebcams, ...linksWithoutAdditional } = links; + const additionalWebcams = normalizeAdditionalWebcams(rawAdditionalWebcams); const fingerprint = await computeFingerprint(videoPath); await updateRegistry(baseDir, (file) => { const existingIndex = file.entries.findIndex((e) => @@ -244,7 +257,8 @@ export async function registerMediaLinks( const entry: MediaLinkEntry = { lastKnownPath: videoPath, fingerprint, - ...links, + ...linksWithoutAdditional, + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), updatedAt: new Date().toISOString(), }; const entries = @@ -258,6 +272,7 @@ export async function registerMediaLinks( export interface MediaLinksLookup { webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; cursorCaptureMode?: CursorCaptureMode; } @@ -314,6 +329,7 @@ export async function findRelocatedMediaByStoredPath( screenVideoPath: match.lastKnownPath, ...(match.webcamVideoPath ? { webcamVideoPath: match.webcamVideoPath } : {}), ...(typeof match.webcamOffsetMs === "number" ? { webcamOffsetMs: match.webcamOffsetMs } : {}), + ...(match.additionalWebcams?.length ? { additionalWebcams: match.additionalWebcams } : {}), ...(match.cursorTelemetryPath ? { cursorTelemetryPath: match.cursorTelemetryPath } : {}), ...(match.cursorCaptureMode ? { cursorCaptureMode: match.cursorCaptureMode } : {}), }; @@ -358,6 +374,7 @@ export async function findMediaLinksByFingerprint( return { ...(match.webcamVideoPath ? { webcamVideoPath: match.webcamVideoPath } : {}), ...(typeof match.webcamOffsetMs === "number" ? { webcamOffsetMs: match.webcamOffsetMs } : {}), + ...(match.additionalWebcams?.length ? { additionalWebcams: match.additionalWebcams } : {}), ...(match.cursorTelemetryPath ? { cursorTelemetryPath: match.cursorTelemetryPath } : {}), ...(match.cursorCaptureMode ? { cursorCaptureMode: match.cursorCaptureMode } : {}), }; diff --git a/electron/media/projectMediaRelinker.test.ts b/electron/media/projectMediaRelinker.test.ts index 2b52d32b5..505a430f4 100644 --- a/electron/media/projectMediaRelinker.test.ts +++ b/electron/media/projectMediaRelinker.test.ts @@ -62,6 +62,54 @@ describe("relinkProjectMedia", () => { expect(logged.join("\n")).toContain(currentWebcamPath); }); + it("relinks additional cameras by index and leaves absent ones alone", async () => { + const currentScreenPath = path.join(tempDir, "recording-43.mp4"); + const currentWebcamPath = path.join(tempDir, "recording-43-webcam.mp4"); + const currentExtra2 = path.join(tempDir, "recording-43-webcam-2.mp4"); + const currentExtra3 = path.join(tempDir, "recording-43-webcam-3.mp4"); + await fs.writeFile(currentScreenPath, "screen bytes"); + await fs.writeFile(currentWebcamPath, "webcam bytes"); + await fs.writeFile(currentExtra2, "extra 2"); + await fs.writeFile(currentExtra3, "extra 3"); + await registerMediaLinks(tempDir, currentScreenPath, { + webcamVideoPath: currentWebcamPath, + additionalWebcams: [ + { path: currentExtra2, label: "Desk" }, + { path: currentExtra3, label: "Wide" }, + ], + }); + + const project = { + assets: [ + { + id: "asset-1", + originalPath: "C:\\Users\\demo\\recording-43.mp4", + sizeBytes: Buffer.byteLength("screen bytes"), + cameraTrack: { + sourcePath: "C:\\Users\\demo\\recording-43-webcam.mp4", + startMs: 0, + offsetMs: 0, + visible: true, + }, + additionalCameraTracks: [ + { sourcePath: "C:\\Users\\demo\\recording-43-webcam-2.mp4", label: "Desk" }, + { sourcePath: "C:\\Users\\demo\\recording-43-webcam-3.mp4", label: "Wide" }, + ], + }, + ], + }; + + const relinked = (await relinkProjectMedia(project, tempDir)) as typeof project; + + expect(relinked.assets[0].cameraTrack.sourcePath).toBe(currentWebcamPath); + expect(relinked.assets[0].additionalCameraTracks.map((t) => t.sourcePath)).toEqual([ + currentExtra2, + currentExtra3, + ]); + expect(relinked.assets[0].additionalCameraTracks.map((t) => t.label)).toEqual(["Desk", "Wide"]); + expect(project.assets[0].additionalCameraTracks[0].sourcePath).toContain("demo"); + }); + it("refuses to relink an asset the document recorded no size for", async () => { // A same-named recording exists and is registered with its webcam, so a // basename match would resolve — that is exactly what must not happen. The diff --git a/electron/media/projectMediaRelinker.ts b/electron/media/projectMediaRelinker.ts index 0c0b8c259..bef7dad02 100644 --- a/electron/media/projectMediaRelinker.ts +++ b/electron/media/projectMediaRelinker.ts @@ -41,11 +41,23 @@ async function resolveAssetMedia( isRecord(cameraTrack) && typeof cameraTrack.sourcePath === "string" && cameraTrack.sourcePath ? cameraTrack.sourcePath : null; + const additionalTracks = Array.isArray(asset.additionalCameraTracks) + ? asset.additionalCameraTracks + : []; + const additionalMissing = await Promise.all( + additionalTracks.map( + async (track) => + isRecord(track) && + typeof track.sourcePath === "string" && + track.sourcePath !== "" && + !(await fileExists(track.sourcePath)), + ), + ); const screenExists = await fileExists(originalPath); const cameraMissing = cameraPath !== null && !(await fileExists(cameraPath)); // Nothing to repair, and this runs on every project open — don't fingerprint // (i.e. open and read) every asset just to confirm what the stats already say. - if (screenExists && !cameraMissing) return asset; + if (screenExists && !cameraMissing && !additionalMissing.some(Boolean)) return asset; let links: RelocatedMediaLookup | null = null; if (screenExists) { @@ -81,10 +93,25 @@ async function resolveAssetMedia( nextCameraTrack = { ...cameraTrack, sourcePath: links.webcamVideoPath }; } + // Extras are matched by index, the same order the recording registered them in. + let additionalChanged = false; + const nextAdditionalTracks = await Promise.all( + additionalTracks.map(async (track, index) => { + const replacement = links.additionalWebcams?.[index]?.path; + if (!additionalMissing[index] || !replacement || !(await fileExists(replacement))) { + return track; + } + console.log(`[media-relink] additional webcam ${index + 2} -> ${replacement}`); + additionalChanged = true; + return { ...track, sourcePath: replacement }; + }), + ); + return { ...asset, originalPath: links.screenVideoPath, ...(nextCameraTrack === cameraTrack ? {} : { cameraTrack: nextCameraTrack }), + ...(additionalChanged ? { additionalCameraTracks: nextAdditionalTracks } : {}), }; } diff --git a/electron/native-bridge/services/compositorViewService.ts b/electron/native-bridge/services/compositorViewService.ts index 6dbfb1461..550d67dcb 100644 --- a/electron/native-bridge/services/compositorViewService.ts +++ b/electron/native-bridge/services/compositorViewService.ts @@ -14,6 +14,7 @@ import type { } from "../../../src/native/contracts"; import type { GifExportJob } from "../../ipc/gifExportJobs"; import type { + ClipCameraInput, ClipInput, CompositorBackend, CompositorParamValue, @@ -775,12 +776,21 @@ export class CompositorViewService { webcamOffsetSec: number, clipIndex: number, sourceTimeSec: number, + additionalCameras: ClipCameraInput[] = [], ): void { const addon = this.ensureAddon(); if (!addon) { return; } - addon.setActiveClip(id, screenPath, webcamPath, webcamOffsetSec, clipIndex, sourceTimeSec); + addon.setActiveClip( + id, + screenPath, + webcamPath, + webcamOffsetSec, + clipIndex, + sourceTimeSec, + additionalCameras, + ); } destroyView(id: number): void { diff --git a/electron/native/README.md b/electron/native/README.md index 5f68fb254..a9c6f8b83 100644 --- a/electron/native/README.md +++ b/electron/native/README.md @@ -85,6 +85,17 @@ Current V2 JSON shape: The current helper implementation supports display/window video capture, system audio loopback, selected-microphone capture, Media Foundation webcam capture, and a DirectShow webcam fallback for virtual cameras that are not exposed through Media Foundation. Webcam frames are currently composed into the primary MP4 as a bottom-right picture-in-picture overlay. Browser `deviceId` values do not always map to Media Foundation symbolic links or WASAPI endpoint IDs, so the renderer passes both browser IDs and user-visible device names. For microphones, the helper tries the requested WASAPI endpoint ID first, then resolves an active capture endpoint by `microphoneDeviceName`, then falls back to the default endpoint. For webcams, Electron resolves a matching DirectShow filter CLSID for the selected label; the helper uses Media Foundation first, then that exact DirectShow filter when the requested camera is absent from Media Foundation. +Several cameras: a `webcams` list records up to four cameras, each into its own MP4, all on the same T0 as the screen and audio. Each entry takes `camDeviceId`, `camDeviceName`, `camClsid` (the DirectShow filter CLSID), `camWidth`, `camHeight`, `camFps` and `camPath`; an entry without `camPath` is skipped. When the list is present and non-empty it replaces the legacy `webcam*` fields; without it those fields still describe one camera, which writes to `webcamPath` when one is given and is drawn into the screen as the inline picture-in-picture otherwise. + +```json +"webcams": [ + { "camDeviceId": "…", "camDeviceName": "Camera A", "camClsid": "{…}", "camWidth": 1920, "camHeight": 1080, "camFps": 30, "camPath": "C:\\path\\recording-123-webcam.mp4" }, + { "camDeviceId": "…", "camDeviceName": "Camera B", "camClsid": "", "camWidth": 1280, "camHeight": 720, "camFps": 30, "camPath": "C:\\path\\recording-123-webcam-2.mp4" } +] +``` + +Every per-camera event carries the camera's `index`, its position in the list after entries without `camPath` have been skipped: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. A camera unplugged mid-take (or whose reads keep failing for a second) is disabled the same way: its file ends at the loss, is still finalized at stop, and is left out of `webcamPaths`. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order — an empty list when every camera that wrote a file of its own was disabled mid-take, so a reader can tell "stopped early" from an older helper that never sends the key. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. + Container: recordings are written as fragmented MP4 (`MFCreateFMPEG4MediaSink` + `MFCreateSinkWriterFromMediaSink`, `MF_MPEG4SINK_MIN_FRAGMENT_DURATION` = 1s) rather than plain MP4. A plain MP4 has no index until `IMFSinkWriter::Finalize()` writes `moov` at the very end, so when the shutdown watchdog force-exits a wedged helper the file on disk holds every frame and no way to read them — that is why issues #252 / #292 / #327 cost the whole recording rather than the frozen tail of it. A fragmented MP4 writes its index up front and its samples in self-describing `moof`+`mdat` pairs, so the same kill leaves a file that plays up to the last complete fragment. This does not fix the freeze; it removes the data loss the freeze causes. Because the fragmented sink needs both output media types at construction, the sink writer is built from a media sink instead of from a URL, and the helper reads the video/audio stream positions back off the sink rather than assuming them. If any of that is unavailable on a machine, the helper retries with the plain container and says so — `container` in the `encoder-selection` event is `fragmented-mp4` or `mp4`, and it reports what was used, not what was asked for. Encoder selection: by default the helper keeps the existing sink-writer path first. If that path fails while setting up H.264, it retries with the Microsoft software H.264 encoder (`mfh264enc.dll`). The key of this retry is registering that encoder locally in the helper process via `MFTRegisterLocalByCLSID`, which makes a software H.264 encoder available even when the machine's hardware encoders are missing or broken; hardware transforms are disabled for the retry only as a secondary guard so the sink writer prefers the locally registered software encoder, not as the fallback mechanism itself. Set `preferSoftwareEncoder: true` in the helper JSON, or set `OPENSCREEN_WGC_PREFER_SOFTWARE_ENCODER=true` before launching Electron, to force the software path from the first attempt. @@ -126,6 +137,12 @@ npm run test:wgc-webcam:win Remove-Item Env:OPENSCREEN_WGC_TEST_WEBCAM_DEVICE_NAME ``` +To check that a camera which cannot be opened costs only itself, record the real camera next to a nonexistent second one listed through `webcams`: + +```powershell +npm run test:wgc-helper:win -- --webcam --missing-second-webcam +``` + To validate a specific native microphone manually: ```powershell diff --git a/electron/native/compositor-view/addon.d.ts b/electron/native/compositor-view/addon.d.ts index e87b830b5..cbc1b8d79 100644 --- a/electron/native/compositor-view/addon.d.ts +++ b/electron/native/compositor-view/addon.d.ts @@ -118,6 +118,14 @@ export interface ExportParamsInput { bitrate?: number; } +/** An additional camera (2-4) of a clip, for `setActiveClip` and the export clip list. + * = napi `ClipCameraInput`. */ +export interface ClipCameraInput { + path: string; + /** Camera source time = screen source time - this. */ + offsetSec: number; +} + /** One timeline clip for the native multiclip export (screen + webcam files + source trim). */ export interface ClipInput { screenPath: string; @@ -126,6 +134,8 @@ export interface ClipInput { sourceEndSec: number; /** webcam source time = screen source time − this. */ webcamOffsetSec: number; + /** Cameras 2-4 (index k-1 = camera k). Only those a layout region shows are decoded. */ + additionalCameras?: ClipCameraInput[]; } /** Which backend the compositor will run on. `"cpu"` = WARP rasterisation + software @@ -198,6 +208,8 @@ export interface CompositorViewAddon { /** Installs the app scene (JSON `SceneDescription`) — layout preset etc. drive the render * instead of the fixture. Invalid JSON is ignored native-side. */ setScene(id: number, sceneJson: string): void; + /** `additionalCameras`: cameras 2-4 of the clip (index k-1 = camera k, empty `path` = none). + * An addon built before it ignores the argument. */ setActiveClip( id: number, screenPath: string, @@ -205,6 +217,7 @@ export interface CompositorViewAddon { webcamOffsetSec: number, clipIndex: number, sourceTimeSec: number, + additionalCameras?: ClipCameraInput[], ): void; destroyView(id: number): void; /** Renders the fixture to `outPath` (C8), auto-pausing live previews. `onProgress` diff --git a/electron/native/wgc-capture/CMakeLists.txt b/electron/native/wgc-capture/CMakeLists.txt index 50466d0d0..009a04677 100644 --- a/electron/native/wgc-capture/CMakeLists.txt +++ b/electron/native/wgc-capture/CMakeLists.txt @@ -39,11 +39,15 @@ add_executable(wgc-capture src/audio_sample_utils.cpp src/audio_sample_utils.h src/desktop_icon_cover.cpp + src/device_selection.cpp + src/device_selection.h src/desktop_icon_cover.h src/dpi_awareness.h src/realtime_scheduling.h src/frame_visibility.cpp src/frame_visibility.h + src/json_fields.cpp + src/json_fields.h src/dshow_webcam_capture.cpp src/dshow_webcam_capture.h src/main.cpp @@ -58,8 +62,12 @@ add_executable(wgc-capture src/wasapi_render_keepalive.cpp src/wasapi_render_keepalive.h src/webcam_capture.cpp + src/webcam_loss.cpp + src/webcam_loss.h src/webcam_format.cpp src/webcam_format.h + src/webcam_config.cpp + src/webcam_config.h src/webcam_capture.h src/wgc_session.cpp src/wgc_session.h @@ -198,3 +206,47 @@ target_link_libraries(mf_encoder_color_test PRIVATE ole32 propsys ) + +add_executable(webcam_config_test + src/json_fields.cpp + src/json_fields.h + src/webcam_config.cpp + src/webcam_config.h + src/webcam_config_test.cpp +) + +target_compile_definitions(webcam_config_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(webcam_config_test PRIVATE /EHsc /W4 /utf-8) + +add_executable(device_selection_test + src/device_selection.cpp + src/device_selection.h + src/device_selection_test.cpp +) + +target_compile_definitions(device_selection_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(device_selection_test PRIVATE /EHsc /W4 /utf-8) + +add_executable(webcam_loss_test + src/webcam_loss.cpp + src/webcam_loss.h + src/webcam_loss_test.cpp +) + +target_compile_definitions(webcam_loss_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(webcam_loss_test PRIVATE /EHsc /W4 /utf-8) diff --git a/electron/native/wgc-capture/src/device_selection.cpp b/electron/native/wgc-capture/src/device_selection.cpp new file mode 100644 index 000000000..3211cbd1d --- /dev/null +++ b/electron/native/wgc-capture/src/device_selection.cpp @@ -0,0 +1,158 @@ +#include "device_selection.h" + +#include +#include + +namespace { + +/** + * Does one of these appear inside the other as WHOLE WORDS? + * + * Plain containment answered for devices that merely share a spelling: a + * requested "Logi" is inside "Logitech", and "Micro" inside "Microphone", + * neither of them as a word. Matching on that resolved a camera nobody asked + * for -- and resolving one is exactly what stops the request reaching the + * DirectShow fallback, where the cameras Media Foundation cannot enumerate live. + * + * Both sides arrive normalized, so a boundary is the start of the string, its + * end, or a space. + */ +bool containsAsWords(const std::wstring& haystack, const std::wstring& needle) { + if (haystack.empty() || needle.empty()) { + return false; + } + size_t pos = haystack.find(needle); + while (pos != std::wstring::npos) { + const bool startsOnBoundary = pos == 0 || haystack[pos - 1] == L' '; + const size_t after = pos + needle.size(); + const bool endsOnBoundary = after == haystack.size() || haystack[after] == L' '; + if (startsOnBoundary && endsOnBoundary) { + return true; + } + pos = haystack.find(needle, pos + 1); + } + return false; +} + +bool containsInsensitive(const std::wstring& haystack, const std::wstring& needle) { + return containsAsWords(haystack, needle) || containsAsWords(needle, haystack); +} + +std::wstring normalizeDeviceName(const std::wstring& value) { + std::wstring normalized; + normalized.reserve(value.size()); + bool lastWasSpace = true; + for (const wchar_t ch : value) { + if (std::iswalnum(ch)) { + normalized.push_back(static_cast(std::towlower(ch))); + lastWasSpace = false; + continue; + } + if (!lastWasSpace) { + normalized.push_back(L' '); + lastWasSpace = true; + } + } + while (!normalized.empty() && normalized.back() == L' ') { + normalized.pop_back(); + } + return normalized; +} + +std::wstring toLower(const std::wstring& value) { + std::wstring lowered; + lowered.reserve(value.size()); + for (const wchar_t ch : value) { + lowered.push_back(static_cast(std::towlower(ch))); + } + return lowered; +} + +} // namespace + +bool DeviceClaims::contains(const std::wstring& identity) const { + const std::wstring normalized = normalizeDeviceIdentity(identity); + return !normalized.empty() && + std::find(identities_.begin(), identities_.end(), normalized) != identities_.end(); +} + +void DeviceClaims::add(const std::wstring& identity) { + const std::wstring normalized = normalizeDeviceIdentity(identity); + if (!normalized.empty() && + std::find(identities_.begin(), identities_.end(), normalized) == identities_.end()) { + identities_.push_back(normalized); + } +} + +std::wstring normalizeDeviceIdentity(const std::wstring& identity) { + const std::wstring lowered = toLower(identity); + const size_t interfaceClass = lowered.rfind(L"#{"); + if (interfaceClass == std::wstring::npos || interfaceClass == 0) { + return lowered; + } + const size_t classEnd = lowered.find(L'}', interfaceClass); + if (classEnd == std::wstring::npos) { + return lowered; + } + return lowered.substr(0, interfaceClass) + lowered.substr(classEnd + 1); +} + +int deviceMatchScore( + const std::wstring& candidateName, + const std::wstring& candidateLink, + const std::wstring& requestedName, + const std::wstring& requestedId) { + int score = 0; + const auto normalizedName = normalizeDeviceName(candidateName); + const auto normalizedLink = normalizeDeviceName(candidateLink); + const auto normalizedRequestedName = normalizeDeviceName(requestedName); + const auto normalizedRequestedId = normalizeDeviceName(requestedId); + + if (!normalizedRequestedName.empty()) { + if (normalizedName == normalizedRequestedName) { + score = std::max(score, 1000); + } + if (containsInsensitive(normalizedName, normalizedRequestedName)) { + score = std::max(score, 900); + } + if (containsInsensitive(normalizedLink, normalizedRequestedName)) { + score = std::max(score, 800); + } + } + + if (!normalizedRequestedId.empty()) { + if (containsInsensitive(normalizedLink, normalizedRequestedId)) { + score = std::max(score, 700); + } + if (containsInsensitive(normalizedName, normalizedRequestedId)) { + score = std::max(score, 600); + } + } + + return score; +} + +int selectUnclaimedDevice( + const std::vector& candidates, + const std::wstring& requestedName, + const std::wstring& requestedId, + const DeviceClaims& claims) { + const bool requested = !requestedName.empty() || !requestedId.empty(); + int selected = -1; + int bestScore = 0; + for (size_t index = 0; index < candidates.size(); ++index) { + if (claims.contains(candidates[index].identity)) { + continue; + } + const int score = + deviceMatchScore(candidates[index].name, candidates[index].identity, requestedName, requestedId); + if (requested && score <= 0) { + continue; + } + if (selected < 0 || score > bestScore) { + selected = static_cast(index); + bestScore = score; + } + } + return selected; +} diff --git a/electron/native/wgc-capture/src/device_selection.h b/electron/native/wgc-capture/src/device_selection.h new file mode 100644 index 000000000..bcdc2ad8f --- /dev/null +++ b/electron/native/wgc-capture/src/device_selection.h @@ -0,0 +1,89 @@ +#pragma once + +#include +#include + +/** + * Which camera devices are already recording in this take. + * + * Two webcams of the same model report the same friendly name, and the id the + * browser hands us is a salted hash that never matches a device path, so name + * matching alone sends every such camera to the FIRST physical device. The + * second open then fails as busy and that camera is lost. Each camera that + * opens adds its device identity here, and later cameras of the take skip it. + * + * Identities are compared normalized (see `normalizeDeviceIdentity`), so a + * device opened through Media Foundation is recognized when it turns up again + * on the DirectShow fallback. + */ +class DeviceClaims { +public: + bool contains(const std::wstring& identity) const; + /** Ignores an empty identity: a device we cannot name cannot be claimed. */ + void add(const std::wstring& identity); + +private: + std::vector identities_; +}; + +/** One device a capture backend enumerated. */ +struct DeviceCandidate { + std::wstring name; + /** The Media Foundation symbolic link or the DirectShow DevicePath. */ + std::wstring identity; +}; + +/** + * The part of a device interface path that names the physical device. + * + * Media Foundation's symbolic link and DirectShow's DevicePath are the same + * `\\?\usb#vid_…##{interface class}\` string, except that + * each registers the camera under its own interface class GUID (measured on a + * Snapdragon front camera: KSCATEGORY_VIDEO_CAMERA against KSCATEGORY_VIDEO). + * Dropping the `#{…}` class and lowercasing leaves the device instance plus the + * reference string, which both share. The reference string is kept because it + * tells apart two cameras of one device (a colour and an IR sensor). A value + * without that shape (a moniker display name) is only lowercased. + */ +std::wstring normalizeDeviceIdentity(const std::wstring& identity); + +/** + * How well a candidate answers a requested name, or 0 for "not this one". + * + * Only decisive matches count: the names being equal once normalized, or one + * containing the other -- which is the ordinary case, since Chromium appends USB + * ids to what the driver reports. + * + * A further tier used to score shared WORDS, to bridge names differing more than + * that. It bridged names that were not the same device. "Logi Capture" and + * "Logitech StreamCam" share no word, yet "logi" sits inside "logitech" and that + * scored high enough to win -- so asking for a camera Media Foundation cannot + * enumerate opened a DIFFERENT camera, instead of returning nothing and letting + * the DirectShow fallback find the real one (getopenscreen/openscreen#405). + * + * Returning 0 is what makes that fallback reachable, so it is a real answer + * rather than a weak match. Keep this in step with + * `electron/recording/deviceNameMatching.ts`, which states the same rules for + * the Electron side and carries their unit tests. + */ +int deviceMatchScore( + const std::wstring& candidateName, + const std::wstring& candidateLink, + const std::wstring& requestedName, + const std::wstring& requestedId); + +/** + * The candidate a camera should open, or -1 for none. + * + * Claimed candidates are skipped; among the rest the best `deviceMatchScore` + * wins, the earlier one on a tie. With nothing requested that is the first + * unclaimed device. With a name or id requested, a candidate scoring 0 is never + * picked -- the same rule as before claims existed, and what lets the caller + * fall back to DirectShow. With an empty claim set this is exactly the old + * selection. + */ +int selectUnclaimedDevice( + const std::vector& candidates, + const std::wstring& requestedName, + const std::wstring& requestedId, + const DeviceClaims& claims); diff --git a/electron/native/wgc-capture/src/device_selection_test.cpp b/electron/native/wgc-capture/src/device_selection_test.cpp new file mode 100644 index 000000000..aa6329f66 --- /dev/null +++ b/electron/native/wgc-capture/src/device_selection_test.cpp @@ -0,0 +1,100 @@ +#include "device_selection.h" + +#include +#include +#include + +namespace { +int failures = 0; +void expect(bool ok, const char* label) { + if (!ok) { + std::printf("FAIL %s\n", label); + ++failures; + } +} +} // namespace + +int main() { + const std::vector twins = { + {L"USB Camera", L"\\\\?\\usb#vid_0c45&pid_6366&mi_00#7&aaa&0&0000#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + {L"USB Camera", L"\\\\?\\usb#vid_0c45&pid_6366&mi_00#7&bbb&0&0000#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + }; + + DeviceClaims none; + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", none) == 0, + "same names, nothing claimed: first"); + + DeviceClaims first; + first.add(twins[0].identity); + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", first) == 1, + "same names, first claimed: second"); + + DeviceClaims both; + both.add(twins[0].identity); + both.add(twins[1].identity); + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", both) == -1, + "same names, both claimed: none"); + expect(selectUnclaimedDevice(twins, L"", L"", both) == -1, + "nothing requested, both claimed: none"); + + DeviceClaims upper; + upper.add(L"\\\\?\\USB#VID_0C45&PID_6366&MI_00#7&AAA&0&0000#{E5323777-F976-4F5B-9B55-B94699C46E44}\\GLOBAL"); + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", upper) == 1, + "claim comparison is case-insensitive"); + + // DirectShow registers the same device under its own interface class GUID. + DeviceClaims viaDirectShow; + viaDirectShow.add( + L"\\\\?\\usb#vid_0c45&pid_6366&mi_00#7&aaa&0&0000#{65e8773d-8f56-11d0-a3b9-00a0c9223196}\\global"); + expect(viaDirectShow.contains(twins[0].identity), "MF link and DirectShow path are one device"); + expect(!viaDirectShow.contains(twins[1].identity), "the twin is a different device"); + + const std::vector mixed = { + {L"Studio Cam", L"\\\\?\\usb#vid_1#a#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + {L"Studio Cam Pro", L"\\\\?\\usb#vid_2#b#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + {L"Other Camera", L"\\\\?\\usb#vid_3#c#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + }; + DeviceClaims exact; + exact.add(mixed[0].identity); + expect(selectUnclaimedDevice(mixed, L"Studio Cam", L"", none) == 0, "exact name wins unclaimed"); + expect(selectUnclaimedDevice(mixed, L"Studio Cam", L"", exact) == 1, + "claimed exact match loses to a weaker unclaimed match"); + DeviceClaims bothStudio = exact; + bothStudio.add(mixed[1].identity); + expect(selectUnclaimedDevice(mixed, L"Studio Cam", L"", bothStudio) == -1, + "a non-matching device is never picked for a request"); + expect(selectUnclaimedDevice(mixed, L"Nonexistent", L"", none) == -1, + "no match for a request: none"); + expect(selectUnclaimedDevice(mixed, L"", L"", none) == 0, "nothing requested: first device"); + expect(selectUnclaimedDevice(mixed, L"", L"", exact) == 1, + "nothing requested: first unclaimed device"); + expect(selectUnclaimedDevice({}, L"", L"", none) == -1, "no candidates: none"); + + // Measured on an ARM64 dev machine: the front camera as Media Foundation and + // as DirectShow list it. + DeviceClaims frontCamera; + frontCamera.add( + L"\\\\?\\display#qcom_avstream_8380#3&2dd9d5f4&0&uid32768#{e5323777-f976-4f5b-9b55-b94699c46e44}" + L"\\{4faeafd4-041b-4e46-85fd-400473891182}"); + expect(frontCamera.contains( + L"\\\\?\\DISPLAY#QCOM_AVSTREAM_8380#3&2DD9D5F4&0&UID32768#{65E8773D-8F56-11D0-A3B9-00A0C9223196}" + L"\\{4FAEAFD4-041B-4E46-85FD-400473891182}"), + "measured MF link and DirectShow path are one device"); + expect(!frontCamera.contains( + L"\\\\?\\display#qcom_avstream_8380#3&2dd9d5f4&0&uid32768#{e5323777-f976-4f5b-9b55-b94699c46e44}" + L"\\{00000000-0000-0000-0000-000000000001}"), + "another reference string on the same device is another camera"); + + DeviceClaims empty; + empty.add(L""); + expect(!empty.contains(L""), "an empty identity is never claimed"); + expect(normalizeDeviceIdentity(L"@device:sw:{ABC}") == L"@device:sw:{abc}", + "identity without an interface class is only lowercased"); + + if (failures == 0) { + std::printf("device_selection_test: all assertions passed\n"); + return 0; + } + std::printf("device_selection_test: %d failure(s)\n", failures); + return 1; +} diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp index 915e3dc4a..837f70f68 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp @@ -2,6 +2,7 @@ #include "realtime_scheduling.h" #include "webcam_format.h" +#include "webcam_loss.h" #include #include @@ -109,6 +110,63 @@ std::array yuvToBgr(int y, int u, int v) { return {clampToByte(blue), clampToByte(green), clampToByte(red)}; } +std::wstring readPropertyString(IPropertyBag* properties, const wchar_t* name) { + VARIANT value; + VariantInit(&value); + std::wstring result; + if (SUCCEEDED(properties->Read(name, &value, nullptr)) && value.vt == VT_BSTR && value.bstrVal) { + result = value.bstrVal; + } + VariantClear(&value); + return result; +} + +/** + * The video input devices DirectShow lists for the filter `clsid`. + * + * The fallback opens its filter by CLSID, which says nothing about WHICH + * device that is. The moniker does: its DevicePath is the same interface path + * Media Foundation reports as the symbolic link, so a device already opened + * there is recognized here. A moniker without a DevicePath (a plain software + * filter) is named by its display name instead. + */ +std::vector enumerateDevicesForClsid(const CLSID& clsid) { + std::vector candidates; + Microsoft::WRL::ComPtr deviceEnumerator; + if (FAILED(CoCreateInstance(CLSID_SystemDeviceEnum, nullptr, CLSCTX_INPROC_SERVER, IID_PPV_ARGS(&deviceEnumerator)))) { + return candidates; + } + Microsoft::WRL::ComPtr monikers; + // S_FALSE: the category is empty. + if (deviceEnumerator->CreateClassEnumerator(CLSID_VideoInputDeviceCategory, &monikers, 0) != S_OK || !monikers) { + return candidates; + } + Microsoft::WRL::ComPtr moniker; + while (monikers->Next(1, &moniker, nullptr) == S_OK) { + Microsoft::WRL::ComPtr properties; + if (SUCCEEDED(moniker->BindToStorage(nullptr, nullptr, IID_PPV_ARGS(&properties)))) { + CLSID monikerClsid{}; + const std::wstring clsidText = readPropertyString(properties.Get(), L"CLSID"); + if (!clsidText.empty() && SUCCEEDED(CLSIDFromString(clsidText.c_str(), &monikerClsid)) && + IsEqualCLSID(monikerClsid, clsid)) { + DeviceCandidate candidate{ + readPropertyString(properties.Get(), L"FriendlyName"), + readPropertyString(properties.Get(), L"DevicePath")}; + if (candidate.identity.empty()) { + LPOLESTR displayName = nullptr; + if (SUCCEEDED(moniker->GetDisplayName(nullptr, nullptr, &displayName)) && displayName) { + candidate.identity = displayName; + CoTaskMemFree(displayName); + } + } + candidates.push_back(std::move(candidate)); + } + } + moniker.Reset(); + } + return candidates; +} + } // namespace struct DirectShowWebcamCapture::Impl { @@ -119,6 +177,8 @@ struct DirectShowWebcamCapture::Impl { Microsoft::WRL::ComPtr sampleGrabber; Microsoft::WRL::ComPtr nullRenderer; Microsoft::WRL::ComPtr mediaControl; + /** Where the graph says the device left; optional, see captureLoop. */ + Microsoft::WRL::ComPtr mediaEvent; bool comInitialized = false; bool running = false; }; @@ -218,6 +278,7 @@ bool DirectShowWebcamCapture::buildGraph( // Every attempt starts from empty filters. A RenderStream that fails can // leave pins connected behind it, and retrying on top of that half-built // graph is how you get a second failure that says nothing about the format. + impl_->mediaEvent.Reset(); impl_->mediaControl.Reset(); impl_->nullRenderer.Reset(); impl_->sampleGrabber.Reset(); @@ -294,7 +355,8 @@ bool DirectShowWebcamCapture::initialize( const std::wstring& directShowClsid, int requestedWidth, int requestedHeight, - int requestedFps) { + int requestedFps, + const DeviceClaims& claims) { (void)deviceId; stop(); delete impl_; @@ -321,6 +383,27 @@ bool DirectShowWebcamCapture::initialize( } selectedDeviceName_ = deviceName.empty() ? directShowClsid : deviceName; + // Every device this filter stands for has already been matched by name in + // Electron, so the only thing left to choose on is which one is free. With + // no moniker naming the filter, the CLSID itself is the identity. + const std::vector candidates = enumerateDevicesForClsid(selectedClsid); + if (candidates.empty()) { + deviceIdentity_ = directShowClsid; + if (claims.contains(deviceIdentity_)) { + std::cerr << "ERROR: DirectShow webcam filter is already recording in this take" << std::endl; + return false; + } + } else { + const int selectedIndex = selectUnclaimedDevice(candidates, L"", L"", claims); + if (selectedIndex < 0) { + std::cerr << "ERROR: Every DirectShow webcam for this filter is already recording in this take" + << std::endl; + return false; + } + deviceIdentity_ = candidates[selectedIndex].identity; + } + std::wcerr << L"INFO: DirectShow webcam device " << deviceIdentity_ << std::endl; + // The camera's own format first, a forced RGB32 conversion only if we cannot // read it. // @@ -355,6 +438,10 @@ bool DirectShowWebcamCapture::initialize( if (!succeeded(impl_->graph.As(&impl_->mediaControl), "QueryInterface(IMediaControl)")) { return false; } + // Best-effort: without it a lost device goes unnoticed, as it always did. + if (FAILED(impl_->graph.As(&impl_->mediaEvent))) { + impl_->mediaEvent.Reset(); + } return true; } @@ -467,6 +554,7 @@ void DirectShowWebcamCapture::stop() { impl_->mediaControl->Stop(); } impl_->running = false; + impl_->mediaEvent.Reset(); impl_->mediaControl.Reset(); impl_->nullRenderer.Reset(); impl_->sampleGrabber.Reset(); @@ -484,6 +572,11 @@ void DirectShowWebcamCapture::captureLoop() { const MmcssThread mmcss(L"Capture"); const HRESULT coinitHr = CoInitializeEx(nullptr, COINIT_MULTITHREADED); while (!stopRequested_ && impl_ && impl_->sampleGrabber) { + // The sample grabber keeps handing back its last buffer after the + // camera is gone, so the frames cannot tell; the graph's events can. + if (deviceLeftGraph()) { + break; + } long bufferSize = 0; HRESULT hr = impl_->sampleGrabber->GetCurrentBuffer(&bufferSize, nullptr); if (SUCCEEDED(hr) && bufferSize > 0) { @@ -500,6 +593,32 @@ void DirectShowWebcamCapture::captureLoop() { } } +bool DirectShowWebcamCapture::deviceLeftGraph() { + if (!impl_->mediaEvent) { + return false; + } + long code = 0; + LONG_PTR param1 = 0; + LONG_PTR param2 = 0; + // A zero timeout drains what is queued without waiting for more. + while (SUCCEEDED(impl_->mediaEvent->GetEvent(&code, ¶m1, ¶m2, 0))) { + impl_->mediaEvent->FreeEventParams(code, param1, param2); + // EC_DEVICE_LOST's second parameter is 0 when the device was removed, + // 1 when it came back; a stream error or abort stops the graph. + const bool lost = (code == EC_DEVICE_LOST && param2 == 0) || code == EC_ERRORABORT || + code == EC_STREAM_ERROR_STOPPED; + if (lost && !lost_.exchange(true)) { + reportWebcamLost(selectedDeviceName_, static_cast(code)); + return true; + } + } + return false; +} + +bool DirectShowWebcamCapture::isLost() const { + return lost_; +} + void DirectShowWebcamCapture::storeFrame(const BYTE* buffer, long length) { const int destinationStride = width_ * 4; const int sourceStride = sourceStride_ > 0 ? sourceStride_ : destinationStride; @@ -594,6 +713,10 @@ int DirectShowWebcamCapture::fps() const { return fps_; } +const std::wstring& DirectShowWebcamCapture::deviceIdentity() const { + return deviceIdentity_; +} + const std::wstring& DirectShowWebcamCapture::selectedDeviceName() const { return selectedDeviceName_; } diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.h b/electron/native/wgc-capture/src/dshow_webcam_capture.h index be4f759a5..5e26098d0 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.h +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.h @@ -1,5 +1,7 @@ #pragma once +#include "device_selection.h" + #include #include @@ -57,7 +59,8 @@ class DirectShowWebcamCapture { const std::wstring& directShowClsid, int requestedWidth, int requestedHeight, - int requestedFps); + int requestedFps, + const DeviceClaims& claims); bool start(); void stop(); bool copyLatestFrame(WebcamFrameSnapshot& destination, uint64_t lastSeenSequence); @@ -66,6 +69,16 @@ class DirectShowWebcamCapture { int height() const; int fps() const; const std::wstring& selectedDeviceName() const; + /** + * The opened device's DevicePath (the moniker's display name when it has + * none, the filter CLSID when no moniker names it), for the take's claims. + */ + const std::wstring& deviceIdentity() const; + /** + * Did the graph report the device gone (EC_DEVICE_LOST removal, a stream + * error, or an abort)? Latched; safe to ask from any thread. + */ + bool isLost() const; void storeFrame(const BYTE* buffer, long length); private: @@ -77,6 +90,8 @@ class DirectShowWebcamCapture { struct Impl; void captureLoop(); + /** Drains the graph's queued events; true once one of them says the device left. */ + bool deviceLeftGraph(); /** * Builds source -> sample grabber -> null renderer and connects it. * @@ -113,6 +128,7 @@ class DirectShowWebcamCapture { Impl* impl_ = nullptr; std::thread thread_; std::atomic stopRequested_ = false; + std::atomic lost_ = false; std::mutex frameMutex_; std::vector latestFrame_; uint64_t latestFrameSequence_ = 0; @@ -123,4 +139,5 @@ class DirectShowWebcamCapture { bool sourceTopDown_ = false; PixelFormat pixelFormat_ = PixelFormat::Bgra; std::wstring selectedDeviceName_; + std::wstring deviceIdentity_; }; diff --git a/electron/native/wgc-capture/src/json_fields.cpp b/electron/native/wgc-capture/src/json_fields.cpp new file mode 100644 index 000000000..e36a43e81 --- /dev/null +++ b/electron/native/wgc-capture/src/json_fields.cpp @@ -0,0 +1,125 @@ +#include "json_fields.h" + +#include +#include + +// Lightweight field lookups over the helper's single JSON argument. They match +// the first occurrence of a key anywhere in the document. + +bool findBool(const std::string& json, const std::string& key, bool fallback) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return fallback; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return fallback; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + if (json.compare(pos, 4, "true") == 0) { + return true; + } + if (json.compare(pos, 5, "false") == 0) { + return false; + } + return fallback; +} + +int64_t findInt64(const std::string& json, const std::string& key, int64_t fallback) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return fallback; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return fallback; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + try { + return std::stoll(json.substr(pos)); + } catch (...) { + return fallback; + } +} + +int findInt(const std::string& json, const std::string& key, int fallback) { + return static_cast(findInt64(json, key, fallback)); +} + +double findDouble(const std::string& json, const std::string& key, double fallback) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return fallback; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return fallback; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + try { + return std::stod(json.substr(pos)); + } catch (...) { + return fallback; + } +} + +std::string findString(const std::string& json, const std::string& key) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return {}; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return {}; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + if (pos >= json.size() || json[pos] != '"') { + return {}; + } + pos += 1; + + std::string result; + while (pos < json.size()) { + const char c = json[pos++]; + if (c == '"') { + break; + } + if (c == '\\' && pos < json.size()) { + const char escaped = json[pos++]; + switch (escaped) { + case '\\': + case '"': + case '/': + result.push_back(escaped); + break; + case 'n': + result.push_back('\n'); + break; + case 'r': + result.push_back('\r'); + break; + case 't': + result.push_back('\t'); + break; + default: + result.push_back(escaped); + break; + } + continue; + } + result.push_back(c); + } + return result; +} diff --git a/electron/native/wgc-capture/src/json_fields.h b/electron/native/wgc-capture/src/json_fields.h new file mode 100644 index 000000000..0921b3381 --- /dev/null +++ b/electron/native/wgc-capture/src/json_fields.h @@ -0,0 +1,10 @@ +#pragma once + +#include +#include + +bool findBool(const std::string& json, const std::string& key, bool fallback); +int64_t findInt64(const std::string& json, const std::string& key, int64_t fallback); +int findInt(const std::string& json, const std::string& key, int fallback); +double findDouble(const std::string& json, const std::string& key, double fallback); +std::string findString(const std::string& json, const std::string& key); diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index 1bad39699..77dc692db 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -8,7 +8,9 @@ #include "wasapi_loopback_capture.h" #include "wasapi_render_keepalive.h" #include "frame_visibility.h" +#include "json_fields.h" #include "webcam_capture.h" +#include "webcam_config.h" #include "wgc_session.h" #include @@ -26,6 +28,7 @@ #include #include #include +#include namespace { @@ -37,7 +40,6 @@ struct CaptureConfig { std::string sourceId; std::string windowHandle; std::string outputPath; - std::string webcamOutputPath; int fps = 60; int width = 0; int height = 0; @@ -138,6 +140,49 @@ struct CaptureControl { } }; +// One camera of the take: its capture, its encoder and the writer-loop state that used +// to be single variables. All cameras share the recording's T0. +struct WebcamStream { + WebcamConfig config; + // The camera's position in the config list, kept when an earlier camera is + // dropped, so every event names the camera the caller asked for. + size_t index = 0; + WebcamCapture capture; + MFEncoder encoder; + bool active = false; + // False only for the legacy inline picture-in-picture camera, which has no + // file of its own and is drawn into the screen frame instead. + bool writeSeparate = false; + std::vector latestFrame; + int latestWidth = 0; + int latestHeight = 0; + uint64_t latestSequence = 0; + bool hasVisibleFrame = false; + int64_t lastTimestampHns = -1; + int64_t nextWriteDueHns = 0; + int64_t nominalIntervalHns = 0; + // Captured in the writer's pull block, submitted after it (issue #115). + Microsoft::WRL::ComPtr pendingSample; + // Owned here because the shutdown watchdog reads the current step's name + // through a raw pointer from another thread. + std::string finalizeStepName; + + // Takes the camera's newest frame if it carries a picture. Returns whether it did. + bool pullVisibleFrame() { + WebcamFrameSnapshot candidate; + if (!capture.copyLatestFrame(candidate, latestSequence) || + !hasVisibleWebcamContent(candidate.data, capture.deliversNv12())) { + return false; + } + latestFrame = std::move(candidate.data); + latestWidth = candidate.width; + latestHeight = candidate.height; + latestSequence = candidate.sequence; + hasVisibleFrame = true; + return true; + } +}; + int readEnvInt(const char* name, int fallback) { char raw[32]{}; const DWORD length = GetEnvironmentVariableA(name, raw, static_cast(sizeof(raw))); @@ -396,122 +441,12 @@ void reportCaptureAdapters(ID3D11Device* device, HMONITOR targetMonitor) { } -bool findBool(const std::string& json, const std::string& key, bool fallback) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return fallback; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return fallback; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - if (json.compare(pos, 4, "true") == 0) { - return true; - } - if (json.compare(pos, 5, "false") == 0) { - return false; - } - return fallback; -} - -int64_t findInt64(const std::string& json, const std::string& key, int64_t fallback) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return fallback; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return fallback; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - try { - return std::stoll(json.substr(pos)); - } catch (...) { - return fallback; - } -} -int findInt(const std::string& json, const std::string& key, int fallback) { - return static_cast(findInt64(json, key, fallback)); -} - -double findDouble(const std::string& json, const std::string& key, double fallback) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return fallback; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return fallback; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - try { - return std::stod(json.substr(pos)); - } catch (...) { - return fallback; - } -} - -std::string findString(const std::string& json, const std::string& key) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return {}; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return {}; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - if (pos >= json.size() || json[pos] != '"') { - return {}; - } - pos += 1; - - std::string result; - while (pos < json.size()) { - const char c = json[pos++]; - if (c == '"') { - break; - } - if (c == '\\' && pos < json.size()) { - const char escaped = json[pos++]; - switch (escaped) { - case '\\': - case '"': - case '/': - result.push_back(escaped); - break; - case 'n': - result.push_back('\n'); - break; - case 'r': - result.push_back('\r'); - break; - case 't': - result.push_back('\t'); - break; - default: - result.push_back(escaped); - break; - } - continue; - } - result.push_back(c); - } - return result; +// `code` stays ahead of `index`: older readers match on the `"code":…` substring. +void printWebcamUnavailable(size_t index, const std::string& deviceName, const char* message) { + std::cout << "{\"event\":\"warning\",\"code\":\"webcam-unavailable\",\"index\":" << index + << ",\"deviceName\":\"" << jsonEscape(deviceName) << "\",\"message\":\"" << message << "\"}" + << std::endl; } std::string parseWindowHandleFromSourceId(const std::string& sourceId) { @@ -585,7 +520,6 @@ bool parseConfig(const std::string& json, CaptureConfig& config) { config.webcamDeviceId = findString(json, "webcamDeviceId"); config.webcamDeviceName = findString(json, "webcamDeviceName"); config.webcamDirectShowClsid = findString(json, "webcamDirectShowClsid"); - config.webcamOutputPath = findString(json, "webcamPath"); config.webcamWidth = findInt(json, "webcamWidth", 0); config.webcamHeight = findInt(json, "webcamHeight", 0); config.webcamFps = findInt(json, "webcamFps", 0); @@ -660,8 +594,9 @@ int wmain(int argc, wchar_t* argv[]) { winrt::init_apartment(winrt::apartment_type::multi_threaded); + const std::string configJson = wideToUtf8(argv[1]); CaptureConfig config; - if (!parseConfig(wideToUtf8(argv[1]), config)) { + if (!parseConfig(configJson, config)) { std::cerr << "ERROR: Failed to parse config JSON" << std::endl; return 1; } @@ -761,42 +696,109 @@ int wmain(int argc, wchar_t* argv[]) { const int pixels = width * height; const int bitrate = pixels >= 3840 * 2160 ? 45'000'000 : pixels >= 2560 * 1440 ? 28'000'000 : 18'000'000; - WebcamCapture webcamCapture; - bool webcamActive = false; - // Decided before initialize(), not after: it selects the capture pixel - // format, and only a camera going to its own file can use NV12 -- an inline - // picture-in-picture composite needs the frame as BGRA. - bool writeSeparateWebcam = config.webcamEnabled && !config.webcamOutputPath.empty(); - if (config.webcamEnabled) { - if (!webcamCapture.initialize( - utf8ToWide(config.webcamDeviceId), - utf8ToWide(config.webcamDeviceName), - utf8ToWide(config.webcamDirectShowClsid), - config.webcamWidth, - config.webcamHeight, - config.webcamFps > 0 ? config.webcamFps : config.fps, - writeSeparateWebcam)) { + // Every camera of the take, in config order. The `webcams` list (or the + // legacy fields with a webcamPath) gives cameras that each write their own + // file. The legacy fields without a webcamPath are the one camera drawn + // into the screen frame as an inline picture-in-picture. + std::vector webcamConfigs = parseWebcamConfigs(configJson); + if (webcamConfigs.empty() && config.webcamEnabled) { + WebcamConfig inlineCamera; + inlineCamera.deviceId = config.webcamDeviceId; + inlineCamera.deviceName = config.webcamDeviceName; + inlineCamera.directShowClsid = config.webcamDirectShowClsid; + inlineCamera.width = config.webcamWidth; + inlineCamera.height = config.webcamHeight; + inlineCamera.fps = config.webcamFps; + webcamConfigs.push_back(std::move(inlineCamera)); + } + + std::vector> webcams; // unique_ptr: WebcamCapture/MFEncoder are not movable + // The devices this take's cameras opened, so two cameras of the same model + // (same name, and a browser id that matches no device) open two devices + // rather than the first one twice. A camera dropped later keeps its claim. + DeviceClaims deviceClaims; + for (size_t index = 0; index < webcamConfigs.size(); ++index) { + auto stream = std::make_unique(); + stream->config = webcamConfigs[index]; + stream->index = index; + stream->finalizeStepName = "webcam-encoder-finalize-" + std::to_string(index); + // Decided before initialize(), not after: it selects the capture pixel + // format, and only a camera going to its own file can use NV12 -- an inline + // picture-in-picture composite needs the frame as BGRA. + stream->writeSeparate = !stream->config.outputPath.empty(); + if (!stream->capture.initialize( + utf8ToWide(stream->config.deviceId), + utf8ToWide(stream->config.deviceName), + utf8ToWide(stream->config.directShowClsid), + stream->config.width, + stream->config.height, + stream->config.fps > 0 ? stream->config.fps : config.fps, + stream->writeSeparate, + deviceClaims)) { // Non-fatal: a screen+audio recording the user can still use is far // better than losing the whole recording because one camera device // didn't match. Report it so the renderer can inform the user (and, // historically, fall back to a browser-recorded webcam sidecar), but - // let capture continue without a native webcam track. - std::cerr << "WARNING: Failed to initialize native webcam capture; continuing without webcam" - << std::endl; - std::cout << "{\"event\":\"warning\",\"code\":\"webcam-unavailable\",\"message\":" - "\"Failed to initialize native webcam capture\"}" - << std::endl; - config.webcamEnabled = false; - writeSeparateWebcam = false; - } else { - std::cout << "{\"event\":\"webcam-format\",\"schemaVersion\":2,\"width\":" << webcamCapture.width() - << ",\"height\":" << webcamCapture.height() - << ",\"fps\":" << webcamCapture.fps() - << ",\"deviceName\":\"" << jsonEscape(wideToUtf8(webcamCapture.selectedDeviceName())) - << "\"}" << std::endl; - // writeSeparateWebcam was decided above, before the pixel format. + // let capture continue without this camera's track. + std::cerr << "WARNING: Failed to initialize native webcam capture for camera " << index + << "; continuing without it" << std::endl; + printWebcamUnavailable( + index, stream->config.deviceName, "Failed to initialize native webcam capture"); + continue; } - } + std::cout << "{\"event\":\"webcam-format\",\"schemaVersion\":2,\"index\":" << index + << ",\"width\":" << stream->capture.width() + << ",\"height\":" << stream->capture.height() + << ",\"fps\":" << stream->capture.fps() + << ",\"deviceName\":\"" << jsonEscape(wideToUtf8(stream->capture.selectedDeviceName())) + << "\"}" << std::endl; + stream->nominalIntervalHns = + static_cast(10'000'000ULL / std::max(1, stream->capture.fps())); + webcams.push_back(std::move(stream)); + } + // Only the legacy single camera without a file of its own is composited + // into the screen frame; every listed camera writes separately. + WebcamStream* inlineWebcam = + webcams.size() == 1 && !webcams[0]->writeSeparate ? webcams[0].get() : nullptr; + // A camera that opened but whose encoder or capture then would not start + // costs that camera, not the take -- with the screen and up to four cameras, + // running out of hardware encoder sessions is the likeliest way to get here. + // Only before the video writer runs: the vector is not touched after that. + const auto dropWebcam = [&](WebcamStream& stream, const char* message) { + std::cerr << "WARNING: " << message << " for camera " << stream.index << "; continuing without it" + << std::endl; + printWebcamUnavailable(stream.index, stream.config.deviceName, message); + stream.capture.stop(); + if (stream.writeSeparate) { + // Balances whatever initialize() got through, then removes the + // file it may have created: nothing was ever written to it. + stream.encoder.finalize(); + // A file that never got created is fine; one that could not be + // removed is left as a 0-byte stub, which the app cleans up. + if (!DeleteFileW(utf8ToWide(stream.config.outputPath).c_str())) { + const DWORD deleteError = GetLastError(); + if (deleteError != ERROR_FILE_NOT_FOUND) { + std::cerr << "WARNING: Could not remove the dropped camera file " + << stream.config.outputPath << " (GetLastError=" << deleteError << ")" + << std::endl; + } + } + } + stream.active = false; + if (inlineWebcam == &stream) { + inlineWebcam = nullptr; + } + }; + const auto eraseDroppedWebcams = [&](const std::vector& dropped) { + webcams.erase( + std::remove_if( + webcams.begin(), + webcams.end(), + [&](const std::unique_ptr& stream) { + return std::find(dropped.begin(), dropped.end(), stream.get()) != dropped.end(); + }), + webcams.end()); + }; WasapiLoopbackCapture loopbackCapture; WasapiLoopbackCapture microphoneCapture; @@ -921,13 +923,13 @@ int wmain(int argc, wchar_t* argv[]) { // // The other two conditions are unchanged and still required: software // encoding and inline webcam PiP both need the frame in system memory, - // which the DXGI path does not produce. config.webcamEnabled, not - // webcamActive -- the latter is only set once webcam capture has started, - // well after this. + // which the DXGI path does not produce. Decided from the initialized + // cameras, not from `active` -- that is only set once webcam capture has + // started, well after this. encoderOptions.useDxgiInput = readEnvInt("OPENSCREEN_WGC_ENABLE_DXGI_INPUT", 0) == 1 && !config.preferSoftwareEncoder && - (!config.webcamEnabled || writeSeparateWebcam); + inlineWebcam == nullptr; MFEncoder encoder; if (!encoder.initialize( @@ -966,8 +968,11 @@ int wmain(int argc, wchar_t* argv[]) { // on stop. << ",\"videoEncoderRuntime\":\"" << encoder.videoEncoderRuntime() << "\"}" << std::endl; - MFEncoder webcamEncoder; - if (writeSeparateWebcam) { + std::vector droppedWebcams; + for (const auto& stream : webcams) { + if (!stream->writeSeparate) { + continue; + } MFEncoderOptions webcamEncoderOptions = encoderOptions; webcamEncoderOptions.injectDefaultSinkWriterFailureOnce = false; webcamEncoderOptions.useDxgiInput = false; @@ -976,26 +981,28 @@ int wmain(int argc, wchar_t* argv[]) { // above 640x480; now that the capture runs at the camera's real // resolution, 8 Mbit/s starves a 1440p or 2160p frame badly enough to // undo the extra pixels. The tiers mirror the screen ladder above. - const int webcamPixels = std::max(1, webcamCapture.width()) * std::max(1, webcamCapture.height()); + const int webcamPixels = + std::max(1, stream->capture.width()) * std::max(1, stream->capture.height()); const int webcamBitrate = webcamPixels >= 3840 * 2160 ? 40'000'000 : webcamPixels >= 2560 * 1440 ? 24'000'000 : webcamPixels >= 1920 * 1080 ? 16'000'000 : webcamPixels >= 1280 * 720 ? 8'000'000 : 4'000'000; - if (!webcamEncoder.initialize( - utf8ToWide(config.webcamOutputPath), - webcamCapture.width(), - webcamCapture.height(), - webcamCapture.fps(), + if (!stream->encoder.initialize( + utf8ToWide(stream->config.outputPath), + stream->capture.width(), + stream->capture.height(), + stream->capture.fps(), webcamBitrate, session.device(), session.context(), nullptr, webcamEncoderOptions)) { - std::cerr << "ERROR: Failed to initialize native webcam encoder" << std::endl; - return 1; + dropWebcam(*stream, "Failed to initialize native webcam encoder"); + droppedWebcams.push_back(stream.get()); } } + eraseDroppedWebcams(droppedWebcams); // By default, no mutex guards frame handoff: writeVideoFrames is the // only thread that ever touches WGC or latestFrameTexture. It pulls each @@ -1019,11 +1026,6 @@ int wmain(int argc, wchar_t* argv[]) { // them is the next bug report, and neither is worth a log line each. std::atomic contendedFrames = 0; Microsoft::WRL::ComPtr latestFrameTexture; - std::vector latestWebcamFrame; - int latestWebcamWidth = 0; - int latestWebcamHeight = 0; - uint64_t latestWebcamSequence = 0; - bool hasVisibleWebcamFrame = false; // Legacy-path-only state. frameMutex guards latestFrameTexture/ // legacyLatestFrameTimestampHns between WGC's callback thread (writer) @@ -1078,7 +1080,6 @@ int wmain(int argc, wchar_t* argv[]) { std::chrono::duration(1.0 / config.fps)); uint64_t frameIndex = 0; int64_t lastEncodedVideoTimestampHns = -1; - int64_t lastWebcamTimestampHns = -1; // Media Foundation's H.264 encoder MFT does not honor irregular input // sample times for a VFR source: it numbers output samples // sequentially at its configured nominal frame rate regardless of the @@ -1089,18 +1090,17 @@ int wmain(int argc, wchar_t* argv[]) { // webcam encoder on a real-time-paced cadence (duplicating the // latest available camera frame when the camera hasn't produced a // newer one yet), so "sample N is at N/fps" is actually correct. - int64_t nextWebcamWriteDueHns = 0; - const int64_t nominalWebcamIntervalHns = - static_cast(10'000'000ULL / std::max(1, webcamCapture.fps())); + // The cadence state is per camera: see WebcamStream. auto nextFrameDue = std::chrono::steady_clock::now(); int64_t firstFrameTimestampHns = -1; int64_t latestFrameTimestampHns = 0; while (!control.stopRequested && !encodeFailed) { Microsoft::WRL::ComPtr videoSample; - Microsoft::WRL::ComPtr webcamSample; bool hasVideoSample = false; - bool hasWebcamSample = false; + for (const auto& stream : webcams) { + stream->pendingSample.Reset(); + } // Whether the picture this tick encodes differs from the last one: // a new WGC frame, or a new camera frame drawn into it. The legacy // callback path cannot tell, so it always reads back. @@ -1177,22 +1177,35 @@ int wmain(int argc, wchar_t* argv[]) { continue; } } - if (webcamActive) { - WebcamFrameSnapshot candidateWebcamFrame; - if (webcamCapture.copyLatestFrame(candidateWebcamFrame, latestWebcamSequence) && - hasVisibleWebcamContent(candidateWebcamFrame.data, webcamCapture.deliversNv12())) { - latestWebcamFrame = std::move(candidateWebcamFrame.data); - latestWebcamWidth = candidateWebcamFrame.width; - latestWebcamHeight = candidateWebcamFrame.height; - latestWebcamSequence = candidateWebcamFrame.sequence; - hasVisibleWebcamFrame = true; - pictureChanged = pictureChanged || !writeSeparateWebcam; + // Screen first, then every camera, as before there was more than one. + for (const auto& stream : webcams) { + if (stream->active && stream->capture.isLost()) { + // Unplugged mid-take: its file ends here instead of + // repeating the last picture to the end, it is left out + // of `webcamPaths` and still finalized at stop. The + // screen and the other cameras go on. + std::cerr << "ERROR: Camera " << stream->index + << " was lost during the take; disabling it" << std::endl; + stream->active = false; + if (inlineWebcam == stream.get()) { + // Nor is its frozen picture drawn into the screen. + inlineWebcam = nullptr; + pictureChanged = true; + } + continue; + } + if (stream->active && stream->pullVisibleFrame()) { + pictureChanged = pictureChanged || !stream->writeSeparate; } } - const BgraFrameView webcamFrame{ - hasVisibleWebcamFrame && !latestWebcamFrame.empty() ? latestWebcamFrame.data() : nullptr, - latestWebcamWidth, - latestWebcamHeight, + // The frame composited into the screen picture, for the legacy + // inline camera only. + const BgraFrameView inlineWebcamFrame{ + inlineWebcam && inlineWebcam->hasVisibleFrame && !inlineWebcam->latestFrame.empty() + ? inlineWebcam->latestFrame.data() + : nullptr, + inlineWebcam ? inlineWebcam->latestWidth : 0, + inlineWebcam ? inlineWebcam->latestHeight : 0, }; const int64_t syntheticTimestampHns = static_cast((frameIndex * 10'000'000ULL) / config.fps); @@ -1210,16 +1223,24 @@ int wmain(int argc, wchar_t* argv[]) { frameTimestampHns = lastEncodedVideoTimestampHns + static_cast(10'000'000ULL / config.fps); } - if (writeSeparateWebcam && webcamFrame.data) { - // Anchor to the same recording-start origin as screen video/audio, - // using real elapsed host-clock time (not a synthetic frame-index - // clock) so a long recording can't accumulate clock-origin drift. - const auto elapsedSinceStart = std::chrono::steady_clock::now() - control.recordingStartedAt; - const int64_t elapsedHns = std::chrono::duration_cast< - std::chrono::duration>>(elapsedSinceStart) - .count(); - const int64_t targetElapsedHns = - std::max(0, elapsedHns - control.pausedDurationHns()); + // Anchor to the same recording-start origin as screen video/audio, + // using real elapsed host-clock time (not a synthetic frame-index + // clock) so a long recording can't accumulate clock-origin drift. + // Read once per tick: every camera of this tick shares it. + int64_t targetElapsedHns = -1; + for (const auto& stream : webcams) { + if (!stream->active || !stream->writeSeparate || !stream->hasVisibleFrame || + stream->latestFrame.empty()) { + continue; + } + if (targetElapsedHns < 0) { + const auto elapsedSinceStart = + std::chrono::steady_clock::now() - control.recordingStartedAt; + const int64_t elapsedHns = std::chrono::duration_cast< + std::chrono::duration>>(elapsedSinceStart) + .count(); + targetElapsedHns = std::max(0, elapsedHns - control.pausedDurationHns()); + } // The H.264 encoder MFT does not honor irregular per-sample // timestamps for a VFR source -- it numbers output samples // sequentially at its configured nominal rate regardless of the @@ -1228,35 +1249,40 @@ int wmain(int argc, wchar_t* argv[]) { // to feed the encoder *at* that nominal cadence, duplicating // the latest available camera frame when the camera hasn't // produced a newer one yet (VFR capture -> CFR encode resampling). - if (targetElapsedHns >= nextWebcamWriteDueHns) { - int64_t webcamTimestampHns = targetElapsedHns; - if (lastWebcamTimestampHns >= 0 && webcamTimestampHns <= lastWebcamTimestampHns) { - webcamTimestampHns = lastWebcamTimestampHns + nominalWebcamIntervalHns; - } - // Capture the sample here, but submit it to the sink - // writer OUTSIDE this block below (issue #115) so a - // slow WriteSample can't hold up the next frame pull. - hasWebcamSample = - webcamCapture.deliversNv12() - ? webcamEncoder.captureNv12Sample( - Nv12FrameView{ - webcamFrame.data, webcamFrame.width, webcamFrame.height}, - webcamTimestampHns, - webcamSample) - : webcamEncoder.captureBgraSample( - webcamFrame, webcamTimestampHns, webcamSample); - if (!hasWebcamSample) { - encodeFailed = true; - control.requestStop(); - break; - } - lastWebcamTimestampHns = webcamTimestampHns; - nextWebcamWriteDueHns += nominalWebcamIntervalHns; - if (nextWebcamWriteDueHns <= targetElapsedHns) { - // Fell behind (e.g. coming out of a pause, or a stall) -- - // resync to now instead of trying to catch up frame-by-frame. - nextWebcamWriteDueHns = targetElapsedHns + nominalWebcamIntervalHns; - } + if (targetElapsedHns < stream->nextWriteDueHns) { + continue; + } + int64_t webcamTimestampHns = targetElapsedHns; + if (stream->lastTimestampHns >= 0 && webcamTimestampHns <= stream->lastTimestampHns) { + webcamTimestampHns = stream->lastTimestampHns + stream->nominalIntervalHns; + } + const BgraFrameView webcamFrame{ + stream->latestFrame.data(), stream->latestWidth, stream->latestHeight}; + // Capture the sample here, but submit it to the sink + // writer OUTSIDE this block below (issue #115) so a + // slow WriteSample can't hold up the next frame pull. + const bool captured = + stream->capture.deliversNv12() + ? stream->encoder.captureNv12Sample( + Nv12FrameView{webcamFrame.data, webcamFrame.width, webcamFrame.height}, + webcamTimestampHns, + stream->pendingSample) + : stream->encoder.captureBgraSample( + webcamFrame, webcamTimestampHns, stream->pendingSample); + if (!captured) { + // One camera failing costs that camera, not the take. + std::cerr << "ERROR: Failed to capture a sample for camera " << stream->index + << "; disabling it" << std::endl; + stream->pendingSample.Reset(); + stream->active = false; + continue; + } + stream->lastTimestampHns = webcamTimestampHns; + stream->nextWriteDueHns += stream->nominalIntervalHns; + if (stream->nextWriteDueHns <= targetElapsedHns) { + // Fell behind (e.g. coming out of a pause, or a stall) -- + // resync to now instead of trying to catch up frame-by-frame. + stream->nextWriteDueHns = targetElapsedHns + stream->nominalIntervalHns; } } if (testStallReadbackMs > 0) { @@ -1294,7 +1320,7 @@ int wmain(int argc, wchar_t* argv[]) { captured = encoder.captureVideoSample( latestFrameTexture.Get(), frameTimestampHns, - !writeSeparateWebcam && webcamFrame.data ? &webcamFrame : nullptr, + inlineWebcamFrame.data ? &inlineWebcamFrame : nullptr, videoSample); } if (!captured) { @@ -1334,10 +1360,15 @@ int wmain(int argc, wchar_t* argv[]) { // Stop detection has nothing to do with this ordering -- that is // CaptureControl::stopMutex/stopCv, checked by the loop condition // above, unrelated to sample submission (issue #252). - if (hasWebcamSample && !webcamEncoder.submitVideoSample(webcamSample.Get())) { - encodeFailed = true; - control.requestStop(); - break; + for (const auto& stream : webcams) { + if (stream->pendingSample && !stream->encoder.submitVideoSample(stream->pendingSample.Get())) { + // Disables this camera only; the screen and the other + // cameras keep recording. + std::cerr << "ERROR: Failed to submit a sample for camera " << stream->index + << "; disabling it" << std::endl; + stream->active = false; + } + stream->pendingSample.Reset(); } if (hasVideoSample && !encoder.submitVideoSample(videoSample.Get())) { encodeFailed = true; @@ -1463,41 +1494,52 @@ int wmain(int argc, wchar_t* argv[]) { stopRenderKeepAliveIfActive(); return 1; } - if (config.webcamEnabled) { - if (!webcamCapture.start()) { - microphoneCapture.stop(); - loopbackCapture.stop(); - stopDeviceWatchIfActive(); - stopRenderKeepAliveIfActive(); - if (audioMixer) { - audioMixer->stop(); - } - std::cerr << "ERROR: Failed to start native webcam capture" << std::endl; - return 1; + const auto stopWebcamCaptures = [&]() { + for (const auto& stream : webcams) { + stream->capture.stop(); + } + }; + droppedWebcams.clear(); + for (const auto& stream : webcams) { + if (!stream->capture.start()) { + dropWebcam(*stream, "Failed to start native webcam capture"); + droppedWebcams.push_back(stream.get()); + continue; } - webcamActive = true; + stream->active = true; + } + eraseDroppedWebcams(droppedWebcams); + if (!webcams.empty()) { + // One 3 s budget for all cameras together, not 3 s each: they warm up + // in parallel, and the screen recording should not start later per camera. const auto webcamDeadline = std::chrono::steady_clock::now() + std::chrono::seconds(3); - while (std::chrono::steady_clock::now() < webcamDeadline && !hasVisibleWebcamFrame) { - WebcamFrameSnapshot candidateWebcamFrame; - if (webcamCapture.copyLatestFrame(candidateWebcamFrame, latestWebcamSequence) && - hasVisibleWebcamContent(candidateWebcamFrame.data, webcamCapture.deliversNv12())) { - latestWebcamFrame = std::move(candidateWebcamFrame.data); - latestWebcamWidth = candidateWebcamFrame.width; - latestWebcamHeight = candidateWebcamFrame.height; - latestWebcamSequence = candidateWebcamFrame.sequence; - hasVisibleWebcamFrame = true; + const auto allVisible = [&]() { + return std::all_of(webcams.begin(), webcams.end(), [](const auto& stream) { + return stream->hasVisibleFrame; + }); + }; + while (std::chrono::steady_clock::now() < webcamDeadline && !allVisible()) { + for (const auto& stream : webcams) { + if (!stream->hasVisibleFrame) { + stream->pullVisibleFrame(); + } + } + if (allVisible()) { break; } std::this_thread::sleep_for(std::chrono::milliseconds(20)); } - if (!hasVisibleWebcamFrame) { - std::cerr << "WARNING: Native webcam started but no visible frame was available before screen capture" - << std::endl; + for (const auto& stream : webcams) { + if (!stream->hasVisibleFrame) { + std::cerr << "WARNING: Native webcam " << stream->index + << " started but no visible frame was available before screen capture" + << std::endl; + } } } if (!session.start()) { - webcamCapture.stop(); + stopWebcamCaptures(); microphoneCapture.stop(); loopbackCapture.stop(); stopDeviceWatchIfActive(); @@ -1547,7 +1589,7 @@ int wmain(int argc, wchar_t* argv[]) { loopbackCapture.stop(); stopDeviceWatchIfActive(); stopRenderKeepAliveIfActive(); - webcamCapture.stop(); + stopWebcamCaptures(); if (audioMixer) { audioMixer->stop(); } @@ -1704,7 +1746,7 @@ int wmain(int argc, wchar_t* argv[]) { logStopStep("render-keepalive"); } beginStopStep("webcam", stepBudgetMs); - webcamCapture.stop(); + stopWebcamCaptures(); logStopStep("webcam"); beginStopStep("audio-mixer", stepBudgetMs); if (audioMixer) { @@ -1800,20 +1842,52 @@ int wmain(int argc, wchar_t* argv[]) { if (!encodeFailed && screenFinalized) { std::cout << "{\"event\":\"recording-stopped\",\"schemaVersion\":2,\"screenPath\":\"" << jsonEscape(config.outputPath) << "\""; - if (writeSeparateWebcam) { - std::cout << ",\"webcamPath\":\"" << jsonEscape(config.webcamOutputPath) << "\""; + // `webcamPath` stays for camera 0, as before; `webcamPaths` lists every + // camera still recording at stop, in index order. Both are printed + // before the camera files are finalized, for the reason above -- so they + // name the files that were being written, and a camera whose finalize + // fails below is reported on stderr, not removed from this list. + // `webcamPaths` is printed, possibly empty, whenever a camera wrote a + // file of its own: its presence is how the app knows a camera missing + // from it stopped early, rather than an older helper that never sent it. + std::vector recordedWebcams; + bool anySeparateWebcam = false; + for (const auto& stream : webcams) { + anySeparateWebcam = anySeparateWebcam || stream->writeSeparate; + if (stream->writeSeparate && stream->active) { + recordedWebcams.push_back(stream.get()); + } + } + if (!recordedWebcams.empty() && recordedWebcams.front()->index == 0) { + std::cout << ",\"webcamPath\":\"" << jsonEscape(recordedWebcams.front()->config.outputPath) + << "\""; + } + if (anySeparateWebcam) { + std::cout << ",\"webcamPaths\":["; + for (size_t i = 0; i < recordedWebcams.size(); ++i) { + std::cout << (i == 0 ? "\"" : ",\"") << jsonEscape(recordedWebcams[i]->config.outputPath) + << "\""; + } + std::cout << "]"; } std::cout << "}" << std::endl; std::cout << "Recording stopped. Output path: " << config.outputPath << std::endl; } + // Every camera that wrote a file is finalized, including one disabled + // mid-take: what it wrote before failing is still worth a playable index. bool webcamFinalized = true; - if (writeSeparateWebcam) { - beginStopStep("webcam-encoder-finalize", shutdownBudgetMs); - webcamFinalized = webcamEncoder.finalize(); - logStopStep("webcam-encoder-finalize"); - if (!webcamFinalized) { - std::cerr << "ERROR: Failed to finalize the webcam recording" << std::endl; + for (const auto& stream : webcams) { + if (!stream->writeSeparate) { + continue; + } + beginStopStep(stream->finalizeStepName.c_str(), shutdownBudgetMs); + const bool finalized = stream->encoder.finalize(); + logStopStep(stream->finalizeStepName.c_str()); + if (!finalized) { + std::cerr << "ERROR: Failed to finalize the webcam recording for camera " << stream->index + << std::endl; + webcamFinalized = false; } } diff --git a/electron/native/wgc-capture/src/mf_encoder.cpp b/electron/native/wgc-capture/src/mf_encoder.cpp index 9eac8c4b4..23bc69331 100644 --- a/electron/native/wgc-capture/src/mf_encoder.cpp +++ b/electron/native/wgc-capture/src/mf_encoder.cpp @@ -786,6 +786,7 @@ bool MFEncoder::initialize( if (!succeeded(MFStartup(MF_VERSION), "MFStartup")) { return false; } + mfStarted_ = true; if (useDxgiInput_ && !initializeDxgiPipeline()) { std::cerr << "WARNING: The GPU DXGI encode path is unavailable on this machine; " @@ -1984,6 +1985,9 @@ bool MFEncoder::finalize() { captureDevice_.Reset(); context_.Reset(); device_.Reset(); - MFShutdown(); + if (mfStarted_) { + MFShutdown(); + mfStarted_ = false; + } return ok; } diff --git a/electron/native/wgc-capture/src/mf_encoder.h b/electron/native/wgc-capture/src/mf_encoder.h index 455624135..4c2034f87 100644 --- a/electron/native/wgc-capture/src/mf_encoder.h +++ b/electron/native/wgc-capture/src/mf_encoder.h @@ -269,6 +269,12 @@ class MFEncoder { int64_t firstTimestampHns_ = -1; int64_t lastTimestampHns_ = -1; bool finalized_ = false; + // Whether initialize() got as far as a successful MFStartup(). finalize() + // may only balance a startup this encoder made: MF's startup count is + // process-wide, and an encoder that was never initialized (a dropped + // camera's) would otherwise shut Media Foundation down under every other + // encoder and camera still running. + bool mfStarted_ = false; bool useDxgiInput_ = false; const char* videoEncoderSelection_ = kVideoEncoderSelectionDefault; const char* videoEncoderRuntime_ = kVideoEncoderRuntimeUnknown; diff --git a/electron/native/wgc-capture/src/mf_encoder_color_test.cpp b/electron/native/wgc-capture/src/mf_encoder_color_test.cpp index 1cc20d6b0..ac33a5e1b 100644 --- a/electron/native/wgc-capture/src/mf_encoder_color_test.cpp +++ b/electron/native/wgc-capture/src/mf_encoder_color_test.cpp @@ -15,12 +15,14 @@ #include #include +#include #include #include #include #include #include #include +#include #include namespace { @@ -358,10 +360,87 @@ void checkRepeatedFrames(ID3D11Device* device, ID3D11DeviceContext* context) { } } +// An encoder that was never initialized -- a camera the helper dropped before +// its encoder was set up -- must not touch Media Foundation when it is +// finalized or destroyed. Its finalize() used to call an unmatched MFShutdown(), +// which in the helper left every other encoder writing nothing (Finalize: +// MF_E_SINK_NO_SAMPLES_PROCESSED). With an initialized encoder alive, as here, +// that takes MF's count to zero under a live sink writer, and the live encoder +// then blocks instead of failing -- so the whole sequence runs on its own thread +// with a deadline, and a regression fails this test instead of hanging the +// build. Needs no ffmpeg. +void checkUninitializedEncoderLeavesOthersRunning() { + Microsoft::WRL::ComPtr device; + Microsoft::WRL::ComPtr context; + if (FAILED(D3D11CreateDevice( + nullptr, D3D_DRIVER_TYPE_HARDWARE, nullptr, D3D11_CREATE_DEVICE_BGRA_SUPPORT, nullptr, 0, + D3D11_SDK_VERSION, &device, nullptr, &context)) && + FAILED(D3D11CreateDevice( + nullptr, D3D_DRIVER_TYPE_WARP, nullptr, D3D11_CREATE_DEVICE_BGRA_SUPPORT, nullptr, 0, + D3D11_SDK_VERSION, &device, nullptr, &context))) { + skip("uninitialized-encoder-leaves-mf-running", "no D3D11 device"); + return; + } + char tempDir[MAX_PATH]{}; + GetTempPathA(MAX_PATH, tempDir); + const std::string path = std::string(tempDir) + "openscreen-mf-encoder-uninitialized-peer.mp4"; + DeleteFileA(path.c_str()); + + // 0 running, 1 passed, 2 failed, 3 skipped. + std::atomic outcome = 0; + std::thread worker([&] { + CoInitializeEx(nullptr, COINIT_MULTITHREADED); + { + MFEncoder live; + if (!live.initialize( + widen(path), kWidth, kHeight, 30, 2'000'000, device.Get(), context.Get(), nullptr, {})) { + outcome = 3; + } else { + { + MFEncoder neverInitialized; + neverInitialized.finalize(); + } // and destroyed, which finalizes again + Microsoft::WRL::ComPtr probe; + std::vector bgra(static_cast(kWidth) * kHeight * 4, 0x80); + const BgraFrameView frame{bgra.data(), kWidth, kHeight}; + bool wrote = SUCCEEDED(MFCreateSample(&probe)); + for (int i = 0; i < kFrames && wrote; i += 1) { + Microsoft::WRL::ComPtr sample; + wrote = live.captureBgraSample(frame, static_cast(i) * 333'333, sample) && + live.submitVideoSample(sample.Get()); + } + outcome = wrote && live.finalize() ? 1 : 2; + } + } + CoUninitialize(); + }); + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(30); + while (outcome == 0 && std::chrono::steady_clock::now() < deadline) { + std::this_thread::sleep_for(std::chrono::milliseconds(20)); + } + if (outcome == 0) { + expect("uninitialized-encoder-leaves-mf-running", false, "the live encoder blocked in Media Foundation"); + std::cout << "ran " << g_ran << " tests\n" << g_failed << " failed" << std::endl; + // The blocked thread can be neither joined nor safely abandoned. + TerminateProcess(GetCurrentProcess(), 1); + } + worker.join(); + DeleteFileA(path.c_str()); + if (outcome == 3) { + skip("uninitialized-encoder-leaves-mf-running", "encoder initialize failed on this host"); + return; + } + expect("uninitialized-encoder-leaves-mf-running", outcome == 1, "write or finalize failed"); +} + } // namespace int main() { checkConverterAgainstReference(); + if (SUCCEEDED(CoInitializeEx(nullptr, COINIT_MULTITHREADED))) { + checkUninitializedEncoderLeavesOthersRunning(); + CoUninitialize(); + } if (!toolsAvailable()) { skip("mf-encoder-color", "ffprobe/ffmpeg not on PATH"); return g_failed == 0 ? 0 : 1; diff --git a/electron/native/wgc-capture/src/webcam_capture.cpp b/electron/native/wgc-capture/src/webcam_capture.cpp index 2a367c67d..45d2ebdd7 100644 --- a/electron/native/wgc-capture/src/webcam_capture.cpp +++ b/electron/native/wgc-capture/src/webcam_capture.cpp @@ -2,6 +2,7 @@ #include "realtime_scheduling.h" #include "webcam_format.h" +#include "webcam_loss.h" #include #include @@ -9,8 +10,8 @@ #include #include -#include #include +#include namespace { @@ -36,114 +37,6 @@ std::wstring readAllocatedString(IMFActivate* activate, REFGUID key) { return result; } -/** - * Does one of these appear inside the other as WHOLE WORDS? - * - * Plain containment answered for devices that merely share a spelling: a - * requested "Logi" is inside "Logitech", and "Micro" inside "Microphone", - * neither of them as a word. Matching on that resolved a camera nobody asked - * for -- and resolving one is exactly what stops the request reaching the - * DirectShow fallback, where the cameras Media Foundation cannot enumerate live. - * - * Both sides arrive normalized, so a boundary is the start of the string, its - * end, or a space. - */ -bool containsAsWords(const std::wstring& haystack, const std::wstring& needle) { - if (haystack.empty() || needle.empty()) { - return false; - } - size_t pos = haystack.find(needle); - while (pos != std::wstring::npos) { - const bool startsOnBoundary = pos == 0 || haystack[pos - 1] == L' '; - const size_t after = pos + needle.size(); - const bool endsOnBoundary = after == haystack.size() || haystack[after] == L' '; - if (startsOnBoundary && endsOnBoundary) { - return true; - } - pos = haystack.find(needle, pos + 1); - } - return false; -} - -bool containsInsensitive(const std::wstring& haystack, const std::wstring& needle) { - return containsAsWords(haystack, needle) || containsAsWords(needle, haystack); -} - -std::wstring normalizeDeviceName(const std::wstring& value) { - std::wstring normalized; - normalized.reserve(value.size()); - bool lastWasSpace = true; - for (const wchar_t ch : value) { - if (std::iswalnum(ch)) { - normalized.push_back(static_cast(std::towlower(ch))); - lastWasSpace = false; - continue; - } - if (!lastWasSpace) { - normalized.push_back(L' '); - lastWasSpace = true; - } - } - while (!normalized.empty() && normalized.back() == L' ') { - normalized.pop_back(); - } - return normalized; -} - -/** - * How well a candidate answers a requested name, or 0 for "not this one". - * - * Only decisive matches count: the names being equal once normalized, or one - * containing the other -- which is the ordinary case, since Chromium appends USB - * ids to what the driver reports. - * - * A further tier used to score shared WORDS, to bridge names differing more than - * that. It bridged names that were not the same device. "Logi Capture" and - * "Logitech StreamCam" share no word, yet "logi" sits inside "logitech" and that - * scored high enough to win -- so asking for a camera Media Foundation cannot - * enumerate opened a DIFFERENT camera, instead of returning nothing and letting - * the DirectShow fallback find the real one (getopenscreen/openscreen#405). - * - * Returning 0 is what makes that fallback reachable, so it is a real answer - * rather than a weak match. Keep this in step with - * `electron/recording/deviceNameMatching.ts`, which states the same rules for - * the Electron side and carries their unit tests. - */ -int deviceMatchScore( - const std::wstring& candidateName, - const std::wstring& candidateLink, - const std::wstring& requestedName, - const std::wstring& requestedId) { - int score = 0; - const auto normalizedName = normalizeDeviceName(candidateName); - const auto normalizedLink = normalizeDeviceName(candidateLink); - const auto normalizedRequestedName = normalizeDeviceName(requestedName); - const auto normalizedRequestedId = normalizeDeviceName(requestedId); - - if (!normalizedRequestedName.empty()) { - if (normalizedName == normalizedRequestedName) { - score = std::max(score, 1000); - } - if (containsInsensitive(normalizedName, normalizedRequestedName)) { - score = std::max(score, 900); - } - if (containsInsensitive(normalizedLink, normalizedRequestedName)) { - score = std::max(score, 800); - } - } - - if (!normalizedRequestedId.empty()) { - if (containsInsensitive(normalizedLink, normalizedRequestedId)) { - score = std::max(score, 700); - } - if (containsInsensitive(normalizedName, normalizedRequestedId)) { - score = std::max(score, 600); - } - } - - return score; -} - } // namespace WebcamCapture::~WebcamCapture() { @@ -157,53 +50,48 @@ bool WebcamCapture::initialize( int requestedWidth, int requestedHeight, int requestedFps, - bool preferNv12) { + bool preferNv12, + DeviceClaims& claims) { fps_ = std::clamp(requestedFps > 0 ? requestedFps : 30, 1, 60); usingDirectShow_ = false; - selectedMatchScore_ = 0; - if (!succeeded(MFStartup(MF_VERSION), "MFStartup(webcam)")) { - if (directShowCapture_.initialize(deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_)) { - usingDirectShow_ = true; - return true; + selectedIdentity_.clear(); + const auto initializeDirectShow = [&]() { + if (!directShowCapture_.initialize( + deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_, claims)) { + if (!deviceId.empty() || !deviceName.empty()) { + std::cerr << "ERROR: Requested webcam device was not found by native Windows webcam providers" + << std::endl; + } + return false; } - return false; + usingDirectShow_ = true; + claims.add(directShowCapture_.deviceIdentity()); + return true; + }; + + if (!succeeded(MFStartup(MF_VERSION), "MFStartup(webcam)")) { + return initializeDirectShow(); } mfStarted_ = true; - if (!selectDevice(deviceId, deviceName)) { + if (!selectDevice(deviceId, deviceName, claims)) { if (mfStarted_) { MFShutdown(); mfStarted_ = false; } - if (directShowCapture_.initialize(deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_)) { - usingDirectShow_ = true; - return true; - } - return false; + return initializeDirectShow(); } - if ((!deviceId.empty() || !deviceName.empty()) && selectedMatchScore_ <= 0) { - if (mediaSource_) { - mediaSource_->Shutdown(); - } - sourceReader_.Reset(); - mediaSource_.Reset(); - if (mfStarted_) { - MFShutdown(); - mfStarted_ = false; - } - if (directShowCapture_.initialize(deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_)) { - usingDirectShow_ = true; - return true; - } - std::cerr << "ERROR: Requested webcam device was not found by native Windows webcam providers" - << std::endl; + if (!configureReader(requestedWidth, requestedHeight, fps_, preferNv12)) { return false; } - - return configureReader(requestedWidth, requestedHeight, fps_, preferNv12); + claims.add(selectedIdentity_); + return true; } -bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstring& deviceName) { +bool WebcamCapture::selectDevice( + const std::wstring& deviceId, + const std::wstring& deviceName, + const DeviceClaims& claims) { Microsoft::WRL::ComPtr attributes; if (!succeeded(MFCreateAttributes(&attributes, 1), "MFCreateAttributes(webcam enumeration)")) { return false; @@ -226,34 +114,40 @@ bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstrin return false; } - UINT32 selectedIndex = 0; - int bestScore = 0; + std::vector candidates; + candidates.reserve(deviceCount); + bool anyClaimed = false; for (UINT32 index = 0; index < deviceCount; index += 1) { - const std::wstring name = readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_FRIENDLY_NAME); - const std::wstring symbolicLink = readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_SOURCE_TYPE_VIDCAP_SYMBOLIC_LINK); - const int score = deviceMatchScore(name, symbolicLink, deviceName, deviceId); - std::wcerr << L"INFO: Native webcam candidate [" << index << L"] name=\"" << name << L"\" score=" << score << std::endl; - if (score > bestScore) { - selectedIndex = index; - bestScore = score; - } - } - - if ((!deviceId.empty() || !deviceName.empty()) && bestScore <= 0) { + DeviceCandidate candidate{ + readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_FRIENDLY_NAME), + readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_SOURCE_TYPE_VIDCAP_SYMBOLIC_LINK)}; + const bool claimed = claims.contains(candidate.identity); + anyClaimed = anyClaimed || claimed; + std::wcerr << L"INFO: Native webcam candidate [" << index << L"] name=\"" << candidate.name + << L"\" score=" << deviceMatchScore(candidate.name, candidate.identity, deviceName, deviceId) + << (claimed ? L" (already recording in this take)" : L"") << std::endl; + candidates.push_back(std::move(candidate)); + } + + const int selectedIndex = selectUnclaimedDevice(candidates, deviceName, deviceId, claims); + if (selectedIndex >= 0) { + selectedDeviceName_ = candidates[selectedIndex].name; + selectedIdentity_ = candidates[selectedIndex].identity; + hr = devices[selectedIndex]->ActivateObject(IID_PPV_ARGS(&mediaSource_)); + } else if (anyClaimed) { + std::cerr << "WARNING: Every matching webcam is already recording in this take; trying DirectShow" + << std::endl; + } else { std::cerr << "WARNING: Requested webcam device was not found by Media Foundation; trying DirectShow" << std::endl; } - selectedMatchScore_ = bestScore; - selectedDeviceName_ = readAllocatedString(devices[selectedIndex], MF_DEVSOURCE_ATTRIBUTE_FRIENDLY_NAME); - hr = devices[selectedIndex]->ActivateObject(IID_PPV_ARGS(&mediaSource_)); - for (UINT32 index = 0; index < deviceCount; index += 1) { devices[index]->Release(); } CoTaskMemFree(devices); - return succeeded(hr, "ActivateObject(webcam)"); + return selectedIndex >= 0 && succeeded(hr, "ActivateObject(webcam)"); } namespace { @@ -511,6 +405,9 @@ void WebcamCapture::captureLoop() { CoInitializeEx(nullptr, COINIT_MULTITHREADED); const auto loopStartedAt = std::chrono::steady_clock::now(); + // Start of the current run of failed reads; empty while reads succeed. + std::optional failureRunStartedAt; + while (!stopRequested_) { DWORD streamIndex = 0; DWORD flags = 0; @@ -536,11 +433,27 @@ void WebcamCapture::captureLoop() { // wise write thousands of identical lines over a long take. readFailures_ += 1; lastReadFailure_ = hr; + const auto now = std::chrono::steady_clock::now(); + if (!failureRunStartedAt) { + failureRunStartedAt = now; + } + const int64_t failingForMs = + std::chrono::duration_cast(now - *failureRunStartedAt).count(); + if (isWebcamLossResult(hr, false, failingForMs)) { + markLost(hr); + break; + } std::this_thread::sleep_for(std::chrono::milliseconds(20)); continue; } + failureRunStartedAt.reset(); if ((flags & MF_SOURCE_READERF_ENDOFSTREAM) != 0) { sawEndOfStream_ = true; + // Stop has its own way out of this loop; end of stream before it + // is the camera leaving. + if (!stopRequested_ && isWebcamLossResult(hr, true, 0)) { + markLost(hr); + } break; } if (!sample) { @@ -621,6 +534,20 @@ void WebcamCapture::captureLoop() { CoUninitialize(); } +void WebcamCapture::markLost(HRESULT hr) { + if (lost_.exchange(true)) { + return; + } + reportWebcamLost(selectedDeviceName_, hr); +} + +bool WebcamCapture::isLost() const { + if (usingDirectShow_) { + return directShowCapture_.isLost(); + } + return lost_; +} + bool WebcamCapture::copyLatestFrame(WebcamFrameSnapshot& destination, uint64_t lastSeenSequence) { if (usingDirectShow_) { return directShowCapture_.copyLatestFrame(destination, lastSeenSequence); diff --git a/electron/native/wgc-capture/src/webcam_capture.h b/electron/native/wgc-capture/src/webcam_capture.h index 949282bf9..5888e5ed9 100644 --- a/electron/native/wgc-capture/src/webcam_capture.h +++ b/electron/native/wgc-capture/src/webcam_capture.h @@ -1,5 +1,6 @@ #pragma once +#include "device_selection.h" #include "dshow_webcam_capture.h" #include @@ -22,6 +23,14 @@ class WebcamCapture { WebcamCapture(const WebcamCapture&) = delete; WebcamCapture& operator=(const WebcamCapture&) = delete; + /** + * Opens the requested camera, skipping devices in `claims`. + * + * `claims` belongs to the take: every camera of it is initialized against + * the same set, in config order, and one that opens adds its device. That + * is what sends a second camera of the same model to the second device + * instead of the busy first one. An empty set selects exactly as before. + */ bool initialize( const std::wstring& deviceId, const std::wstring& deviceName, @@ -29,7 +38,8 @@ class WebcamCapture { int requestedWidth, int requestedHeight, int requestedFps, - bool preferNv12); + bool preferNv12, + DeviceClaims& claims); bool start(); void stop(); bool copyLatestFrame(WebcamFrameSnapshot& destination, uint64_t lastSeenSequence); @@ -46,17 +56,30 @@ class WebcamCapture { */ bool deliversNv12() const; const std::wstring& selectedDeviceName() const; + /** + * Has the camera left the take -- unplugged, or its driver given up? + * + * Latched: once true it stays true, and the camera delivers no more + * frames. Safe to ask from any thread. + */ + bool isLost() const; private: - bool selectDevice(const std::wstring& deviceId, const std::wstring& deviceName); + bool selectDevice( + const std::wstring& deviceId, + const std::wstring& deviceName, + const DeviceClaims& claims); bool configureReader(int requestedWidth, int requestedHeight, int requestedFps, bool preferNv12); void captureLoop(); + /** Latches `lost_` and says so once; see `isWebcamLossResult`. */ + void markLost(HRESULT hr); Microsoft::WRL::ComPtr mediaSource_; Microsoft::WRL::ComPtr sourceReader_; DirectShowWebcamCapture directShowCapture_; std::thread thread_; std::atomic stopRequested_ = false; + std::atomic lost_ = false; std::mutex frameMutex_; std::vector latestFrame_; uint64_t latestFrameSequence_ = 0; @@ -86,6 +109,7 @@ class WebcamCapture { /** Where the loop's wall time goes, in microseconds. */ uint64_t readSampleUs_ = 0; uint64_t storeUs_ = 0; - int selectedMatchScore_ = 0; std::wstring selectedDeviceName_; + /** The symbolic link of the device this capture opened, for the take's claims. */ + std::wstring selectedIdentity_; }; diff --git a/electron/native/wgc-capture/src/webcam_config.cpp b/electron/native/wgc-capture/src/webcam_config.cpp new file mode 100644 index 000000000..d437b9d26 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_config.cpp @@ -0,0 +1,92 @@ +#include "webcam_config.h" + +#include "json_fields.h" + +#include + +namespace { + +// The find* helpers search the whole document, so each array entry is sliced +// out first to keep its fields from being confused with the top-level ones. +std::vector splitTopLevelObjects(const std::string& json, const std::string& arrayKey) { + std::vector objects; + size_t pos = json.find("\"" + arrayKey + "\""); + if (pos == std::string::npos) { + return objects; + } + pos = json.find('[', pos); + if (pos == std::string::npos) { + return objects; + } + ++pos; + + int depth = 0; + bool inString = false; + size_t objectStart = 0; + for (; pos < json.size(); ++pos) { + const char c = json[pos]; + if (inString) { + if (c == '\\') { + ++pos; + } else if (c == '"') { + inString = false; + } + continue; + } + if (c == '"') { + inString = true; + } else if (c == '{') { + if (depth == 0) { + objectStart = pos; + } + ++depth; + } else if (c == '}') { + if (depth > 0 && --depth == 0) { + objects.push_back(json.substr(objectStart, pos - objectStart + 1)); + } + } else if (c == ']' && depth == 0) { + break; + } + } + return objects; +} + +} // namespace + +std::vector parseWebcamConfigs(const std::string& json) { + std::vector configs; + for (const std::string& entry : splitTopLevelObjects(json, "webcams")) { + if (configs.size() >= kMaxWebcams) { + break; + } + WebcamConfig config; + config.outputPath = findString(entry, "camPath"); + if (config.outputPath.empty()) { + continue; + } + config.deviceId = findString(entry, "camDeviceId"); + config.deviceName = findString(entry, "camDeviceName"); + config.directShowClsid = findString(entry, "camClsid"); + config.width = findInt(entry, "camWidth", 0); + config.height = findInt(entry, "camHeight", 0); + config.fps = findInt(entry, "camFps", 0); + configs.push_back(std::move(config)); + } + if (!configs.empty() || !findBool(json, "webcamEnabled", false)) { + return configs; + } + + WebcamConfig legacy; + legacy.outputPath = findString(json, "webcamPath"); + if (legacy.outputPath.empty()) { + return configs; + } + legacy.deviceId = findString(json, "webcamDeviceId"); + legacy.deviceName = findString(json, "webcamDeviceName"); + legacy.directShowClsid = findString(json, "webcamDirectShowClsid"); + legacy.width = findInt(json, "webcamWidth", 0); + legacy.height = findInt(json, "webcamHeight", 0); + legacy.fps = findInt(json, "webcamFps", 0); + configs.push_back(std::move(legacy)); + return configs; +} diff --git a/electron/native/wgc-capture/src/webcam_config.h b/electron/native/wgc-capture/src/webcam_config.h new file mode 100644 index 000000000..b0c0998ea --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_config.h @@ -0,0 +1,17 @@ +#pragma once + +#include +#include +#include + +struct WebcamConfig { + std::string deviceId, deviceName, directShowClsid, outputPath; + int width = 0, height = 0, fps = 0; +}; + +constexpr size_t kMaxWebcams = 4; + +// The cameras to record, in order. Reads the `webcams` list when present and +// non-empty; otherwise the legacy single-camera fields (webcamEnabled, webcamDeviceId, +// webcamDeviceName, webcamDirectShowClsid, webcamWidth/Height/Fps, webcamPath). +std::vector parseWebcamConfigs(const std::string& json); diff --git a/electron/native/wgc-capture/src/webcam_config_test.cpp b/electron/native/wgc-capture/src/webcam_config_test.cpp new file mode 100644 index 000000000..1696698bf --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_config_test.cpp @@ -0,0 +1,63 @@ +#include "webcam_config.h" + +#include +#include + +namespace { +int failures = 0; +void expect(bool ok, const char* label) { + if (!ok) { + std::printf("FAIL %s\n", label); + ++failures; + } +} +} // namespace + +int main() { + const std::string legacy = + R"({"fps":60,"width":2560,"height":1440,"webcamEnabled":true,"webcamDeviceId":"id1",)" + R"("webcamDeviceName":"Cam One","webcamDirectShowClsid":"{A}","webcamWidth":1920,)" + R"("webcamHeight":1080,"webcamFps":30,"outputs":{"screenPath":"s.mp4","webcamPath":"w.mp4"}})"; + auto l = parseWebcamConfigs(legacy); + expect(l.size() == 1, "legacy: one camera"); + expect(l.size() == 1 && l[0].deviceName == "Cam One" && l[0].outputPath == "w.mp4" && + l[0].width == 1920 && l[0].fps == 30, + "legacy: fields"); + + expect(parseWebcamConfigs(R"({"webcamEnabled":false,"webcamPath":"w.mp4"})").empty(), + "legacy disabled: none"); + + const std::string list = + R"({"fps":60,"width":2560,"webcamEnabled":true,"webcamDeviceName":"Cam One","webcamPath":"w.mp4",)" + R"("webcams":[{"camDeviceId":"id1","camDeviceName":"Cam One","camClsid":"{A}","camWidth":1920,)" + R"("camHeight":1080,"camFps":30,"camPath":"w.mp4"},)" + R"({"camDeviceId":"id2","camDeviceName":"Desk \"Cam\"","camClsid":"","camWidth":1280,)" + R"("camHeight":720,"camFps":30,"camPath":"w-2.mp4"}]})"; + auto m = parseWebcamConfigs(list); + expect(m.size() == 2, "list: two cameras"); + expect(m.size() == 2 && m[1].deviceName == "Desk \"Cam\"" && m[1].outputPath == "w-2.mp4" && + m[1].width == 1280 && m[1].height == 720, + "list: second camera fields, escaped quote"); + expect(m.size() == 2 && m[0].fps == 30, "list: entry fps not the top-level fps"); + + expect(parseWebcamConfigs(R"({"webcamEnabled":true,"webcamPath":"w.mp4","webcams":[]})").size() == 1, + "empty list falls back to legacy"); + + std::string five = R"({"webcams":[)"; + for (int i = 0; i < 5; ++i) { + five += (i ? "," : ""); + five += R"({"camDeviceName":"C","camPath":"p)" + std::to_string(i) + R"(.mp4"})"; + } + five += "]}"; + expect(parseWebcamConfigs(five).size() == 4, "capped at four"); + + expect(parseWebcamConfigs(R"({"webcams":[{"camDeviceName":"No path"}]})").empty(), + "entry without camPath is skipped"); + + if (failures == 0) { + std::printf("webcam_config_test: all assertions passed\n"); + return 0; + } + std::printf("webcam_config_test: %d assertion(s) failed\n", failures); + return 1; +} diff --git a/electron/native/wgc-capture/src/webcam_loss.cpp b/electron/native/wgc-capture/src/webcam_loss.cpp new file mode 100644 index 000000000..7548f84c2 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_loss.cpp @@ -0,0 +1,28 @@ +#include "webcam_loss.h" + +#include + +#include + +bool isWebcamLossResult(HRESULT hr, bool endOfStream, int64_t consecutiveFailureMs) { + if (SUCCEEDED(hr)) { + return endOfStream; + } + if (hr == MF_E_VIDEO_RECORDING_DEVICE_INVALIDATED || hr == MF_E_HW_MFT_FAILED_START_STREAMING) { + return true; + } + return consecutiveFailureMs >= kWebcamLossFailureRunMs; +} + +void reportWebcamLost(const std::wstring& deviceName, HRESULT hr) { + std::string name; + const int size = WideCharToMultiByte( + CP_UTF8, 0, deviceName.data(), static_cast(deviceName.size()), nullptr, 0, nullptr, nullptr); + if (size > 0) { + name.resize(static_cast(size)); + WideCharToMultiByte( + CP_UTF8, 0, deviceName.data(), static_cast(deviceName.size()), name.data(), size, nullptr, nullptr); + } + std::cerr << "WARNING: Webcam lost during the take: " << name << " hr=0x" << std::hex + << static_cast(hr) << std::dec << std::endl; +} diff --git a/electron/native/wgc-capture/src/webcam_loss.h b/electron/native/wgc-capture/src/webcam_loss.h new file mode 100644 index 000000000..64af1f987 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_loss.h @@ -0,0 +1,34 @@ +#pragma once + +#include + +#include +#include + +/** How long reads may keep failing, back to back, before the camera counts as gone. */ +constexpr int64_t kWebcamLossFailureRunMs = 1000; + +/** + * Does this read result mean the camera has left the take? + * + * Media Foundation reports an unplugged camera as + * MF_E_VIDEO_RECORDING_DEVICE_INVALIDATED on every following ReadSample, and a + * camera whose driver gave up as MF_E_HW_MFT_FAILED_START_STREAMING; both are + * final. End of stream while the take is still running is the same thing said + * differently. Any other failure is forgiven once -- a camera can drop a read -- + * but not for `kWebcamLossFailureRunMs` in a row: a reader that fails that long + * delivers nothing, and treating it as alive is what froze a file on its last + * picture for the rest of the take. + * + * `consecutiveFailureMs` is the time since the first failure of the current + * run; a successful read ends the run. Ignored when `hr` succeeded. + */ +bool isWebcamLossResult(HRESULT hr, bool endOfStream, int64_t consecutiveFailureMs); + +/** + * The one line both capture backends print when a camera leaves the take. + * + * `hr` is the read result that decided it, or the DirectShow event code when + * the graph reported the loss. + */ +void reportWebcamLost(const std::wstring& deviceName, HRESULT hr); diff --git a/electron/native/wgc-capture/src/webcam_loss_test.cpp b/electron/native/wgc-capture/src/webcam_loss_test.cpp new file mode 100644 index 000000000..2a8462f8d --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_loss_test.cpp @@ -0,0 +1,41 @@ +#include "webcam_loss.h" + +#include + +#include + +namespace { +int failures = 0; +void expect(bool ok, const char* label) { + if (!ok) { + std::printf("FAIL %s\n", label); + ++failures; + } +} +} // namespace + +int main() { + // The code the on-site test logged after the C920 was unplugged. + expect(isWebcamLossResult(MF_E_VIDEO_RECORDING_DEVICE_INVALIDATED, false, 0), + "device invalidated: lost on the first read"); + expect(isWebcamLossResult(MF_E_HW_MFT_FAILED_START_STREAMING, false, 0), + "hardware failed to start streaming: lost"); + expect(isWebcamLossResult(S_OK, true, 0), "end of stream during the take: lost"); + + expect(!isWebcamLossResult(E_FAIL, false, 0), "one other failure: not lost"); + expect(!isWebcamLossResult(E_FAIL, false, kWebcamLossFailureRunMs - 1), + "other failures for just under a second: not lost"); + expect(isWebcamLossResult(E_FAIL, false, kWebcamLossFailureRunMs), + "other failures for a second: lost"); + expect(isWebcamLossResult(MF_E_NOTACCEPTING, false, 5000), "any failure code, long run: lost"); + + expect(!isWebcamLossResult(S_OK, false, 0), "S_OK: not lost"); + expect(!isWebcamLossResult(S_OK, false, 5000), "S_OK ends a run of failures: not lost"); + + if (failures == 0) { + std::printf("webcam_loss_test: all assertions passed\n"); + return 0; + } + std::printf("webcam_loss_test: %d failure(s)\n", failures); + return 1; +} diff --git a/electron/recording/nativeWindowsCaptureStop.test.ts b/electron/recording/nativeWindowsCaptureStop.test.ts index e277eceec..0def7936f 100644 --- a/electron/recording/nativeWindowsCaptureStop.test.ts +++ b/electron/recording/nativeWindowsCaptureStop.test.ts @@ -7,9 +7,13 @@ import { NATIVE_WINDOWS_SALVAGEABLE_OUTPUT_BYTES, readMicrophoneDefaulted, readMicrophoneUnavailable, + readReportedWebcamPaths, readSecondaryWindowsApplied, readStoppedPath, + readStoppedWebcamPaths, + readUnavailableWebcamIndices, readWebcamFormat, + readWebcamFormatAt, readWebcamUnavailable, terminateNativeWindowsCapture, waitForNativeWindowsCaptureStop, @@ -539,3 +543,75 @@ describe("terminateNativeWindowsCapture", () => { expect(helper.killCalls).toBe(1); }); }); + +describe("per-camera helper events", () => { + const unavailable = (i?: number) => + `{"event":"warning","code":"webcam-unavailable"${i === undefined ? "" : `,"index":${i}`},"message":"x"}`; + + it("collects unavailable cameras by index, an old event counting as camera 0", () => { + expect(readUnavailableWebcamIndices(`${unavailable(1)}\n${unavailable(3)}`)).toEqual([1, 3]); + expect(readUnavailableWebcamIndices(unavailable())).toEqual([0]); + expect(readUnavailableWebcamIndices("nothing")).toEqual([]); + }); + + it("ignores other warnings and reads a prefixed unavailable event", () => { + const other = '{"event":"warning","code":"microphone-defaulted","index":2}'; + const prefixed = `INFO: camera lost ${unavailable(2)}`; + expect(readUnavailableWebcamIndices(`${other}\n${prefixed}`)).toEqual([2]); + }); + + it("reads the format of a given camera", () => { + const out = [ + '{"event":"webcam-format","schemaVersion":2,"index":0,"width":1920,"height":1080,"fps":30,"deviceName":"A"}', + '{"event":"webcam-format","schemaVersion":2,"index":1,"width":1280,"height":720,"fps":30,"deviceName":"B"}', + ].join("\n"); + expect(readWebcamFormatAt(out, 1)?.width).toBe(1280); + expect(readWebcamFormatAt(out, 0)?.width).toBe(1920); + expect(readWebcamFormatAt(out, 2)).toBeNull(); + }); + + it("reads a format event glued to a diagnostic prefix, the last one winning", () => { + const out = [ + 'INFO: DirectShow webcam connected subtype NV12 {"event":"webcam-format","index":1,"width":640,"height":480}', + 'INFO: again {"event":"webcam-format","index":1,"width":1280,"height":720}', + ].join("\n"); + expect(readWebcamFormatAt(out, 1)?.width).toBe(1280); + }); + + it("treats a format without index as camera 0", () => { + expect(readWebcamFormatAt('{"event":"webcam-format","width":800}', 0)?.width).toBe(800); + }); + + it("reads every stopped camera path, falling back to the single legacy path", () => { + expect( + readStoppedWebcamPaths( + '{"event":"recording-stopped","webcamPath":"a.mp4","webcamPaths":["a.mp4","b.mp4"]}', + ), + ).toEqual(["a.mp4", "b.mp4"]); + expect(readStoppedWebcamPaths('{"event":"recording-stopped","webcamPath":"a.mp4"}')).toEqual([ + "a.mp4", + ]); + expect(readStoppedWebcamPaths('{"event":"recording-stopped"}')).toEqual([]); + }); + + it("tells a reported webcamPaths list apart from its absence", () => { + expect( + readReportedWebcamPaths('{"event":"recording-stopped","webcamPaths":["a.mp4","b.mp4"]}'), + ).toEqual(["a.mp4", "b.mp4"]); + expect(readReportedWebcamPaths('{"event":"recording-stopped","webcamPaths":[]}')).toEqual([]); + // An old helper sends only the legacy path; that is "unknown", not "none". + expect( + readReportedWebcamPaths('{"event":"recording-stopped","webcamPath":"a.mp4"}'), + ).toBeNull(); + expect(readReportedWebcamPaths("Recording stopped. Output path: s.mp4")).toBeNull(); + }); + + it("reads JSON-escaped Windows paths behind a prefix", () => { + const line = + 'INFO: x {"event":"recording-stopped","schemaVersion":2,"screenPath":"C:\\\\Users\\\\me\\\\s.mp4","webcamPath":"C:\\\\Users\\\\me\\\\s-webcam.mp4","webcamPaths":["C:\\\\Users\\\\me\\\\s-webcam.mp4","C:\\\\Users\\\\me\\\\s-webcam-2.mp4"]}'; + expect(readStoppedWebcamPaths(line)).toEqual([ + "C:\\Users\\me\\s-webcam.mp4", + "C:\\Users\\me\\s-webcam-2.mp4", + ]); + }); +}); diff --git a/electron/recording/nativeWindowsCaptureStop.ts b/electron/recording/nativeWindowsCaptureStop.ts index 478d30e61..05c2a0f1f 100644 --- a/electron/recording/nativeWindowsCaptureStop.ts +++ b/electron/recording/nativeWindowsCaptureStop.ts @@ -218,6 +218,93 @@ export function readWebcamFormat(output: string) { } } +type HelperEvent = Record; + +/** + * Every `{"event":""…}` object in the output, in order. + * + * Locates each object start and slices it with `findObjectEnd` instead of + * parsing whole lines, because diagnostics can be glued in front of an event + * (see `readWebcamFormat`). Slices that do not parse are skipped. + */ +function readHelperEvents(output: string, name: string): HelperEvent[] { + const needle = `{"event":"${name}"`; + const events: HelperEvent[] = []; + let from = 0; + for (;;) { + const start = output.indexOf(needle, from); + if (start === -1) { + return events; + } + const end = findObjectEnd(output, start); + if (end === -1) { + return events; + } + from = end + 1; + try { + const parsed: unknown = JSON.parse(output.slice(start, end + 1)); + if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) { + events.push(parsed as HelperEvent); + } + } catch { + // Not a complete JSON object; keep looking. + } + } +} + +/** A helper event's camera index; an event without one comes from an old helper (camera 0). */ +function eventCameraIndex(event: HelperEvent) { + return typeof event.index === "number" ? event.index : 0; +} + +/** Indices of every camera the helper reported as unavailable (no index = camera 0). */ +export function readUnavailableWebcamIndices(output: string): number[] { + return readHelperEvents(output, "warning") + .filter((event) => event.code === "webcam-unavailable") + .map(eventCameraIndex); +} + +/** The last `webcam-format` event of the given camera (no index = camera 0), or null. */ +export function readWebcamFormatAt( + output: string, + index: number, +): ReturnType { + const match = readHelperEvents(output, "webcam-format") + .filter((event) => eventCameraIndex(event) === index) + .at(-1); + return (match as ReturnType) ?? null; +} + +/** + * The `webcamPaths` list of the last `recording-stopped` event, or null when + * there is no such event or it carried no `webcamPaths` key. + * + * Null is the signal that matters: only a helper that sends the key says which + * cameras were still recording at stop, so only then may a camera missing from + * it be called one that stopped early. An old helper, or a stop without the + * event, must read as "unknown", never as "every camera stopped". + */ +export function readReportedWebcamPaths(output: string): string[] | null { + const event = readHelperEvents(output, "recording-stopped").at(-1); + if (!event || !Array.isArray(event.webcamPaths)) { + return null; + } + return event.webcamPaths.filter((entry): entry is string => typeof entry === "string"); +} + +/** + * Camera files the helper reported at stop: `webcamPaths`, else the single + * legacy `webcamPath`, else none. + */ +export function readStoppedWebcamPaths(output: string): string[] { + const reported = readReportedWebcamPaths(output); + if (reported) { + return reported; + } + const event = readHelperEvents(output, "recording-stopped").at(-1); + return typeof event?.webcamPath === "string" && event.webcamPath ? [event.webcamPath] : []; +} + /** * The most useful line of a failed helper run, for a toast. * diff --git a/electron/recording/nativeWindowsWebcams.test.ts b/electron/recording/nativeWindowsWebcams.test.ts new file mode 100644 index 000000000..f7187851c --- /dev/null +++ b/electron/recording/nativeWindowsWebcams.test.ts @@ -0,0 +1,265 @@ +import { describe, expect, it } from "vitest"; +import { + additionalWebcamLabels, + buildHelperWebcamConfig, + collectStoppedWebcams, + dedupeAdditionalWebcams, + isWebcamSidecarFile, + labelsOfUnavailableAdditionalWebcams, + labelsOfWebcamsStoppedEarly, + stripWebcamSuffix, + webcamOutputPath, +} from "./nativeWindowsWebcams"; + +describe("nativeWindowsWebcams", () => { + it("names camera files", () => { + expect(webcamOutputPath("C:\\r", "rec-", 7, 1)).toMatch(/rec-7-webcam\.mp4$/); + expect(webcamOutputPath("C:\\r", "rec-", 7, 3)).toMatch(/rec-7-webcam-3\.mp4$/); + }); + + it("recognizes every camera file of a recording and nothing else", () => { + for (const f of ["rec-7-webcam.mp4", "rec-7-webcam-2.mp4", "rec-7-webcam-4.webm"]) { + expect(isWebcamSidecarFile(f)).toBe(true); + } + for (const f of ["rec-7.mp4", "rec-7-webcamera.mp4", "rec-7-webcam-x.mp4"]) { + expect(isWebcamSidecarFile(f)).toBe(false); + } + expect(stripWebcamSuffix("rec-7-webcam-2")).toBe("rec-7"); + expect(stripWebcamSuffix("rec-7-webcam")).toBe("rec-7"); + expect(stripWebcamSuffix("rec-7")).toBe("rec-7"); + }); + + it("dedupes a device that equals camera 1, duplicates, and caps at three", () => { + const extras = [ + { deviceId: "a", deviceName: "Front" }, + { deviceId: "b", deviceName: "Desk" }, + { deviceId: "b", deviceName: "Desk" }, + { deviceName: "Side" }, + { deviceName: "Top" }, + { deviceName: "Fifth" }, + ]; + expect( + dedupeAdditionalWebcams({ deviceId: "a", deviceName: "Front" }, extras).map( + (e) => e.deviceName, + ), + ).toEqual(["Desk", "Side", "Top"]); + }); + + it("drops extras that name no device at all", () => { + expect(dedupeAdditionalWebcams(null, [{ deviceName: " " }, { deviceName: "Desk" }])).toEqual([ + { deviceName: "Desk" }, + ]); + }); + + it("skips malformed entries from IPC", () => { + const junk = [ + null, + { deviceName: 3 }, + { deviceId: 4, deviceName: "X" }, + { deviceName: "Desk" }, + ]; + expect(dedupeAdditionalWebcams(null, junk as unknown as Array<{ deviceName: string }>)).toEqual( + [{ deviceName: "Desk" }], + ); + }); + + it("never matches an extra with an id to an id-less camera 1 by name", () => { + // Two cameras of the same model: same name, and camera 1 came without an id. + expect( + dedupeAdditionalWebcams({ deviceName: "USB Camera" }, [ + { deviceId: "b", deviceName: "USB Camera" }, + ]), + ).toEqual([{ deviceId: "b", deviceName: "USB Camera" }]); + // Without an id on either side the name is all there is, and it still matches. + expect( + dedupeAdditionalWebcams({ deviceName: "USB Camera" }, [{ deviceName: "USB Camera" }]), + ).toEqual([]); + }); + + describe("labelsOfWebcamsStoppedEarly", () => { + const front = String.raw`C:\Rec\r-webcam.mp4`; + const desk = String.raw`C:\Rec\r-webcam-2.mp4`; + const side = String.raw`C:\Rec\r-webcam-3.mp4`; + const requested = [ + { path: front, label: "Front" }, + { path: desk, label: "Desk" }, + { path: side, label: "Side" }, + ]; + const sizes = new Map([ + [front, 100], + [desk, 100], + [side, 0], + ]); + + it("names a kept camera the helper no longer listed, camera 1 included", () => { + expect( + labelsOfWebcamsStoppedEarly({ + requested, + sizes, + helperWebcamPaths: [desk], + }), + ).toEqual(["Front"]); + }); + + it("leaves out a camera whose file was not kept: that one was not recorded", () => { + expect(labelsOfWebcamsStoppedEarly({ requested, sizes, helperWebcamPaths: [] })).toEqual([ + "Front", + "Desk", + ]); + }); + + it("flags nothing without a webcamPaths key", () => { + expect(labelsOfWebcamsStoppedEarly({ requested, sizes, helperWebcamPaths: null })).toEqual( + [], + ); + }); + + it("matches paths regardless of case and separator", () => { + expect( + labelsOfWebcamsStoppedEarly({ + requested, + sizes, + helperWebcamPaths: ["c:/rec/R-WEBCAM.mp4", String.raw`c:\REC//r-webcam-2.MP4`], + }), + ).toEqual([]); + }); + }); + + it("drops an empty additional camera file and names it", () => { + const r = collectStoppedWebcams({ + camera1Enabled: true, + requested: [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + { path: "w-3.mp4", label: "Side" }, + ], + sizes: new Map([ + ["w.mp4", 100], + ["w-2.mp4", 0], + ["w-3.mp4", 50], + ]), + }); + expect(r).toEqual({ + camera1: "w.mp4", + additional: [{ path: "w-3.mp4", label: "Side" }], + dropped: ["Desk"], + }); + }); + + it("keeps camera 1 and reports a camera whose file never appeared as not recorded", () => { + const r = collectStoppedWebcams({ + camera1Enabled: true, + requested: [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + ], + sizes: new Map([["w.mp4", 100]]), + }); + expect(r).toEqual({ camera1: "w.mp4", additional: [], dropped: ["Desk"] }); + }); + + it("keeps extras when camera 1 is lost, leaving camera 1 out of dropped", () => { + const r = collectStoppedWebcams({ + camera1Enabled: true, + requested: [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + ], + sizes: new Map([ + ["w.mp4", 0], + ["w-2.mp4", 10], + ]), + }); + expect(r).toEqual({ additional: [{ path: "w-2.mp4", label: "Desk" }], dropped: [] }); + }); + + it("maps unavailable helper indices to the labels of the extras", () => { + const requested = [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + { path: "w-3.mp4", label: "Side" }, + ]; + expect(labelsOfUnavailableAdditionalWebcams(requested, [0, 2, 2, 9])).toEqual(["Side"]); + }); + + it("labels extras by device name, else Camera ", () => { + expect(additionalWebcamLabels("Front", [{ deviceName: "Desk" }, { deviceName: " " }])).toEqual( + ["Desk", "Camera 3"], + ); + }); + + it("tells cameras with the same name apart by occurrence, counting camera 1", () => { + expect( + additionalWebcamLabels("USB Camera", [ + { deviceName: "USB Camera" }, + { deviceName: "Desk" }, + { deviceName: " USB Camera " }, + ]), + ).toEqual(["USB Camera (2)", "Desk", "USB Camera (3)"]); + expect( + additionalWebcamLabels(undefined, [ + { deviceName: "USB Camera" }, + { deviceName: "USB Camera" }, + ]), + ).toEqual(["USB Camera", "USB Camera (2)"]); + }); + + it("builds a start config with the legacy fields and a list of every camera", () => { + const config = buildHelperWebcamConfig({ + camera1: { + enabled: true, + deviceId: "id-1", + deviceName: "Front", + width: 1280, + height: 720, + fps: 30, + }, + camera1Clsid: "{c1}", + camera1Path: "C:\\r\\rec-7-webcam.mp4", + extras: [ + { deviceId: "id-2", deviceName: "Desk", clsid: null, path: "C:\\r\\rec-7-webcam-2.mp4" }, + ], + }); + const parsed = JSON.parse(JSON.stringify(config)); + expect(parsed).toMatchObject({ + webcamEnabled: true, + webcamDeviceId: "id-1", + webcamDeviceName: "Front", + webcamDirectShowClsid: "{c1}", + webcamWidth: 1280, + webcamHeight: 720, + webcamFps: 30, + }); + expect(parsed.webcams).toEqual([ + { + camDeviceId: "id-1", + camDeviceName: "Front", + camClsid: "{c1}", + camWidth: 1280, + camHeight: 720, + camFps: 30, + camPath: "C:\\r\\rec-7-webcam.mp4", + }, + { + camDeviceId: "id-2", + camDeviceName: "Desk", + camClsid: null, + camWidth: 1280, + camHeight: 720, + camFps: 30, + camPath: "C:\\r\\rec-7-webcam-2.mp4", + }, + ]); + }); + + it("sends no camera list while camera 1 is off", () => { + const config = buildHelperWebcamConfig({ + camera1: { enabled: false, width: 1280, height: 720, fps: 30 }, + camera1Clsid: null, + camera1Path: "C:\\r\\rec-7-webcam.mp4", + extras: [{ deviceName: "Desk", clsid: null, path: "C:\\r\\rec-7-webcam-2.mp4" }], + }); + expect(config.webcamEnabled).toBe(false); + expect(config.webcams).toEqual([]); + }); +}); diff --git a/electron/recording/nativeWindowsWebcams.ts b/electron/recording/nativeWindowsWebcams.ts new file mode 100644 index 000000000..9c8b462b5 --- /dev/null +++ b/electron/recording/nativeWindowsWebcams.ts @@ -0,0 +1,269 @@ +/** + * Pure helpers for recording several cameras with the native Windows helper. + * + * Camera 1 stays exactly what it always was: the helper's legacy `webcam*` + * fields, the file `-webcam.mp4` and the session's + * `webcamVideoPath`. Cameras 2-4 ride along in the helper's `webcams` list and + * in the session's `additionalWebcams`. Kept out of `electron/ipc/handlers.ts` + * because that module cannot be loaded from a test. + */ +import path from "node:path"; +import { type AdditionalWebcam, MAX_ADDITIONAL_WEBCAMS } from "../../src/lib/recordingSession"; + +/** `-webcam` (camera 1) or `-webcam-<2..9>` at the end of a file's base name. */ +const WEBCAM_SUFFIX = /-webcam(?:-[2-9])?$/; + +/** Camera 1 → `-webcam.mp4`, camera n ≥ 2 → `-webcam-.mp4`. */ +export function webcamOutputPath( + dir: string, + prefix: string, + recordingId: number, + cameraNumber: number, +): string { + const suffix = cameraNumber <= 1 ? "-webcam" : `-webcam-${cameraNumber}`; + return path.join(dir, `${prefix}${recordingId}${suffix}.mp4`); +} + +/** Whether a file name is one of a recording's camera files (any extension). */ +export function isWebcamSidecarFile(fileName: string): boolean { + return WEBCAM_SUFFIX.test(path.parse(fileName).name); +} + +/** The recording's base name for a camera file's base name; other names pass through. */ +export function stripWebcamSuffix(baseName: string): string { + return baseName.replace(WEBCAM_SUFFIX, ""); +} + +/** One camera in the helper config's `webcams` list (keys are the helper's). */ +export interface HelperWebcamEntry { + camDeviceId: string | null; + camDeviceName: string; + camClsid: string | null; + camWidth: number; + camHeight: number; + camFps: number; + camPath: string; +} + +type DeviceRef = { deviceId?: string; deviceName?: string }; + +/** + * Same device: by id when both sides carry one, otherwise by name — but only + * when neither carries an id. Two webcams of the same model share a name, so an + * id on one side and none on the other says nothing about whether they are the + * same device, and matching by name there would silently drop the second one. + */ +function isSameDevice(a: DeviceRef, b: DeviceRef) { + if (a.deviceId && b.deviceId) { + return a.deviceId === b.deviceId; + } + if (a.deviceId || b.deviceId) { + return false; + } + const name = a.deviceName?.trim(); + return Boolean(name) && name === b.deviceName?.trim(); +} + +/** + * The additional cameras worth asking the helper for: no malformed entry, no + * entry that names no device, none equal to camera 1, no duplicates, and at most + * {@link MAX_ADDITIONAL_WEBCAMS}. + */ +export function dedupeAdditionalWebcams( + camera1: DeviceRef | null, + extras: T[], +): T[] { + const kept: T[] = []; + for (const extra of extras) { + if (kept.length >= MAX_ADDITIONAL_WEBCAMS) { + break; + } + // The list crosses IPC, so a malformed entry is skipped rather than trusted. + if ( + !extra || + typeof extra.deviceName !== "string" || + (extra.deviceId !== undefined && typeof extra.deviceId !== "string") + ) { + continue; + } + if (!extra.deviceId && !extra.deviceName.trim()) { + continue; + } + if (camera1 && isSameDevice(camera1, extra)) { + continue; + } + if (kept.some((other) => isSameDevice(other, extra))) { + continue; + } + kept.push(extra); + } + return kept; +} + +/** + * Labels of the additional cameras, in order: the device name, else + * `Camera ` (n counts camera 1, so the first extra is camera 2). + * + * Two webcams of the same model report the same name, and a label is all the + * user is told about a camera that could not be opened or was dropped. So a + * label already used — by camera 1 or an earlier extra — gets " (2)", " (3)" + * by occurrence. Camera 1's own label is never changed. + */ +export function additionalWebcamLabels( + camera1Name: string | undefined, + extras: Array<{ deviceName: string }>, +): string[] { + const seen = new Map(); + const camera1Label = camera1Name?.trim(); + if (camera1Label) { + seen.set(camera1Label, 1); + } + return extras.map((extra, i) => { + const label = extra.deviceName.trim() || `Camera ${i + 2}`; + const occurrence = (seen.get(label) ?? 0) + 1; + seen.set(label, occurrence); + return occurrence > 1 ? `${label} (${occurrence})` : label; + }); +} + +/** + * The camera part of the helper config: the unchanged legacy `webcam*` fields + * for camera 1 plus the `webcams` list (camera 1 first, then the extras). + * + * Extras are recorded only while camera 1 is on (the HUD's camera toggle + * governs every camera), so with camera 1 off the list is empty. Every extra + * uses camera 1's requested size and rate: the quality setting is one for all. + */ +export function buildHelperWebcamConfig(input: { + camera1: { + enabled: boolean; + deviceId?: string; + deviceName?: string; + width: number; + height: number; + fps: number; + }; + camera1Clsid: string | null; + camera1Path: string; + extras: Array<{ deviceId?: string; deviceName: string; clsid: string | null; path: string }>; +}) { + const { camera1 } = input; + const entry = ( + device: { deviceId?: string; deviceName?: string }, + clsid: string | null, + camPath: string, + ): HelperWebcamEntry => ({ + camDeviceId: device.deviceId ?? null, + camDeviceName: device.deviceName ?? "", + camClsid: clsid, + camWidth: camera1.width, + camHeight: camera1.height, + camFps: camera1.fps, + camPath, + }); + const webcams: HelperWebcamEntry[] = camera1.enabled + ? [ + entry(camera1, input.camera1Clsid, input.camera1Path), + ...input.extras.map((extra) => entry(extra, extra.clsid, extra.path)), + ] + : []; + return { + webcamEnabled: camera1.enabled, + webcamDeviceId: camera1.deviceId ?? null, + webcamDeviceName: camera1.deviceName ?? null, + webcamDirectShowClsid: input.camera1Clsid, + webcamWidth: camera1.width, + webcamHeight: camera1.height, + webcamFps: camera1.fps, + webcams, + }; +} + +/** + * Labels of the additional cameras the helper reported unavailable. The + * helper's index is the position in the `webcams` list, which is the position + * in `requested` (camera 1 first). Index 0 is camera 1 and is left out: the + * caller reports camera 1 through its own `webcamUnavailable` flag. The labels + * come from our own list because the helper's `deviceName` can be empty. + */ +export function labelsOfUnavailableAdditionalWebcams( + requested: Array<{ label: string }>, + indices: number[], +): string[] { + const labels: string[] = []; + for (const index of new Set(indices)) { + const camera = index > 0 ? requested[index] : undefined; + if (camera) { + labels.push(camera.label); + } + } + return labels; +} + +/** + * Which requested cameras made it into the take. + * + * A camera is kept when its file exists with size > 0. The helper's own list + * at stop is deliberately not consulted: it names the cameras still recording + * at that moment, so a camera that died mid-take with a playable partial file + * would be missing from it, and an old helper names only camera 1. A file that + * never appeared is simply absent from `sizes`. + * + * `camera1Enabled` says whether `requested[0]` is camera 1. Camera 1 comes back + * as `camera1` (undefined when lost) and is never in `dropped`, which names + * only the additional cameras not kept — the caller reports camera 1 through + * its own `webcamDropped` flag. + */ +export function collectStoppedWebcams(input: { + camera1Enabled: boolean; + requested: Array<{ path: string; label: string }>; + sizes: Map; +}): { camera1?: string; additional: AdditionalWebcam[]; dropped: string[] } { + const isKept = (filePath: string) => (input.sizes.get(filePath) ?? 0) > 0; + const [first, ...rest] = input.requested; + const camera1 = input.camera1Enabled && first && isKept(first.path) ? first.path : undefined; + const extras = input.camera1Enabled ? rest : input.requested; + const additional: AdditionalWebcam[] = []; + const dropped: string[] = []; + for (const camera of extras) { + if (isKept(camera.path)) { + additional.push({ path: camera.path, label: camera.label }); + } else { + dropped.push(camera.label); + } + } + return { ...(camera1 ? { camera1 } : {}), additional, dropped }; +} + +/** A path compared the way Windows does: case-insensitive, either separator. */ +function comparablePath(filePath: string) { + return filePath.replace(/[\\/]+/g, "\\").toLowerCase(); +} + +/** + * Labels of the cameras that stopped early: their file was kept (size > 0, the + * same test as {@link collectStoppedWebcams}) but the helper did not list it in + * `recording-stopped.webcamPaths`, the cameras still recording at stop. That is + * a camera the helper disabled mid-take; its partial file stays in the take, + * and the user is told which camera it was. + * + * `helperWebcamPaths` is null when the event carried no `webcamPaths` key (an + * old helper, or no event at all) — then nothing is known and nothing is + * flagged. Camera 1 counts like any other, so its entry needs a real label. + */ +export function labelsOfWebcamsStoppedEarly(input: { + requested: Array<{ path: string; label: string }>; + sizes: Map; + helperWebcamPaths: string[] | null; +}): string[] { + if (!input.helperWebcamPaths) { + return []; + } + const stillRecording = new Set(input.helperWebcamPaths.map(comparablePath)); + return input.requested + .filter( + (camera) => + (input.sizes.get(camera.path) ?? 0) > 0 && !stillRecording.has(comparablePath(camera.path)), + ) + .map((camera) => camera.label); +} diff --git a/package-lock.json b/package-lock.json index 3f8dceacc..638a465bf 100644 --- a/package-lock.json +++ b/package-lock.json @@ -37,6 +37,7 @@ "clsx": "^2.1.1", "electron-updater": "^6.8.9", "i18next": "^23.16.0", + "js-aruco2": "2.0.0", "langchain": "^1.2.39", "lucide-react": "^0.545.0", "mediabunny": "^1.40.1", @@ -8316,6 +8317,15 @@ "url": "https://github.com/sponsors/panva" } }, + "node_modules/js-aruco2": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/js-aruco2/-/js-aruco2-2.0.0.tgz", + "integrity": "sha512-2R9P7INqIkg1fsGBazSDLChmfwJpbaYN26gO6wFr1TLIOrvwQf0YzuhwVZAPF8frQ4cjdppAN0dRmExEd/6wqw==", + "license": "MIT", + "engines": { + "node": ">=12.0.0" + } + }, "node_modules/js-tiktoken": { "version": "1.0.21", "resolved": "https://registry.npmjs.org/js-tiktoken/-/js-tiktoken-1.0.21.tgz", diff --git a/package.json b/package.json index f707b8dbc..2cc3c0e42 100644 --- a/package.json +++ b/package.json @@ -125,6 +125,7 @@ "clsx": "^2.1.1", "electron-updater": "^6.8.9", "i18next": "^23.16.0", + "js-aruco2": "2.0.0", "langchain": "^1.2.39", "lucide-react": "^0.545.0", "mediabunny": "^1.40.1", diff --git a/scripts/build-windows-wgc-helper.mjs b/scripts/build-windows-wgc-helper.mjs index ab4b26c00..927a7a392 100644 --- a/scripts/build-windows-wgc-helper.mjs +++ b/scripts/build-windows-wgc-helper.mjs @@ -116,6 +116,30 @@ if (!fs.existsSync(webcamFormatTestPath)) { await run(webcamFormatTestPath, [], { cwd: BUILD_DIR }); console.log(`Passed ${webcamFormatTestPath}`); +const webcamConfigTestPath = path.join(BUILD_DIR, "webcam_config_test.exe"); +if (!fs.existsSync(webcamConfigTestPath)) { + throw new Error(`WGC helper build completed but ${webcamConfigTestPath} was not found.`); +} +// Guards how the helper reads the list of cameras (and the legacy single-camera fields). +await run(webcamConfigTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${webcamConfigTestPath}`); + +const deviceSelectionTestPath = path.join(BUILD_DIR, "device_selection_test.exe"); +if (!fs.existsSync(deviceSelectionTestPath)) { + throw new Error(`WGC helper build completed but ${deviceSelectionTestPath} was not found.`); +} +// Guards that two cameras of the same model in one take open two devices, not one twice. +await run(deviceSelectionTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${deviceSelectionTestPath}`); + +const webcamLossTestPath = path.join(BUILD_DIR, "webcam_loss_test.exe"); +if (!fs.existsSync(webcamLossTestPath)) { + throw new Error(`WGC helper build completed but ${webcamLossTestPath} was not found.`); +} +// Guards that an unplugged camera ends its file instead of freezing on its last picture. +await run(webcamLossTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${webcamLossTestPath}`); + const frameVisibilityTestPath = path.join(BUILD_DIR, "frame_visibility_test.exe"); if (!fs.existsSync(frameVisibilityTestPath)) { throw new Error(`WGC helper build completed but ${frameVisibilityTestPath} was not found.`); diff --git a/scripts/editor-multicam-smoke.mjs b/scripts/editor-multicam-smoke.mjs new file mode 100644 index 000000000..33222c6c3 --- /dev/null +++ b/scripts/editor-multicam-smoke.mjs @@ -0,0 +1,592 @@ +// Multi-camera editor smoke run (dev tool, not part of any test suite). +// +// Drives the built Electron app with Playwright through the multi-camera editor: adds a +// "camera + inset" section from the layout menu, picks its cameras in the inspector, converts +// a Full Camera section (key C) to "screen + camera" and back, undoes and redoes that, drags +// a PiP window in the preview, applies a perspective correction in the calibration dialog, +// saves, relaunches the app and checks the project came back unchanged. After every step it +// reads the saved project file and asserts the document state, and it captures the live +// preview inside each section. +// +// Preview captures are taken at the middle of a section's VISIBLE span: its span on the ruler +// minus the trims of its clip. A section eases in and out over its first and last moments, so +// a capture near either end of what is visible shows the transition, not the layout (a trim +// that cuts into a section moves its visible start, and the glide with it). The test section +// is placed clear of the project's trims for the same reason. With camera 1 in the large +// place, the run also measures the composed frame's outer pixel columns and rows against the +// wallpaper colour, so a frame that does not fill edge to edge fails. +// +// Usage (Windows, from the repo root): +// npm run build-vite +// npm run build:native:compositor # then copy compositor_view.node into the bin dir below +// node scripts/editor-multicam-smoke.mjs --source --out [--bin ] +// +// --source id of an existing project with at least two cameras on its first clip +// (e.g. proj_7bc8…). It is NOT modified: the run copies it to a new project id +// with a fresh `updatedAt`, so the editor opens the copy, and deletes the copy at +// the end. Back up %APPDATA%\openscreen\projects first anyway: the editor may +// touch other state while it runs. +// --out directory for screenshots, the Electron log and results.json. +// --bin native bin dir holding compositor_view.node and the ffmpeg DLLs +// (default electron/native/bin/win32-). +// +// The app is launched with the repo root as its entry (not dist-electron/main.js, which would +// move userData to Roaming\Electron) and with OPENSCREEN_DISABLE_CONTENT_PROTECTION=1. +// Exit code 1 when any check failed. +import { randomUUID } from "node:crypto"; +import fs from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { _electron as electron } from "playwright"; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); + +function arg(name, fallback) { + const i = process.argv.indexOf(`--${name}`); + return i >= 0 && process.argv[i + 1] ? process.argv[i + 1] : fallback; +} + +const SOURCE_ID = arg("source"); +const OUT = arg("out"); +const BIN = arg("bin", path.join(ROOT, "electron", "native", "bin", `win32-${process.arch}`)); +if (!SOURCE_ID || !OUT) { + console.error("usage: node scripts/editor-multicam-smoke.mjs --source --out "); + process.exit(2); +} +fs.mkdirSync(OUT, { recursive: true }); + +const PROJECTS = path.join(process.env.APPDATA ?? "", "openscreen", "projects"); +const COPY_ID = `proj_${randomUUID()}`; +const COPY_FILE = path.join(PROJECTS, `${COPY_ID}.openscreen`); + +const sleep = (ms) => new Promise((r) => setTimeout(r, ms)); +const results = []; +function note(name, ok, detail = "") { + results.push({ name, ok, detail }); + console.log(`${ok ? "PASS" : "FAIL"} ${name} ${detail}`); +} + +function readDoc() { + return JSON.parse(fs.readFileSync(COPY_FILE, "utf8")); +} +function cameraState(doc = readDoc()) { + const legacy = doc.legacyEditor ?? {}; + return { + layouts: legacy.cameraLayoutRegions ?? [], + fulls: legacy.cameraFullscreenRegions ?? [], + settings: legacy.cameraSettings ?? null, + }; +} +// Saves are asynchronous: poll the file until the predicate holds or the time is up. +async function waitState(predicate, timeoutMs = 8000) { + const end = Date.now() + timeoutMs; + let state = cameraState(); + while (Date.now() < end) { + state = cameraState(); + try { + if (predicate(state)) return { ok: true, state }; + } catch { + // a half-written state: keep polling + } + await sleep(200); + } + return { ok: false, state }; +} +/** + * The part of a camera section the viewer sees, in ruler seconds: its span minus the trims of + * its clip (trims are in source seconds; a clip maps source to ruler time by a fixed offset). + * The longest remaining piece, or null when a trim hides the whole section. + */ +function visibleSpan(row, doc) { + const clips = doc.timeline.clips; + const clip = + clips.find((c) => c.id === row.clipId) ?? + clips.find( + (c) => c.timelineStartSec * 1000 <= row.startMs && row.startMs < c.timelineEndSec * 1000, + ); + let pieces = [[row.startMs / 1000, row.endMs / 1000]]; + for (const trim of doc.timeline.trimRanges ?? []) { + if (!clip || trim.assetId !== clip.assetId) continue; + if (trim.clipId && trim.clipId !== clip.id) continue; + const offset = clip.timelineStartSec - clip.sourceStartSec; + const [cutStart, cutEnd] = [trim.startSec + offset, trim.endSec + offset]; + pieces = pieces + .flatMap(([start, end]) => [ + [start, Math.min(end, cutStart)], + [Math.max(start, cutEnd), end], + ]) + .filter(([start, end]) => end - start > 1e-3); + } + pieces.sort((a, b) => b[1] - b[0] - (a[1] - a[0])); + return pieces[0] ?? null; +} +/** Where a steady-state capture of a section goes: the middle of its visible span. */ +function steadyMid(row, doc = readDoc()) { + const span = visibleSpan(row, doc); + return span ? (span[0] + span[1]) / 2 : (row.startMs + row.endMs) / 2000; +} +/** The first time from `fromSec` on with `lengthSec` free of the first clip's trims. */ +function clearOfTrims(fromSec, lengthSec, doc = readDoc()) { + const clip = doc.timeline.clips[0]; + const offset = clip.timelineStartSec - clip.sourceStartSec; + const trims = [...(doc.timeline.trimRanges ?? [])].sort((a, b) => a.startSec - b.startSec); + let at = fromSec; + for (const trim of trims) { + if (trim.assetId !== clip.assetId || (trim.clipId && trim.clipId !== clip.id)) continue; + const [cutStart, cutEnd] = [trim.startSec + offset, trim.endSec + offset]; + if (at < cutEnd && at + lengthSec > cutStart) at = Math.ceil(cutEnd) + 1; + } + return at; +} + +const brief = (rows) => + rows + .map( + (r) => + `${r.template ?? "full"}[${(r.slots ?? []).map((s) => s.camera).join(",")}]@${( + r.startMs / 1000 + ).toFixed(2)}-${(r.endMs / 1000).toFixed(2)}`, + ) + .join(" "); + +// 1. The copy the editor opens: newest `updatedAt`, fresh id, otherwise the source verbatim. +const source = JSON.parse(fs.readFileSync(path.join(PROJECTS, `${SOURCE_ID}.openscreen`), "utf8")); +source.project.id = COPY_ID; +source.project.title = `Multi-camera smoke ${new Date().toISOString()}`; +source.project.updatedAt = new Date().toISOString(); +fs.writeFileSync(COPY_FILE, JSON.stringify(source)); +console.log(`copy: ${COPY_FILE}`); + +const log = fs.createWriteStream(path.join(OUT, "electron.log")); +const consoleErrors = []; + +async function launch() { + const app = await electron.launch({ + args: [ROOT, "--lang=en-US"], + cwd: ROOT, + env: { + ...process.env, + OPENSCREEN_COMPOSITOR_VIEW_NODE: path.join(BIN, "compositor_view.node"), + OPENSCREEN_DISABLE_CONTENT_PROTECTION: "1", + PATH: `${BIN};${process.env.PATH}`, + }, + timeout: 60_000, + }); + const proc = app.process(); + proc.stdout?.on("data", (d) => log.write(d)); + proc.stderr?.on("data", (d) => log.write(d)); + const hud = await app.firstWindow({ timeout: 60_000 }); + await hud.waitForLoadState("domcontentloaded"); + // Switching closes the HUD window under the evaluate, so its rejection is expected. + await hud.evaluate(() => window.electronAPI.switchToEditor()).catch(() => undefined); + let editor = null; + for (let i = 0; i < 60 && !editor; i++) { + editor = app.windows().find((w) => w.url().includes("windowType=editor")) ?? null; + if (!editor) await sleep(500); + } + if (!editor) throw new Error("editor window did not open"); + editor.on("console", (m) => m.type() === "error" && consoleErrors.push(m.text())); + editor.on("pageerror", (e) => consoleErrors.push(`pageerror: ${e.message}`)); + await editor.waitForLoadState("domcontentloaded"); + const preview = editor.getByTestId("preview"); + await preview.waitFor({ state: "visible", timeout: 30_000 }); + await sleep(8000); // project load + first composed frame + return { app, proc, editor, preview }; +} + +async function close({ app, proc }) { + await Promise.race([app.close(), sleep(8000)]); + try { + proc.kill(); + } catch { + // already gone + } +} + +function driver(editor, preview) { + const time = async () => Number(await preview.getAttribute("data-current-time-sec")); + const canvas = editor.locator('[class*="_tlRulerRow_"] [class*="_tlCanvas_"]').first(); + let totalSec = null; + const clickRulerAt = async (frac) => { + const box = await canvas.boundingBox(); + if (!box) throw new Error("no ruler"); + await editor.mouse.click(box.x + box.width * frac, box.y + box.height / 2); + await sleep(400); + }; + // The ruler maps x linearly onto [0, total]; one probe click measures the total. + const seekTo = async (sec) => { + if (totalSec === null) { + await clickRulerAt(0.5); + totalSec = (await time()) / 0.5; + } + await clickRulerAt(sec / totalSec); + await sleep(1200); // the compositor's frame for the new time + return time(); + }; + const xOf = async (sec) => { + const box = await canvas.boundingBox(); + return box.x + (box.width * sec) / totalSec; + }; + // Clicks the camera-lane pill under `sec`: Full Camera and layout sections share that lane. + const selectPillAt = async (sec) => { + const x = await xOf(sec); + const pills = editor.locator('[class*="_lanePill_"][class*="_laneCameraFullscreen_"]'); + const n = await pills.count(); + for (let i = 0; i < n; i++) { + const b = await pills.nth(i).boundingBox(); + if (b && x >= b.x && x <= b.x + b.width) { + await editor.mouse.click(x, b.y + b.height / 2); + await sleep(500); + return true; + } + } + return false; + }; + const shot = async (name) => { + const file = path.join(OUT, `${name}.png`); + await preview.screenshot({ path: file }); + return file; + }; + const cameraPillCount = () => + editor.locator('[class*="_lanePill_"][class*="_laneCameraFullscreen_"]').count(); + // The composed frame alone (the preview element also holds the pane around it), decoded in + // the page: the four outermost columns and rows on the left and top, each as RGB pixels. + const frameEdges = async (name) => { + const file = path.join(OUT, `${name}-frame.png`); + const png = await preview.locator("canvas").first().screenshot({ path: file }); + const measured = await editor.evaluate(async (b64) => { + const img = new Image(); + img.src = `data:image/png;base64,${b64}`; + await img.decode(); + const c = document.createElement("canvas"); + c.width = img.width; + c.height = img.height; + const g = c.getContext("2d"); + g.drawImage(img, 0, 0); + const px = g.getImageData(0, 0, c.width, c.height).data; + const at = (x, y) => { + const i = (y * c.width + x) * 4; + return [px[i], px[i + 1], px[i + 2]]; + }; + const column = (x) => Array.from({ length: c.height }, (_, y) => at(x, y)); + const row = (y) => Array.from({ length: c.width }, (_, x) => at(x, y)); + return { + size: [c.width, c.height], + left: [0, 1, 2, 3].map(column), + top: [0, 1, 2, 3].map(row), + }; + }, png.toString("base64")); + return { file, ...measured }; + }; + return { time, seekTo, selectPillAt, shot, cameraPillCount, frameEdges }; +} + +const colourDistance = (a, b) => Math.hypot(a[0] - b[0], a[1] - b[1], a[2] - b[2]); +const meanColour = (pixels) => + [0, 1, 2].map((k) => Math.round(pixels.reduce((sum, p) => sum + p[k], 0) / pixels.length)); + +/** + * How much wallpaper a frame shows along its left and top edges, measured against a plain + * frame (no camera section) where the wallpaper lies at exactly those pixels. Per line (edge + * columns 0-3, rows 0-3): the share of its pixels within a small colour distance of the + * plain frame's pixel at the same place; the 5 % nearest each corner are left out (rounded + * corners). A line that is not part of the picture in the plain frame either (a resampling + * seam: it jumps away from its inner neighbour there, while a wallpaper gradient does not) is + * reported and skipped. + */ +function wallpaperAtEdges(frame, plain) { + const report = {}; + for (const side of ["left", "top"]) { + report[side] = frame[side].map((line, i) => { + const ref = plain[side][i]; + const inner = plain[side][i + 1] ?? plain[side][i - 1]; + const cut = Math.round(line.length * 0.05); + const at = Array.from({ length: line.length - 2 * cut }, (_, k) => k + cut); + const jump = at.reduce((sum, k) => sum + colourDistance(ref[k], inner[k]), 0) / at.length; + const wallpaper = at.filter((k) => colourDistance(line[k], ref[k]) < 24).length; + return { + line: i, + seam: jump > 40, + wallpaperShare: Number((wallpaper / at.length).toFixed(3)), + mean: meanColour(at.map((k) => line[k])), + plainMean: meanColour(at.map((k) => ref[k])), + }; + }); + } + return report; +} + +const shots = []; +let exitCode = 0; +let session = null; +try { + session = await launch(); + const { editor, preview } = session; + const d = driver(editor, preview); + const before = cameraState(); + note( + "the copy opens with no layout sections", + before.layouts.length === 0 && before.fulls.length === 1, + `fulls ${brief(before.fulls)}`, + ); + note("one camera-lane pill on open", (await d.cameraPillCount()) === 1); + + // The wallpaper along the frame's edges, read where no section is (the plain screen layout, + // whose padding shows the wallpaper). The project's own Full Camera sits at ~17-19 s. + const plainAt = clearOfTrims(12, 1); + await d.seekTo(plainAt); + const plain = await d.frameEdges("0-plain"); + note("a plain frame is captured as the wallpaper reference", true, `t=${plainAt}`); + + // A. Camera + inset from the layout menu, clear of the project's trims (a trim inside the + // section would move its visible start, see the header). + const startA = clearOfTrims(3, 2); + await d.seekTo(startA); + await editor.getByRole("button", { name: /^Add layout/ }).click(); + await editor.getByRole("button", { name: /Camera \+ inset/ }).click(); + let r = await waitState( + (s) => s.layouts.length === 1 && s.layouts[0].template === "camera-full-pip", + ); + note("A: the layout menu adds a camera-full-pip section", r.ok, brief(r.state.layouts)); + const sectionA = r.state.layouts[0]; + const midA = sectionA ? steadyMid(sectionA) : startA + 1; + const visibleA = sectionA ? visibleSpan(sectionA, readDoc()) : null; + note( + "A: the section lies clear of the project's trims", + Boolean(visibleA) && + Math.abs(visibleA[0] - sectionA.startMs / 1000) < 1e-3 && + Math.abs(visibleA[1] - sectionA.endMs / 1000) < 1e-3, + `visible ${JSON.stringify(visibleA)}, captures at ${midA.toFixed(3)}`, + ); + note( + "A: its two places show the clip's two cameras", + Boolean(sectionA) && + sectionA.slots.length === 2 && + new Set(sectionA.slots.map((s) => s.camera)).size === 2, + JSON.stringify(sectionA?.slots), + ); + const tA = await d.seekTo(midA); + shots.push(await d.shot("A-camera-full-pip")); + await editor.screenshot({ path: path.join(OUT, "A-window.png") }); + const boxes = await editor.evaluate(() => { + const box = (el) => { + const r = el?.getBoundingClientRect(); + return r ? [r.x, r.y, r.width, r.height].map(Math.round) : null; + }; + const preview = document.querySelector('[data-testid="preview"]'); + return { + window: [window.innerWidth, window.innerHeight], + preview: box(preview), + frame: box(preview?.querySelector('[class*="_previewFrame_"]')), + canvas: box(preview?.querySelector("canvas")), + }; + }); + fs.writeFileSync(path.join(OUT, "A-boxes.json"), JSON.stringify(boxes, null, 2)); + note("A: preview captured inside the section", Math.abs(tA - midA) < 0.3, `t=${tA.toFixed(2)}`); + + // B. Choose cameras in the inspector: camera 1 into the large place, then camera 2 again. + note("B: the section is selectable on the timeline", await d.selectPillAt(midA)); + const place1 = editor.getByRole("combobox", { name: /^Place 1/ }); + await place1.waitFor({ state: "visible", timeout: 5000 }); + for (const camera of [0, 1]) { + await place1.selectOption(String(camera)); + r = await waitState( + (s) => + s.layouts.length === 1 && + s.layouts[0].slots[0].camera === camera && + s.layouts[0].slots[1].camera === 1 - camera, + ); + note( + `B: the inspector puts camera ${camera + 1} into place 1 (the other moves to place 2)`, + r.ok, + JSON.stringify(r.state.layouts[0]?.slots), + ); + // Steady state at the middle of the visible span, captured with nothing selected (the + // ruler click clears the selection, so the section is picked again afterwards). + await d.seekTo(midA); + shots.push(await d.shot(`B-camera${camera + 1}-large`)); + const edges = wallpaperAtEdges(await d.frameEdges(`B-camera${camera + 1}-large`), plain); + fs.writeFileSync( + path.join(OUT, `B-camera${camera + 1}-large-edges.json`), + JSON.stringify(edges, null, 2), + ); + const lines = [...edges.left, ...edges.top].filter((l) => !l.seam); + const describe = (side) => + edges[side] + .map((l) => (l.seam ? `${l.line}:seam` : `${l.line}:${l.wallpaperShare}`)) + .join(" "); + note( + `B: camera ${camera + 1} large fills the frame edge to edge`, + lines.length >= 6 && lines.every((l) => l.wallpaperShare < 0.1), + `wallpaper share per line, left ${describe("left")} | top ${describe("top")}`, + ); + await d.selectPillAt(midA); + await place1.waitFor({ state: "visible", timeout: 5000 }); + } + + // C. A Full Camera section (key C) at 9 s, converted to screen + camera and back. + await d.seekTo(clearOfTrims(9, 2)); + await editor.keyboard.press("c"); + r = await waitState((s) => s.fulls.length === 2); + note("C: key C adds a Full Camera section", r.ok, brief(r.state.fulls)); + const fullC = r.state.fulls.find((f) => f.startMs > 8000 && f.startMs < 10_000); + const midC = fullC ? steadyMid(fullC) : 10; + await d.seekTo(midC); + shots.push(await d.shot("C1-full-camera")); + note("C: the Full Camera section is selectable", await d.selectPillAt(midC)); + const templates = editor.getByRole("group", { name: "Layout" }); + await templates.getByRole("button", { name: /Screen \+ camera/ }).click(); + r = await waitState( + (s) => + s.fulls.length === 1 && + s.layouts.length === 2 && + s.layouts.some((l) => l.template === "screen-pip" && Math.abs(l.startMs - fullC.startMs) < 1), + ); + note( + "C: Full Camera → screen-pip moves the section into the layout list", + r.ok, + brief(r.state.layouts), + ); + await sleep(1000); + shots.push(await d.shot("C2-screen-pip")); + await templates.getByRole("button", { name: /^Full camera/ }).click(); + r = await waitState( + (s) => + s.fulls.length === 2 && + s.layouts.length === 1 && + s.fulls.some((f) => Math.abs(f.startMs - fullC.startMs) < 1), + ); + note("C: screen-pip → Full Camera moves it back", r.ok, `fulls ${brief(r.state.fulls)}`); + await sleep(1000); + shots.push(await d.shot("C3-full-camera-again")); + + // D. Undo and redo the conversion back. + await editor.keyboard.press("Control+z"); + r = await waitState( + (s) => s.fulls.length === 1 && s.layouts.some((l) => l.template === "screen-pip"), + ); + note("D: Ctrl+Z restores the screen-pip section", r.ok, brief(r.state.layouts)); + await editor.keyboard.press("Control+Shift+z"); + r = await waitState((s) => s.fulls.length === 2 && s.layouts.length === 1); + note("D: Ctrl+Shift+Z redoes the Full Camera", r.ok, `fulls ${brief(r.state.fulls)}`); + + // E. Drag the inset window of section A in the preview. + await d.seekTo(midA); + await d.selectPillAt(midA); + const place = editor.getByTestId("layout-place").first(); + await place.waitFor({ state: "visible", timeout: 5000 }); + note( + "E: the selected section shows its PiP window in the preview", + true, + `${await editor.getByTestId("layout-place").count()} place(s)`, + ); + const pb = await place.boundingBox(); + const rectBefore = cameraState().layouts[0]?.slots[1]?.rect ?? null; + await editor.mouse.move(pb.x + pb.width / 2, pb.y + pb.height / 2); + await editor.mouse.down(); + for (let i = 1; i <= 10; i++) { + await editor.mouse.move(pb.x + pb.width / 2 - 12 * i, pb.y + pb.height / 2 - 6 * i); + await sleep(30); + } + await editor.mouse.up(); + r = await waitState((s) => { + const rect = s.layouts[0]?.slots[1]?.rect; + return Boolean(rect) && JSON.stringify(rect) !== JSON.stringify(rectBefore); + }); + note( + "E: dragging the PiP window stores its rect", + r.ok, + JSON.stringify(r.state.layouts[0]?.slots[1]?.rect), + ); + await sleep(1000); + shots.push(await d.shot("E-pip-dragged")); + + // F. Calibration: perspective of camera 2, from the camera list. + await d.seekTo(midA); // the ruler click also clears the selection + const rail = editor.getByRole("button", { name: "Camera layout", exact: true }); + if ((await rail.getAttribute("aria-pressed")) !== "true") await rail.click(); + await editor + .getByTestId("camera-row-1") + .getByRole("button", { name: /Correct perspective/ }) + .click(); + const dialog = editor.getByRole("dialog"); + await dialog.waitFor({ state: "visible", timeout: 5000 }); + await dialog + .getByText(/Loading the camera picture/) + .waitFor({ state: "detached", timeout: 15_000 }); + note( + "F: the dialog loads the camera still", + !(await dialog.getByText(/could not be loaded/).count()), + ); + await dialog.getByRole("button", { name: "Detect markers" }).click(); + const status = dialog.getByRole("status"); + const statusText = (await status.innerText()).trim(); + note("F: marker detection reports a result", statusText.length > 0, statusText); + const h0 = dialog.getByTestId("calibration-handle-0"); + await h0.focus(); + for (let i = 0; i < 4; i++) await editor.keyboard.press("Shift+ArrowRight"); + for (let i = 0; i < 4; i++) await editor.keyboard.press("Shift+ArrowDown"); + note("F: moving a corner clears the marker message", (await status.innerText()).trim() === ""); + const h2 = await dialog.getByTestId("calibration-handle-2").boundingBox(); + await editor.mouse.move(h2.x + h2.width / 2, h2.y + h2.height / 2); + await editor.mouse.down(); + for (let i = 1; i <= 8; i++) { + await editor.mouse.move(h2.x + h2.width / 2 - 5 * i, h2.y + h2.height / 2 - 3 * i); + await sleep(30); + } + await editor.mouse.up(); + await sleep(500); + await dialog.screenshot({ path: path.join(OUT, "F1-calibration-dialog.png") }); + shots.push(path.join(OUT, "F1-calibration-dialog.png")); + await dialog.getByRole("button", { name: "Apply", exact: true }).click(); + r = await waitState((s) => { + const p = s.settings?.[1]?.perspective; + return Boolean(p) && p.corners.length === 4 && Math.abs(p.aspect - 297 / 210) < 1e-6; + }); + const corners = r.state.settings?.[1]?.perspective?.corners; + note( + "F: Apply stores camera 2's perspective (A4 landscape)", + r.ok && corners[0].x > 0.15 && corners[2].x < 0.85, + JSON.stringify(r.state.settings?.[1]?.perspective), + ); + await d.seekTo(midA); + await sleep(1000); + shots.push(await d.shot("F2-perspective-in-section")); + + // G. Save, relaunch, compare. + await editor.keyboard.press("Control+s"); + await sleep(1500); + const saved = cameraState(); + await close(session); + session = null; + + session = await launch(); + const d2 = driver(session.editor, session.preview); + const reopened = cameraState(); + note( + "G: the reopened project keeps sections and camera settings", + JSON.stringify(reopened) === JSON.stringify(saved), + `${reopened.layouts.length} layout(s), ${reopened.fulls.length} full camera(s)`, + ); + note("G: the timeline shows all three camera sections", (await d2.cameraPillCount()) === 3); + await d2.seekTo(midA); + shots.push(await d2.shot("G-reopened-section-A")); + await d2.seekTo(midC); + shots.push(await d2.shot("G-reopened-full-camera")); + fs.writeFileSync(path.join(OUT, "final-state.json"), JSON.stringify(reopened, null, 2)); +} catch (error) { + note("run completed without an exception", false, String(error?.stack ?? error)); +} finally { + note( + "no console errors in the editor", + consoleErrors.length === 0, + consoleErrors.slice(0, 5).join(" | "), + ); + if (session) await close(session); + log.end(); + fs.rmSync(COPY_FILE, { force: true }); + console.log(`removed copy: ${!fs.existsSync(COPY_FILE)}`); + fs.writeFileSync(path.join(OUT, "results.json"), JSON.stringify({ results, shots }, null, 2)); + exitCode = results.every((x) => x.ok) ? 0 : 1; +} +process.exit(exitCode); diff --git a/scripts/multicam-fixture.mjs b/scripts/multicam-fixture.mjs new file mode 100644 index 000000000..ac5b460fe --- /dev/null +++ b/scripts/multicam-fixture.mjs @@ -0,0 +1,933 @@ +// Several cameras in the picture, measured end to end (dev tool, not part of the build). +// +// Generates synthetic sources and two projects, runs the real headless export, and checks the +// exported pixels against the camera layout templates and the glide the compositor plans. +// +// node scripts/multicam-fixture.mjs generate +// screen.mp4 (1920x1080 dark grey, 8 s), cam0.mp4 (solid red), cam1.mp4 (solid green), +// cam2.mp4 (a 16x9 checkerboard warped "in perspective" onto known corners), all +// 1280x720, plus multicam.openscreen (three cameras, four layout regions, a perspective +// on camera 2) and onecam.openscreen (the same take with camera 1 only, no regions). +// Also board43.mp4 (a 12x9 board, 4:3, warped onto the same corners) and two more +// projects: persp.openscreen (camera 1 IS that board with a 4:3 perspective, in its +// default PiP and then in a Full Camera section) and seam.openscreen (a camera-1 Full +// Camera section directly followed by a camera-full-pip [2, 1] layout region). +// node scripts/multicam-fixture.mjs export +// Runs `electron . export` and prints the wall-clock time. Run from a built tree +// (`npm run build-vite`) with OPENSCREEN_COMPOSITOR_VIEW_NODE pointing at the addon and +// its ffmpeg DLLs first on PATH. +// node scripts/multicam-fixture.mjs check [multicam|persp|seam] +// Extracts frames by index and prints a pass/fail table; exits 1 on any failure. +// The project defaults to multicam. +// +// The expected rects (`pipRect`, `regionLayers`, `plannedRects`) MIRROR the templates and +// the planner; they are not an independent derivation. The template PiPs anchor on camera +// 1's default PiP, which the check measures (red box at 0.5 s) instead of recomputing it. +// +// ffmpeg: OPENSCREEN_FFMPEG, else the vendored build under crates/thirdparty. + +import { spawn, spawnSync } from "node:child_process"; +import fs from "node:fs"; +import { createRequire } from "node:module"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; + +const ROOT = path.join(path.dirname(fileURLToPath(import.meta.url)), ".."); +const FFMPEG = + process.env.OPENSCREEN_FFMPEG ?? + path.join(ROOT, "crates", "thirdparty", "ffmpeg-n8.1.2-win64-lgpl-shared", "bin", "ffmpeg.exe"); + +const DURATION_S = 8; +const CAM_W = 1280; +const CAM_H = 720; +const SQUARE_PX = 80; // 16 x 9 squares on the 1280x720 board +// Where the board's corners land in camera 2's picture (pixels): TL, TR, BR, BL. +const BOARD_CORNERS_PX = [ + [160, 90], + [1050, 40], + [1180, 680], + [70, 600], +]; +const BOARD_ASPECT = CAM_W / CAM_H; +// Camera 1's board in persp.openscreen: 12 x 9 squares, so its corrected picture is 4:3 while +// the camera's own frame is 16:9 — a box laid out from the raw camera would stretch it. +const BOARD43 = { cols: 12, rows: 9, aspect: 4 / 3 }; + +// Mirrors src/lib/cameraLayoutTemplates.ts. +const PIP_WIDTH_FRAC = 0.22; +const PIP_MARGIN_FRAC = 0.025; +const PIP_GAP_FRAC = 0.02; +// Mirrors crates/compositor/src/regions.rs. +const TRANSITION_WINDOW_S = 1.01505; +const FULLSCREEN_LEAD_OUT_WINDOW_S = TRANSITION_WINDOW_S * 1.5; +const EXPORT_FPS = 60; +const TOLERANCE_PX = 2; + +const PERSP_REGIONS = [ + { id: "lay_persp_full", startMs: 3000, endMs: 6000, template: "camera-full", cameras: [0] }, +]; +const SEAM_REGIONS = [ + { id: "lay_seam_full", startMs: 1000, endMs: 3000, template: "camera-full", cameras: [0] }, + { id: "lay_seam_pip", startMs: 3000, endMs: 5000, template: "camera-full-pip", cameras: [1, 0] }, +]; + +const REGIONS = [ + { id: "lay_screen_pip", startMs: 1000, endMs: 3000, template: "screen-pip", cameras: [0, 1] }, + { id: "lay_full_pip", startMs: 3000, endMs: 5000, template: "camera-full-pip", cameras: [1, 0] }, + { id: "lay_side", startMs: 5500, endMs: 7000, template: "side-by-side", cameras: [0, 1] }, + { id: "lay_board", startMs: 7000, endMs: 8000, template: "camera-full", cameras: [2] }, +]; + +function run(cmd, args) { + const result = spawnSync(cmd, args, { encoding: "utf8", maxBuffer: 1 << 28 }); + if (result.status !== 0) { + throw new Error( + `${path.basename(cmd)} failed (${result.status}): ${result.stderr?.slice(-2000)}`, + ); + } + return result; +} + +function encode(input, filter, out) { + run(FFMPEG, [ + "-y", + "-hide_banner", + "-loglevel", + "error", + "-f", + "lavfi", + "-i", + input, + ...(filter ? ["-vf", filter] : []), + "-t", + String(DURATION_S), + "-pix_fmt", + "yuv420p", + "-c:v", + "libopenh264", + "-b:v", + "12M", + out, + ]); +} + +/** Unit square (0,0),(1,0),(1,1),(0,1) onto four points (Heckbert), as `cameraPerspective.ts`. */ +function homographyFromUnitSquare([p0, p1, p2, p3]) { + const dx1 = p1[0] - p2[0]; + const dx2 = p3[0] - p2[0]; + const dx3 = p0[0] - p1[0] + p2[0] - p3[0]; + const dy1 = p1[1] - p2[1]; + const dy2 = p3[1] - p2[1]; + const dy3 = p0[1] - p1[1] + p2[1] - p3[1]; + const den = dx1 * dy2 - dx2 * dy1; + const g = (dx3 * dy2 - dx2 * dy3) / den; + const h = (dx1 * dy3 - dx3 * dy1) / den; + return [ + p1[0] - p0[0] + g * p1[0], + p3[0] - p0[0] + h * p3[0], + p0[0], + p1[1] - p0[1] + g * p1[1], + p3[1] - p0[1] + h * p3[1], + p0[1], + g, + h, + 1, + ]; +} + +function invert3([a, b, c, d, e, f, g, h, i]) { + const det = a * (e * i - f * h) - b * (d * i - f * g) + c * (d * h - e * g); + return [ + (e * i - f * h) / det, + (c * h - b * i) / det, + (b * f - c * e) / det, + (f * g - d * i) / det, + (a * i - c * g) / det, + (c * d - a * f) / det, + (d * h - e * g) / det, + (b * g - a * h) / det, + (a * e - b * d) / det, + ]; +} + +function generate(dir) { + fs.mkdirSync(dir, { recursive: true }); + const media = { + screen: path.join(dir, "screen.mp4"), + cam0: path.join(dir, "cam0.mp4"), + cam1: path.join(dir, "cam1.mp4"), + cam2: path.join(dir, "cam2.mp4"), + }; + encode(`color=c=0x303030:s=1920x1080:r=30:d=${DURATION_S}`, null, media.screen); + encode(`color=c=red:s=${CAM_W}x${CAM_H}:r=30:d=${DURATION_S}`, null, media.cam0); + encode(`color=c=0x00ff00:s=${CAM_W}x${CAM_H}:r=30:d=${DURATION_S}`, null, media.cam1); + // ffmpeg's `perspective` filter is GPL-only, and the vendored build is LGPL: the board is + // drawn straight in perspective instead. Each output pixel centre is mapped back onto the + // board's unit square by the inverse homography, so the true corners are exact. + const inv = invert3(homographyFromUnitSquare(BOARD_CORNERS_PX)); + const [s, u, w] = [0, 3, 6].map((r) => `(${inv[r]}*(X+0.5)+${inv[r + 1]}*(Y+0.5)+${inv[r + 2]})`); + const board = (cols, rows) => + `geq=lum='st(0,${s}/${w});st(1,${u}/${w});` + + `if(lt(ld(0),0)+gte(ld(0),1)+lt(ld(1),0)+gte(ld(1),1),128,` + + `if(mod(floor(ld(0)*${cols})+floor(ld(1)*${rows}),2),235,16))':cb=128:cr=128`; + const black = `color=c=black:s=${CAM_W}x${CAM_H}:r=30:d=${DURATION_S},format=yuv420p`; + encode(black, board(CAM_W / SQUARE_PX, CAM_H / SQUARE_PX), media.cam2); + media.board43 = path.join(dir, "board43.mp4"); + encode(black, board(BOARD43.cols, BOARD43.rows), media.board43); + + const track = (sourcePath, label) => ({ + sourcePath, + startMs: 0, + offsetMs: 0, + visible: true, + width: CAM_W, + height: CAM_H, + ...(label === undefined ? {} : { label }), + }); + const corners = BOARD_CORNERS_PX.map(([x, y]) => ({ x: x / CAM_W, y: y / CAM_H })); + const slotted = (regions) => + regions.map(({ cameras, ...region }) => ({ + ...region, + slots: cameras.map((camera) => ({ camera })), + })); + // kind: "multicam" | "onecam" | "persp" | "seam". + const project = (kind) => ({ + schemaVersion: 8, + project: { + id: `proj_${kind}_fixture`, + title: `${kind} fixture`, + createdAt: "2026-10-05T00:00:00.000Z", + updatedAt: "2026-10-05T00:00:00.000Z", + primaryAssetId: "asset_screen", + }, + assets: [ + { + id: "asset_screen", + kind: "video", + label: "screen.mp4", + originalPath: media.screen, + durationSec: DURATION_S, + sizeBytes: fs.statSync(media.screen).size, + video: { codec: "unknown", width: 1920, height: 1080, fps: 30 }, + cameraTrack: track(kind === "persp" ? media.board43 : media.cam0), + ...(kind === "multicam" + ? { + additionalCameraTracks: [track(media.cam1, "Green"), track(media.cam2, "Board")], + } + : {}), + ...(kind === "seam" ? { additionalCameraTracks: [track(media.cam1, "Green")] } : {}), + }, + ], + transcript: null, + transcripts: [], + timeline: { + clips: [ + { + id: "clip_main", + assetId: "asset_screen", + sourceStartSec: 0, + sourceEndSec: DURATION_S, + timelineStartSec: 0, + timelineEndSec: DURATION_S, + wordRefs: [], + origin: "user", + reason: "Fixture", + }, + ], + gaps: [], + trimRanges: [], + muteRanges: [], + speedRanges: [], + captionRanges: [], + }, + annotations: [], + zoomRanges: [], + audioTracks: [], + legacyEditor: { + wallpaper: "#101010", + wallpaperMotion: "none", + frame: "none", + shadowIntensity: 0, + backgroundBlur: 0, + motionBlurAmount: 0, + webcamLayoutPreset: "picture-in-picture", + webcamMaskShape: "rectangle", + webcamRoundness: 0.1, + webcamBackgroundMode: "none", + ...(kind === "multicam" + ? { + cameraLayoutRegions: slotted(REGIONS), + cameraSettings: [null, null, { perspective: { corners, aspect: BOARD_ASPECT } }], + } + : {}), + ...(kind === "persp" + ? { + cameraLayoutRegions: slotted(PERSP_REGIONS), + cameraSettings: [{ perspective: { corners, aspect: BOARD43.aspect } }], + } + : {}), + ...(kind === "seam" ? { cameraLayoutRegions: slotted(SEAM_REGIONS) } : {}), + }, + }); + for (const kind of ["multicam", "onecam", "persp", "seam"]) { + const file = path.join(dir, `${kind}.openscreen`); + fs.writeFileSync(file, `${JSON.stringify(project(kind), null, 2)}\n`); + console.log(`wrote ${file}`); + } +} + +/** + * Runs the headless export and prints its wall-clock time, and the render time alone: from + * the first progress line to "Exported", which leaves Electron's start-up and the project + * load out. + */ +function exportProject(projectPath, outPath) { + const electron = createRequire(import.meta.url)("electron"); + const started = performance.now(); + let firstProgress = null; + let exported = null; + const child = spawn(electron, [ROOT, "export", projectPath, "-o", outPath], { + cwd: ROOT, + stdio: ["ignore", "pipe", "pipe"], + }); + const watch = (chunk, sink) => { + const text = chunk.toString(); + if (firstProgress === null && text.includes("Exporting")) firstProgress = performance.now(); + if (exported === null && text.includes("Exported")) exported = performance.now(); + sink.write(text); + }; + child.stdout.on("data", (chunk) => watch(chunk, process.stdout)); + child.stderr.on("data", (chunk) => watch(chunk, process.stderr)); + child.once("exit", (code) => { + const seconds = (performance.now() - started) / 1000; + const render = + firstProgress !== null && exported !== null + ? `${((exported - firstProgress) / 1000).toFixed(2)} s` + : "unknown"; + console.log(`export exit=${code} wall=${seconds.toFixed(2)} s render=${render}`); + process.exitCode = code ?? 1; + }); +} + +// --- checking --------------------------------------------------------------------------- + +/** Minimal P6 reader: {width, height, data} with 3 bytes per pixel. */ +function readPpm(file) { + const buf = fs.readFileSync(file); + const fields = []; + let i = 0; + while (fields.length < 4) { + while (/\s/.test(String.fromCharCode(buf[i]))) i++; + if (buf[i] === 0x23) { + while (buf[i] !== 0x0a) i++; + continue; + } + let token = ""; + while (!/\s/.test(String.fromCharCode(buf[i]))) token += String.fromCharCode(buf[i++]); + fields.push(token); + } + if (fields[0] !== "P6" || fields[3] !== "255") throw new Error(`${file}: not an 8-bit P6`); + return { width: Number(fields[1]), height: Number(fields[2]), data: buf.subarray(i + 1) }; +} + +/** One ffmpeg pass that writes the given frame indices as PPMs, in order. */ +function extractFrames(video, indices, dir) { + fs.rmSync(dir, { recursive: true, force: true }); + fs.mkdirSync(dir, { recursive: true }); + const select = indices.map((n) => `eq(n\\,${n})`).join("+"); + run(FFMPEG, [ + "-hide_banner", + "-loglevel", + "error", + "-i", + video, + "-vf", + `select='${select}'`, + "-fps_mode", + "passthrough", + "-c:v", + "ppm", + path.join(dir, "f%04d.ppm"), + ]); + return new Map( + indices.map((n, k) => [n, readPpm(path.join(dir, `f${String(k + 1).padStart(4, "0")}.ppm`))]), + ); +} + +const isRed = (r, g, b) => r > 150 && g < 100 && b < 100; +const isGreen = (r, g, b) => g > 150 && r < 100 && b < 100; + +/** Bounding box [x0, y0, x1, y1) of the pixels `test` accepts, and how many there are. */ +function bbox(frame, test) { + let x0 = Infinity; + let y0 = Infinity; + let x1 = -1; + let y1 = -1; + let count = 0; + const { width, height, data } = frame; + for (let y = 0; y < height; y++) { + for (let x = 0; x < width; x++) { + const o = (y * width + x) * 3; + if (!test(data[o], data[o + 1], data[o + 2])) continue; + count++; + if (x < x0) x0 = x; + if (x > x1) x1 = x; + if (y < y0) y0 = y; + if (y > y1) y1 = y; + } + } + return count === 0 ? null : { box: [x0, y0, x1 + 1, y1 + 1], count }; +} + +/** + * A template PiP of a 16:9 camera, like `pipLayer`: anchored on camera 1's default PiP + * (`anchor`, [x, y, w, h] fractions: its width, right and bottom edges), stacking leftwards; + * the corner constants without one. + */ +function pipRect(index, frame, anchor) { + const w = anchor ? anchor[2] : PIP_WIDTH_FRAC; + const h = (w * frame.width) / (16 / 9) / frame.height; + const firstRight = anchor ? anchor[0] + anchor[2] : 1 - PIP_MARGIN_FRAC; + const right = firstRight - index * (w + PIP_GAP_FRAC); + const bottom = anchor + ? anchor[1] + anchor[3] + : 1 - (PIP_MARGIN_FRAC * frame.width) / frame.height; + return [right - w, bottom - h, w, h]; +} + +/** Each region's layers as {camera: [x, y, w, h]} (fractions), like `resolveCameraLayout`. */ +function regionLayers(region, frame, anchor) { + const out = {}; + region.cameras.forEach((camera, i) => { + if (region.template === "camera-full" || (region.template === "camera-full-pip" && i === 0)) { + out[camera] = [0, 0, 1, 1]; + } else if (region.template === "side-by-side") { + out[camera] = i === 0 ? [0, 0, 0.5, 1] : [0.5, 0, 0.5, 1]; + } else { + out[camera] = pipRect(region.template === "camera-full-pip" ? i - 1 : i, frame, anchor); + } + }); + return out; +} + +function cubicBezier(x1, y1, x2, y2, x) { + const sample = (a, b, t) => ((1 - 3 * b + 3 * a) * t + (3 * b - 6 * a)) * t * t + 3 * a * t; + let lo = 0; + let hi = 1; + for (let i = 0; i < 60; i++) { + const mid = (lo + hi) / 2; + if (sample(x1, x2, mid) < x) lo = mid; + else hi = mid; + } + return sample(y1, y2, (lo + hi) / 2); +} +const ease = (t) => cubicBezier(0.16, 1, 0.3, 1, Math.min(1, Math.max(0, t))); + +/** + * The rect each camera is planned at, at source time t: `camera_layers_at` in + * crates/compositor/src/camera_layers.rs for these regions (no speed regions, so the screen + * clock is the source clock). `fallback` is camera 0's default rect outside every region. + * Opacity is ignored: the check only looks at where a camera is. + */ +function plannedRects(t, frame, fallback, regions = REGIONS) { + const sec = (ms) => ms / 1000; + // A camera-1 camera-full region is a Full Camera section, not a layout region; it is a + // layout region's neighbour only at a seam, with camera 1 on the whole frame. + const isFullCamera = (r) => r.template === "camera-full" && r.cameras[0] === 0; + const layouts = regions.filter((r) => !isFullCamera(r)); + const index = layouts.findIndex((r) => sec(r.startMs) <= t && t <= sec(r.endMs)); + const fallbackLayers = { 0: fallback }; + if (index < 0) return fallbackLayers; + const region = layouts[index]; + const [start, end] = [sec(region.startMs), sec(region.endMs)]; + const prev = regions.find((r) => r !== region && Math.abs(sec(r.endMs) - start) <= 0.001); + const hasNext = regions.some((r) => r !== region && Math.abs(sec(r.startMs) - end) <= 0.001); + const current = regionLayers(region, frame, fallback); + const half = (end - start) / 2; + const winIn = Math.min(TRANSITION_WINDOW_S, half); + const winOut = Math.min(FULLSCREEN_LEAD_OUT_WINDOW_S, half); + const blend = (from, to, k) => { + const out = {}; + for (const [camera, rect] of Object.entries(to)) { + const a = from[camera]; + out[camera] = a ? rect.map((v, i) => a[i] + (v - a[i]) * k) : rect; + } + for (const [camera, rect] of Object.entries(from)) out[camera] ??= rect; + return out; + }; + if (t - start < winIn) { + return blend( + prev ? regionLayers(prev, frame, fallback) : fallbackLayers, + current, + ease((t - start) / winIn), + ); + } + if (!hasNext && end - t < winOut) { + return blend(current, fallbackLayers, 1 - ease((end - t) / winOut)); + } + return current; +} + +const toPx = ([x, y, w, h], frame) => [ + x * frame.width, + y * frame.height, + (x + w) * frame.width, + (y + h) * frame.height, +]; +/** Largest edge distance (px) between a measured box and a planned rect. */ +function boxError(measured, rect, frame) { + if (!measured) return Infinity; + const want = toPx(rect, frame); + return Math.max(...measured.box.map((v, i) => Math.abs(v - want[i]))); +} +const fmtBox = (m) => (m ? `[${m.box.join(", ")}]` : "none"); +const fmtRect = (rect, frame) => + `[${toPx(rect, frame) + .map((v) => v.toFixed(1)) + .join(", ")}]`; + +/** Sub-pixel positions where the luminance along a line crosses mid-grey. */ +function edgesAlong(frame, horizontal, at) { + const { width, height, data } = frame; + const n = horizontal ? width : height; + const lum = (i) => { + const o = horizontal ? (at * width + i) * 3 : (i * width + at) * 3; + return (data[o] + data[o + 1] + data[o + 2]) / 3; + }; + const edges = []; + for (let i = 1; i < n; i++) { + const a = lum(i - 1) - 128; + const b = lum(i) - 128; + if (a < 0 !== b < 0) edges.push(i - 1 + a / (a - b) + 0.5); + } + return edges; +} + +/** + * The rectified board, cover-fitted into the frame like any other camera: squares of one size + * `s = max(W / cols, H / rows)`, centred, so its edges sit at `centre + k * s` on both axes. + * Checks three rows and three columns: edges where the square grid puts them (right angles and + * equal square sizes), and equally spaced within each line. + */ +function boardCheck(frame, cols = CAM_W / SQUARE_PX, rows = CAM_H / SQUARE_PX) { + const { width, height } = frame; + const size = Math.max(width / cols, height / rows); + const along = (centre, count, extent) => { + const out = []; + for (let k = -count; k <= count; k++) { + const e = centre + (k - count / 2) * size; + if (e > 3 && e < extent - 3) out.push(e); + } + return out; + }; + const lines = [ + ...[1, 4, 7].map((r) => ({ + horizontal: true, + at: Math.round(height / 2 + (r - rows / 2 + 0.5) * size), + want: along(width / 2, cols, width), + })), + ...[2, Math.floor(cols / 2), cols - 3].map((c) => ({ + horizontal: false, + at: Math.round(width / 2 + (c - cols / 2 + 0.5) * size), + want: along(height / 2, rows, height), + })), + ]; + let spacingError = 0; + let gridError = 0; + const found = []; + const steps = { horizontal: [], vertical: [] }; + for (const line of lines) { + const edges = edgesAlong(frame, line.horizontal, line.at); + found.push(`${edges.length}/${line.want.length}`); + if (edges.length < 2) { + spacingError = Infinity; + gridError = Infinity; + continue; + } + const step = (edges[edges.length - 1] - edges[0]) / (edges.length - 1); + steps[line.horizontal ? "horizontal" : "vertical"].push(step); + edges.forEach((e, k) => { + spacingError = Math.max(spacingError, Math.abs(e - (edges[0] + k * step))); + }); + if (edges.length !== line.want.length) { + gridError = Infinity; + continue; + } + edges.forEach((e, k) => { + gridError = Math.max(gridError, Math.abs(e - line.want[k])); + }); + } + const mean = (v) => v.reduce((x, y) => x + y, 0) / v.length; + return { + lines: lines.length, + found, + size, + spacingError, + gridError, + stepH: mean(steps.horizontal), + stepV: mean(steps.vertical), + }; +} + +function report(rows) { + for (const r of rows) { + const verdict = r.pass === null ? "INFO" : r.pass ? "PASS" : "FAIL"; + console.log(`${verdict} | ${r.name} | ${r.measured} | expected ${r.expected}`); + } + const checked = rows.filter((r) => r.pass !== null); + const failed = checked.filter((r) => !r.pass).length; + console.log(`${checked.length - failed}/${checked.length} passed`); + process.exitCode = failed === 0 ? 0 : 1; +} + +const fractionsOf = (found, frame) => + found + ? [ + found.box[0] / frame.width, + found.box[1] / frame.height, + (found.box[2] - found.box[0]) / frame.width, + (found.box[3] - found.box[1]) / frame.height, + ] + : null; + +const isWhite = (r, g, b) => r > 200 && g > 200 && b > 200; + +/** + * The board inside a box (`box` = [x0, y0, x1, y1) px): the mean square step across (from + * three rows) and down (from three columns), from the luminance edges strictly inside it. + */ +function boardInBox(frame, box, cols, rows) { + const [x0, y0, x1, y1] = box; + const { width, data } = frame; + const lum = (x, y) => { + const o = (y * width + x) * 3; + return (data[o] + data[o + 1] + data[o + 2]) / 3 - 128; + }; + const edges = (horizontal, at) => { + const out = []; + const [lo, hi] = horizontal ? [x0 + 3, x1 - 3] : [y0 + 3, y1 - 3]; + for (let i = lo + 1; i < hi; i++) { + const a = horizontal ? lum(i - 1, at) : lum(at, i - 1); + const b = horizontal ? lum(i, at) : lum(at, i); + if (a < 0 !== b < 0) out.push(i - 1 + a / (a - b) + 0.5); + } + return out; + }; + const step = (e) => (e.length < 2 ? Number.NaN : (e[e.length - 1] - e[0]) / (e.length - 1)); + const [bw, bh] = [x1 - x0, y1 - y0]; + const across = [1, Math.floor(rows / 2), rows - 2].map((r) => + step(edges(true, Math.round(y0 + ((r + 0.5) * bh) / rows))), + ); + const down = [2, Math.floor(cols / 2), cols - 3].map((c) => + step(edges(false, Math.round(x0 + ((c + 0.5) * bw) / cols))), + ); + const mean = (v) => v.reduce((x, y) => x + y, 0) / v.length; + return { stepH: mean(across), stepV: mean(down) }; +} + +function checkPersp(video) { + const frameAt = (sec) => Math.round(sec * EXPORT_FPS); + const frames = extractFrames( + video, + [frameAt(1), frameAt(4.25)], + path.join(os.tmpdir(), "openscreen-multicam-frames"), + ); + const rows = []; + const add = (name, pass, measured, expected) => rows.push({ name, pass, measured, expected }); + const { cols, rows: boardRows, aspect } = BOARD43; + + // Default PiP: the box takes the perspective's 4:3, and its squares are square. + const pipFrame = frames.get(frameAt(1)); + const pip = bbox(pipFrame, isWhite); + const pipAspect = pip ? (pip.box[2] - pip.box[0]) / (pip.box[3] - pip.box[1]) : Number.NaN; + const pipHeight = pip ? pip.box[3] - pip.box[1] : 0; + add( + "1.0 s camera 1 PiP box has the perspective's 4:3", + Math.abs(pipAspect * pipHeight - aspect * pipHeight) <= TOLERANCE_PX, + `${fmtBox(pip)} ratio ${pipAspect.toFixed(4)}`, + `${aspect.toFixed(4)} within ${TOLERANCE_PX} px of width`, + ); + const inPip = pip ? boardInBox(pipFrame, pip.box, cols, boardRows) : { stepH: NaN, stepV: NaN }; + add( + "1.0 s PiP checkerboard squares square (step across vs down)", + Math.abs(inPip.stepH - inPip.stepV) <= TOLERANCE_PX, + `${inPip.stepH.toFixed(2)} x ${inPip.stepV.toFixed(2)} px`, + `equal within ${TOLERANCE_PX} px`, + ); + + // Full Camera: the 4:3 picture cover-fitted into the frame, squares square and on the grid. + const board = boardCheck(frames.get(frameAt(4.25)), cols, boardRows); + add( + "4.25 s Full Camera checkerboard edges equally spaced", + board.spacingError <= TOLERANCE_PX, + `${board.lines} lines, edges ${board.found.join(" ")}, max err ${board.spacingError.toFixed(2)} px`, + `<= ${TOLERANCE_PX} px`, + ); + add( + "4.25 s Full Camera checkerboard squares square (step across vs down)", + Math.abs(board.stepH - board.stepV) <= TOLERANCE_PX, + `${board.stepH.toFixed(2)} x ${board.stepV.toFixed(2)} px`, + `equal within ${TOLERANCE_PX} px`, + ); + add( + "4.25 s Full Camera checkerboard on the cover-fitted square grid", + board.gridError <= TOLERANCE_PX, + `max err ${board.gridError.toFixed(2)} px`, + `squares of ${board.size.toFixed(1)} px, centred`, + ); + report(rows); +} + +function checkSeam(video) { + const t = (n) => n / EXPORT_FPS; + const frameAt = (sec) => Math.round(sec * EXPORT_FPS); + const sweep = []; + // From after the Full Camera section's own lead-in (1.0 s + 1.015 s) across the seam. + for (let n = frameAt(2.05); n <= frameAt(4.2); n += EXPORT_FPS / 30) sweep.push(n); + const indices = [...new Set([frameAt(0.5), frameAt(4.6), ...sweep])].sort((a, b) => a - b); + const frames = extractFrames( + video, + indices, + path.join(os.tmpdir(), "openscreen-multicam-frames"), + ); + const rows = []; + const add = (name, pass, measured, expected) => rows.push({ name, pass, measured, expected }); + const f05 = frames.get(frameAt(0.5)); + const fallback = fractionsOf(bbox(f05, isRed), f05); + add("0.5 s camera 1 default PiP present", !!fallback, fmtBox(bbox(f05, isRed)), "red box"); + + // Up to the seam camera 1 fills the frame: no lead-out of the Full Camera section. + let notFull = 0; + let worstFull = 0; + // From the seam on, camera 1 is on the planned glide from the frame to its PiP slot, and + // its box only ever shrinks: it never passes through the default PiP and back. (The slot + // IS the default PiP's rect — template PiPs anchor on it — so where it ends proves + // nothing; where the glide starts does.) + let worstGlide = 0; + let grows = 0; + let lastArea = Infinity; + let firstAfterSeam = null; + const trace = []; + for (const n of sweep) { + const frame = frames.get(n); + const got = bbox(frame, isRed); + const area = got ? (got.box[2] - got.box[0]) * (got.box[3] - got.box[1]) : 0; + trace.push(`${t(n).toFixed(2)}:${got ? got.box[0] : "-"}`); + if (t(n) < 3) { + const err = boxError(got, [0, 0, 1, 1], frame); + worstFull = Math.max(worstFull, err); + if (err > TOLERANCE_PX) notFull++; + } else { + const want = plannedRects(t(n), frame, fallback, SEAM_REGIONS)[0]; + worstGlide = Math.max(worstGlide, boxError(got, want, frame)); + if (area > lastArea + 4 * frame.width) grows++; + if (t(n) > 3 && firstAfterSeam === null) firstAfterSeam = { n, got, width: frame.width }; + } + lastArea = t(n) >= 3 ? area : lastArea; + } + add( + "2.05-3.0 s camera 1 fills the frame up to the seam", + notFull === 0, + `${notFull} frames off, max err ${worstFull.toFixed(2)} px`, + `0 (<= ${TOLERANCE_PX} px)`, + ); + add( + "3.0-4.2 s camera 1 on the planned glide from the frame", + worstGlide <= TOLERANCE_PX, + `max err ${worstGlide.toFixed(2)} px; left edge ${trace.filter((_, k) => k % 6 === 0).join(" ")}`, + `<= ${TOLERANCE_PX} px`, + ); + add("3.0-4.2 s camera 1 box never grows back", grows === 0, `${grows} growing steps`, "0"); + { + // The first frame after the seam is barely into the glide: still nearly the whole frame. + // Gliding from the default PiP instead, it would already be PiP-sized. + const first = firstAfterSeam; + const w = first?.got ? first.got.box[2] - first.got.box[0] : 0; + add( + `${first ? t(first.n).toFixed(3) : "?"} s glide starts from the frame, not the default PiP`, + !!first && w >= 0.75 * first.width, + first ? `${fmtBox(first.got)} width ${w} px` : "no frame", + `>= ${first ? (0.75 * first.width).toFixed(0) : "?"} px wide`, + ); + } + const f46 = frames.get(frameAt(4.6)); + const slot = plannedRects(t(frameAt(4.6)), f46, fallback, SEAM_REGIONS); + const red = bbox(f46, isRed); + const redErr = boxError(red, slot[0], f46); + add( + "4.6 s camera 1 in its PiP slot", + redErr <= TOLERANCE_PX, + `${fmtBox(red)} err ${redErr.toFixed(2)} px`, + fmtRect(slot[0], f46), + ); + const green = bbox(f46, isGreen); + add( + "4.6 s camera 2 fills the frame under it", + boxError(green, [0, 0, 1, 1], f46) <= TOLERANCE_PX, + fmtBox(green), + fmtRect([0, 0, 1, 1], f46), + ); + report(rows); +} + +function check(video) { + const t = (n) => n / EXPORT_FPS; + const frameAt = (sec) => Math.round(sec * EXPORT_FPS); + const glide = []; + for (let n = frameAt(2.9); n <= frameAt(3.6); n += EXPORT_FPS / 30) glide.push(n); + const indices = [ + ...new Set([ + frameAt(0.5), + frameAt(2), + frameAt(4), + frameAt(6), + frameAt(6.5), + frameAt(7.5), + ...glide, + ]), + ].sort((a, b) => a - b); + const frames = extractFrames( + video, + indices, + path.join(os.tmpdir(), "openscreen-multicam-frames"), + ); + const rows = []; + const add = (name, pass, measured, expected) => rows.push({ name, pass, measured, expected }); + + const f05 = frames.get(frameAt(0.5)); + const fallbackBox = bbox(f05, isRed); + const fallback = fallbackBox + ? [ + fallbackBox.box[0] / f05.width, + fallbackBox.box[1] / f05.height, + (fallbackBox.box[2] - fallbackBox.box[0]) / f05.width, + (fallbackBox.box[3] - fallbackBox.box[1]) / f05.height, + ] + : null; + add( + "0.5 s camera 1 default PiP present (reference for 6.0 s)", + !!fallbackBox, + fmtBox(fallbackBox), + "red box", + ); + add("0.5 s no other camera", !bbox(f05, isGreen), fmtBox(bbox(f05, isGreen)), "no green"); + + const atRect = (sec, camera, test, label) => { + const frame = frames.get(frameAt(sec)); + const want = plannedRects(t(frameAt(sec)), frame, fallback)[camera]; + const got = bbox(frame, test); + const err = boxError(got, want, frame); + add( + `${sec.toFixed(1)} s ${label}`, + err <= TOLERANCE_PX, + `${fmtBox(got)} err ${err.toFixed(2)} px`, + fmtRect(want, frame), + ); + }; + atRect(2, 0, isRed, "red PiP (screen-pip slot 0)"); + atRect(2, 1, isGreen, "green PiP (screen-pip slot 1)"); + atRect(4, 1, isGreen, "green fills the frame (camera-full-pip)"); + atRect(4, 0, isRed, "red PiP over it"); + + // The glide across 3.0 s: green present in every frame, on the planned rect, and growing. + let worst = 0; + let missing = 0; + let shrinks = 0; + let lastArea = 0; + const trace = []; + for (const n of glide) { + const frame = frames.get(n); + const got = bbox(frame, isGreen); + if (!got || got.count < 1000) missing++; + const want = plannedRects(t(n), frame, fallback)[1]; + worst = Math.max(worst, boxError(got, want, frame)); + const area = got ? (got.box[2] - got.box[0]) * (got.box[3] - got.box[1]) : 0; + if (area + 4 * frame.width < lastArea) shrinks++; + lastArea = area; + trace.push(`${t(n).toFixed(3)}:${got ? got.box[0] : "-"}`); + } + add( + `2.9-3.6 s green present in all ${glide.length} frames`, + missing === 0, + `${missing} without green`, + "0", + ); + add( + "2.9-3.6 s green on the planned glide rect", + worst <= TOLERANCE_PX, + `max err ${worst.toFixed(2)} px`, + `<= ${TOLERANCE_PX} px`, + ); + add("2.9-3.6 s green box never shrinks", shrinks === 0, `${shrinks} shrinking steps`, "0"); + const firstMoving = glide.find((n) => t(n) > 3); + const lastGlide = glide[glide.length - 1]; + const k0 = plannedRects(t(firstMoving), frames.get(firstMoving), fallback)[1]; + add( + "2.9-3.6 s glide starts at the 3.0 s boundary", + boxError( + bbox(frames.get(frameAt(2.9)), isGreen), + pipRect(1, frames.get(frameAt(2.9)), fallback), + frames.get(frameAt(2.9)), + ) <= TOLERANCE_PX && k0[0] < pipRect(1, frames.get(firstMoving), fallback)[0], + `left edge ${trace.slice(0, 6).join(" ")} ... ${trace[trace.length - 1]}`, + `PiP until 3.0, then x -> 0 (by ${t(lastGlide).toFixed(2)} s planned x ${(plannedRects(t(lastGlide), frames.get(lastGlide), fallback)[1][0] * frames.get(lastGlide).width).toFixed(1)})`, + ); + + { + // Green (slot 1) is drawn after red, so it hides whatever of red's still-gliding rect + // overhangs the right half: red's visible right edge is green's left edge. + const frame = frames.get(frameAt(6)); + const planned = plannedRects(t(frameAt(6)), frame, fallback); + const [x, y, w, h] = planned[0]; + const visible = [x, y, Math.min(x + w, planned[1][0]) - x, h]; + const got = bbox(frame, isRed); + const err = boxError(got, visible, frame); + add( + "6.0 s red, left half (lead-in from the default, k = ease(2/3)), under green", + err <= TOLERANCE_PX, + `${fmtBox(got)} err ${err.toFixed(2)} px`, + fmtRect(visible, frame), + ); + const settled = boxError(got, [0, 0, 0.5, 1], frame); + add("6.0 s red vs the settled left half", null, `err ${settled.toFixed(2)} px`, "information"); + } + atRect(6, 1, isGreen, "green, right half (fading in)"); + atRect(6.5, 0, isRed, "red, settled left half"); + atRect(6.5, 1, isGreen, "green, settled right half"); + + const board = boardCheck(frames.get(frameAt(7.5))); + add( + "7.5 s checkerboard edges equally spaced", + board.spacingError <= TOLERANCE_PX, + `${board.lines} lines, edges ${board.found.join(" ")}, max err ${board.spacingError.toFixed(2)} px`, + `<= ${TOLERANCE_PX} px`, + ); + add( + "7.5 s checkerboard squares square (step across vs down)", + Math.abs(board.stepH - board.stepV) <= TOLERANCE_PX, + `${board.stepH.toFixed(2)} x ${board.stepV.toFixed(2)} px`, + `equal within ${TOLERANCE_PX} px`, + ); + add( + "7.5 s checkerboard on the cover-fitted square grid", + board.gridError <= TOLERANCE_PX, + `max err ${board.gridError.toFixed(2)} px`, + `squares of ${board.size.toFixed(1)} px, centred`, + ); + + report(rows); +} + +const [command, a, b] = process.argv.slice(2); +if (command === "generate" && a) generate(path.resolve(a)); +else if (command === "export" && a && b) exportProject(path.resolve(a), path.resolve(b)); +else if (command === "check" && a && (b ?? "multicam") === "multicam") check(path.resolve(a)); +else if (command === "check" && a && b === "persp") checkPersp(path.resolve(a)); +else if (command === "check" && a && b === "seam") checkSeam(path.resolve(a)); +else { + console.error( + "usage: multicam-fixture.mjs generate | export | check [multicam|persp|seam]", + ); + process.exitCode = 2; +} diff --git a/scripts/test-windows-wgc-helper.mjs b/scripts/test-windows-wgc-helper.mjs index 0ff318bc0..62e1d9fde 100644 --- a/scripts/test-windows-wgc-helper.mjs +++ b/scripts/test-windows-wgc-helper.mjs @@ -39,6 +39,15 @@ const WITH_WINDOW_POPUP = process.argv.includes("--window-popup"); const WITH_WEBCAM = process.env.OPENSCREEN_WGC_TEST_WEBCAM === "true" || process.argv.includes("--webcam"); +/** + * Adds a second camera that does not exist to a `--webcam` run, listed through + * the `webcams` config next to the real one: the helper must drop only that + * camera, say so with its index, and still record the first. + */ +const WITH_MISSING_SECOND_WEBCAM = + process.env.OPENSCREEN_WGC_TEST_MISSING_SECOND_WEBCAM === "true" || + process.argv.includes("--missing-second-webcam"); +const MISSING_WEBCAM_NAME = "OpenScreen Nonexistent Camera"; const CAPTURE_CURSOR = process.env.OPENSCREEN_WGC_TEST_CAPTURE_CURSOR === "true" || process.argv.includes("--capture-cursor"); @@ -107,6 +116,9 @@ const STOP_LATENCY_BUDGET_MS = 15_000; if (WITH_SOFTWARE_ENCODER && WITH_SOFTWARE_FALLBACK) { throw new Error("--software-encoder and --software-fallback are mutually exclusive"); } +if (WITH_MISSING_SECOND_WEBCAM && !WITH_WEBCAM) { + throw new Error("--missing-second-webcam needs --webcam"); +} function runHelper( config, @@ -860,6 +872,9 @@ const outputPath = path.join( `openscreen-wgc-helper-${WITH_WEBCAM ? "webcam" : WITH_WINDOW ? "window" : WITH_SYSTEM_AUDIO || WITH_MICROPHONE ? "audio" : "video"}-${process.pid}-${Date.now()}-${randomUUID()}.mp4`, ); const webcamOutputPath = WITH_WEBCAM ? outputPath.replace(/\.mp4$/i, "-webcam.mp4") : null; +const missingWebcamOutputPath = WITH_MISSING_SECOND_WEBCAM + ? outputPath.replace(/\.mp4$/i, "-webcam-2.mp4") + : null; const fixtureWindow = WITH_WINDOW ? await startFixtureWindow() : null; @@ -904,6 +919,29 @@ const config = { }, }; +if (WITH_MISSING_SECOND_WEBCAM) { + config.webcams = [ + { + camDeviceId: config.webcamDeviceId, + camDeviceName: config.webcamDeviceName, + camClsid: config.webcamDirectShowClsid, + camWidth: config.webcamWidth, + camHeight: config.webcamHeight, + camFps: config.webcamFps, + camPath: webcamOutputPath, + }, + { + camDeviceId: "", + camDeviceName: MISSING_WEBCAM_NAME, + camClsid: "", + camWidth: 1280, + camHeight: 720, + camFps: 30, + camPath: missingWebcamOutputPath, + }, + ]; +} + if (WITH_WINDOW_POPUP) { const scriptPath = path.join(os.tmpdir(), `openscreen-popup-fixture-${process.pid}.ps1`); fs.writeFileSync(scriptPath, POPUP_FIXTURE_SCRIPT); @@ -1111,6 +1149,35 @@ if ( if (WITH_WEBCAM && !webcamStreams.some((stream) => stream.codec_type === "video")) { throw new Error(`WGC helper webcam output has no video stream: ${webcamOutputPath}`); } +if (WITH_MISSING_SECOND_WEBCAM) { + const unavailable = result.stdout + .split(/\r?\n/) + .filter((line) => line.includes('"code":"webcam-unavailable"')) + .map((line) => JSON.parse(line.slice(line.indexOf('{"event"')))); + if (!unavailable.some((event) => event.index === 1 && event.deviceName === MISSING_WEBCAM_NAME)) { + throw new Error( + `WGC helper did not report camera 1 (${MISSING_WEBCAM_NAME}) as unavailable: ${result.stdout}`, + ); + } + const stoppedLine = result.stdout + .split(/\r?\n/) + .find((line) => line.includes('"event":"recording-stopped"')); + const stopped = stoppedLine + ? JSON.parse(stoppedLine.slice(stoppedLine.indexOf('{"event"'))) + : null; + if (JSON.stringify(stopped?.webcamPaths) !== JSON.stringify([webcamOutputPath])) { + throw new Error( + `recording-stopped.webcamPaths should list only the real camera: ${stoppedLine ?? "missing"}`, + ); + } + if (fs.existsSync(missingWebcamOutputPath) && fs.statSync(missingWebcamOutputPath).size > 0) { + throw new Error(`WGC helper wrote a file for the missing camera: ${missingWebcamOutputPath}`); + } + console.log("WGC helper missing-second-webcam check passed", { + unavailable, + webcamPaths: stopped.webcamPaths, + }); +} if ( (CAPTURE_CURSOR && !cursorCapture) || (cursorCapture && diff --git a/src/cli/CliExportRunner.clipList.test.ts b/src/cli/CliExportRunner.clipList.test.ts new file mode 100644 index 000000000..39696c400 --- /dev/null +++ b/src/cli/CliExportRunner.clipList.test.ts @@ -0,0 +1,71 @@ +import { describe, expect, it } from "vitest"; +import { type AxcutAsset, type AxcutDocument, axcutSchemaVersion } from "@/lib/ai-edition/schema"; +import { buildNativeClipList } from "./CliExportRunner"; + +function doc(asset: AxcutAsset): AxcutDocument { + return { + schemaVersion: axcutSchemaVersion, + project: { + id: "proj_1", + title: "Cli", + createdAt: "2026-10-05T10:00:00Z", + updatedAt: "2026-10-05T10:00:00Z", + primaryAssetId: asset.id, + }, + assets: [asset], + transcript: null, + transcripts: [], + timeline: { + clips: [ + { + id: "c1", + assetId: asset.id, + sourceStartSec: 0, + sourceEndSec: 10, + timelineStartSec: 0, + timelineEndSec: 10, + wordRefs: [], + origin: "user", + reason: "", + }, + ], + gaps: [], + trimRanges: [], + muteRanges: [], + speedRanges: [], + captionRanges: [], + }, + annotations: [], + zoomRanges: [], + audioTracks: [], + legacyEditor: null, + }; +} + +const ASSET: AxcutAsset = { + id: "a1", + kind: "video", + label: "asset", + originalPath: "/tmp/a.mp4", + cameraTrack: null, +}; + +describe("CLI export clip list", () => { + it("carries the asset's extra cameras", () => { + const clips = buildNativeClipList( + doc({ + ...ASSET, + additionalCameraTracks: [ + { sourcePath: "/tmp/cam2.mp4", startMs: 1000, offsetMs: -250, visible: true, label: "" }, + ], + }), + ); + expect(clips).toHaveLength(1); + expect(clips[0].additionalCameras).toEqual([{ path: "/tmp/cam2.mp4", offsetSec: 0.75 }]); + }); + + it("sends no extra cameras for a one-camera asset", () => { + const clips = buildNativeClipList(doc(ASSET)); + expect(clips[0]).not.toHaveProperty("additionalCameras"); + }); +}); diff --git a/src/cli/CliExportRunner.tsx b/src/cli/CliExportRunner.tsx index de080359e..52f28ff45 100644 --- a/src/cli/CliExportRunner.tsx +++ b/src/cli/CliExportRunner.tsx @@ -26,7 +26,7 @@ import { appendAutoZoomSuggestions, collectAutoZoomSuggestionsForDocument, } from "@/lib/ai-edition/timeline/apply-auto-zooms"; -import { assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; import { resolveClipSourceEndSec } from "@/lib/ai-edition/timeline/clipDuration"; import type { CliDoneResult, CliExportRequest } from "@/lib/cliContracts"; import { GIF_SIZE_PRESETS, type GifSizePreset } from "@/lib/exporter"; @@ -79,7 +79,7 @@ function replaceExtension(filePath: string, newExtension: string): string { /** Mirrors ExportDialog.buildNativeClipList: trim-narrowed visible clips mapped * onto the native multiclip contract. Kept in lock-step with * buildSceneDescription so export and scene agree on the clip stream. */ -function buildNativeClipList(axcutDocument: AxcutDocument): CompositorClipInput[] { +export function buildNativeClipList(axcutDocument: AxcutDocument): CompositorClipInput[] { const assetById = new Map(axcutDocument.assets.map((asset) => [asset.id, asset])); return resolveVisibleClips(axcutDocument).flatMap((clip) => { const asset = assetById.get(clip.assetId); @@ -87,6 +87,8 @@ function buildNativeClipList(axcutDocument: AxcutDocument): CompositorClipInput[ return []; } const camera = assetCameraSource(asset); + // Cameras 2-4, only sent when the asset has any (same rule as `buildSceneDescription`). + const additionalCameras = assetAdditionalCameraSources(asset); const sourceEndSec = resolveClipSourceEndSec(clip, asset); return [ { @@ -96,6 +98,7 @@ function buildNativeClipList(axcutDocument: AxcutDocument): CompositorClipInput[ sourceEndSec, webcamOffsetSec: camera.offsetSec, hasAudio: true, + ...(additionalCameras.length > 0 ? { additionalCameras } : {}), }, ]; }); diff --git a/src/components/ai-edition/CameraCalibrationModal.test.tsx b/src/components/ai-edition/CameraCalibrationModal.test.tsx new file mode 100644 index 000000000..0e68013dc --- /dev/null +++ b/src/components/ai-edition/CameraCalibrationModal.test.tsx @@ -0,0 +1,330 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { fireEvent, render, screen, waitFor } from "@testing-library/react"; +import { afterEach, beforeEach, describe, expect, it, vi } from "vitest"; +import type { CameraSettings } from "@/components/video-editor/types"; + +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: (scope: string) => (key: string) => `${scope}.${key}`, +})); + +// A small still: 50 x 40 RGBA. jsdom has neither `ImageData` nor a video decoder. +const IMAGE = { width: 50, height: 40, data: new Uint8ClampedArray(50 * 40 * 4) } as ImageData; +vi.mock("@/lib/ai-edition/timeline/grabFrame", () => ({ + grabFrame: vi.fn(async () => IMAGE), +})); + +const detectCornerMarkers = vi.fn(); +vi.mock("@/lib/arucoMarkers", () => ({ + detectCornerMarkers: (...args: unknown[]) => detectCornerMarkers(...args), +})); +const printMarkerSheet = vi.fn(); +vi.mock("@/lib/markerSheet", () => ({ + printMarkerSheet: (...args: unknown[]) => printMarkerSheet(...args), +})); + +import { CameraCalibrationModal } from "./CameraCalibrationModal"; + +const CAMERA = { index: 1, label: "Desk", src: "file:///cam.mp4", timeSec: 2 }; + +function renderModal( + mode: "perspective" | "crop", + initial: CameraSettings | null, + onApply = vi.fn(), + onClose = vi.fn(), +) { + render( + , + ); + return { onApply, onClose }; +} + +const apply = () => screen.getByRole("button", { name: "dialogs.cameraCalibration.apply" }); +const stillLoaded = () => + waitFor(() => expect(screen.queryByText("dialogs.cameraCalibration.loading")).toBeNull()); + +describe("CameraCalibrationModal", () => { + beforeEach(() => { + // jsdom has no 2D canvas: `getContext` would log "not implemented" and return null. The + // dialog draws nothing without a context, which is what these tests assume. + vi.spyOn(HTMLCanvasElement.prototype, "getContext").mockReturnValue(null); + }); + afterEach(() => { + vi.restoreAllMocks(); + }); + + it("apply is disabled for a crossed quad", async () => { + renderModal("perspective", { + perspective: { + corners: [ + { x: 0.1, y: 0.1 }, + { x: 0.9, y: 0.9 }, + { x: 0.9, y: 0.1 }, + { x: 0.1, y: 0.9 }, + ], + aspect: 1, + }, + }); + await stillLoaded(); + expect(apply()).toBeDisabled(); + expect(screen.getByRole("alert")).toHaveTextContent("dialogs.cameraCalibration.invalidQuad"); + }); + + it("apply stores corners, aspect and margin in one call", async () => { + const { onApply, onClose } = renderModal("perspective", null); + await stillLoaded(); + fireEvent.click(screen.getByRole("button", { name: "1:1" })); + fireEvent.change(screen.getByRole("slider"), { target: { value: "10" } }); + fireEvent.click(apply()); + expect(onApply).toHaveBeenCalledTimes(1); + expect(onApply).toHaveBeenCalledWith({ + perspective: { + corners: [ + { x: 0.15, y: 0.15 }, + { x: 0.85, y: 0.15 }, + { x: 0.85, y: 0.85 }, + { x: 0.15, y: 0.85 }, + ], + aspect: 1, + margin: 0.1, + }, + }); + expect(onClose).toHaveBeenCalled(); + }); + + it("arrow keys move the focused handle", async () => { + const { onApply } = renderModal("perspective", null); + await stillLoaded(); + const handle = screen.getByTestId("calibration-handle-0"); + handle.focus(); + // One image pixel right, then ten down. + fireEvent.keyDown(handle, { key: "ArrowRight" }); + fireEvent.keyDown(handle, { key: "ArrowDown", shiftKey: true }); + fireEvent.click(apply()); + const corners = onApply.mock.calls[0][0].perspective.corners; + expect(corners[0].x).toBeCloseTo(0.15 + 1 / 50, 9); + expect(corners[0].y).toBeCloseTo(0.15 + 10 / 40, 9); + expect(corners[1]).toEqual({ x: 0.85, y: 0.15 }); + // The A4 landscape default. + expect(onApply.mock.calls[0][0].perspective.aspect).toBeCloseTo(297 / 210, 9); + }); + + it("crop mode stores a crop", async () => { + const { onApply } = renderModal("crop", null); + await stillLoaded(); + const crop = screen.getByTestId("calibration-crop"); + // Shrink from the right edge, then move right. + for (let i = 0; i < 20; i++) fireEvent.keyDown(crop, { key: "ArrowLeft", shiftKey: true }); + for (let i = 0; i < 5; i++) fireEvent.keyDown(crop, { key: "ArrowRight" }); + fireEvent.click(apply()); + expect(onApply).toHaveBeenCalledTimes(1); + const stored = onApply.mock.calls[0][0].crop; + expect(stored.x).toBeCloseTo(0.05, 9); + expect(stored.y).toBe(0); + expect(stored.width).toBeCloseTo(0.8, 9); + expect(stored.height).toBe(1); + }); + + it("cancel stores nothing", async () => { + const { onApply, onClose } = renderModal("perspective", null); + await stillLoaded(); + fireEvent.keyDown(screen.getByTestId("calibration-handle-2"), { key: "ArrowLeft" }); + fireEvent.click(screen.getByRole("button", { name: "common.actions.cancel" })); + expect(onClose).toHaveBeenCalled(); + expect(onApply).not.toHaveBeenCalled(); + }); + + it("an invalid free ratio says why and marks the field", async () => { + renderModal("perspective", null); + await stillLoaded(); + fireEvent.click(screen.getByRole("button", { name: "dialogs.cameraCalibration.formats.free" })); + const input = screen.getByRole("spinbutton"); + expect(input).toHaveAttribute("aria-invalid", "false"); + fireEvent.change(input, { target: { value: "50" } }); + expect(input).toHaveAttribute("aria-invalid", "true"); + expect(screen.getByRole("alert")).toHaveTextContent("dialogs.cameraCalibration.invalidRatio"); + expect(apply()).toBeDisabled(); + }); + + it("a drag ends on pointercancel and its listeners go with the dialog", async () => { + const add = vi.spyOn(window, "addEventListener"); + const remove = vi.spyOn(window, "removeEventListener"); + const { unmount } = render( + , + ); + await stillLoaded(); + const frame = screen.getByTestId("calibration-frame"); + vi.spyOn(frame, "getBoundingClientRect").mockReturnValue(new DOMRect(0, 0, 100, 80)); + const crop = screen.getByTestId("calibration-crop"); + const dragTypes = ["pointermove", "pointerup", "pointercancel"]; + // The drag's own listeners: added after `from`, still registered unless removed since. + const dragListeners = (from: number) => + add.mock.calls.slice(from).filter(([type]) => dragTypes.includes(type)); + const stillAttached = (from: number) => + dragListeners(from).filter( + ([type, fn]) => !remove.mock.calls.some(([t, f]) => t === type && f === fn), + ).length; + + let from = add.mock.calls.length; + fireEvent.pointerDown(crop, { clientX: 10, clientY: 10 }); + expect(dragListeners(from)).toHaveLength(3); + fireEvent(window, new Event("pointercancel")); + expect(stillAttached(from)).toBe(0); + + // A drag still running when the dialog goes away is ended with it. + from = add.mock.calls.length; + fireEvent.pointerDown(crop, { clientX: 10, clientY: 10 }); + expect(stillAttached(from)).toBe(3); + unmount(); + expect(stillAttached(from)).toBe(0); + }); + + it("detect fills the handles", async () => { + const found = [ + { x: 0.2, y: 0.25 }, + { x: 0.8, y: 0.2 }, + { x: 0.85, y: 0.9 }, + { x: 0.1, y: 0.8 }, + ]; + detectCornerMarkers.mockReturnValueOnce(found); + const { onApply } = renderModal("perspective", null); + await stillLoaded(); + fireEvent.click( + screen.getByRole("button", { name: "dialogs.cameraCalibration.detectMarkers" }), + ); + expect(detectCornerMarkers).toHaveBeenCalledWith(IMAGE); + expect(screen.getByRole("status")).toHaveTextContent("dialogs.cameraCalibration.markersFound"); + fireEvent.click(apply()); + expect(onApply.mock.calls[0][0].perspective.corners).toEqual(found); + }); + + it("detect without four markers shows the message and keeps the handles", async () => { + detectCornerMarkers.mockReturnValueOnce(null); + const { onApply } = renderModal("perspective", null); + await stillLoaded(); + fireEvent.click( + screen.getByRole("button", { name: "dialogs.cameraCalibration.detectMarkers" }), + ); + expect(screen.getByRole("status")).toHaveTextContent( + "dialogs.cameraCalibration.markersNotFound", + ); + fireEvent.click(apply()); + expect(onApply.mock.calls[0][0].perspective.corners).toEqual([ + { x: 0.15, y: 0.15 }, + { x: 0.85, y: 0.15 }, + { x: 0.85, y: 0.85 }, + { x: 0.15, y: 0.85 }, + ]); + }); + + it("the marker message goes away once a handle moves", async () => { + detectCornerMarkers.mockReturnValueOnce(null); + renderModal("perspective", null); + await stillLoaded(); + const detect = () => + fireEvent.click( + screen.getByRole("button", { name: "dialogs.cameraCalibration.detectMarkers" }), + ); + detect(); + expect(screen.getByRole("status")).toHaveTextContent( + "dialogs.cameraCalibration.markersNotFound", + ); + fireEvent.keyDown(screen.getByTestId("calibration-handle-1"), { key: "ArrowLeft" }); + expect(screen.getByRole("status")).toBeEmptyDOMElement(); + + // A drag that starts clears it too, before the pointer even moves. + detectCornerMarkers.mockReturnValueOnce(null); + detect(); + expect(screen.getByRole("status")).toHaveTextContent( + "dialogs.cameraCalibration.markersNotFound", + ); + const frame = screen.getByTestId("calibration-frame"); + vi.spyOn(frame, "getBoundingClientRect").mockReturnValue(new DOMRect(0, 0, 100, 80)); + // Corner 0 sits at 15 % / 15 % of the frame. + fireEvent.pointerDown(frame, { clientX: 15, clientY: 12 }); + expect(screen.getByRole("status")).toBeEmptyDOMElement(); + fireEvent(window, new Event("pointerup")); + }); + + it("print hands the translated texts to the sheet", async () => { + renderModal("perspective", null); + await stillLoaded(); + fireEvent.click( + screen.getByRole("button", { name: "dialogs.cameraCalibration.printMarkerSheet" }), + ); + expect(printMarkerSheet).toHaveBeenCalledWith({ + instruction: "dialogs.cameraCalibration.markerSheetInstruction", + labels: [ + "0 – dialogs.cameraCalibration.corners.topLeft", + "1 – dialogs.cameraCalibration.corners.topRight", + "2 – dialogs.cameraCalibration.corners.bottomRight", + "3 – dialogs.cameraCalibration.corners.bottomLeft", + ], + }); + }); + + it("reset removes the stored perspective", async () => { + const { onApply } = renderModal("perspective", { + perspective: { + corners: [ + { x: 0.1, y: 0.1 }, + { x: 0.9, y: 0.1 }, + { x: 0.9, y: 0.9 }, + { x: 0.1, y: 0.9 }, + ], + aspect: 2, + }, + }); + await stillLoaded(); + fireEvent.click(screen.getByRole("button", { name: "dialogs.cameraCalibration.reset" })); + expect(onApply).toHaveBeenCalledWith({ perspective: undefined }); + }); + + it("perspective mode says a stored crop is ignored", async () => { + renderModal("perspective", { crop: { x: 0.1, y: 0.1, width: 0.5, height: 0.5 } }); + await stillLoaded(); + expect(screen.getByText("dialogs.cameraCalibration.cropIgnored")).toBeInTheDocument(); + }); + + it("says nothing about the crop when there is none", async () => { + renderModal("perspective", null); + await stillLoaded(); + expect(screen.queryByText("dialogs.cameraCalibration.cropIgnored")).toBeNull(); + }); + + it("crop mode does not show the perspective note", async () => { + renderModal("crop", { crop: { x: 0.1, y: 0.1, width: 0.5, height: 0.5 } }); + await stillLoaded(); + expect(screen.queryByText("dialogs.cameraCalibration.cropIgnored")).toBeNull(); + }); + + it("the caller can report camera 1's crop, which lives outside the settings", async () => { + render( + , + ); + await stillLoaded(); + expect(screen.getByText("dialogs.cameraCalibration.cropIgnored")).toBeInTheDocument(); + }); +}); diff --git a/src/components/ai-edition/CameraCalibrationModal.tsx b/src/components/ai-edition/CameraCalibrationModal.tsx new file mode 100644 index 000000000..d0e0438db --- /dev/null +++ b/src/components/ai-edition/CameraCalibrationModal.tsx @@ -0,0 +1,711 @@ +// The camera calibration dialog. Perspective mode: four corner handles on a still of the camera, +// dragged with the pointer or nudged with the arrow keys, a loupe for the corner being placed, +// the target format and margin, and a live preview of the corrected picture. "Detect markers" +// places the corners on the four printed markers (`arucoMarkers.ts`); "Print marker sheet" +// prints them (`markerSheet.ts`). Crop mode: a rectangle with corner and edge handles. Built on +// `ModalShell`, so the editor's shortcuts and undo stay blocked while it is open; Apply hands +// back one settings patch (one undo step). + +import { + type CSSProperties, + type KeyboardEvent as ReactKeyboardEvent, + type PointerEvent as ReactPointerEvent, + useEffect, + useMemo, + useRef, + useState, +} from "react"; +import type { + CameraPerspective, + CameraPoint, + CameraSettings, + CropRegion, +} from "@/components/video-editor/types"; +import { useScopedT } from "@/contexts/I18nContext"; +import { + type Corners, + type CropEdges, + clampCrop, + clampPoint, + insetCorners, + isFullCrop, + isValidQuad, + loupeSourceRect, + moveCrop, + nearestHandle, + previewSize, + renderRectified, + resizeCrop, +} from "@/lib/ai-edition/timeline/calibrationGeometry"; +import { grabFrame } from "@/lib/ai-edition/timeline/grabFrame"; +import { detectCornerMarkers } from "@/lib/arucoMarkers"; +import { printMarkerSheet } from "@/lib/markerSheet"; +import type { CalibrationMode } from "./CamerasSection"; +import { previewBoxStyle } from "./cropDraft"; +import { ModalShell } from "./Modals"; +import styles from "./NewEditorShell.module.css"; +import { ChoiceRow } from "./RightPanes"; + +/** The camera being calibrated and the still it is calibrated on. */ +export interface CalibrationCamera { + /** 0 = camera 1. */ + index: number; + label: string; + /** Video URL of the camera's file. */ + src: string; + /** Time of that file shown by the playhead. */ + timeSec: number; +} + +export interface CameraCalibrationModalProps { + open: boolean; + camera: CalibrationCamera; + mode: CalibrationMode; + /** The camera's stored settings; the dialog starts from them. */ + initial: CameraSettings | null; + /** The camera has a crop, which a perspective replaces. Camera 1 keeps its crop outside + * `initial`, so the caller says; defaults to `initial.crop`. */ + hasCrop?: boolean; + /** The change to store (a key set to `undefined` removes it); called at most once. */ + onApply: (patch: Partial) => void; + onClose: () => void; +} + +type FormatId = "a4Portrait" | "a4Landscape" | "wide" | "standard" | "square" | "free"; + +const FORMATS: ReadonlyArray<{ id: FormatId; aspect: number | null }> = [ + { id: "a4Portrait", aspect: 210 / 297 }, + { id: "a4Landscape", aspect: 297 / 210 }, + { id: "wide", aspect: 16 / 9 }, + { id: "standard", aspect: 4 / 3 }, + { id: "square", aspect: 1 }, + { id: "free", aspect: null }, +]; + +const MIN_ASPECT = 0.1; +const MAX_ASPECT = 10; +const MAX_MARGIN_PCT = 20; +/** How far from a corner (screen px) a press still grabs it. */ +const HANDLE_RADIUS_PX = 24; +const LOUPE_SIZE_PX = 120; +const LOUPE_ZOOM = 4; +const PREVIEW_LONG_SIDE_PX = 320; +/** One arrow press in crop mode, as a fraction of the image. */ +const CROP_STEP = 0.01; + +const CORNER_KEYS = ["topLeft", "topRight", "bottomRight", "bottomLeft"] as const; +const CROP_CORNERS = ["nw", "ne", "sw", "se"] as const; +const CROP_EDGES = ["n", "s", "w", "e"] as const; +const FULL_CROP: CropRegion = { x: 0, y: 0, width: 1, height: 1 }; + +function formatOf(aspect: number): FormatId { + const match = FORMATS.find((f) => f.aspect !== null && Math.abs(f.aspect - aspect) < 1e-3); + return match?.id ?? "free"; +} + +function parseAspect(text: string): number | null { + const value = Number(text.replace(",", ".")); + return Number.isFinite(value) && value >= MIN_ASPECT && value <= MAX_ASPECT ? value : null; +} + +function copyCorners(c: Corners): Corners { + return [{ ...c[0] }, { ...c[1] }, { ...c[2] }, { ...c[3] }]; +} + +/** + * A pointer drag on `window` until release or cancel; `onMove` gets the offset from the press. + * Returns a disposer that removes the listeners early (the dialog closing mid-drag). + */ +function trackDrag(e: ReactPointerEvent, onMove: (dxPx: number, dyPx: number) => void) { + const startX = e.clientX; + const startY = e.clientY; + const move = (ev: PointerEvent) => onMove(ev.clientX - startX, ev.clientY - startY); + const stop = () => { + window.removeEventListener("pointermove", move); + window.removeEventListener("pointerup", stop); + window.removeEventListener("pointercancel", stop); + }; + window.addEventListener("pointermove", move); + window.addEventListener("pointerup", stop); + window.addEventListener("pointercancel", stop); + return stop; +} + +export function CameraCalibrationModal({ + open, + camera, + mode, + initial, + hasCrop = initial?.crop != null, + onApply, + onClose, +}: CameraCalibrationModalProps) { + const t = useScopedT("dialogs"); + const tc = useScopedT("common"); + // The dialog is mounted per opening, so the stored settings seed the draft once. + const stored = initial?.perspective; + const [corners, setCorners] = useState(() => + stored ? copyCorners(stored.corners) : insetCorners(), + ); + const [format, setFormat] = useState(() => + stored ? formatOf(stored.aspect) : "a4Landscape", + ); + const [freeAspect, setFreeAspect] = useState(() => + stored ? String(Math.round(stored.aspect * 1000) / 1000) : "1.5", + ); + const [marginPct, setMarginPct] = useState(() => Math.round((stored?.margin ?? 0) * 100)); + const [crop, setCrop] = useState(() => initial?.crop ?? FULL_CROP); + const [image, setImage] = useState(null); + const [loadFailed, setLoadFailed] = useState(false); + const [activeHandle, setActiveHandle] = useState(null); + const [markerResult, setMarkerResult] = useState<"found" | "notFound" | null>(null); + + const frameRef = useRef(null); + const stillRef = useRef(null); + const loupeRef = useRef(null); + const previewRef = useRef(null); + const handleRefs = useRef>([]); + // The running drag's disposer: a new drag or unmounting ends the previous one. + const stopDragRef = useRef<(() => void) | null>(null); + + useEffect( + () => () => { + stopDragRef.current?.(); + stopDragRef.current = null; + }, + [], + ); + + // The marker message describes the corners as detected; any manual change makes it stale. + const startDrag = (e: ReactPointerEvent, onMove: (dxPx: number, dyPx: number) => void) => { + setMarkerResult(null); + stopDragRef.current?.(); + stopDragRef.current = trackDrag(e, onMove); + }; + + // The still, at full resolution. + useEffect(() => { + let cancelled = false; + setImage(null); + setLoadFailed(false); + grabFrame(camera.src, camera.timeSec) + .then((data) => { + if (!cancelled) setImage(data); + }) + .catch(() => { + if (!cancelled) setLoadFailed(true); + }); + return () => { + cancelled = true; + }; + }, [camera.src, camera.timeSec]); + + // Paint the still. No 2D context (jsdom) leaves the canvas blank; the handles still work. + useEffect(() => { + const canvas = stillRef.current; + if (!canvas || !image) return; + canvas.width = image.width; + canvas.height = image.height; + canvas.getContext("2d")?.putImageData(image, 0, 0); + }, [image]); + + const aspect = + format === "free" + ? parseAspect(freeAspect) + : (FORMATS.find((f) => f.id === format)?.aspect ?? null); + const margin = marginPct / 100; + const quadValid = isValidQuad(corners); + // Memoized: the preview resamples whenever this object changes, not on every render. + const perspective = useMemo( + () => + quadValid && aspect !== null ? { corners, aspect, ...(margin > 0 ? { margin } : {}) } : null, + [corners, aspect, margin, quadValid], + ); + const preview = previewSize(aspect ?? 1, PREVIEW_LONG_SIDE_PX); + + // The corrected picture, resampled on every change of the corners, format or margin. + useEffect(() => { + if (mode !== "perspective") return; + const canvas = previewRef.current; + const ctx = canvas?.getContext("2d"); + if (!canvas || !ctx) return; + ctx.clearRect(0, 0, canvas.width, canvas.height); + if (!image || !perspective) return; + const pixels = renderRectified(image, perspective, canvas.width, canvas.height); + if (pixels) ctx.putImageData(new ImageData(pixels, canvas.width, canvas.height), 0, 0); + }, [mode, image, perspective]); + + // The loupe: the area under the active corner, magnified, crisp pixels and a crosshair. + useEffect(() => { + const canvas = loupeRef.current; + const source = stillRef.current; + const ctx = canvas?.getContext("2d"); + if (!canvas || !source || !ctx || !image || activeHandle === null) return; + const c = corners[activeHandle]; + const center = { x: c.x * image.width, y: c.y * image.height }; + const r = loupeSourceRect(center, image.width, image.height, LOUPE_SIZE_PX, LOUPE_ZOOM); + ctx.imageSmoothingEnabled = false; + ctx.clearRect(0, 0, LOUPE_SIZE_PX, LOUPE_SIZE_PX); + ctx.drawImage(source, r.x, r.y, r.width, r.height, 0, 0, LOUPE_SIZE_PX, LOUPE_SIZE_PX); + const cx = ((center.x - r.x) / r.width) * LOUPE_SIZE_PX; + const cy = ((center.y - r.y) / r.height) * LOUPE_SIZE_PX; + ctx.strokeStyle = "rgb(255 80 80)"; + ctx.lineWidth = 1; + ctx.beginPath(); + ctx.moveTo(cx, 0); + ctx.lineTo(cx, LOUPE_SIZE_PX); + ctx.moveTo(0, cy); + ctx.lineTo(LOUPE_SIZE_PX, cy); + ctx.stroke(); + }, [image, corners, activeHandle]); + + const setCorner = (index: number, p: CameraPoint) => { + setMarkerResult(null); + setCorners((prev) => { + const next = copyCorners(prev); + next[index] = clampPoint(p); + return next; + }); + }; + + // A press anywhere near a corner grabs it, so a corner is found without aiming at its dot. + const onFramePointerDown = (e: ReactPointerEvent) => { + const frame = frameRef.current; + if (!frame) return; + const r = frame.getBoundingClientRect(); + if (r.width <= 0 || r.height <= 0) return; + const points = corners.map((c) => ({ x: c.x * r.width, y: c.y * r.height })); + const hit = nearestHandle( + points, + { x: e.clientX - r.left, y: e.clientY - r.top }, + HANDLE_RADIUS_PX, + ); + if (hit === null) return; + e.preventDefault(); + handleRefs.current[hit]?.focus(); + setActiveHandle(hit); + const start = corners[hit]; + startDrag(e, (dx, dy) => + setCorner(hit, { x: start.x + dx / r.width, y: start.y + dy / r.height }), + ); + }; + + // The arrows move the focused corner by one image pixel, ten with Shift. + const onHandleKeyDown = (index: number) => (e: ReactKeyboardEvent) => { + const dx = e.key === "ArrowLeft" ? -1 : e.key === "ArrowRight" ? 1 : 0; + const dy = e.key === "ArrowUp" ? -1 : e.key === "ArrowDown" ? 1 : 0; + if (dx === 0 && dy === 0) return; + // The editor shell seeks on the arrows, from window: keep them here. + e.preventDefault(); + e.nativeEvent.stopPropagation(); + const step = e.shiftKey ? 10 : 1; + const c = corners[index]; + setCorner(index, { + x: c.x + (dx * step) / (image?.width ?? 1000), + y: c.y + (dy * step) / (image?.height ?? 1000), + }); + }; + + const startCropMove = (e: ReactPointerEvent) => { + const frame = frameRef.current; + if (!frame) return; + e.preventDefault(); + e.stopPropagation(); + const r = frame.getBoundingClientRect(); + if (r.width <= 0 || r.height <= 0) return; + const start = crop; + startDrag(e, (dx, dy) => setCrop(moveCrop(start, dx / r.width, dy / r.height))); + }; + + const startCropResize = (edges: CropEdges) => (e: ReactPointerEvent) => { + const frame = frameRef.current; + if (!frame) return; + e.preventDefault(); + e.stopPropagation(); + const r = frame.getBoundingClientRect(); + if (r.width <= 0 || r.height <= 0) return; + const start = crop; + startDrag(e, (dx, dy) => setCrop(resizeCrop(start, edges, dx / r.width, dy / r.height))); + }; + + // The arrows move the crop; Shift + the arrows resize it from its bottom-right corner. + const onCropKeyDown = (e: ReactKeyboardEvent) => { + const dx = e.key === "ArrowLeft" ? -1 : e.key === "ArrowRight" ? 1 : 0; + const dy = e.key === "ArrowUp" ? -1 : e.key === "ArrowDown" ? 1 : 0; + if (dx === 0 && dy === 0) return; + e.preventDefault(); + e.nativeEvent.stopPropagation(); + if (e.shiftKey) { + setCrop( + resizeCrop(crop, { right: dx !== 0, bottom: dy !== 0 }, dx * CROP_STEP, dy * CROP_STEP), + ); + return; + } + setCrop(moveCrop(crop, dx * CROP_STEP, dy * CROP_STEP)); + }; + + // The four printed markers place the corners; without all four the corners stay put. + const detectMarkers = () => { + if (!image) return; + const found = detectCornerMarkers(image); + if (found) setCorners(copyCorners(found)); + setMarkerResult(found ? "found" : "notFound"); + }; + + const printSheet = () => { + // Marker ID n belongs on corner n, the handles' order. + const label = (id: number) => `${id} – ${t(`cameraCalibration.corners.${CORNER_KEYS[id]}`)}`; + printMarkerSheet({ + instruction: t("cameraCalibration.markerSheetInstruction"), + labels: [label(0), label(1), label(2), label(3)], + }); + }; + + const isPerspective = mode === "perspective"; + const canApply = isPerspective ? perspective !== null : true; + const hasStored = isPerspective ? Boolean(initial?.perspective) : Boolean(initial?.crop); + + const apply = () => { + if (isPerspective) { + if (!perspective) return; + onApply({ perspective: { ...perspective, corners: copyCorners(perspective.corners) } }); + } else { + const next = clampCrop(crop); + onApply({ crop: isFullCrop(next) ? undefined : next }); + } + onClose(); + }; + + // Reset removes the stored correction (or crop) altogether. + const reset = () => { + setMarkerResult(null); + onApply(isPerspective ? { perspective: undefined } : { crop: undefined }); + onClose(); + }; + + const imageAspect = image ? image.width / image.height : 16 / 9; + const title = isPerspective + ? t("cameraCalibration.titlePerspective", { camera: camera.label }) + : t("cameraCalibration.titleCrop", { camera: camera.label }); + // The loupe sits in the corner away from the corner being placed. + const active = activeHandle === null ? null : corners[activeHandle]; + const loupeOnRight = active !== null && active.x < 0.5 && active.y < 0.5; + + return ( + +

+ {isPerspective ? t("cameraCalibration.perspectiveHelp") : t("cameraCalibration.cropHelp")} +

+ {isPerspective && hasCrop ? ( +

+ {t("cameraCalibration.cropIgnored")} +

+ ) : null} +
+ + {image === null ? ( +

+ {loadFailed ? t("cameraCalibration.loadFailed") : t("cameraCalibration.loading")} +

+ ) : null} + {isPerspective ? ( + <> + + `${c.x},${c.y}`).join(" ")} + fill={quadValid ? "rgb(80 160 255 / 0.15)" : "rgb(255 80 80 / 0.2)"} + stroke={quadValid ? "rgb(255 255 255 / 0.9)" : "rgb(255 80 80)"} + strokeWidth={1.5} + vectorEffect="non-scaling-stroke" + /> + + {corners.map((c, i) => ( +
+ + {isPerspective ? ( +
+ + +

+ {markerResult === "found" + ? t("cameraCalibration.markersFound") + : markerResult === "notFound" + ? t("cameraCalibration.markersNotFound") + : null} +

+
+ ) : null} + + {isPerspective ? ( +
+
+ {t("cameraCalibration.format")} + + label={t("cameraCalibration.format")} + columns={3} + options={[ + { value: "a4Portrait", label: t("cameraCalibration.formats.a4Portrait") }, + { value: "a4Landscape", label: t("cameraCalibration.formats.a4Landscape") }, + { value: "wide", label: "16:9" }, + { value: "standard", label: "4:3" }, + { value: "square", label: "1:1" }, + { value: "free", label: t("cameraCalibration.formats.free") }, + ]} + value={format} + onChange={setFormat} + /> + {format === "free" ? ( + + ) : null} + {format === "free" && aspect === null ? ( + + ) : null} + +
+
+ {t("cameraCalibration.preview")} + +
+
+ ) : null} + + {isPerspective && !quadValid ? ( + + ) : null} + +
+ +
+ + +
+
+
+ ); +} diff --git a/src/components/ai-edition/CamerasSection.test.tsx b/src/components/ai-edition/CamerasSection.test.tsx new file mode 100644 index 000000000..a382aaa5f --- /dev/null +++ b/src/components/ai-edition/CamerasSection.test.tsx @@ -0,0 +1,185 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { fireEvent, render, screen } from "@testing-library/react"; +import { describe, expect, it, vi } from "vitest"; +import type { AxcutDocument } from "@/lib/ai-edition/schema"; + +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: (scope: string) => (key: string, vars?: Record) => + vars?.n !== undefined + ? `${scope}.${key}#${vars.n}${vars.label ? `|${vars.label}` : ""}` + : `${scope}.${key}`, +})); +vi.mock("@/lib/ai-edition/timeline/grabFrame", () => ({ + grabFrameDataUrl: vi.fn(() => Promise.resolve("data:image/png;base64,AAAA")), +})); + +import { readFileSync } from "node:fs"; +import { act } from "@testing-library/react"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import { grabFrameDataUrl } from "@/lib/ai-edition/timeline/grabFrame"; +import { CamerasSection, calibrationCameraAt, cameraHasCrop } from "./CamerasSection"; + +const track = (sourcePath: string, extra: Record = {}) => ({ + sourcePath, + visible: true, + startMs: 0, + offsetMs: 0, + ...extra, +}); + +function makeDoc(extraCameras: number): AxcutDocument { + return { + assets: [ + { + id: "a1", + cameraTrack: track("C:/cam1.mp4"), + additionalCameraTracks: Array.from({ length: extraCameras }, (_, i) => + track(`C:/cam${i + 2}.mp4`, { label: i === 0 ? "Desk" : "" }), + ), + }, + ], + timeline: { + clips: [ + { + id: "c1", + assetId: "a1", + sourceStartSec: 0, + sourceEndSec: 10, + timelineStartSec: 0, + timelineEndSec: 10, + }, + ], + }, + } as unknown as AxcutDocument; +} + +const render2 = (extra: number, setCameraSettings = vi.fn(), onOpenCalibration = vi.fn()) => { + render( + , + ); + return { setCameraSettings, onOpenCalibration }; +}; + +describe("CamerasSection", () => { + it("lists every camera of the clip", () => { + render2(2); + expect(screen.getByTestId("camera-row-0")).toHaveTextContent("settings.cameras.cameraN#1"); + expect(screen.getByTestId("camera-row-1")).toHaveTextContent( + "settings.cameras.cameraNamed#2|Desk", + ); + expect(screen.getByTestId("camera-row-2")).toHaveTextContent("settings.cameras.cameraN#3"); + }); + + it("camera 1 offers only the perspective", () => { + const { onOpenCalibration } = render2(1); + const row = screen.getByTestId("camera-row-0"); + expect(row).toHaveTextContent("settings.cameras.camera1Hint"); + expect(row.querySelectorAll("button")).toHaveLength(1); + fireEvent.click(row.querySelector("button") as Element); + expect(onOpenCalibration).toHaveBeenCalledWith(0, "perspective"); + }); + + it("rotating camera 2 calls setCameraSettings", () => { + const { setCameraSettings, onOpenCalibration } = render2(1); + const row = screen.getByTestId("camera-row-1"); + fireEvent.click( + row.querySelector("button[aria-pressed]:not([aria-pressed='true'])") as Element, + ); + expect(setCameraSettings).toHaveBeenCalledWith(1, { rotation: 180 }); + const buttons = Array.from(row.querySelectorAll("button")); + fireEvent.click(buttons.find((b) => b.textContent === "settings.cameras.crop") as Element); + expect(onOpenCalibration).toHaveBeenCalledWith(1, "crop"); + }); + + it("follows the playhead from the store without a prop", async () => { + vi.useFakeTimers(); + useProjectStore.setState({ currentTimeSec: 1 }); + render( + , + ); + act(() => { + useProjectStore.setState({ currentTimeSec: 4 }); + }); + act(() => { + vi.advanceTimersByTime(300); + }); + expect(vi.mocked(grabFrameDataUrl).mock.calls.at(-1)?.[1]).toBe(4); + vi.useRealTimers(); + }); + + it("keeps the playhead subscription out of LayoutPane", () => { + const source = readFileSync("src/components/ai-edition/RightPanes.tsx", "utf8"); + const start = source.indexOf("export function LayoutPane("); + const end = source.indexOf("/** The tightest the frame gets"); + expect(source.slice(start, end)).not.toContain("currentTimeSec"); + }); + + it("finds the calibration still of a camera at the playhead", () => { + const t = (key: string, vars?: Record) => + vars?.n !== undefined ? `${key}#${vars.n}${vars.label ? `|${vars.label}` : ""}` : key; + const doc = makeDoc(1); + const desk = calibrationCameraAt(doc, 2, 1, t); + expect(desk).toMatchObject({ index: 1, label: "cameras.cameraNamed#2|Desk", timeSec: 2 }); + expect(desk?.src).toMatch(/^file:\/\/.*cam2\.mp4$/); + expect(calibrationCameraAt(doc, 2, 0, t)?.label).toBe("cameras.cameraN#1"); + // No such camera, or no document: nothing to calibrate. + expect(calibrationCameraAt(doc, 2, 3, t)).toBeNull(); + expect(calibrationCameraAt(null, 2, 0, t)).toBeNull(); + }); + + it("a perspective disables the crop with a hint", () => { + const onOpenCalibration = vi.fn(); + render( + , + ); + const row = screen.getByTestId("camera-row-1"); + const crop = screen.getByRole("button", { name: "settings.cameras.crop" }); + expect(crop).toBeDisabled(); + expect(row).toHaveTextContent("settings.cameras.cropOffWithPerspective"); + fireEvent.click(crop); + expect(onOpenCalibration).not.toHaveBeenCalled(); + }); + + it("without a perspective the crop stays available and no hint shows", () => { + render2(1); + expect(screen.getByRole("button", { name: "settings.cameras.crop" })).toBeEnabled(); + expect(screen.queryByText("settings.cameras.cropOffWithPerspective")).toBeNull(); + }); + + it("knows whether a camera has a crop", () => { + const crop = { x: 0.1, y: 0.1, width: 0.5, height: 0.5 }; + expect(cameraHasCrop(null, [null, { crop }], 1)).toBe(true); + expect(cameraHasCrop(null, [null, { mirror: true }], 1)).toBe(false); + const withCamera1Crop = (webcamCropRegion: unknown) => + ({ legacyEditor: { webcamCropRegion } }) as unknown as AxcutDocument; + expect(cameraHasCrop(withCamera1Crop(crop), [], 0)).toBe(true); + expect(cameraHasCrop(withCamera1Crop({ x: 0, y: 0, width: 1, height: 1 }), [], 0)).toBe(false); + expect(cameraHasCrop(makeDoc(0), [], 0)).toBe(false); + }); +}); diff --git a/src/components/ai-edition/CamerasSection.tsx b/src/components/ai-edition/CamerasSection.tsx new file mode 100644 index 000000000..92585ae43 --- /dev/null +++ b/src/components/ai-edition/CamerasSection.tsx @@ -0,0 +1,245 @@ +// The "Cameras" section of the layout pane: every camera of the clip under the playhead, with +// a thumbnail and the per-camera settings. Camera 1 keeps its rotation, mirror and crop in +// the controls above (older fields), so its row only offers the perspective correction. + +import { useEffect, useMemo, useState } from "react"; +import { toFileUrl } from "@/components/video-editor/projectPersistence"; +import type { CameraSettings, CropRegion } from "@/components/video-editor/types"; +import { useScopedT } from "@/contexts/I18nContext"; +import type { AxcutDocument } from "@/lib/ai-edition/schema"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { camerasForClipAt, type ProjectCamera } from "@/lib/ai-edition/timeline/cameraList"; +import { grabFrameDataUrl } from "@/lib/ai-edition/timeline/grabFrame"; +import { locateVirtualPosition } from "@/lib/ai-edition/timeline/virtual-preview"; +import type { CameraRotation } from "@/lib/cameraOrientation"; +import styles from "./NewEditorShell.module.css"; +import { ChoiceRow, Toggle } from "./RightPanes"; + +/** What the calibration dialog edits: the picture's crop, or its perspective. */ +export type CalibrationMode = "crop" | "perspective"; + +const THUMBNAIL_SIDE = 96; +/** The playhead moves every frame during playback; a still is only worth grabbing once it rests. */ +const THUMBNAIL_DEBOUNCE_MS = 250; + +export interface CamerasSectionProps { + document: AxcutDocument | null; + /** Overrides the playhead on the ruler; by default it is read from the project store here, so + * only this section re-renders while the playhead moves. */ + playheadSec?: number; + /** `legacyEditor.cameraSettings`, normalized: index 0 = camera 1. */ + cameraSettings: (CameraSettings | null)[]; + setCameraSettings: (index: number, patch: Partial | null) => void | Promise; + /** Opens the calibration dialog for a camera. */ + onOpenCalibration?: (cameraIndex: number, mode: CalibrationMode) => void; +} + +export type CameraStill = { src: string; timeSec: number }; + +/** Where each camera's file is, and which time of it the playhead shows. */ +export function stillsAt( + document: AxcutDocument | null, + playheadSec: number, +): Map { + const stills = new Map(); + if (!document) return stills; + const position = locateVirtualPosition(document.timeline.clips, playheadSec); + if (!position) return stills; + const asset = document.assets.find((a) => a.id === position.clip.assetId); + const sources = [assetCameraSource(asset), ...assetAdditionalCameraSources(asset)]; + sources.forEach((source, index) => { + if (!source.path) return; + const src = /^(https?|blob|data):/.test(source.path) ? source.path : toFileUrl(source.path); + stills.set(index, { src, timeSec: Math.max(0, position.sourceTimeSec - source.offsetSec) }); + }); + return stills; +} + +/** + * Whether a camera has a crop stored. Camera 1 keeps its crop in `webcamCropRegion` (a + * full-frame rect means none); the others in `cameraSettings[k].crop`. + */ +export function cameraHasCrop( + document: AxcutDocument | null, + cameraSettings: (CameraSettings | null)[], + index: number, +): boolean { + if (index !== 0) return cameraSettings[index]?.crop != null; + const legacy = document?.legacyEditor as Record | null | undefined; + const crop = legacy?.webcamCropRegion as Partial | undefined; + if (!crop) return false; + const coversFrame = (v: unknown) => typeof v !== "number" || v >= 1 - 1e-6; + return !(coversFrame(crop.width) && coversFrame(crop.height)); +} + +/** The camera the calibration dialog opens on: its label and its still at the playhead. */ +export function calibrationCameraAt( + document: AxcutDocument | null, + playheadSec: number, + index: number, + t: (key: string, vars?: Record) => string, +): { index: number; label: string; src: string; timeSec: number } | null { + if (!document) return null; + const camera = camerasForClipAt(document, playheadSec, t).find((c) => c.index === index); + const still = stillsAt(document, playheadSec).get(index); + if (!camera?.available || !still) return null; + return { index, label: camera.label, ...still }; +} + +function CameraThumbnail({ still, label }: { still: CameraStill | undefined; label: string }) { + const [url, setUrl] = useState(null); + const src = still?.src; + const timeSec = still?.timeSec; + useEffect(() => { + if (src === undefined || timeSec === undefined) { + setUrl(null); + return; + } + let cancelled = false; + const timer = setTimeout(() => { + grabFrameDataUrl(src, timeSec, THUMBNAIL_SIDE) + .then((dataUrl) => { + if (!cancelled) setUrl(dataUrl); + }) + .catch(() => { + if (!cancelled) setUrl(null); + }); + }, THUMBNAIL_DEBOUNCE_MS); + return () => { + cancelled = true; + clearTimeout(timer); + }; + }, [src, timeSec]); + return ( +
+ {url ? ( + {label} + ) : null} +
+ ); +} + +const BUTTON = `${styles.btn} ${styles.btnSecondary}`; + +export function CamerasSection({ + document, + playheadSec: playheadOverride, + cameraSettings, + setCameraSettings, + onOpenCalibration, +}: CamerasSectionProps) { + const ts = useScopedT("settings"); + const storePlayheadSec = useProjectStore((s) => s.currentTimeSec); + const playheadSec = playheadOverride ?? storePlayheadSec; + const cameras: ProjectCamera[] = useMemo( + () => (document ? camerasForClipAt(document, playheadSec, ts) : []), + [document, playheadSec, ts], + ); + const stills = useMemo(() => stillsAt(document, playheadSec), [document, playheadSec]); + if (cameras.length === 0) return null; + return ( + <> +
{ts("cameras.title")}
+ {cameras.map((camera) => { + const settings = cameraSettings[camera.index] ?? null; + const isFirst = camera.index === 0; + // A perspective replaces the crop at render, so the crop is not offered then. + const hasPerspective = settings?.perspective != null; + return ( +
+
+ +
+
{camera.label}
+ {camera.available ? null : ( +

{ts("cameras.unavailable")}

+ )} +
+
+ {isFirst ?

{ts("cameras.camera1Hint")}

: null} + {isFirst ? null : ( + <> + + label={ts("cameras.rotation")} + options={[ + { value: 0, label: "0°" }, + { value: 180, label: "180°" }, + ]} + value={settings?.rotation ?? 0} + onChange={(rotation) => void setCameraSettings(camera.index, { rotation })} + /> +
+ {ts("cameras.mirror")} + void setCameraSettings(camera.index, { mirror })} + /> +
+ + )} +
+ {isFirst ? null : ( + + )} + + {!isFirst && settings ? ( + + ) : null} +
+ {!isFirst && hasPerspective ? ( +

{ts("cameras.cropOffWithPerspective")}

+ ) : null} +
+ ); + })} + + ); +} diff --git a/src/components/ai-edition/ExportDialog.params.test.tsx b/src/components/ai-edition/ExportDialog.params.test.tsx index d78ad091c..76cc992c2 100644 --- a/src/components/ai-edition/ExportDialog.params.test.tsx +++ b/src/components/ai-edition/ExportDialog.params.test.tsx @@ -228,3 +228,47 @@ describe("ExportDialog format settings", () => { } }); }); + +describe("ExportDialog clip list", () => { + beforeEach(() => { + window.electronAPI = { + pickExportSavePath: vi.fn(async () => ({ path: "/tmp/out.mp4" })), + onNativeExportProgress: vi.fn(() => noop), + } as unknown as ElectronAPI; + }); + + afterEach(() => { + cleanup(); + vi.clearAllMocks(); + }); + + it("hands the native export the asset's extra cameras", async () => { + const withCameras: AxcutDocument = { + ...DOC, + assets: [ + { + ...DOC.assets[0], + additionalCameraTracks: [ + { sourcePath: "/tmp/cam2.mp4", startMs: 500, offsetMs: 250, visible: true, label: "" }, + { sourcePath: "/tmp/cam3.mp4", startMs: 0, offsetMs: 0, visible: false, label: "" }, + ], + }, + ], + }; + renderDialog(withCameras); + await exportMp4(); + const clips = vi.mocked(exportMultiNative).mock.calls.at(-1)?.[0]; + expect(clips?.[0]?.additionalCameras).toEqual([ + { path: "/tmp/cam2.mp4", offsetSec: 0.75 }, + // A hidden track keeps its slot, so the indices stay aligned with the tracks. + { path: "", offsetSec: 0 }, + ]); + }); + + it("sends no extra cameras for a one-camera asset", async () => { + renderDialog(); + await exportMp4(); + const clips = vi.mocked(exportMultiNative).mock.calls.at(-1)?.[0]; + expect(clips?.[0]).not.toHaveProperty("additionalCameras"); + }); +}); diff --git a/src/components/ai-edition/ExportDialog.tsx b/src/components/ai-edition/ExportDialog.tsx index 47b08b08a..be340bb60 100644 --- a/src/components/ai-edition/ExportDialog.tsx +++ b/src/components/ai-edition/ExportDialog.tsx @@ -19,7 +19,7 @@ import { } from "@/lib/ai-edition/document/outputFormat"; import type { AxcutDocument } from "@/lib/ai-edition/schema"; import { getEditorSettings } from "@/lib/ai-edition/store/editorSettings"; -import { assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; import { resolveClipSourceEndSec } from "@/lib/ai-edition/timeline/clipDuration"; import { type ExportFormat, @@ -117,6 +117,8 @@ function buildNativeClipList(document: AxcutDocument): CompositorClipInput[] { return []; } const camera = assetCameraSource(asset); + // Cameras 2-4, only sent when the asset has any (same rule as `buildSceneDescription`). + const additionalCameras = assetAdditionalCameraSources(asset); // sourceEndSec is optional in the schema (unknown until probed) — fall back through // the single canonical precedence used by every consumer (clip.probe → asset.duration // → timeline-length guess). See `resolveClipSourceEndSec` for the full order. @@ -132,6 +134,7 @@ function buildNativeClipList(document: AxcutDocument): CompositorClipInput[] { sourceEndSec, webcamOffsetSec: camera.offsetSec, hasAudio: true, + ...(additionalCameras.length > 0 ? { additionalCameras } : {}), }, ]; }); diff --git a/src/components/ai-edition/NativeCompositorOverlay.test.tsx b/src/components/ai-edition/NativeCompositorOverlay.test.tsx index ff3522665..f776feab7 100644 --- a/src/components/ai-edition/NativeCompositorOverlay.test.tsx +++ b/src/components/ai-edition/NativeCompositorOverlay.test.tsx @@ -18,20 +18,23 @@ import { publishNativePosition } from "@/native/nativeSync"; const native = vi.hoisted(() => ({ setActiveClip: vi.fn(async () => ({ ok: true })), setNativePlaying: vi.fn(), + setNativeScene: vi.fn(), })); +const i18n = vi.hoisted(() => ({ locale: "en" })); vi.mock("@/native", () => ({ pushAllNativeParams: vi.fn(), setActiveClip: native.setActiveClip, setCurrentNativeViewId: vi.fn(), setNativePlaying: native.setNativePlaying, - setNativeScene: vi.fn(), + setNativeScene: native.setNativeScene, subscribeNativeCompositor: () => () => undefined, useIsCpuCompositor: () => false, useNativeCompositorView: () => ({ viewId: 7, error: null }), })); vi.mock("@/contexts/I18nContext", () => ({ + useI18n: () => ({ locale: i18n.locale }), useScopedT: () => (key: string) => key, })); @@ -154,10 +157,31 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(8, true); - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 10); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 10, []); expect(native.setNativePlaying).not.toHaveBeenCalledWith(false); }); + it("sends the asset's extra cameras with the clip", async () => { + const document = makeDocument(); + document.assets[0] = { + ...document.assets[0], + additionalCameraTracks: [ + { sourcePath: "/cam-2.mp4", startMs: 500, offsetMs: 0, visible: true, label: "" }, + { sourcePath: "/cam-3.mp4", startMs: 0, offsetMs: 0, visible: false, label: "" }, + ], + }; + useProjectStore.setState({ document }); + await mountAtFirstClip(); + publishNativePosition({ clipIndex: 0, sourceTimeSec: 4.9 }, now); + + setPlayhead(8, true); + + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 10, [ + { path: "/cam-2.mp4", offsetSec: 0.5 }, + { path: "", offsetSec: 0 }, + ]); + }); + it("re-anchors a view that stays behind, once the gap holds", async () => { await mountAtFirstClip(); // The view stalled 0.4 s behind the playhead, inside the first clip. @@ -170,7 +194,7 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(2.52, true); expect(native.setActiveClip).toHaveBeenCalledTimes(1); - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 2.52); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 2.52, []); }); // At 16× a frame that takes 30 ms to arrive shows the playhead 0.48 s of programme ago. @@ -225,8 +249,8 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(time, false); } - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 14.4); - expect(native.setActiveClip).toHaveBeenLastCalledWith(7, "/take.mp4", "", 0, 0, 14.4); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 14.4, []); + expect(native.setActiveClip).toHaveBeenLastCalledWith(7, "/take.mp4", "", 0, 0, 14.4, []); }); it("still sends the clip on a change while paused", async () => { @@ -236,7 +260,7 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(6, false); - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 8); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 8, []); }); // An addon that reports no position cannot be read, so it is driven as before. @@ -249,3 +273,43 @@ describe("NativeCompositorOverlay while playing", () => { expect(native.setActiveClip).toHaveBeenCalledTimes(1); }); }); + +// The desk-view label is generated text inside the scene, so the scene is rebuilt when the +// user switches language — otherwise the preview keeps the old label until the next edit. +describe("NativeCompositorOverlay on a language change", () => { + beforeEach(() => { + vi.clearAllMocks(); + i18n.locale = "en"; + useProjectStore.setState({ + projectId: "proj_sync", + document: makeDocument(), + revision: 1, + status: "ready", + error: null, + sourceDurationSec: 12, + currentTimeSec: 1, + playing: false, + dirty: false, + lastSavedAt: new Date(), + }); + }); + + afterEach(() => { + cleanup(); + i18n.locale = "en"; + useProjectStore.getState().clear(); + }); + + it("pushes the scene again when the locale changes", async () => { + const { rerender } = render(); + await act(async () => { + await Promise.resolve(); + }); + native.setNativeScene.mockClear(); + + i18n.locale = "de"; + rerender(); + + expect(native.setNativeScene).toHaveBeenCalledTimes(1); + }); +}); diff --git a/src/components/ai-edition/NativeCompositorOverlay.tsx b/src/components/ai-edition/NativeCompositorOverlay.tsx index d60bbcd20..84a67ce39 100644 --- a/src/components/ai-edition/NativeCompositorOverlay.tsx +++ b/src/components/ai-edition/NativeCompositorOverlay.tsx @@ -1,10 +1,10 @@ import { useEffect, useMemo, useRef, useSyncExternalStore } from "react"; -import { useScopedT } from "@/contexts/I18nContext"; +import { useI18n, useScopedT } from "@/contexts/I18nContext"; import { readSpeedRegions } from "@/lib/ai-edition/document/timeline"; import { noteUiProbeClipSwitch } from "@/lib/ai-edition/perf/uiFrameProbe"; import { getEditorSettings } from "@/lib/ai-edition/store/editorSettings"; import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; -import { assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; import { findActiveSpeedRegion, type SpeedRegion } from "@/lib/ai-edition/timeline/speed"; import { resolveNativePosition } from "@/lib/ai-edition/timeline/timelineMap"; import { @@ -129,6 +129,8 @@ export function NativeCompositorOverlay() { sources: sources ?? undefined, }); const t = useScopedT("editor"); + // The scene carries translated text (the desk-view label): a language switch rebuilds it. + const { locale } = useI18n(); // No usable GPU: the preview still renders every effect, just slowly (~8 fps with // everything on). Nothing is disabled — the output stays identical to the GPU path — // so this is a notice, not a degradation warning. @@ -147,7 +149,7 @@ export function NativeCompositorOverlay() { // `_webcamSizeRevision` ci-dessus) : le layout preset et cie pilotent le rendu (remplace le // layout fixture). Effet APRÈS celui du viewId ci-dessus → currentViewId est déjà publié // quand on pousse. - // biome-ignore lint/correctness/useExhaustiveDependencies: size revision + // biome-ignore lint/correctness/useExhaustiveDependencies: size revision, locale useEffect(() => { if (viewId === null || !document) { return; @@ -163,8 +165,9 @@ export function NativeCompositorOverlay() { // re-trigger this effect when the probed-size cache mutates; the actual value is // re-read fresh via getWebcamNativeSize() above on every run (biome flags this as // an "unnecessary" dependency, but removing it would mean a probed webcam size - // arriving after mount never gets pushed to native). - }, [viewId, document, sources, _webcamSizeRevision]); + // arriving after mount never gets pushed to native). `locale` is the same kind of + // trigger: the label text is read inside `buildSceneDescription`. + }, [viewId, document, sources, _webcamSizeRevision, locale]); // SYNCHRO COMPLETE DES PARAMS, en un seul endroit. // @@ -276,6 +279,7 @@ export function NativeCompositorOverlay() { camera.offsetSec, activeClipIndex, activeSourceTimeSec, + assetAdditionalCameraSources(asset), ) .then(() => { if (pendingTargetClipIdRef.current !== targetClipId) { @@ -352,6 +356,7 @@ export function NativeCompositorOverlay() { camera.offsetSec, activeClipIndex, activeSourceTimeSec, + assetAdditionalCameraSources(asset), ).catch((error: unknown) => { console.warn("[compositor-view] re-anchoring the preview failed:", error); }); diff --git a/src/components/ai-edition/NewEditorShell.module.css b/src/components/ai-edition/NewEditorShell.module.css index 30b3c0947..162591937 100644 --- a/src/components/ai-edition/NewEditorShell.module.css +++ b/src/components/ai-edition/NewEditorShell.module.css @@ -763,6 +763,26 @@ visibility: hidden; display: block; } +.layoutPlace { + /* Move/resize hitbox of a layout section's camera window, over the native-drawn picture. */ + position: absolute; + z-index: 3; + box-sizing: border-box; + border: 1px dashed var(--accent); + cursor: move; + touch-action: none; +} +.layoutPlaceHandle { + position: absolute; + width: 12px; + height: 12px; + transform: translate(-50%, -50%); + box-sizing: border-box; + border: 2px solid var(--accent); + border-radius: 2px; + background: #fff; + touch-action: none; +} /* ─── transport bar (lives inside .timelineHead, next to the tool icons) ─ */ .transport { display: flex; diff --git a/src/components/ai-edition/NewEditorShell.tsx b/src/components/ai-edition/NewEditorShell.tsx index ee12161bd..032b878cd 100644 --- a/src/components/ai-edition/NewEditorShell.tsx +++ b/src/components/ai-edition/NewEditorShell.tsx @@ -2,6 +2,7 @@ import { useCallback, useEffect, useMemo, useRef, useState } from "react"; import { toast } from "sonner"; import type { EditorProjectData } from "@/components/video-editor/projectPersistence"; import { toFileUrl } from "@/components/video-editor/projectPersistence"; +import type { CameraSettings } from "@/components/video-editor/types"; import { useEditorDialogActions } from "@/contexts/EditorDialogsContext"; import { useScopedT } from "@/contexts/I18nContext"; import { useShortcuts } from "@/contexts/ShortcutsContext"; @@ -33,6 +34,12 @@ import { } from "@/lib/ai-edition/schema"; import { useMcpDocumentHost } from "@/lib/ai-edition/store/mcpDocumentHost"; import { saveWithDeadline, useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import { + copySourceKey, + pasteHitsCameraSection, + pasteIdPrefix, + pasteTarget, +} from "@/lib/ai-edition/store/regionClipboardKinds"; import { useAssetTranscriptions, useAutoTranscription, @@ -44,6 +51,7 @@ import { future as redoStack, past as undoStack } from "@/lib/ai-edition/store/u import { useChatPromptBus } from "@/lib/ai-edition/store/useChatPromptBus"; import { useSequentialTimelineOps } from "@/lib/ai-edition/store/useSequentialTimelineOps"; import { useTimeline } from "@/lib/ai-edition/store/useTimeline"; +import { showCameraSectionOutcome } from "@/lib/ai-edition/timeline/cameraSectionNotice"; import { isGeneratedAssetId } from "@/lib/ai-edition/timeline/clip-parts"; import { mergeCloseCuts } from "@/lib/ai-edition/timeline/cut-breath"; import { newRegionDurationSec } from "@/lib/ai-edition/timeline/newRegionDuration"; @@ -58,6 +66,8 @@ import { nativeBridgeClient } from "@/native"; import type { AiEditionProjectSummary } from "@/native/contracts"; import { resolveVisibleClips } from "@/native/sceneDescription"; import { useNativePlaybackSync } from "@/native/useNativePlaybackSync"; +import { type CalibrationCamera, CameraCalibrationModal } from "./CameraCalibrationModal"; +import { type CalibrationMode, calibrationCameraAt, cameraHasCrop } from "./CamerasSection"; import { ExportDialog } from "./ExportDialog"; import { insertionsEnabled } from "./insertionsEnabled"; import { ChatStripPanel } from "./LeftPanel"; @@ -192,6 +202,8 @@ export async function runLoadedMetadataWrite( export function NewEditorShell() { const te = useScopedT("editor"); + const tt = useScopedT("timeline"); + const ts = useScopedT("settings"); useMcpDocumentHost(); const document = useProjectStore((s) => s.document); const projectId = useProjectStore((s) => s.projectId); @@ -266,6 +278,20 @@ export function NewEditorShell() { // "Edit clip" rail button — a single shell-level instance instead of one // mounted per trigger site. const [editClipTarget, setEditClipTarget] = useState(null); + // The camera calibration dialog (perspective or crop), opened from the layout pane's camera + // list. The still is taken at the playhead as it opens; one instance for the editor. + const [calibration, setCalibration] = useState<{ + camera: CalibrationCamera; + mode: CalibrationMode; + } | null>(null); + const openCalibration = useCallback( + (cameraIndex: number, mode: CalibrationMode) => { + const { document: doc, currentTimeSec } = useProjectStore.getState(); + const camera = calibrationCameraAt(doc, currentTimeSec, cameraIndex, ts); + if (camera) setCalibration({ camera, mode }); + }, + [ts], + ); const [exportOpen, setExportOpen] = useState(false); const [unsavedPrompt, setUnsavedPrompt] = useState<{ action: "close" | "new" | "open" | "record"; @@ -340,6 +366,15 @@ export function NewEditorShell() { saveDocument, }); + // Every per-camera settings write (Cameras section toggles, reset, calibration apply) goes + // through the shared queue, so a toggle cannot race a calibration apply on a stale document. + const setTimelineCameraSettings = tl.setCameraSettings; + const setCameraSettingsQueued = useCallback( + (index: number, patch: Partial | null) => + enqueueTimelineWrite(() => setTimelineCameraSettings(index, patch)), + [enqueueTimelineWrite, setTimelineCameraSettings], + ); + const promptUnsaved = useCallback( (action: "close" | "new" | "open" | "record"): Promise => { if (!dirty) return Promise.resolve("discard"); @@ -1082,7 +1117,9 @@ export function NewEditorShell() { // Land it at the playhead, keeping the copied length. const timeMs = Math.round(useProjectStore.getState().currentTimeSec * 1000); const src = snapshot.region as { startMs: number; endMs: number }; - const prefix = snapshot.kind === "annotation" ? "ann" : snapshot.kind; + const prefix = pasteIdPrefix(snapshot.kind); + const target = pasteTarget(snapshot.kind); + if (!target) return; // Audio re-ventilates through its own anchorer, which advances each // fragment's source offset — the generic one would copy the offset into @@ -1120,32 +1157,24 @@ export function NewEditorShell() { () => createId(prefix), ); - if (snapshot.kind === "zoom") { - await saveDocument( - { - ...doc, - zoomRanges: [...doc.zoomRanges, ...anchored] as typeof doc.zoomRanges, - }, - { history: true }, - ); - } else if (snapshot.kind === "annotation") { - await saveDocument( - { - ...doc, - annotations: [...doc.annotations, ...anchored] as typeof doc.annotations, - }, - { history: true }, - ); + if (target.store === "document") { + const rows = doc[target.key] as unknown[]; + await saveDocument({ ...doc, [target.key]: [...rows, ...anchored] }, { history: true }); } else { - // speed and cameraFullscreen are both plain spans on legacyEditor. - const key = snapshot.kind === "speed" ? "speedRegions" : "cameraFullscreenRegions"; const legacy = (doc.legacyEditor as Record) ?? {}; - const prev = (legacy[key] as unknown[]) ?? []; + // Full Camera and layout sections share one lane: a paste that would land on the + // other list is refused like an add is, instead of being dropped by the scene. A Full + // Camera over Full Camera is not refused: those merge, as they do on add. + if ( + target.key !== "speedRegions" && + pasteHitsCameraSection(legacy, target.key, pasted.startMs, pasted.endMs) + ) { + showCameraSectionOutcome("occupied", tt); + return; + } + const prev = (legacy[target.key] as unknown[]) ?? []; await saveDocument( - { - ...doc, - legacyEditor: { ...legacy, [key]: [...prev, ...anchored] }, - }, + { ...doc, legacyEditor: { ...legacy, [target.key]: [...prev, ...anchored] } }, { history: true }, ); } @@ -1153,7 +1182,7 @@ export function NewEditorShell() { // `tl` belongs here now that the trim branch calls tl.addTrim: useTimeline // returns a fresh object each render, so memoizing on saveDocument alone // would paste through a callback holding a stale document. - }, [saveDocument, tl, te]); + }, [saveDocument, tl, te, tt]); // Copy the SELECTED pill. Reads the same arrays the lanes render, so what gets // copied is what the user is looking at — the old version dug into the raw @@ -1197,14 +1226,9 @@ export function NewEditorShell() { return; } - const source = - sel.kind === "zoom" - ? tl.zoomRegions - : sel.kind === "annotation" - ? tl.annotationRegions - : sel.kind === "speed" - ? tl.speedRegions - : tl.cameraFullscreenRegions; + const sourceKey = copySourceKey(sel.kind); + if (!sourceKey) return; + const source = tl[sourceKey]; const region = (source as Array<{ id: string }>).find((r) => r.id === sel.id); if (!region) return; copyRegion({ kind: sel.kind, region: region as unknown as Record }); @@ -1386,7 +1410,9 @@ export function NewEditorShell() { } if (matchesShortcut(e, shortcuts.addCameraFullscreen, isMac)) { e.preventDefault(); - void tl.addCameraFullscreen(newRegionDurationSec()); + void tl.addCameraFullscreen(newRegionDurationSec()).then((outcome) => { + showCameraSectionOutcome(outcome, tt); + }); return; } @@ -1433,6 +1459,7 @@ export function NewEditorShell() { isMac, togglePlay, handleSeek, + tt, ]); const showTimeline = mode !== "rec"; @@ -1622,6 +1649,12 @@ export function NewEditorShell() { selectedZoomRegionId={tl.selection?.kind === "zoom" ? tl.selection.id : null} onZoomFocusChange={tl.updateZoomFocusLive} onZoomFocusCommit={() => void tl.commitZoomFocus()} + cameraLayoutRegions={tl.cameraLayoutRegions} + selectedLayoutRegionId={ + tl.selection?.kind === "cameraLayout" ? tl.selection.id : null + } + onLayoutSlotRectLive={tl.updateLayoutSlotRectLive} + onLayoutSlotRectCommit={() => void tl.commitLayoutSlotRect()} annotationRegions={tl.annotationRegions} selectedAnnotationId={ tl.selection?.kind === "annotation" ? tl.selection.id : null @@ -1655,6 +1688,8 @@ export function NewEditorShell() { clips={tl.clips} onEditClip={setEditClipTarget} transcriptProps={transcriptProps} + onOpenCalibration={openCalibration} + setCameraSettings={setCameraSettingsQueued} /> ) : mode === "media" ? ( @@ -1746,6 +1781,18 @@ export function NewEditorShell() { setEditClipTarget(null); }} /> + {calibration ? ( + void setCameraSettingsQueued(calibration.camera.index, patch)} + onClose={() => setCalibration(null)} + /> + ) : null} { diff --git a/src/components/ai-edition/Preview.tsx b/src/components/ai-edition/Preview.tsx index f4ad0a5c5..23f032c00 100644 --- a/src/components/ai-edition/Preview.tsx +++ b/src/components/ai-edition/Preview.tsx @@ -1,5 +1,9 @@ import { useCallback, useEffect, useMemo, useState } from "react"; -import type { CameraFullscreenRegion, ZoomFocus } from "@/components/video-editor/types"; +import type { + CameraFullscreenRegion, + NormalizedRect, + ZoomFocus, +} from "@/components/video-editor/types"; import { useScopedT } from "@/contexts/I18nContext"; import type { AxcutAnnotationRegion, @@ -11,6 +15,7 @@ import type { } from "@/lib/ai-edition/schema"; import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; import type { SpeedRegion } from "@/lib/ai-edition/timeline/speed"; +import type { AnchoredCameraLayoutRegion } from "@/lib/cameraLayouts"; import { EditorEmptyState } from "./EditorEmptyState"; import styles from "./NewEditorShell.module.css"; import { PreviewCanvas } from "./PreviewCanvas"; @@ -43,6 +48,10 @@ interface PreviewProps { selectedZoomRegionId?: string | null; onZoomFocusChange?: (id: string, focus: ZoomFocus) => void; onZoomFocusCommit?: () => void; + cameraLayoutRegions?: AnchoredCameraLayoutRegion[]; + selectedLayoutRegionId?: string | null; + onLayoutSlotRectLive?: (id: string, slotIndex: number, rect: NormalizedRect) => void; + onLayoutSlotRectCommit?: () => void; annotationRegions?: AxcutAnnotationRegion[]; selectedAnnotationId?: string | null; onSelectAnnotation?: (id: string) => void; @@ -78,6 +87,10 @@ export function Preview({ selectedZoomRegionId, onZoomFocusChange, onZoomFocusCommit, + cameraLayoutRegions, + selectedLayoutRegionId, + onLayoutSlotRectLive, + onLayoutSlotRectCommit, annotationRegions, selectedAnnotationId, onSelectAnnotation, @@ -221,6 +234,10 @@ export function Preview({ selectedZoomRegionId={selectedZoomRegionId} onZoomFocusChange={onZoomFocusChange} onZoomFocusCommit={onZoomFocusCommit} + cameraLayoutRegions={cameraLayoutRegions} + selectedLayoutRegionId={selectedLayoutRegionId} + onLayoutSlotRectLive={onLayoutSlotRectLive} + onLayoutSlotRectCommit={onLayoutSlotRectCommit} annotationRegions={annotationRegions} selectedAnnotationId={selectedAnnotationId} onSelectAnnotation={onSelectAnnotation} diff --git a/src/components/ai-edition/PreviewCanvas.layoutPlaces.test.tsx b/src/components/ai-edition/PreviewCanvas.layoutPlaces.test.tsx new file mode 100644 index 000000000..3baa78fa3 --- /dev/null +++ b/src/components/ai-edition/PreviewCanvas.layoutPlaces.test.tsx @@ -0,0 +1,177 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { cleanup, fireEvent, render, screen } from "@testing-library/react"; +import type { ComponentProps } from "react"; +import { afterEach, beforeAll, describe, expect, it, vi } from "vitest"; +import type { AxcutAsset, AxcutClip } from "@/lib/ai-edition/schema"; +import { createEmptyDocument } from "@/lib/ai-edition/schema"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import type { AnchoredCameraLayoutRegion } from "@/lib/cameraLayouts"; +import { PreviewCanvas } from "./PreviewCanvas"; + +vi.mock("@/native/client", () => ({ nativeBridgeClient: { aiEdition: {} } })); +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: (scope: string) => (key: string) => `${scope}.${key}`, +})); +// The pixel and media layers are not under test: only the DOM hitboxes are. +vi.mock("./NativeCompositorOverlay", () => ({ NativeCompositorOverlay: () => null })); +vi.mock("./VirtualPreview", () => ({ VirtualPreview: () => null })); +vi.mock("./WebcamOverlay", () => ({ WebcamOverlay: () => null })); +vi.mock("./ZoomFocusOverlay", () => ({ ZoomFocusOverlay: () => null })); +vi.mock("./AnnotationLayer", () => ({ AnnotationLayer: () => null })); + +const track = (path: string) => ({ + sourcePath: path, + startMs: 0, + offsetMs: 0, + visible: true, + width: 1920, + height: 1080, +}); + +const asset = { + id: "a1", + kind: "video", + label: "a", + originalPath: "/screen.mp4", + cameraTrack: track("/c1.mp4"), + additionalCameraTracks: [{ ...track("/c2.mp4"), label: "" }], +} as unknown as AxcutAsset; + +const clip = { + id: "c1", + assetId: "a1", + sourceStartSec: 0, + sourceEndSec: 10, + timelineStartSec: 0, + timelineEndSec: 10, + wordRefs: [], + origin: "user", + reason: "", +} as AxcutClip; + +const region = ( + template: AnchoredCameraLayoutRegion["template"], + cameras: number[], +): AnchoredCameraLayoutRegion => ({ + id: "l1", + startMs: 0, + endMs: 10_000, + template, + slots: cameras.map((camera) => ({ camera })), + clipId: "c1", + assetId: "a1", + sourceStartSec: 0, + sourceEndSec: 10, +}); + +type Props = ComponentProps; + +function renderCanvas(over: Partial) { + const document = createEmptyDocument({ projectId: "p", title: "t" }); + useProjectStore.setState({ + projectId: "p", + document: { ...document, assets: [asset] }, + }); + const props: Props = { + videoSources: [], + clips: [clip], + seekTarget: null, + onTimeChange: vi.fn(), + onSeek: vi.fn(), + onLoadedMetadata: vi.fn(), + onVideoElement: vi.fn(), + currentTimeSec: 5, + onLayoutSlotRectLive: vi.fn(), + onLayoutSlotRectCommit: vi.fn(), + ...over, + }; + render( +
+ +
, + ); + return props; +} + +beforeAll(() => { + // jsdom lays nothing out: give every element a 1000×500 box and a pointer-capture API. + vi.spyOn(HTMLElement.prototype, "getBoundingClientRect").mockReturnValue({ + left: 0, + top: 0, + width: 1000, + height: 500, + right: 1000, + bottom: 500, + x: 0, + y: 0, + toJSON: () => ({}), + }); + HTMLElement.prototype.setPointerCapture = vi.fn(); + HTMLElement.prototype.releasePointerCapture = vi.fn(); +}); + +afterEach(() => { + cleanup(); + useProjectStore.getState().clear(); +}); + +describe("PreviewCanvas layout places", () => { + it("a selected layout section shows a handle per pip place", () => { + const regions = [region("screen-pip", [0, 1])]; + renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: "l1" }); + expect(screen.getAllByTestId("layout-place")).toHaveLength(2); + expect(screen.getAllByTestId("layout-place-handle")).toHaveLength(2); + }); + + it("gives a frame-filling place no handle", () => { + const regions = [region("camera-full-pip", [1, 0])]; + renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: "l1" }); + expect(screen.getAllByTestId("layout-place-handle")).toHaveLength(1); + }); + + it("shows nothing when the section is not selected or the playhead is outside it", () => { + const regions = [region("screen-pip", [0, 1])]; + renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: null }); + expect(screen.queryByTestId("layout-place")).toBeNull(); + cleanup(); + renderCanvas({ + cameraLayoutRegions: [{ ...regions[0], endMs: 2000 }], + selectedLayoutRegionId: "l1", + }); + expect(screen.queryByTestId("layout-place")).toBeNull(); + }); + + it("moves a place live and commits once on release", () => { + const regions = [region("camera-full-pip", [1, 0])]; + const onLive = vi.fn>(); + const props = renderCanvas({ + cameraLayoutRegions: regions, + selectedLayoutRegionId: "l1", + onLayoutSlotRectLive: onLive, + }); + const place = screen.getByTestId("layout-place"); + fireEvent.pointerDown(place, { pointerId: 1, clientX: 500, clientY: 250 }); + fireEvent.pointerMove(place, { pointerId: 1, clientX: 400, clientY: 200 }); + fireEvent.pointerMove(place, { pointerId: 1, clientX: 300, clientY: 150 }); + fireEvent.pointerUp(place, { pointerId: 1 }); + expect(onLive).toHaveBeenCalledTimes(2); + const [[id, slotIndex, first], [, , last]] = onLive.mock.calls; + // Slot 1 is the PiP of camera-full-pip (slot 0 fills the frame). + expect([id, slotIndex]).toEqual(["l1", 1]); + // Both moves are measured from the grab: the second one went twice as far. + expect(first.x - last.x).toBeCloseTo(0.1); + expect(first.y - last.y).toBeCloseTo(0.1); + expect(last.width).toBe(first.width); + expect(props.onLayoutSlotRectCommit).toHaveBeenCalledTimes(1); + }); + + it("leaves no undo step for a click without a move", () => { + const regions = [region("screen-pip", [0])]; + const props = renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: "l1" }); + const place = screen.getByTestId("layout-place"); + fireEvent.pointerDown(place, { pointerId: 1, clientX: 500, clientY: 250 }); + fireEvent.pointerUp(place, { pointerId: 1 }); + expect(props.onLayoutSlotRectCommit).not.toHaveBeenCalled(); + }); +}); diff --git a/src/components/ai-edition/PreviewCanvas.tsx b/src/components/ai-edition/PreviewCanvas.tsx index c817fde3c..8ed2c57b0 100644 --- a/src/components/ai-edition/PreviewCanvas.tsx +++ b/src/components/ai-edition/PreviewCanvas.tsx @@ -28,6 +28,7 @@ import { type CameraFullscreenRegion, type CropRegion, DEFAULT_CROP_REGION, + type NormalizedRect, type WebcamLayoutPreset, type WebcamMaskShape, type ZoomFocus, @@ -47,9 +48,23 @@ import type { } from "@/lib/ai-edition/schema"; import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; import { useEditorSettings } from "@/lib/ai-edition/store/useEditorSettings"; -import { resolveActiveCameraTrack } from "@/lib/ai-edition/timeline/camera"; +import { + assetAdditionalCameraSources, + assetCameraSource, + resolveActiveCameraTrack, +} from "@/lib/ai-edition/timeline/camera"; +import { + inwardCorner, + moveSlotRect, + type PipPlace, + pipPlacesOf, + resizeSlotRect, + type SlotCorner, +} from "@/lib/ai-edition/timeline/layoutSlotDrag"; import type { SpeedRegion } from "@/lib/ai-edition/timeline/speed"; +import { resolvePillIds } from "@/lib/ai-edition/timeline/timelineMap"; import { locateVirtualPosition } from "@/lib/ai-edition/timeline/virtual-preview"; +import { type AnchoredCameraLayoutRegion, normalizeCameraSettings } from "@/lib/cameraLayouts"; import { computeCameraFullscreenRect, computeCompositeLayout, @@ -61,7 +76,12 @@ import { webcamAnchorAt } from "@/lib/projectDefaults"; import { wallpaperStyle } from "@/lib/wallpaper"; import { getCssClipPath } from "@/lib/webcamMaskShapes"; import { computeCameraFullscreenProgress } from "@/lib/zoomMath/cameraFullscreenUtils"; -import { webcamBoxSourceSize } from "@/native/sceneDescription"; +import { + camera0PerspectiveOf, + cameraLayoutContextOf, + clipPipLayoutOf, + webcamBoxSourceSize, +} from "@/native/sceneDescription"; import { getWebcamNativeSize, getWebcamNativeSizeRevision, @@ -92,6 +112,13 @@ interface PreviewCanvasProps { selectedZoomRegionId?: string | null; onZoomFocusChange?: (id: string, focus: ZoomFocus) => void; onZoomFocusCommit?: () => void; + /** Layout sections; the selected one's PiP places get move/resize hitboxes. */ + cameraLayoutRegions?: AnchoredCameraLayoutRegion[]; + selectedLayoutRegionId?: string | null; + /** Live edit of a place's rect during a gesture (no undo step). */ + onLayoutSlotRectLive?: (id: string, slotIndex: number, rect: NormalizedRect) => void; + /** End of the gesture: one undo step. */ + onLayoutSlotRectCommit?: () => void; annotationRegions?: AxcutAnnotationRegion[]; selectedAnnotationId?: string | null; onSelectAnnotation?: (id: string) => void; @@ -243,6 +270,8 @@ export function PreviewCanvas(props: PreviewCanvasProps) { getWebcamNativeSizeRevision, () => 0, ); + // A perspective on camera 1 gives the box its corrected ratio, as in the scene. + const camera0Perspective = useMemo(() => camera0PerspectiveOf(document), [document]); // biome-ignore lint/correctness/useExhaustiveDependencies: the revision re-reads the probed-size cache const webcamSourceSize = useMemo( () => @@ -250,8 +279,9 @@ export function PreviewCanvas(props: PreviewCanvasProps) { activeCameraTrack, activeCameraTrack?.sourcePath ? getWebcamNativeSize(activeCameraTrack.sourcePath) : null, settings.webcamCropRegion, + camera0Perspective, ), - [activeCameraTrack, settings.webcamCropRegion, webcamSizeRevision], + [activeCameraTrack, settings.webcamCropRegion, camera0Perspective, webcamSizeRevision], ); const formatFill = useMemo(() => (document ? isFormatFillActive(document) : false), [document]); @@ -351,6 +381,60 @@ export function PreviewCanvas(props: PreviewCanvasProps) { // camera-less clip, so this is belt-and-braces rather than the only guard. const showWebcamSlot = Boolean(layout?.webcamRect && activeClipHasCamera); const [isPlaying, setIsPlaying] = useState(false); + + // The selected layout section's row under the playhead: one pill can span several clip- + // anchored rows, and the one under the playhead names the asset whose cameras are drawn. + const layoutRow = useMemo(() => { + const id = props.selectedLayoutRegionId; + const regions = props.cameraLayoutRegions; + if (!id || !regions) return null; + const pill = new Set(resolvePillIds(regions, id)); + const nowMs = props.currentTimeSec * 1000; + return regions.find((r) => pill.has(r.id) && nowMs >= r.startMs && nowMs < r.endMs) ?? null; + }, [props.selectedLayoutRegionId, props.cameraLayoutRegions, props.currentTimeSec]); + // Its PiP places, through the context the scene builds, so each hitbox sits on the window + // the compositor draws. Frame-filling places get none. + // biome-ignore lint/correctness/useExhaustiveDependencies: the revision re-reads the probed-size cache + const layoutPlaces = useMemo((): PipPlace[] => { + if (!layoutRow || frameSize.width <= 0 || frameSize.height <= 0) return []; + const asset = assets.find((a) => a.id === (layoutRow.assetId ?? activeClip?.assetId)); + if (!asset) return []; + const sources = [assetCameraSource(asset), ...assetAdditionalCameraSources(asset)]; + const legacy = document?.legacyEditor as Record | null | undefined; + const camera0Path = asset.cameraTrack?.sourcePath; + const maskShape = settings.webcamMaskShape as WebcamMaskShape; + const ctx = cameraLayoutContextOf({ + frame: frameSize, + asset, + cameraSettings: normalizeCameraSettings(legacy?.cameraSettings), + webcamCropRegion: settings.webcamCropRegion, + probedCamera0Size: camera0Path ? getWebcamNativeSize(camera0Path) : null, + clipLayout: clipPipLayoutOf(layout, frameSize, maskShape), + webcamMaskShape: maskShape, + webcamRoundness: settings.webcamRoundness, + pipPreset: + resolveWebcamLayoutPreset( + settings.webcamLayoutPreset as WebcamLayoutPreset, + activeClipHasCamera, + ) === "picture-in-picture", + }); + return pipPlacesOf(layoutRow, ctx).filter((p) => (sources[p.camera]?.path ?? "") !== ""); + }, [ + layoutRow, + frameSize, + assets, + activeClip, + document, + layout, + activeClipHasCamera, + settings.webcamCropRegion, + settings.webcamMaskShape, + settings.webcamRoundness, + settings.webcamLayoutPreset, + webcamSizeRevision, + ]); + const editsLayoutPlaces = + layoutPlaces.length > 0 && !isPlaying && props.onLayoutSlotRectLive !== undefined; const handleVideoElement = useMemo(() => props.onVideoElement, [props.onVideoElement]); // L'élément `