From 742d3d865a787728e4cc6a784155ed1b41e42017 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sat, 3 Oct 2026 21:45:00 +0200 Subject: [PATCH 01/22] feat(wgc): read a list of cameras from the helper config --- electron/native/wgc-capture/CMakeLists.txt | 20 +++ .../native/wgc-capture/src/json_fields.cpp | 125 ++++++++++++++++++ electron/native/wgc-capture/src/json_fields.h | 10 ++ electron/native/wgc-capture/src/main.cpp | 118 +---------------- .../native/wgc-capture/src/webcam_config.cpp | 92 +++++++++++++ .../native/wgc-capture/src/webcam_config.h | 17 +++ .../wgc-capture/src/webcam_config_test.cpp | 63 +++++++++ scripts/build-windows-wgc-helper.mjs | 8 ++ 8 files changed, 336 insertions(+), 117 deletions(-) create mode 100644 electron/native/wgc-capture/src/json_fields.cpp create mode 100644 electron/native/wgc-capture/src/json_fields.h create mode 100644 electron/native/wgc-capture/src/webcam_config.cpp create mode 100644 electron/native/wgc-capture/src/webcam_config.h create mode 100644 electron/native/wgc-capture/src/webcam_config_test.cpp diff --git a/electron/native/wgc-capture/CMakeLists.txt b/electron/native/wgc-capture/CMakeLists.txt index 50466d0d0..84aa89853 100644 --- a/electron/native/wgc-capture/CMakeLists.txt +++ b/electron/native/wgc-capture/CMakeLists.txt @@ -44,6 +44,8 @@ add_executable(wgc-capture src/realtime_scheduling.h src/frame_visibility.cpp src/frame_visibility.h + src/json_fields.cpp + src/json_fields.h src/dshow_webcam_capture.cpp src/dshow_webcam_capture.h src/main.cpp @@ -60,6 +62,8 @@ add_executable(wgc-capture src/webcam_capture.cpp src/webcam_format.cpp src/webcam_format.h + src/webcam_config.cpp + src/webcam_config.h src/webcam_capture.h src/wgc_session.cpp src/wgc_session.h @@ -198,3 +202,19 @@ target_link_libraries(mf_encoder_color_test PRIVATE ole32 propsys ) + +add_executable(webcam_config_test + src/json_fields.cpp + src/json_fields.h + src/webcam_config.cpp + src/webcam_config.h + src/webcam_config_test.cpp +) + +target_compile_definitions(webcam_config_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(webcam_config_test PRIVATE /EHsc /W4 /utf-8) diff --git a/electron/native/wgc-capture/src/json_fields.cpp b/electron/native/wgc-capture/src/json_fields.cpp new file mode 100644 index 000000000..e36a43e81 --- /dev/null +++ b/electron/native/wgc-capture/src/json_fields.cpp @@ -0,0 +1,125 @@ +#include "json_fields.h" + +#include +#include + +// Lightweight field lookups over the helper's single JSON argument. They match +// the first occurrence of a key anywhere in the document. + +bool findBool(const std::string& json, const std::string& key, bool fallback) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return fallback; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return fallback; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + if (json.compare(pos, 4, "true") == 0) { + return true; + } + if (json.compare(pos, 5, "false") == 0) { + return false; + } + return fallback; +} + +int64_t findInt64(const std::string& json, const std::string& key, int64_t fallback) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return fallback; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return fallback; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + try { + return std::stoll(json.substr(pos)); + } catch (...) { + return fallback; + } +} + +int findInt(const std::string& json, const std::string& key, int fallback) { + return static_cast(findInt64(json, key, fallback)); +} + +double findDouble(const std::string& json, const std::string& key, double fallback) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return fallback; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return fallback; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + try { + return std::stod(json.substr(pos)); + } catch (...) { + return fallback; + } +} + +std::string findString(const std::string& json, const std::string& key) { + auto pos = json.find("\"" + key + "\""); + if (pos == std::string::npos) { + return {}; + } + pos = json.find(':', pos); + if (pos == std::string::npos) { + return {}; + } + pos += 1; + while (pos < json.size() && std::isspace(static_cast(json[pos]))) { + pos += 1; + } + if (pos >= json.size() || json[pos] != '"') { + return {}; + } + pos += 1; + + std::string result; + while (pos < json.size()) { + const char c = json[pos++]; + if (c == '"') { + break; + } + if (c == '\\' && pos < json.size()) { + const char escaped = json[pos++]; + switch (escaped) { + case '\\': + case '"': + case '/': + result.push_back(escaped); + break; + case 'n': + result.push_back('\n'); + break; + case 'r': + result.push_back('\r'); + break; + case 't': + result.push_back('\t'); + break; + default: + result.push_back(escaped); + break; + } + continue; + } + result.push_back(c); + } + return result; +} diff --git a/electron/native/wgc-capture/src/json_fields.h b/electron/native/wgc-capture/src/json_fields.h new file mode 100644 index 000000000..0921b3381 --- /dev/null +++ b/electron/native/wgc-capture/src/json_fields.h @@ -0,0 +1,10 @@ +#pragma once + +#include +#include + +bool findBool(const std::string& json, const std::string& key, bool fallback); +int64_t findInt64(const std::string& json, const std::string& key, int64_t fallback); +int findInt(const std::string& json, const std::string& key, int fallback); +double findDouble(const std::string& json, const std::string& key, double fallback); +std::string findString(const std::string& json, const std::string& key); diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index 1bad39699..5ba80cd49 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -8,6 +8,7 @@ #include "wasapi_loopback_capture.h" #include "wasapi_render_keepalive.h" #include "frame_visibility.h" +#include "json_fields.h" #include "webcam_capture.h" #include "wgc_session.h" @@ -396,123 +397,6 @@ void reportCaptureAdapters(ID3D11Device* device, HMONITOR targetMonitor) { } -bool findBool(const std::string& json, const std::string& key, bool fallback) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return fallback; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return fallback; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - if (json.compare(pos, 4, "true") == 0) { - return true; - } - if (json.compare(pos, 5, "false") == 0) { - return false; - } - return fallback; -} - -int64_t findInt64(const std::string& json, const std::string& key, int64_t fallback) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return fallback; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return fallback; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - try { - return std::stoll(json.substr(pos)); - } catch (...) { - return fallback; - } -} - -int findInt(const std::string& json, const std::string& key, int fallback) { - return static_cast(findInt64(json, key, fallback)); -} - -double findDouble(const std::string& json, const std::string& key, double fallback) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return fallback; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return fallback; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - try { - return std::stod(json.substr(pos)); - } catch (...) { - return fallback; - } -} - -std::string findString(const std::string& json, const std::string& key) { - auto pos = json.find("\"" + key + "\""); - if (pos == std::string::npos) { - return {}; - } - pos = json.find(':', pos); - if (pos == std::string::npos) { - return {}; - } - pos += 1; - while (pos < json.size() && std::isspace(static_cast(json[pos]))) { - pos += 1; - } - if (pos >= json.size() || json[pos] != '"') { - return {}; - } - pos += 1; - - std::string result; - while (pos < json.size()) { - const char c = json[pos++]; - if (c == '"') { - break; - } - if (c == '\\' && pos < json.size()) { - const char escaped = json[pos++]; - switch (escaped) { - case '\\': - case '"': - case '/': - result.push_back(escaped); - break; - case 'n': - result.push_back('\n'); - break; - case 'r': - result.push_back('\r'); - break; - case 't': - result.push_back('\t'); - break; - default: - result.push_back(escaped); - break; - } - continue; - } - result.push_back(c); - } - return result; -} std::string parseWindowHandleFromSourceId(const std::string& sourceId) { constexpr char prefix[] = "window:"; diff --git a/electron/native/wgc-capture/src/webcam_config.cpp b/electron/native/wgc-capture/src/webcam_config.cpp new file mode 100644 index 000000000..d437b9d26 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_config.cpp @@ -0,0 +1,92 @@ +#include "webcam_config.h" + +#include "json_fields.h" + +#include + +namespace { + +// The find* helpers search the whole document, so each array entry is sliced +// out first to keep its fields from being confused with the top-level ones. +std::vector splitTopLevelObjects(const std::string& json, const std::string& arrayKey) { + std::vector objects; + size_t pos = json.find("\"" + arrayKey + "\""); + if (pos == std::string::npos) { + return objects; + } + pos = json.find('[', pos); + if (pos == std::string::npos) { + return objects; + } + ++pos; + + int depth = 0; + bool inString = false; + size_t objectStart = 0; + for (; pos < json.size(); ++pos) { + const char c = json[pos]; + if (inString) { + if (c == '\\') { + ++pos; + } else if (c == '"') { + inString = false; + } + continue; + } + if (c == '"') { + inString = true; + } else if (c == '{') { + if (depth == 0) { + objectStart = pos; + } + ++depth; + } else if (c == '}') { + if (depth > 0 && --depth == 0) { + objects.push_back(json.substr(objectStart, pos - objectStart + 1)); + } + } else if (c == ']' && depth == 0) { + break; + } + } + return objects; +} + +} // namespace + +std::vector parseWebcamConfigs(const std::string& json) { + std::vector configs; + for (const std::string& entry : splitTopLevelObjects(json, "webcams")) { + if (configs.size() >= kMaxWebcams) { + break; + } + WebcamConfig config; + config.outputPath = findString(entry, "camPath"); + if (config.outputPath.empty()) { + continue; + } + config.deviceId = findString(entry, "camDeviceId"); + config.deviceName = findString(entry, "camDeviceName"); + config.directShowClsid = findString(entry, "camClsid"); + config.width = findInt(entry, "camWidth", 0); + config.height = findInt(entry, "camHeight", 0); + config.fps = findInt(entry, "camFps", 0); + configs.push_back(std::move(config)); + } + if (!configs.empty() || !findBool(json, "webcamEnabled", false)) { + return configs; + } + + WebcamConfig legacy; + legacy.outputPath = findString(json, "webcamPath"); + if (legacy.outputPath.empty()) { + return configs; + } + legacy.deviceId = findString(json, "webcamDeviceId"); + legacy.deviceName = findString(json, "webcamDeviceName"); + legacy.directShowClsid = findString(json, "webcamDirectShowClsid"); + legacy.width = findInt(json, "webcamWidth", 0); + legacy.height = findInt(json, "webcamHeight", 0); + legacy.fps = findInt(json, "webcamFps", 0); + configs.push_back(std::move(legacy)); + return configs; +} diff --git a/electron/native/wgc-capture/src/webcam_config.h b/electron/native/wgc-capture/src/webcam_config.h new file mode 100644 index 000000000..b0c0998ea --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_config.h @@ -0,0 +1,17 @@ +#pragma once + +#include +#include +#include + +struct WebcamConfig { + std::string deviceId, deviceName, directShowClsid, outputPath; + int width = 0, height = 0, fps = 0; +}; + +constexpr size_t kMaxWebcams = 4; + +// The cameras to record, in order. Reads the `webcams` list when present and +// non-empty; otherwise the legacy single-camera fields (webcamEnabled, webcamDeviceId, +// webcamDeviceName, webcamDirectShowClsid, webcamWidth/Height/Fps, webcamPath). +std::vector parseWebcamConfigs(const std::string& json); diff --git a/electron/native/wgc-capture/src/webcam_config_test.cpp b/electron/native/wgc-capture/src/webcam_config_test.cpp new file mode 100644 index 000000000..1696698bf --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_config_test.cpp @@ -0,0 +1,63 @@ +#include "webcam_config.h" + +#include +#include + +namespace { +int failures = 0; +void expect(bool ok, const char* label) { + if (!ok) { + std::printf("FAIL %s\n", label); + ++failures; + } +} +} // namespace + +int main() { + const std::string legacy = + R"({"fps":60,"width":2560,"height":1440,"webcamEnabled":true,"webcamDeviceId":"id1",)" + R"("webcamDeviceName":"Cam One","webcamDirectShowClsid":"{A}","webcamWidth":1920,)" + R"("webcamHeight":1080,"webcamFps":30,"outputs":{"screenPath":"s.mp4","webcamPath":"w.mp4"}})"; + auto l = parseWebcamConfigs(legacy); + expect(l.size() == 1, "legacy: one camera"); + expect(l.size() == 1 && l[0].deviceName == "Cam One" && l[0].outputPath == "w.mp4" && + l[0].width == 1920 && l[0].fps == 30, + "legacy: fields"); + + expect(parseWebcamConfigs(R"({"webcamEnabled":false,"webcamPath":"w.mp4"})").empty(), + "legacy disabled: none"); + + const std::string list = + R"({"fps":60,"width":2560,"webcamEnabled":true,"webcamDeviceName":"Cam One","webcamPath":"w.mp4",)" + R"("webcams":[{"camDeviceId":"id1","camDeviceName":"Cam One","camClsid":"{A}","camWidth":1920,)" + R"("camHeight":1080,"camFps":30,"camPath":"w.mp4"},)" + R"({"camDeviceId":"id2","camDeviceName":"Desk \"Cam\"","camClsid":"","camWidth":1280,)" + R"("camHeight":720,"camFps":30,"camPath":"w-2.mp4"}]})"; + auto m = parseWebcamConfigs(list); + expect(m.size() == 2, "list: two cameras"); + expect(m.size() == 2 && m[1].deviceName == "Desk \"Cam\"" && m[1].outputPath == "w-2.mp4" && + m[1].width == 1280 && m[1].height == 720, + "list: second camera fields, escaped quote"); + expect(m.size() == 2 && m[0].fps == 30, "list: entry fps not the top-level fps"); + + expect(parseWebcamConfigs(R"({"webcamEnabled":true,"webcamPath":"w.mp4","webcams":[]})").size() == 1, + "empty list falls back to legacy"); + + std::string five = R"({"webcams":[)"; + for (int i = 0; i < 5; ++i) { + five += (i ? "," : ""); + five += R"({"camDeviceName":"C","camPath":"p)" + std::to_string(i) + R"(.mp4"})"; + } + five += "]}"; + expect(parseWebcamConfigs(five).size() == 4, "capped at four"); + + expect(parseWebcamConfigs(R"({"webcams":[{"camDeviceName":"No path"}]})").empty(), + "entry without camPath is skipped"); + + if (failures == 0) { + std::printf("webcam_config_test: all assertions passed\n"); + return 0; + } + std::printf("webcam_config_test: %d assertion(s) failed\n", failures); + return 1; +} diff --git a/scripts/build-windows-wgc-helper.mjs b/scripts/build-windows-wgc-helper.mjs index ab4b26c00..143fa689c 100644 --- a/scripts/build-windows-wgc-helper.mjs +++ b/scripts/build-windows-wgc-helper.mjs @@ -116,6 +116,14 @@ if (!fs.existsSync(webcamFormatTestPath)) { await run(webcamFormatTestPath, [], { cwd: BUILD_DIR }); console.log(`Passed ${webcamFormatTestPath}`); +const webcamConfigTestPath = path.join(BUILD_DIR, "webcam_config_test.exe"); +if (!fs.existsSync(webcamConfigTestPath)) { + throw new Error(`WGC helper build completed but ${webcamConfigTestPath} was not found.`); +} +// Guards how the helper reads the list of cameras (and the legacy single-camera fields). +await run(webcamConfigTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${webcamConfigTestPath}`); + const frameVisibilityTestPath = path.join(BUILD_DIR, "frame_visibility_test.exe"); if (!fs.existsSync(frameVisibilityTestPath)) { throw new Error(`WGC helper build completed but ${frameVisibilityTestPath} was not found.`); From d09a1ac6c41e16f7956ff8c9db713ca4073b93d8 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:04:28 +0200 Subject: [PATCH 02/22] feat(wgc): record every camera in the config on the shared clock Each camera of the webcams list (or the legacy single-camera fields) gets its own capture and encoder on the recording's T0. A camera that cannot be opened is dropped with an indexed webcam-unavailable warning; one whose samples fail mid-take is disabled on its own. recording-stopped keeps webcamPath and adds webcamPaths. MFEncoder now only balances an MFStartup it made: a dropped camera's never-initialized encoder ran an unmatched MFShutdown from its destructor and stopped every other encoder in the process. --- electron/native/README.md | 17 + electron/native/wgc-capture/src/main.cpp | 410 ++++++++++++------ .../native/wgc-capture/src/mf_encoder.cpp | 6 +- electron/native/wgc-capture/src/mf_encoder.h | 6 + scripts/test-windows-wgc-helper.mjs | 67 +++ 5 files changed, 363 insertions(+), 143 deletions(-) diff --git a/electron/native/README.md b/electron/native/README.md index 5f68fb254..a66efbef8 100644 --- a/electron/native/README.md +++ b/electron/native/README.md @@ -85,6 +85,17 @@ Current V2 JSON shape: The current helper implementation supports display/window video capture, system audio loopback, selected-microphone capture, Media Foundation webcam capture, and a DirectShow webcam fallback for virtual cameras that are not exposed through Media Foundation. Webcam frames are currently composed into the primary MP4 as a bottom-right picture-in-picture overlay. Browser `deviceId` values do not always map to Media Foundation symbolic links or WASAPI endpoint IDs, so the renderer passes both browser IDs and user-visible device names. For microphones, the helper tries the requested WASAPI endpoint ID first, then resolves an active capture endpoint by `microphoneDeviceName`, then falls back to the default endpoint. For webcams, Electron resolves a matching DirectShow filter CLSID for the selected label; the helper uses Media Foundation first, then that exact DirectShow filter when the requested camera is absent from Media Foundation. +Several cameras: a `webcams` list records up to four cameras, each into its own MP4, all on the same T0 as the screen and audio. Each entry takes `camDeviceId`, `camDeviceName`, `camClsid` (the DirectShow filter CLSID), `camWidth`, `camHeight`, `camFps` and `camPath`; an entry without `camPath` is skipped. When the list is present and non-empty it replaces the legacy `webcam*` fields; without it those fields still describe one camera, which writes to `webcamPath` when one is given and is drawn into the screen as the inline picture-in-picture otherwise. + +```json +"webcams": [ + { "camDeviceId": "…", "camDeviceName": "Camera A", "camClsid": "{…}", "camWidth": 1920, "camHeight": 1080, "camFps": 30, "camPath": "C:\\path\\recording-123-webcam.mp4" }, + { "camDeviceId": "…", "camDeviceName": "Camera B", "camClsid": "", "camWidth": 1280, "camHeight": 720, "camFps": 30, "camPath": "C:\\path\\recording-123-webcam-2.mp4" } +] +``` + +Every per-camera event carries the camera's `index` in that list: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. + Container: recordings are written as fragmented MP4 (`MFCreateFMPEG4MediaSink` + `MFCreateSinkWriterFromMediaSink`, `MF_MPEG4SINK_MIN_FRAGMENT_DURATION` = 1s) rather than plain MP4. A plain MP4 has no index until `IMFSinkWriter::Finalize()` writes `moov` at the very end, so when the shutdown watchdog force-exits a wedged helper the file on disk holds every frame and no way to read them — that is why issues #252 / #292 / #327 cost the whole recording rather than the frozen tail of it. A fragmented MP4 writes its index up front and its samples in self-describing `moof`+`mdat` pairs, so the same kill leaves a file that plays up to the last complete fragment. This does not fix the freeze; it removes the data loss the freeze causes. Because the fragmented sink needs both output media types at construction, the sink writer is built from a media sink instead of from a URL, and the helper reads the video/audio stream positions back off the sink rather than assuming them. If any of that is unavailable on a machine, the helper retries with the plain container and says so — `container` in the `encoder-selection` event is `fragmented-mp4` or `mp4`, and it reports what was used, not what was asked for. Encoder selection: by default the helper keeps the existing sink-writer path first. If that path fails while setting up H.264, it retries with the Microsoft software H.264 encoder (`mfh264enc.dll`). The key of this retry is registering that encoder locally in the helper process via `MFTRegisterLocalByCLSID`, which makes a software H.264 encoder available even when the machine's hardware encoders are missing or broken; hardware transforms are disabled for the retry only as a secondary guard so the sink writer prefers the locally registered software encoder, not as the fallback mechanism itself. Set `preferSoftwareEncoder: true` in the helper JSON, or set `OPENSCREEN_WGC_PREFER_SOFTWARE_ENCODER=true` before launching Electron, to force the software path from the first attempt. @@ -126,6 +137,12 @@ npm run test:wgc-webcam:win Remove-Item Env:OPENSCREEN_WGC_TEST_WEBCAM_DEVICE_NAME ``` +To check that a camera which cannot be opened costs only itself, record the real camera next to a nonexistent second one listed through `webcams`: + +```powershell +npm run test:wgc-helper:win -- --webcam --missing-second-webcam +``` + To validate a specific native microphone manually: ```powershell diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index 5ba80cd49..a0785dab5 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -10,6 +10,7 @@ #include "frame_visibility.h" #include "json_fields.h" #include "webcam_capture.h" +#include "webcam_config.h" #include "wgc_session.h" #include @@ -27,6 +28,7 @@ #include #include #include +#include namespace { @@ -38,7 +40,6 @@ struct CaptureConfig { std::string sourceId; std::string windowHandle; std::string outputPath; - std::string webcamOutputPath; int fps = 60; int width = 0; int height = 0; @@ -139,6 +140,49 @@ struct CaptureControl { } }; +// One camera of the take: its capture, its encoder and the writer-loop state that used +// to be single variables. All cameras share the recording's T0. +struct WebcamStream { + WebcamConfig config; + // The camera's position in the config list, kept when an earlier camera is + // dropped, so every event names the camera the caller asked for. + size_t index = 0; + WebcamCapture capture; + MFEncoder encoder; + bool active = false; + // False only for the legacy inline picture-in-picture camera, which has no + // file of its own and is drawn into the screen frame instead. + bool writeSeparate = false; + std::vector latestFrame; + int latestWidth = 0; + int latestHeight = 0; + uint64_t latestSequence = 0; + bool hasVisibleFrame = false; + int64_t lastTimestampHns = -1; + int64_t nextWriteDueHns = 0; + int64_t nominalIntervalHns = 0; + // Captured in the writer's pull block, submitted after it (issue #115). + Microsoft::WRL::ComPtr pendingSample; + // Owned here because the shutdown watchdog reads the current step's name + // through a raw pointer from another thread. + std::string finalizeStepName; + + // Takes the camera's newest frame if it carries a picture. Returns whether it did. + bool pullVisibleFrame() { + WebcamFrameSnapshot candidate; + if (!capture.copyLatestFrame(candidate, latestSequence) || + !hasVisibleWebcamContent(candidate.data, capture.deliversNv12())) { + return false; + } + latestFrame = std::move(candidate.data); + latestWidth = candidate.width; + latestHeight = candidate.height; + latestSequence = candidate.sequence; + hasVisibleFrame = true; + return true; + } +}; + int readEnvInt(const char* name, int fallback) { char raw[32]{}; const DWORD length = GetEnvironmentVariableA(name, raw, static_cast(sizeof(raw))); @@ -469,7 +513,6 @@ bool parseConfig(const std::string& json, CaptureConfig& config) { config.webcamDeviceId = findString(json, "webcamDeviceId"); config.webcamDeviceName = findString(json, "webcamDeviceName"); config.webcamDirectShowClsid = findString(json, "webcamDirectShowClsid"); - config.webcamOutputPath = findString(json, "webcamPath"); config.webcamWidth = findInt(json, "webcamWidth", 0); config.webcamHeight = findInt(json, "webcamHeight", 0); config.webcamFps = findInt(json, "webcamFps", 0); @@ -544,8 +587,9 @@ int wmain(int argc, wchar_t* argv[]) { winrt::init_apartment(winrt::apartment_type::multi_threaded); + const std::string configJson = wideToUtf8(argv[1]); CaptureConfig config; - if (!parseConfig(wideToUtf8(argv[1]), config)) { + if (!parseConfig(configJson, config)) { std::cerr << "ERROR: Failed to parse config JSON" << std::endl; return 1; } @@ -645,42 +689,67 @@ int wmain(int argc, wchar_t* argv[]) { const int pixels = width * height; const int bitrate = pixels >= 3840 * 2160 ? 45'000'000 : pixels >= 2560 * 1440 ? 28'000'000 : 18'000'000; - WebcamCapture webcamCapture; - bool webcamActive = false; - // Decided before initialize(), not after: it selects the capture pixel - // format, and only a camera going to its own file can use NV12 -- an inline - // picture-in-picture composite needs the frame as BGRA. - bool writeSeparateWebcam = config.webcamEnabled && !config.webcamOutputPath.empty(); - if (config.webcamEnabled) { - if (!webcamCapture.initialize( - utf8ToWide(config.webcamDeviceId), - utf8ToWide(config.webcamDeviceName), - utf8ToWide(config.webcamDirectShowClsid), - config.webcamWidth, - config.webcamHeight, - config.webcamFps > 0 ? config.webcamFps : config.fps, - writeSeparateWebcam)) { + // Every camera of the take, in config order. The `webcams` list (or the + // legacy fields with a webcamPath) gives cameras that each write their own + // file. The legacy fields without a webcamPath are the one camera drawn + // into the screen frame as an inline picture-in-picture. + std::vector webcamConfigs = parseWebcamConfigs(configJson); + if (webcamConfigs.empty() && config.webcamEnabled) { + WebcamConfig inlineCamera; + inlineCamera.deviceId = config.webcamDeviceId; + inlineCamera.deviceName = config.webcamDeviceName; + inlineCamera.directShowClsid = config.webcamDirectShowClsid; + inlineCamera.width = config.webcamWidth; + inlineCamera.height = config.webcamHeight; + inlineCamera.fps = config.webcamFps; + webcamConfigs.push_back(std::move(inlineCamera)); + } + + std::vector> webcams; // unique_ptr: WebcamCapture/MFEncoder are not movable + for (size_t index = 0; index < webcamConfigs.size(); ++index) { + auto stream = std::make_unique(); + stream->config = webcamConfigs[index]; + stream->index = index; + stream->finalizeStepName = "webcam-encoder-finalize-" + std::to_string(index); + // Decided before initialize(), not after: it selects the capture pixel + // format, and only a camera going to its own file can use NV12 -- an inline + // picture-in-picture composite needs the frame as BGRA. + stream->writeSeparate = !stream->config.outputPath.empty(); + if (!stream->capture.initialize( + utf8ToWide(stream->config.deviceId), + utf8ToWide(stream->config.deviceName), + utf8ToWide(stream->config.directShowClsid), + stream->config.width, + stream->config.height, + stream->config.fps > 0 ? stream->config.fps : config.fps, + stream->writeSeparate)) { // Non-fatal: a screen+audio recording the user can still use is far // better than losing the whole recording because one camera device // didn't match. Report it so the renderer can inform the user (and, // historically, fall back to a browser-recorded webcam sidecar), but - // let capture continue without a native webcam track. - std::cerr << "WARNING: Failed to initialize native webcam capture; continuing without webcam" - << std::endl; - std::cout << "{\"event\":\"warning\",\"code\":\"webcam-unavailable\",\"message\":" - "\"Failed to initialize native webcam capture\"}" - << std::endl; - config.webcamEnabled = false; - writeSeparateWebcam = false; - } else { - std::cout << "{\"event\":\"webcam-format\",\"schemaVersion\":2,\"width\":" << webcamCapture.width() - << ",\"height\":" << webcamCapture.height() - << ",\"fps\":" << webcamCapture.fps() - << ",\"deviceName\":\"" << jsonEscape(wideToUtf8(webcamCapture.selectedDeviceName())) - << "\"}" << std::endl; - // writeSeparateWebcam was decided above, before the pixel format. + // let capture continue without this camera's track. `code` stays + // ahead of `index`: older readers match on the `"code":…` substring. + std::cerr << "WARNING: Failed to initialize native webcam capture for camera " << index + << "; continuing without it" << std::endl; + std::cout << "{\"event\":\"warning\",\"code\":\"webcam-unavailable\",\"index\":" << index + << ",\"deviceName\":\"" << jsonEscape(stream->config.deviceName) + << "\",\"message\":\"Failed to initialize native webcam capture\"}" << std::endl; + continue; } + std::cout << "{\"event\":\"webcam-format\",\"schemaVersion\":2,\"index\":" << index + << ",\"width\":" << stream->capture.width() + << ",\"height\":" << stream->capture.height() + << ",\"fps\":" << stream->capture.fps() + << ",\"deviceName\":\"" << jsonEscape(wideToUtf8(stream->capture.selectedDeviceName())) + << "\"}" << std::endl; + stream->nominalIntervalHns = + static_cast(10'000'000ULL / std::max(1, stream->capture.fps())); + webcams.push_back(std::move(stream)); } + // Only the legacy single camera without a file of its own is composited + // into the screen frame; every listed camera writes separately. + WebcamStream* const inlineWebcam = + webcams.size() == 1 && !webcams[0]->writeSeparate ? webcams[0].get() : nullptr; WasapiLoopbackCapture loopbackCapture; WasapiLoopbackCapture microphoneCapture; @@ -805,13 +874,13 @@ int wmain(int argc, wchar_t* argv[]) { // // The other two conditions are unchanged and still required: software // encoding and inline webcam PiP both need the frame in system memory, - // which the DXGI path does not produce. config.webcamEnabled, not - // webcamActive -- the latter is only set once webcam capture has started, - // well after this. + // which the DXGI path does not produce. Decided from the initialized + // cameras, not from `active` -- that is only set once webcam capture has + // started, well after this. encoderOptions.useDxgiInput = readEnvInt("OPENSCREEN_WGC_ENABLE_DXGI_INPUT", 0) == 1 && !config.preferSoftwareEncoder && - (!config.webcamEnabled || writeSeparateWebcam); + inlineWebcam == nullptr; MFEncoder encoder; if (!encoder.initialize( @@ -850,8 +919,10 @@ int wmain(int argc, wchar_t* argv[]) { // on stop. << ",\"videoEncoderRuntime\":\"" << encoder.videoEncoderRuntime() << "\"}" << std::endl; - MFEncoder webcamEncoder; - if (writeSeparateWebcam) { + for (const auto& stream : webcams) { + if (!stream->writeSeparate) { + continue; + } MFEncoderOptions webcamEncoderOptions = encoderOptions; webcamEncoderOptions.injectDefaultSinkWriterFailureOnce = false; webcamEncoderOptions.useDxgiInput = false; @@ -860,23 +931,25 @@ int wmain(int argc, wchar_t* argv[]) { // above 640x480; now that the capture runs at the camera's real // resolution, 8 Mbit/s starves a 1440p or 2160p frame badly enough to // undo the extra pixels. The tiers mirror the screen ladder above. - const int webcamPixels = std::max(1, webcamCapture.width()) * std::max(1, webcamCapture.height()); + const int webcamPixels = + std::max(1, stream->capture.width()) * std::max(1, stream->capture.height()); const int webcamBitrate = webcamPixels >= 3840 * 2160 ? 40'000'000 : webcamPixels >= 2560 * 1440 ? 24'000'000 : webcamPixels >= 1920 * 1080 ? 16'000'000 : webcamPixels >= 1280 * 720 ? 8'000'000 : 4'000'000; - if (!webcamEncoder.initialize( - utf8ToWide(config.webcamOutputPath), - webcamCapture.width(), - webcamCapture.height(), - webcamCapture.fps(), + if (!stream->encoder.initialize( + utf8ToWide(stream->config.outputPath), + stream->capture.width(), + stream->capture.height(), + stream->capture.fps(), webcamBitrate, session.device(), session.context(), nullptr, webcamEncoderOptions)) { - std::cerr << "ERROR: Failed to initialize native webcam encoder" << std::endl; + std::cerr << "ERROR: Failed to initialize native webcam encoder for camera " + << stream->index << std::endl; return 1; } } @@ -903,11 +976,6 @@ int wmain(int argc, wchar_t* argv[]) { // them is the next bug report, and neither is worth a log line each. std::atomic contendedFrames = 0; Microsoft::WRL::ComPtr latestFrameTexture; - std::vector latestWebcamFrame; - int latestWebcamWidth = 0; - int latestWebcamHeight = 0; - uint64_t latestWebcamSequence = 0; - bool hasVisibleWebcamFrame = false; // Legacy-path-only state. frameMutex guards latestFrameTexture/ // legacyLatestFrameTimestampHns between WGC's callback thread (writer) @@ -962,7 +1030,6 @@ int wmain(int argc, wchar_t* argv[]) { std::chrono::duration(1.0 / config.fps)); uint64_t frameIndex = 0; int64_t lastEncodedVideoTimestampHns = -1; - int64_t lastWebcamTimestampHns = -1; // Media Foundation's H.264 encoder MFT does not honor irregular input // sample times for a VFR source: it numbers output samples // sequentially at its configured nominal frame rate regardless of the @@ -973,18 +1040,17 @@ int wmain(int argc, wchar_t* argv[]) { // webcam encoder on a real-time-paced cadence (duplicating the // latest available camera frame when the camera hasn't produced a // newer one yet), so "sample N is at N/fps" is actually correct. - int64_t nextWebcamWriteDueHns = 0; - const int64_t nominalWebcamIntervalHns = - static_cast(10'000'000ULL / std::max(1, webcamCapture.fps())); + // The cadence state is per camera: see WebcamStream. auto nextFrameDue = std::chrono::steady_clock::now(); int64_t firstFrameTimestampHns = -1; int64_t latestFrameTimestampHns = 0; while (!control.stopRequested && !encodeFailed) { Microsoft::WRL::ComPtr videoSample; - Microsoft::WRL::ComPtr webcamSample; bool hasVideoSample = false; - bool hasWebcamSample = false; + for (const auto& stream : webcams) { + stream->pendingSample.Reset(); + } // Whether the picture this tick encodes differs from the last one: // a new WGC frame, or a new camera frame drawn into it. The legacy // callback path cannot tell, so it always reads back. @@ -1061,22 +1127,20 @@ int wmain(int argc, wchar_t* argv[]) { continue; } } - if (webcamActive) { - WebcamFrameSnapshot candidateWebcamFrame; - if (webcamCapture.copyLatestFrame(candidateWebcamFrame, latestWebcamSequence) && - hasVisibleWebcamContent(candidateWebcamFrame.data, webcamCapture.deliversNv12())) { - latestWebcamFrame = std::move(candidateWebcamFrame.data); - latestWebcamWidth = candidateWebcamFrame.width; - latestWebcamHeight = candidateWebcamFrame.height; - latestWebcamSequence = candidateWebcamFrame.sequence; - hasVisibleWebcamFrame = true; - pictureChanged = pictureChanged || !writeSeparateWebcam; + // Screen first, then every camera, as before there was more than one. + for (const auto& stream : webcams) { + if (stream->active && stream->pullVisibleFrame()) { + pictureChanged = pictureChanged || !stream->writeSeparate; } } - const BgraFrameView webcamFrame{ - hasVisibleWebcamFrame && !latestWebcamFrame.empty() ? latestWebcamFrame.data() : nullptr, - latestWebcamWidth, - latestWebcamHeight, + // The frame composited into the screen picture, for the legacy + // inline camera only. + const BgraFrameView inlineWebcamFrame{ + inlineWebcam && inlineWebcam->hasVisibleFrame && !inlineWebcam->latestFrame.empty() + ? inlineWebcam->latestFrame.data() + : nullptr, + inlineWebcam ? inlineWebcam->latestWidth : 0, + inlineWebcam ? inlineWebcam->latestHeight : 0, }; const int64_t syntheticTimestampHns = static_cast((frameIndex * 10'000'000ULL) / config.fps); @@ -1094,16 +1158,24 @@ int wmain(int argc, wchar_t* argv[]) { frameTimestampHns = lastEncodedVideoTimestampHns + static_cast(10'000'000ULL / config.fps); } - if (writeSeparateWebcam && webcamFrame.data) { - // Anchor to the same recording-start origin as screen video/audio, - // using real elapsed host-clock time (not a synthetic frame-index - // clock) so a long recording can't accumulate clock-origin drift. - const auto elapsedSinceStart = std::chrono::steady_clock::now() - control.recordingStartedAt; - const int64_t elapsedHns = std::chrono::duration_cast< - std::chrono::duration>>(elapsedSinceStart) - .count(); - const int64_t targetElapsedHns = - std::max(0, elapsedHns - control.pausedDurationHns()); + // Anchor to the same recording-start origin as screen video/audio, + // using real elapsed host-clock time (not a synthetic frame-index + // clock) so a long recording can't accumulate clock-origin drift. + // Read once per tick: every camera of this tick shares it. + int64_t targetElapsedHns = -1; + for (const auto& stream : webcams) { + if (!stream->active || !stream->writeSeparate || !stream->hasVisibleFrame || + stream->latestFrame.empty()) { + continue; + } + if (targetElapsedHns < 0) { + const auto elapsedSinceStart = + std::chrono::steady_clock::now() - control.recordingStartedAt; + const int64_t elapsedHns = std::chrono::duration_cast< + std::chrono::duration>>(elapsedSinceStart) + .count(); + targetElapsedHns = std::max(0, elapsedHns - control.pausedDurationHns()); + } // The H.264 encoder MFT does not honor irregular per-sample // timestamps for a VFR source -- it numbers output samples // sequentially at its configured nominal rate regardless of the @@ -1112,35 +1184,40 @@ int wmain(int argc, wchar_t* argv[]) { // to feed the encoder *at* that nominal cadence, duplicating // the latest available camera frame when the camera hasn't // produced a newer one yet (VFR capture -> CFR encode resampling). - if (targetElapsedHns >= nextWebcamWriteDueHns) { - int64_t webcamTimestampHns = targetElapsedHns; - if (lastWebcamTimestampHns >= 0 && webcamTimestampHns <= lastWebcamTimestampHns) { - webcamTimestampHns = lastWebcamTimestampHns + nominalWebcamIntervalHns; - } - // Capture the sample here, but submit it to the sink - // writer OUTSIDE this block below (issue #115) so a - // slow WriteSample can't hold up the next frame pull. - hasWebcamSample = - webcamCapture.deliversNv12() - ? webcamEncoder.captureNv12Sample( - Nv12FrameView{ - webcamFrame.data, webcamFrame.width, webcamFrame.height}, - webcamTimestampHns, - webcamSample) - : webcamEncoder.captureBgraSample( - webcamFrame, webcamTimestampHns, webcamSample); - if (!hasWebcamSample) { - encodeFailed = true; - control.requestStop(); - break; - } - lastWebcamTimestampHns = webcamTimestampHns; - nextWebcamWriteDueHns += nominalWebcamIntervalHns; - if (nextWebcamWriteDueHns <= targetElapsedHns) { - // Fell behind (e.g. coming out of a pause, or a stall) -- - // resync to now instead of trying to catch up frame-by-frame. - nextWebcamWriteDueHns = targetElapsedHns + nominalWebcamIntervalHns; - } + if (targetElapsedHns < stream->nextWriteDueHns) { + continue; + } + int64_t webcamTimestampHns = targetElapsedHns; + if (stream->lastTimestampHns >= 0 && webcamTimestampHns <= stream->lastTimestampHns) { + webcamTimestampHns = stream->lastTimestampHns + stream->nominalIntervalHns; + } + const BgraFrameView webcamFrame{ + stream->latestFrame.data(), stream->latestWidth, stream->latestHeight}; + // Capture the sample here, but submit it to the sink + // writer OUTSIDE this block below (issue #115) so a + // slow WriteSample can't hold up the next frame pull. + const bool captured = + stream->capture.deliversNv12() + ? stream->encoder.captureNv12Sample( + Nv12FrameView{webcamFrame.data, webcamFrame.width, webcamFrame.height}, + webcamTimestampHns, + stream->pendingSample) + : stream->encoder.captureBgraSample( + webcamFrame, webcamTimestampHns, stream->pendingSample); + if (!captured) { + // One camera failing costs that camera, not the take. + std::cerr << "ERROR: Failed to capture a sample for camera " << stream->index + << "; disabling it" << std::endl; + stream->pendingSample.Reset(); + stream->active = false; + continue; + } + stream->lastTimestampHns = webcamTimestampHns; + stream->nextWriteDueHns += stream->nominalIntervalHns; + if (stream->nextWriteDueHns <= targetElapsedHns) { + // Fell behind (e.g. coming out of a pause, or a stall) -- + // resync to now instead of trying to catch up frame-by-frame. + stream->nextWriteDueHns = targetElapsedHns + stream->nominalIntervalHns; } } if (testStallReadbackMs > 0) { @@ -1178,7 +1255,7 @@ int wmain(int argc, wchar_t* argv[]) { captured = encoder.captureVideoSample( latestFrameTexture.Get(), frameTimestampHns, - !writeSeparateWebcam && webcamFrame.data ? &webcamFrame : nullptr, + inlineWebcamFrame.data ? &inlineWebcamFrame : nullptr, videoSample); } if (!captured) { @@ -1218,10 +1295,15 @@ int wmain(int argc, wchar_t* argv[]) { // Stop detection has nothing to do with this ordering -- that is // CaptureControl::stopMutex/stopCv, checked by the loop condition // above, unrelated to sample submission (issue #252). - if (hasWebcamSample && !webcamEncoder.submitVideoSample(webcamSample.Get())) { - encodeFailed = true; - control.requestStop(); - break; + for (const auto& stream : webcams) { + if (stream->pendingSample && !stream->encoder.submitVideoSample(stream->pendingSample.Get())) { + // Disables this camera only; the screen and the other + // cameras keep recording. + std::cerr << "ERROR: Failed to submit a sample for camera " << stream->index + << "; disabling it" << std::endl; + stream->active = false; + } + stream->pendingSample.Reset(); } if (hasVideoSample && !encoder.submitVideoSample(videoSample.Get())) { encodeFailed = true; @@ -1347,8 +1429,14 @@ int wmain(int argc, wchar_t* argv[]) { stopRenderKeepAliveIfActive(); return 1; } - if (config.webcamEnabled) { - if (!webcamCapture.start()) { + const auto stopWebcamCaptures = [&]() { + for (const auto& stream : webcams) { + stream->capture.stop(); + } + }; + for (const auto& stream : webcams) { + if (!stream->capture.start()) { + stopWebcamCaptures(); microphoneCapture.stop(); loopbackCapture.stop(); stopDeviceWatchIfActive(); @@ -1356,32 +1444,43 @@ int wmain(int argc, wchar_t* argv[]) { if (audioMixer) { audioMixer->stop(); } - std::cerr << "ERROR: Failed to start native webcam capture" << std::endl; + std::cerr << "ERROR: Failed to start native webcam capture for camera " << stream->index + << std::endl; return 1; } - webcamActive = true; + stream->active = true; + } + if (!webcams.empty()) { + // One 3 s budget for all cameras together, not 3 s each: they warm up + // in parallel, and the screen recording should not start later per camera. const auto webcamDeadline = std::chrono::steady_clock::now() + std::chrono::seconds(3); - while (std::chrono::steady_clock::now() < webcamDeadline && !hasVisibleWebcamFrame) { - WebcamFrameSnapshot candidateWebcamFrame; - if (webcamCapture.copyLatestFrame(candidateWebcamFrame, latestWebcamSequence) && - hasVisibleWebcamContent(candidateWebcamFrame.data, webcamCapture.deliversNv12())) { - latestWebcamFrame = std::move(candidateWebcamFrame.data); - latestWebcamWidth = candidateWebcamFrame.width; - latestWebcamHeight = candidateWebcamFrame.height; - latestWebcamSequence = candidateWebcamFrame.sequence; - hasVisibleWebcamFrame = true; + const auto allVisible = [&]() { + return std::all_of(webcams.begin(), webcams.end(), [](const auto& stream) { + return stream->hasVisibleFrame; + }); + }; + while (std::chrono::steady_clock::now() < webcamDeadline && !allVisible()) { + for (const auto& stream : webcams) { + if (!stream->hasVisibleFrame) { + stream->pullVisibleFrame(); + } + } + if (allVisible()) { break; } std::this_thread::sleep_for(std::chrono::milliseconds(20)); } - if (!hasVisibleWebcamFrame) { - std::cerr << "WARNING: Native webcam started but no visible frame was available before screen capture" - << std::endl; + for (const auto& stream : webcams) { + if (!stream->hasVisibleFrame) { + std::cerr << "WARNING: Native webcam " << stream->index + << " started but no visible frame was available before screen capture" + << std::endl; + } } } if (!session.start()) { - webcamCapture.stop(); + stopWebcamCaptures(); microphoneCapture.stop(); loopbackCapture.stop(); stopDeviceWatchIfActive(); @@ -1431,7 +1530,7 @@ int wmain(int argc, wchar_t* argv[]) { loopbackCapture.stop(); stopDeviceWatchIfActive(); stopRenderKeepAliveIfActive(); - webcamCapture.stop(); + stopWebcamCaptures(); if (audioMixer) { audioMixer->stop(); } @@ -1588,7 +1687,7 @@ int wmain(int argc, wchar_t* argv[]) { logStopStep("render-keepalive"); } beginStopStep("webcam", stepBudgetMs); - webcamCapture.stop(); + stopWebcamCaptures(); logStopStep("webcam"); beginStopStep("audio-mixer", stepBudgetMs); if (audioMixer) { @@ -1684,20 +1783,47 @@ int wmain(int argc, wchar_t* argv[]) { if (!encodeFailed && screenFinalized) { std::cout << "{\"event\":\"recording-stopped\",\"schemaVersion\":2,\"screenPath\":\"" << jsonEscape(config.outputPath) << "\""; - if (writeSeparateWebcam) { - std::cout << ",\"webcamPath\":\"" << jsonEscape(config.webcamOutputPath) << "\""; + // `webcamPath` stays for camera 0, as before; `webcamPaths` lists every + // camera still recording at stop, in index order. Both are printed + // before the camera files are finalized, for the reason above -- so they + // name the files that were being written, and a camera whose finalize + // fails below is reported on stderr, not removed from this list. + std::vector recordedWebcams; + for (const auto& stream : webcams) { + if (stream->writeSeparate && stream->active) { + recordedWebcams.push_back(stream.get()); + } + } + if (!recordedWebcams.empty() && recordedWebcams.front()->index == 0) { + std::cout << ",\"webcamPath\":\"" << jsonEscape(recordedWebcams.front()->config.outputPath) + << "\""; + } + if (!recordedWebcams.empty()) { + std::cout << ",\"webcamPaths\":["; + for (size_t i = 0; i < recordedWebcams.size(); ++i) { + std::cout << (i == 0 ? "\"" : ",\"") << jsonEscape(recordedWebcams[i]->config.outputPath) + << "\""; + } + std::cout << "]"; } std::cout << "}" << std::endl; std::cout << "Recording stopped. Output path: " << config.outputPath << std::endl; } + // Every camera that wrote a file is finalized, including one disabled + // mid-take: what it wrote before failing is still worth a playable index. bool webcamFinalized = true; - if (writeSeparateWebcam) { - beginStopStep("webcam-encoder-finalize", shutdownBudgetMs); - webcamFinalized = webcamEncoder.finalize(); - logStopStep("webcam-encoder-finalize"); - if (!webcamFinalized) { - std::cerr << "ERROR: Failed to finalize the webcam recording" << std::endl; + for (const auto& stream : webcams) { + if (!stream->writeSeparate) { + continue; + } + beginStopStep(stream->finalizeStepName.c_str(), shutdownBudgetMs); + const bool finalized = stream->encoder.finalize(); + logStopStep(stream->finalizeStepName.c_str()); + if (!finalized) { + std::cerr << "ERROR: Failed to finalize the webcam recording for camera " << stream->index + << std::endl; + webcamFinalized = false; } } diff --git a/electron/native/wgc-capture/src/mf_encoder.cpp b/electron/native/wgc-capture/src/mf_encoder.cpp index 9eac8c4b4..23bc69331 100644 --- a/electron/native/wgc-capture/src/mf_encoder.cpp +++ b/electron/native/wgc-capture/src/mf_encoder.cpp @@ -786,6 +786,7 @@ bool MFEncoder::initialize( if (!succeeded(MFStartup(MF_VERSION), "MFStartup")) { return false; } + mfStarted_ = true; if (useDxgiInput_ && !initializeDxgiPipeline()) { std::cerr << "WARNING: The GPU DXGI encode path is unavailable on this machine; " @@ -1984,6 +1985,9 @@ bool MFEncoder::finalize() { captureDevice_.Reset(); context_.Reset(); device_.Reset(); - MFShutdown(); + if (mfStarted_) { + MFShutdown(); + mfStarted_ = false; + } return ok; } diff --git a/electron/native/wgc-capture/src/mf_encoder.h b/electron/native/wgc-capture/src/mf_encoder.h index 455624135..4c2034f87 100644 --- a/electron/native/wgc-capture/src/mf_encoder.h +++ b/electron/native/wgc-capture/src/mf_encoder.h @@ -269,6 +269,12 @@ class MFEncoder { int64_t firstTimestampHns_ = -1; int64_t lastTimestampHns_ = -1; bool finalized_ = false; + // Whether initialize() got as far as a successful MFStartup(). finalize() + // may only balance a startup this encoder made: MF's startup count is + // process-wide, and an encoder that was never initialized (a dropped + // camera's) would otherwise shut Media Foundation down under every other + // encoder and camera still running. + bool mfStarted_ = false; bool useDxgiInput_ = false; const char* videoEncoderSelection_ = kVideoEncoderSelectionDefault; const char* videoEncoderRuntime_ = kVideoEncoderRuntimeUnknown; diff --git a/scripts/test-windows-wgc-helper.mjs b/scripts/test-windows-wgc-helper.mjs index 0ff318bc0..62e1d9fde 100644 --- a/scripts/test-windows-wgc-helper.mjs +++ b/scripts/test-windows-wgc-helper.mjs @@ -39,6 +39,15 @@ const WITH_WINDOW_POPUP = process.argv.includes("--window-popup"); const WITH_WEBCAM = process.env.OPENSCREEN_WGC_TEST_WEBCAM === "true" || process.argv.includes("--webcam"); +/** + * Adds a second camera that does not exist to a `--webcam` run, listed through + * the `webcams` config next to the real one: the helper must drop only that + * camera, say so with its index, and still record the first. + */ +const WITH_MISSING_SECOND_WEBCAM = + process.env.OPENSCREEN_WGC_TEST_MISSING_SECOND_WEBCAM === "true" || + process.argv.includes("--missing-second-webcam"); +const MISSING_WEBCAM_NAME = "OpenScreen Nonexistent Camera"; const CAPTURE_CURSOR = process.env.OPENSCREEN_WGC_TEST_CAPTURE_CURSOR === "true" || process.argv.includes("--capture-cursor"); @@ -107,6 +116,9 @@ const STOP_LATENCY_BUDGET_MS = 15_000; if (WITH_SOFTWARE_ENCODER && WITH_SOFTWARE_FALLBACK) { throw new Error("--software-encoder and --software-fallback are mutually exclusive"); } +if (WITH_MISSING_SECOND_WEBCAM && !WITH_WEBCAM) { + throw new Error("--missing-second-webcam needs --webcam"); +} function runHelper( config, @@ -860,6 +872,9 @@ const outputPath = path.join( `openscreen-wgc-helper-${WITH_WEBCAM ? "webcam" : WITH_WINDOW ? "window" : WITH_SYSTEM_AUDIO || WITH_MICROPHONE ? "audio" : "video"}-${process.pid}-${Date.now()}-${randomUUID()}.mp4`, ); const webcamOutputPath = WITH_WEBCAM ? outputPath.replace(/\.mp4$/i, "-webcam.mp4") : null; +const missingWebcamOutputPath = WITH_MISSING_SECOND_WEBCAM + ? outputPath.replace(/\.mp4$/i, "-webcam-2.mp4") + : null; const fixtureWindow = WITH_WINDOW ? await startFixtureWindow() : null; @@ -904,6 +919,29 @@ const config = { }, }; +if (WITH_MISSING_SECOND_WEBCAM) { + config.webcams = [ + { + camDeviceId: config.webcamDeviceId, + camDeviceName: config.webcamDeviceName, + camClsid: config.webcamDirectShowClsid, + camWidth: config.webcamWidth, + camHeight: config.webcamHeight, + camFps: config.webcamFps, + camPath: webcamOutputPath, + }, + { + camDeviceId: "", + camDeviceName: MISSING_WEBCAM_NAME, + camClsid: "", + camWidth: 1280, + camHeight: 720, + camFps: 30, + camPath: missingWebcamOutputPath, + }, + ]; +} + if (WITH_WINDOW_POPUP) { const scriptPath = path.join(os.tmpdir(), `openscreen-popup-fixture-${process.pid}.ps1`); fs.writeFileSync(scriptPath, POPUP_FIXTURE_SCRIPT); @@ -1111,6 +1149,35 @@ if ( if (WITH_WEBCAM && !webcamStreams.some((stream) => stream.codec_type === "video")) { throw new Error(`WGC helper webcam output has no video stream: ${webcamOutputPath}`); } +if (WITH_MISSING_SECOND_WEBCAM) { + const unavailable = result.stdout + .split(/\r?\n/) + .filter((line) => line.includes('"code":"webcam-unavailable"')) + .map((line) => JSON.parse(line.slice(line.indexOf('{"event"')))); + if (!unavailable.some((event) => event.index === 1 && event.deviceName === MISSING_WEBCAM_NAME)) { + throw new Error( + `WGC helper did not report camera 1 (${MISSING_WEBCAM_NAME}) as unavailable: ${result.stdout}`, + ); + } + const stoppedLine = result.stdout + .split(/\r?\n/) + .find((line) => line.includes('"event":"recording-stopped"')); + const stopped = stoppedLine + ? JSON.parse(stoppedLine.slice(stoppedLine.indexOf('{"event"'))) + : null; + if (JSON.stringify(stopped?.webcamPaths) !== JSON.stringify([webcamOutputPath])) { + throw new Error( + `recording-stopped.webcamPaths should list only the real camera: ${stoppedLine ?? "missing"}`, + ); + } + if (fs.existsSync(missingWebcamOutputPath) && fs.statSync(missingWebcamOutputPath).size > 0) { + throw new Error(`WGC helper wrote a file for the missing camera: ${missingWebcamOutputPath}`); + } + console.log("WGC helper missing-second-webcam check passed", { + unavailable, + webcamPaths: stopped.webcamPaths, + }); +} if ( (CAPTURE_CURSOR && !cursorCapture) || (cursorCapture && From e86b382ee5b362be7851fb2d6d718ab9836e43f1 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:38:05 +0200 Subject: [PATCH 03/22] fix(wgc): drop a camera whose encoder cannot start A camera whose encoder initialize() or capture start() fails is now warned about with the indexed webcam-unavailable event and dropped, instead of ending the take; its encoder is finalized and its empty file removed. The screen encoder's failure stays fatal. mf_encoder_color_test pins the MFEncoder fix: finalizing a never-initialized encoder must leave a live one able to write and finalize. --- electron/native/README.md | 2 +- electron/native/wgc-capture/src/main.cpp | 70 +++++++++++----- .../wgc-capture/src/mf_encoder_color_test.cpp | 79 +++++++++++++++++++ 3 files changed, 130 insertions(+), 21 deletions(-) diff --git a/electron/native/README.md b/electron/native/README.md index a66efbef8..316622866 100644 --- a/electron/native/README.md +++ b/electron/native/README.md @@ -94,7 +94,7 @@ Several cameras: a `webcams` list records up to four cameras, each into its own ] ``` -Every per-camera event carries the camera's `index` in that list: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. +Every per-camera event carries the camera's `index` in that list: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. Container: recordings are written as fragmented MP4 (`MFCreateFMPEG4MediaSink` + `MFCreateSinkWriterFromMediaSink`, `MF_MPEG4SINK_MIN_FRAGMENT_DURATION` = 1s) rather than plain MP4. A plain MP4 has no index until `IMFSinkWriter::Finalize()` writes `moov` at the very end, so when the shutdown watchdog force-exits a wedged helper the file on disk holds every frame and no way to read them — that is why issues #252 / #292 / #327 cost the whole recording rather than the frozen tail of it. A fragmented MP4 writes its index up front and its samples in self-describing `moof`+`mdat` pairs, so the same kill leaves a file that plays up to the last complete fragment. This does not fix the freeze; it removes the data loss the freeze causes. Because the fragmented sink needs both output media types at construction, the sink writer is built from a media sink instead of from a URL, and the helper reads the video/audio stream positions back off the sink rather than assuming them. If any of that is unavailable on a machine, the helper retries with the plain container and says so — `container` in the `encoder-selection` event is `fragmented-mp4` or `mp4`, and it reports what was used, not what was asked for. diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index a0785dab5..20e14382c 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -442,6 +442,13 @@ void reportCaptureAdapters(ID3D11Device* device, HMONITOR targetMonitor) { +// `code` stays ahead of `index`: older readers match on the `"code":…` substring. +void printWebcamUnavailable(size_t index, const std::string& deviceName, const char* message) { + std::cout << "{\"event\":\"warning\",\"code\":\"webcam-unavailable\",\"index\":" << index + << ",\"deviceName\":\"" << jsonEscape(deviceName) << "\",\"message\":\"" << message << "\"}" + << std::endl; +} + std::string parseWindowHandleFromSourceId(const std::string& sourceId) { constexpr char prefix[] = "window:"; if (sourceId.rfind(prefix, 0) != 0) { @@ -727,13 +734,11 @@ int wmain(int argc, wchar_t* argv[]) { // better than losing the whole recording because one camera device // didn't match. Report it so the renderer can inform the user (and, // historically, fall back to a browser-recorded webcam sidecar), but - // let capture continue without this camera's track. `code` stays - // ahead of `index`: older readers match on the `"code":…` substring. + // let capture continue without this camera's track. std::cerr << "WARNING: Failed to initialize native webcam capture for camera " << index << "; continuing without it" << std::endl; - std::cout << "{\"event\":\"warning\",\"code\":\"webcam-unavailable\",\"index\":" << index - << ",\"deviceName\":\"" << jsonEscape(stream->config.deviceName) - << "\",\"message\":\"Failed to initialize native webcam capture\"}" << std::endl; + printWebcamUnavailable( + index, stream->config.deviceName, "Failed to initialize native webcam capture"); continue; } std::cout << "{\"event\":\"webcam-format\",\"schemaVersion\":2,\"index\":" << index @@ -748,8 +753,38 @@ int wmain(int argc, wchar_t* argv[]) { } // Only the legacy single camera without a file of its own is composited // into the screen frame; every listed camera writes separately. - WebcamStream* const inlineWebcam = + WebcamStream* inlineWebcam = webcams.size() == 1 && !webcams[0]->writeSeparate ? webcams[0].get() : nullptr; + // A camera that opened but whose encoder or capture then would not start + // costs that camera, not the take -- with the screen and up to four cameras, + // running out of hardware encoder sessions is the likeliest way to get here. + // Only before the video writer runs: the vector is not touched after that. + const auto dropWebcam = [&](WebcamStream& stream, const char* message) { + std::cerr << "WARNING: " << message << " for camera " << stream.index << "; continuing without it" + << std::endl; + printWebcamUnavailable(stream.index, stream.config.deviceName, message); + stream.capture.stop(); + if (stream.writeSeparate) { + // Balances whatever initialize() got through, then removes the + // file it may have created: nothing was ever written to it. + stream.encoder.finalize(); + DeleteFileW(utf8ToWide(stream.config.outputPath).c_str()); + } + stream.active = false; + if (inlineWebcam == &stream) { + inlineWebcam = nullptr; + } + }; + const auto eraseDroppedWebcams = [&](const std::vector& dropped) { + webcams.erase( + std::remove_if( + webcams.begin(), + webcams.end(), + [&](const std::unique_ptr& stream) { + return std::find(dropped.begin(), dropped.end(), stream.get()) != dropped.end(); + }), + webcams.end()); + }; WasapiLoopbackCapture loopbackCapture; WasapiLoopbackCapture microphoneCapture; @@ -919,6 +954,7 @@ int wmain(int argc, wchar_t* argv[]) { // on stop. << ",\"videoEncoderRuntime\":\"" << encoder.videoEncoderRuntime() << "\"}" << std::endl; + std::vector droppedWebcams; for (const auto& stream : webcams) { if (!stream->writeSeparate) { continue; @@ -948,11 +984,11 @@ int wmain(int argc, wchar_t* argv[]) { session.context(), nullptr, webcamEncoderOptions)) { - std::cerr << "ERROR: Failed to initialize native webcam encoder for camera " - << stream->index << std::endl; - return 1; + dropWebcam(*stream, "Failed to initialize native webcam encoder"); + droppedWebcams.push_back(stream.get()); } } + eraseDroppedWebcams(droppedWebcams); // By default, no mutex guards frame handoff: writeVideoFrames is the // only thread that ever touches WGC or latestFrameTexture. It pulls each @@ -1434,22 +1470,16 @@ int wmain(int argc, wchar_t* argv[]) { stream->capture.stop(); } }; + droppedWebcams.clear(); for (const auto& stream : webcams) { if (!stream->capture.start()) { - stopWebcamCaptures(); - microphoneCapture.stop(); - loopbackCapture.stop(); - stopDeviceWatchIfActive(); - stopRenderKeepAliveIfActive(); - if (audioMixer) { - audioMixer->stop(); - } - std::cerr << "ERROR: Failed to start native webcam capture for camera " << stream->index - << std::endl; - return 1; + dropWebcam(*stream, "Failed to start native webcam capture"); + droppedWebcams.push_back(stream.get()); + continue; } stream->active = true; } + eraseDroppedWebcams(droppedWebcams); if (!webcams.empty()) { // One 3 s budget for all cameras together, not 3 s each: they warm up // in parallel, and the screen recording should not start later per camera. diff --git a/electron/native/wgc-capture/src/mf_encoder_color_test.cpp b/electron/native/wgc-capture/src/mf_encoder_color_test.cpp index 1cc20d6b0..ac33a5e1b 100644 --- a/electron/native/wgc-capture/src/mf_encoder_color_test.cpp +++ b/electron/native/wgc-capture/src/mf_encoder_color_test.cpp @@ -15,12 +15,14 @@ #include #include +#include #include #include #include #include #include #include +#include #include namespace { @@ -358,10 +360,87 @@ void checkRepeatedFrames(ID3D11Device* device, ID3D11DeviceContext* context) { } } +// An encoder that was never initialized -- a camera the helper dropped before +// its encoder was set up -- must not touch Media Foundation when it is +// finalized or destroyed. Its finalize() used to call an unmatched MFShutdown(), +// which in the helper left every other encoder writing nothing (Finalize: +// MF_E_SINK_NO_SAMPLES_PROCESSED). With an initialized encoder alive, as here, +// that takes MF's count to zero under a live sink writer, and the live encoder +// then blocks instead of failing -- so the whole sequence runs on its own thread +// with a deadline, and a regression fails this test instead of hanging the +// build. Needs no ffmpeg. +void checkUninitializedEncoderLeavesOthersRunning() { + Microsoft::WRL::ComPtr device; + Microsoft::WRL::ComPtr context; + if (FAILED(D3D11CreateDevice( + nullptr, D3D_DRIVER_TYPE_HARDWARE, nullptr, D3D11_CREATE_DEVICE_BGRA_SUPPORT, nullptr, 0, + D3D11_SDK_VERSION, &device, nullptr, &context)) && + FAILED(D3D11CreateDevice( + nullptr, D3D_DRIVER_TYPE_WARP, nullptr, D3D11_CREATE_DEVICE_BGRA_SUPPORT, nullptr, 0, + D3D11_SDK_VERSION, &device, nullptr, &context))) { + skip("uninitialized-encoder-leaves-mf-running", "no D3D11 device"); + return; + } + char tempDir[MAX_PATH]{}; + GetTempPathA(MAX_PATH, tempDir); + const std::string path = std::string(tempDir) + "openscreen-mf-encoder-uninitialized-peer.mp4"; + DeleteFileA(path.c_str()); + + // 0 running, 1 passed, 2 failed, 3 skipped. + std::atomic outcome = 0; + std::thread worker([&] { + CoInitializeEx(nullptr, COINIT_MULTITHREADED); + { + MFEncoder live; + if (!live.initialize( + widen(path), kWidth, kHeight, 30, 2'000'000, device.Get(), context.Get(), nullptr, {})) { + outcome = 3; + } else { + { + MFEncoder neverInitialized; + neverInitialized.finalize(); + } // and destroyed, which finalizes again + Microsoft::WRL::ComPtr probe; + std::vector bgra(static_cast(kWidth) * kHeight * 4, 0x80); + const BgraFrameView frame{bgra.data(), kWidth, kHeight}; + bool wrote = SUCCEEDED(MFCreateSample(&probe)); + for (int i = 0; i < kFrames && wrote; i += 1) { + Microsoft::WRL::ComPtr sample; + wrote = live.captureBgraSample(frame, static_cast(i) * 333'333, sample) && + live.submitVideoSample(sample.Get()); + } + outcome = wrote && live.finalize() ? 1 : 2; + } + } + CoUninitialize(); + }); + const auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(30); + while (outcome == 0 && std::chrono::steady_clock::now() < deadline) { + std::this_thread::sleep_for(std::chrono::milliseconds(20)); + } + if (outcome == 0) { + expect("uninitialized-encoder-leaves-mf-running", false, "the live encoder blocked in Media Foundation"); + std::cout << "ran " << g_ran << " tests\n" << g_failed << " failed" << std::endl; + // The blocked thread can be neither joined nor safely abandoned. + TerminateProcess(GetCurrentProcess(), 1); + } + worker.join(); + DeleteFileA(path.c_str()); + if (outcome == 3) { + skip("uninitialized-encoder-leaves-mf-running", "encoder initialize failed on this host"); + return; + } + expect("uninitialized-encoder-leaves-mf-running", outcome == 1, "write or finalize failed"); +} + } // namespace int main() { checkConverterAgainstReference(); + if (SUCCEEDED(CoInitializeEx(nullptr, COINIT_MULTITHREADED))) { + checkUninitializedEncoderLeavesOthersRunning(); + CoUninitialize(); + } if (!toolsAvailable()) { skip("mf-encoder-color", "ffprobe/ffmpeg not on PATH"); return g_failed == 0 ? 0 : 1; From bce2d6d82550fca4bf52962be7b195f2a69b6b8e Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:44:41 +0200 Subject: [PATCH 04/22] feat(recording): read per-camera helper events --- .../nativeWindowsCaptureStop.test.ts | 63 ++++++++++++++++ .../recording/nativeWindowsCaptureStop.ts | 72 +++++++++++++++++++ 2 files changed, 135 insertions(+) diff --git a/electron/recording/nativeWindowsCaptureStop.test.ts b/electron/recording/nativeWindowsCaptureStop.test.ts index e277eceec..6391569c2 100644 --- a/electron/recording/nativeWindowsCaptureStop.test.ts +++ b/electron/recording/nativeWindowsCaptureStop.test.ts @@ -9,7 +9,10 @@ import { readMicrophoneUnavailable, readSecondaryWindowsApplied, readStoppedPath, + readStoppedWebcamPaths, + readUnavailableWebcamIndices, readWebcamFormat, + readWebcamFormatAt, readWebcamUnavailable, terminateNativeWindowsCapture, waitForNativeWindowsCaptureStop, @@ -539,3 +542,63 @@ describe("terminateNativeWindowsCapture", () => { expect(helper.killCalls).toBe(1); }); }); + +describe("per-camera helper events", () => { + const unavailable = (i?: number) => + `{"event":"warning","code":"webcam-unavailable"${i === undefined ? "" : `,"index":${i}`},"message":"x"}`; + + it("collects unavailable cameras by index, an old event counting as camera 0", () => { + expect(readUnavailableWebcamIndices(`${unavailable(1)}\n${unavailable(3)}`)).toEqual([1, 3]); + expect(readUnavailableWebcamIndices(unavailable())).toEqual([0]); + expect(readUnavailableWebcamIndices("nothing")).toEqual([]); + }); + + it("ignores other warnings and reads a prefixed unavailable event", () => { + const other = '{"event":"warning","code":"microphone-defaulted","index":2}'; + const prefixed = `INFO: camera lost ${unavailable(2)}`; + expect(readUnavailableWebcamIndices(`${other}\n${prefixed}`)).toEqual([2]); + }); + + it("reads the format of a given camera", () => { + const out = [ + '{"event":"webcam-format","schemaVersion":2,"index":0,"width":1920,"height":1080,"fps":30,"deviceName":"A"}', + '{"event":"webcam-format","schemaVersion":2,"index":1,"width":1280,"height":720,"fps":30,"deviceName":"B"}', + ].join("\n"); + expect(readWebcamFormatAt(out, 1)?.width).toBe(1280); + expect(readWebcamFormatAt(out, 0)?.width).toBe(1920); + expect(readWebcamFormatAt(out, 2)).toBeNull(); + }); + + it("reads a format event glued to a diagnostic prefix, the last one winning", () => { + const out = [ + 'INFO: DirectShow webcam connected subtype NV12 {"event":"webcam-format","index":1,"width":640,"height":480}', + 'INFO: again {"event":"webcam-format","index":1,"width":1280,"height":720}', + ].join("\n"); + expect(readWebcamFormatAt(out, 1)?.width).toBe(1280); + }); + + it("treats a format without index as camera 0", () => { + expect(readWebcamFormatAt('{"event":"webcam-format","width":800}', 0)?.width).toBe(800); + }); + + it("reads every stopped camera path, falling back to the single legacy path", () => { + expect( + readStoppedWebcamPaths( + '{"event":"recording-stopped","webcamPath":"a.mp4","webcamPaths":["a.mp4","b.mp4"]}', + ), + ).toEqual(["a.mp4", "b.mp4"]); + expect(readStoppedWebcamPaths('{"event":"recording-stopped","webcamPath":"a.mp4"}')).toEqual([ + "a.mp4", + ]); + expect(readStoppedWebcamPaths('{"event":"recording-stopped"}')).toEqual([]); + }); + + it("reads JSON-escaped Windows paths behind a prefix", () => { + const line = + 'INFO: x {"event":"recording-stopped","schemaVersion":2,"screenPath":"C:\\\\Users\\\\me\\\\s.mp4","webcamPath":"C:\\\\Users\\\\me\\\\s-webcam.mp4","webcamPaths":["C:\\\\Users\\\\me\\\\s-webcam.mp4","C:\\\\Users\\\\me\\\\s-webcam-2.mp4"]}'; + expect(readStoppedWebcamPaths(line)).toEqual([ + "C:\\Users\\me\\s-webcam.mp4", + "C:\\Users\\me\\s-webcam-2.mp4", + ]); + }); +}); diff --git a/electron/recording/nativeWindowsCaptureStop.ts b/electron/recording/nativeWindowsCaptureStop.ts index 478d30e61..824cf4b3c 100644 --- a/electron/recording/nativeWindowsCaptureStop.ts +++ b/electron/recording/nativeWindowsCaptureStop.ts @@ -218,6 +218,78 @@ export function readWebcamFormat(output: string) { } } +type HelperEvent = Record; + +/** + * Every `{"event":""…}` object in the output, in order. + * + * Locates each object start and slices it with `findObjectEnd` instead of + * parsing whole lines, because diagnostics can be glued in front of an event + * (see `readWebcamFormat`). Slices that do not parse are skipped. + */ +function readHelperEvents(output: string, name: string): HelperEvent[] { + const needle = `{"event":"${name}"`; + const events: HelperEvent[] = []; + let from = 0; + for (;;) { + const start = output.indexOf(needle, from); + if (start === -1) { + return events; + } + const end = findObjectEnd(output, start); + if (end === -1) { + return events; + } + from = end + 1; + try { + const parsed: unknown = JSON.parse(output.slice(start, end + 1)); + if (parsed !== null && typeof parsed === "object" && !Array.isArray(parsed)) { + events.push(parsed as HelperEvent); + } + } catch { + // Not a complete JSON object; keep looking. + } + } +} + +/** A helper event's camera index; an event without one comes from an old helper (camera 0). */ +function eventCameraIndex(event: HelperEvent) { + return typeof event.index === "number" ? event.index : 0; +} + +/** Indices of every camera the helper reported as unavailable (no index = camera 0). */ +export function readUnavailableWebcamIndices(output: string): number[] { + return readHelperEvents(output, "warning") + .filter((event) => event.code === "webcam-unavailable") + .map(eventCameraIndex); +} + +/** The last `webcam-format` event of the given camera (no index = camera 0), or null. */ +export function readWebcamFormatAt( + output: string, + index: number, +): ReturnType { + const match = readHelperEvents(output, "webcam-format") + .filter((event) => eventCameraIndex(event) === index) + .at(-1); + return (match as ReturnType) ?? null; +} + +/** + * Camera files the helper reported at stop: `webcamPaths`, else the single + * legacy `webcamPath`, else none. + */ +export function readStoppedWebcamPaths(output: string): string[] { + const event = readHelperEvents(output, "recording-stopped").at(-1); + if (!event) { + return []; + } + if (Array.isArray(event.webcamPaths)) { + return event.webcamPaths.filter((entry): entry is string => typeof entry === "string"); + } + return typeof event.webcamPath === "string" && event.webcamPath ? [event.webcamPath] : []; +} + /** * The most useful line of a failed helper run, for a toast. * From c0b235a87625ea6d7cefd0f3bed34cd995b4c10e Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 11:52:58 +0200 Subject: [PATCH 05/22] feat(recording): keep additional cameras in the session and media links --- electron/electron-env.d.ts | 9 +-- electron/ipc/handlers.ts | 84 +++++++++++++++++++---- electron/media/mediaLinksRegistry.test.ts | 23 +++++++ electron/media/mediaLinksRegistry.ts | 18 ++++- src/lib/recordingSession.test.ts | 56 +++++++++++++++ src/lib/recordingSession.ts | 46 +++++++++++++ 6 files changed, 216 insertions(+), 20 deletions(-) create mode 100644 src/lib/recordingSession.test.ts diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index 26f2f5308..54e58e428 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -353,12 +353,9 @@ interface Window { /** Why this recording ended before it was stopped, when it did. */ warning?: string; }>; - findRecordingCamera: (videoPath: string) => Promise<{ - success: boolean; - webcamVideoPath?: string; - offsetMs?: number; - error?: string; - }>; + findRecordingCamera: ( + videoPath: string, + ) => Promise; readBinaryFile: (filePath: string) => Promise<{ success: boolean; data?: ArrayBuffer; diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 81e706c08..4e93d8d12 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -34,7 +34,9 @@ import { } from "../../src/lib/nativeMacRecording"; import type { NativeWindowsRecordingRequest } from "../../src/lib/nativeWindowsRecording"; import { + type AdditionalWebcam, type CursorCaptureMode, + type FindRecordingCameraResult, normalizeCursorCaptureMode, normalizeProjectMedia, normalizeRecordingSession, @@ -598,9 +600,25 @@ async function getApprovedProjectSession( throw new Error("Project references an invalid or unsupported webcam video path"); } - return webcamVideoPath - ? { screenVideoPath, webcamVideoPath, createdAt: Date.now() } - : { screenVideoPath, createdAt: Date.now() }; + // Additional cameras go through the same approval as camera 1; one that no + // longer resolves is dropped rather than failing the whole project. + const additionalWebcams: AdditionalWebcam[] = []; + for (const extra of media.additionalWebcams ?? []) { + const approved = await approveReadableVideoPath( + await resolveWithSiblingFallback(extra.path), + trustedDirs, + ); + if (approved) { + additionalWebcams.push({ path: approved, label: extra.label }); + } + } + + return { + screenVideoPath, + ...(webcamVideoPath ? { webcamVideoPath } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), + createdAt: Date.now(), + }; } type SelectedSource = { @@ -1341,6 +1359,7 @@ async function registerRecordingMediaLinks( options: { webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorCaptureMode?: CursorCaptureMode; }, ) { @@ -1355,6 +1374,9 @@ async function registerRecordingMediaLinks( ...(options.webcamVideoPath && Number.isFinite(options.webcamOffsetMs) ? { webcamOffsetMs: options.webcamOffsetMs } : {}), + ...(options.additionalWebcams?.length + ? { additionalWebcams: options.additionalWebcams } + : {}), ...(hasCursorTelemetry ? { cursorTelemetryPath } : {}), ...(options.cursorCaptureMode ? { cursorCaptureMode: options.cursorCaptureMode } : {}), }); @@ -1792,10 +1814,34 @@ async function loadRecordedSessionForVideoPath( } } + if (session.additionalWebcams) { + const approvedExtras: AdditionalWebcam[] = []; + for (const extra of session.additionalWebcams) { + let extraPath: string | null = extra.path; + if (!isPathAllowed(extraPath)) { + extraPath = await approveReadableVideoPath(extraPath, [ + path.dirname(manifestPath), + RECORDINGS_DIR, + ]); + } + if (extraPath) { + approvedExtras.push({ path: extraPath, label: extra.label }); + } + } + if (approvedExtras.length > 0) { + session.additionalWebcams = approvedExtras; + } else { + delete session.additionalWebcams; + } + } + approveFilePath(session.screenVideoPath); if (session.webcamVideoPath) { approveFilePath(session.webcamVideoPath); } + for (const extra of session.additionalWebcams ?? []) { + approveFilePath(extra.path); + } return session; } catch (error) { const nodeError = error as NodeJS.ErrnoException; @@ -1815,6 +1861,7 @@ async function loadRecordedSessionForVideoPath( async function resolveMediaLinksForVideo(videoPath: string): Promise<{ webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; resolvedVia: "sidecar" | "fingerprint" | "none"; }> { @@ -1825,6 +1872,7 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ .then(() => true) .catch(() => false); + const sessionAdditionalWebcams = session?.additionalWebcams ?? []; if (session?.webcamVideoPath || hasCursorTelemetry) { // Opportunistic backfill so the link survives a later move even if this // recording predates the registry, or if its sidecar doesn't travel with it. @@ -1833,6 +1881,9 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ ...(session?.webcamVideoPath && Number.isFinite(session.webcamOffsetMs) ? { webcamOffsetMs: session.webcamOffsetMs } : {}), + ...(sessionAdditionalWebcams.length > 0 + ? { additionalWebcams: sessionAdditionalWebcams } + : {}), ...(hasCursorTelemetry ? { cursorTelemetryPath } : {}), }).catch((error) => console.warn("[media-links] backfill failed:", error)); @@ -1843,6 +1894,9 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ webcamOffsetMs: session.webcamOffsetMs ?? 0, } : {}), + ...(sessionAdditionalWebcams.length > 0 + ? { additionalWebcams: sessionAdditionalWebcams } + : {}), ...(hasCursorTelemetry ? { cursorTelemetryPath } : {}), resolvedVia: "sidecar", }; @@ -1856,8 +1910,19 @@ async function resolveMediaLinksForVideo(videoPath: string): Promise<{ webcamVideoPath = (await approveReadableVideoPath(webcamVideoPath, [RECORDINGS_DIR])) ?? undefined; } + const additionalWebcams: AdditionalWebcam[] = []; + for (const extra of links.additionalWebcams ?? []) { + let extraPath: string | null = extra.path; + if (!isPathAllowed(extraPath)) { + extraPath = await approveReadableVideoPath(extraPath, [RECORDINGS_DIR]); + } + if (extraPath) { + additionalWebcams.push({ path: extraPath, label: extra.label }); + } + } return { ...(webcamVideoPath ? { webcamVideoPath, webcamOffsetMs: links.webcamOffsetMs ?? 0 } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), ...(links.cursorTelemetryPath ? { cursorTelemetryPath: links.cursorTelemetryPath } : {}), resolvedVia: "fingerprint", }; @@ -4727,15 +4792,7 @@ export function registerIpcHandlers( // `addAsset` in the new editor's project store. ipcMain.handle( "find-recording-camera", - async ( - _event, - videoPath: string, - ): Promise<{ - success: boolean; - webcamVideoPath?: string; - offsetMs?: number; - error?: string; - }> => { + async (_event, videoPath: string): Promise => { try { const normalized = normalizeVideoSourcePath(videoPath); if (!normalized || !isPathAllowed(normalized)) { @@ -4749,6 +4806,9 @@ export function registerIpcHandlers( success: true, webcamVideoPath: resolution.webcamVideoPath, offsetMs: resolution.webcamOffsetMs ?? 0, + ...(resolution.additionalWebcams?.length + ? { additionalWebcams: resolution.additionalWebcams } + : {}), }; } catch (err) { return { diff --git a/electron/media/mediaLinksRegistry.test.ts b/electron/media/mediaLinksRegistry.test.ts index 7c76a7580..0aae45269 100644 --- a/electron/media/mediaLinksRegistry.test.ts +++ b/electron/media/mediaLinksRegistry.test.ts @@ -187,6 +187,29 @@ describe("mediaLinksRegistry", () => { } }); + it("round-trips additional cameras and resolves old records without them", async () => { + const screenPath = path.join(tempDir, "multi.webm"); + const oldScreenPath = path.join(tempDir, "old.webm"); + await writeFileOfSize(screenPath, 5000, "m"); + await writeFileOfSize(oldScreenPath, 5000, "o"); + const additionalWebcams = [ + { path: path.join(tempDir, "multi-webcam-2.mp4"), label: "Desk" }, + { path: path.join(tempDir, "multi-webcam-3.mp4"), label: "" }, + ]; + await registerMediaLinks(tempDir, screenPath, { + webcamVideoPath: path.join(tempDir, "multi-webcam.mp4"), + additionalWebcams, + }); + await registerMediaLinks(tempDir, oldScreenPath, { + webcamVideoPath: path.join(tempDir, "old-webcam.mp4"), + }); + + const resolved = await findMediaLinksByFingerprint(tempDir, screenPath); + expect(resolved?.additionalWebcams).toEqual(additionalWebcams); + const old = await findMediaLinksByFingerprint(tempDir, oldScreenPath); + expect(old).not.toHaveProperty("additionalWebcams"); + }); + it("returns null when there is no matching fingerprint", async () => { const unknownPath = path.join(tempDir, "unknown.webm"); await writeFileOfSize(unknownPath, 1000, "z"); diff --git a/electron/media/mediaLinksRegistry.ts b/electron/media/mediaLinksRegistry.ts index ec0ad76de..69d119d51 100644 --- a/electron/media/mediaLinksRegistry.ts +++ b/electron/media/mediaLinksRegistry.ts @@ -19,7 +19,11 @@ import fs from "node:fs/promises"; import path from "node:path"; -import type { CursorCaptureMode } from "../../src/lib/recordingSession"; +import { + type AdditionalWebcam, + type CursorCaptureMode, + normalizeAdditionalWebcams, +} from "../../src/lib/recordingSession"; // ponytail: `baseDir` is passed in by every caller (RECORDINGS_DIR in // electron/ipc/handlers.ts) rather than imported here, so this module has no @@ -40,6 +44,7 @@ export interface MediaLinkEntry { fingerprint: MediaFingerprint; webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; cursorCaptureMode?: CursorCaptureMode; updatedAt: string; @@ -98,6 +103,7 @@ function normalizeEntry(candidate: unknown): MediaLinkEntry | null { if (!candidate || typeof candidate !== "object") return null; const raw = candidate as Partial; const fp = raw.fingerprint; + const additionalWebcams = normalizeAdditionalWebcams(raw.additionalWebcams); if ( typeof raw.lastKnownPath !== "string" || !fp || @@ -116,6 +122,7 @@ function normalizeEntry(candidate: unknown): MediaLinkEntry | null { }, ...(typeof raw.webcamVideoPath === "string" ? { webcamVideoPath: raw.webcamVideoPath } : {}), ...(typeof raw.webcamOffsetMs === "number" ? { webcamOffsetMs: raw.webcamOffsetMs } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), ...(typeof raw.cursorTelemetryPath === "string" ? { cursorTelemetryPath: raw.cursorTelemetryPath } : {}), @@ -220,6 +227,7 @@ async function updateRegistry( export interface MediaLinksToRegister { webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; cursorCaptureMode?: CursorCaptureMode; } @@ -236,6 +244,8 @@ export async function registerMediaLinks( links: MediaLinksToRegister, ): Promise { if (!links.webcamVideoPath && !links.cursorTelemetryPath) return; + const { additionalWebcams: rawAdditionalWebcams, ...linksWithoutAdditional } = links; + const additionalWebcams = normalizeAdditionalWebcams(rawAdditionalWebcams); const fingerprint = await computeFingerprint(videoPath); await updateRegistry(baseDir, (file) => { const existingIndex = file.entries.findIndex((e) => @@ -244,7 +254,8 @@ export async function registerMediaLinks( const entry: MediaLinkEntry = { lastKnownPath: videoPath, fingerprint, - ...links, + ...linksWithoutAdditional, + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), updatedAt: new Date().toISOString(), }; const entries = @@ -258,6 +269,7 @@ export async function registerMediaLinks( export interface MediaLinksLookup { webcamVideoPath?: string; webcamOffsetMs?: number; + additionalWebcams?: AdditionalWebcam[]; cursorTelemetryPath?: string; cursorCaptureMode?: CursorCaptureMode; } @@ -314,6 +326,7 @@ export async function findRelocatedMediaByStoredPath( screenVideoPath: match.lastKnownPath, ...(match.webcamVideoPath ? { webcamVideoPath: match.webcamVideoPath } : {}), ...(typeof match.webcamOffsetMs === "number" ? { webcamOffsetMs: match.webcamOffsetMs } : {}), + ...(match.additionalWebcams?.length ? { additionalWebcams: match.additionalWebcams } : {}), ...(match.cursorTelemetryPath ? { cursorTelemetryPath: match.cursorTelemetryPath } : {}), ...(match.cursorCaptureMode ? { cursorCaptureMode: match.cursorCaptureMode } : {}), }; @@ -358,6 +371,7 @@ export async function findMediaLinksByFingerprint( return { ...(match.webcamVideoPath ? { webcamVideoPath: match.webcamVideoPath } : {}), ...(typeof match.webcamOffsetMs === "number" ? { webcamOffsetMs: match.webcamOffsetMs } : {}), + ...(match.additionalWebcams?.length ? { additionalWebcams: match.additionalWebcams } : {}), ...(match.cursorTelemetryPath ? { cursorTelemetryPath: match.cursorTelemetryPath } : {}), ...(match.cursorCaptureMode ? { cursorCaptureMode: match.cursorCaptureMode } : {}), }; diff --git a/src/lib/recordingSession.test.ts b/src/lib/recordingSession.test.ts new file mode 100644 index 000000000..8ca842b0d --- /dev/null +++ b/src/lib/recordingSession.test.ts @@ -0,0 +1,56 @@ +import { describe, expect, it } from "vitest"; +import { normalizeProjectMedia, normalizeRecordingSession } from "./recordingSession"; + +describe("additionalWebcams", () => { + it("normalizes a session without additionalWebcams unchanged", () => { + expect(normalizeProjectMedia({ screenVideoPath: "/s.mp4", webcamVideoPath: "/w.mp4" })).toEqual( + { + screenVideoPath: "/s.mp4", + webcamVideoPath: "/w.mp4", + }, + ); + }); + it("keeps valid entries in order", () => { + const media = normalizeProjectMedia({ + screenVideoPath: "/s.mp4", + additionalWebcams: [ + { path: " /w-2.mp4 ", label: "Desk" }, + { path: "/w-3.mp4", label: "Side" }, + ], + }); + expect(media?.additionalWebcams).toEqual([ + { path: "/w-2.mp4", label: "Desk" }, + { path: "/w-3.mp4", label: "Side" }, + ]); + }); + it("drops invalid additionalWebcams entries and caps the list", () => { + const media = normalizeProjectMedia({ + screenVideoPath: "/s.mp4", + additionalWebcams: [ + { path: "", label: "x" }, + { label: "no path" }, + "junk", + { path: "/a.mp4" }, + { path: "/b.mp4", label: "B" }, + { path: "/c.mp4", label: "C" }, + { path: "/d.mp4", label: "D" }, + ], + }); + expect(media?.additionalWebcams).toEqual([ + { path: "/a.mp4", label: "" }, + { path: "/b.mp4", label: "B" }, + { path: "/c.mp4", label: "C" }, + ]); + expect( + normalizeProjectMedia({ screenVideoPath: "/s.mp4", additionalWebcams: [] }), + ).not.toHaveProperty("additionalWebcams"); + }); + it("survives a session round trip", () => { + const session = normalizeRecordingSession({ + screenVideoPath: "/s.mp4", + createdAt: 1, + additionalWebcams: [{ path: "/w-2.mp4", label: "Desk" }], + }); + expect(normalizeRecordingSession(JSON.parse(JSON.stringify(session)))).toEqual(session); + }); +}); diff --git a/src/lib/recordingSession.ts b/src/lib/recordingSession.ts index 5fd06bc25..2da7603af 100644 --- a/src/lib/recordingSession.ts +++ b/src/lib/recordingSession.ts @@ -1,3 +1,12 @@ +/** A recorded camera beyond camera 1 (which stays `webcamVideoPath`). */ +export interface AdditionalWebcam { + path: string; + label: string; +} + +/** Camera 1 plus at most this many additional cameras (4 in total). */ +export const MAX_ADDITIONAL_WEBCAMS = 3; + export interface ProjectMedia { screenVideoPath: string; webcamVideoPath?: string; @@ -10,6 +19,8 @@ export interface ProjectMedia { * that much extra leading footage instead of showing stale camera frames. */ webcamOffsetMs?: number; + /** Cameras 2-4, in recording order. Omitted when there are none. */ + additionalWebcams?: AdditionalWebcam[]; cursorCaptureMode?: CursorCaptureMode; } @@ -53,6 +64,29 @@ function normalizePath(value: unknown): string | undefined { return trimmed ? trimmed : undefined; } +export function normalizeAdditionalWebcams(value: unknown): AdditionalWebcam[] { + if (!Array.isArray(value)) { + return []; + } + + const result: AdditionalWebcam[] = []; + for (const entry of value) { + if (result.length >= MAX_ADDITIONAL_WEBCAMS) { + break; + } + if (!entry || typeof entry !== "object") { + continue; + } + const raw = entry as Partial; + const entryPath = normalizePath(raw.path); + if (!entryPath) { + continue; + } + result.push({ path: entryPath, label: typeof raw.label === "string" ? raw.label : "" }); + } + return result; +} + export function normalizeProjectMedia(candidate: unknown): ProjectMedia | null { if (!candidate || typeof candidate !== "object") { return null; @@ -66,6 +100,7 @@ export function normalizeProjectMedia(candidate: unknown): ProjectMedia | null { } const webcamVideoPath = normalizePath(raw.webcamVideoPath); + const additionalWebcams = normalizeAdditionalWebcams(raw.additionalWebcams); const cursorCaptureMode = normalizeCursorCaptureMode(raw.cursorCaptureMode); const webcamOffsetMs = typeof raw.webcamOffsetMs === "number" && Number.isFinite(raw.webcamOffsetMs) @@ -76,6 +111,7 @@ export function normalizeProjectMedia(candidate: unknown): ProjectMedia | null { screenVideoPath, ...(webcamVideoPath ? { webcamVideoPath } : {}), ...(webcamOffsetMs !== undefined ? { webcamOffsetMs } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), ...(cursorCaptureMode ? { cursorCaptureMode } : {}), }; } @@ -99,3 +135,13 @@ export function normalizeRecordingSession(candidate: unknown): RecordingSession : Date.now(), }; } + +/** Result of the `find-recording-camera` IPC, shared by the handler and the renderer typing. */ +export interface FindRecordingCameraResult { + success: boolean; + webcamVideoPath?: string; + offsetMs?: number; + /** Cameras 2-4 of the same recording, already approved for reading. */ + additionalWebcams?: AdditionalWebcam[]; + error?: string; +} From 1dead65f84c2a80cdc34ce06517b65d35f7d9d76 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:01:42 +0200 Subject: [PATCH 06/22] feat(recording): start and stop several cameras on Windows --- electron/electron-env.d.ts | 5 + electron/ipc/handlers.ts | 199 ++++++++++++----- .../recording/nativeWindowsWebcams.test.ts | 179 ++++++++++++++++ electron/recording/nativeWindowsWebcams.ts | 202 ++++++++++++++++++ src/lib/nativeWindowsRecording.ts | 10 + 5 files changed, 545 insertions(+), 50 deletions(-) create mode 100644 electron/recording/nativeWindowsWebcams.test.ts create mode 100644 electron/recording/nativeWindowsWebcams.ts diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index 54e58e428..f0ed35c2f 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -177,6 +177,11 @@ interface Window { * saved without it. Still a success — the screen video is intact. */ webcamDropped?: boolean; + /** + * Device names of additional cameras (2-4) that produced nothing usable + * and were left out of the session. Camera 1 is `webcamDropped`. + */ + droppedWebcams?: string[]; }>; pauseNativeWindowsRecording: () => Promise<{ success: boolean; diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 4e93d8d12..835a7a231 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -124,11 +124,21 @@ import { readMicrophoneDefaulted, readMicrophoneUnavailable, readSecondaryWindowsApplied, - readWebcamFormat, - readWebcamUnavailable, + readStoppedWebcamPaths, + readUnavailableWebcamIndices, + readWebcamFormatAt, terminateNativeWindowsCapture, waitForNativeWindowsCaptureStop, } from "../recording/nativeWindowsCaptureStop"; +import { + buildHelperWebcamConfig, + collectStoppedWebcams, + dedupeAdditionalWebcams, + isWebcamSidecarFile, + labelsOfUnavailableAdditionalWebcams, + stripWebcamSuffix, + webcamOutputPath, +} from "../recording/nativeWindowsWebcams"; import { patchWebmDurationOnDisk } from "../recording/webm-duration"; import { reindexRecordingOnDisk } from "../recording/webm-seek-index"; import { @@ -744,6 +754,11 @@ let nativeWindowsCaptureProcess: ChildProcessWithoutNullStreams | null = null; let nativeWindowsCaptureOutput = ""; let nativeWindowsCaptureTargetPath: string | null = null; let nativeWindowsCaptureWebcamTargetPath: string | null = null; +/** + * Cameras 2-4 of the running take, in helper order after camera 1, each with + * the label it is reported under. Only paths generated here ever land in it. + */ +let nativeWindowsCaptureAdditionalWebcamTargets: Array<{ path: string; label: string }> = []; let nativeWindowsCaptureRecordingId: number | null = null; let nativeWindowsCursorOffsetMs = 0; let nativeWindowsCursorCaptureMode: CursorCaptureMode = "editable-overlay"; @@ -770,6 +785,7 @@ function resetNativeWindowsCaptureState() { nativeWindowsCaptureProcess = null; nativeWindowsCaptureTargetPath = null; nativeWindowsCaptureWebcamTargetPath = null; + nativeWindowsCaptureAdditionalWebcamTargets = []; nativeWindowsCaptureRecordingId = null; nativeWindowsCursorOffsetMs = 0; nativeWindowsCursorCaptureMode = "editable-overlay"; @@ -797,12 +813,12 @@ async function salvageNativeWindowsFragmentedCapture(screenVideoPath: string | n */ async function removeNativeWindowsCaptureOutputs( screenVideoPath: string | null, - webcamVideoPath: string | null, + webcamVideoPaths: Array, options: { onlyIfUnusable?: boolean } = {}, ) { const targets = [ screenVideoPath, - webcamVideoPath, + ...webcamVideoPaths, screenVideoPath ? `${screenVideoPath}.cursor.json` : null, ]; @@ -1758,9 +1774,7 @@ function setCurrentRecordingSessionState(session: RecordingSession | null) { function getSessionManifestPathForVideo(videoPath: string) { const parsedPath = path.parse(videoPath); - const baseName = parsedPath.name.endsWith("-webcam") - ? parsedPath.name.slice(0, -"-webcam".length) - : parsedPath.name; + const baseName = stripWebcamSuffix(parsedPath.name); return path.join(parsedPath.dir, `${baseName}${RECORDING_SESSION_SUFFIX}`); } @@ -2856,10 +2870,7 @@ export function registerIpcHandlers( ? request.recordingId : Date.now(); const outputPath = path.join(RECORDINGS_DIR, `${RECORDING_FILE_PREFIX}${recordingId}.mp4`); - const webcamOutputPath = path.join( - RECORDINGS_DIR, - `${RECORDING_FILE_PREFIX}${recordingId}-webcam.mp4`, - ); + const webcamPath = webcamOutputPath(RECORDINGS_DIR, RECORDING_FILE_PREFIX, recordingId, 1); const sourceDisplay = request.source.type === "display" && typeof request.source.displayId === "number" ? (screen.getAllDisplays().find((display) => display.id === request.source.displayId) ?? @@ -2878,6 +2889,22 @@ export function registerIpcHandlers( const webcamDirectShowClsid = request.webcam.enabled ? await resolveDirectShowWebcamClsid(request.webcam.deviceName) : null; + // Cameras 2-4 only while camera 1 is on: its toggle governs every + // camera. Files are numbered from 2 in the order they are sent. + const additionalWebcams = request.webcam.enabled + ? await Promise.all( + dedupeAdditionalWebcams( + request.webcam, + Array.isArray(request.additionalWebcams) ? request.additionalWebcams : [], + ).map(async (extra, i) => ({ + deviceId: extra.deviceId, + deviceName: extra.deviceName, + label: extra.deviceName?.trim() || `Camera ${i + 2}`, + clsid: await resolveDirectShowWebcamClsid(extra.deviceName), + path: webcamOutputPath(RECORDINGS_DIR, RECORDING_FILE_PREFIX, recordingId, i + 2), + })), + ) + : []; const cursorCaptureMode = normalizeCursorCaptureMode(request.cursor?.mode) ?? "editable-overlay"; const envPreferSoftwareEncoder = (process.env.OPENSCREEN_WGC_PREFER_SOFTWARE_ENCODER ?? "") @@ -2909,13 +2936,12 @@ export function registerIpcHandlers( microphoneDeviceId: request.audio.microphone.deviceId ?? null, microphoneDeviceName: request.audio.microphone.deviceName ?? null, microphoneGain: request.audio.microphone.gain, - webcamEnabled: request.webcam.enabled, - webcamDeviceId: request.webcam.deviceId ?? null, - webcamDeviceName: request.webcam.deviceName ?? null, - webcamDirectShowClsid, - webcamWidth: request.webcam.width, - webcamHeight: request.webcam.height, - webcamFps: request.webcam.fps, + ...buildHelperWebcamConfig({ + camera1: request.webcam, + camera1Clsid: webcamDirectShowClsid, + camera1Path: webcamPath, + extras: additionalWebcams, + }), captureCursor: cursorCaptureMode === "system", cursorCaptureMode, hideDesktopIcons: @@ -2923,7 +2949,7 @@ export function registerIpcHandlers( appSettings.getSnapshot().recording.hideDesktopIcons, outputs: { screenPath: outputPath, - webcamPath: webcamOutputPath, + webcamPath, }, source: { type: request.source.type, @@ -2945,6 +2971,10 @@ export function registerIpcHandlers( source: request.source, audio: request.audio, webcam: request.webcam, + additionalWebcams: additionalWebcams.map(({ label, path: cameraPath }) => ({ + label, + path: cameraPath, + })), encoder: { preferSoftwareEncoder }, cursor: { mode: cursorCaptureMode }, // Both spaces, deliberately: the helper's own errors quote the physical @@ -2959,7 +2989,10 @@ export function registerIpcHandlers( await fs.mkdir(RECORDINGS_DIR, { recursive: true }); nativeWindowsCaptureOutput = ""; nativeWindowsCaptureTargetPath = outputPath; - nativeWindowsCaptureWebcamTargetPath = request.webcam.enabled ? webcamOutputPath : null; + nativeWindowsCaptureWebcamTargetPath = request.webcam.enabled ? webcamPath : null; + nativeWindowsCaptureAdditionalWebcamTargets = additionalWebcams.map( + ({ label, path: cameraPath }) => ({ label, path: cameraPath }), + ); nativeWindowsCaptureRecordingId = recordingId; nativeWindowsCursorOffsetMs = 0; nativeWindowsCursorCaptureMode = cursorCaptureMode; @@ -2995,7 +3028,13 @@ export function registerIpcHandlers( cursorCaptureMode === "editable-overlay" ? Math.max(0, captureStartedAtMs - cursorStartTimeMs) : 0; - const webcamFormat = readWebcamFormat(nativeWindowsCaptureOutput); + // Index-aware readers: with several cameras the helper prints one + // format line and possibly one unavailable warning per camera, and + // only index 0 (or no index, from an old helper) is camera 1. + const webcamFormat = readWebcamFormatAt(nativeWindowsCaptureOutput, 0); + const unavailableWebcamIndices = new Set( + readUnavailableWebcamIndices(nativeWindowsCaptureOutput), + ); const encoderSelection = readNativeWindowsEncoderSelection(nativeWindowsCaptureOutput); // Captured now because stop may have no helper left to ask. A helper // killed mid-recording is exactly the case where this matters most. @@ -3004,6 +3043,9 @@ export function registerIpcHandlers( captureStartedAtMs, cursorOffsetMs: nativeWindowsCursorOffsetMs, webcamFormat, + additionalWebcamFormats: nativeWindowsCaptureAdditionalWebcamTargets.map((_, i) => + readWebcamFormatAt(nativeWindowsCaptureOutput, i + 1), + ), encoderSelection, // Logged only: menus missing from a window take on Windows before 11 // 24H2 are a platform limit, not something to put in front of the user. @@ -3023,8 +3065,7 @@ export function registerIpcHandlers( // missing `webcamFormat`: absence of the format line also means "the // line could not be parsed", which would put a red toast on a recording // whose camera is working perfectly. - const webcamUnavailable = - request.webcam.enabled && readWebcamUnavailable(nativeWindowsCaptureOutput); + const webcamUnavailable = request.webcam.enabled && unavailableWebcamIndices.has(0); // Same shape as the camera notice: the helper records the Windows // default input rather than failing, so this take is usable but is // almost certainly the wrong microphone. @@ -3042,6 +3083,28 @@ export function registerIpcHandlers( deviceName: request.webcam.deviceName, }); } + // Helper indices follow the `webcams` list: camera 1 at 0, extras after. + const startedWebcams = [ + { path: webcamPath, label: request.webcam.deviceName ?? "" }, + ...nativeWindowsCaptureAdditionalWebcamTargets, + ]; + const unavailableWebcams = request.webcam.enabled + ? labelsOfUnavailableAdditionalWebcams(startedWebcams, [...unavailableWebcamIndices]) + : []; + if (unavailableWebcams.length > 0) { + console.warn( + "[native-wgc] recording without additional cameras the helper could not open", + { + unavailableWebcams, + }, + ); + // Already reported now; the helper deleted their files, so leaving + // them in the targets would report them a second time at stop. + nativeWindowsCaptureAdditionalWebcamTargets = + nativeWindowsCaptureAdditionalWebcamTargets.filter( + (_, i) => !unavailableWebcamIndices.has(i + 1), + ); + } return { success: true, @@ -3051,6 +3114,7 @@ export function registerIpcHandlers( videoEncoderSelection: encoderSelection?.video ?? null, videoEncoderRuntime: encoderSelection?.videoEncoderRuntime ?? null, webcamUnavailable, + ...(unavailableWebcams.length > 0 ? { unavailableWebcams } : {}), microphoneDefaulted, }; } catch (error) { @@ -3386,6 +3450,11 @@ export function registerIpcHandlers( const proc = nativeWindowsCaptureProcess; const preferredPath = nativeWindowsCaptureTargetPath; const preferredWebcamPath = nativeWindowsCaptureWebcamTargetPath; + const additionalWebcamTargets = nativeWindowsCaptureAdditionalWebcamTargets; + const allWebcamPaths = [ + preferredWebcamPath, + ...additionalWebcamTargets.map((target) => target.path), + ]; const recordingId = nativeWindowsCaptureRecordingId ?? Date.now(); const cursorCaptureMode = nativeWindowsCursorCaptureMode; @@ -3408,7 +3477,7 @@ export function registerIpcHandlers( if (!exited) { detachNativeWindowsCaptureOutputDrain(); } - await removeNativeWindowsCaptureOutputs(preferredPath, preferredWebcamPath); + await removeNativeWindowsCaptureOutputs(preferredPath, allWebcamPaths); return { success: true, discarded: true }; } finally { // Unconditional. Killing a wedged helper can itself throw, and @@ -3482,7 +3551,7 @@ export function registerIpcHandlers( // explain. Size-gate it anyway: throwing away a recording to tidy // up after a failed stop is the worse mistake of the two, and the // gate is the same one the salvage check above uses. - await removeNativeWindowsCaptureOutputs(preferredPath, preferredWebcamPath, { + await removeNativeWindowsCaptureOutputs(preferredPath, allWebcamPaths, { onlyIfUnusable: true, }); // The helper log goes to console/diagnostics above, not into this @@ -3518,31 +3587,56 @@ export function registerIpcHandlers( shiftPendingCursorTelemetry(nativeWindowsCursorOffsetMs); await writePendingCursorTelemetry(screenVideoPath); } - let webcamVideoPath: string | undefined; - if (preferredWebcamPath) { - try { - // Size, not just existence. A camera that opened but delivered no - // frame still gets a file created for it, and its `Finalize()` then - // fails, leaving nought bytes on disk. Admitting that file put a - // camera track in the document pointing at something no demuxer can - // read, and the preview compositor answers an unreadable camera by - // drawing the SCREEN recording inside the little camera rectangle — - // which is how a webcam that never recorded showed up as the desktop - // duplicated into its own corner (getopenscreen/openscreen#387). - const webcamStat = await fs.stat(preferredWebcamPath); - webcamVideoPath = webcamStat.size > 0 ? preferredWebcamPath : undefined; - if (!webcamVideoPath) { - console.warn("[native-wgc] the webcam file is empty; saving without a camera", { - path: preferredWebcamPath, - }); - } - } catch { - webcamVideoPath = undefined; + // Size, not just existence. A camera that opened but delivered no frame + // still gets a file created for it, and its `Finalize()` then fails, + // leaving nought bytes on disk. Admitting that file put a camera track in + // the document pointing at something no demuxer can read, and the preview + // compositor answers an unreadable camera by drawing the SCREEN recording + // inside the little camera rectangle — which is how a webcam that never + // recorded showed up as the desktop duplicated into its own corner + // (getopenscreen/openscreen#387). Every camera is judged by its own file, + // not by the helper's list at stop (see `collectStoppedWebcams`). + const requestedWebcams = [ + ...(preferredWebcamPath ? [{ path: preferredWebcamPath, label: "" }] : []), + ...additionalWebcamTargets, + ]; + const webcamSizes = new Map(); + for (const camera of requestedWebcams) { + const stat = await fs.stat(camera.path).catch(() => null); + if (stat) { + webcamSizes.set(camera.path, stat.size); } } - const session: RecordingSession = webcamVideoPath - ? { screenVideoPath, webcamVideoPath, createdAt: recordingId, cursorCaptureMode } - : { screenVideoPath, createdAt: recordingId, cursorCaptureMode }; + const stoppedWebcams = collectStoppedWebcams({ + camera1Enabled: Boolean(preferredWebcamPath), + requested: requestedWebcams, + sizes: webcamSizes, + }); + const webcamVideoPath = stoppedWebcams.camera1; + const additionalWebcams = stoppedWebcams.additional; + if (preferredWebcamPath && !webcamVideoPath && webcamSizes.has(preferredWebcamPath)) { + console.warn("[native-wgc] the webcam file is empty; saving without a camera", { + path: preferredWebcamPath, + }); + } + if (stoppedWebcams.dropped.length > 0) { + console.warn("[native-wgc] additional cameras produced nothing usable", { + dropped: stoppedWebcams.dropped, + helperWebcamPaths: readStoppedWebcamPaths(nativeWindowsCaptureOutput), + }); + } + // Generated by the start handler, never taken from the renderer or the + // helper; approved like the session's other media. + for (const extra of additionalWebcams) { + approveFilePath(extra.path); + } + const session: RecordingSession = { + screenVideoPath, + ...(webcamVideoPath ? { webcamVideoPath } : {}), + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), + createdAt: recordingId, + cursorCaptureMode, + }; setCurrentRecordingSessionState(session); currentProjectPath = null; @@ -3551,7 +3645,11 @@ export function registerIpcHandlers( `${path.parse(screenVideoPath).name}${RECORDING_SESSION_SUFFIX}`, ); await fs.writeFile(sessionManifestPath, JSON.stringify(session, null, 2), "utf-8"); - await registerRecordingMediaLinks(screenVideoPath, { webcamVideoPath, cursorCaptureMode }); + await registerRecordingMediaLinks(screenVideoPath, { + webcamVideoPath, + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), + cursorCaptureMode, + }); return { success: true, @@ -3566,6 +3664,7 @@ export function registerIpcHandlers( // unreported, the user would find out in the editor — which is exactly // the silence this change exists to end. webcamDropped: Boolean(preferredWebcamPath) && !webcamVideoPath, + ...(stoppedWebcams.dropped.length > 0 ? { droppedWebcams: stoppedWebcams.dropped } : {}), message: recovered ? "Native Windows recording recovered from a failed stop" : "Native Windows recording session stored successfully", @@ -4034,7 +4133,7 @@ export function registerIpcHandlers( const files = await fs.readdir(RECORDINGS_DIR); const videoFiles = files.filter( - (file) => file.endsWith(".webm") && !file.endsWith("-webcam.webm"), + (file) => file.endsWith(".webm") && !isWebcamSidecarFile(file), ); if (videoFiles.length === 0) { diff --git a/electron/recording/nativeWindowsWebcams.test.ts b/electron/recording/nativeWindowsWebcams.test.ts new file mode 100644 index 000000000..bb4e7c039 --- /dev/null +++ b/electron/recording/nativeWindowsWebcams.test.ts @@ -0,0 +1,179 @@ +import { describe, expect, it } from "vitest"; +import { + buildHelperWebcamConfig, + collectStoppedWebcams, + dedupeAdditionalWebcams, + isWebcamSidecarFile, + labelsOfUnavailableAdditionalWebcams, + stripWebcamSuffix, + webcamOutputPath, +} from "./nativeWindowsWebcams"; + +describe("nativeWindowsWebcams", () => { + it("names camera files", () => { + expect(webcamOutputPath("C:\\r", "rec-", 7, 1)).toMatch(/rec-7-webcam\.mp4$/); + expect(webcamOutputPath("C:\\r", "rec-", 7, 3)).toMatch(/rec-7-webcam-3\.mp4$/); + }); + + it("recognizes every camera file of a recording and nothing else", () => { + for (const f of ["rec-7-webcam.mp4", "rec-7-webcam-2.mp4", "rec-7-webcam-4.webm"]) { + expect(isWebcamSidecarFile(f)).toBe(true); + } + for (const f of ["rec-7.mp4", "rec-7-webcamera.mp4", "rec-7-webcam-x.mp4"]) { + expect(isWebcamSidecarFile(f)).toBe(false); + } + expect(stripWebcamSuffix("rec-7-webcam-2")).toBe("rec-7"); + expect(stripWebcamSuffix("rec-7-webcam")).toBe("rec-7"); + expect(stripWebcamSuffix("rec-7")).toBe("rec-7"); + }); + + it("dedupes a device that equals camera 1, duplicates, and caps at three", () => { + const extras = [ + { deviceId: "a", deviceName: "Front" }, + { deviceId: "b", deviceName: "Desk" }, + { deviceId: "b", deviceName: "Desk" }, + { deviceName: "Side" }, + { deviceName: "Top" }, + { deviceName: "Fifth" }, + ]; + expect( + dedupeAdditionalWebcams({ deviceId: "a", deviceName: "Front" }, extras).map( + (e) => e.deviceName, + ), + ).toEqual(["Desk", "Side", "Top"]); + }); + + it("drops extras that name no device at all", () => { + expect(dedupeAdditionalWebcams(null, [{ deviceName: " " }, { deviceName: "Desk" }])).toEqual([ + { deviceName: "Desk" }, + ]); + }); + + it("skips malformed entries from IPC", () => { + const junk = [ + null, + { deviceName: 3 }, + { deviceId: 4, deviceName: "X" }, + { deviceName: "Desk" }, + ]; + expect(dedupeAdditionalWebcams(null, junk as unknown as Array<{ deviceName: string }>)).toEqual( + [{ deviceName: "Desk" }], + ); + }); + + it("drops an empty additional camera file and names it", () => { + const r = collectStoppedWebcams({ + camera1Enabled: true, + requested: [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + { path: "w-3.mp4", label: "Side" }, + ], + sizes: new Map([ + ["w.mp4", 100], + ["w-2.mp4", 0], + ["w-3.mp4", 50], + ]), + }); + expect(r).toEqual({ + camera1: "w.mp4", + additional: [{ path: "w-3.mp4", label: "Side" }], + dropped: ["Desk"], + }); + }); + + it("keeps camera 1 and reports a camera whose file never appeared as not recorded", () => { + const r = collectStoppedWebcams({ + camera1Enabled: true, + requested: [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + ], + sizes: new Map([["w.mp4", 100]]), + }); + expect(r).toEqual({ camera1: "w.mp4", additional: [], dropped: ["Desk"] }); + }); + + it("keeps extras when camera 1 is lost, leaving camera 1 out of dropped", () => { + const r = collectStoppedWebcams({ + camera1Enabled: true, + requested: [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + ], + sizes: new Map([ + ["w.mp4", 0], + ["w-2.mp4", 10], + ]), + }); + expect(r).toEqual({ additional: [{ path: "w-2.mp4", label: "Desk" }], dropped: [] }); + }); + + it("maps unavailable helper indices to the labels of the extras", () => { + const requested = [ + { path: "w.mp4", label: "Front" }, + { path: "w-2.mp4", label: "Desk" }, + { path: "w-3.mp4", label: "Side" }, + ]; + expect(labelsOfUnavailableAdditionalWebcams(requested, [0, 2, 2, 9])).toEqual(["Side"]); + }); + + it("builds a start config with the legacy fields and a list of every camera", () => { + const config = buildHelperWebcamConfig({ + camera1: { + enabled: true, + deviceId: "id-1", + deviceName: "Front", + width: 1280, + height: 720, + fps: 30, + }, + camera1Clsid: "{c1}", + camera1Path: "C:\\r\\rec-7-webcam.mp4", + extras: [ + { deviceId: "id-2", deviceName: "Desk", clsid: null, path: "C:\\r\\rec-7-webcam-2.mp4" }, + ], + }); + const parsed = JSON.parse(JSON.stringify(config)); + expect(parsed).toMatchObject({ + webcamEnabled: true, + webcamDeviceId: "id-1", + webcamDeviceName: "Front", + webcamDirectShowClsid: "{c1}", + webcamWidth: 1280, + webcamHeight: 720, + webcamFps: 30, + }); + expect(parsed.webcams).toEqual([ + { + camDeviceId: "id-1", + camDeviceName: "Front", + camClsid: "{c1}", + camWidth: 1280, + camHeight: 720, + camFps: 30, + camPath: "C:\\r\\rec-7-webcam.mp4", + }, + { + camDeviceId: "id-2", + camDeviceName: "Desk", + camClsid: null, + camWidth: 1280, + camHeight: 720, + camFps: 30, + camPath: "C:\\r\\rec-7-webcam-2.mp4", + }, + ]); + }); + + it("sends no camera list while camera 1 is off", () => { + const config = buildHelperWebcamConfig({ + camera1: { enabled: false, width: 1280, height: 720, fps: 30 }, + camera1Clsid: null, + camera1Path: "C:\\r\\rec-7-webcam.mp4", + extras: [{ deviceName: "Desk", clsid: null, path: "C:\\r\\rec-7-webcam-2.mp4" }], + }); + expect(config.webcamEnabled).toBe(false); + expect(config.webcams).toEqual([]); + }); +}); diff --git a/electron/recording/nativeWindowsWebcams.ts b/electron/recording/nativeWindowsWebcams.ts new file mode 100644 index 000000000..42e87365a --- /dev/null +++ b/electron/recording/nativeWindowsWebcams.ts @@ -0,0 +1,202 @@ +/** + * Pure helpers for recording several cameras with the native Windows helper. + * + * Camera 1 stays exactly what it always was: the helper's legacy `webcam*` + * fields, the file `-webcam.mp4` and the session's + * `webcamVideoPath`. Cameras 2-4 ride along in the helper's `webcams` list and + * in the session's `additionalWebcams`. Kept out of `electron/ipc/handlers.ts` + * because that module cannot be loaded from a test. + */ +import path from "node:path"; +import { type AdditionalWebcam, MAX_ADDITIONAL_WEBCAMS } from "../../src/lib/recordingSession"; + +/** `-webcam` (camera 1) or `-webcam-<2..9>` at the end of a file's base name. */ +const WEBCAM_SUFFIX = /-webcam(?:-[2-9])?$/; + +/** Camera 1 → `-webcam.mp4`, camera n ≥ 2 → `-webcam-.mp4`. */ +export function webcamOutputPath( + dir: string, + prefix: string, + recordingId: number, + cameraNumber: number, +): string { + const suffix = cameraNumber <= 1 ? "-webcam" : `-webcam-${cameraNumber}`; + return path.join(dir, `${prefix}${recordingId}${suffix}.mp4`); +} + +/** Whether a file name is one of a recording's camera files (any extension). */ +export function isWebcamSidecarFile(fileName: string): boolean { + return WEBCAM_SUFFIX.test(path.parse(fileName).name); +} + +/** The recording's base name for a camera file's base name; other names pass through. */ +export function stripWebcamSuffix(baseName: string): string { + return baseName.replace(WEBCAM_SUFFIX, ""); +} + +/** One camera in the helper config's `webcams` list (keys are the helper's). */ +export interface HelperWebcamEntry { + camDeviceId: string | null; + camDeviceName: string; + camClsid: string | null; + camWidth: number; + camHeight: number; + camFps: number; + camPath: string; +} + +type DeviceRef = { deviceId?: string; deviceName?: string }; + +/** Same device: by id when both sides carry one, otherwise by name. */ +function isSameDevice(a: DeviceRef, b: DeviceRef) { + if (a.deviceId && b.deviceId) { + return a.deviceId === b.deviceId; + } + const name = a.deviceName?.trim(); + return Boolean(name) && name === b.deviceName?.trim(); +} + +/** + * The additional cameras worth asking the helper for: no malformed entry, no + * entry that names no device, none equal to camera 1, no duplicates, and at most + * {@link MAX_ADDITIONAL_WEBCAMS}. + */ +export function dedupeAdditionalWebcams( + camera1: DeviceRef | null, + extras: T[], +): T[] { + const kept: T[] = []; + for (const extra of extras) { + if (kept.length >= MAX_ADDITIONAL_WEBCAMS) { + break; + } + // The list crosses IPC, so a malformed entry is skipped rather than trusted. + if ( + !extra || + typeof extra.deviceName !== "string" || + (extra.deviceId !== undefined && typeof extra.deviceId !== "string") + ) { + continue; + } + if (!extra.deviceId && !extra.deviceName.trim()) { + continue; + } + if (camera1 && isSameDevice(camera1, extra)) { + continue; + } + if (kept.some((other) => isSameDevice(other, extra))) { + continue; + } + kept.push(extra); + } + return kept; +} + +/** + * The camera part of the helper config: the unchanged legacy `webcam*` fields + * for camera 1 plus the `webcams` list (camera 1 first, then the extras). + * + * Extras are recorded only while camera 1 is on (the HUD's camera toggle + * governs every camera), so with camera 1 off the list is empty. Every extra + * uses camera 1's requested size and rate: the quality setting is one for all. + */ +export function buildHelperWebcamConfig(input: { + camera1: { + enabled: boolean; + deviceId?: string; + deviceName?: string; + width: number; + height: number; + fps: number; + }; + camera1Clsid: string | null; + camera1Path: string; + extras: Array<{ deviceId?: string; deviceName: string; clsid: string | null; path: string }>; +}) { + const { camera1 } = input; + const entry = ( + device: { deviceId?: string; deviceName?: string }, + clsid: string | null, + camPath: string, + ): HelperWebcamEntry => ({ + camDeviceId: device.deviceId ?? null, + camDeviceName: device.deviceName ?? "", + camClsid: clsid, + camWidth: camera1.width, + camHeight: camera1.height, + camFps: camera1.fps, + camPath, + }); + const webcams: HelperWebcamEntry[] = camera1.enabled + ? [ + entry(camera1, input.camera1Clsid, input.camera1Path), + ...input.extras.map((extra) => entry(extra, extra.clsid, extra.path)), + ] + : []; + return { + webcamEnabled: camera1.enabled, + webcamDeviceId: camera1.deviceId ?? null, + webcamDeviceName: camera1.deviceName ?? null, + webcamDirectShowClsid: input.camera1Clsid, + webcamWidth: camera1.width, + webcamHeight: camera1.height, + webcamFps: camera1.fps, + webcams, + }; +} + +/** + * Labels of the additional cameras the helper reported unavailable. The + * helper's index is the position in the `webcams` list, which is the position + * in `requested` (camera 1 first). Index 0 is camera 1 and is left out: the + * caller reports camera 1 through its own `webcamUnavailable` flag. The labels + * come from our own list because the helper's `deviceName` can be empty. + */ +export function labelsOfUnavailableAdditionalWebcams( + requested: Array<{ label: string }>, + indices: number[], +): string[] { + const labels: string[] = []; + for (const index of new Set(indices)) { + const camera = index > 0 ? requested[index] : undefined; + if (camera) { + labels.push(camera.label); + } + } + return labels; +} + +/** + * Which requested cameras made it into the take. + * + * A camera is kept when its file exists with size > 0. The helper's own list + * at stop is deliberately not consulted: it names the cameras still recording + * at that moment, so a camera that died mid-take with a playable partial file + * would be missing from it, and an old helper names only camera 1. A file that + * never appeared is simply absent from `sizes`. + * + * `camera1Enabled` says whether `requested[0]` is camera 1. Camera 1 comes back + * as `camera1` (undefined when lost) and is never in `dropped`, which names + * only the additional cameras not kept — the caller reports camera 1 through + * its own `webcamDropped` flag. + */ +export function collectStoppedWebcams(input: { + camera1Enabled: boolean; + requested: Array<{ path: string; label: string }>; + sizes: Map; +}): { camera1?: string; additional: AdditionalWebcam[]; dropped: string[] } { + const isKept = (filePath: string) => (input.sizes.get(filePath) ?? 0) > 0; + const [first, ...rest] = input.requested; + const camera1 = input.camera1Enabled && first && isKept(first.path) ? first.path : undefined; + const extras = input.camera1Enabled ? rest : input.requested; + const additional: AdditionalWebcam[] = []; + const dropped: string[] = []; + for (const camera of extras) { + if (isKept(camera.path)) { + additional.push({ path: camera.path, label: camera.label }); + } else { + dropped.push(camera.label); + } + } + return { ...(camera1 ? { camera1 } : {}), additional, dropped }; +} diff --git a/src/lib/nativeWindowsRecording.ts b/src/lib/nativeWindowsRecording.ts index 5d5d92b31..c014cf243 100644 --- a/src/lib/nativeWindowsRecording.ts +++ b/src/lib/nativeWindowsRecording.ts @@ -34,6 +34,11 @@ export type NativeWindowsRecordingRequest = { height: number; fps: number; }; + /** + * Cameras 2-4, recorded at camera 1's size and rate. Only sent while camera 1 + * is enabled; the main process drops duplicates of camera 1 and caps the list. + */ + additionalWebcams?: Array<{ deviceId?: string; deviceName: string }>; cursor: { mode: import("./recordingSession").CursorCaptureMode; }; @@ -60,6 +65,11 @@ export type NativeWindowsRecordingStartResult = { * but the user has to be told, or they discover it in the editor. */ webcamUnavailable?: boolean; + /** + * Device names of additional cameras (2-4) the helper could not open; this + * take records without them. Camera 1 is reported by `webcamUnavailable`. + */ + unavailableWebcams?: string[]; /** * A microphone was asked for but could not be named, so the helper captured * whatever Windows calls the default input. The take is fine; the voice on it From 9665e7be0815f8d5bb7ef2ddf278ee03aef88ebd Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:19:19 +0200 Subject: [PATCH 07/22] fix(wgc): never open the same camera twice in one take Two webcams of the same model report the same name, and the browser id never matches a device path, so every such camera selected the first device and the second open failed as busy. The take now owns a claim set: each camera that opens adds its device (MF symbolic link or DirectShow DevicePath, normalized so both paths agree), and later cameras pick the best unclaimed match. The selection rule lives in device_selection.{h,cpp} with its own unit test. --- electron/native/wgc-capture/CMakeLists.txt | 16 ++ .../wgc-capture/src/device_selection.cpp | 158 +++++++++++++ .../native/wgc-capture/src/device_selection.h | 89 ++++++++ .../wgc-capture/src/device_selection_test.cpp | 100 +++++++++ .../wgc-capture/src/dshow_webcam_capture.cpp | 85 ++++++- .../wgc-capture/src/dshow_webcam_capture.h | 11 +- electron/native/wgc-capture/src/main.cpp | 7 +- .../native/wgc-capture/src/webcam_capture.cpp | 210 +++++------------- .../native/wgc-capture/src/webcam_capture.h | 20 +- scripts/build-windows-wgc-helper.mjs | 8 + 10 files changed, 539 insertions(+), 165 deletions(-) create mode 100644 electron/native/wgc-capture/src/device_selection.cpp create mode 100644 electron/native/wgc-capture/src/device_selection.h create mode 100644 electron/native/wgc-capture/src/device_selection_test.cpp diff --git a/electron/native/wgc-capture/CMakeLists.txt b/electron/native/wgc-capture/CMakeLists.txt index 84aa89853..870f4003a 100644 --- a/electron/native/wgc-capture/CMakeLists.txt +++ b/electron/native/wgc-capture/CMakeLists.txt @@ -39,6 +39,8 @@ add_executable(wgc-capture src/audio_sample_utils.cpp src/audio_sample_utils.h src/desktop_icon_cover.cpp + src/device_selection.cpp + src/device_selection.h src/desktop_icon_cover.h src/dpi_awareness.h src/realtime_scheduling.h @@ -218,3 +220,17 @@ target_compile_definitions(webcam_config_test PRIVATE ) target_compile_options(webcam_config_test PRIVATE /EHsc /W4 /utf-8) + +add_executable(device_selection_test + src/device_selection.cpp + src/device_selection.h + src/device_selection_test.cpp +) + +target_compile_definitions(device_selection_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(device_selection_test PRIVATE /EHsc /W4 /utf-8) diff --git a/electron/native/wgc-capture/src/device_selection.cpp b/electron/native/wgc-capture/src/device_selection.cpp new file mode 100644 index 000000000..3211cbd1d --- /dev/null +++ b/electron/native/wgc-capture/src/device_selection.cpp @@ -0,0 +1,158 @@ +#include "device_selection.h" + +#include +#include + +namespace { + +/** + * Does one of these appear inside the other as WHOLE WORDS? + * + * Plain containment answered for devices that merely share a spelling: a + * requested "Logi" is inside "Logitech", and "Micro" inside "Microphone", + * neither of them as a word. Matching on that resolved a camera nobody asked + * for -- and resolving one is exactly what stops the request reaching the + * DirectShow fallback, where the cameras Media Foundation cannot enumerate live. + * + * Both sides arrive normalized, so a boundary is the start of the string, its + * end, or a space. + */ +bool containsAsWords(const std::wstring& haystack, const std::wstring& needle) { + if (haystack.empty() || needle.empty()) { + return false; + } + size_t pos = haystack.find(needle); + while (pos != std::wstring::npos) { + const bool startsOnBoundary = pos == 0 || haystack[pos - 1] == L' '; + const size_t after = pos + needle.size(); + const bool endsOnBoundary = after == haystack.size() || haystack[after] == L' '; + if (startsOnBoundary && endsOnBoundary) { + return true; + } + pos = haystack.find(needle, pos + 1); + } + return false; +} + +bool containsInsensitive(const std::wstring& haystack, const std::wstring& needle) { + return containsAsWords(haystack, needle) || containsAsWords(needle, haystack); +} + +std::wstring normalizeDeviceName(const std::wstring& value) { + std::wstring normalized; + normalized.reserve(value.size()); + bool lastWasSpace = true; + for (const wchar_t ch : value) { + if (std::iswalnum(ch)) { + normalized.push_back(static_cast(std::towlower(ch))); + lastWasSpace = false; + continue; + } + if (!lastWasSpace) { + normalized.push_back(L' '); + lastWasSpace = true; + } + } + while (!normalized.empty() && normalized.back() == L' ') { + normalized.pop_back(); + } + return normalized; +} + +std::wstring toLower(const std::wstring& value) { + std::wstring lowered; + lowered.reserve(value.size()); + for (const wchar_t ch : value) { + lowered.push_back(static_cast(std::towlower(ch))); + } + return lowered; +} + +} // namespace + +bool DeviceClaims::contains(const std::wstring& identity) const { + const std::wstring normalized = normalizeDeviceIdentity(identity); + return !normalized.empty() && + std::find(identities_.begin(), identities_.end(), normalized) != identities_.end(); +} + +void DeviceClaims::add(const std::wstring& identity) { + const std::wstring normalized = normalizeDeviceIdentity(identity); + if (!normalized.empty() && + std::find(identities_.begin(), identities_.end(), normalized) == identities_.end()) { + identities_.push_back(normalized); + } +} + +std::wstring normalizeDeviceIdentity(const std::wstring& identity) { + const std::wstring lowered = toLower(identity); + const size_t interfaceClass = lowered.rfind(L"#{"); + if (interfaceClass == std::wstring::npos || interfaceClass == 0) { + return lowered; + } + const size_t classEnd = lowered.find(L'}', interfaceClass); + if (classEnd == std::wstring::npos) { + return lowered; + } + return lowered.substr(0, interfaceClass) + lowered.substr(classEnd + 1); +} + +int deviceMatchScore( + const std::wstring& candidateName, + const std::wstring& candidateLink, + const std::wstring& requestedName, + const std::wstring& requestedId) { + int score = 0; + const auto normalizedName = normalizeDeviceName(candidateName); + const auto normalizedLink = normalizeDeviceName(candidateLink); + const auto normalizedRequestedName = normalizeDeviceName(requestedName); + const auto normalizedRequestedId = normalizeDeviceName(requestedId); + + if (!normalizedRequestedName.empty()) { + if (normalizedName == normalizedRequestedName) { + score = std::max(score, 1000); + } + if (containsInsensitive(normalizedName, normalizedRequestedName)) { + score = std::max(score, 900); + } + if (containsInsensitive(normalizedLink, normalizedRequestedName)) { + score = std::max(score, 800); + } + } + + if (!normalizedRequestedId.empty()) { + if (containsInsensitive(normalizedLink, normalizedRequestedId)) { + score = std::max(score, 700); + } + if (containsInsensitive(normalizedName, normalizedRequestedId)) { + score = std::max(score, 600); + } + } + + return score; +} + +int selectUnclaimedDevice( + const std::vector& candidates, + const std::wstring& requestedName, + const std::wstring& requestedId, + const DeviceClaims& claims) { + const bool requested = !requestedName.empty() || !requestedId.empty(); + int selected = -1; + int bestScore = 0; + for (size_t index = 0; index < candidates.size(); ++index) { + if (claims.contains(candidates[index].identity)) { + continue; + } + const int score = + deviceMatchScore(candidates[index].name, candidates[index].identity, requestedName, requestedId); + if (requested && score <= 0) { + continue; + } + if (selected < 0 || score > bestScore) { + selected = static_cast(index); + bestScore = score; + } + } + return selected; +} diff --git a/electron/native/wgc-capture/src/device_selection.h b/electron/native/wgc-capture/src/device_selection.h new file mode 100644 index 000000000..bcdc2ad8f --- /dev/null +++ b/electron/native/wgc-capture/src/device_selection.h @@ -0,0 +1,89 @@ +#pragma once + +#include +#include + +/** + * Which camera devices are already recording in this take. + * + * Two webcams of the same model report the same friendly name, and the id the + * browser hands us is a salted hash that never matches a device path, so name + * matching alone sends every such camera to the FIRST physical device. The + * second open then fails as busy and that camera is lost. Each camera that + * opens adds its device identity here, and later cameras of the take skip it. + * + * Identities are compared normalized (see `normalizeDeviceIdentity`), so a + * device opened through Media Foundation is recognized when it turns up again + * on the DirectShow fallback. + */ +class DeviceClaims { +public: + bool contains(const std::wstring& identity) const; + /** Ignores an empty identity: a device we cannot name cannot be claimed. */ + void add(const std::wstring& identity); + +private: + std::vector identities_; +}; + +/** One device a capture backend enumerated. */ +struct DeviceCandidate { + std::wstring name; + /** The Media Foundation symbolic link or the DirectShow DevicePath. */ + std::wstring identity; +}; + +/** + * The part of a device interface path that names the physical device. + * + * Media Foundation's symbolic link and DirectShow's DevicePath are the same + * `\\?\usb#vid_…##{interface class}\` string, except that + * each registers the camera under its own interface class GUID (measured on a + * Snapdragon front camera: KSCATEGORY_VIDEO_CAMERA against KSCATEGORY_VIDEO). + * Dropping the `#{…}` class and lowercasing leaves the device instance plus the + * reference string, which both share. The reference string is kept because it + * tells apart two cameras of one device (a colour and an IR sensor). A value + * without that shape (a moniker display name) is only lowercased. + */ +std::wstring normalizeDeviceIdentity(const std::wstring& identity); + +/** + * How well a candidate answers a requested name, or 0 for "not this one". + * + * Only decisive matches count: the names being equal once normalized, or one + * containing the other -- which is the ordinary case, since Chromium appends USB + * ids to what the driver reports. + * + * A further tier used to score shared WORDS, to bridge names differing more than + * that. It bridged names that were not the same device. "Logi Capture" and + * "Logitech StreamCam" share no word, yet "logi" sits inside "logitech" and that + * scored high enough to win -- so asking for a camera Media Foundation cannot + * enumerate opened a DIFFERENT camera, instead of returning nothing and letting + * the DirectShow fallback find the real one (getopenscreen/openscreen#405). + * + * Returning 0 is what makes that fallback reachable, so it is a real answer + * rather than a weak match. Keep this in step with + * `electron/recording/deviceNameMatching.ts`, which states the same rules for + * the Electron side and carries their unit tests. + */ +int deviceMatchScore( + const std::wstring& candidateName, + const std::wstring& candidateLink, + const std::wstring& requestedName, + const std::wstring& requestedId); + +/** + * The candidate a camera should open, or -1 for none. + * + * Claimed candidates are skipped; among the rest the best `deviceMatchScore` + * wins, the earlier one on a tie. With nothing requested that is the first + * unclaimed device. With a name or id requested, a candidate scoring 0 is never + * picked -- the same rule as before claims existed, and what lets the caller + * fall back to DirectShow. With an empty claim set this is exactly the old + * selection. + */ +int selectUnclaimedDevice( + const std::vector& candidates, + const std::wstring& requestedName, + const std::wstring& requestedId, + const DeviceClaims& claims); diff --git a/electron/native/wgc-capture/src/device_selection_test.cpp b/electron/native/wgc-capture/src/device_selection_test.cpp new file mode 100644 index 000000000..aa6329f66 --- /dev/null +++ b/electron/native/wgc-capture/src/device_selection_test.cpp @@ -0,0 +1,100 @@ +#include "device_selection.h" + +#include +#include +#include + +namespace { +int failures = 0; +void expect(bool ok, const char* label) { + if (!ok) { + std::printf("FAIL %s\n", label); + ++failures; + } +} +} // namespace + +int main() { + const std::vector twins = { + {L"USB Camera", L"\\\\?\\usb#vid_0c45&pid_6366&mi_00#7&aaa&0&0000#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + {L"USB Camera", L"\\\\?\\usb#vid_0c45&pid_6366&mi_00#7&bbb&0&0000#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + }; + + DeviceClaims none; + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", none) == 0, + "same names, nothing claimed: first"); + + DeviceClaims first; + first.add(twins[0].identity); + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", first) == 1, + "same names, first claimed: second"); + + DeviceClaims both; + both.add(twins[0].identity); + both.add(twins[1].identity); + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", both) == -1, + "same names, both claimed: none"); + expect(selectUnclaimedDevice(twins, L"", L"", both) == -1, + "nothing requested, both claimed: none"); + + DeviceClaims upper; + upper.add(L"\\\\?\\USB#VID_0C45&PID_6366&MI_00#7&AAA&0&0000#{E5323777-F976-4F5B-9B55-B94699C46E44}\\GLOBAL"); + expect(selectUnclaimedDevice(twins, L"USB Camera", L"", upper) == 1, + "claim comparison is case-insensitive"); + + // DirectShow registers the same device under its own interface class GUID. + DeviceClaims viaDirectShow; + viaDirectShow.add( + L"\\\\?\\usb#vid_0c45&pid_6366&mi_00#7&aaa&0&0000#{65e8773d-8f56-11d0-a3b9-00a0c9223196}\\global"); + expect(viaDirectShow.contains(twins[0].identity), "MF link and DirectShow path are one device"); + expect(!viaDirectShow.contains(twins[1].identity), "the twin is a different device"); + + const std::vector mixed = { + {L"Studio Cam", L"\\\\?\\usb#vid_1#a#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + {L"Studio Cam Pro", L"\\\\?\\usb#vid_2#b#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + {L"Other Camera", L"\\\\?\\usb#vid_3#c#{e5323777-f976-4f5b-9b55-b94699c46e44}\\global"}, + }; + DeviceClaims exact; + exact.add(mixed[0].identity); + expect(selectUnclaimedDevice(mixed, L"Studio Cam", L"", none) == 0, "exact name wins unclaimed"); + expect(selectUnclaimedDevice(mixed, L"Studio Cam", L"", exact) == 1, + "claimed exact match loses to a weaker unclaimed match"); + DeviceClaims bothStudio = exact; + bothStudio.add(mixed[1].identity); + expect(selectUnclaimedDevice(mixed, L"Studio Cam", L"", bothStudio) == -1, + "a non-matching device is never picked for a request"); + expect(selectUnclaimedDevice(mixed, L"Nonexistent", L"", none) == -1, + "no match for a request: none"); + expect(selectUnclaimedDevice(mixed, L"", L"", none) == 0, "nothing requested: first device"); + expect(selectUnclaimedDevice(mixed, L"", L"", exact) == 1, + "nothing requested: first unclaimed device"); + expect(selectUnclaimedDevice({}, L"", L"", none) == -1, "no candidates: none"); + + // Measured on an ARM64 dev machine: the front camera as Media Foundation and + // as DirectShow list it. + DeviceClaims frontCamera; + frontCamera.add( + L"\\\\?\\display#qcom_avstream_8380#3&2dd9d5f4&0&uid32768#{e5323777-f976-4f5b-9b55-b94699c46e44}" + L"\\{4faeafd4-041b-4e46-85fd-400473891182}"); + expect(frontCamera.contains( + L"\\\\?\\DISPLAY#QCOM_AVSTREAM_8380#3&2DD9D5F4&0&UID32768#{65E8773D-8F56-11D0-A3B9-00A0C9223196}" + L"\\{4FAEAFD4-041B-4E46-85FD-400473891182}"), + "measured MF link and DirectShow path are one device"); + expect(!frontCamera.contains( + L"\\\\?\\display#qcom_avstream_8380#3&2dd9d5f4&0&uid32768#{e5323777-f976-4f5b-9b55-b94699c46e44}" + L"\\{00000000-0000-0000-0000-000000000001}"), + "another reference string on the same device is another camera"); + + DeviceClaims empty; + empty.add(L""); + expect(!empty.contains(L""), "an empty identity is never claimed"); + expect(normalizeDeviceIdentity(L"@device:sw:{ABC}") == L"@device:sw:{abc}", + "identity without an interface class is only lowercased"); + + if (failures == 0) { + std::printf("device_selection_test: all assertions passed\n"); + return 0; + } + std::printf("device_selection_test: %d failure(s)\n", failures); + return 1; +} diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp index 915e3dc4a..c7773505a 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp @@ -109,6 +109,63 @@ std::array yuvToBgr(int y, int u, int v) { return {clampToByte(blue), clampToByte(green), clampToByte(red)}; } +std::wstring readPropertyString(IPropertyBag* properties, const wchar_t* name) { + VARIANT value; + VariantInit(&value); + std::wstring result; + if (SUCCEEDED(properties->Read(name, &value, nullptr)) && value.vt == VT_BSTR && value.bstrVal) { + result = value.bstrVal; + } + VariantClear(&value); + return result; +} + +/** + * The video input devices DirectShow lists for the filter `clsid`. + * + * The fallback opens its filter by CLSID, which says nothing about WHICH + * device that is. The moniker does: its DevicePath is the same interface path + * Media Foundation reports as the symbolic link, so a device already opened + * there is recognized here. A moniker without a DevicePath (a plain software + * filter) is named by its display name instead. + */ +std::vector enumerateDevicesForClsid(const CLSID& clsid) { + std::vector candidates; + Microsoft::WRL::ComPtr deviceEnumerator; + if (FAILED(CoCreateInstance(CLSID_SystemDeviceEnum, nullptr, CLSCTX_INPROC_SERVER, IID_PPV_ARGS(&deviceEnumerator)))) { + return candidates; + } + Microsoft::WRL::ComPtr monikers; + // S_FALSE: the category is empty. + if (deviceEnumerator->CreateClassEnumerator(CLSID_VideoInputDeviceCategory, &monikers, 0) != S_OK || !monikers) { + return candidates; + } + Microsoft::WRL::ComPtr moniker; + while (monikers->Next(1, &moniker, nullptr) == S_OK) { + Microsoft::WRL::ComPtr properties; + if (SUCCEEDED(moniker->BindToStorage(nullptr, nullptr, IID_PPV_ARGS(&properties)))) { + CLSID monikerClsid{}; + const std::wstring clsidText = readPropertyString(properties.Get(), L"CLSID"); + if (!clsidText.empty() && SUCCEEDED(CLSIDFromString(clsidText.c_str(), &monikerClsid)) && + IsEqualCLSID(monikerClsid, clsid)) { + DeviceCandidate candidate{ + readPropertyString(properties.Get(), L"FriendlyName"), + readPropertyString(properties.Get(), L"DevicePath")}; + if (candidate.identity.empty()) { + LPOLESTR displayName = nullptr; + if (SUCCEEDED(moniker->GetDisplayName(nullptr, nullptr, &displayName)) && displayName) { + candidate.identity = displayName; + CoTaskMemFree(displayName); + } + } + candidates.push_back(std::move(candidate)); + } + } + moniker.Reset(); + } + return candidates; +} + } // namespace struct DirectShowWebcamCapture::Impl { @@ -294,7 +351,8 @@ bool DirectShowWebcamCapture::initialize( const std::wstring& directShowClsid, int requestedWidth, int requestedHeight, - int requestedFps) { + int requestedFps, + const DeviceClaims& claims) { (void)deviceId; stop(); delete impl_; @@ -321,6 +379,27 @@ bool DirectShowWebcamCapture::initialize( } selectedDeviceName_ = deviceName.empty() ? directShowClsid : deviceName; + // Every device this filter stands for has already been matched by name in + // Electron, so the only thing left to choose on is which one is free. With + // no moniker naming the filter, the CLSID itself is the identity. + const std::vector candidates = enumerateDevicesForClsid(selectedClsid); + if (candidates.empty()) { + deviceIdentity_ = directShowClsid; + if (claims.contains(deviceIdentity_)) { + std::cerr << "ERROR: DirectShow webcam filter is already recording in this take" << std::endl; + return false; + } + } else { + const int selectedIndex = selectUnclaimedDevice(candidates, L"", L"", claims); + if (selectedIndex < 0) { + std::cerr << "ERROR: Every DirectShow webcam for this filter is already recording in this take" + << std::endl; + return false; + } + deviceIdentity_ = candidates[selectedIndex].identity; + } + std::wcerr << L"INFO: DirectShow webcam device " << deviceIdentity_ << std::endl; + // The camera's own format first, a forced RGB32 conversion only if we cannot // read it. // @@ -594,6 +673,10 @@ int DirectShowWebcamCapture::fps() const { return fps_; } +const std::wstring& DirectShowWebcamCapture::deviceIdentity() const { + return deviceIdentity_; +} + const std::wstring& DirectShowWebcamCapture::selectedDeviceName() const { return selectedDeviceName_; } diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.h b/electron/native/wgc-capture/src/dshow_webcam_capture.h index be4f759a5..2c4393288 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.h +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.h @@ -1,5 +1,7 @@ #pragma once +#include "device_selection.h" + #include #include @@ -57,7 +59,8 @@ class DirectShowWebcamCapture { const std::wstring& directShowClsid, int requestedWidth, int requestedHeight, - int requestedFps); + int requestedFps, + const DeviceClaims& claims); bool start(); void stop(); bool copyLatestFrame(WebcamFrameSnapshot& destination, uint64_t lastSeenSequence); @@ -66,6 +69,11 @@ class DirectShowWebcamCapture { int height() const; int fps() const; const std::wstring& selectedDeviceName() const; + /** + * The opened device's DevicePath (the moniker's display name when it has + * none, the filter CLSID when no moniker names it), for the take's claims. + */ + const std::wstring& deviceIdentity() const; void storeFrame(const BYTE* buffer, long length); private: @@ -123,4 +131,5 @@ class DirectShowWebcamCapture { bool sourceTopDown_ = false; PixelFormat pixelFormat_ = PixelFormat::Bgra; std::wstring selectedDeviceName_; + std::wstring deviceIdentity_; }; diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index 20e14382c..e7a18082e 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -713,6 +713,10 @@ int wmain(int argc, wchar_t* argv[]) { } std::vector> webcams; // unique_ptr: WebcamCapture/MFEncoder are not movable + // The devices this take's cameras opened, so two cameras of the same model + // (same name, and a browser id that matches no device) open two devices + // rather than the first one twice. A camera dropped later keeps its claim. + DeviceClaims deviceClaims; for (size_t index = 0; index < webcamConfigs.size(); ++index) { auto stream = std::make_unique(); stream->config = webcamConfigs[index]; @@ -729,7 +733,8 @@ int wmain(int argc, wchar_t* argv[]) { stream->config.width, stream->config.height, stream->config.fps > 0 ? stream->config.fps : config.fps, - stream->writeSeparate)) { + stream->writeSeparate, + deviceClaims)) { // Non-fatal: a screen+audio recording the user can still use is far // better than losing the whole recording because one camera device // didn't match. Report it so the renderer can inform the user (and, diff --git a/electron/native/wgc-capture/src/webcam_capture.cpp b/electron/native/wgc-capture/src/webcam_capture.cpp index 2a367c67d..b67e1ceac 100644 --- a/electron/native/wgc-capture/src/webcam_capture.cpp +++ b/electron/native/wgc-capture/src/webcam_capture.cpp @@ -9,7 +9,6 @@ #include #include -#include #include namespace { @@ -36,114 +35,6 @@ std::wstring readAllocatedString(IMFActivate* activate, REFGUID key) { return result; } -/** - * Does one of these appear inside the other as WHOLE WORDS? - * - * Plain containment answered for devices that merely share a spelling: a - * requested "Logi" is inside "Logitech", and "Micro" inside "Microphone", - * neither of them as a word. Matching on that resolved a camera nobody asked - * for -- and resolving one is exactly what stops the request reaching the - * DirectShow fallback, where the cameras Media Foundation cannot enumerate live. - * - * Both sides arrive normalized, so a boundary is the start of the string, its - * end, or a space. - */ -bool containsAsWords(const std::wstring& haystack, const std::wstring& needle) { - if (haystack.empty() || needle.empty()) { - return false; - } - size_t pos = haystack.find(needle); - while (pos != std::wstring::npos) { - const bool startsOnBoundary = pos == 0 || haystack[pos - 1] == L' '; - const size_t after = pos + needle.size(); - const bool endsOnBoundary = after == haystack.size() || haystack[after] == L' '; - if (startsOnBoundary && endsOnBoundary) { - return true; - } - pos = haystack.find(needle, pos + 1); - } - return false; -} - -bool containsInsensitive(const std::wstring& haystack, const std::wstring& needle) { - return containsAsWords(haystack, needle) || containsAsWords(needle, haystack); -} - -std::wstring normalizeDeviceName(const std::wstring& value) { - std::wstring normalized; - normalized.reserve(value.size()); - bool lastWasSpace = true; - for (const wchar_t ch : value) { - if (std::iswalnum(ch)) { - normalized.push_back(static_cast(std::towlower(ch))); - lastWasSpace = false; - continue; - } - if (!lastWasSpace) { - normalized.push_back(L' '); - lastWasSpace = true; - } - } - while (!normalized.empty() && normalized.back() == L' ') { - normalized.pop_back(); - } - return normalized; -} - -/** - * How well a candidate answers a requested name, or 0 for "not this one". - * - * Only decisive matches count: the names being equal once normalized, or one - * containing the other -- which is the ordinary case, since Chromium appends USB - * ids to what the driver reports. - * - * A further tier used to score shared WORDS, to bridge names differing more than - * that. It bridged names that were not the same device. "Logi Capture" and - * "Logitech StreamCam" share no word, yet "logi" sits inside "logitech" and that - * scored high enough to win -- so asking for a camera Media Foundation cannot - * enumerate opened a DIFFERENT camera, instead of returning nothing and letting - * the DirectShow fallback find the real one (getopenscreen/openscreen#405). - * - * Returning 0 is what makes that fallback reachable, so it is a real answer - * rather than a weak match. Keep this in step with - * `electron/recording/deviceNameMatching.ts`, which states the same rules for - * the Electron side and carries their unit tests. - */ -int deviceMatchScore( - const std::wstring& candidateName, - const std::wstring& candidateLink, - const std::wstring& requestedName, - const std::wstring& requestedId) { - int score = 0; - const auto normalizedName = normalizeDeviceName(candidateName); - const auto normalizedLink = normalizeDeviceName(candidateLink); - const auto normalizedRequestedName = normalizeDeviceName(requestedName); - const auto normalizedRequestedId = normalizeDeviceName(requestedId); - - if (!normalizedRequestedName.empty()) { - if (normalizedName == normalizedRequestedName) { - score = std::max(score, 1000); - } - if (containsInsensitive(normalizedName, normalizedRequestedName)) { - score = std::max(score, 900); - } - if (containsInsensitive(normalizedLink, normalizedRequestedName)) { - score = std::max(score, 800); - } - } - - if (!normalizedRequestedId.empty()) { - if (containsInsensitive(normalizedLink, normalizedRequestedId)) { - score = std::max(score, 700); - } - if (containsInsensitive(normalizedName, normalizedRequestedId)) { - score = std::max(score, 600); - } - } - - return score; -} - } // namespace WebcamCapture::~WebcamCapture() { @@ -157,53 +48,48 @@ bool WebcamCapture::initialize( int requestedWidth, int requestedHeight, int requestedFps, - bool preferNv12) { + bool preferNv12, + DeviceClaims& claims) { fps_ = std::clamp(requestedFps > 0 ? requestedFps : 30, 1, 60); usingDirectShow_ = false; - selectedMatchScore_ = 0; - if (!succeeded(MFStartup(MF_VERSION), "MFStartup(webcam)")) { - if (directShowCapture_.initialize(deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_)) { - usingDirectShow_ = true; - return true; + selectedIdentity_.clear(); + const auto initializeDirectShow = [&]() { + if (!directShowCapture_.initialize( + deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_, claims)) { + if (!deviceId.empty() || !deviceName.empty()) { + std::cerr << "ERROR: Requested webcam device was not found by native Windows webcam providers" + << std::endl; + } + return false; } - return false; + usingDirectShow_ = true; + claims.add(directShowCapture_.deviceIdentity()); + return true; + }; + + if (!succeeded(MFStartup(MF_VERSION), "MFStartup(webcam)")) { + return initializeDirectShow(); } mfStarted_ = true; - if (!selectDevice(deviceId, deviceName)) { + if (!selectDevice(deviceId, deviceName, claims)) { if (mfStarted_) { MFShutdown(); mfStarted_ = false; } - if (directShowCapture_.initialize(deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_)) { - usingDirectShow_ = true; - return true; - } - return false; + return initializeDirectShow(); } - if ((!deviceId.empty() || !deviceName.empty()) && selectedMatchScore_ <= 0) { - if (mediaSource_) { - mediaSource_->Shutdown(); - } - sourceReader_.Reset(); - mediaSource_.Reset(); - if (mfStarted_) { - MFShutdown(); - mfStarted_ = false; - } - if (directShowCapture_.initialize(deviceId, deviceName, directShowClsid, requestedWidth, requestedHeight, fps_)) { - usingDirectShow_ = true; - return true; - } - std::cerr << "ERROR: Requested webcam device was not found by native Windows webcam providers" - << std::endl; + if (!configureReader(requestedWidth, requestedHeight, fps_, preferNv12)) { return false; } - - return configureReader(requestedWidth, requestedHeight, fps_, preferNv12); + claims.add(selectedIdentity_); + return true; } -bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstring& deviceName) { +bool WebcamCapture::selectDevice( + const std::wstring& deviceId, + const std::wstring& deviceName, + const DeviceClaims& claims) { Microsoft::WRL::ComPtr attributes; if (!succeeded(MFCreateAttributes(&attributes, 1), "MFCreateAttributes(webcam enumeration)")) { return false; @@ -226,34 +112,40 @@ bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstrin return false; } - UINT32 selectedIndex = 0; - int bestScore = 0; + std::vector candidates; + candidates.reserve(deviceCount); + bool anyClaimed = false; for (UINT32 index = 0; index < deviceCount; index += 1) { - const std::wstring name = readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_FRIENDLY_NAME); - const std::wstring symbolicLink = readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_SOURCE_TYPE_VIDCAP_SYMBOLIC_LINK); - const int score = deviceMatchScore(name, symbolicLink, deviceName, deviceId); - std::wcerr << L"INFO: Native webcam candidate [" << index << L"] name=\"" << name << L"\" score=" << score << std::endl; - if (score > bestScore) { - selectedIndex = index; - bestScore = score; - } - } - - if ((!deviceId.empty() || !deviceName.empty()) && bestScore <= 0) { + DeviceCandidate candidate{ + readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_FRIENDLY_NAME), + readAllocatedString(devices[index], MF_DEVSOURCE_ATTRIBUTE_SOURCE_TYPE_VIDCAP_SYMBOLIC_LINK)}; + const bool claimed = claims.contains(candidate.identity); + anyClaimed = anyClaimed || claimed; + std::wcerr << L"INFO: Native webcam candidate [" << index << L"] name=\"" << candidate.name + << L"\" score=" << deviceMatchScore(candidate.name, candidate.identity, deviceName, deviceId) + << (claimed ? L" (already recording in this take)" : L"") << std::endl; + candidates.push_back(std::move(candidate)); + } + + const int selectedIndex = selectUnclaimedDevice(candidates, deviceName, deviceId, claims); + if (selectedIndex >= 0) { + selectedDeviceName_ = candidates[selectedIndex].name; + selectedIdentity_ = candidates[selectedIndex].identity; + hr = devices[selectedIndex]->ActivateObject(IID_PPV_ARGS(&mediaSource_)); + } else if (anyClaimed) { + std::cerr << "WARNING: Every matching webcam is already recording in this take; trying DirectShow" + << std::endl; + } else { std::cerr << "WARNING: Requested webcam device was not found by Media Foundation; trying DirectShow" << std::endl; } - selectedMatchScore_ = bestScore; - selectedDeviceName_ = readAllocatedString(devices[selectedIndex], MF_DEVSOURCE_ATTRIBUTE_FRIENDLY_NAME); - hr = devices[selectedIndex]->ActivateObject(IID_PPV_ARGS(&mediaSource_)); - for (UINT32 index = 0; index < deviceCount; index += 1) { devices[index]->Release(); } CoTaskMemFree(devices); - return succeeded(hr, "ActivateObject(webcam)"); + return selectedIndex >= 0 && succeeded(hr, "ActivateObject(webcam)"); } namespace { diff --git a/electron/native/wgc-capture/src/webcam_capture.h b/electron/native/wgc-capture/src/webcam_capture.h index 949282bf9..1d9e1fe5f 100644 --- a/electron/native/wgc-capture/src/webcam_capture.h +++ b/electron/native/wgc-capture/src/webcam_capture.h @@ -1,5 +1,6 @@ #pragma once +#include "device_selection.h" #include "dshow_webcam_capture.h" #include @@ -22,6 +23,14 @@ class WebcamCapture { WebcamCapture(const WebcamCapture&) = delete; WebcamCapture& operator=(const WebcamCapture&) = delete; + /** + * Opens the requested camera, skipping devices in `claims`. + * + * `claims` belongs to the take: every camera of it is initialized against + * the same set, in config order, and one that opens adds its device. That + * is what sends a second camera of the same model to the second device + * instead of the busy first one. An empty set selects exactly as before. + */ bool initialize( const std::wstring& deviceId, const std::wstring& deviceName, @@ -29,7 +38,8 @@ class WebcamCapture { int requestedWidth, int requestedHeight, int requestedFps, - bool preferNv12); + bool preferNv12, + DeviceClaims& claims); bool start(); void stop(); bool copyLatestFrame(WebcamFrameSnapshot& destination, uint64_t lastSeenSequence); @@ -48,7 +58,10 @@ class WebcamCapture { const std::wstring& selectedDeviceName() const; private: - bool selectDevice(const std::wstring& deviceId, const std::wstring& deviceName); + bool selectDevice( + const std::wstring& deviceId, + const std::wstring& deviceName, + const DeviceClaims& claims); bool configureReader(int requestedWidth, int requestedHeight, int requestedFps, bool preferNv12); void captureLoop(); @@ -86,6 +99,7 @@ class WebcamCapture { /** Where the loop's wall time goes, in microseconds. */ uint64_t readSampleUs_ = 0; uint64_t storeUs_ = 0; - int selectedMatchScore_ = 0; std::wstring selectedDeviceName_; + /** The symbolic link of the device this capture opened, for the take's claims. */ + std::wstring selectedIdentity_; }; diff --git a/scripts/build-windows-wgc-helper.mjs b/scripts/build-windows-wgc-helper.mjs index 143fa689c..8f687dae9 100644 --- a/scripts/build-windows-wgc-helper.mjs +++ b/scripts/build-windows-wgc-helper.mjs @@ -124,6 +124,14 @@ if (!fs.existsSync(webcamConfigTestPath)) { await run(webcamConfigTestPath, [], { cwd: BUILD_DIR }); console.log(`Passed ${webcamConfigTestPath}`); +const deviceSelectionTestPath = path.join(BUILD_DIR, "device_selection_test.exe"); +if (!fs.existsSync(deviceSelectionTestPath)) { + throw new Error(`WGC helper build completed but ${deviceSelectionTestPath} was not found.`); +} +// Guards that two cameras of the same model in one take open two devices, not one twice. +await run(deviceSelectionTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${deviceSelectionTestPath}`); + const frameVisibilityTestPath = path.join(BUILD_DIR, "frame_visibility_test.exe"); if (!fs.existsSync(frameVisibilityTestPath)) { throw new Error(`WGC helper build completed but ${frameVisibilityTestPath} was not found.`); From e45dad0fea4aae84d6c1dab76ccbd94116287d4f Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:19:19 +0200 Subject: [PATCH 08/22] fix(recording): tell cameras with the same name apart A label already used by camera 1 or an earlier extra gets " (2)", " (3)" by occurrence, so unavailable and dropped cameras of the same model can be told apart. --- electron/electron-env.d.ts | 5 ++-- electron/ipc/handlers.ts | 27 ++++++++++--------- .../recording/nativeWindowsWebcams.test.ts | 23 ++++++++++++++++ electron/recording/nativeWindowsWebcams.ts | 26 ++++++++++++++++++ src/lib/nativeWindowsRecording.ts | 5 ++-- 5 files changed, 70 insertions(+), 16 deletions(-) diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index f0ed35c2f..99b2a3f79 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -178,8 +178,9 @@ interface Window { */ webcamDropped?: boolean; /** - * Device names of additional cameras (2-4) that produced nothing usable - * and were left out of the session. Camera 1 is `webcamDropped`. + * Labels (device name, else `Camera `) of additional cameras (2-4) that + * produced nothing usable and were left out of the session. Camera 1 is + * `webcamDropped`. */ droppedWebcams?: string[]; }>; diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 835a7a231..097d97706 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -131,6 +131,7 @@ import { waitForNativeWindowsCaptureStop, } from "../recording/nativeWindowsCaptureStop"; import { + additionalWebcamLabels, buildHelperWebcamConfig, collectStoppedWebcams, dedupeAdditionalWebcams, @@ -2891,20 +2892,22 @@ export function registerIpcHandlers( : null; // Cameras 2-4 only while camera 1 is on: its toggle governs every // camera. Files are numbered from 2 in the order they are sent. - const additionalWebcams = request.webcam.enabled - ? await Promise.all( - dedupeAdditionalWebcams( - request.webcam, - Array.isArray(request.additionalWebcams) ? request.additionalWebcams : [], - ).map(async (extra, i) => ({ - deviceId: extra.deviceId, - deviceName: extra.deviceName, - label: extra.deviceName?.trim() || `Camera ${i + 2}`, - clsid: await resolveDirectShowWebcamClsid(extra.deviceName), - path: webcamOutputPath(RECORDINGS_DIR, RECORDING_FILE_PREFIX, recordingId, i + 2), - })), + const keptExtras = request.webcam.enabled + ? dedupeAdditionalWebcams( + request.webcam, + Array.isArray(request.additionalWebcams) ? request.additionalWebcams : [], ) : []; + const extraLabels = additionalWebcamLabels(request.webcam.deviceName, keptExtras); + const additionalWebcams = await Promise.all( + keptExtras.map(async (extra, i) => ({ + deviceId: extra.deviceId, + deviceName: extra.deviceName, + label: extraLabels[i], + clsid: await resolveDirectShowWebcamClsid(extra.deviceName), + path: webcamOutputPath(RECORDINGS_DIR, RECORDING_FILE_PREFIX, recordingId, i + 2), + })), + ); const cursorCaptureMode = normalizeCursorCaptureMode(request.cursor?.mode) ?? "editable-overlay"; const envPreferSoftwareEncoder = (process.env.OPENSCREEN_WGC_PREFER_SOFTWARE_ENCODER ?? "") diff --git a/electron/recording/nativeWindowsWebcams.test.ts b/electron/recording/nativeWindowsWebcams.test.ts index bb4e7c039..ffd0ea56e 100644 --- a/electron/recording/nativeWindowsWebcams.test.ts +++ b/electron/recording/nativeWindowsWebcams.test.ts @@ -1,5 +1,6 @@ import { describe, expect, it } from "vitest"; import { + additionalWebcamLabels, buildHelperWebcamConfig, collectStoppedWebcams, dedupeAdditionalWebcams, @@ -118,6 +119,28 @@ describe("nativeWindowsWebcams", () => { expect(labelsOfUnavailableAdditionalWebcams(requested, [0, 2, 2, 9])).toEqual(["Side"]); }); + it("labels extras by device name, else Camera ", () => { + expect(additionalWebcamLabels("Front", [{ deviceName: "Desk" }, { deviceName: " " }])).toEqual( + ["Desk", "Camera 3"], + ); + }); + + it("tells cameras with the same name apart by occurrence, counting camera 1", () => { + expect( + additionalWebcamLabels("USB Camera", [ + { deviceName: "USB Camera" }, + { deviceName: "Desk" }, + { deviceName: " USB Camera " }, + ]), + ).toEqual(["USB Camera (2)", "Desk", "USB Camera (3)"]); + expect( + additionalWebcamLabels(undefined, [ + { deviceName: "USB Camera" }, + { deviceName: "USB Camera" }, + ]), + ).toEqual(["USB Camera", "USB Camera (2)"]); + }); + it("builds a start config with the legacy fields and a list of every camera", () => { const config = buildHelperWebcamConfig({ camera1: { diff --git a/electron/recording/nativeWindowsWebcams.ts b/electron/recording/nativeWindowsWebcams.ts index 42e87365a..08a797082 100644 --- a/electron/recording/nativeWindowsWebcams.ts +++ b/electron/recording/nativeWindowsWebcams.ts @@ -92,6 +92,32 @@ export function dedupeAdditionalWebcams` (n counts camera 1, so the first extra is camera 2). + * + * Two webcams of the same model report the same name, and a label is all the + * user is told about a camera that could not be opened or was dropped. So a + * label already used — by camera 1 or an earlier extra — gets " (2)", " (3)" + * by occurrence. Camera 1's own label is never changed. + */ +export function additionalWebcamLabels( + camera1Name: string | undefined, + extras: Array<{ deviceName: string }>, +): string[] { + const seen = new Map(); + const camera1Label = camera1Name?.trim(); + if (camera1Label) { + seen.set(camera1Label, 1); + } + return extras.map((extra, i) => { + const label = extra.deviceName.trim() || `Camera ${i + 2}`; + const occurrence = (seen.get(label) ?? 0) + 1; + seen.set(label, occurrence); + return occurrence > 1 ? `${label} (${occurrence})` : label; + }); +} + /** * The camera part of the helper config: the unchanged legacy `webcam*` fields * for camera 1 plus the `webcams` list (camera 1 first, then the extras). diff --git a/src/lib/nativeWindowsRecording.ts b/src/lib/nativeWindowsRecording.ts index c014cf243..b5924f84f 100644 --- a/src/lib/nativeWindowsRecording.ts +++ b/src/lib/nativeWindowsRecording.ts @@ -66,8 +66,9 @@ export type NativeWindowsRecordingStartResult = { */ webcamUnavailable?: boolean; /** - * Device names of additional cameras (2-4) the helper could not open; this - * take records without them. Camera 1 is reported by `webcamUnavailable`. + * Labels (device name, else `Camera `) of additional cameras (2-4) the + * helper could not open; this take records without them. Camera 1 is + * reported by `webcamUnavailable`. */ unavailableWebcams?: string[]; /** From 1a419933d44fc4a2ad00dc134500f5a377e0484a Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:29:22 +0200 Subject: [PATCH 09/22] feat(project): keep a recording's additional cameras --- electron/media/projectMediaRelinker.test.ts | 48 ++++++++++++ electron/media/projectMediaRelinker.ts | 29 ++++++- src/lib/ai-edition/schema/index.test.ts | 28 +++++++ src/lib/ai-edition/schema/index.ts | 18 +++++ src/lib/ai-edition/store/projectStore.test.ts | 76 +++++++++++++++++++ src/lib/ai-edition/store/projectStore.ts | 23 +++++- 6 files changed, 220 insertions(+), 2 deletions(-) diff --git a/electron/media/projectMediaRelinker.test.ts b/electron/media/projectMediaRelinker.test.ts index 2b52d32b5..505a430f4 100644 --- a/electron/media/projectMediaRelinker.test.ts +++ b/electron/media/projectMediaRelinker.test.ts @@ -62,6 +62,54 @@ describe("relinkProjectMedia", () => { expect(logged.join("\n")).toContain(currentWebcamPath); }); + it("relinks additional cameras by index and leaves absent ones alone", async () => { + const currentScreenPath = path.join(tempDir, "recording-43.mp4"); + const currentWebcamPath = path.join(tempDir, "recording-43-webcam.mp4"); + const currentExtra2 = path.join(tempDir, "recording-43-webcam-2.mp4"); + const currentExtra3 = path.join(tempDir, "recording-43-webcam-3.mp4"); + await fs.writeFile(currentScreenPath, "screen bytes"); + await fs.writeFile(currentWebcamPath, "webcam bytes"); + await fs.writeFile(currentExtra2, "extra 2"); + await fs.writeFile(currentExtra3, "extra 3"); + await registerMediaLinks(tempDir, currentScreenPath, { + webcamVideoPath: currentWebcamPath, + additionalWebcams: [ + { path: currentExtra2, label: "Desk" }, + { path: currentExtra3, label: "Wide" }, + ], + }); + + const project = { + assets: [ + { + id: "asset-1", + originalPath: "C:\\Users\\demo\\recording-43.mp4", + sizeBytes: Buffer.byteLength("screen bytes"), + cameraTrack: { + sourcePath: "C:\\Users\\demo\\recording-43-webcam.mp4", + startMs: 0, + offsetMs: 0, + visible: true, + }, + additionalCameraTracks: [ + { sourcePath: "C:\\Users\\demo\\recording-43-webcam-2.mp4", label: "Desk" }, + { sourcePath: "C:\\Users\\demo\\recording-43-webcam-3.mp4", label: "Wide" }, + ], + }, + ], + }; + + const relinked = (await relinkProjectMedia(project, tempDir)) as typeof project; + + expect(relinked.assets[0].cameraTrack.sourcePath).toBe(currentWebcamPath); + expect(relinked.assets[0].additionalCameraTracks.map((t) => t.sourcePath)).toEqual([ + currentExtra2, + currentExtra3, + ]); + expect(relinked.assets[0].additionalCameraTracks.map((t) => t.label)).toEqual(["Desk", "Wide"]); + expect(project.assets[0].additionalCameraTracks[0].sourcePath).toContain("demo"); + }); + it("refuses to relink an asset the document recorded no size for", async () => { // A same-named recording exists and is registered with its webcam, so a // basename match would resolve — that is exactly what must not happen. The diff --git a/electron/media/projectMediaRelinker.ts b/electron/media/projectMediaRelinker.ts index 0c0b8c259..bef7dad02 100644 --- a/electron/media/projectMediaRelinker.ts +++ b/electron/media/projectMediaRelinker.ts @@ -41,11 +41,23 @@ async function resolveAssetMedia( isRecord(cameraTrack) && typeof cameraTrack.sourcePath === "string" && cameraTrack.sourcePath ? cameraTrack.sourcePath : null; + const additionalTracks = Array.isArray(asset.additionalCameraTracks) + ? asset.additionalCameraTracks + : []; + const additionalMissing = await Promise.all( + additionalTracks.map( + async (track) => + isRecord(track) && + typeof track.sourcePath === "string" && + track.sourcePath !== "" && + !(await fileExists(track.sourcePath)), + ), + ); const screenExists = await fileExists(originalPath); const cameraMissing = cameraPath !== null && !(await fileExists(cameraPath)); // Nothing to repair, and this runs on every project open — don't fingerprint // (i.e. open and read) every asset just to confirm what the stats already say. - if (screenExists && !cameraMissing) return asset; + if (screenExists && !cameraMissing && !additionalMissing.some(Boolean)) return asset; let links: RelocatedMediaLookup | null = null; if (screenExists) { @@ -81,10 +93,25 @@ async function resolveAssetMedia( nextCameraTrack = { ...cameraTrack, sourcePath: links.webcamVideoPath }; } + // Extras are matched by index, the same order the recording registered them in. + let additionalChanged = false; + const nextAdditionalTracks = await Promise.all( + additionalTracks.map(async (track, index) => { + const replacement = links.additionalWebcams?.[index]?.path; + if (!additionalMissing[index] || !replacement || !(await fileExists(replacement))) { + return track; + } + console.log(`[media-relink] additional webcam ${index + 2} -> ${replacement}`); + additionalChanged = true; + return { ...track, sourcePath: replacement }; + }), + ); + return { ...asset, originalPath: links.screenVideoPath, ...(nextCameraTrack === cameraTrack ? {} : { cameraTrack: nextCameraTrack }), + ...(additionalChanged ? { additionalCameraTracks: nextAdditionalTracks } : {}), }; } diff --git a/src/lib/ai-edition/schema/index.test.ts b/src/lib/ai-edition/schema/index.test.ts index 38aabd870..bb8148e9c 100644 --- a/src/lib/ai-edition/schema/index.test.ts +++ b/src/lib/ai-edition/schema/index.test.ts @@ -1,6 +1,7 @@ import { describe, expect, it } from "vitest"; import { migrateRawDocumentToCurrent } from "../document/migrate"; import { + additionalCameraTrackSchema, annotationRegionSchema, assetSchema, audioTrackSchema, @@ -275,6 +276,33 @@ describe("axcut-schema v8", () => { ).not.toThrow(); }); + describe("additionalCameraTracks", () => { + const base = { id: "asset_1", kind: "video", label: "x", originalPath: "/x.mp4" }; + + it("leaves an asset without the key untouched", () => { + expect(assetSchema.parse(base)).not.toHaveProperty("additionalCameraTracks"); + }); + + it("fills defaults on each entry", () => { + const asset = assetSchema.parse({ + ...base, + additionalCameraTracks: [{ sourcePath: "/a.mp4" }, { sourcePath: "/b.mp4", label: "Desk" }], + }); + expect(asset.additionalCameraTracks).toEqual([ + { sourcePath: "/a.mp4", startMs: 0, offsetMs: 0, visible: true, label: "" }, + { sourcePath: "/b.mp4", startMs: 0, offsetMs: 0, visible: true, label: "Desk" }, + ]); + expect(additionalCameraTrackSchema.safeParse({ sourcePath: "" }).success).toBe(false); + }); + + it("rejects more than three additional cameras", () => { + const entries = Array.from({ length: 5 }, (_, i) => ({ sourcePath: `/c${i}.mp4` })); + expect(assetSchema.safeParse({ ...base, additionalCameraTracks: entries }).success).toBe( + false, + ); + }); + }); + it("assetSchema defaults cameraTrack to null", () => { const asset = assetSchema.parse({ id: "asset_1", diff --git a/src/lib/ai-edition/schema/index.ts b/src/lib/ai-edition/schema/index.ts index 9c5760f19..02702afc1 100644 --- a/src/lib/ai-edition/schema/index.ts +++ b/src/lib/ai-edition/schema/index.ts @@ -156,6 +156,23 @@ export const cameraTrackSchema = z .nullable() .default(null); +// Cameras 2-4 of a multi-camera recording. Camera 1 stays `cameraTrack`; these only +// store the extra files (the editor still shows camera 1) and mirror its fields. +// +// Optional and absent on every document written before multi-camera recording, +// exactly like `cameraTrack.width`: additive, so no schema-version bump, and an +// older build simply drops the key on save. +export const additionalCameraTrackSchema = z.object({ + sourcePath: z.string().min(1), + startMs: z.number().nonnegative().default(0), + offsetMs: z.number().int().default(0), + visible: z.boolean().default(true), + width: z.number().int().positive().optional(), + height: z.number().int().positive().optional(), + label: z.string().default(""), +}); +export type AxcutAdditionalCameraTrack = z.infer; + // Why a media can never be transcribed. Only the DETERMINISTIC verdicts live // here: a container with no audio track (a screen recording captured with no // mic and no system audio — the common case) fails identically on every @@ -190,6 +207,7 @@ export const assetSchema = z.object({ // no schema-version bump (an older build simply drops the key on save). transcriptionFailure: assetTranscriptionFailureSchema.nullish(), cameraTrack: cameraTrackSchema, + additionalCameraTracks: z.array(additionalCameraTrackSchema).max(3).optional(), }); // A crop is a sub-rectangle of the source video, expressed as fractions diff --git a/src/lib/ai-edition/store/projectStore.test.ts b/src/lib/ai-edition/store/projectStore.test.ts index b49bae656..ce3daf45d 100644 --- a/src/lib/ai-edition/store/projectStore.test.ts +++ b/src/lib/ai-edition/store/projectStore.test.ts @@ -251,6 +251,82 @@ describe("useProjectStore", () => { expect(camera?.offsetMs).toBe(-193); }); + it("addAsset keeps a recording's additional cameras beside camera 1", async () => { + useProjectStore.setState({ + projectId: "proj_test", + document: sampleDoc, + revision: 1, + status: "ready", + error: null, + }); + bridgeMocks.save.mockImplementation(async (document) => ({ success: true, document })); + // biome-ignore lint/suspicious/noExplicitAny: test-only stub of the legacy contextBridge surface + (window as any).electronAPI.findRecordingCamera.mockResolvedValue({ + success: true, + webcamVideoPath: "/w.mp4", + offsetMs: 0, + additionalWebcams: [{ path: "/w-2.mp4", label: "Desk" }], + }); + bridgeMocks.addAsset.mockResolvedValue({ + assetId: "asset_1", + document: { + ...sampleDoc, + assets: [ + { id: "asset_1", kind: "video", label: "screen.mp4", originalPath: "/tmp/screen.mp4" }, + ], + project: { ...sampleDoc.project, primaryAssetId: "asset_1" }, + }, + }); + + await useProjectStore.getState().addAsset("/tmp/screen.mp4"); + await new Promise((resolve) => setTimeout(resolve, 0)); + + const asset = useProjectStore.getState().document?.assets[0]; + expect(asset?.cameraTrack?.sourcePath).toBe("/w.mp4"); + expect(asset?.additionalCameraTracks).toHaveLength(1); + expect(asset?.additionalCameraTracks?.[0]).toMatchObject({ + sourcePath: "/w-2.mp4", + label: "Desk", + startMs: 0, + offsetMs: 0, + visible: true, + }); + }); + + it("addAsset adds no additionalCameraTracks key when the recording has no extras", async () => { + useProjectStore.setState({ + projectId: "proj_test", + document: sampleDoc, + revision: 1, + status: "ready", + error: null, + }); + bridgeMocks.save.mockImplementation(async (document) => ({ success: true, document })); + // biome-ignore lint/suspicious/noExplicitAny: test-only stub of the legacy contextBridge surface + (window as any).electronAPI.findRecordingCamera.mockResolvedValue({ + success: true, + webcamVideoPath: "/w.mp4", + offsetMs: 0, + }); + bridgeMocks.addAsset.mockResolvedValue({ + assetId: "asset_1", + document: { + ...sampleDoc, + assets: [ + { id: "asset_1", kind: "video", label: "screen.mp4", originalPath: "/tmp/screen.mp4" }, + ], + project: { ...sampleDoc.project, primaryAssetId: "asset_1" }, + }, + }); + + await useProjectStore.getState().addAsset("/tmp/screen.mp4"); + await new Promise((resolve) => setTimeout(resolve, 0)); + + const asset = useProjectStore.getState().document?.assets[0]; + expect(asset?.cameraTrack?.sourcePath).toBe("/w.mp4"); + expect(asset).not.toHaveProperty("additionalCameraTracks"); + }); + it("addAsset stays silent (no toast) when a plain imported video has no camera", async () => { useProjectStore.setState({ projectId: "proj_test", diff --git a/src/lib/ai-edition/store/projectStore.ts b/src/lib/ai-edition/store/projectStore.ts index ec60b230b..351989dd0 100644 --- a/src/lib/ai-edition/store/projectStore.ts +++ b/src/lib/ai-edition/store/projectStore.ts @@ -381,6 +381,21 @@ export const useProjectStore = create((set, get) => ({ const camDims = await probeVideoDimensions(toFileUrl(camera.webcamVideoPath)).catch( () => null, ); + // Cameras 2-4 get the same treatment as camera 1 (same T0 on Windows, so the + // same offset) and the same non-fatal dimension probe. + const extras = await Promise.all( + (camera.additionalWebcams ?? []).map(async (extra) => { + const dims = await probeVideoDimensions(toFileUrl(extra.path)).catch(() => null); + return { + sourcePath: extra.path, + label: extra.label, + startMs: 0, + offsetMs: Math.round(camera.offsetMs ?? 0), + visible: true, + ...(dims ?? {}), + }; + }), + ); const linked = { sourcePath: camera.webcamVideoPath, startMs: 0, @@ -403,7 +418,13 @@ export const useProjectStore = create((set, get) => ({ const next: AxcutDocument = { ...document, assets: document.assets.map((a) => - a.id === addedAsset.id ? { ...a, cameraTrack: linked } : a, + a.id === addedAsset.id + ? { + ...a, + cameraTrack: linked, + ...(extras.length > 0 ? { additionalCameraTracks: extras } : {}), + } + : a, ), }; // Only adopt the linked document if it actually reached disk -- otherwise From aa84135dc27e2e9e11e4eda8c9910c8298fe3791 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:40:03 +0200 Subject: [PATCH 10/22] feat(hud): pick additional cameras for a Windows recording --- electron/app-settings.test.ts | 55 +++++++ electron/app-settings.ts | 27 ++++ electron/ipc/handlers.ts | 3 + electron/ipc/recordingPrefs.test.ts | 1 + .../ai-edition/v4/EditorShellV4.module.css | 44 ++++++ .../ai-edition/v4/RecStage.test.tsx | 2 + src/components/ai-edition/v4/RecStage.tsx | 30 ++++ .../launch/AdditionalCamerasList.test.tsx | 137 ++++++++++++++++++ .../launch/AdditionalCamerasList.tsx | 99 +++++++++++++ src/components/launch/HudDeviceSettings.tsx | 26 ++++ src/components/launch/LaunchWindow.module.css | 10 ++ src/components/launch/LaunchWindow.tsx | 40 +++++ src/hooks/useNativeWindowsCaptureAvailable.ts | 27 ++++ src/hooks/useScreenRecorder.noCamera.test.tsx | 1 + .../useScreenRecorder.prefsRace.test.tsx | 1 + src/hooks/useScreenRecorder.ts | 47 ++++++ src/i18n/locales/ar/launch.json | 3 + src/i18n/locales/cs/launch.json | 3 + src/i18n/locales/de/launch.json | 3 + src/i18n/locales/en/launch.json | 3 + src/i18n/locales/es/launch.json | 3 + src/i18n/locales/fr/launch.json | 3 + src/i18n/locales/it/launch.json | 3 + src/i18n/locales/ja-JP/launch.json | 3 + src/i18n/locales/ko-KR/launch.json | 3 + src/i18n/locales/pt-BR/launch.json | 3 + src/i18n/locales/ru/launch.json | 3 + src/i18n/locales/tr/launch.json | 3 + src/i18n/locales/vi/launch.json | 3 + src/i18n/locales/zh-CN/launch.json | 3 + src/i18n/locales/zh-TW/launch.json | 3 + src/lib/additionalWebcams.test.ts | 60 ++++++++ src/lib/additionalWebcams.ts | 36 +++++ 33 files changed, 691 insertions(+) create mode 100644 src/components/launch/AdditionalCamerasList.test.tsx create mode 100644 src/components/launch/AdditionalCamerasList.tsx create mode 100644 src/hooks/useNativeWindowsCaptureAvailable.ts create mode 100644 src/lib/additionalWebcams.test.ts create mode 100644 src/lib/additionalWebcams.ts diff --git a/electron/app-settings.test.ts b/electron/app-settings.test.ts index f3781734f..f7590fdb6 100644 --- a/electron/app-settings.test.ts +++ b/electron/app-settings.test.ts @@ -165,4 +165,59 @@ describe("app settings store", () => { expect(new AppSettingsStore(dir).getSnapshot().recording.camQuality).toBe("2160p"); }); + + it("reads a missing additional camera list as empty", () => { + const dir = temp(); + writeFileSync( + path.join(dir, "recording-settings.json"), + JSON.stringify({ camEnabled: true }), + "utf8", + ); + + expect(new AppSettingsStore(dir).getSnapshot().recording.camAdditionalDevices).toEqual([]); + }); + + it("drops additional camera entries without a name and keeps at most three", () => { + const store = new AppSettingsStore(temp()); + store.setRecordingPreferences({ + camAdditionalDevices: [ + { id: "a", name: "A" }, + { id: null, name: "B" }, + { id: "x", name: "" }, + { id: "c", name: "C" }, + { id: "d", name: "D" }, + { id: "e", name: "E" }, + ], + }); + + expect(store.getSnapshot().recording.camAdditionalDevices).toEqual([ + { id: "a", name: "A" }, + { id: null, name: "B" }, + { id: "c", name: "C" }, + ]); + }); + + it("ignores junk in a stored additional camera list", () => { + const dir = temp(); + writeFileSync( + path.join(dir, "recording-settings.json"), + JSON.stringify({ + camAdditionalDevices: [null, 3, "x", { name: 5 }, { id: 7, name: "Kept" }, { name: "Ok" }], + }), + "utf8", + ); + + expect(new AppSettingsStore(dir).getSnapshot().recording.camAdditionalDevices).toEqual([ + { id: null, name: "Kept" }, + { id: null, name: "Ok" }, + ]); + }); + + it("rejects an additional camera list that is not a list", () => { + const store = new AppSettingsStore(temp()); + + expect(() => store.setRecordingPreferences({ camAdditionalDevices: "x" as never })).toThrow( + TypeError, + ); + }); }); diff --git a/electron/app-settings.ts b/electron/app-settings.ts index bef025bf3..d1efb9aae 100644 --- a/electron/app-settings.ts +++ b/electron/app-settings.ts @@ -15,6 +15,8 @@ export interface RecordingPreferences { camEnabled: boolean; camDeviceId: string | null; camDeviceName: string | null; + /** Cameras 2-4 of a native Windows recording, in the order they were picked. At most three. */ + camAdditionalDevices: Array<{ id: string | null; name: string }>; /** Capture resolution for the camera. See WEBCAM_QUALITY_PRESETS. */ camQuality: WebcamQualityId; systemAudioEnabled: boolean; @@ -38,6 +40,7 @@ export const DEFAULT_RECORDING_PREFERENCES: RecordingPreferences = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: DEFAULT_WEBCAM_QUALITY, systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", @@ -101,6 +104,22 @@ const bool = (value: unknown, fallback: boolean) => (typeof value === "boolean" const nullableString = (value: unknown, fallback: string | null) => value === null || typeof value === "string" ? value : fallback; +const MAX_ADDITIONAL_CAMERAS = 3; + +/** Keeps the entries that name a camera, in order, up to the cap; everything else is junk. */ +function additionalCameras(value: unknown): RecordingPreferences["camAdditionalDevices"] { + if (!Array.isArray(value)) return []; + const kept: RecordingPreferences["camAdditionalDevices"] = []; + for (const entry of value) { + if (kept.length >= MAX_ADDITIONAL_CAMERAS) break; + if (!entry || typeof entry !== "object") continue; + const { id, name } = entry as Record; + if (typeof name !== "string" || name.length === 0) continue; + kept.push({ id: typeof id === "string" ? id : null, name }); + } + return kept; +} + function parseRecording(raw: RawSettings): RecordingPreferences { return { micEnabled: bool(raw.micEnabled, DEFAULT_RECORDING_PREFERENCES.micEnabled), @@ -109,6 +128,7 @@ function parseRecording(raw: RawSettings): RecordingPreferences { camEnabled: bool(raw.camEnabled, DEFAULT_RECORDING_PREFERENCES.camEnabled), camDeviceId: nullableString(raw.camDeviceId, DEFAULT_RECORDING_PREFERENCES.camDeviceId), camDeviceName: nullableString(raw.camDeviceName, DEFAULT_RECORDING_PREFERENCES.camDeviceName), + camAdditionalDevices: additionalCameras(raw.camAdditionalDevices), // Unset in every settings file written before the camera had a quality // setting, and `webcamQualityFrom` answers those with the default. camQuality: webcamQualityFrom(raw.camQuality), @@ -169,6 +189,10 @@ function validateRecordingPatch(patch: Partial): void { if ((key.endsWith("Enabled") || key === "hideDesktopIcons") && typeof value !== "boolean") { throw new TypeError(`${key} must be a boolean`); } + if (key === "camAdditionalDevices") { + if (!Array.isArray(value)) throw new TypeError("camAdditionalDevices must be a list"); + continue; + } if ( (key.endsWith("DeviceId") || key.endsWith("DeviceName")) && value !== null && @@ -207,6 +231,9 @@ export class AppSettingsStore { const next = Object.fromEntries( Object.entries(patch).filter(([, value]) => value !== undefined), ) as Partial; + if (next.camAdditionalDevices) { + next.camAdditionalDevices = additionalCameras(next.camAdditionalDevices); + } atomicWrite(this.userData, { ...raw, ...current, ...next }); return this.getSnapshot(); } diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 097d97706..191250fcd 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -686,6 +686,8 @@ export interface RecordingPrefs { camDeviceId: string | null; /** Camera label paired with the preferred id for restart-safe resolution. */ camDeviceName: string | null; + /** Cameras 2-4 of a native Windows recording, in pick order. At most three. */ + camAdditionalDevices: Array<{ id: string | null; name: string }>; /** Capture resolution for the camera. See WEBCAM_QUALITY_PRESETS. */ camQuality: WebcamQualityId; systemAudioEnabled: boolean; @@ -701,6 +703,7 @@ const defaultRecordingPrefs: RecordingPrefs = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: DEFAULT_WEBCAM_QUALITY, systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", diff --git a/electron/ipc/recordingPrefs.test.ts b/electron/ipc/recordingPrefs.test.ts index 2c3317af9..7d2d056a1 100644 --- a/electron/ipc/recordingPrefs.test.ts +++ b/electron/ipc/recordingPrefs.test.ts @@ -19,6 +19,7 @@ const defaults: RecordingPrefs = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", diff --git a/src/components/ai-edition/v4/EditorShellV4.module.css b/src/components/ai-edition/v4/EditorShellV4.module.css index ee3d11fab..b4a1449cc 100644 --- a/src/components/ai-edition/v4/EditorShellV4.module.css +++ b/src/components/ai-edition/v4/EditorShellV4.module.css @@ -1000,6 +1000,50 @@ border-color: var(--border-hi); background: var(--surface-2); } +/* Cameras 2-4, under the camera row: one checkbox line per extra camera. */ +.recExtraCams { + display: flex; + flex-direction: column; + gap: 2px; + padding: 10px 16px 13px; + border-bottom: 1px solid var(--border-soft); +} +.recExtraCamsTitle { + color: var(--fg-2); + font-size: 13px; + font-weight: 500; + margin-bottom: 4px; +} +.recExtraCam { + display: flex; + align-items: center; + justify-content: space-between; + gap: 8px; + padding: 6px 10px; + border: 1px solid var(--border); + border-radius: 9px; + background: var(--surface-hi); + color: var(--fg); + font-size: 12.5px; + cursor: pointer; + text-align: left; +} +.recExtraCam:hover:not(:disabled) { + border-color: var(--border-hi); + background: var(--surface-2); +} +.recExtraCam:disabled { + opacity: 0.45; + cursor: not-allowed; +} +.recExtraCamOn { + border-color: var(--border-hi); +} +.recExtraCamsHint { + color: var(--muted); + font-size: 12px; + margin-top: 4px; +} .recSelect { height: 32px; max-width: 200px; diff --git a/src/components/ai-edition/v4/RecStage.test.tsx b/src/components/ai-edition/v4/RecStage.test.tsx index b79308df3..58af434b1 100644 --- a/src/components/ai-edition/v4/RecStage.test.tsx +++ b/src/components/ai-edition/v4/RecStage.test.tsx @@ -290,6 +290,7 @@ describe("RecStage controls", () => { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", @@ -501,6 +502,7 @@ describe("RecStage controls", () => { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", diff --git a/src/components/ai-edition/v4/RecStage.tsx b/src/components/ai-edition/v4/RecStage.tsx index 658ed6b4d..6f488f444 100644 --- a/src/components/ai-edition/v4/RecStage.tsx +++ b/src/components/ai-edition/v4/RecStage.tsx @@ -13,6 +13,7 @@ import { ZoomIn, } from "lucide-react"; import { useEffect, useId, useRef, useState } from "react"; +import { AdditionalCamerasList } from "@/components/launch/AdditionalCamerasList"; import { AudioLevelMeter } from "@/components/ui/audio-level-meter"; import { Tooltip } from "@/components/ui/tooltip"; import { useScopedT } from "@/contexts/I18nContext"; @@ -21,6 +22,7 @@ import { useCameraDevices } from "@/hooks/useCameraDevices"; import { useCameraPreviewStream } from "@/hooks/useCameraPreviewStream"; import { useEditableCursorAvailable } from "@/hooks/useEditableCursorAvailable"; import { useMicrophoneDevices } from "@/hooks/useMicrophoneDevices"; +import { useNativeWindowsCaptureAvailable } from "@/hooks/useNativeWindowsCaptureAvailable"; import { usePortalOwnsSource } from "@/hooks/usePortalOwnsSource"; import { canRecordMicrophone, getPlatform } from "@/utils/platformUtils"; import styles from "./EditorShellV4.module.css"; @@ -32,6 +34,7 @@ interface RecordingPrefsState { camEnabled: boolean; camDeviceId: string | null; camDeviceName: string | null; + camAdditionalDevices: Array<{ id: string | null; name: string }>; systemAudioEnabled: boolean; cursorCaptureMode: "editable-overlay" | "system"; hideDesktopIcons: boolean; @@ -45,6 +48,7 @@ const DEFAULT_PREFS: RecordingPrefsState = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", hideDesktopIcons: false, @@ -79,6 +83,8 @@ export function RecStage({ onClose?: () => void; }) { const t = useScopedT("editor"); + const tLaunch = useScopedT("launch"); + const nativeWindowsCapture = useNativeWindowsCaptureAvailable(); const [prefs, setPrefsState] = useState(DEFAULT_PREFS); // Bumped by every local change and every pushed snapshot, so the re-read after a // failed write can tell that something newer has landed since. @@ -478,6 +484,30 @@ export function RecStage({ + {/* Cameras 2-4 are recorded beside camera 1 by the native Windows helper only, so the + list waits for camera 1 and, without that helper, stays visible but disabled. */} + {prefs.camEnabled && camDevices.devices.length > 1 ? ( +
+ updatePrefs({ camAdditionalDevices: next })} + disabled={!nativeWindowsCapture} + labels={{ + title: tLaunch("webcam.additionalCameras"), + hint: tLaunch("webcam.additionalCamerasHint"), + }} + classes={{ + title: styles.recExtraCamsTitle, + item: styles.recExtraCam, + itemActive: styles.recExtraCamOn, + hint: styles.recExtraCamsHint, + }} + /> +
+ ) : null} + {/* Without its native helper the browser records, and it always draws the system cursor into the video: there is nothing to switch, so neither this row nor Auto-zoom, which reads what the editable cursor records, is shown. Same rule as the HUD's button. */} diff --git a/src/components/launch/AdditionalCamerasList.test.tsx b/src/components/launch/AdditionalCamerasList.test.tsx new file mode 100644 index 000000000..061724417 --- /dev/null +++ b/src/components/launch/AdditionalCamerasList.test.tsx @@ -0,0 +1,137 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { cleanup, fireEvent, render, screen } from "@testing-library/react"; +import { useState } from "react"; +import { afterEach, describe, expect, it, vi } from "vitest"; +import { type AdditionalCameraChoice, AdditionalCamerasList } from "./AdditionalCamerasList"; + +afterEach(cleanup); + +const labels = { + title: "Additional cameras", + hint: "Only with native Windows recording", +}; + +const device = (id: string) => ({ deviceId: id, label: `Camera ${id}`, groupId: id }); +const choice = (id: string): AdditionalCameraChoice => ({ id, name: `Camera ${id}` }); + +describe("AdditionalCamerasList", () => { + it("the additional list never offers camera 1", () => { + render( + , + ); + + const items = screen.getAllByRole("menuitemcheckbox"); + expect(items.map((item) => item.textContent)).toEqual(["Camera B", "Camera C"]); + }); + + it("keeps the order of selection", () => { + const onChange = vi.fn(); + function Harness() { + const [selected, setSelected] = useState([]); + return ( + { + onChange(next); + setSelected(next); + }} + disabled={false} + labels={labels} + /> + ); + } + render(); + + fireEvent.click(screen.getByRole("menuitemcheckbox", { name: "Camera C" })); + fireEvent.click(screen.getByRole("menuitemcheckbox", { name: "Camera B" })); + + expect(onChange).toHaveBeenLastCalledWith([choice("C"), choice("B")]); + expect(screen.getByRole("menuitemcheckbox", { name: "Camera C" })).toHaveAttribute( + "aria-checked", + "true", + ); + }); + + it("unchecks a camera that is already selected", () => { + const onChange = vi.fn(); + render( + , + ); + + fireEvent.click(screen.getByRole("menuitemcheckbox", { name: "Camera B" })); + + expect(onChange).toHaveBeenLastCalledWith([choice("C")]); + }); + + it("caps at three", () => { + const onChange = vi.fn(); + render( + , + ); + + const fourth = screen.getByRole("menuitemcheckbox", { name: "Camera E" }); + expect(fourth).toBeDisabled(); + expect(screen.getByRole("menuitemcheckbox", { name: "Camera B" })).toBeEnabled(); + fireEvent.click(fourth); + expect(onChange).not.toHaveBeenCalled(); + }); + + it("is disabled with a hint without native Windows recording", () => { + const onChange = vi.fn(); + render( + , + ); + + const item = screen.getByRole("menuitemcheckbox", { name: "Camera B" }); + expect(item).toBeDisabled(); + expect(screen.getByText(labels.hint)).toBeInTheDocument(); + fireEvent.click(item); + expect(onChange).not.toHaveBeenCalled(); + }); + + it("shows no hint while it is usable", () => { + render( + , + ); + + expect(screen.queryByText(labels.hint)).not.toBeInTheDocument(); + }); +}); diff --git a/src/components/launch/AdditionalCamerasList.tsx b/src/components/launch/AdditionalCamerasList.tsx new file mode 100644 index 000000000..4d8c83b6e --- /dev/null +++ b/src/components/launch/AdditionalCamerasList.tsx @@ -0,0 +1,99 @@ +import { Check } from "lucide-react"; +import type { CameraDevice } from "../../hooks/useCameraDevices"; +import styles from "./LaunchWindow.module.css"; + +/** Camera 1 plus this many more is the most a recording can hold. */ +export const MAX_ADDITIONAL_CAMERAS = 3; + +/** A picked extra camera, stored the way the recording prefs keep it. */ +export interface AdditionalCameraChoice { + id: string | null; + name: string; +} + +export interface AdditionalCamerasLabels { + title: string; + /** Shown while the list is disabled: why it cannot be used here. */ + hint: string; +} + +/** Class names, so the editor's Rec stage can restyle the list without a second component. */ +export interface AdditionalCamerasClasses { + title?: string; + item?: string; + itemActive?: string; + hint?: string; +} + +interface AdditionalCamerasListProps { + devices: CameraDevice[]; + /** Camera 1: never offered again here. */ + primaryDeviceId: string | undefined; + selected: AdditionalCameraChoice[]; + onChange: (next: AdditionalCameraChoice[]) => void; + disabled: boolean; + labels: AdditionalCamerasLabels; + classes?: AdditionalCamerasClasses; +} + +function isSameCamera(choice: AdditionalCameraChoice, device: CameraDevice): boolean { + return choice.id !== null ? choice.id === device.deviceId : choice.name === device.label; +} + +/** + * Checkbox list for cameras 2-4. Selection keeps the order of the clicks, because that order is + * the order the cameras get their `-webcam-N` files in. + */ +export function AdditionalCamerasList({ + devices, + primaryDeviceId, + selected, + onChange, + disabled, + labels, + classes, +}: AdditionalCamerasListProps) { + const offered = devices.filter((device) => device.deviceId !== primaryDeviceId); + // A saved pick whose camera is unplugged is neither shown nor counted: it cannot be recorded. + const present = selected.filter((choice) => + offered.some((device) => isSameCamera(choice, device)), + ); + const full = present.length >= MAX_ADDITIONAL_CAMERAS; + + const toggle = (device: CameraDevice) => { + if (disabled) return; + const isOn = present.some((choice) => isSameCamera(choice, device)); + if (isOn) { + onChange(present.filter((choice) => !isSameCamera(choice, device))); + } else if (!full) { + onChange([...present, { id: device.deviceId, name: device.label }]); + } + }; + + return ( + <> +
{labels.title}
+ {offered.map((device) => { + const isOn = present.some((choice) => isSameCamera(choice, device)); + const blocked = disabled || (!isOn && full); + return ( + + ); + })} + {disabled ?
{labels.hint}
: null} + + ); +} diff --git a/src/components/launch/HudDeviceSettings.tsx b/src/components/launch/HudDeviceSettings.tsx index 164b2f370..65c1a9931 100644 --- a/src/components/launch/HudDeviceSettings.tsx +++ b/src/components/launch/HudDeviceSettings.tsx @@ -6,6 +6,11 @@ import { useCameraPreviewStream } from "../../hooks/useCameraPreviewStream"; import type { MicrophoneDevice } from "../../hooks/useMicrophoneDevices"; import { WEBCAM_QUALITY_IDS, type WebcamQualityId } from "../../hooks/webcamCaptureTarget"; import { Tooltip } from "../ui/tooltip"; +import { + type AdditionalCameraChoice, + type AdditionalCamerasLabels, + AdditionalCamerasList, +} from "./AdditionalCamerasList"; import styles from "./LaunchWindow.module.css"; const LEVEL_SEGMENTS = 12; @@ -31,6 +36,15 @@ export interface HudDeviceSettingsLabels { cameraQualityOptions: Record; } +/** Cameras 2-4. Left out while camera 1 is off, since they are only recorded beside it. */ +export interface HudAdditionalCameras { + selected: AdditionalCameraChoice[]; + onChange: (next: AdditionalCameraChoice[]) => void; + /** No native Windows recording: the list stays visible, with its hint, but cannot be used. */ + disabled: boolean; + labels: AdditionalCamerasLabels; +} + /** Segmented input-level bar, driven by the live analyser. */ const LevelMeter = memo(function LevelMeter({ level }: { level: number }) { const lit = Math.round((Math.min(100, Math.max(0, level)) / 100) * LEVEL_SEGMENTS); @@ -101,6 +115,7 @@ export const HudDeviceSettings = memo(function HudDeviceSettings({ canCheckForUpdates, checkingForUpdates, cameraQuality, + additionalCameras, onSelectCameraQuality, onSelectMic, onSelectCamera, @@ -123,6 +138,7 @@ export const HudDeviceSettings = memo(function HudDeviceSettings({ canCheckForUpdates: boolean; checkingForUpdates: boolean; cameraQuality: WebcamQualityId; + additionalCameras?: HudAdditionalCameras; onSelectCameraQuality: (quality: WebcamQualityId) => void; onSelectMic: (device: MicrophoneDevice) => void; onSelectCamera: (device: CameraDevice) => void; @@ -223,6 +239,16 @@ export const HudDeviceSettings = memo(function HudDeviceSettings({ ); }) )} + {hasCamera && additionalCameras && cameraDevices.length > 1 ? ( + + ) : null} {hasCamera ? ( <> {/* Below the device list, because it qualifies the camera picked diff --git a/src/components/launch/LaunchWindow.module.css b/src/components/launch/LaunchWindow.module.css index 0f522b32e..59d700dbd 100644 --- a/src/components/launch/LaunchWindow.module.css +++ b/src/components/launch/LaunchWindow.module.css @@ -322,6 +322,16 @@ outline: none; } +.languageMenuItem:disabled { + opacity: 0.45; + cursor: not-allowed; +} + +.languageMenuItem:disabled:hover { + background: transparent; + color: rgba(255, 255, 255, 0.72); +} + .languageMenuItemActive { background: rgba(16, 185, 129, 0.14); color: #ffffff; diff --git a/src/components/launch/LaunchWindow.tsx b/src/components/launch/LaunchWindow.tsx index ca9dd076a..7aad7ccc4 100644 --- a/src/components/launch/LaunchWindow.tsx +++ b/src/components/launch/LaunchWindow.tsx @@ -12,11 +12,13 @@ import { type MicrophoneDevice, useMicrophoneDevices, } from "../../hooks/useMicrophoneDevices"; +import { useNativeWindowsCaptureAvailable } from "../../hooks/useNativeWindowsCaptureAvailable"; import { usePortalOwnsSource } from "../../hooks/usePortalOwnsSource"; import { useRememberedSourceName } from "../../hooks/useRememberedSourceName"; import { useScreenRecorder } from "../../hooks/useScreenRecorder"; import type { WebcamQualityId } from "../../hooks/webcamCaptureTarget"; import { requestCameraAccess } from "../../lib/requestCameraAccess"; +import type { AdditionalCameraChoice } from "./AdditionalCamerasList"; import { HudCameraButton, HudCursorButton, @@ -119,6 +121,8 @@ export function LaunchWindow() { setWebcamQuality, webcamDeviceName, setWebcamDeviceName, + webcamAdditionalDevices, + setWebcamAdditionalDevices, cursorCaptureMode, setCursorCaptureMode, softwareEncoderFallbackNoticeVisible, @@ -156,6 +160,7 @@ export function LaunchWindow() { */ const portalOwnsSource = usePortalOwnsSource(); + const nativeWindowsCapture = useNativeWindowsCaptureAvailable(); const isVertical = trayLayout === "vertical"; const isPopoverOpen = isLanguageMenuOpen || isDeviceSettingsOpen; const controlsLocked = recording || saving; @@ -861,6 +866,7 @@ export function LaunchWindow() { camDeviceId?: string; camDeviceName?: string; camQuality?: WebcamQualityId; + camAdditionalDevices?: AdditionalCameraChoice[]; micEnabled?: boolean; micDeviceId?: string; micDeviceName?: string; @@ -942,6 +948,17 @@ export function LaunchWindow() { [controlsLocked, persistRecordingPrefs, setWebcamQuality], ); + const handleChangeAdditionalCameras = useCallback( + (next: AdditionalCameraChoice[]) => { + // Same guard as the quality: a panel left open mid-take must not rewrite what the take + // was started with. + if (controlsLocked) return; + setWebcamAdditionalDevices(next); + persistRecordingPrefs({ camAdditionalDevices: next }); + }, + [controlsLocked, persistRecordingPrefs, setWebcamAdditionalDevices], + ); + const toggleDeviceSettings = useCallback(() => { if (controlsLocked) return; setIsLanguageMenuOpen(false); @@ -1095,6 +1112,28 @@ export function LaunchWindow() { [t, tCommon], ); + const additionalCameras = useMemo( + () => + webcamEnabled + ? { + selected: webcamAdditionalDevices, + onChange: handleChangeAdditionalCameras, + disabled: !nativeWindowsCapture, + labels: { + title: t("webcam.additionalCameras"), + hint: t("webcam.additionalCamerasHint"), + }, + } + : undefined, + [ + handleChangeAdditionalCameras, + nativeWindowsCapture, + t, + webcamAdditionalDevices, + webcamEnabled, + ], + ); + const versionLabel = appInfo ? t("deviceSettings.version", { version: appInfo.version }) : null; const hasNotices = Boolean(systemLocaleSuggestion) || softwareEncoderFallbackNoticeVisible; @@ -1321,6 +1360,7 @@ export function LaunchWindow() { canCheckForUpdates={(appInfo?.canCheckForUpdates ?? false) && !recording} checkingForUpdates={isCheckingForUpdates} cameraQuality={webcamQuality} + additionalCameras={additionalCameras} onSelectCameraQuality={handleSelectCameraQuality} onSelectMic={handleSelectMicDevice} onSelectCamera={handleSelectCameraDevice} diff --git a/src/hooks/useNativeWindowsCaptureAvailable.ts b/src/hooks/useNativeWindowsCaptureAvailable.ts new file mode 100644 index 000000000..1a6806c39 --- /dev/null +++ b/src/hooks/useNativeWindowsCaptureAvailable.ts @@ -0,0 +1,27 @@ +import { useEffect, useState } from "react"; +import { getPlatform } from "@/utils/platformUtils"; + +/** + * Whether this machine records through the native Windows helper, the only path that can hold + * more than one camera. `false` until the helper has answered, and for a failed probe: an option + * that cannot work is shown disabled rather than offered. + */ +export function useNativeWindowsCaptureAvailable(): boolean { + const [available, setAvailable] = useState(false); + + useEffect(() => { + const probe = window.electronAPI?.isNativeWindowsCaptureAvailable; + if (getPlatform() !== "win32" || !probe) return; + let cancelled = false; + void probe() + .then((result) => { + if (!cancelled) setAvailable(result.success && result.available); + }) + .catch(() => {}); + return () => { + cancelled = true; + }; + }, []); + + return available; +} diff --git a/src/hooks/useScreenRecorder.noCamera.test.tsx b/src/hooks/useScreenRecorder.noCamera.test.tsx index cd75f16b8..b2ba793d6 100644 --- a/src/hooks/useScreenRecorder.noCamera.test.tsx +++ b/src/hooks/useScreenRecorder.noCamera.test.tsx @@ -30,6 +30,7 @@ function prefs(camEnabled: boolean): RecordingPrefs { camEnabled, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", diff --git a/src/hooks/useScreenRecorder.prefsRace.test.tsx b/src/hooks/useScreenRecorder.prefsRace.test.tsx index ae84c2cd7..83d2197dd 100644 --- a/src/hooks/useScreenRecorder.prefsRace.test.tsx +++ b/src/hooks/useScreenRecorder.prefsRace.test.tsx @@ -25,6 +25,7 @@ function prefs(micEnabled: boolean): RecordingPrefs { camEnabled: false, camDeviceId: null, camDeviceName: null, + camAdditionalDevices: [], camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", diff --git a/src/hooks/useScreenRecorder.ts b/src/hooks/useScreenRecorder.ts index 915fcf959..a5a88cdfa 100644 --- a/src/hooks/useScreenRecorder.ts +++ b/src/hooks/useScreenRecorder.ts @@ -2,6 +2,7 @@ import { fixWebmDuration } from "@fix-webm-duration/fix"; import { useCallback, useEffect, useRef, useState } from "react"; import { toast } from "sonner"; import { useScopedT } from "@/contexts/I18nContext"; +import { type AdditionalCameraPick, resolveAdditionalWebcams } from "@/lib/additionalWebcams"; import { mixAudioTracks, nativeMicrophoneGain } from "@/lib/audioMix"; import { type NativeLinuxRecordingRequest, @@ -32,6 +33,16 @@ import { } from "./webcamCaptureTarget"; import { webcamDeviceIdentityFrom } from "./webcamDeviceIdentity"; +/** The cameras the system lists right now; empty when it cannot say. */ +async function listPresentCameras(): Promise> { + try { + const devices = await navigator.mediaDevices.enumerateDevices(); + return devices.filter((device) => device.kind === "videoinput"); + } catch { + return []; + } +} + const TARGET_FRAME_RATE = 60; const MIN_FRAME_RATE = 30; const TARGET_WIDTH = 3840; @@ -110,6 +121,9 @@ type UseScreenRecorderReturn = { setWebcamQuality: (quality: WebcamQualityId) => void; webcamDeviceName: string | undefined; setWebcamDeviceName: (deviceName: string | undefined) => void; + /** Cameras 2-4, in pick order. Recorded only by the native Windows path. */ + webcamAdditionalDevices: AdditionalCameraPick[]; + setWebcamAdditionalDevices: (devices: AdditionalCameraPick[]) => void; systemAudioEnabled: boolean; setSystemAudioEnabled: (enabled: boolean) => void; webcamEnabled: boolean; @@ -268,6 +282,11 @@ export function useScreenRecorder(): UseScreenRecorderReturn { useEffect(() => { tRef.current = t; }, [t]); + // Same reason as `tRef`: read by `finalizeNativeWindowsRecording`. + const tLaunchRef = useRef(tLaunch); + useEffect(() => { + tLaunchRef.current = tLaunch; + }, [tLaunch]); useEffect(() => { return window.electronAPI?.onNativeMacSystemAudioUnavailable?.(() => { toast.warning(tRef.current("recording.systemAudioUnavailable")); @@ -283,6 +302,9 @@ export function useScreenRecorder(): UseScreenRecorderReturn { const [webcamDeviceId, setWebcamDeviceId] = useState(undefined); const [webcamQuality, setWebcamQuality] = useState(DEFAULT_WEBCAM_QUALITY); const [webcamDeviceName, setWebcamDeviceName] = useState(undefined); + const [webcamAdditionalDevices, setWebcamAdditionalDevices] = useState( + [], + ); const [systemAudioEnabled, setSystemAudioEnabled] = useState(false); const [webcamEnabled, setWebcamEnabledState] = useState(false); const [cursorCaptureMode, setCursorCaptureMode] = useState("editable-overlay"); @@ -307,6 +329,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { camDeviceId?: string | null; camDeviceName?: string | null; camQuality?: WebcamQualityId | null; + camAdditionalDevices?: AdditionalCameraPick[] | null; systemAudioEnabled: boolean; cursorCaptureMode: CursorCaptureMode; }) => { @@ -325,6 +348,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { setWebcamDeviceName(prefs.camDeviceName ?? undefined); } setWebcamQuality(webcamQualityFrom(prefs.camQuality)); + setWebcamAdditionalDevices(prefs.camAdditionalDevices ?? []); setSystemAudioEnabled(prefs.systemAudioEnabled); setCursorCaptureMode(prefs.cursorCaptureMode); setRecordingPrefsLoaded(true); @@ -782,6 +806,13 @@ export function useScreenRecorder(): UseScreenRecorderReturn { if (result.webcamDropped) { toast.error(tRef.current("recording.cameraCaptureUnavailable")); } + if (result.droppedWebcams?.length) { + toast.error( + tLaunchRef.current("webcam.camerasNotRecorded", { + names: result.droppedWebcams.join(", "), + }), + ); + } if (result.session) { await window.electronAPI.setCurrentRecordingSession(result.session); } else if (result.path) { @@ -1258,6 +1289,13 @@ export function useScreenRecorder(): UseScreenRecorderReturn { // MediaRecorder against the helper's own process-spawn/WGC-init latency. stopWebcamPreviewStream(); } + const additionalWebcams = webcamEnabled + ? resolveAdditionalWebcams( + webcamAdditionalDevices, + await listPresentCameras(), + webcamIdentity.deviceId, + ) + : []; const request: NativeWindowsRecordingRequest = { recordingId: activeRecordingId, preferSoftwareEncoder: loadUserPreferences().preferSoftwareEncoder, @@ -1292,6 +1330,8 @@ export function useScreenRecorder(): UseScreenRecorderReturn { height: webcamPresetFor(webcamQuality).height, fps: WEBCAM_TARGET_FRAME_RATE, }, + // Cameras 2-4 ride along only with camera 1: with it off the helper gets no extras. + ...(additionalWebcams.length > 0 ? { additionalWebcams } : {}), cursor: { mode: cursorCaptureMode, }, @@ -1308,6 +1348,11 @@ export function useScreenRecorder(): UseScreenRecorderReturn { if (result.webcamUnavailable) { toast.error(t("recording.cameraCaptureUnavailable")); } + if (result.unavailableWebcams?.length) { + toast.error( + tLaunch("webcam.camerasNotRecorded", { names: result.unavailableWebcams.join(", ") }), + ); + } if (result.microphoneDefaulted) { toast.error(t("recording.microphoneDefaulted")); } @@ -2454,6 +2499,8 @@ export function useScreenRecorder(): UseScreenRecorderReturn { setWebcamQuality, webcamDeviceName, setWebcamDeviceName, + webcamAdditionalDevices, + setWebcamAdditionalDevices, systemAudioEnabled, setSystemAudioEnabled, webcamEnabled, diff --git a/src/i18n/locales/ar/launch.json b/src/i18n/locales/ar/launch.json index ef613d15d..858d9d8e9 100644 --- a/src/i18n/locales/ar/launch.json +++ b/src/i18n/locales/ar/launch.json @@ -48,6 +48,9 @@ "camera": "الكاميرا", "cameraDevice": "جهاز الكاميرا", "quality": "جودة الكاميرا", + "additionalCameras": "كاميرات إضافية", + "additionalCamerasHint": "فقط مع التسجيل الأصلي في Windows", + "camerasNotRecorded": "لم يتم تسجيلها: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/cs/launch.json b/src/i18n/locales/cs/launch.json index d9912da51..d79eb409b 100644 --- a/src/i18n/locales/cs/launch.json +++ b/src/i18n/locales/cs/launch.json @@ -48,6 +48,9 @@ "camera": "Kamera", "cameraDevice": "Zařízení kamery", "quality": "Kvalita kamery", + "additionalCameras": "Další kamery", + "additionalCamerasHint": "Pouze s nativním nahráváním ve Windows", + "camerasNotRecorded": "Nenahráno: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/de/launch.json b/src/i18n/locales/de/launch.json index b4dab82a9..2ba2310d7 100644 --- a/src/i18n/locales/de/launch.json +++ b/src/i18n/locales/de/launch.json @@ -48,6 +48,9 @@ "camera": "Kamera", "cameraDevice": "Kameragerät", "quality": "Kameraqualität", + "additionalCameras": "Weitere Kameras", + "additionalCamerasHint": "Nur bei nativer Windows-Aufnahme", + "camerasNotRecorded": "Nicht aufgenommen: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/en/launch.json b/src/i18n/locales/en/launch.json index 97263f184..2489623aa 100644 --- a/src/i18n/locales/en/launch.json +++ b/src/i18n/locales/en/launch.json @@ -48,6 +48,9 @@ "camera": "Camera", "cameraDevice": "Camera device", "quality": "Camera quality", + "additionalCameras": "Additional cameras", + "additionalCamerasHint": "Only with native Windows recording", + "camerasNotRecorded": "Not recorded: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/es/launch.json b/src/i18n/locales/es/launch.json index 3114921ec..565d4a852 100644 --- a/src/i18n/locales/es/launch.json +++ b/src/i18n/locales/es/launch.json @@ -48,6 +48,9 @@ "camera": "Cámara", "cameraDevice": "Dispositivo de cámara", "quality": "Calidad de la cámara", + "additionalCameras": "Cámaras adicionales", + "additionalCamerasHint": "Solo con la grabación nativa de Windows", + "camerasNotRecorded": "No grabadas: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/fr/launch.json b/src/i18n/locales/fr/launch.json index c956fb37a..0cc342866 100644 --- a/src/i18n/locales/fr/launch.json +++ b/src/i18n/locales/fr/launch.json @@ -48,6 +48,9 @@ "camera": "Caméra", "cameraDevice": "Périphérique caméra", "quality": "Qualité de la caméra", + "additionalCameras": "Caméras supplémentaires", + "additionalCamerasHint": "Uniquement avec l'enregistrement natif sous Windows", + "camerasNotRecorded": "Non enregistrées : {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/it/launch.json b/src/i18n/locales/it/launch.json index 9897b92eb..2290d693a 100644 --- a/src/i18n/locales/it/launch.json +++ b/src/i18n/locales/it/launch.json @@ -48,6 +48,9 @@ "camera": "Fotocamera", "cameraDevice": "Dispositivo fotocamera", "quality": "Qualità della fotocamera", + "additionalCameras": "Fotocamere aggiuntive", + "additionalCamerasHint": "Solo con la registrazione nativa di Windows", + "camerasNotRecorded": "Non registrate: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/ja-JP/launch.json b/src/i18n/locales/ja-JP/launch.json index 9f2323eec..4d7fcf3ce 100644 --- a/src/i18n/locales/ja-JP/launch.json +++ b/src/i18n/locales/ja-JP/launch.json @@ -48,6 +48,9 @@ "camera": "カメラ", "cameraDevice": "カメラデバイス", "quality": "カメラの画質", + "additionalCameras": "追加のカメラ", + "additionalCamerasHint": "Windows のネイティブ録画でのみ使用できます", + "camerasNotRecorded": "録画されませんでした: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/ko-KR/launch.json b/src/i18n/locales/ko-KR/launch.json index 5ce2b8e72..641fb53d8 100644 --- a/src/i18n/locales/ko-KR/launch.json +++ b/src/i18n/locales/ko-KR/launch.json @@ -48,6 +48,9 @@ "camera": "카메라", "cameraDevice": "카메라 장치", "quality": "카메라 화질", + "additionalCameras": "추가 카메라", + "additionalCamerasHint": "Windows 기본 녹화에서만 사용할 수 있습니다", + "camerasNotRecorded": "녹화되지 않음: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/pt-BR/launch.json b/src/i18n/locales/pt-BR/launch.json index 886615088..276d494e0 100644 --- a/src/i18n/locales/pt-BR/launch.json +++ b/src/i18n/locales/pt-BR/launch.json @@ -48,6 +48,9 @@ "camera": "Câmera", "cameraDevice": "Dispositivo de câmera", "quality": "Qualidade da câmera", + "additionalCameras": "Câmeras adicionais", + "additionalCamerasHint": "Somente com a gravação nativa do Windows", + "camerasNotRecorded": "Não gravadas: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/ru/launch.json b/src/i18n/locales/ru/launch.json index cdc103c0b..1d99fd9b5 100644 --- a/src/i18n/locales/ru/launch.json +++ b/src/i18n/locales/ru/launch.json @@ -48,6 +48,9 @@ "camera": "Камера", "cameraDevice": "Устройство камеры", "quality": "Качество камеры", + "additionalCameras": "Дополнительные камеры", + "additionalCamerasHint": "Только при встроенной записи в Windows", + "camerasNotRecorded": "Не записаны: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/tr/launch.json b/src/i18n/locales/tr/launch.json index 530157dd2..22c00b419 100644 --- a/src/i18n/locales/tr/launch.json +++ b/src/i18n/locales/tr/launch.json @@ -48,6 +48,9 @@ "camera": "Kamera", "cameraDevice": "Kamera cihazı", "quality": "Kamera kalitesi", + "additionalCameras": "Ek kameralar", + "additionalCamerasHint": "Yalnızca yerel Windows kaydıyla", + "camerasNotRecorded": "Kaydedilmedi: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/vi/launch.json b/src/i18n/locales/vi/launch.json index 53f452e27..0416dce15 100644 --- a/src/i18n/locales/vi/launch.json +++ b/src/i18n/locales/vi/launch.json @@ -48,6 +48,9 @@ "camera": "Máy ảnh", "cameraDevice": "Thiết bị máy ảnh", "quality": "Chất lượng camera", + "additionalCameras": "Camera bổ sung", + "additionalCamerasHint": "Chỉ với tính năng quay gốc của Windows", + "camerasNotRecorded": "Không được ghi: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/zh-CN/launch.json b/src/i18n/locales/zh-CN/launch.json index 7a61e466e..03cb0f418 100644 --- a/src/i18n/locales/zh-CN/launch.json +++ b/src/i18n/locales/zh-CN/launch.json @@ -48,6 +48,9 @@ "camera": "摄像头", "cameraDevice": "摄像头设备", "quality": "摄像头画质", + "additionalCameras": "附加摄像头", + "additionalCamerasHint": "仅支持 Windows 原生录制", + "camerasNotRecorded": "未录制:{{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/zh-TW/launch.json b/src/i18n/locales/zh-TW/launch.json index 2e460592e..97bb58696 100644 --- a/src/i18n/locales/zh-TW/launch.json +++ b/src/i18n/locales/zh-TW/launch.json @@ -48,6 +48,9 @@ "camera": "攝影機", "cameraDevice": "攝影機裝置", "quality": "攝影機畫質", + "additionalCameras": "額外攝影機", + "additionalCamerasHint": "僅限 Windows 原生錄影", + "camerasNotRecorded": "未錄製:{{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/lib/additionalWebcams.test.ts b/src/lib/additionalWebcams.test.ts new file mode 100644 index 000000000..f3889a36f --- /dev/null +++ b/src/lib/additionalWebcams.test.ts @@ -0,0 +1,60 @@ +import { describe, expect, it } from "vitest"; +import { resolveAdditionalWebcams } from "./additionalWebcams"; + +const present = [ + { deviceId: "a", label: "Cam A" }, + { deviceId: "b", label: "Cam B" }, + { deviceId: "c", label: "Cam C" }, + { deviceId: "d", label: "Cam D" }, + { deviceId: "e", label: "Cam E" }, +]; + +describe("resolveAdditionalWebcams", () => { + it("keeps the order of the picks and reports the current id and label", () => { + expect( + resolveAdditionalWebcams( + [ + { id: "c", name: "Cam C" }, + { id: "b", name: "Cam B" }, + ], + present, + "a", + ), + ).toEqual([ + { deviceId: "c", deviceName: "Cam C" }, + { deviceId: "b", deviceName: "Cam B" }, + ]); + }); + + it("drops cameras that are no longer present and falls back to the name when the id changed", () => { + expect( + resolveAdditionalWebcams( + [ + { id: "gone", name: "Unplugged" }, + { id: "old-id", name: "Cam B" }, + ], + present, + "a", + ), + ).toEqual([{ deviceId: "b", deviceName: "Cam B" }]); + }); + + it("never repeats camera 1 or a camera picked twice", () => { + expect( + resolveAdditionalWebcams( + [ + { id: "a", name: "Cam A" }, + { id: "b", name: "Cam B" }, + { id: null, name: "Cam B" }, + ], + present, + "a", + ), + ).toEqual([{ deviceId: "b", deviceName: "Cam B" }]); + }); + + it("caps at three", () => { + const picks = ["b", "c", "d", "e"].map((id) => ({ id, name: `Cam ${id.toUpperCase()}` })); + expect(resolveAdditionalWebcams(picks, present, "a")).toHaveLength(3); + }); +}); diff --git a/src/lib/additionalWebcams.ts b/src/lib/additionalWebcams.ts new file mode 100644 index 000000000..c386821f1 --- /dev/null +++ b/src/lib/additionalWebcams.ts @@ -0,0 +1,36 @@ +export interface AdditionalCameraPick { + id: string | null; + name: string; +} + +export interface PresentCamera { + deviceId: string; + label: string; +} + +const MAX_ADDITIONAL_WEBCAMS = 3; + +/** + * The cameras 2-4 of a native Windows request: the saved picks that are still plugged in, under + * the id and label the system reports now (an id can change between sessions, the pick is + * matched by id first and by name second). Camera 1, repeats and anything past the cap are left + * out; the order of the picks is kept. + */ +export function resolveAdditionalWebcams( + picks: AdditionalCameraPick[], + present: PresentCamera[], + primaryDeviceId: string | undefined, +): Array<{ deviceId: string; deviceName: string }> { + const resolved: Array<{ deviceId: string; deviceName: string }> = []; + for (const pick of picks) { + if (resolved.length >= MAX_ADDITIONAL_WEBCAMS) break; + const device = + (pick.id !== null + ? present.find((candidate) => candidate.deviceId === pick.id) + : undefined) ?? present.find((candidate) => candidate.label === pick.name); + if (!device || device.deviceId === primaryDeviceId) continue; + if (resolved.some((entry) => entry.deviceId === device.deviceId)) continue; + resolved.push({ deviceId: device.deviceId, deviceName: device.label }); + } + return resolved; +} From 66bb3b51d36c3d8ab628e08b4c3c280de58aceb8 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:43:20 +0200 Subject: [PATCH 11/22] fix(hud): note why the capture probe failed --- src/hooks/useNativeWindowsCaptureAvailable.ts | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/src/hooks/useNativeWindowsCaptureAvailable.ts b/src/hooks/useNativeWindowsCaptureAvailable.ts index 1a6806c39..c6a0faf57 100644 --- a/src/hooks/useNativeWindowsCaptureAvailable.ts +++ b/src/hooks/useNativeWindowsCaptureAvailable.ts @@ -17,7 +17,9 @@ export function useNativeWindowsCaptureAvailable(): boolean { .then((result) => { if (!cancelled) setAvailable(result.success && result.available); }) - .catch(() => {}); + .catch((error) => { + console.warn("Could not probe native Windows capture:", error); + }); return () => { cancelled = true; }; From a8fea4b2b65f1feb41d14710ccac3d48057723ec Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 12:50:00 +0200 Subject: [PATCH 12/22] docs(e2e): several cameras on Windows --- .../testing/manual-e2e-checklist.md | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/technical-documentation/testing/manual-e2e-checklist.md b/technical-documentation/testing/manual-e2e-checklist.md index 60095b9e2..ec4a4fd09 100644 --- a/technical-documentation/testing/manual-e2e-checklist.md +++ b/technical-documentation/testing/manual-e2e-checklist.md @@ -219,6 +219,18 @@ The helper waits up to 3 s for the camera's first visible frame before starting - [ ] Take two recordings of one scene, holding a page of small printed text at arm's length: one at 640x480, from a camera that only offers 640x480 or from a release older than #875, and one at 1080p or 4K. Add a Full Camera segment over the same moment in each project, export MP4 at 1080p, extract a frame with `ffmpeg -ss -i -frames:v 1 frame.png`, and view both at 100%. The 1080p take resolves the text and the 640x480 take, upscaled about threefold, does not. With neither reference at hand, log `skipped: no 640x480 reference`. +### Several cameras (Windows) + +Up to four cameras record into one take on the native Windows path: camera 1 as today, up to three more picked under the HUD's *Additional cameras*. macOS, Linux and the browser recorder stay single-camera, and the section shows disabled with a hint there. Each extra camera is its own file beside the recording: `-webcam-2.mp4`, `-webcam-3.mp4`, `-webcam-4.mp4`. + +- [ ] Several cameras (Windows). Pick a second camera under *Additional cameras*, record about 10 s and clap in front of both cameras. Both camera files exist next to the recording (`-webcam.mp4`, `-webcam-2.mp4`) and both play. The clap lands at the same time in both and against the screen. The editor shows camera 1 as before. After saving and reopening, the project still lists both cameras. +- [ ] Order of the runs: start with two cameras of different names (Logitech Brio plus C920), then two identical Brios. Record at 1080p first, and at 4K afterwards as a load test: two 4K cameras on one USB controller may lack the bandwidth, so a camera that drops out there is a finding about the controller first and the app second. Log the controller layout next to the result. +- [ ] With two identical Brios, which physical Brio becomes camera 1 follows Windows' enumeration order, not the HUD choice or its preview. A swap against the preview is a known limitation, not a defect. Wave a hand into one camera at a time to tell the files apart, and note which is which. +- [ ] Read the helper output (`Save diagnostics`, or `(Get-Content diag.json -Raw | ConvertFrom-Json).helperOutput.windows -split "`n" | Select-String "webcam"`). Camera 2's `Native webcam candidate` lines list the second Brio, and neither the IR or any other interface of the first Brio nor a candidate marked `already recording in this take`. Each camera prints its own `webcam-format` event with its `index`, and `recording-stopped` carries both paths in `webcamPaths`. +- [ ] A camera that cannot be opened or records nothing is named in a "Not recorded: …" notice after the take, and the take is kept with the cameras that worked. Provoke it by unplugging the second camera, or holding it open in another app, just before *Start recording*, and check the helper output for `{"event":"warning","code":"webcam-unavailable","index":1,…}`. Camera 1 and the screen are unaffected. +- [ ] With the first camera turned off in the HUD, record again and confirm no additional camera is recorded (additional cameras follow camera 1). +- [ ] Probe every file: `ffprobe -show_entries format=duration` of each camera file is within 0.2 s of the screen file's. + ## Editor opens and loads the project - [ ] Confirm the editor opens after a successful stop, in Edit mode, with the expected project title and asset and a "Recording added to a new project" toast. @@ -854,3 +866,4 @@ On macOS 15.2+ it opens at launch, until it has been closed once, while one of i | 2026-10-03 | installed `v2.0.0-rc.13` (NSIS from the release, sha512 matched `latest.yml`), installed over rc.12 for all users, About reports `2.0.0-rc.13` | Windows 11 Home 26200, 1920×1080 @ 125 %, AMD Radeon iGPU, built-in USB2.0 HD UVC WebCam | Partial — 1 defect | **Scoped to the app changes between rc.8, the last computer-use pass, and rc.13**; the export path itself is covered by the rc.13 row above. Real OS mouse and keyboard input throughout; computer-use screenshots came back uniformly grey again, so the app was observed through `PrintWindow` on its own windows and DOM reads over CDP. **Defect: the webcam shows its first frame when a clip is entered past the webcam's last frame from another clip.** Two clips of one take (screen video 50.98 s, webcam 51.24 s, audio and clip 51.31 s): scrubbing from clip 2 to 51.2 s in clip 1 shows the screen's last frame, as 227b9b84 intends, but the webcam's frame 0 (2 of 2; ground truth from ffmpeg: an empty chair from 18.7 s to the end). Scrubbing to the same point from inside clip 1, or to 51.0 s from clip 2, shows the right webcam frame. Cause: `open_and_seek_clip` and `seek_pair` in `crates/compositor/src/live.rs` seek the webcam with `seek_to` and fall back to `seek_to(0.0)`, where `present_frame` uses `seek_to_or_last`; unchanged on `main`. Paused scrub only, in a ~70 ms window at the end of each clip. **Passed:** #961: one take with the countdown overlay, Record mode, editor work and an export, then closing the editor left 0 processes. #966: after rebinding *Add zoom* to Q and *Add audio* to J, the empty-lane hints read "Appuyez sur Q pour ajouter un zoom" and "Appuyez sur J pour ajouter un audio, V pour enregistrer une voix off", and Q adds a zoom. Wordmark menu in French: every row 31 px (one line), menu 261 px wide, "2.0.0-rc.13" in one piece. Preview: playback from 0:12 to 0:45.8 across a trim (0:15.0 → 0:18.5) and a 1.5× region with no canvas stall over 150 ms, 32–50 picture changes a second, wall time matching the edit; stops at 0:51.3 on the screen's and the webcam's last frames. #960 (#964): A → B → A with the playhead at 0:06, the preview is pixel-identical to before (mean difference 0, against 77.9 for project B). Background image → gradient → colour followed in the preview. #968: a media file whose read the OS refused (`icacls` deny on the test file) — *Regenerate* shows the translated "Impossible de lire l'audio de ce média…" hint inline and in the "Échec de la transcription" toast, without the IPC wrapper; the main process logs `MediaUnreadableError … Permission denied`. MP4 export from the GUI, 1752×1080 30 fps: 2930 frames, video 97.633 s against audio 97.666 s, zoom, background, webcam and cursor present in extracted frames, −18.3 LUFS. **Also seen:** the green "Aucune parole détectée" chip from the earlier successful run stays next to the unreadable hint; each clip of the take still carries 0.32 s of audio past its video (what is left of #942), and the export ends each clip at its video (97.63 s, against 98.22 s from the clip lengths the timeline shows). **Not run:** the HUD with no camera (#967; a camera is attached), the macOS HUD fix (c2a8559e), tray, microphone and system-audio content, GIF, macOS and Linux. | | 2026-10-03 | installed `v2.0.0-rc.13`, CI-built DMG `Openscreen-macOS-Apple-Silicon-2.0.0-rc.13.dmg` from the GitHub Release (tag `018d22ab`), installed over rc.12, `spctl`: accepted, Notarized Developer ID (M4LK7C6S84), About reports `2.0.0-rc.13` | macOS 26.5 (25F71), Mac mini M1, 3840×2160, no camera, no microphone | Partial, no defect | **Scoped to the app changes between rc.8 and rc.13, on macOS**: the Windows row above covers the same diff on Windows and leaves out #967, c2a8559e and macOS, which this row takes. Real OS mouse and keyboard input through computer-use (screenshots worked on this Mac); Apple's picker is owned by Control Centre and cannot be granted, so the operator clicked *Share Entire Screen* for each take. **c2a8559e (HUD kept out of takes after the editor round trip):** take 1 picked from the HUD's record button (`excludedWindowIds: [582]`), stop, editor, Record mode showed the source as *Choose what to record*, *Start recording* reopened Apple's picker instead of reusing the pick, take 2 recorded with `excludedWindowIds: [639]`; frames extracted from both files (28:26 and 13:04) show no HUD where it sat. **#967 (no camera on this Mac):** with `camEnabled: true` and no camera identity in `recording-settings.json`, launch rewrote it to `false` and the HUD showed the camera off; same with a stale identity (`camDeviceId` + `camDeviceName` of an unplugged camera); clicking the toggle keeps it off with the "Camera access is blocked" toast; Record mode reads *Camera Off*. **#961:** closing the editor after take 2 (countdown overlay, Record mode handover) left 0 processes; so did closing it after a long editor session (eight project opens through the native panel, transcriptions, Media mode, an export, the export dialog's save panel cancelled). **#966:** with *Add zoom* bound to X in `shortcuts.json`, the empty zoom lane reads "Press X to add zoom", and "Appuyez sur X pour ajouter un zoom" / "Drücke X, …" in French and German. **Wordmark menu (3bee2d25):** French and German, every row on one line, "2.0.0-rc.13" in one piece. **#960 (b834416c):** project A with captions → project B → A again: captions show at once at 0:06.4, 0:10.2 and 1:05.2 without touching *Show captions* (0:05.3 shows none, a 4.77 to 5.57 s silence in the saved words). **Preview:** playback from 0:12 to the end across four transcript silence cuts, a 2× and a 16× region: picture and captions follow, the readout advances with the edit, stops at 1:06.2; no stall seen, 8 `[pipeline] décodage` opens for one seek and four cuts (consistent with a preload per cut; not compared with rc.12). **227b9b84 / 83fbcecc:** a two-clip project whose first clip is a file with 8 s of video and 10 s of audio (no camera): seeking from clip 2 to 0:09.5 shows clip 1's last frame, the same as 0:07.7, no stderr. A video-only file plays with its picture (e73097c8). **#968:** project whose first asset is `chmod 000`: `MediaUnreadableError … Permission denied` in the main log, clip 2 still transcribed, the asset card shows "Transcription failed" and "Couldn't read this media's audio. Make sure OpenScreen can access the folder it's in, then regenerate." without the IPC wrapper; after `chmod 644`, *Regenerate* runs again (not persisted as no-audio). **Metal compositor (8caaa20d, dd5e914b, 0624c148, 2fb994eb):** laptop frame, 3D cursor, click impact, image background blurred 53 %, Aurora, captions; MP4 1080p60 export of 3374 frames, video and audio 56.23 s against 56.21 s computed (66.154 s minus 5.09 s of cuts minus the 2× and 16× gains), full decode clean; extracted frames show the laptop with its shadow, the blurred background, the 3D cursor with its shadow, rounded screen corners and captions. During playback the preview followed image → image, blur 52 → 1 %, image → colour and back. **Also seen:** the playhead keeps its time when another project is opened (0:41.0 from A shown in B and C). With a hand-written `camDeviceName` and a null `camDeviceId`, the HUD keeps the camera on and the file at `true` (`useCameraHudSync` compares the empty selection with the empty id and returns early); the app always writes both together, so this state is artificial. Out of scope: a 10 s extract with speech under loud video sound transcribes as no speech in Auto (detected `en`, p 0.35) and as "*Musique*" in French. **Not run:** webcam and microphone (none on this Mac), so the Windows row's webcam-past-the-end defect cannot be checked here; click impact not confirmed (the fixture's cursor sidecar has no clicks, and none was identified around the nine clicks of take 2 at preview scale); Aurora motion on a plain colour; the 4× preview lag of the rc.12 row (not re-measured: this pass played 2× and 16× regions without timing the lag); `Esc`; tray; GIF; A/B against rc.12; Gatekeeper (the DMG came through `gh`, no quarantine flag); Intel. | | 2026-10-03 | installed `v2.0.0-rc.15` (build run 37124242835, tag `ccd8b449`, NSIS sha512 matched `latest.yml`), installed over rc.12 for all users, About reports `2.0.0-rc.15` (`win32 x64 · nsis`, Electron 41.2.1, Chromium 146.0.7680.188, Node 24.14.1). `v2.0.0` was promoted from this candidate during the pass and differs from it only by the version bump (`eff86dcb`), so the results hold for the stable code | Windows 11 Home 26200, 1920×1080 @ 125 %, AMD Ryzen 5 7520U / Radeon iGPU, built-in USB2.0 HD UVC WebCam (best mode 1280×720@30), Microphone Array (AMD) | Partial — 6 defects | **Whole-file pass on Windows.** Real OS mouse and keyboard input throughout. Computer-use screenshots came back grey again, so the app was observed through `PrintWindow` on its own windows and DOM reads over CDP, and every picture and sound claim below was measured on the file with ffprobe/ffmpeg. **Defects:** (1) #1004, HUD drag at 125 %: the HUD window grows about 1.5 px per drag step (1180×882 → 1232×935 over ~43 steps) and the bar jumps by half the growth on release; after a long drag down the bar sat partly under the taskbar (centre y 1042, work area ends at 1032). (2) #1005, *Edit clip*: Reset, Cancel and Apply sit inside the scrolling body. In a 1240×1000 window (794 css px tall) the card is clipped above Apply, and a click where Apply should be lands on the backdrop, closing the dialog and discarding the edit. It fits when the window is maximized. (3) #1006, the rail's "Choose a clip to edit" menu renders under the timeline toolbar: Clip 2 and Clip 3 are covered (`elementFromPoint` returns `_tlToolbar`) and clicks on them do nothing. (4) #1007, a clip duplicated with Ctrl+C / Ctrl+V does not carry its anchored zoom or Full Camera region: `duplicateClip` in `src/lib/ai-edition/document/timeline.ts` copies only `trimRanges`. (5) #1008, zoom repel across a clip boundary: a zoom dragged into a neighbour that ends at its clip's end jumps past it into the next clip, shrinking from 12.97 to 8.23 s and leaving a zero-length fragment on the first clip. Ctrl+Z restores it. (6) #1009, the wordmark menu does not close on a click on the bare top bar. The header is a `-webkit-app-region: drag` area, which never delivers the document `mousedown` that `AppMenu` listens for. It does close on the preview, on a top-bar button and on the wordmark. **Passed: HUD and capture.** One HUD, `[content-protection] OFF` logged; horizontal ↔ vertical layout; every tooltip, idle and recording, beside or above the bar and unclipped; language menu (15 locales); camera and microphone toggles in one click; device settings: three inputs, level meter, live camera preview, camera quality row (1080p persisted across a relaunch; hand-edited 720p and an absent key both read 4K). Hide bar, then the tray overflow icon brings the HUD back; Quit leaves 0 processes; a second launch exits 0. With no source the record tooltip reads "Choose a screen or window to record" and one click opens the picker, then starts. While recording: source button disabled, toggles `aria-disabled`, gear inert, tray tooltip "Recording: Tout l'écran". Pause freezes the timer and resume advances it; restart starts a new helper and deletes the first file; cancel deletes the file and opens no editor. Stop opens the editor with "Recording added to a new project". Screen H.264 High bt709 1920×1080; AAC 48 kHz; webcam 1280×720 ~30 fps at 8 Mbit/s, the camera's best mode, not upscaled, matching `webcamFormat`. Microphone-only, system-audio-only and all-off takes behave; window source 1270×668, non-black; an odd 1001×601 window gives an even 992×596 file; a context menu inside the window is recorded and cut at its edge. Without the flag the HUD and an open Notes window are absent from the take. Record mode: rows, live camera preview, microphone meter, device switch, source modal with badge, *Start recording* hands over to the HUD; auto-zoom on gives one merged zoom over three clicks, off gives none, and the setting persists; editable cursor off hides the auto-zoom row and the Cursor facet; *Hide desktop icons* shows a bare wallpaper and restores the icons after. **Passed: editor.** Rename, Saved dot, tabs, chat panel toggle, tooltips with key chips, dividers; play, pause, seek playing and paused, ±1 frame arrows, stop at end; ruler click and drag, navigator narrow and pan, labels without collision, playhead within 0.5 px; Ctrl+S toast with a stable top bar. Clips: media card drag and *Add to timeline*, reorder, Edit clip grips, 1:1 crop held through a corner drag, Free, Cancel, pencil, delete. Trim, zoom (levels, custom, out-of-range message, focus drag, 3D modes), speed (presets, custom, 16× cap message, playback at 16×), annotation (text, sizes, plates, palette, Typewriter, image, arrow, blur mosaic and smooth, oval, clamped, dragged into padding); Ctrl+D, Ctrl+Z, Ctrl+Shift+Z. Modifiers inside a trim fire on the parked frame. Copy and paste of zoom, annotation and trim with their toasts and properties; an empty copy is a no-op; deleting a clip removes its modifiers; Edit clip clamps them; a Full Camera region across a junction splits and merges as clips move. Imported music: fragments across a junction with `offsetMs`, −18 dB and 1 s fades, gain, *Reset audio*; voiceover lands at the playhead. Transcript: clip order, silences, monotonic word seeks, live cue, Backspace skip and restore, silence trim; captions in four styles and six anchors; French translation with a *Display* row, and deleting it leaves the words untouched; typing in the packaged gate does nothing; a double-click correction survives *Regenerate as English* (model reused); 101 languages, detected language on the card. Composition: colour, gradient and one-colour gradient, wallpaper, animations move during playback only (21–23 distinct frames in 3 s, None 1), blur, 9:16 with Whole and Follow cursor, Auto padding with even borders, shadows, four frame styles in both themes. Camera layout: four presets, mirror, square, size (default 40, max 60), roundness, positions; background Cutout, Blur with an intensity slider, Custom image and colour, and the custom-colour mask reaches a 720p export; webcam crop corner zoom (124 %) and frame move. Audio facet: +6.5 dB raises the export from −17.2 to −12.7 LUFS with the peak held at −1.3 dBFS; reset returns 0 dB. Cursor facet: hide, style, size (to 5.8 of 6) and 3D cursor all show in the preview. Depth of field appears once a zoom has a 3D preset and changes the preview. Wordmark menu: rows as listed, About row version, arrows wrap both ways; Keyboard Shortcuts as one dialog: Z → Q changes the lane hint, Q adds a zoom, Z does nothing, reset restores Z; Ctrl+O; AI settings as one dialog; light and dark themes, dialogs included; French UI with one-line menu rows (#969), back to English; About lists the versions and channel, and Copy puts the same block on the clipboard; the HUD panel's version row and Check for Updates ("Checking…", disabled, the same dialog as the menu, re-enabled after). AI chat (deepseek): "add one zoom at 2x from 5 s to 8 s" lands as 5–8 s at 2.20× with the reason given, is undone by one Ctrl+Z and redone. Closing the editor quits the app and writes `editor-window.json` (`maximized: false`); after a relaunch the last project's settings and the main project's two clips, crop, regions, music, voiceover and captions are as left, and a paste before any copy changes nothing. A schema-4 project from July opens, migrates to schema 8, and keeps its clip, seven zooms and annotation. **Export:** MP4 1080p30 of the two-clip project, 10288 frames, audio and video within one frame, −16.0 LUFS and −1.4 dBTP, annotation upright, caption plate at x = 108 for a 10 % inset, Full Camera where mapped, music ducked about 10 dB under speech with fades at the track's own edges, muted track absent, voiceover not stretched by a 2× region. GIF 25 fps: cancel removes the partial file and leaves the existing destination untouched; the full render (~127 s, #952) gives 434 frames, infinite loop, consistent in ffmpeg and GDI+. **Also seen:** a take with a ~4 s pause had audio 1.09 s past its video, against 0.02–0.17 s without a pause; three dedicated paused takes gave 0.40, 0.15 and 0.03 s. The export ends each clip at its last video frame (10288 frames against the 10320 the progress counted), so the timeline and the export disagree by that tail. The editor reopens 4 px taller than it closed at 125 %. **Not run:** the A/V offset across a pause measured directly (the flash-and-beep rig never painted on the captured desktop), the HUD with no camera (needs the camera disabled in Windows), tray right-click menus (Update Settings, Stop Recording, Save Diagnostics: Explorer is granted click-only), the Alt Help menu (an Alt tap showed nothing, inconclusive), Ctrl/Shift+wheel on the timeline, cursor auto-hide, smoothing, motion blur and click effects, annotation animations other than Typewriter, Depth of field in an export, chat rewind, history and Smart cuts, a software-encoder take, the crackle and hole scan, multiple displays, 4K and 1440p cameras, 150 % DPI, macOS and Linux. | +| 2026-10-04 | dev `feat/multi-camera-recording` at `212c07f7` (feature head `23ea0751` + the HUD probe lint fix); `wgc-capture.exe` rebuilt from this tree by `npm run build:native:win` | Windows 11 on ARM64 (Snapdragon X Elite), one built-in camera (`AI Front Camera`), driven without computer-use | **Partial — the two-webcam run on site is outstanding; sub-project not complete until logged** | **Ran:** unit suite (`npm run test`, 304 of 305 files, the one failure `LeftPanel.copyMessage.test.tsx` passed alone, a load flake), both `tsc`, lint (0 errors, 26 warnings, the base count), `i18n:check`; the helper's native unit tests (`node scripts/build-windows-wgc-helper.mjs`, all pass); helper smoke runs `npm run test:wgc-helper:win`, `-- --webcam` and `-- --webcam --missing-second-webcam` (index 1 dropped and named, camera 1 kept); a direct helper run listing the one camera twice: camera 1 recorded, camera 2 dropped with `webcam-unavailable` index 1 and "already recording in this take", `webcamPaths` held one file, exit 0, camera file 4.81 s against the screen's 4.80 s. **Not run:** the app click-through (HUD *Additional cameras*, editor, save and reopen, the session manifest's `additionalWebcams`, the "Not recorded" notice): no computer-use here and only one camera. **Not covered at all:** two real cameras recording together, identical-name Brios, 1080p and 4K load, clap sync, macOS and Linux. | From 1607b29ce2422518d18fb74e26dbb3be676e0bf3 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:18:35 +0200 Subject: [PATCH 13/22] fix(project): no fixed limit on additional camera tracks The recorder caps at three extra cameras, but a validator that ships with a cap cannot be loosened later for older builds. The HUD and Electron still cap. --- src/lib/ai-edition/schema/index.test.ts | 8 ++++---- src/lib/ai-edition/schema/index.ts | 4 +++- 2 files changed, 7 insertions(+), 5 deletions(-) diff --git a/src/lib/ai-edition/schema/index.test.ts b/src/lib/ai-edition/schema/index.test.ts index bb8148e9c..adc99c418 100644 --- a/src/lib/ai-edition/schema/index.test.ts +++ b/src/lib/ai-edition/schema/index.test.ts @@ -295,11 +295,11 @@ describe("axcut-schema v8", () => { expect(additionalCameraTrackSchema.safeParse({ sourcePath: "" }).success).toBe(false); }); - it("rejects more than three additional cameras", () => { + it("accepts more than three additional cameras: the list has no fixed limit", () => { const entries = Array.from({ length: 5 }, (_, i) => ({ sourcePath: `/c${i}.mp4` })); - expect(assetSchema.safeParse({ ...base, additionalCameraTracks: entries }).success).toBe( - false, - ); + const parsed = assetSchema.safeParse({ ...base, additionalCameraTracks: entries }); + expect(parsed.success).toBe(true); + expect(parsed.data?.additionalCameraTracks).toHaveLength(5); }); }); diff --git a/src/lib/ai-edition/schema/index.ts b/src/lib/ai-edition/schema/index.ts index 02702afc1..495b93b09 100644 --- a/src/lib/ai-edition/schema/index.ts +++ b/src/lib/ai-edition/schema/index.ts @@ -207,7 +207,9 @@ export const assetSchema = z.object({ // no schema-version bump (an older build simply drops the key on save). transcriptionFailure: assetTranscriptionFailureSchema.nullish(), cameraTrack: cameraTrackSchema, - additionalCameraTracks: z.array(additionalCameraTrackSchema).max(3).optional(), + // No fixed limit on purpose: the recorder caps at three today, but a + // validator that ships with a cap cannot be loosened for older builds. + additionalCameraTracks: z.array(additionalCameraTrackSchema).optional(), }); // A crop is a sub-rectangle of the source video, expressed as fractions From 3f3eb2c5d64c12d95fd84bf078762985e3dd80f4 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:18:56 +0200 Subject: [PATCH 14/22] fix(recording): never match identical cameras by name across an id Two webcams of the same model share a name. When an extra carries a deviceId and camera 1 does not, the name says nothing about whether they are the same device, so the extra is no longer dropped as a duplicate of camera 1. --- electron/recording/nativeWindowsWebcams.test.ts | 13 +++++++++++++ electron/recording/nativeWindowsWebcams.ts | 10 +++++++++- 2 files changed, 22 insertions(+), 1 deletion(-) diff --git a/electron/recording/nativeWindowsWebcams.test.ts b/electron/recording/nativeWindowsWebcams.test.ts index ffd0ea56e..9c86d4fde 100644 --- a/electron/recording/nativeWindowsWebcams.test.ts +++ b/electron/recording/nativeWindowsWebcams.test.ts @@ -62,6 +62,19 @@ describe("nativeWindowsWebcams", () => { ); }); + it("never matches an extra with an id to an id-less camera 1 by name", () => { + // Two cameras of the same model: same name, and camera 1 came without an id. + expect( + dedupeAdditionalWebcams({ deviceName: "USB Camera" }, [ + { deviceId: "b", deviceName: "USB Camera" }, + ]), + ).toEqual([{ deviceId: "b", deviceName: "USB Camera" }]); + // Without an id on either side the name is all there is, and it still matches. + expect( + dedupeAdditionalWebcams({ deviceName: "USB Camera" }, [{ deviceName: "USB Camera" }]), + ).toEqual([]); + }); + it("drops an empty additional camera file and names it", () => { const r = collectStoppedWebcams({ camera1Enabled: true, diff --git a/electron/recording/nativeWindowsWebcams.ts b/electron/recording/nativeWindowsWebcams.ts index 08a797082..23e0bb492 100644 --- a/electron/recording/nativeWindowsWebcams.ts +++ b/electron/recording/nativeWindowsWebcams.ts @@ -47,11 +47,19 @@ export interface HelperWebcamEntry { type DeviceRef = { deviceId?: string; deviceName?: string }; -/** Same device: by id when both sides carry one, otherwise by name. */ +/** + * Same device: by id when both sides carry one, otherwise by name — but only + * when neither carries an id. Two webcams of the same model share a name, so an + * id on one side and none on the other says nothing about whether they are the + * same device, and matching by name there would silently drop the second one. + */ function isSameDevice(a: DeviceRef, b: DeviceRef) { if (a.deviceId && b.deviceId) { return a.deviceId === b.deviceId; } + if (a.deviceId || b.deviceId) { + return false; + } const name = a.deviceName?.trim(); return Boolean(name) && name === b.deviceName?.trim(); } From 21c848ee852f0d8502d89e3d5853df4f0fea4ecf Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:19:39 +0200 Subject: [PATCH 15/22] fix(wgc): report a camera file that could not be removed The helper deletes the file of a camera it drops at start, but ignored a failed DeleteFileW. It now logs a WARNING with the path and GetLastError, and Electron keeps the dropped cameras' paths so stop and discard remove a 0-byte stub left behind. --- electron/ipc/handlers.ts | 35 ++++++++++++++++++++++++ electron/native/wgc-capture/src/main.cpp | 11 +++++++- 2 files changed, 45 insertions(+), 1 deletion(-) diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 191250fcd..520e18787 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -763,6 +763,11 @@ let nativeWindowsCaptureWebcamTargetPath: string | null = null; * the label it is reported under. Only paths generated here ever land in it. */ let nativeWindowsCaptureAdditionalWebcamTargets: Array<{ path: string; label: string }> = []; +/** + * Files of cameras the helper dropped at start. The helper deletes them, but a + * delete that failed leaves a 0-byte stub; stop and discard remove it if empty. + */ +let nativeWindowsCaptureDroppedWebcamPaths: string[] = []; let nativeWindowsCaptureRecordingId: number | null = null; let nativeWindowsCursorOffsetMs = 0; let nativeWindowsCursorCaptureMode: CursorCaptureMode = "editable-overlay"; @@ -790,6 +795,7 @@ function resetNativeWindowsCaptureState() { nativeWindowsCaptureTargetPath = null; nativeWindowsCaptureWebcamTargetPath = null; nativeWindowsCaptureAdditionalWebcamTargets = []; + nativeWindowsCaptureDroppedWebcamPaths = []; nativeWindowsCaptureRecordingId = null; nativeWindowsCursorOffsetMs = 0; nativeWindowsCursorCaptureMode = "editable-overlay"; @@ -848,6 +854,22 @@ async function removeNativeWindowsCaptureOutputs( } } } + +/** Removes the 0-byte stubs of cameras dropped at start; anything with data stays. */ +async function removeEmptyNativeWindowsWebcamFiles(paths: string[]) { + for (const target of paths) { + if (!isPathWithinDir(target, RECORDINGS_DIR)) { + continue; + } + const stats = await fs.stat(target).catch(() => null); + if (stats?.size !== 0) { + continue; + } + await fs.rm(target, { force: true }).catch((error) => { + console.warn("[native-wgc] could not remove an empty camera file:", target, error); + }); + } +} let nativeMacCaptureProcess: ChildProcessWithoutNullStreams | null = null; /** * Apple's system picker session (macOS 15.2+), started on first use and kept for the app's @@ -2999,6 +3021,7 @@ export function registerIpcHandlers( nativeWindowsCaptureAdditionalWebcamTargets = additionalWebcams.map( ({ label, path: cameraPath }) => ({ label, path: cameraPath }), ); + nativeWindowsCaptureDroppedWebcamPaths = []; nativeWindowsCaptureRecordingId = recordingId; nativeWindowsCursorOffsetMs = 0; nativeWindowsCursorCaptureMode = cursorCaptureMode; @@ -3097,6 +3120,13 @@ export function registerIpcHandlers( const unavailableWebcams = request.webcam.enabled ? labelsOfUnavailableAdditionalWebcams(startedWebcams, [...unavailableWebcamIndices]) : []; + // The helper deletes a dropped camera's file but may fail to; a stub + // left behind is removed at stop or discard if it is still empty. + nativeWindowsCaptureDroppedWebcamPaths = request.webcam.enabled + ? startedWebcams + .filter((_, i) => unavailableWebcamIndices.has(i)) + .map((camera) => camera.path) + : []; if (unavailableWebcams.length > 0) { console.warn( "[native-wgc] recording without additional cameras the helper could not open", @@ -3457,9 +3487,13 @@ export function registerIpcHandlers( const preferredPath = nativeWindowsCaptureTargetPath; const preferredWebcamPath = nativeWindowsCaptureWebcamTargetPath; const additionalWebcamTargets = nativeWindowsCaptureAdditionalWebcamTargets; + const droppedWebcamPaths = nativeWindowsCaptureDroppedWebcamPaths; + // Start-dropped cameras ride along so a discard or a failed stop also + // removes a stub the helper could not delete (both are empty). const allWebcamPaths = [ preferredWebcamPath, ...additionalWebcamTargets.map((target) => target.path), + ...droppedWebcamPaths, ]; const recordingId = nativeWindowsCaptureRecordingId ?? Date.now(); const cursorCaptureMode = nativeWindowsCursorCaptureMode; @@ -3631,6 +3665,7 @@ export function registerIpcHandlers( helperWebcamPaths: readStoppedWebcamPaths(nativeWindowsCaptureOutput), }); } + await removeEmptyNativeWindowsWebcamFiles(droppedWebcamPaths); // Generated by the start handler, never taken from the renderer or the // helper; approved like the session's other media. for (const extra of additionalWebcams) { diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index e7a18082e..09983e45d 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -773,7 +773,16 @@ int wmain(int argc, wchar_t* argv[]) { // Balances whatever initialize() got through, then removes the // file it may have created: nothing was ever written to it. stream.encoder.finalize(); - DeleteFileW(utf8ToWide(stream.config.outputPath).c_str()); + // A file that never got created is fine; one that could not be + // removed is left as a 0-byte stub, which the app cleans up. + if (!DeleteFileW(utf8ToWide(stream.config.outputPath).c_str())) { + const DWORD deleteError = GetLastError(); + if (deleteError != ERROR_FILE_NOT_FOUND) { + std::cerr << "WARNING: Could not remove the dropped camera file " + << stream.config.outputPath << " (GetLastError=" << deleteError << ")" + << std::endl; + } + } } stream.active = false; if (inlineWebcam == &stream) { From 31e8d3190d6c90bcb1645b6e6b6b66723c96c709 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:20:00 +0200 Subject: [PATCH 16/22] feat(recording): name a camera that stopped early A camera the helper disables mid-take keeps its partial file in the take, but nobody was told. A camera whose file was kept (size > 0) yet is missing from recording-stopped.webcamPaths is now named in a "Stopped early" notice after the take, camera 1 included. Paths compare case-insensitively with either separator. Only an event that carries webcamPaths can say so, so the helper now prints the list, possibly empty, whenever a camera wrote a file of its own; an older helper or a missing event never produces the notice. --- electron/electron-env.d.ts | 5 ++ electron/ipc/handlers.ts | 23 ++++++++- electron/native/README.md | 2 +- electron/native/wgc-capture/src/main.cpp | 7 ++- .../nativeWindowsCaptureStop.test.ts | 13 +++++ .../recording/nativeWindowsCaptureStop.ts | 29 ++++++++--- .../recording/nativeWindowsWebcams.test.ts | 50 +++++++++++++++++++ electron/recording/nativeWindowsWebcams.ts | 33 ++++++++++++ ...eScreenRecorder.nativeStopFailure.test.tsx | 16 ++++++ src/hooks/useScreenRecorder.ts | 9 ++++ src/i18n/locales/ar/launch.json | 1 + src/i18n/locales/cs/launch.json | 1 + src/i18n/locales/de/launch.json | 1 + src/i18n/locales/en/launch.json | 1 + src/i18n/locales/es/launch.json | 1 + src/i18n/locales/fr/launch.json | 1 + src/i18n/locales/it/launch.json | 1 + src/i18n/locales/ja-JP/launch.json | 1 + src/i18n/locales/ko-KR/launch.json | 1 + src/i18n/locales/pt-BR/launch.json | 1 + src/i18n/locales/ru/launch.json | 1 + src/i18n/locales/tr/launch.json | 1 + src/i18n/locales/vi/launch.json | 1 + src/i18n/locales/zh-CN/launch.json | 1 + src/i18n/locales/zh-TW/launch.json | 1 + 25 files changed, 192 insertions(+), 10 deletions(-) diff --git a/electron/electron-env.d.ts b/electron/electron-env.d.ts index 99b2a3f79..1c28c8b4f 100644 --- a/electron/electron-env.d.ts +++ b/electron/electron-env.d.ts @@ -183,6 +183,11 @@ interface Window { * `webcamDropped`. */ droppedWebcams?: string[]; + /** + * Labels of cameras (camera 1 included) that the helper disabled mid-take. + * Their partial files are kept in the session; the user is told which. + */ + webcamsStoppedEarly?: string[]; }>; pauseNativeWindowsRecording: () => Promise<{ success: boolean; diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index 520e18787..f1cbfe31f 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -123,6 +123,7 @@ import { NATIVE_WINDOWS_SALVAGEABLE_OUTPUT_BYTES, readMicrophoneDefaulted, readMicrophoneUnavailable, + readReportedWebcamPaths, readSecondaryWindowsApplied, readStoppedWebcamPaths, readUnavailableWebcamIndices, @@ -137,6 +138,7 @@ import { dedupeAdditionalWebcams, isWebcamSidecarFile, labelsOfUnavailableAdditionalWebcams, + labelsOfWebcamsStoppedEarly, stripWebcamSuffix, webcamOutputPath, } from "../recording/nativeWindowsWebcams"; @@ -763,6 +765,8 @@ let nativeWindowsCaptureWebcamTargetPath: string | null = null; * the label it is reported under. Only paths generated here ever land in it. */ let nativeWindowsCaptureAdditionalWebcamTargets: Array<{ path: string; label: string }> = []; +/** Camera 1's label for notices: its device name, else "Camera 1". */ +let nativeWindowsCaptureWebcamLabel = "Camera 1"; /** * Files of cameras the helper dropped at start. The helper deletes them, but a * delete that failed leaves a 0-byte stub; stop and discard remove it if empty. @@ -795,6 +799,7 @@ function resetNativeWindowsCaptureState() { nativeWindowsCaptureTargetPath = null; nativeWindowsCaptureWebcamTargetPath = null; nativeWindowsCaptureAdditionalWebcamTargets = []; + nativeWindowsCaptureWebcamLabel = "Camera 1"; nativeWindowsCaptureDroppedWebcamPaths = []; nativeWindowsCaptureRecordingId = null; nativeWindowsCursorOffsetMs = 0; @@ -3021,6 +3026,7 @@ export function registerIpcHandlers( nativeWindowsCaptureAdditionalWebcamTargets = additionalWebcams.map( ({ label, path: cameraPath }) => ({ label, path: cameraPath }), ); + nativeWindowsCaptureWebcamLabel = request.webcam.deviceName?.trim() || "Camera 1"; nativeWindowsCaptureDroppedWebcamPaths = []; nativeWindowsCaptureRecordingId = recordingId; nativeWindowsCursorOffsetMs = 0; @@ -3487,6 +3493,7 @@ export function registerIpcHandlers( const preferredPath = nativeWindowsCaptureTargetPath; const preferredWebcamPath = nativeWindowsCaptureWebcamTargetPath; const additionalWebcamTargets = nativeWindowsCaptureAdditionalWebcamTargets; + const camera1Label = nativeWindowsCaptureWebcamLabel; const droppedWebcamPaths = nativeWindowsCaptureDroppedWebcamPaths; // Start-dropped cameras ride along so a discard or a failed stop also // removes a stub the helper could not delete (both are empty). @@ -3637,7 +3644,7 @@ export function registerIpcHandlers( // (getopenscreen/openscreen#387). Every camera is judged by its own file, // not by the helper's list at stop (see `collectStoppedWebcams`). const requestedWebcams = [ - ...(preferredWebcamPath ? [{ path: preferredWebcamPath, label: "" }] : []), + ...(preferredWebcamPath ? [{ path: preferredWebcamPath, label: camera1Label }] : []), ...additionalWebcamTargets, ]; const webcamSizes = new Map(); @@ -3665,6 +3672,19 @@ export function registerIpcHandlers( helperWebcamPaths: readStoppedWebcamPaths(nativeWindowsCaptureOutput), }); } + // A camera the helper disabled mid-take keeps its partial file (it is + // in the take) but is named, so the user knows why it ends early. Only + // a helper that sends `webcamPaths` can tell; otherwise nothing is said. + const webcamsStoppedEarly = labelsOfWebcamsStoppedEarly({ + requested: requestedWebcams, + sizes: webcamSizes, + helperWebcamPaths: readReportedWebcamPaths(nativeWindowsCaptureOutput), + }); + if (webcamsStoppedEarly.length > 0) { + console.warn("[native-wgc] cameras stopped before the end of the take", { + webcamsStoppedEarly, + }); + } await removeEmptyNativeWindowsWebcamFiles(droppedWebcamPaths); // Generated by the start handler, never taken from the renderer or the // helper; approved like the session's other media. @@ -3706,6 +3726,7 @@ export function registerIpcHandlers( // the silence this change exists to end. webcamDropped: Boolean(preferredWebcamPath) && !webcamVideoPath, ...(stoppedWebcams.dropped.length > 0 ? { droppedWebcams: stoppedWebcams.dropped } : {}), + ...(webcamsStoppedEarly.length > 0 ? { webcamsStoppedEarly } : {}), message: recovered ? "Native Windows recording recovered from a failed stop" : "Native Windows recording session stored successfully", diff --git a/electron/native/README.md b/electron/native/README.md index 316622866..931a91471 100644 --- a/electron/native/README.md +++ b/electron/native/README.md @@ -94,7 +94,7 @@ Several cameras: a `webcams` list records up to four cameras, each into its own ] ``` -Every per-camera event carries the camera's `index` in that list: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. +Every per-camera event carries the camera's `index` in that list: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order — an empty list when every camera that wrote a file of its own was disabled mid-take, so a reader can tell "stopped early" from an older helper that never sends the key. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. Container: recordings are written as fragmented MP4 (`MFCreateFMPEG4MediaSink` + `MFCreateSinkWriterFromMediaSink`, `MF_MPEG4SINK_MIN_FRAGMENT_DURATION` = 1s) rather than plain MP4. A plain MP4 has no index until `IMFSinkWriter::Finalize()` writes `moov` at the very end, so when the shutdown watchdog force-exits a wedged helper the file on disk holds every frame and no way to read them — that is why issues #252 / #292 / #327 cost the whole recording rather than the frozen tail of it. A fragmented MP4 writes its index up front and its samples in self-describing `moof`+`mdat` pairs, so the same kill leaves a file that plays up to the last complete fragment. This does not fix the freeze; it removes the data loss the freeze causes. Because the fragmented sink needs both output media types at construction, the sink writer is built from a media sink instead of from a URL, and the helper reads the video/audio stream positions back off the sink rather than assuming them. If any of that is unavailable on a machine, the helper retries with the plain container and says so — `container` in the `encoder-selection` event is `fragmented-mp4` or `mp4`, and it reports what was used, not what was asked for. diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index 09983e45d..f6b4c8ffd 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -1832,8 +1832,13 @@ int wmain(int argc, wchar_t* argv[]) { // before the camera files are finalized, for the reason above -- so they // name the files that were being written, and a camera whose finalize // fails below is reported on stderr, not removed from this list. + // `webcamPaths` is printed, possibly empty, whenever a camera wrote a + // file of its own: its presence is how the app knows a camera missing + // from it stopped early, rather than an older helper that never sent it. std::vector recordedWebcams; + bool anySeparateWebcam = false; for (const auto& stream : webcams) { + anySeparateWebcam = anySeparateWebcam || stream->writeSeparate; if (stream->writeSeparate && stream->active) { recordedWebcams.push_back(stream.get()); } @@ -1842,7 +1847,7 @@ int wmain(int argc, wchar_t* argv[]) { std::cout << ",\"webcamPath\":\"" << jsonEscape(recordedWebcams.front()->config.outputPath) << "\""; } - if (!recordedWebcams.empty()) { + if (anySeparateWebcam) { std::cout << ",\"webcamPaths\":["; for (size_t i = 0; i < recordedWebcams.size(); ++i) { std::cout << (i == 0 ? "\"" : ",\"") << jsonEscape(recordedWebcams[i]->config.outputPath) diff --git a/electron/recording/nativeWindowsCaptureStop.test.ts b/electron/recording/nativeWindowsCaptureStop.test.ts index 6391569c2..0def7936f 100644 --- a/electron/recording/nativeWindowsCaptureStop.test.ts +++ b/electron/recording/nativeWindowsCaptureStop.test.ts @@ -7,6 +7,7 @@ import { NATIVE_WINDOWS_SALVAGEABLE_OUTPUT_BYTES, readMicrophoneDefaulted, readMicrophoneUnavailable, + readReportedWebcamPaths, readSecondaryWindowsApplied, readStoppedPath, readStoppedWebcamPaths, @@ -593,6 +594,18 @@ describe("per-camera helper events", () => { expect(readStoppedWebcamPaths('{"event":"recording-stopped"}')).toEqual([]); }); + it("tells a reported webcamPaths list apart from its absence", () => { + expect( + readReportedWebcamPaths('{"event":"recording-stopped","webcamPaths":["a.mp4","b.mp4"]}'), + ).toEqual(["a.mp4", "b.mp4"]); + expect(readReportedWebcamPaths('{"event":"recording-stopped","webcamPaths":[]}')).toEqual([]); + // An old helper sends only the legacy path; that is "unknown", not "none". + expect( + readReportedWebcamPaths('{"event":"recording-stopped","webcamPath":"a.mp4"}'), + ).toBeNull(); + expect(readReportedWebcamPaths("Recording stopped. Output path: s.mp4")).toBeNull(); + }); + it("reads JSON-escaped Windows paths behind a prefix", () => { const line = 'INFO: x {"event":"recording-stopped","schemaVersion":2,"screenPath":"C:\\\\Users\\\\me\\\\s.mp4","webcamPath":"C:\\\\Users\\\\me\\\\s-webcam.mp4","webcamPaths":["C:\\\\Users\\\\me\\\\s-webcam.mp4","C:\\\\Users\\\\me\\\\s-webcam-2.mp4"]}'; diff --git a/electron/recording/nativeWindowsCaptureStop.ts b/electron/recording/nativeWindowsCaptureStop.ts index 824cf4b3c..05c2a0f1f 100644 --- a/electron/recording/nativeWindowsCaptureStop.ts +++ b/electron/recording/nativeWindowsCaptureStop.ts @@ -275,19 +275,34 @@ export function readWebcamFormatAt( return (match as ReturnType) ?? null; } +/** + * The `webcamPaths` list of the last `recording-stopped` event, or null when + * there is no such event or it carried no `webcamPaths` key. + * + * Null is the signal that matters: only a helper that sends the key says which + * cameras were still recording at stop, so only then may a camera missing from + * it be called one that stopped early. An old helper, or a stop without the + * event, must read as "unknown", never as "every camera stopped". + */ +export function readReportedWebcamPaths(output: string): string[] | null { + const event = readHelperEvents(output, "recording-stopped").at(-1); + if (!event || !Array.isArray(event.webcamPaths)) { + return null; + } + return event.webcamPaths.filter((entry): entry is string => typeof entry === "string"); +} + /** * Camera files the helper reported at stop: `webcamPaths`, else the single * legacy `webcamPath`, else none. */ export function readStoppedWebcamPaths(output: string): string[] { - const event = readHelperEvents(output, "recording-stopped").at(-1); - if (!event) { - return []; + const reported = readReportedWebcamPaths(output); + if (reported) { + return reported; } - if (Array.isArray(event.webcamPaths)) { - return event.webcamPaths.filter((entry): entry is string => typeof entry === "string"); - } - return typeof event.webcamPath === "string" && event.webcamPath ? [event.webcamPath] : []; + const event = readHelperEvents(output, "recording-stopped").at(-1); + return typeof event?.webcamPath === "string" && event.webcamPath ? [event.webcamPath] : []; } /** diff --git a/electron/recording/nativeWindowsWebcams.test.ts b/electron/recording/nativeWindowsWebcams.test.ts index 9c86d4fde..f7187851c 100644 --- a/electron/recording/nativeWindowsWebcams.test.ts +++ b/electron/recording/nativeWindowsWebcams.test.ts @@ -6,6 +6,7 @@ import { dedupeAdditionalWebcams, isWebcamSidecarFile, labelsOfUnavailableAdditionalWebcams, + labelsOfWebcamsStoppedEarly, stripWebcamSuffix, webcamOutputPath, } from "./nativeWindowsWebcams"; @@ -75,6 +76,55 @@ describe("nativeWindowsWebcams", () => { ).toEqual([]); }); + describe("labelsOfWebcamsStoppedEarly", () => { + const front = String.raw`C:\Rec\r-webcam.mp4`; + const desk = String.raw`C:\Rec\r-webcam-2.mp4`; + const side = String.raw`C:\Rec\r-webcam-3.mp4`; + const requested = [ + { path: front, label: "Front" }, + { path: desk, label: "Desk" }, + { path: side, label: "Side" }, + ]; + const sizes = new Map([ + [front, 100], + [desk, 100], + [side, 0], + ]); + + it("names a kept camera the helper no longer listed, camera 1 included", () => { + expect( + labelsOfWebcamsStoppedEarly({ + requested, + sizes, + helperWebcamPaths: [desk], + }), + ).toEqual(["Front"]); + }); + + it("leaves out a camera whose file was not kept: that one was not recorded", () => { + expect(labelsOfWebcamsStoppedEarly({ requested, sizes, helperWebcamPaths: [] })).toEqual([ + "Front", + "Desk", + ]); + }); + + it("flags nothing without a webcamPaths key", () => { + expect(labelsOfWebcamsStoppedEarly({ requested, sizes, helperWebcamPaths: null })).toEqual( + [], + ); + }); + + it("matches paths regardless of case and separator", () => { + expect( + labelsOfWebcamsStoppedEarly({ + requested, + sizes, + helperWebcamPaths: ["c:/rec/R-WEBCAM.mp4", String.raw`c:\REC//r-webcam-2.MP4`], + }), + ).toEqual([]); + }); + }); + it("drops an empty additional camera file and names it", () => { const r = collectStoppedWebcams({ camera1Enabled: true, diff --git a/electron/recording/nativeWindowsWebcams.ts b/electron/recording/nativeWindowsWebcams.ts index 23e0bb492..9c8b462b5 100644 --- a/electron/recording/nativeWindowsWebcams.ts +++ b/electron/recording/nativeWindowsWebcams.ts @@ -234,3 +234,36 @@ export function collectStoppedWebcams(input: { } return { ...(camera1 ? { camera1 } : {}), additional, dropped }; } + +/** A path compared the way Windows does: case-insensitive, either separator. */ +function comparablePath(filePath: string) { + return filePath.replace(/[\\/]+/g, "\\").toLowerCase(); +} + +/** + * Labels of the cameras that stopped early: their file was kept (size > 0, the + * same test as {@link collectStoppedWebcams}) but the helper did not list it in + * `recording-stopped.webcamPaths`, the cameras still recording at stop. That is + * a camera the helper disabled mid-take; its partial file stays in the take, + * and the user is told which camera it was. + * + * `helperWebcamPaths` is null when the event carried no `webcamPaths` key (an + * old helper, or no event at all) — then nothing is known and nothing is + * flagged. Camera 1 counts like any other, so its entry needs a real label. + */ +export function labelsOfWebcamsStoppedEarly(input: { + requested: Array<{ path: string; label: string }>; + sizes: Map; + helperWebcamPaths: string[] | null; +}): string[] { + if (!input.helperWebcamPaths) { + return []; + } + const stillRecording = new Set(input.helperWebcamPaths.map(comparablePath)); + return input.requested + .filter( + (camera) => + (input.sizes.get(camera.path) ?? 0) > 0 && !stillRecording.has(comparablePath(camera.path)), + ) + .map((camera) => camera.label); +} diff --git a/src/hooks/useScreenRecorder.nativeStopFailure.test.tsx b/src/hooks/useScreenRecorder.nativeStopFailure.test.tsx index 928ac5886..5e761e3ea 100644 --- a/src/hooks/useScreenRecorder.nativeStopFailure.test.tsx +++ b/src/hooks/useScreenRecorder.nativeStopFailure.test.tsx @@ -137,4 +137,20 @@ describe("useScreenRecorder native Windows stop failure", () => { expect(api.switchToEditor).toHaveBeenCalled(); expect(view.result.current.recording).toBe(false); }); + + it("names the cameras that stopped early and still opens the editor", async () => { + vi.mocked(toast.warning).mockClear(); + api.stopNativeWindowsRecording.mockResolvedValue({ + success: true, + path: "C:\\rec\\a.mp4", + webcamsStoppedEarly: ["Front", "Desk"], + }); + + const view = renderHook(() => useScreenRecorder()); + await startNativeRecording(view); + await pressStop(view); + + expect(toast.warning).toHaveBeenCalledWith("webcam.camerasStoppedEarly"); + expect(api.switchToEditor).toHaveBeenCalled(); + }); }); diff --git a/src/hooks/useScreenRecorder.ts b/src/hooks/useScreenRecorder.ts index a5a88cdfa..177d7667b 100644 --- a/src/hooks/useScreenRecorder.ts +++ b/src/hooks/useScreenRecorder.ts @@ -813,6 +813,15 @@ export function useScreenRecorder(): UseScreenRecorderReturn { }), ); } + // Kept in the take, but shorter than it: the helper disabled these + // cameras mid-take, so the editor shows them ending early. + if (result.webcamsStoppedEarly?.length) { + toast.warning( + tLaunchRef.current("webcam.camerasStoppedEarly", { + names: result.webcamsStoppedEarly.join(", "), + }), + ); + } if (result.session) { await window.electronAPI.setCurrentRecordingSession(result.session); } else if (result.path) { diff --git a/src/i18n/locales/ar/launch.json b/src/i18n/locales/ar/launch.json index 858d9d8e9..365845783 100644 --- a/src/i18n/locales/ar/launch.json +++ b/src/i18n/locales/ar/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "كاميرات إضافية", "additionalCamerasHint": "فقط مع التسجيل الأصلي في Windows", "camerasNotRecorded": "لم يتم تسجيلها: {{names}}", + "camerasStoppedEarly": "توقفت مبكرًا: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/cs/launch.json b/src/i18n/locales/cs/launch.json index d79eb409b..89b34c7c3 100644 --- a/src/i18n/locales/cs/launch.json +++ b/src/i18n/locales/cs/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Další kamery", "additionalCamerasHint": "Pouze s nativním nahráváním ve Windows", "camerasNotRecorded": "Nenahráno: {{names}}", + "camerasStoppedEarly": "Zastaveno předčasně: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/de/launch.json b/src/i18n/locales/de/launch.json index 2ba2310d7..de7926b50 100644 --- a/src/i18n/locales/de/launch.json +++ b/src/i18n/locales/de/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Weitere Kameras", "additionalCamerasHint": "Nur bei nativer Windows-Aufnahme", "camerasNotRecorded": "Nicht aufgenommen: {{names}}", + "camerasStoppedEarly": "Vorzeitig beendet: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/en/launch.json b/src/i18n/locales/en/launch.json index 2489623aa..478e0c5ef 100644 --- a/src/i18n/locales/en/launch.json +++ b/src/i18n/locales/en/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Additional cameras", "additionalCamerasHint": "Only with native Windows recording", "camerasNotRecorded": "Not recorded: {{names}}", + "camerasStoppedEarly": "Stopped early: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/es/launch.json b/src/i18n/locales/es/launch.json index 565d4a852..85fd72d02 100644 --- a/src/i18n/locales/es/launch.json +++ b/src/i18n/locales/es/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Cámaras adicionales", "additionalCamerasHint": "Solo con la grabación nativa de Windows", "camerasNotRecorded": "No grabadas: {{names}}", + "camerasStoppedEarly": "Se detuvieron antes de tiempo: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/fr/launch.json b/src/i18n/locales/fr/launch.json index 0cc342866..1f6815556 100644 --- a/src/i18n/locales/fr/launch.json +++ b/src/i18n/locales/fr/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Caméras supplémentaires", "additionalCamerasHint": "Uniquement avec l'enregistrement natif sous Windows", "camerasNotRecorded": "Non enregistrées : {{names}}", + "camerasStoppedEarly": "Arrêtées prématurément : {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/it/launch.json b/src/i18n/locales/it/launch.json index 2290d693a..e9f7fa36b 100644 --- a/src/i18n/locales/it/launch.json +++ b/src/i18n/locales/it/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Fotocamere aggiuntive", "additionalCamerasHint": "Solo con la registrazione nativa di Windows", "camerasNotRecorded": "Non registrate: {{names}}", + "camerasStoppedEarly": "Interrotte in anticipo: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/ja-JP/launch.json b/src/i18n/locales/ja-JP/launch.json index 4d7fcf3ce..7c28f6731 100644 --- a/src/i18n/locales/ja-JP/launch.json +++ b/src/i18n/locales/ja-JP/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "追加のカメラ", "additionalCamerasHint": "Windows のネイティブ録画でのみ使用できます", "camerasNotRecorded": "録画されませんでした: {{names}}", + "camerasStoppedEarly": "途中で停止しました: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/ko-KR/launch.json b/src/i18n/locales/ko-KR/launch.json index 641fb53d8..706dab4bc 100644 --- a/src/i18n/locales/ko-KR/launch.json +++ b/src/i18n/locales/ko-KR/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "추가 카메라", "additionalCamerasHint": "Windows 기본 녹화에서만 사용할 수 있습니다", "camerasNotRecorded": "녹화되지 않음: {{names}}", + "camerasStoppedEarly": "중간에 중지됨: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/pt-BR/launch.json b/src/i18n/locales/pt-BR/launch.json index 276d494e0..eabcbebba 100644 --- a/src/i18n/locales/pt-BR/launch.json +++ b/src/i18n/locales/pt-BR/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Câmeras adicionais", "additionalCamerasHint": "Somente com a gravação nativa do Windows", "camerasNotRecorded": "Não gravadas: {{names}}", + "camerasStoppedEarly": "Pararam antes do fim: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/ru/launch.json b/src/i18n/locales/ru/launch.json index 1d99fd9b5..dbbb750b1 100644 --- a/src/i18n/locales/ru/launch.json +++ b/src/i18n/locales/ru/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Дополнительные камеры", "additionalCamerasHint": "Только при встроенной записи в Windows", "camerasNotRecorded": "Не записаны: {{names}}", + "camerasStoppedEarly": "Остановились раньше времени: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/tr/launch.json b/src/i18n/locales/tr/launch.json index 22c00b419..b8f3309d1 100644 --- a/src/i18n/locales/tr/launch.json +++ b/src/i18n/locales/tr/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Ek kameralar", "additionalCamerasHint": "Yalnızca yerel Windows kaydıyla", "camerasNotRecorded": "Kaydedilmedi: {{names}}", + "camerasStoppedEarly": "Erken durdu: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/vi/launch.json b/src/i18n/locales/vi/launch.json index 0416dce15..ae776c2d0 100644 --- a/src/i18n/locales/vi/launch.json +++ b/src/i18n/locales/vi/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "Camera bổ sung", "additionalCamerasHint": "Chỉ với tính năng quay gốc của Windows", "camerasNotRecorded": "Không được ghi: {{names}}", + "camerasStoppedEarly": "Đã dừng sớm: {{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/zh-CN/launch.json b/src/i18n/locales/zh-CN/launch.json index 03cb0f418..9c2cbeae3 100644 --- a/src/i18n/locales/zh-CN/launch.json +++ b/src/i18n/locales/zh-CN/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "附加摄像头", "additionalCamerasHint": "仅支持 Windows 原生录制", "camerasNotRecorded": "未录制:{{names}}", + "camerasStoppedEarly": "提前停止:{{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" diff --git a/src/i18n/locales/zh-TW/launch.json b/src/i18n/locales/zh-TW/launch.json index 97bb58696..f9c72b322 100644 --- a/src/i18n/locales/zh-TW/launch.json +++ b/src/i18n/locales/zh-TW/launch.json @@ -51,6 +51,7 @@ "additionalCameras": "額外攝影機", "additionalCamerasHint": "僅限 Windows 原生錄影", "camerasNotRecorded": "未錄製:{{names}}", + "camerasStoppedEarly": "提前停止:{{names}}", "quality1080p": "1080p", "quality1440p": "1440p", "quality2160p": "4K" From 20f7044ba937b91428998a92406dc567f07db0bc Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sun, 4 Oct 2026 13:20:15 +0200 Subject: [PATCH 17/22] docs(e2e): identical cameras and screen pacing with several cameras The checklist now notes that with several identical cameras plugged in the recorded ones follow Windows' enumeration order, adds a 1-vs-2-camera screen pacing comparison at 4K (#945) and a stopped-early check. The helper README says the camera index counts after entries without camPath are skipped, and the extras-need-camera-1 assumption (R6) is noted where links drop them. --- electron/ipc/handlers.ts | 3 +++ electron/media/mediaLinksRegistry.ts | 3 +++ electron/native/README.md | 2 +- technical-documentation/testing/manual-e2e-checklist.md | 3 +++ 4 files changed, 10 insertions(+), 1 deletion(-) diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index f1cbfe31f..8270c69da 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -4960,6 +4960,9 @@ export function registerIpcHandlers( return { success: false, error: "Video path has not been approved" }; } const resolution = await resolveMediaLinksForVideo(normalized); + // Additional cameras are only returned alongside camera 1. That relies + // on R6 (extras are recorded only while camera 1 is on), so extras + // without camera 1 means camera 1's file came out empty. if (!resolution.webcamVideoPath) { return { success: false, error: "No camera attached to this recording" }; } diff --git a/electron/media/mediaLinksRegistry.ts b/electron/media/mediaLinksRegistry.ts index 69d119d51..56bd7d57f 100644 --- a/electron/media/mediaLinksRegistry.ts +++ b/electron/media/mediaLinksRegistry.ts @@ -243,6 +243,9 @@ export async function registerMediaLinks( videoPath: string, links: MediaLinksToRegister, ): Promise { + // Extras alone register nothing. That relies on R6: additional cameras are + // recorded only while camera 1 is on, so a take with extras but no camera 1 + // is one whose camera 1 file came out empty — a rare loss accepted here. if (!links.webcamVideoPath && !links.cursorTelemetryPath) return; const { additionalWebcams: rawAdditionalWebcams, ...linksWithoutAdditional } = links; const additionalWebcams = normalizeAdditionalWebcams(rawAdditionalWebcams); diff --git a/electron/native/README.md b/electron/native/README.md index 931a91471..d9856d052 100644 --- a/electron/native/README.md +++ b/electron/native/README.md @@ -94,7 +94,7 @@ Several cameras: a `webcams` list records up to four cameras, each into its own ] ``` -Every per-camera event carries the camera's `index` in that list: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order — an empty list when every camera that wrote a file of its own was disabled mid-take, so a reader can tell "stopped early" from an older helper that never sends the key. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. +Every per-camera event carries the camera's `index`, its position in the list after entries without `camPath` have been skipped: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order — an empty list when every camera that wrote a file of its own was disabled mid-take, so a reader can tell "stopped early" from an older helper that never sends the key. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. Container: recordings are written as fragmented MP4 (`MFCreateFMPEG4MediaSink` + `MFCreateSinkWriterFromMediaSink`, `MF_MPEG4SINK_MIN_FRAGMENT_DURATION` = 1s) rather than plain MP4. A plain MP4 has no index until `IMFSinkWriter::Finalize()` writes `moov` at the very end, so when the shutdown watchdog force-exits a wedged helper the file on disk holds every frame and no way to read them — that is why issues #252 / #292 / #327 cost the whole recording rather than the frozen tail of it. A fragmented MP4 writes its index up front and its samples in self-describing `moof`+`mdat` pairs, so the same kill leaves a file that plays up to the last complete fragment. This does not fix the freeze; it removes the data loss the freeze causes. Because the fragmented sink needs both output media types at construction, the sink writer is built from a media sink instead of from a URL, and the helper reads the video/audio stream positions back off the sink rather than assuming them. If any of that is unavailable on a machine, the helper retries with the plain container and says so — `container` in the `encoder-selection` event is `fragmented-mp4` or `mp4`, and it reports what was used, not what was asked for. diff --git a/technical-documentation/testing/manual-e2e-checklist.md b/technical-documentation/testing/manual-e2e-checklist.md index ec4a4fd09..3fb4cc255 100644 --- a/technical-documentation/testing/manual-e2e-checklist.md +++ b/technical-documentation/testing/manual-e2e-checklist.md @@ -226,8 +226,11 @@ Up to four cameras record into one take on the native Windows path: camera 1 as - [ ] Several cameras (Windows). Pick a second camera under *Additional cameras*, record about 10 s and clap in front of both cameras. Both camera files exist next to the recording (`-webcam.mp4`, `-webcam-2.mp4`) and both play. The clap lands at the same time in both and against the screen. The editor shows camera 1 as before. After saving and reopening, the project still lists both cameras. - [ ] Order of the runs: start with two cameras of different names (Logitech Brio plus C920), then two identical Brios. Record at 1080p first, and at 4K afterwards as a load test: two 4K cameras on one USB controller may lack the bandwidth, so a camera that drops out there is a finding about the controller first and the app second. Log the controller layout next to the result. - [ ] With two identical Brios, which physical Brio becomes camera 1 follows Windows' enumeration order, not the HUD choice or its preview. A swap against the preview is a known limitation, not a defect. Wave a hand into one camera at a time to tell the files apart, and note which is which. +- [ ] Identical cameras are matched by name, so with several of one model plugged in, *which* of them get recorded also follows Windows' enumeration order, not the HUD picks: with three Brios attached and two picked, the third may be recorded instead of a picked one. Plug in only the identical cameras you want to record, and note any mismatch as this known limitation. +- [ ] **[4K camera]** Screen pacing with several cameras. Screen pacing is a known weak spot (#945), and a pacing regression under camera load is easy to misread as a USB problem. Record the same 20 s screen scene twice at 4K camera quality, once with one camera and once with two, and compare the screen file of each: `ffprobe -v error -select_streams v:0 -count_frames -show_entries stream=nb_read_frames,avg_frame_rate .mp4`, plus the helper's `[pacing] frames= elapsed_ms=` line from the diagnostics (and `[frame-drops]` on the GPU input path). Frames over elapsed time should match the chosen screen rate in both runs; the helper does not log individual missed ticks, so a shortfall shows only there. A two-camera run whose screen falls clearly short of the one-camera run is a pacing finding for the helper, not a camera or controller one, whatever the camera files look like. Log both numbers. - [ ] Read the helper output (`Save diagnostics`, or `(Get-Content diag.json -Raw | ConvertFrom-Json).helperOutput.windows -split "`n" | Select-String "webcam"`). Camera 2's `Native webcam candidate` lines list the second Brio, and neither the IR or any other interface of the first Brio nor a candidate marked `already recording in this take`. Each camera prints its own `webcam-format` event with its `index`, and `recording-stopped` carries both paths in `webcamPaths`. - [ ] A camera that cannot be opened or records nothing is named in a "Not recorded: …" notice after the take, and the take is kept with the cameras that worked. Provoke it by unplugging the second camera, or holding it open in another app, just before *Start recording*, and check the helper output for `{"event":"warning","code":"webcam-unavailable","index":1,…}`. Camera 1 and the screen are unaffected. +- [ ] A camera the helper disables mid-take (its log says `ERROR: Failed to capture a sample for camera ; disabling it`, or `submit`) keeps its partial file in the take and is named in a "Stopped early: …" notice after the take; `recording-stopped.webcamPaths` lists only the cameras still recording at stop. Try unplugging the second camera halfway through; if that leaves no such line, the encoder never failed and there is nothing to check, so log `skipped: could not provoke a mid-take camera failure`. - [ ] With the first camera turned off in the HUD, record again and confirm no additional camera is recorded (additional cameras follow camera 1). - [ ] Probe every file: `ffprobe -show_entries format=duration` of each camera file is within 0.2 s of the screen file's. From 38d838b9a596d7794de71a753291d41a3b163b5f Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:27:55 +0200 Subject: [PATCH 18/22] docs(e2e): log the first real two-camera run on Windows on ARM --- technical-documentation/testing/manual-e2e-checklist.md | 1 + 1 file changed, 1 insertion(+) diff --git a/technical-documentation/testing/manual-e2e-checklist.md b/technical-documentation/testing/manual-e2e-checklist.md index 3fb4cc255..479a0910e 100644 --- a/technical-documentation/testing/manual-e2e-checklist.md +++ b/technical-documentation/testing/manual-e2e-checklist.md @@ -870,3 +870,4 @@ On macOS 15.2+ it opens at launch, until it has been closed once, while one of i | 2026-10-03 | installed `v2.0.0-rc.13`, CI-built DMG `Openscreen-macOS-Apple-Silicon-2.0.0-rc.13.dmg` from the GitHub Release (tag `018d22ab`), installed over rc.12, `spctl`: accepted, Notarized Developer ID (M4LK7C6S84), About reports `2.0.0-rc.13` | macOS 26.5 (25F71), Mac mini M1, 3840×2160, no camera, no microphone | Partial, no defect | **Scoped to the app changes between rc.8 and rc.13, on macOS**: the Windows row above covers the same diff on Windows and leaves out #967, c2a8559e and macOS, which this row takes. Real OS mouse and keyboard input through computer-use (screenshots worked on this Mac); Apple's picker is owned by Control Centre and cannot be granted, so the operator clicked *Share Entire Screen* for each take. **c2a8559e (HUD kept out of takes after the editor round trip):** take 1 picked from the HUD's record button (`excludedWindowIds: [582]`), stop, editor, Record mode showed the source as *Choose what to record*, *Start recording* reopened Apple's picker instead of reusing the pick, take 2 recorded with `excludedWindowIds: [639]`; frames extracted from both files (28:26 and 13:04) show no HUD where it sat. **#967 (no camera on this Mac):** with `camEnabled: true` and no camera identity in `recording-settings.json`, launch rewrote it to `false` and the HUD showed the camera off; same with a stale identity (`camDeviceId` + `camDeviceName` of an unplugged camera); clicking the toggle keeps it off with the "Camera access is blocked" toast; Record mode reads *Camera Off*. **#961:** closing the editor after take 2 (countdown overlay, Record mode handover) left 0 processes; so did closing it after a long editor session (eight project opens through the native panel, transcriptions, Media mode, an export, the export dialog's save panel cancelled). **#966:** with *Add zoom* bound to X in `shortcuts.json`, the empty zoom lane reads "Press X to add zoom", and "Appuyez sur X pour ajouter un zoom" / "Drücke X, …" in French and German. **Wordmark menu (3bee2d25):** French and German, every row on one line, "2.0.0-rc.13" in one piece. **#960 (b834416c):** project A with captions → project B → A again: captions show at once at 0:06.4, 0:10.2 and 1:05.2 without touching *Show captions* (0:05.3 shows none, a 4.77 to 5.57 s silence in the saved words). **Preview:** playback from 0:12 to the end across four transcript silence cuts, a 2× and a 16× region: picture and captions follow, the readout advances with the edit, stops at 1:06.2; no stall seen, 8 `[pipeline] décodage` opens for one seek and four cuts (consistent with a preload per cut; not compared with rc.12). **227b9b84 / 83fbcecc:** a two-clip project whose first clip is a file with 8 s of video and 10 s of audio (no camera): seeking from clip 2 to 0:09.5 shows clip 1's last frame, the same as 0:07.7, no stderr. A video-only file plays with its picture (e73097c8). **#968:** project whose first asset is `chmod 000`: `MediaUnreadableError … Permission denied` in the main log, clip 2 still transcribed, the asset card shows "Transcription failed" and "Couldn't read this media's audio. Make sure OpenScreen can access the folder it's in, then regenerate." without the IPC wrapper; after `chmod 644`, *Regenerate* runs again (not persisted as no-audio). **Metal compositor (8caaa20d, dd5e914b, 0624c148, 2fb994eb):** laptop frame, 3D cursor, click impact, image background blurred 53 %, Aurora, captions; MP4 1080p60 export of 3374 frames, video and audio 56.23 s against 56.21 s computed (66.154 s minus 5.09 s of cuts minus the 2× and 16× gains), full decode clean; extracted frames show the laptop with its shadow, the blurred background, the 3D cursor with its shadow, rounded screen corners and captions. During playback the preview followed image → image, blur 52 → 1 %, image → colour and back. **Also seen:** the playhead keeps its time when another project is opened (0:41.0 from A shown in B and C). With a hand-written `camDeviceName` and a null `camDeviceId`, the HUD keeps the camera on and the file at `true` (`useCameraHudSync` compares the empty selection with the empty id and returns early); the app always writes both together, so this state is artificial. Out of scope: a 10 s extract with speech under loud video sound transcribes as no speech in Auto (detected `en`, p 0.35) and as "*Musique*" in French. **Not run:** webcam and microphone (none on this Mac), so the Windows row's webcam-past-the-end defect cannot be checked here; click impact not confirmed (the fixture's cursor sidecar has no clicks, and none was identified around the nine clicks of take 2 at preview scale); Aurora motion on a plain colour; the 4× preview lag of the rc.12 row (not re-measured: this pass played 2× and 16× regions without timing the lag); `Esc`; tray; GIF; A/B against rc.12; Gatekeeper (the DMG came through `gh`, no quarantine flag); Intel. | | 2026-10-03 | installed `v2.0.0-rc.15` (build run 37124242835, tag `ccd8b449`, NSIS sha512 matched `latest.yml`), installed over rc.12 for all users, About reports `2.0.0-rc.15` (`win32 x64 · nsis`, Electron 41.2.1, Chromium 146.0.7680.188, Node 24.14.1). `v2.0.0` was promoted from this candidate during the pass and differs from it only by the version bump (`eff86dcb`), so the results hold for the stable code | Windows 11 Home 26200, 1920×1080 @ 125 %, AMD Ryzen 5 7520U / Radeon iGPU, built-in USB2.0 HD UVC WebCam (best mode 1280×720@30), Microphone Array (AMD) | Partial — 6 defects | **Whole-file pass on Windows.** Real OS mouse and keyboard input throughout. Computer-use screenshots came back grey again, so the app was observed through `PrintWindow` on its own windows and DOM reads over CDP, and every picture and sound claim below was measured on the file with ffprobe/ffmpeg. **Defects:** (1) #1004, HUD drag at 125 %: the HUD window grows about 1.5 px per drag step (1180×882 → 1232×935 over ~43 steps) and the bar jumps by half the growth on release; after a long drag down the bar sat partly under the taskbar (centre y 1042, work area ends at 1032). (2) #1005, *Edit clip*: Reset, Cancel and Apply sit inside the scrolling body. In a 1240×1000 window (794 css px tall) the card is clipped above Apply, and a click where Apply should be lands on the backdrop, closing the dialog and discarding the edit. It fits when the window is maximized. (3) #1006, the rail's "Choose a clip to edit" menu renders under the timeline toolbar: Clip 2 and Clip 3 are covered (`elementFromPoint` returns `_tlToolbar`) and clicks on them do nothing. (4) #1007, a clip duplicated with Ctrl+C / Ctrl+V does not carry its anchored zoom or Full Camera region: `duplicateClip` in `src/lib/ai-edition/document/timeline.ts` copies only `trimRanges`. (5) #1008, zoom repel across a clip boundary: a zoom dragged into a neighbour that ends at its clip's end jumps past it into the next clip, shrinking from 12.97 to 8.23 s and leaving a zero-length fragment on the first clip. Ctrl+Z restores it. (6) #1009, the wordmark menu does not close on a click on the bare top bar. The header is a `-webkit-app-region: drag` area, which never delivers the document `mousedown` that `AppMenu` listens for. It does close on the preview, on a top-bar button and on the wordmark. **Passed: HUD and capture.** One HUD, `[content-protection] OFF` logged; horizontal ↔ vertical layout; every tooltip, idle and recording, beside or above the bar and unclipped; language menu (15 locales); camera and microphone toggles in one click; device settings: three inputs, level meter, live camera preview, camera quality row (1080p persisted across a relaunch; hand-edited 720p and an absent key both read 4K). Hide bar, then the tray overflow icon brings the HUD back; Quit leaves 0 processes; a second launch exits 0. With no source the record tooltip reads "Choose a screen or window to record" and one click opens the picker, then starts. While recording: source button disabled, toggles `aria-disabled`, gear inert, tray tooltip "Recording: Tout l'écran". Pause freezes the timer and resume advances it; restart starts a new helper and deletes the first file; cancel deletes the file and opens no editor. Stop opens the editor with "Recording added to a new project". Screen H.264 High bt709 1920×1080; AAC 48 kHz; webcam 1280×720 ~30 fps at 8 Mbit/s, the camera's best mode, not upscaled, matching `webcamFormat`. Microphone-only, system-audio-only and all-off takes behave; window source 1270×668, non-black; an odd 1001×601 window gives an even 992×596 file; a context menu inside the window is recorded and cut at its edge. Without the flag the HUD and an open Notes window are absent from the take. Record mode: rows, live camera preview, microphone meter, device switch, source modal with badge, *Start recording* hands over to the HUD; auto-zoom on gives one merged zoom over three clicks, off gives none, and the setting persists; editable cursor off hides the auto-zoom row and the Cursor facet; *Hide desktop icons* shows a bare wallpaper and restores the icons after. **Passed: editor.** Rename, Saved dot, tabs, chat panel toggle, tooltips with key chips, dividers; play, pause, seek playing and paused, ±1 frame arrows, stop at end; ruler click and drag, navigator narrow and pan, labels without collision, playhead within 0.5 px; Ctrl+S toast with a stable top bar. Clips: media card drag and *Add to timeline*, reorder, Edit clip grips, 1:1 crop held through a corner drag, Free, Cancel, pencil, delete. Trim, zoom (levels, custom, out-of-range message, focus drag, 3D modes), speed (presets, custom, 16× cap message, playback at 16×), annotation (text, sizes, plates, palette, Typewriter, image, arrow, blur mosaic and smooth, oval, clamped, dragged into padding); Ctrl+D, Ctrl+Z, Ctrl+Shift+Z. Modifiers inside a trim fire on the parked frame. Copy and paste of zoom, annotation and trim with their toasts and properties; an empty copy is a no-op; deleting a clip removes its modifiers; Edit clip clamps them; a Full Camera region across a junction splits and merges as clips move. Imported music: fragments across a junction with `offsetMs`, −18 dB and 1 s fades, gain, *Reset audio*; voiceover lands at the playhead. Transcript: clip order, silences, monotonic word seeks, live cue, Backspace skip and restore, silence trim; captions in four styles and six anchors; French translation with a *Display* row, and deleting it leaves the words untouched; typing in the packaged gate does nothing; a double-click correction survives *Regenerate as English* (model reused); 101 languages, detected language on the card. Composition: colour, gradient and one-colour gradient, wallpaper, animations move during playback only (21–23 distinct frames in 3 s, None 1), blur, 9:16 with Whole and Follow cursor, Auto padding with even borders, shadows, four frame styles in both themes. Camera layout: four presets, mirror, square, size (default 40, max 60), roundness, positions; background Cutout, Blur with an intensity slider, Custom image and colour, and the custom-colour mask reaches a 720p export; webcam crop corner zoom (124 %) and frame move. Audio facet: +6.5 dB raises the export from −17.2 to −12.7 LUFS with the peak held at −1.3 dBFS; reset returns 0 dB. Cursor facet: hide, style, size (to 5.8 of 6) and 3D cursor all show in the preview. Depth of field appears once a zoom has a 3D preset and changes the preview. Wordmark menu: rows as listed, About row version, arrows wrap both ways; Keyboard Shortcuts as one dialog: Z → Q changes the lane hint, Q adds a zoom, Z does nothing, reset restores Z; Ctrl+O; AI settings as one dialog; light and dark themes, dialogs included; French UI with one-line menu rows (#969), back to English; About lists the versions and channel, and Copy puts the same block on the clipboard; the HUD panel's version row and Check for Updates ("Checking…", disabled, the same dialog as the menu, re-enabled after). AI chat (deepseek): "add one zoom at 2x from 5 s to 8 s" lands as 5–8 s at 2.20× with the reason given, is undone by one Ctrl+Z and redone. Closing the editor quits the app and writes `editor-window.json` (`maximized: false`); after a relaunch the last project's settings and the main project's two clips, crop, regions, music, voiceover and captions are as left, and a paste before any copy changes nothing. A schema-4 project from July opens, migrates to schema 8, and keeps its clip, seven zooms and annotation. **Export:** MP4 1080p30 of the two-clip project, 10288 frames, audio and video within one frame, −16.0 LUFS and −1.4 dBTP, annotation upright, caption plate at x = 108 for a 10 % inset, Full Camera where mapped, music ducked about 10 dB under speech with fades at the track's own edges, muted track absent, voiceover not stretched by a 2× region. GIF 25 fps: cancel removes the partial file and leaves the existing destination untouched; the full render (~127 s, #952) gives 434 frames, infinite loop, consistent in ffmpeg and GDI+. **Also seen:** a take with a ~4 s pause had audio 1.09 s past its video, against 0.02–0.17 s without a pause; three dedicated paused takes gave 0.40, 0.15 and 0.03 s. The export ends each clip at its last video frame (10288 frames against the 10320 the progress counted), so the timeline and the export disagree by that tail. The editor reopens 4 px taller than it closed at 125 %. **Not run:** the A/V offset across a pause measured directly (the flash-and-beep rig never painted on the captured desktop), the HUD with no camera (needs the camera disabled in Windows), tray right-click menus (Update Settings, Stop Recording, Save Diagnostics: Explorer is granted click-only), the Alt Help menu (an Alt tap showed nothing, inconclusive), Ctrl/Shift+wheel on the timeline, cursor auto-hide, smoothing, motion blur and click effects, annotation animations other than Typewriter, Depth of field in an export, chat rewind, history and Smart cuts, a software-encoder take, the crackle and hole scan, multiple displays, 4K and 1440p cameras, 150 % DPI, macOS and Linux. | | 2026-10-04 | dev `feat/multi-camera-recording` at `212c07f7` (feature head `23ea0751` + the HUD probe lint fix); `wgc-capture.exe` rebuilt from this tree by `npm run build:native:win` | Windows 11 on ARM64 (Snapdragon X Elite), one built-in camera (`AI Front Camera`), driven without computer-use | **Partial — the two-webcam run on site is outstanding; sub-project not complete until logged** | **Ran:** unit suite (`npm run test`, 304 of 305 files, the one failure `LeftPanel.copyMessage.test.tsx` passed alone, a load flake), both `tsc`, lint (0 errors, 26 warnings, the base count), `i18n:check`; the helper's native unit tests (`node scripts/build-windows-wgc-helper.mjs`, all pass); helper smoke runs `npm run test:wgc-helper:win`, `-- --webcam` and `-- --webcam --missing-second-webcam` (index 1 dropped and named, camera 1 kept); a direct helper run listing the one camera twice: camera 1 recorded, camera 2 dropped with `webcam-unavailable` index 1 and "already recording in this take", `webcamPaths` held one file, exit 0, camera file 4.81 s against the screen's 4.80 s. **Not run:** the app click-through (HUD *Additional cameras*, editor, save and reopen, the session manifest's `additionalWebcams`, the "Not recorded" notice): no computer-use here and only one camera. **Not covered at all:** two real cameras recording together, identical-name Brios, 1080p and 4K load, clap sync, macOS and Linux. | +| 2026-10-05 | dev `feat/multi-camera-recording` at `86d49475`; `wgc-capture.exe` from `electron/native/wgc-capture/build` (4 Oct 13:05, `findstr` hit the fix-wave string *Could not remove the dropped camera file* and the control `encoder-selection`); `compositor_view.node` built from this tree by `npm run build:native:compositor` (ARM64) and loaded through `OPENSCREEN_COMPOSITOR_VIEW_NODE` next to the vendored win32-arm64 ffmpeg | Windows 11 on ARM64 (Snapdragon X Elite), **Logitech BRIO** (USB) + built-in `AI Front Camera`, 2560×1440 display | **Partial — two real cameras pass; Brio + C920, two identical Brios, 4K load and a mid-take camera loss not run yet** | *Helper:* Brio as camera 1 + built-in as camera 2 at 1080p30, 38 s take with three claps — both camera files 1146 frames / 38.19 s, screen 38.20 s at a steady 30 fps (`[pacing] frames=1146 elapsed_ms=38201`), camera 2 skipped the claimed Brio (`already recording in this take`), `webcamPaths` listed both. Clap sync by per-frame motion cross-correlation: 0 frames over the whole take, ≤ 1 frame (33 ms) per 3 s window; an 8-frame strip of the strongest clap shows the hands meeting and opening in the same frames on both. *App (HUD driven by the user):* Brio as camera 1, built-in ticked under *Additional cameras* — three files written, session `additionalWebcams` with label `AI Front Camera`, no *Not recorded* notice; Studio shows screen + camera 1 as before; the project holds `additionalCameraTracks` (1920×1080, offset 0, label) and keeps it after reopen, an edit and save. | The editor cannot select camera 2 yet — by design, sub-projects 2/3. First dev launch had no `compositor_view.node` in the worktree (blank Studio: `native addon not present`) — environment, not product. | From b4d10be032250cb5262a22d129ba103b91c95c23 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:49:53 +0200 Subject: [PATCH 19/22] fix(wgc): stop and report a camera that is unplugged mid-take A failed ReadSample was counted and retried forever, so an unplugged camera kept its file running to the end on its last picture and stayed in webcamPaths. The Media Foundation capture now latches lost on a device invalidated or hardware start failure, on end of stream during the take, or after a second of consecutive read failures; the DirectShow fallback latches it on EC_DEVICE_LOST (removal), EC_ERRORABORT or EC_STREAM_ERROR_STOPPED. The writer loop disables a lost camera like any mid-take failure, so its file ends at the loss and the app names it as stopped early. --- electron/native/README.md | 2 +- electron/native/wgc-capture/CMakeLists.txt | 16 ++++++++ .../wgc-capture/src/dshow_webcam_capture.cpp | 40 ++++++++++++++++++ .../wgc-capture/src/dshow_webcam_capture.h | 8 ++++ electron/native/wgc-capture/src/main.cpp | 15 +++++++ .../native/wgc-capture/src/webcam_capture.cpp | 35 ++++++++++++++++ .../native/wgc-capture/src/webcam_capture.h | 10 +++++ .../native/wgc-capture/src/webcam_loss.cpp | 28 +++++++++++++ electron/native/wgc-capture/src/webcam_loss.h | 34 +++++++++++++++ .../wgc-capture/src/webcam_loss_test.cpp | 41 +++++++++++++++++++ scripts/build-windows-wgc-helper.mjs | 8 ++++ 11 files changed, 236 insertions(+), 1 deletion(-) create mode 100644 electron/native/wgc-capture/src/webcam_loss.cpp create mode 100644 electron/native/wgc-capture/src/webcam_loss.h create mode 100644 electron/native/wgc-capture/src/webcam_loss_test.cpp diff --git a/electron/native/README.md b/electron/native/README.md index d9856d052..a9c6f8b83 100644 --- a/electron/native/README.md +++ b/electron/native/README.md @@ -94,7 +94,7 @@ Several cameras: a `webcams` list records up to four cameras, each into its own ] ``` -Every per-camera event carries the camera's `index`, its position in the list after entries without `camPath` have been skipped: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order — an empty list when every camera that wrote a file of its own was disabled mid-take, so a reader can tell "stopped early" from an older helper that never sends the key. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. +Every per-camera event carries the camera's `index`, its position in the list after entries without `camPath` have been skipped: `webcam-format` (`{"event":"webcam-format","schemaVersion":2,"index":0,"width":…,"height":…,"fps":…,"deviceName":"…"}`) for each camera that opened, and `{"event":"warning","code":"webcam-unavailable","index":1,"deviceName":"…","message":"…"}` for each that did not (`code` comes before `index` so older substring readers still match). A camera that cannot be opened, or whose encoder or capture will not start (the message says which), is dropped and the rest of the take goes on; a camera whose encoder rejects a sample mid-take is disabled on its own, without stopping the screen or the other cameras. A camera unplugged mid-take (or whose reads keep failing for a second) is disabled the same way: its file ends at the loss, is still finalized at stop, and is left out of `webcamPaths`. `recording-stopped` keeps `webcamPath` (camera 0, when it recorded) and adds `webcamPaths`, the files of every camera still recording at stop, in index order — an empty list when every camera that wrote a file of its own was disabled mid-take, so a reader can tell "stopped early" from an older helper that never sends the key. Both are printed before the camera files are finalized (see the stop sequence), so a camera whose finalize fails is reported on stderr and by a non-zero exit, not removed from the list. Each camera finalizes in its own `[stop-timing]` step, `webcam-encoder-finalize-`. Container: recordings are written as fragmented MP4 (`MFCreateFMPEG4MediaSink` + `MFCreateSinkWriterFromMediaSink`, `MF_MPEG4SINK_MIN_FRAGMENT_DURATION` = 1s) rather than plain MP4. A plain MP4 has no index until `IMFSinkWriter::Finalize()` writes `moov` at the very end, so when the shutdown watchdog force-exits a wedged helper the file on disk holds every frame and no way to read them — that is why issues #252 / #292 / #327 cost the whole recording rather than the frozen tail of it. A fragmented MP4 writes its index up front and its samples in self-describing `moof`+`mdat` pairs, so the same kill leaves a file that plays up to the last complete fragment. This does not fix the freeze; it removes the data loss the freeze causes. Because the fragmented sink needs both output media types at construction, the sink writer is built from a media sink instead of from a URL, and the helper reads the video/audio stream positions back off the sink rather than assuming them. If any of that is unavailable on a machine, the helper retries with the plain container and says so — `container` in the `encoder-selection` event is `fragmented-mp4` or `mp4`, and it reports what was used, not what was asked for. diff --git a/electron/native/wgc-capture/CMakeLists.txt b/electron/native/wgc-capture/CMakeLists.txt index 870f4003a..009a04677 100644 --- a/electron/native/wgc-capture/CMakeLists.txt +++ b/electron/native/wgc-capture/CMakeLists.txt @@ -62,6 +62,8 @@ add_executable(wgc-capture src/wasapi_render_keepalive.cpp src/wasapi_render_keepalive.h src/webcam_capture.cpp + src/webcam_loss.cpp + src/webcam_loss.h src/webcam_format.cpp src/webcam_format.h src/webcam_config.cpp @@ -234,3 +236,17 @@ target_compile_definitions(device_selection_test PRIVATE ) target_compile_options(device_selection_test PRIVATE /EHsc /W4 /utf-8) + +add_executable(webcam_loss_test + src/webcam_loss.cpp + src/webcam_loss.h + src/webcam_loss_test.cpp +) + +target_compile_definitions(webcam_loss_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(webcam_loss_test PRIVATE /EHsc /W4 /utf-8) diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp index c7773505a..837f70f68 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp @@ -2,6 +2,7 @@ #include "realtime_scheduling.h" #include "webcam_format.h" +#include "webcam_loss.h" #include #include @@ -176,6 +177,8 @@ struct DirectShowWebcamCapture::Impl { Microsoft::WRL::ComPtr sampleGrabber; Microsoft::WRL::ComPtr nullRenderer; Microsoft::WRL::ComPtr mediaControl; + /** Where the graph says the device left; optional, see captureLoop. */ + Microsoft::WRL::ComPtr mediaEvent; bool comInitialized = false; bool running = false; }; @@ -275,6 +278,7 @@ bool DirectShowWebcamCapture::buildGraph( // Every attempt starts from empty filters. A RenderStream that fails can // leave pins connected behind it, and retrying on top of that half-built // graph is how you get a second failure that says nothing about the format. + impl_->mediaEvent.Reset(); impl_->mediaControl.Reset(); impl_->nullRenderer.Reset(); impl_->sampleGrabber.Reset(); @@ -434,6 +438,10 @@ bool DirectShowWebcamCapture::initialize( if (!succeeded(impl_->graph.As(&impl_->mediaControl), "QueryInterface(IMediaControl)")) { return false; } + // Best-effort: without it a lost device goes unnoticed, as it always did. + if (FAILED(impl_->graph.As(&impl_->mediaEvent))) { + impl_->mediaEvent.Reset(); + } return true; } @@ -546,6 +554,7 @@ void DirectShowWebcamCapture::stop() { impl_->mediaControl->Stop(); } impl_->running = false; + impl_->mediaEvent.Reset(); impl_->mediaControl.Reset(); impl_->nullRenderer.Reset(); impl_->sampleGrabber.Reset(); @@ -563,6 +572,11 @@ void DirectShowWebcamCapture::captureLoop() { const MmcssThread mmcss(L"Capture"); const HRESULT coinitHr = CoInitializeEx(nullptr, COINIT_MULTITHREADED); while (!stopRequested_ && impl_ && impl_->sampleGrabber) { + // The sample grabber keeps handing back its last buffer after the + // camera is gone, so the frames cannot tell; the graph's events can. + if (deviceLeftGraph()) { + break; + } long bufferSize = 0; HRESULT hr = impl_->sampleGrabber->GetCurrentBuffer(&bufferSize, nullptr); if (SUCCEEDED(hr) && bufferSize > 0) { @@ -579,6 +593,32 @@ void DirectShowWebcamCapture::captureLoop() { } } +bool DirectShowWebcamCapture::deviceLeftGraph() { + if (!impl_->mediaEvent) { + return false; + } + long code = 0; + LONG_PTR param1 = 0; + LONG_PTR param2 = 0; + // A zero timeout drains what is queued without waiting for more. + while (SUCCEEDED(impl_->mediaEvent->GetEvent(&code, ¶m1, ¶m2, 0))) { + impl_->mediaEvent->FreeEventParams(code, param1, param2); + // EC_DEVICE_LOST's second parameter is 0 when the device was removed, + // 1 when it came back; a stream error or abort stops the graph. + const bool lost = (code == EC_DEVICE_LOST && param2 == 0) || code == EC_ERRORABORT || + code == EC_STREAM_ERROR_STOPPED; + if (lost && !lost_.exchange(true)) { + reportWebcamLost(selectedDeviceName_, static_cast(code)); + return true; + } + } + return false; +} + +bool DirectShowWebcamCapture::isLost() const { + return lost_; +} + void DirectShowWebcamCapture::storeFrame(const BYTE* buffer, long length) { const int destinationStride = width_ * 4; const int sourceStride = sourceStride_ > 0 ? sourceStride_ : destinationStride; diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.h b/electron/native/wgc-capture/src/dshow_webcam_capture.h index 2c4393288..5e26098d0 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.h +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.h @@ -74,6 +74,11 @@ class DirectShowWebcamCapture { * none, the filter CLSID when no moniker names it), for the take's claims. */ const std::wstring& deviceIdentity() const; + /** + * Did the graph report the device gone (EC_DEVICE_LOST removal, a stream + * error, or an abort)? Latched; safe to ask from any thread. + */ + bool isLost() const; void storeFrame(const BYTE* buffer, long length); private: @@ -85,6 +90,8 @@ class DirectShowWebcamCapture { struct Impl; void captureLoop(); + /** Drains the graph's queued events; true once one of them says the device left. */ + bool deviceLeftGraph(); /** * Builds source -> sample grabber -> null renderer and connects it. * @@ -121,6 +128,7 @@ class DirectShowWebcamCapture { Impl* impl_ = nullptr; std::thread thread_; std::atomic stopRequested_ = false; + std::atomic lost_ = false; std::mutex frameMutex_; std::vector latestFrame_; uint64_t latestFrameSequence_ = 0; diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index f6b4c8ffd..77dc692db 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -1179,6 +1179,21 @@ int wmain(int argc, wchar_t* argv[]) { } // Screen first, then every camera, as before there was more than one. for (const auto& stream : webcams) { + if (stream->active && stream->capture.isLost()) { + // Unplugged mid-take: its file ends here instead of + // repeating the last picture to the end, it is left out + // of `webcamPaths` and still finalized at stop. The + // screen and the other cameras go on. + std::cerr << "ERROR: Camera " << stream->index + << " was lost during the take; disabling it" << std::endl; + stream->active = false; + if (inlineWebcam == stream.get()) { + // Nor is its frozen picture drawn into the screen. + inlineWebcam = nullptr; + pictureChanged = true; + } + continue; + } if (stream->active && stream->pullVisibleFrame()) { pictureChanged = pictureChanged || !stream->writeSeparate; } diff --git a/electron/native/wgc-capture/src/webcam_capture.cpp b/electron/native/wgc-capture/src/webcam_capture.cpp index b67e1ceac..45d2ebdd7 100644 --- a/electron/native/wgc-capture/src/webcam_capture.cpp +++ b/electron/native/wgc-capture/src/webcam_capture.cpp @@ -2,6 +2,7 @@ #include "realtime_scheduling.h" #include "webcam_format.h" +#include "webcam_loss.h" #include #include @@ -10,6 +11,7 @@ #include #include #include +#include namespace { @@ -403,6 +405,9 @@ void WebcamCapture::captureLoop() { CoInitializeEx(nullptr, COINIT_MULTITHREADED); const auto loopStartedAt = std::chrono::steady_clock::now(); + // Start of the current run of failed reads; empty while reads succeed. + std::optional failureRunStartedAt; + while (!stopRequested_) { DWORD streamIndex = 0; DWORD flags = 0; @@ -428,11 +433,27 @@ void WebcamCapture::captureLoop() { // wise write thousands of identical lines over a long take. readFailures_ += 1; lastReadFailure_ = hr; + const auto now = std::chrono::steady_clock::now(); + if (!failureRunStartedAt) { + failureRunStartedAt = now; + } + const int64_t failingForMs = + std::chrono::duration_cast(now - *failureRunStartedAt).count(); + if (isWebcamLossResult(hr, false, failingForMs)) { + markLost(hr); + break; + } std::this_thread::sleep_for(std::chrono::milliseconds(20)); continue; } + failureRunStartedAt.reset(); if ((flags & MF_SOURCE_READERF_ENDOFSTREAM) != 0) { sawEndOfStream_ = true; + // Stop has its own way out of this loop; end of stream before it + // is the camera leaving. + if (!stopRequested_ && isWebcamLossResult(hr, true, 0)) { + markLost(hr); + } break; } if (!sample) { @@ -513,6 +534,20 @@ void WebcamCapture::captureLoop() { CoUninitialize(); } +void WebcamCapture::markLost(HRESULT hr) { + if (lost_.exchange(true)) { + return; + } + reportWebcamLost(selectedDeviceName_, hr); +} + +bool WebcamCapture::isLost() const { + if (usingDirectShow_) { + return directShowCapture_.isLost(); + } + return lost_; +} + bool WebcamCapture::copyLatestFrame(WebcamFrameSnapshot& destination, uint64_t lastSeenSequence) { if (usingDirectShow_) { return directShowCapture_.copyLatestFrame(destination, lastSeenSequence); diff --git a/electron/native/wgc-capture/src/webcam_capture.h b/electron/native/wgc-capture/src/webcam_capture.h index 1d9e1fe5f..5888e5ed9 100644 --- a/electron/native/wgc-capture/src/webcam_capture.h +++ b/electron/native/wgc-capture/src/webcam_capture.h @@ -56,6 +56,13 @@ class WebcamCapture { */ bool deliversNv12() const; const std::wstring& selectedDeviceName() const; + /** + * Has the camera left the take -- unplugged, or its driver given up? + * + * Latched: once true it stays true, and the camera delivers no more + * frames. Safe to ask from any thread. + */ + bool isLost() const; private: bool selectDevice( @@ -64,12 +71,15 @@ class WebcamCapture { const DeviceClaims& claims); bool configureReader(int requestedWidth, int requestedHeight, int requestedFps, bool preferNv12); void captureLoop(); + /** Latches `lost_` and says so once; see `isWebcamLossResult`. */ + void markLost(HRESULT hr); Microsoft::WRL::ComPtr mediaSource_; Microsoft::WRL::ComPtr sourceReader_; DirectShowWebcamCapture directShowCapture_; std::thread thread_; std::atomic stopRequested_ = false; + std::atomic lost_ = false; std::mutex frameMutex_; std::vector latestFrame_; uint64_t latestFrameSequence_ = 0; diff --git a/electron/native/wgc-capture/src/webcam_loss.cpp b/electron/native/wgc-capture/src/webcam_loss.cpp new file mode 100644 index 000000000..7548f84c2 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_loss.cpp @@ -0,0 +1,28 @@ +#include "webcam_loss.h" + +#include + +#include + +bool isWebcamLossResult(HRESULT hr, bool endOfStream, int64_t consecutiveFailureMs) { + if (SUCCEEDED(hr)) { + return endOfStream; + } + if (hr == MF_E_VIDEO_RECORDING_DEVICE_INVALIDATED || hr == MF_E_HW_MFT_FAILED_START_STREAMING) { + return true; + } + return consecutiveFailureMs >= kWebcamLossFailureRunMs; +} + +void reportWebcamLost(const std::wstring& deviceName, HRESULT hr) { + std::string name; + const int size = WideCharToMultiByte( + CP_UTF8, 0, deviceName.data(), static_cast(deviceName.size()), nullptr, 0, nullptr, nullptr); + if (size > 0) { + name.resize(static_cast(size)); + WideCharToMultiByte( + CP_UTF8, 0, deviceName.data(), static_cast(deviceName.size()), name.data(), size, nullptr, nullptr); + } + std::cerr << "WARNING: Webcam lost during the take: " << name << " hr=0x" << std::hex + << static_cast(hr) << std::dec << std::endl; +} diff --git a/electron/native/wgc-capture/src/webcam_loss.h b/electron/native/wgc-capture/src/webcam_loss.h new file mode 100644 index 000000000..64af1f987 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_loss.h @@ -0,0 +1,34 @@ +#pragma once + +#include + +#include +#include + +/** How long reads may keep failing, back to back, before the camera counts as gone. */ +constexpr int64_t kWebcamLossFailureRunMs = 1000; + +/** + * Does this read result mean the camera has left the take? + * + * Media Foundation reports an unplugged camera as + * MF_E_VIDEO_RECORDING_DEVICE_INVALIDATED on every following ReadSample, and a + * camera whose driver gave up as MF_E_HW_MFT_FAILED_START_STREAMING; both are + * final. End of stream while the take is still running is the same thing said + * differently. Any other failure is forgiven once -- a camera can drop a read -- + * but not for `kWebcamLossFailureRunMs` in a row: a reader that fails that long + * delivers nothing, and treating it as alive is what froze a file on its last + * picture for the rest of the take. + * + * `consecutiveFailureMs` is the time since the first failure of the current + * run; a successful read ends the run. Ignored when `hr` succeeded. + */ +bool isWebcamLossResult(HRESULT hr, bool endOfStream, int64_t consecutiveFailureMs); + +/** + * The one line both capture backends print when a camera leaves the take. + * + * `hr` is the read result that decided it, or the DirectShow event code when + * the graph reported the loss. + */ +void reportWebcamLost(const std::wstring& deviceName, HRESULT hr); diff --git a/electron/native/wgc-capture/src/webcam_loss_test.cpp b/electron/native/wgc-capture/src/webcam_loss_test.cpp new file mode 100644 index 000000000..2a8462f8d --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_loss_test.cpp @@ -0,0 +1,41 @@ +#include "webcam_loss.h" + +#include + +#include + +namespace { +int failures = 0; +void expect(bool ok, const char* label) { + if (!ok) { + std::printf("FAIL %s\n", label); + ++failures; + } +} +} // namespace + +int main() { + // The code the on-site test logged after the C920 was unplugged. + expect(isWebcamLossResult(MF_E_VIDEO_RECORDING_DEVICE_INVALIDATED, false, 0), + "device invalidated: lost on the first read"); + expect(isWebcamLossResult(MF_E_HW_MFT_FAILED_START_STREAMING, false, 0), + "hardware failed to start streaming: lost"); + expect(isWebcamLossResult(S_OK, true, 0), "end of stream during the take: lost"); + + expect(!isWebcamLossResult(E_FAIL, false, 0), "one other failure: not lost"); + expect(!isWebcamLossResult(E_FAIL, false, kWebcamLossFailureRunMs - 1), + "other failures for just under a second: not lost"); + expect(isWebcamLossResult(E_FAIL, false, kWebcamLossFailureRunMs), + "other failures for a second: lost"); + expect(isWebcamLossResult(MF_E_NOTACCEPTING, false, 5000), "any failure code, long run: lost"); + + expect(!isWebcamLossResult(S_OK, false, 0), "S_OK: not lost"); + expect(!isWebcamLossResult(S_OK, false, 5000), "S_OK ends a run of failures: not lost"); + + if (failures == 0) { + std::printf("webcam_loss_test: all assertions passed\n"); + return 0; + } + std::printf("webcam_loss_test: %d failure(s)\n", failures); + return 1; +} diff --git a/scripts/build-windows-wgc-helper.mjs b/scripts/build-windows-wgc-helper.mjs index 8f687dae9..927a7a392 100644 --- a/scripts/build-windows-wgc-helper.mjs +++ b/scripts/build-windows-wgc-helper.mjs @@ -132,6 +132,14 @@ if (!fs.existsSync(deviceSelectionTestPath)) { await run(deviceSelectionTestPath, [], { cwd: BUILD_DIR }); console.log(`Passed ${deviceSelectionTestPath}`); +const webcamLossTestPath = path.join(BUILD_DIR, "webcam_loss_test.exe"); +if (!fs.existsSync(webcamLossTestPath)) { + throw new Error(`WGC helper build completed but ${webcamLossTestPath} was not found.`); +} +// Guards that an unplugged camera ends its file instead of freezing on its last picture. +await run(webcamLossTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${webcamLossTestPath}`); + const frameVisibilityTestPath = path.join(BUILD_DIR, "frame_visibility_test.exe"); if (!fs.existsSync(frameVisibilityTestPath)) { throw new Error(`WGC helper build completed but ${frameVisibilityTestPath} was not found.`); From 0aeb9145065a4bed9725dee0bc6499c70adc948c Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 5 Oct 2026 08:53:23 +0200 Subject: [PATCH 20/22] docs(e2e): log the on-site camera runs and the unplug fix --- technical-documentation/testing/manual-e2e-checklist.md | 1 + 1 file changed, 1 insertion(+) diff --git a/technical-documentation/testing/manual-e2e-checklist.md b/technical-documentation/testing/manual-e2e-checklist.md index 479a0910e..9e95eb6a8 100644 --- a/technical-documentation/testing/manual-e2e-checklist.md +++ b/technical-documentation/testing/manual-e2e-checklist.md @@ -871,3 +871,4 @@ On macOS 15.2+ it opens at launch, until it has been closed once, while one of i | 2026-10-03 | installed `v2.0.0-rc.15` (build run 37124242835, tag `ccd8b449`, NSIS sha512 matched `latest.yml`), installed over rc.12 for all users, About reports `2.0.0-rc.15` (`win32 x64 · nsis`, Electron 41.2.1, Chromium 146.0.7680.188, Node 24.14.1). `v2.0.0` was promoted from this candidate during the pass and differs from it only by the version bump (`eff86dcb`), so the results hold for the stable code | Windows 11 Home 26200, 1920×1080 @ 125 %, AMD Ryzen 5 7520U / Radeon iGPU, built-in USB2.0 HD UVC WebCam (best mode 1280×720@30), Microphone Array (AMD) | Partial — 6 defects | **Whole-file pass on Windows.** Real OS mouse and keyboard input throughout. Computer-use screenshots came back grey again, so the app was observed through `PrintWindow` on its own windows and DOM reads over CDP, and every picture and sound claim below was measured on the file with ffprobe/ffmpeg. **Defects:** (1) #1004, HUD drag at 125 %: the HUD window grows about 1.5 px per drag step (1180×882 → 1232×935 over ~43 steps) and the bar jumps by half the growth on release; after a long drag down the bar sat partly under the taskbar (centre y 1042, work area ends at 1032). (2) #1005, *Edit clip*: Reset, Cancel and Apply sit inside the scrolling body. In a 1240×1000 window (794 css px tall) the card is clipped above Apply, and a click where Apply should be lands on the backdrop, closing the dialog and discarding the edit. It fits when the window is maximized. (3) #1006, the rail's "Choose a clip to edit" menu renders under the timeline toolbar: Clip 2 and Clip 3 are covered (`elementFromPoint` returns `_tlToolbar`) and clicks on them do nothing. (4) #1007, a clip duplicated with Ctrl+C / Ctrl+V does not carry its anchored zoom or Full Camera region: `duplicateClip` in `src/lib/ai-edition/document/timeline.ts` copies only `trimRanges`. (5) #1008, zoom repel across a clip boundary: a zoom dragged into a neighbour that ends at its clip's end jumps past it into the next clip, shrinking from 12.97 to 8.23 s and leaving a zero-length fragment on the first clip. Ctrl+Z restores it. (6) #1009, the wordmark menu does not close on a click on the bare top bar. The header is a `-webkit-app-region: drag` area, which never delivers the document `mousedown` that `AppMenu` listens for. It does close on the preview, on a top-bar button and on the wordmark. **Passed: HUD and capture.** One HUD, `[content-protection] OFF` logged; horizontal ↔ vertical layout; every tooltip, idle and recording, beside or above the bar and unclipped; language menu (15 locales); camera and microphone toggles in one click; device settings: three inputs, level meter, live camera preview, camera quality row (1080p persisted across a relaunch; hand-edited 720p and an absent key both read 4K). Hide bar, then the tray overflow icon brings the HUD back; Quit leaves 0 processes; a second launch exits 0. With no source the record tooltip reads "Choose a screen or window to record" and one click opens the picker, then starts. While recording: source button disabled, toggles `aria-disabled`, gear inert, tray tooltip "Recording: Tout l'écran". Pause freezes the timer and resume advances it; restart starts a new helper and deletes the first file; cancel deletes the file and opens no editor. Stop opens the editor with "Recording added to a new project". Screen H.264 High bt709 1920×1080; AAC 48 kHz; webcam 1280×720 ~30 fps at 8 Mbit/s, the camera's best mode, not upscaled, matching `webcamFormat`. Microphone-only, system-audio-only and all-off takes behave; window source 1270×668, non-black; an odd 1001×601 window gives an even 992×596 file; a context menu inside the window is recorded and cut at its edge. Without the flag the HUD and an open Notes window are absent from the take. Record mode: rows, live camera preview, microphone meter, device switch, source modal with badge, *Start recording* hands over to the HUD; auto-zoom on gives one merged zoom over three clicks, off gives none, and the setting persists; editable cursor off hides the auto-zoom row and the Cursor facet; *Hide desktop icons* shows a bare wallpaper and restores the icons after. **Passed: editor.** Rename, Saved dot, tabs, chat panel toggle, tooltips with key chips, dividers; play, pause, seek playing and paused, ±1 frame arrows, stop at end; ruler click and drag, navigator narrow and pan, labels without collision, playhead within 0.5 px; Ctrl+S toast with a stable top bar. Clips: media card drag and *Add to timeline*, reorder, Edit clip grips, 1:1 crop held through a corner drag, Free, Cancel, pencil, delete. Trim, zoom (levels, custom, out-of-range message, focus drag, 3D modes), speed (presets, custom, 16× cap message, playback at 16×), annotation (text, sizes, plates, palette, Typewriter, image, arrow, blur mosaic and smooth, oval, clamped, dragged into padding); Ctrl+D, Ctrl+Z, Ctrl+Shift+Z. Modifiers inside a trim fire on the parked frame. Copy and paste of zoom, annotation and trim with their toasts and properties; an empty copy is a no-op; deleting a clip removes its modifiers; Edit clip clamps them; a Full Camera region across a junction splits and merges as clips move. Imported music: fragments across a junction with `offsetMs`, −18 dB and 1 s fades, gain, *Reset audio*; voiceover lands at the playhead. Transcript: clip order, silences, monotonic word seeks, live cue, Backspace skip and restore, silence trim; captions in four styles and six anchors; French translation with a *Display* row, and deleting it leaves the words untouched; typing in the packaged gate does nothing; a double-click correction survives *Regenerate as English* (model reused); 101 languages, detected language on the card. Composition: colour, gradient and one-colour gradient, wallpaper, animations move during playback only (21–23 distinct frames in 3 s, None 1), blur, 9:16 with Whole and Follow cursor, Auto padding with even borders, shadows, four frame styles in both themes. Camera layout: four presets, mirror, square, size (default 40, max 60), roundness, positions; background Cutout, Blur with an intensity slider, Custom image and colour, and the custom-colour mask reaches a 720p export; webcam crop corner zoom (124 %) and frame move. Audio facet: +6.5 dB raises the export from −17.2 to −12.7 LUFS with the peak held at −1.3 dBFS; reset returns 0 dB. Cursor facet: hide, style, size (to 5.8 of 6) and 3D cursor all show in the preview. Depth of field appears once a zoom has a 3D preset and changes the preview. Wordmark menu: rows as listed, About row version, arrows wrap both ways; Keyboard Shortcuts as one dialog: Z → Q changes the lane hint, Q adds a zoom, Z does nothing, reset restores Z; Ctrl+O; AI settings as one dialog; light and dark themes, dialogs included; French UI with one-line menu rows (#969), back to English; About lists the versions and channel, and Copy puts the same block on the clipboard; the HUD panel's version row and Check for Updates ("Checking…", disabled, the same dialog as the menu, re-enabled after). AI chat (deepseek): "add one zoom at 2x from 5 s to 8 s" lands as 5–8 s at 2.20× with the reason given, is undone by one Ctrl+Z and redone. Closing the editor quits the app and writes `editor-window.json` (`maximized: false`); after a relaunch the last project's settings and the main project's two clips, crop, regions, music, voiceover and captions are as left, and a paste before any copy changes nothing. A schema-4 project from July opens, migrates to schema 8, and keeps its clip, seven zooms and annotation. **Export:** MP4 1080p30 of the two-clip project, 10288 frames, audio and video within one frame, −16.0 LUFS and −1.4 dBTP, annotation upright, caption plate at x = 108 for a 10 % inset, Full Camera where mapped, music ducked about 10 dB under speech with fades at the track's own edges, muted track absent, voiceover not stretched by a 2× region. GIF 25 fps: cancel removes the partial file and leaves the existing destination untouched; the full render (~127 s, #952) gives 434 frames, infinite loop, consistent in ffmpeg and GDI+. **Also seen:** a take with a ~4 s pause had audio 1.09 s past its video, against 0.02–0.17 s without a pause; three dedicated paused takes gave 0.40, 0.15 and 0.03 s. The export ends each clip at its last video frame (10288 frames against the 10320 the progress counted), so the timeline and the export disagree by that tail. The editor reopens 4 px taller than it closed at 125 %. **Not run:** the A/V offset across a pause measured directly (the flash-and-beep rig never painted on the captured desktop), the HUD with no camera (needs the camera disabled in Windows), tray right-click menus (Update Settings, Stop Recording, Save Diagnostics: Explorer is granted click-only), the Alt Help menu (an Alt tap showed nothing, inconclusive), Ctrl/Shift+wheel on the timeline, cursor auto-hide, smoothing, motion blur and click effects, annotation animations other than Typewriter, Depth of field in an export, chat rewind, history and Smart cuts, a software-encoder take, the crackle and hole scan, multiple displays, 4K and 1440p cameras, 150 % DPI, macOS and Linux. | | 2026-10-04 | dev `feat/multi-camera-recording` at `212c07f7` (feature head `23ea0751` + the HUD probe lint fix); `wgc-capture.exe` rebuilt from this tree by `npm run build:native:win` | Windows 11 on ARM64 (Snapdragon X Elite), one built-in camera (`AI Front Camera`), driven without computer-use | **Partial — the two-webcam run on site is outstanding; sub-project not complete until logged** | **Ran:** unit suite (`npm run test`, 304 of 305 files, the one failure `LeftPanel.copyMessage.test.tsx` passed alone, a load flake), both `tsc`, lint (0 errors, 26 warnings, the base count), `i18n:check`; the helper's native unit tests (`node scripts/build-windows-wgc-helper.mjs`, all pass); helper smoke runs `npm run test:wgc-helper:win`, `-- --webcam` and `-- --webcam --missing-second-webcam` (index 1 dropped and named, camera 1 kept); a direct helper run listing the one camera twice: camera 1 recorded, camera 2 dropped with `webcam-unavailable` index 1 and "already recording in this take", `webcamPaths` held one file, exit 0, camera file 4.81 s against the screen's 4.80 s. **Not run:** the app click-through (HUD *Additional cameras*, editor, save and reopen, the session manifest's `additionalWebcams`, the "Not recorded" notice): no computer-use here and only one camera. **Not covered at all:** two real cameras recording together, identical-name Brios, 1080p and 4K load, clap sync, macOS and Linux. | | 2026-10-05 | dev `feat/multi-camera-recording` at `86d49475`; `wgc-capture.exe` from `electron/native/wgc-capture/build` (4 Oct 13:05, `findstr` hit the fix-wave string *Could not remove the dropped camera file* and the control `encoder-selection`); `compositor_view.node` built from this tree by `npm run build:native:compositor` (ARM64) and loaded through `OPENSCREEN_COMPOSITOR_VIEW_NODE` next to the vendored win32-arm64 ffmpeg | Windows 11 on ARM64 (Snapdragon X Elite), **Logitech BRIO** (USB) + built-in `AI Front Camera`, 2560×1440 display | **Partial — two real cameras pass; Brio + C920, two identical Brios, 4K load and a mid-take camera loss not run yet** | *Helper:* Brio as camera 1 + built-in as camera 2 at 1080p30, 38 s take with three claps — both camera files 1146 frames / 38.19 s, screen 38.20 s at a steady 30 fps (`[pacing] frames=1146 elapsed_ms=38201`), camera 2 skipped the claimed Brio (`already recording in this take`), `webcamPaths` listed both. Clap sync by per-frame motion cross-correlation: 0 frames over the whole take, ≤ 1 frame (33 ms) per 3 s window; an 8-frame strip of the strongest clap shows the hands meeting and opening in the same frames on both. *App (HUD driven by the user):* Brio as camera 1, built-in ticked under *Additional cameras* — three files written, session `additionalWebcams` with label `AI Front Camera`, no *Not recorded* notice; Studio shows screen + camera 1 as before; the project holds `additionalCameraTracks` (1920×1080, offset 0, label) and keeps it after reopen, an edit and save. | The editor cannot select camera 2 yet — by design, sub-projects 2/3. First dev launch had no `compositor_view.node` in the worktree (blank Studio: `native addon not present`) — environment, not product. | +| 2026-10-05 | dev `feat/multi-camera-recording` at `e0067391` (`fix(wgc): stop and report a camera that is unplugged mid-take`); `wgc-capture.exe` rebuilt by `npm run build:native:win`, `findstr` hit *Webcam lost during the take* and the control `encoder-selection` in `build` and `binwin32-x64` | Windows 11 on ARM64 (Snapdragon X Elite), **two Logitech BRIO** (identical names, USB instances `C&1EF69AFB` and `9&299A5C35`), **HD Pro Webcam C920**, built-in `AI Front Camera`; helper-level runs with a `webcams` list | **Partial — every on-site case passes at 1080p; 4K with a second camera paces cameras at 27.8 fps; the unplug notice was checked at helper level, not in the app UI** | *Brio + C920, 1080p:* 310/310/310 frames (camera 1, camera 2, screen). *Two identical Brios:* camera 2 skipped the claimed Brio (`already recording in this take`) and opened the other one — two different pictures; which Brio became camera 1 followed enumeration order, not intent (known limitation). *All four cameras at once, 1080p:* four different pictures, 392 frames each, screen 392 at a steady 30 fps. *4K:* one Brio alone 3840×2160 at 30 fps (401/401); two Brios requested at 4K → the Brio on a cheap USB cable only offers 1080p even alone (USB 2 link — hardware), and with 4K + 1080p both cameras paced at **27.8 fps** (363 frames in 13.07 s), screen 29.5 fps — all cameras are converted and encoded on the screen writer thread; follow-up: a worker thread per camera. *Unplug the C920 mid-take:* before the fix the C920 file ran on with a frozen picture and was still listed (`readFailures=737 lastReadHr=0xc00d3ea2`); after `e0067391` the helper logs `Webcam lost during the take: HD Pro Webcam C920 hr=0xc00d3ea2`, the C920 file ends at the loss (200 frames / 6.66 s), the Brio and screen run the full 27.7 s, and `webcamPaths` lists only the Brio — the input Electron turns into *Stopped early: …*. | The *Stopped early* toast itself was not watched in the app; it is unit-tested (`labelsOfWebcamsStoppedEarly`, renderer stop test). | From 08576219773b703e118f9b03f3a8f25e1f4b2ab1 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Sat, 3 Oct 2026 10:07:18 +0200 Subject: [PATCH 21/22] feat(camera): desk view for Full Camera sections, with a covered tilt MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A Full Camera section can now show a webcam tilted down onto the desk: "Desk view" in the inspector turns the camera 180°, switches the mirror off for that section (otherwise text on the page reads back to front) and shows the whole camera frame instead of the face crop. The timeline marks such sections with a rotate icon. The moment the camera is tilted is covered at both ends of the section: the whole camera picture is blurred and dimmed for the Full Camera grow (and the shrink), fading in and out around it, with a translated "Desk mode" label over it that can be switched off per section. Preview and export render it identically through the compositor's frame plan. - Rust: per-frame orientation (u/v bound swaps, crop bypass) and cover strength, a `LayerCB.cover` lane (HLSL, WGSL, MSL), the blur clamped to the camera's valid area so aligned decoder padding never bleeds in; the label is drawn at the cover strength of the same frame, so it fades exactly with the cover. - App: `rotation` / `mirror` / `deskLabel` on Full Camera regions (defaults are not stored), store and persistence, inspector controls, one generated label annotation per projected piece, translations in every locale. - Tests for each layer; WGSL is validated with naga on every host. --- crates/compositor/src/compositor_linux.rs | 39 ++- crates/compositor/src/compositor_macos.rs | 25 +- crates/compositor/src/compositor_windows.rs | 30 ++- crates/compositor/src/frame_geometry.rs | 249 +++++++++++++++++- crates/compositor/src/regions.rs | 213 ++++++++++++++- crates/compositor/src/scene.rs | 12 +- crates/compositor/src/shaders.hlsl | 27 +- crates/compositor/src/shaders.metal | 38 ++- crates/compositor/src/text_anim.rs | 60 ++++- crates/compositor/src/vk_shaders/layer.wgsl | 24 +- .../NativeCompositorOverlay.test.tsx | 45 +++- .../ai-edition/NativeCompositorOverlay.tsx | 11 +- .../ai-edition/v4/EditorShellV4.module.css | 7 + .../ai-edition/v4/FloatingInspector.test.tsx | 83 ++++++ .../ai-edition/v4/FloatingInspector.tsx | 64 +++++ src/components/ai-edition/v4/V4Timeline.tsx | 15 +- .../video-editor/projectPersistence.test.ts | 30 +++ .../video-editor/projectPersistence.ts | 6 + src/components/video-editor/types.ts | 9 +- src/i18n/locales/ar/settings.json | 12 + src/i18n/locales/cs/settings.json | 12 + src/i18n/locales/de/settings.json | 12 + src/i18n/locales/en/settings.json | 12 + src/i18n/locales/es/settings.json | 12 + src/i18n/locales/fr/settings.json | 12 + src/i18n/locales/it/settings.json | 12 + src/i18n/locales/ja-JP/settings.json | 12 + src/i18n/locales/ko-KR/settings.json | 12 + src/i18n/locales/pt-BR/settings.json | 12 + src/i18n/locales/ru/settings.json | 12 + src/i18n/locales/tr/settings.json | 12 + src/i18n/locales/vi/settings.json | 12 + src/i18n/locales/zh-CN/settings.json | 12 + src/i18n/locales/zh-TW/settings.json | 12 + .../store/documentWriteAudit.test.ts | 12 + src/lib/ai-edition/store/useTimeline.test.ts | 107 ++++++++ src/lib/ai-edition/store/useTimeline.ts | 65 ++++- .../ai-edition/timeline/timelineMap.test.ts | 12 + src/lib/cameraOrientation.test.ts | 82 ++++++ src/lib/cameraOrientation.ts | 52 ++++ src/lib/deskCover.ts | 9 + src/native/sceneDescription.test.ts | 159 ++++++++++- src/native/sceneDescription.ts | 120 ++++++++- .../testing/manual-e2e-checklist.md | 7 + 44 files changed, 1715 insertions(+), 77 deletions(-) create mode 100644 src/lib/cameraOrientation.test.ts create mode 100644 src/lib/cameraOrientation.ts create mode 100644 src/lib/deskCover.ts diff --git a/crates/compositor/src/compositor_linux.rs b/crates/compositor/src/compositor_linux.rs index f7c5204b6..c2b8bc17b 100644 --- a/crates/compositor/src/compositor_linux.rs +++ b/crates/compositor/src/compositor_linux.rs @@ -67,12 +67,12 @@ fn layer_source(models: bool) -> String { /// en parcourant les 18 wallpapers livres) en laissant le jeu actif resident. const IMG_CACHE_BUDGET_BYTES: u64 = 512 * 1024 * 1024; -/// Taille du buffer uniforme d'un calque : `LayerCB` entier (176 octets), le `struct Layer` de +/// Taille du buffer uniforme d'un calque : `LayerCB` entier (192 octets), le `struct Layer` de /// `layer.wgsl`. `blur.wgsl` n'en lit que les 128 premiers. const LAYER_BYTES: u64 = std::mem::size_of::() as u64; /// `&LayerCB` -> ses octets. `LayerCB` est `#[repr(C, align(16))]`, son layout EST le buffer -/// uniforme WGSL (dix vec4 et un vec2 + 2 f32 = 176 octets). +/// uniforme WGSL (douze vec4 = 192 octets). fn layer_bytes(cb: &LayerCB) -> &[u8] { unsafe { std::slice::from_raw_parts(cb as *const LayerCB as *const u8, LAYER_BYTES as usize) } } @@ -2485,17 +2485,16 @@ impl Compositor { let [cu0, cv0, cu1, cv1] = crate::frame_geometry::webcam_source_rect( [wcw, wch], [wtw as f32, wth as f32], - scene_ref - .as_ref() - .and_then(|scene| scene.layout.webcam_crop), + if g.webcam.full_frame { + None + } else { + scene_ref.as_ref().and_then(|scene| scene.layout.webcam_crop) + }, g.w_px[0] / g.w_px[1].max(0.0001), ); - // MIROIR : on inverse l'intervalle u. Le VS interpole `src` - // lineairement et `fs_main` ne re-clampe pas `i.uv`, donc un - // intervalle a l'envers suffit -- aucune retouche du WGSL. Apres le - // cover-crop les deux bornes sont strictement a l'interieur de la - // texture, donc le sampler ClampToEdge ne bave pas sur les bords. - let (u0, u1) = if lp.webcam_mirror { (cu1, cu0) } else { (cu0, cu1) }; + // Mirror and the desk-shot turn are both bound swaps: u for horizontal, v for vertical. + let (u0, u1) = if g.webcam.flip_u { (cu1, cu0) } else { (cu0, cu1) }; + let (v0, v1) = if g.webcam.flip_v { (cv1, cv0) } else { (cv0, cv1) }; // `src_prev` doit valoir EXACTEMENT le `src` de ce draw, miroir // compris : le shader s'en sert pour reconstruire l'UV de la frame // precedente, et un rect source qui ne correspond pas au calque @@ -2503,7 +2502,7 @@ impl Compositor { // n'a jamais ete affichee. Seul `dst_prev` porte le mouvement. let cb = LayerCB { dst: g.w_dst, - src: [u0, cv0, u1, cv1], + src: [u0, v0, u1, v1], quad_px: g.w_px, radius_px: g.w_radius, mode: 0.0, @@ -2515,9 +2514,10 @@ impl Compositor { // `fx.z` = mode, `fx.w` = intensite du flou. Contrat commun aux // trois back-ends, cf. `layer.wgsl` et `webcam-segmentation.md`. fx: [w_valid[0], w_valid[1], effect_code, blur_intensity], - src_prev: [u0, cv0, u1, cv1], + src_prev: [u0, v0, u1, v1], dst_prev: g.w_dst_prev, mb: [g.mb_taps, g.mb_amount, 1.0, 0.0], + cover: [g.webcam_cover, 0.04 * g.w_px[0].min(g.w_px[1]) * g.webcam_cover, 0.35, 0.0], ..Default::default() }; // Le masque est lie par `make_bind` sur tous les draws, pas seulement @@ -2753,6 +2753,13 @@ impl Compositor { if text.content.trim().is_empty() { continue; } + // The desk label shows only while the camera is covered: skip the rest of + // its section before rasterizing anything. + if crate::text_anim::is_desk_cover(text.animation.as_deref()) + && g.webcam_cover <= 0.0 + { + continue; + } let color = parse_hex(&text.color).unwrap_or([1.0, 1.0, 1.0, 1.0]); let background = parse_hex(&text.background_color).unwrap_or([0.0, 0.0, 0.0, 0.0]); @@ -2787,10 +2794,11 @@ impl Compositor { // et remis a l'echelle de la sortie, comme la taille de // police : en px absolus la meme animation sauterait deux // fois plus haut dans un rendu 4K que dans l'apercu. - let anim = crate::text_anim::text_animation_state( + let anim = crate::text_anim::annotation_text_state( text.animation.as_deref(), (g.source_t - a.start_sec as f32) * 1000.0, ((a.end_sec - a.start_sec) * 1000.0) as f32, + g.webcam_cover, ); let anim_px = rh / crate::text_anim::ANIMATION_REFERENCE_HEIGHT; let (mut ax, mut ay, mut aw, mut ah) = ( @@ -4888,6 +4896,9 @@ mod tests { clip_index: None, start_sec: 2.0, end_sec: 8.0, + rotation: 0, + mirror: None, + full_frame: false, }); // 5 s : la montee (~1 s) est finie, le retour (~1,5 s) n'a pas commence. comp.set_timeline_time(Some(5.0)); diff --git a/crates/compositor/src/compositor_macos.rs b/crates/compositor/src/compositor_macos.rs index 5440cfb51..493504be3 100644 --- a/crates/compositor/src/compositor_macos.rs +++ b/crates/compositor/src/compositor_macos.rs @@ -1673,6 +1673,13 @@ impl Compositor { if text.content.trim().is_empty() { continue; } + // The desk label shows only while the camera is covered: skip the rest of + // its section before rasterizing anything. + if crate::text_anim::is_desk_cover(text.animation.as_deref()) + && g.webcam_cover <= 0.0 + { + continue; + } let spec = crate::text::TextSpec { content: text.content.clone(), color: parse_hex(&text.color).unwrap_or([1.0, 1.0, 1.0, 1.0]), @@ -1706,10 +1713,11 @@ impl Compositor { }) else { continue; }; - let anim = crate::text_anim::text_animation_state( + let anim = crate::text_anim::annotation_text_state( text.animation.as_deref(), (t - a.start_sec as f32) * 1000.0, ((a.end_sec - a.start_sec) * 1000.0) as f32, + g.webcam_cover, ); let anim_px = rh / crate::text_anim::ANIMATION_REFERENCE_HEIGHT; let (mut ax, mut ay, mut aw, mut ah) = ( @@ -2663,10 +2671,16 @@ impl Compositor { let [cu0, cv0, cu1, cv1] = crate::frame_geometry::webcam_source_rect( [wcw, wch], [wtw as f32, wth as f32], - scene_ref.as_ref().and_then(|scene| scene.layout.webcam_crop), + if g.webcam.full_frame { + None + } else { + scene_ref.as_ref().and_then(|scene| scene.layout.webcam_crop) + }, g.w_px[0] / g.w_px[1].max(0.0001), ); - let (u0, u1) = if lp.webcam_mirror { (cu1, cu0) } else { (cu0, cu1) }; + // Mirror and the desk-shot turn are both bound swaps: u for horizontal, v for vertical. + let (u0, u1) = if g.webcam.flip_u { (cu1, cu0) } else { (cu0, cu1) }; + let (v0, v1) = if g.webcam.flip_v { (cv1, cv0) } else { (cv0, cv1) }; let webcam_is_block = matches!( g.scene_preset.as_deref(), Some("dual-frame") | Some("vertical-stack") @@ -2733,7 +2747,7 @@ impl Compositor { enc, &LayerCB { dst: g.w_dst, - src: [u0, cv0, u1, cv1], + src: [u0, v0, u1, v1], quad_px: g.w_px, radius_px: g.w_radius, mode: 0.0, @@ -2741,9 +2755,10 @@ impl Compositor { // plus lu, le fond ayant déjà été peint sous la caméra. color: [0.0, 0.0, 0.0, 1.0], fx: [w_valid[0], w_valid[1], effect_code, blur_intensity], - src_prev: [u0, cv0, u1, cv1], + src_prev: [u0, v0, u1, v1], dst_prev: g.w_dst_prev, mb: [g.mb_taps, g.mb_amount, 1.0, 0.0], + cover: [g.webcam_cover, 0.04 * g.w_px[0].min(g.w_px[1]) * g.webcam_cover, 0.35, 0.0], ..Default::default() }, wy, diff --git a/crates/compositor/src/compositor_windows.rs b/crates/compositor/src/compositor_windows.rs index 11952713d..ebf106f79 100644 --- a/crates/compositor/src/compositor_windows.rs +++ b/crates/compositor/src/compositor_windows.rs @@ -2438,13 +2438,16 @@ impl Compositor { let [su0, sv0, su1, sv1] = crate::frame_geometry::webcam_source_rect( [wcw, wch], [wtw as f32, wth as f32], - scene_ref - .as_ref() - .and_then(|scene| scene.layout.webcam_crop), + if g.webcam.full_frame { + None + } else { + scene_ref.as_ref().and_then(|scene| scene.layout.webcam_crop) + }, w_px[0] / w_px[1].max(0.0001), ); - // miroir = échanger les bornes u du rect source (flip horizontal). - let (u0, u1) = if lp.webcam_mirror { (su1, su0) } else { (su0, su1) }; + // Mirror and the desk-shot turn are both bound swaps: u for horizontal, v for vertical. + let (u0, u1) = if g.webcam.flip_u { (su1, su0) } else { (su0, su1) }; + let (v0, v1) = if g.webcam.flip_v { (sv1, sv0) } else { (sv0, sv1) }; if lp.has_webcam { // L'ombre portée appartient à la bulle flottante PiP : elle se retire avec elle // (`shape_fade`), pour qu'au plein écran plus rien n'encadre la caméra. C'est une @@ -2508,7 +2511,7 @@ impl Compositor { self.draw_video( &LayerCB { dst: w_dst, - src: [u0, sv0, u1, sv1], + src: [u0, v0, u1, v1], quad_px: w_px, radius_px: w_radius, mode: 0.0, @@ -2516,9 +2519,10 @@ impl Compositor { // plus lu, le fond ayant déjà été peint sous la caméra. color: [0.0, 0.0, 0.0, 1.0], fx: [w_valid[0], w_valid[1], effect_code, blur_intensity], - src_prev: [u0, sv0, u1, sv1], // src fixe (pas de zoom webcam) + src_prev: [u0, v0, u1, v1], // src fixe (pas de zoom webcam) dst_prev: w_dst_prev, mb: [mb_taps, mb_amount, 1.0, 0.0], + cover: [g.webcam_cover, 0.04 * w_px[0].min(w_px[1]) * g.webcam_cover, 0.35, 0.0], ..Default::default() }, &wy, @@ -2741,6 +2745,13 @@ impl Compositor { if text.content.trim().is_empty() { continue; } + // The desk label shows only while the camera is covered: skip the rest of + // its section before rasterizing anything. + if crate::text_anim::is_desk_cover(text.animation.as_deref()) + && g.webcam_cover <= 0.0 + { + continue; + } // `font_size_rel` est une fraction de la HAUTEUR DE LA BOÎTE D'ANCRAGE — rect // écran, ou cadre de sortie pour un sous-titre (cf. le contrat et // `annotationScale.ts`) : on la ramène en pixels de sortie ici, avec le même @@ -2786,10 +2797,13 @@ impl Compositor { // dispose ici) : dans une région accélérée, elle défile donc au rythme du // clip. À vitesse 1 — le cas de toutes les annotations existantes — c'est // exactement le timing de l'aperçu DOM. - let anim = crate::text_anim::text_animation_state( + // The desk label is the exception: its opacity is the camera cover, which + // runs on the screen clock (`annotation_text_state`). + let anim = crate::text_anim::annotation_text_state( text.animation.as_deref(), (t - annotation.start_sec as f32) * 1000.0, ((annotation.end_sec - annotation.start_sec) * 1000.0) as f32, + g.webcam_cover, ); // Les décalages sont donnés à la hauteur de référence : on les ramène à la // sortie, comme la taille de police, pour que l'animation ait la même diff --git a/crates/compositor/src/frame_geometry.rs b/crates/compositor/src/frame_geometry.rs index a8ef080e4..a719b9673 100644 --- a/crates/compositor/src/frame_geometry.rs +++ b/crates/compositor/src/frame_geometry.rs @@ -27,7 +27,7 @@ use crate::config::Cfg; use crate::scene::{Scene, SceneCrop}; -/// Constant buffer d'un calque : **176 octets**, un par draw. +/// Constant buffer d'un calque : **192 octets**, un par draw. /// /// C'est le contrat partagé par les trois côtés — `cbuffer Layer` dans `shaders.hlsl`, /// `struct Layer` dans `shaders.metal` et `vk_shaders/layer.wgsl`, et ce struct. Ils doivent @@ -35,13 +35,13 @@ use crate::scene::{Scene, SceneCrop}; /// produit un shader qui lit `color` là où on a écrit `fx`. /// /// `align(16)` vient de la version macOS ; sous `repr(C)` seul, les offsets sont déjà -/// 0/16/32/40/44/48/64/80/96/112/128/144/160 des deux côtés — l'alignement Rust ne change que +/// 0/16/32/40/44/48/64/80/96/112/128/144/160/176 des deux côtés — l'alignement Rust ne change que /// l'adresse du struct, pas son contenu, et Windows le `copy_nonoverlapping` dans un /// constant buffer mappé où l'alignement source est sans effet. Les deux formes étaient /// donc compatibles ; les unifier évite qu'elles cessent de l'être. /// /// (Le commentaire d'origine annonçait « 64 octets ». Il n'a jamais été juste : dix champs, -/// trente-deux `f32`. Les trois derniers, le flou de mouvement de l'écran incliné, en font 176.) +/// trente-deux `f32`. Les trois suivants, le flou de mouvement de l'écran incliné, en font 176, et `cover`, le voile de la vue bureau, 192.) #[repr(C, align(16))] #[derive(Clone, Copy, Default)] pub struct LayerCB { @@ -62,6 +62,9 @@ pub struct LayerCB { pub trail_a: [f32; 4], pub trail_b: [f32; 4], pub trail_mb: [f32; 4], + /// Desk-view cover of the webcam layer: x = strength 0..1, y = blur radius (quad px), + /// z = dim factor, w unused. Zero everywhere else. + pub cover: [f32; 4], } impl LayerCB { @@ -330,6 +333,33 @@ pub(crate) fn cover_crop_uv(visible: [f32; 2], tex: [f32; 2], box_ar: f32) -> (f (u0, v0, u1, v1) } +/// How the webcam texture is laid onto its quad this frame. The backends swap the source +/// rect's u bounds for `flip_u` and its v bounds for `flip_v` — 180° is both — and drop the +/// webcam crop for `full_frame`. Derived per frame because a Full Camera region can turn +/// the camera for a desk shot. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub struct WebcamOrientation { + pub flip_u: bool, + pub flip_v: bool, + pub full_frame: bool, +} + +pub fn webcam_orientation( + region: Option<&crate::scene::SceneCameraFullscreenRegion>, + layout_mirror: bool, +) -> WebcamOrientation { + let Some(r) = region else { + return WebcamOrientation { flip_u: layout_mirror, flip_v: false, full_frame: false }; + }; + let turned = r.rotation == 180; + let mirror = r.mirror.unwrap_or(layout_mirror); + WebcamOrientation { + flip_u: mirror ^ turned, + flip_v: turned, + full_frame: turned && r.full_frame, + } +} + /// Camera equivalent of the screen crop pipeline: apply the user crop first, then a centred /// cover-crop inside that authored window so arbitrary layout slots never stretch the image. pub(crate) fn webcam_source_rect( @@ -1908,6 +1938,10 @@ pub struct FrameGeometry { pub w_px: [f32; 2], pub w_radius: f32, pub shape_fade: f32, + /// Orientation of the webcam this frame (mirror, desk-shot turn, crop bypass). + pub webcam: WebcamOrientation, + /// Desk-view cover strength of the webcam this frame, 0..1 (`camera_fullscreen_cover_at`). + pub webcam_cover: f32, /// Cadre autour de l'écran : chrome de fenêtre plat (mode 14) ou appareil modelé (mode 17). /// `None` : aucun, et le rendu est celui d'avant le cadre, à l'octet. `Some` : `s_dst` est /// déjà la boîte rétrécie, et `s_radius` le rayon des coins de l'écran — des seuls coins BAS @@ -3061,6 +3095,11 @@ pub fn plan_frame(input: &FrameGeometryInput) -> FrameGeometry { crate::regions::camera_fullscreen_progress_at(cam_regions, source_t_prev, &clock); let shape_fade = crate::regions::camera_fullscreen_shape_at(cam_regions, source_t, &clock); + let webcam = webcam_orientation( + crate::regions::camera_fullscreen_region_at(cam_regions, source_t, &clock), + lp.webcam_mirror, + ); + let webcam_cover = crate::regions::camera_fullscreen_cover_at(cam_regions, source_t, &clock); // rétrécissement réactif : la webcam garde 70 % de sa taille pendant un zoom actif, quel // que soit son niveau (elle suivait 1/zoom, et rétrécissait donc d'autant plus que le zoom // était profond : ×0,6 au zoom maximal). L'enveloppe est celle de la région : elle descend @@ -3504,6 +3543,8 @@ pub fn plan_frame(input: &FrameGeometryInput) -> FrameGeometry { w_px, w_radius, shape_fade, + webcam, + webcam_cover, window_frame, screen_mask, } @@ -6305,6 +6346,7 @@ mod tests { .into_iter() .chain(c.quad_px) .chain([c.radius_px, c.mode]) + .chain(c.cover) .map(f32::to_bits) .collect() }) @@ -7302,7 +7344,7 @@ mod tests { #[test] fn layer_cb_matches_the_shader_constant_buffer() { use std::mem::{align_of, offset_of, size_of}; - assert_eq!(size_of::(), 176); + assert_eq!(size_of::(), 192); assert_eq!(align_of::(), 16); for (name, got, want) in [ ("dst", offset_of!(LayerCB, dst), 0), @@ -7318,11 +7360,102 @@ mod tests { ("trail_a", offset_of!(LayerCB, trail_a), 128), ("trail_b", offset_of!(LayerCB, trail_b), 144), ("trail_mb", offset_of!(LayerCB, trail_mb), 160), + ("cover", offset_of!(LayerCB, cover), 176), ] { assert_eq!(got, want, "offset de `{name}`"); } } + /// A turned Full Camera section from 1 s to 9 s: the plan carries the cover strength, full in + /// the hold after the start and zero in the steady part. + #[test] + fn the_frame_plan_carries_the_cover() { + let cfg = crate::config::all().pop().expect("cfg"); + let json = zoomed_golden_scene_json().replace( + r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#, + r#""zoomRegions":[],"cameraFullscreenRegions":[{"clipIndex":0,"startSec":1.0,"endSec":9.0,"rotation":180,"fullFrame":true}]"#, + ); + let scene = Scene::from_json(&json).expect("scene"); + let cover_at = |t: f32| { + plan_frame(&FrameGeometryInput { timeline_t_override: Some(t), ..golden_input(&scene, &cfg) }) + .webcam_cover + }; + assert_eq!(cover_at(1.5), 1.0); + assert_eq!(cover_at(5.0), 0.0); + } + + /// The desk label's opacity at the compositor caller boundary, against the cover of the same + /// frame plan: equal at every sampled frame, and the label (spanned over its whole section by + /// the app) is on screen wherever the cover is not 0. Three sections: the review's 2.6 s one + /// at 1×, where the label's old fade (from its own 1.15 s window) was half gone while the + /// cover still held 1; a long one; and a section inside a 2× speed region, whose cover runs + /// on the screen clock. + #[test] + fn the_desk_label_fades_with_the_cover() { + use crate::text_anim::{annotation_text_state, DESK_COVER_ANIMATION}; + let cfg = crate::config::all().pop().expect("cfg"); + let zoom = r#""zoomRegions":[{"clipIndex":0,"startSec":0.0,"endSec":5.0,"scale":2.0,"focusX":0.5,"focusY":0.3,"rotation":"none"}]"#; + let section = |start: f32, end: f32, speed: Option<&str>| { + let regions = format!( + r#""zoomRegions":[],"cameraFullscreenRegions":[{{"clipIndex":0,"startSec":{start},"endSec":{end},"rotation":180,"fullFrame":true}}]{}"#, + speed.map(|s| format!(",{s}")).unwrap_or_default() + ); + Scene::from_json(&zoomed_golden_scene_json().replace(zoom, ®ions)).expect("scene") + }; + // (label, cover) at source time `t` of a label spanning [start, end]. + let sample = |scene: &Scene, start: f32, end: f32, t: f32| { + let g = plan_frame(&FrameGeometryInput { + timeline_t_override: Some(t), + ..golden_input(scene, &cfg) + }); + let label = annotation_text_state( + Some(DESK_COVER_ANIMATION), + (t - start) * 1000.0, + (end - start) * 1000.0, + g.webcam_cover, + ); + (label.opacity, g.webcam_cover) + }; + let speed = r#""speedRegions":[{"clipIndex":0,"startSec":1.0,"endSec":9.0,"speed":2.0}]"#; + let cases = [ + ("2.6 s at 1x", section(1.0, 3.6, None), 1.0f32, 3.6f32), + ("long at 1x", section(1.0, 9.0, None), 1.0, 9.0), + ("2.6 s on screen inside 2x", section(2.0, 7.2, Some(speed)), 2.0, 7.2), + ]; + for (name, scene, start, end) in &cases { + let mut t = start - 0.1; + while t < end + 0.1 { + let (label, cover) = sample(scene, *start, *end, t); + assert_eq!(label, cover, "{name}: t = {t}"); + if cover > 0.0 { + assert!(t >= *start && t < *end, "{name}: cover {cover} outside the label at {t}"); + } + t += 1.0 / 120.0; + } + } + // The review's frames: the cover still holds 1 at 0.9 s and 1.0 s (the old label was at + // ~0.5 and below). The 0.28 s left between the holds is two 0.14 s fades meeting at + // +1.1575 s, where label and cover touch 0 together before rising into the end hold. + let (_, short, s0, s1) = &cases[0]; + for at in [0.5, 0.9, 1.0] { + assert_eq!(sample(short, *s0, *s1, s0 + at), (1.0, 1.0), "at +{at} s"); + } + let (mid, _) = sample(short, *s0, *s1, s0 + 1.09); + assert!(mid > 0.0 && mid < 1.0, "mid-fade {mid}"); + let (low, low_cover) = sample(short, *s0, *s1, s0 + 1.1575); + assert!(low < 1e-3 && low == low_cover, "junction {low} vs {low_cover}"); + assert_eq!(sample(short, *s0, *s1, s1 - 1.25), (1.0, 1.0), "end hold"); + // Inside 2x the same 2.6 s of screen time takes 5.2 s of source: every screen instant + // matches the 1x section's, which the source-time label could not do. + let (_, fast, f0, f1) = &cases[2]; + for k in 0..=26 { + let screen = k as f32 * 0.1; + let slow = sample(short, *s0, *s1, s0 + screen); + let quick = sample(fast, *f0, *f1, f0 + 2.0 * screen); + assert!((slow.0 - quick.0).abs() < 1e-3, "screen +{screen} s: {slow:?} vs {quick:?}"); + } + } + /// Le pivot doit rester collé à `center` quand le sprite grandit — c'est exactement ce qui /// était cassé (ancrage centré en dur : la pointe s'éloignait proportionnellement à la /// taille). On dessine la même flèche à deux tailles et on vérifie que le point désigné @@ -7730,6 +7863,8 @@ mod tests { w_px: [0.0, 0.0], w_radius: 0.0, shape_fade: 0.0, + webcam: WebcamOrientation::default(), + webcam_cover: 0.0, window_frame: None, screen_mask: None, } @@ -8794,4 +8929,110 @@ mod tests { println!("left ×2 : {:.3} côté proche, {:.3} côté lointain", near / at_focus, far / at_focus); assert!(near > 1.04 * at_focus && far < 0.96 * at_focus, "{near} {at_focus} {far}"); } + + use crate::scene::SceneCameraFullscreenRegion; + + fn cam_region(rotation: u16, mirror: Option) -> SceneCameraFullscreenRegion { + SceneCameraFullscreenRegion { + clip_index: None, + start_sec: 0.0, + end_sec: 1.0, + rotation, + mirror, + full_frame: rotation == 180, + } + } + + #[test] + fn without_a_region_the_webcam_is_laid_as_before() { + for m in [false, true] { + assert_eq!( + webcam_orientation(None, m), + WebcamOrientation { flip_u: m, flip_v: false, full_frame: false } + ); + } + } + + /// 180° is both axes swapped; a mirror on top cancels the horizontal one. + #[test] + fn a_turned_region_swaps_both_axes_and_a_mirror_cancels_one() { + let turned = cam_region(180, Some(false)); + assert_eq!( + webcam_orientation(Some(&turned), true), + WebcamOrientation { flip_u: true, flip_v: true, full_frame: true } + ); + let turned_mirrored = cam_region(180, Some(true)); + assert_eq!( + webcam_orientation(Some(&turned_mirrored), false), + WebcamOrientation { flip_u: false, flip_v: true, full_frame: true } + ); + } + + #[test] + fn a_plain_region_keeps_the_layout_mirror_and_an_unknown_rotation_is_none() { + assert!(webcam_orientation(Some(&cam_region(0, None)), true).flip_u); + let odd = cam_region(90, None); + assert_eq!( + webcam_orientation(Some(&odd), false), + WebcamOrientation { flip_u: false, flip_v: false, full_frame: false } + ); + } + + /// `layer.wgsl` is only compiled on Linux, in CI, where a name or syntax error breaks every + /// draw. Parse and validate it on every host through wgpu's own naga, and pin the desk-view + /// cover: the fragment entry point must call the radius-taking blur kernel, and the kernel + /// must keep its taps inside the picture's valid area. + /// + /// The file does not declare `LAYER_MODELS`: `compositor_linux::layer_source` prefixes it, + /// once per pipeline. Both variants are checked here the same way, the one without the 3D + /// models first, since that is the pipeline the webcam (mode 0) is drawn with. + #[test] + fn layer_wgsl_validates_and_its_fragment_applies_the_desk_view_cover() { + use wgpu::naga; + fn calls(block: &naga::Block, target: naga::Handle) -> usize { + block + .iter() + .map(|st| match st { + naga::Statement::Call { function, .. } => usize::from(*function == target), + naga::Statement::Block(b) => calls(b, target), + naga::Statement::If { accept, reject, .. } => calls(accept, target) + calls(reject, target), + naga::Statement::Loop { body, continuing, .. } => calls(body, target) + calls(continuing, target), + naga::Statement::Switch { cases, .. } => cases.iter().map(|c| calls(&c.body, target)).sum(), + _ => 0, + }) + .sum() + } + + for models in [false, true] { + // The same prefix as `compositor_linux::layer_source`, which only builds on Linux. + let source = + format!("const LAYER_MODELS: bool = {models};\n{}", include_str!("vk_shaders/layer.wgsl")); + let module = naga::front::wgsl::parse_str(&source) + .unwrap_or_else(|e| panic!("layer.wgsl (LAYER_MODELS = {models}) parses: {e:?}")); + naga::valid::Validator::new(naga::valid::ValidationFlags::all(), naga::valid::Capabilities::all()) + .validate(&module) + .unwrap_or_else(|e| panic!("layer.wgsl (LAYER_MODELS = {models}) validates: {e:?}")); + let (kernel, f) = module + .functions + .iter() + .find(|(_, f)| f.name.as_deref() == Some("blur_webcam_radius")) + .expect("blur_webcam_radius exists"); + assert_eq!(f.arguments.len(), 5, "uv, max_r_px, qpx, local_px, valid"); + // The taps are clamped to the valid part of an aligned decoder texture, half a texel + // in: the kernel reads the texture size to know what half a texel is. + assert!( + f.expressions.iter().any(|(_, e)| matches!( + e, + naga::Expression::ImageQuery { query: naga::ImageQuery::Size { .. }, .. } + )), + "blur_webcam_radius reads the texture size for its half-texel clamp" + ); + let fs = module.entry_points.iter().find(|e| e.name == "fs_main").expect("fs_main"); + assert_eq!( + calls(&fs.function.body, kernel), + 1, + "fs_main blurs the covered camera once (LAYER_MODELS = {models})" + ); + } + } } diff --git a/crates/compositor/src/regions.rs b/crates/compositor/src/regions.rs index d14c37773..c00cc4d83 100644 --- a/crates/compositor/src/regions.rs +++ b/crates/compositor/src/regions.rs @@ -1042,6 +1042,65 @@ pub fn camera_fullscreen_shape_at( 1.0 - smoothstep(0.5, 1.0, camera_fullscreen_phase_at(regions, t, clock)) } +/// The Full Camera region in effect at `t`: the one with the strongest phase, so a region's +/// orientation and its grow/shrink always refer to the same region. `None` wherever the +/// envelope is 0 — outside every region and on their bounds. +pub fn camera_fullscreen_region_at<'a>( + regions: &'a [SceneCameraFullscreenRegion], + t: f32, + clock: &ScreenClock, +) -> Option<&'a SceneCameraFullscreenRegion> { + let mut best: Option<(&'a SceneCameraFullscreenRegion, f32)> = None; + for r in regions { + let phase = camera_fullscreen_region_phase(r, t, clock); + if phase > 0.0 && best.map_or(true, |(_, b)| phase > b) { + best = Some((r, phase)); + } + } + best.map(|(r, _)| r) +} + +/// How long a turned section's camera takes to sharpen after the hold, and to blur again +/// before the shrink. See `camera_fullscreen_cover_at`. +pub const DESK_COVER_FADE_S: f32 = 0.5; + +/// Cover strength of the webcam at `t` (0 = sharp, 1 = fully blurred and dimmed). Only a +/// turned section (a camera tilted onto the desk) is covered: the camera is moving at both +/// ends of such a section, so the picture is hidden for exactly the grow and the shrink, plus a +/// short fade into and out of the steady part. Measured on screen time like the grow, and 0 +/// on and outside the bounds like the grow. +pub fn camera_fullscreen_cover_at( + regions: &[SceneCameraFullscreenRegion], + t: f32, + clock: &ScreenClock, +) -> f32 { + let Some(r) = camera_fullscreen_region_at(regions, t, clock) else { + return 0.0; + }; + if r.rotation != 180 { + return 0.0; + } + let (start, end, t) = (clock.at(r.start_sec as f32), clock.at(r.end_sec as f32), clock.at(t)); + // Each hold gets at most half the section; the two fades share what is left, so the start + // fade ends at or before the point where the end fade begins and the cover never jumps. + let len = end - start; + let half = len * 0.5; + let hold_in = TRANSITION_WINDOW_S.min(half); + let hold_out = FULLSCREEN_LEAD_OUT_WINDOW_S.min(half); + let steady = (len - hold_in - hold_out).max(0.0); + let fade = DESK_COVER_FADE_S.min(steady * 0.5); + let side = |since: f32, hold: f32, fade: f32| -> f32 { + if since <= hold { + 1.0 + } else if fade > 0.0 && since < hold + fade { + 1.0 - smoothstep(0.0, fade, since - hold) + } else { + 0.0 + } + }; + side(t - start, hold_in, fade).max(side(end - t, hold_out, fade)) +} + // ============ Rotation 3D (tilt perspective, présets iso/left/right) ================ // Port de `computeRotation3DContainScale` (TS, `types.ts`) — même formule, même ordre de // composition ("CSS rotateX rotateY rotateZ s'applique droite-à-gauche : Z d'abord, puis Y, @@ -1977,6 +2036,9 @@ mod zoom_focus_tests { clip_index: None, start_sec: 18.0 * k, end_sec: 22.0 * k, + rotation: 0, + mirror: None, + full_frame: false, }] }; let (sped, plain) = (zooms(4.0), zooms(1.0)); @@ -2003,11 +2065,160 @@ mod zoom_focus_tests { } } + fn cam(start: f64, end: f64, rotation: u16) -> SceneCameraFullscreenRegion { + SceneCameraFullscreenRegion { + clip_index: None, + start_sec: start, + end_sec: end, + rotation, + mirror: None, + full_frame: rotation != 0, + } + } + + /// Orientation and transition must name the same region: inside it the region, at its + /// bounds and outside nothing, exactly where the progress envelope is 0. + #[test] + fn the_full_camera_region_at_t_is_the_one_the_envelope_uses() { + let r = [cam(10.0, 20.0, 180), cam(30.0, 40.0, 0)]; + let clock = ScreenClock::default(); + assert_eq!(camera_fullscreen_region_at(&r, 15.0, &clock).map(|c| c.rotation), Some(180)); + assert_eq!(camera_fullscreen_region_at(&r, 35.0, &clock).map(|c| c.rotation), Some(0)); + for t in [5.0, 10.0, 20.0, 25.0] { + assert!(camera_fullscreen_region_at(&r, t, &clock).is_none(), "t={t}"); + assert_eq!(camera_fullscreen_progress_at(&r, t, &clock), 0.0, "t={t}"); + } + } + + + /// The hold covers exactly the grow; then the picture sharpens over the fade. + #[test] + fn a_turned_section_is_covered_while_the_camera_moves() { + let r = [cam(10.0, 30.0, 180)]; + let clock = ScreenClock::default(); + let c = |t: f32| camera_fullscreen_cover_at(&r, t, &clock); + assert_eq!(c(10.0 + 0.01), 1.0, "start of the hold"); + assert_eq!(c(10.0 + TRANSITION_WINDOW_S - 0.01), 1.0, "end of the hold"); + let mid_fade = c(10.0 + TRANSITION_WINDOW_S + DESK_COVER_FADE_S / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-3, "half way through the fade: {mid_fade}"); + assert_eq!(c(20.0), 0.0, "steady part is sharp"); + // Mirrored at the end: fade up, then hold for the whole shrink. + let mid_rise = c(30.0 - FULLSCREEN_LEAD_OUT_WINDOW_S - DESK_COVER_FADE_S / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-3, "half way up: {mid_rise}"); + assert_eq!(c(30.0 - FULLSCREEN_LEAD_OUT_WINDOW_S + 0.01), 1.0); + assert_eq!(c(30.0 - 0.01), 1.0); + } + + #[test] + fn the_cover_is_zero_on_and_outside_the_bounds() { + let r = [cam(10.0, 30.0, 180)]; + let clock = ScreenClock::default(); + for t in [0.0, 9.99, 10.0, 30.0, 30.01, 40.0] { + assert_eq!(camera_fullscreen_cover_at(&r, t, &clock), 0.0, "t={t}"); + } + } + + #[test] + fn a_plain_section_is_never_covered() { + let r = [cam(10.0, 30.0, 0)]; + let clock = ScreenClock::default(); + for step in 0..=400 { + let t = step as f32 * 0.1; + assert_eq!(camera_fullscreen_cover_at(&r, t, &clock), 0.0, "t={t}"); + } + } + + /// Each hold gets at most half the section and the fades share what is left, so a section + /// shorter than both holds is covered throughout and a longer one keeps its full fades. + #[test] + fn short_sections_shrink_the_fades_first() { + let clock = ScreenClock::default(); + let short = [cam(10.0, 11.5, 180)]; // half = 0.75 < hold-in: no fade at all + for step in 1..150 { + let t = 10.0 + step as f32 * 0.01; + let c = camera_fullscreen_cover_at(&short, t, &clock); + assert!((0.0..=1.0).contains(&c), "t={t}: {c}"); + assert_eq!(c, 1.0, "a section shorter than both holds is covered throughout (t={t})"); + } + // 3.6 s: steady = 3.6 - 1.015 - 1.523 = 1.06, so both ends keep their full 0.5 s fade. + let medium = [cam(10.0, 13.6, 180)]; + let c = |t: f32| camera_fullscreen_cover_at(&medium, t, &clock); + let mid_fade = c(10.0 + TRANSITION_WINDOW_S + DESK_COVER_FADE_S / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-3, "start fade half way: {mid_fade}"); + let mid_rise = c(13.6 - FULLSCREEN_LEAD_OUT_WINDOW_S - DESK_COVER_FADE_S / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-3, "end fade half way: {mid_rise}"); + } + + /// 2.6 s: hold-in 1.015, hold-out 1.3 (half), steady 0.285 — the two fades share it, + /// 0.1425 s each, so the start fade ends exactly where the end fade begins. + #[test] + fn medium_sections_share_the_steady_part_between_the_fades() { + let clock = ScreenClock::default(); + let r = [cam(10.0, 12.6, 180)]; + let c = |t: f32| camera_fullscreen_cover_at(&r, t, &clock); + let fade = (2.6 - TRANSITION_WINDOW_S - 1.3) / 2.0; + let meet = 10.0 + TRANSITION_WINDOW_S + fade; + let mid_fade = c(10.0 + TRANSITION_WINDOW_S + fade / 2.0); + assert!((mid_fade - 0.5).abs() < 1e-2, "start fade half way: {mid_fade}"); + assert!(c(meet) < 1e-2, "both fades are near zero where they meet: {}", c(meet)); + let mid_rise = c(12.6 - 1.3 - fade / 2.0); + assert!((mid_rise - 0.5).abs() < 1e-2, "end fade half way: {mid_rise}"); + assert_eq!(c(12.6 - 1.3 + 0.01), 1.0, "the end hold is half the section"); + } + + /// No section length makes the cover jump: sampled every 1 ms, no step exceeds 0.1. (At + /// 10 ms the shortest legitimate fade — 42 ms in a 2.2 s section — already steps ~0.3 per + /// sample, so the finer grid is what tells a ramp from a jump.) + #[test] + fn the_cover_is_continuous_for_every_section_length() { + let clock = ScreenClock::default(); + for len in [2.0_f32, 2.2, 2.6, 3.0, 3.5, 4.0, 4.5] { + let r = [cam(10.0, 10.0 + len as f64, 180)]; + let steps = (len * 1000.0).round() as i32; + let mut prev = camera_fullscreen_cover_at(&r, 10.0 + 0.0005, &clock); + for step in 1..steps { + let t = 10.0 + step as f32 * 0.001; + let c = camera_fullscreen_cover_at(&r, t, &clock); + assert!((0.0..=1.0).contains(&c), "len={len} t={t}: {c}"); + assert!((c - prev).abs() <= 0.1, "len={len} t={t}: {prev} -> {c}"); + prev = c; + } + } + } + + /// Under a speed change the windows are measured on screen time, like the grow. + #[test] + fn the_cover_runs_on_screen_time() { + let clock = ScreenClock::new(&[speed(None, 0.0, 1000.0, 2.0)], 0); + let r = [cam(10.0, 40.0, 180)]; + // At 2x, the 1.015 s hold spans 2.03 s of source time. + assert_eq!(camera_fullscreen_cover_at(&r, 10.0 + 1.9, &clock), 1.0); + assert!(camera_fullscreen_cover_at(&r, 10.0 + 3.6, &clock) < 1.0); + } + #[test] + fn the_region_fields_parse_and_default() { + let plain: SceneCameraFullscreenRegion = + serde_json::from_str(r#"{"startSec":1,"endSec":2}"#).unwrap(); + assert_eq!((plain.rotation, plain.mirror, plain.full_frame), (0, None, false)); + let desk: SceneCameraFullscreenRegion = serde_json::from_str( + r#"{"startSec":1,"endSec":2,"rotation":180,"mirror":false,"fullFrame":true}"#, + ) + .unwrap(); + assert_eq!((desk.rotation, desk.mirror, desk.full_frame), (180, Some(false), true)); + } + /// Les coins suivent la phase, pas le rect. Avec `1 - progrès`, ils étaient carrés presque /// tout du long de la montée et ne revenaient qu'à la toute fin du retour. #[test] fn full_camera_corners_dissolve_late_and_come_back_early() { - let r = [SceneCameraFullscreenRegion { clip_index: None, start_sec: 10.0, end_sec: 20.0 }]; + let r = [SceneCameraFullscreenRegion { + clip_index: None, + start_sec: 10.0, + end_sec: 20.0, + rotation: 0, + mirror: None, + full_frame: false, + }]; let clock = ScreenClock::default(); let at = |t: f32| { (camera_fullscreen_progress_at(&r, t, &clock), camera_fullscreen_shape_at(&r, t, &clock)) diff --git a/crates/compositor/src/scene.rs b/crates/compositor/src/scene.rs index bddd7d55c..a7ca6c13a 100644 --- a/crates/compositor/src/scene.rs +++ b/crates/compositor/src/scene.rs @@ -528,7 +528,8 @@ pub struct SceneSpeedRegion { /// Une zone "Full Camera" de la timeline (temps en secondes) : la caméra PREND tout le cadre /// pendant cette fenêtre (plein écran net — ni marge, ni arrondi, ni masque, ni fond derrière). -/// Pas de champs au-delà des bornes temporelles (miroir de `CameraFullscreenRegion`, TS). +/// The orientation fields are resolved by the app (`sceneDescription.ts`): a desk shot +/// arrives as `rotation: 180`, `mirror: false`, `fullFrame: true`. #[derive(Debug, Clone, Copy, Deserialize)] #[serde(rename_all = "camelCase")] pub struct SceneCameraFullscreenRegion { @@ -537,6 +538,15 @@ pub struct SceneCameraFullscreenRegion { pub clip_index: Option, pub start_sec: f64, pub end_sec: f64, + /// 0 or 180. Anything else is treated as 0 by `frame_geometry::webcam_orientation`. + #[serde(default)] + pub rotation: u16, + /// Mirror inside this region; `None` = the layout's `webcam_mirror`. + #[serde(default)] + pub mirror: Option, + /// Ignore `layout.webcam_crop` inside this region. + #[serde(default)] + pub full_frame: bool, } /// Rendu du curseur. diff --git a/crates/compositor/src/shaders.hlsl b/crates/compositor/src/shaders.hlsl index c17d207d9..8e1b1bc2a 100644 --- a/crates/compositor/src/shaders.hlsl +++ b/crates/compositor/src/shaders.hlsl @@ -23,6 +23,7 @@ cbuffer Layer : register(b0) float4 trail_a; // mode 8 : coins TL, TR du plan à la frame précédente (px locaux, comme fx) ; mode 18 incliné : en fractions de sortie float4 trail_b; // mode 8 : coins BR, BL du plan à la frame précédente (comme src_prev) ; mode 18 incliné : en fractions de sortie float4 trail_mb; // mode 8 : x = taps, y = force du flou de mouvement (ceux du mode 0) ; mode 18 incliné : le `mb` du mode 8 (profondeur de champ), et `color.xy` sa lampe ; 0 ailleurs + float4 cover; // x = desk-view cover 0..1, y = blur radius (quad px), z = dim, w unused }; // Mode 15 (curseur modélisé) : le détail des emplacements est dans `frame_geometry.rs`, en tête // de la section « Curseur modélisé » (`cursor_model_cb`). Mode 17 (appareil modelé) : en tête de @@ -502,10 +503,15 @@ static const float3 VOGEL_TAPS[21] = { float3(-0.633036, -0.758588, 0.087119) }; -float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px) +// Vogel-disc blur of the camera texture at a radius in quad pixels. `valid` is the part of the +// texture the picture fills (fx.xy): decoders allocate aligned textures (a 1080-line camera in a +// 1088-line texture), so each tap is clamped half a chroma texel inside it, never into padding. +float3 blur_webcam_radius(float2 uv, float max_r_px, float2 qpx, float2 local_px, float2 valid) { - float max_r_px = max(intensity, 0.0) * 22.0 + 1.5; float2 step = max_r_px / max(qpx, 1.0); + float cw, ch; + texUV.GetDimensions(cw, ch); + float2 hi = max(valid - 0.5 / max(float2(cw, ch), 1.0), 0.0); // Interleaved Gradient Noise pour rotation aléatoire par pixel float noise = frac(52.9829189 * frac(0.06711056 * local_px.x + 0.00583715 * local_px.y)); float angle = noise * 6.2831853; @@ -518,12 +524,17 @@ float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px) float2 p = VOGEL_TAPS[k].xy; float w = VOGEL_TAPS[k].z; float2 rot_p = float2(p.x * c - p.y * s, p.x * s + p.y * c); - sum += sample_yuv(saturate(uv + rot_p * step)) * w; + sum += sample_yuv(clamp(uv + rot_p * step, 0.0, hi)) * w; total += w; } return sum / max(total, 1e-4); } +float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px, float2 valid) +{ + return blur_webcam_radius(uv, max(intensity, 0.0) * 22.0 + 1.5, qpx, local_px, valid); +} + // ============ Curseur MODÉLISÉ (mode 15) ============ // L'état courant en objet 3D, lancé de rayons par pixel. Deux sortes d'objets : // - un curseur SCULPTÉ (`trail_a.x` > 0) : la flèche ou la main d'un des cinq thèmes d'origine, @@ -4028,13 +4039,21 @@ float4 ps_main(VSOut i) : SV_Target } else if (effect > 1.5) { - rgb = lerp(blur_webcam_bg(uv_now, fx.w, quad_px, i.local), rgb, person); + rgb = lerp(blur_webcam_bg(uv_now, fx.w, quad_px, i.local, fx.xy), rgb, person); } else { alpha_mask = person; } } + + // Desk-view cover: the camera is being tilted, so the whole picture is blurred and + // dimmed (cover.x = strength, cover.y = radius in quad px, cover.z = dim at full cover). + if (cover.x > 0.001) + { + float3 hidden = blur_webcam_radius(uv_now, cover.y, quad_px, i.local, fx.xy); + rgb = lerp(rgb, hidden, cover.x) * (1.0 - cover.z * cover.x); + } } else { diff --git a/crates/compositor/src/shaders.metal b/crates/compositor/src/shaders.metal index 7ca87d3f2..baf787b1c 100644 --- a/crates/compositor/src/shaders.metal +++ b/crates/compositor/src/shaders.metal @@ -48,7 +48,7 @@ using namespace metal; // ================================================================================= // // Le moteur côté CPU upload ce buffer via `setVertexBytes` (vertex stage) et -// `setFragmentBytes` (fragment stage) avant chaque draw — la copie est de 176 octets, +// `setFragmentBytes` (fragment stage) avant chaque draw — la copie est de 192 octets, // ce qui est sous le seuil d'alignement 4K de Metal pour le mode « immediate ». struct Layer @@ -66,6 +66,7 @@ struct Layer float4 trail_a; // mode 8 : coins TL, TR du plan à la frame précédente (px locaux, comme fx) ; mode 18 incliné : en fractions de sortie float4 trail_b; // mode 8 : coins BR, BL du plan à la frame précédente (comme src_prev) ; mode 18 incliné : en fractions de sortie float4 trail_mb; // mode 8 : x = taps, y = force du flou de mouvement (ceux du mode 0) ; 0 ailleurs + float4 cover; // x = desk-view cover 0..1, y = blur radius (quad px), z = dim, w unused }; // Mode 15 (curseur modélisé) : le détail des emplacements est dans `frame_geometry.rs`, en tête // de la section « Curseur modélisé » (`cursor_model_cb`). Mode 17 (appareil modelé) : en tête de @@ -424,12 +425,17 @@ constant float3 VOGEL_TAPS[21] = { float3(-0.633036, -0.758588, 0.087119) }; -inline float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px, - texture2d texY, - texture2d texUV) +// Vogel-disc blur of the camera texture at a radius in quad pixels. `valid` is the part of the +// texture the picture fills (fx.xy): decoders allocate aligned textures (a 1080-line camera in a +// 1088-line texture), so each tap is clamped half a chroma texel inside it, never into padding. +inline float3 blur_webcam_radius(float2 uv, float max_r_px, float2 qpx, float2 local_px, + float2 valid, + texture2d texY, + texture2d texUV) { - float max_r_px = max(intensity, 0.0) * 22.0 + 1.5; float2 step = max_r_px / max(qpx, float2(1.0)); + float2 chroma = max(float2(float(texUV.get_width()), float(texUV.get_height())), float2(1.0)); + float2 hi = max(valid - 0.5 / chroma, float2(0.0)); float noise = fract(52.9829189 * fract(0.06711056 * local_px.x + 0.00583715 * local_px.y)); float angle = noise * 6.2831853; float s = sin(angle); @@ -441,12 +447,21 @@ inline float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 loca float2 p = VOGEL_TAPS[k].xy; float w = VOGEL_TAPS[k].z; float2 rot_p = float2(p.x * c - p.y * s, p.x * s + p.y * c); - sum += sample_yuv(saturate(uv + rot_p * step), texY, texUV) * w; + sum += sample_yuv(clamp(uv + rot_p * step, float2(0.0), hi), texY, texUV) * w; total += w; } return sum / max(total, 1e-4); } +inline float3 blur_webcam_bg(float2 uv, float intensity, float2 qpx, float2 local_px, + float2 valid, + texture2d texY, + texture2d texUV) +{ + return blur_webcam_radius(uv, max(intensity, 0.0) * 22.0 + 1.5, qpx, local_px, valid, texY, + texUV); +} + // Hash 2D -> [0,1) sans sin(). Miroir de `hash12` côté HLSL. inline float hash12(float2 p) { @@ -3772,13 +3787,22 @@ fragment float4 ps_main(VSOut i [[stage_in]], } else if (effect > 1.5) { - rgb = mix(blur_webcam_bg(uv_now, layer.fx.w, layer.quad_px, i.local, texY, texUV), rgb, person); + rgb = mix(blur_webcam_bg(uv_now, layer.fx.w, layer.quad_px, i.local, layer.fx.xy, texY, texUV), rgb, person); } else { alpha_mask = person; } } + + // Desk-view cover: the camera is being tilted, so the whole picture is blurred and + // dimmed (cover.x = strength, cover.y = radius in quad px, cover.z = dim at full cover). + if (layer.cover.x > 0.001) + { + float3 hidden = blur_webcam_radius(uv_now, layer.cover.y, layer.quad_px, i.local, layer.fx.xy, + texY, texUV); + rgb = mix(rgb, hidden, layer.cover.x) * (1.0 - layer.cover.z * layer.cover.x); + } } else { diff --git a/crates/compositor/src/text_anim.rs b/crates/compositor/src/text_anim.rs index d6d3bade1..ae3887f98 100644 --- a/crates/compositor/src/text_anim.rs +++ b/crates/compositor/src/text_anim.rs @@ -4,7 +4,8 @@ //! amplitudes. Les sept animations étaient déjà nommées dans le schéma, traduites dans les treize //! langues et transportées jusqu'ici par la scène — mais rien ne les jouait. Reprendre les //! constantes du TS plutôt que d'en réinventer garantit qu'un projet fait à l'époque de l'aperçu -//! DOM s'anime toujours pareil. +//! DOM s'anime toujours pareil. Exception : l'étiquette de la vue bureau (`deskCover`), que seul +//! le compositeur joue et qui n'a pas de courbe propre (cf. `annotation_text_state`). /// Les décalages ci-dessous sont exprimés en px À CETTE HAUTEUR : l'appelant les met à l'échelle /// de la sortie, exactement comme la taille de police (cf. `annotationScale.ts`). En pixels @@ -17,6 +18,10 @@ pub const TEXT_ANIMATION_DURATION_MS: f32 = 700.0; /// fin de l'annotation est une coupe sèche alors que son arrivée est animée. pub const TEXT_EXIT_DURATION_MS: f32 = 300.0; +/// The desk-view label's animation. The app generates it for a turned Full Camera section and +/// never stores it in a document, so it exists only here, not in `annotationTextAnimation.ts`. +pub const DESK_COVER_ANIMATION: &str = "deskCover"; + #[derive(Debug, Clone, Copy, PartialEq)] pub struct TextAnimationState { pub opacity: f32, @@ -37,6 +42,11 @@ fn clamp01(v: f32) -> f32 { v.clamp(0.0, 1.0) } +fn smoothstep01(v: f32) -> f32 { + let t = clamp01(v); + t * t * (3.0 - 2.0 * t) +} + fn ease_out_cubic(v: f32) -> f32 { let t = clamp01(v); 1.0 - (1.0 - t).powi(3) @@ -106,6 +116,30 @@ pub fn text_animation_state( } } +/// `true` for the desk-view label, whose opacity is the camera cover (`annotation_text_state`). +pub fn is_desk_cover(animation: Option<&str>) -> bool { + animation == Some(DESK_COVER_ANIMATION) +} + +/// What the compositors draw a text annotation with. Ordinary animations run on +/// `text_animation_state` (source time, unchanged). The desk-view label has no timing of its +/// own: its opacity IS `cover`, the webcam's cover strength this frame +/// (`camera_fullscreen_cover_at`, already on the screen clock). A second fade derived from the +/// label's own window drifted from the cover on short sections and inside speed regions; one +/// value read twice cannot. The app spans the label over its whole section, so the cover alone +/// decides when it shows. +pub fn annotation_text_state( + animation: Option<&str>, + elapsed_ms: f32, + region_ms: f32, + cover: f32, +) -> TextAnimationState { + if is_desk_cover(animation) { + return TextAnimationState { opacity: clamp01(cover), ..TextAnimationState::IDLE }; + } + text_animation_state(animation, elapsed_ms, region_ms) +} + #[cfg(test)] mod tests { use super::*; @@ -245,4 +279,28 @@ mod tests { fn an_empty_region_does_not_divide_by_zero() { assert!(text_animation_state(Some("fade"), 0.0, 0.0).opacity.is_finite()); } + + #[test] + fn the_desk_label_takes_the_cover_and_nothing_else() { + // Whatever the label's own window says, only the cover counts; the motion fields stay idle. + for cover in [0.0, 0.21, 0.45, 1.0] { + for (elapsed, region) in [(0.0, 2600.0), (900.0, 1150.0), (5000.0, LONG)] { + let s = annotation_text_state(Some(DESK_COVER_ANIMATION), elapsed, region, cover); + assert_eq!(s, TextAnimationState { opacity: cover, ..TextAnimationState::IDLE }); + } + } + } + + #[test] + fn ordinary_animations_ignore_the_cover() { + for name in [None, Some("fade"), Some("rise"), Some("pop"), Some("typewriter"), Some("pulse")] { + for at in [0.0, 200.0, 650.0, 9_900.0] { + assert_eq!( + annotation_text_state(name, at, LONG, 0.37), + text_animation_state(name, at, LONG), + "{name:?} at {at}" + ); + } + } + } } diff --git a/crates/compositor/src/vk_shaders/layer.wgsl b/crates/compositor/src/vk_shaders/layer.wgsl index ecf4960a1..1916cef1f 100644 --- a/crates/compositor/src/vk_shaders/layer.wgsl +++ b/crates/compositor/src/vk_shaders/layer.wgsl @@ -40,6 +40,7 @@ struct Layer { trail_a: vec4, // mode 8 : coins TL, TR du plan a la frame precedente (px locaux, comme fx) ; mode 18 incline : en fractions de sortie trail_b: vec4, // mode 8 : coins BR, BL du plan a la frame precedente (comme src_prev) ; mode 18 incline : en fractions de sortie trail_mb: vec4, // mode 8 : x = taps, y = force du flou de mouvement (ceux du mode 0) ; mode 18 incline : le `mb` du mode 8 (profondeur de champ), et `color.xy` sa lampe ; 0 ailleurs + cover: vec4, // x = desk-view cover 0..1, y = blur radius (quad px), z = dim, w unused } @group(0) @binding(0) var layer: Layer; @@ -560,9 +561,13 @@ const VOGEL_TAPS = array, 21>( vec3(-0.633036, -0.758588, 0.087119) ); -fn blur_webcam_bg(uv: vec2, intensity: f32, qpx: vec2, local_px: vec2) -> vec3 { - let max_r_px = max(intensity, 0.0) * 22.0 + 1.5; +// Vogel-disc blur of the camera texture at a radius in quad pixels. `valid` is the part of the +// texture the picture fills (fx.xy): decoders allocate aligned textures (a 1080-line camera in a +// 1088-line texture), so each tap is clamped half a chroma texel inside it, never into padding. +fn blur_webcam_radius(uv: vec2, max_r_px: f32, qpx: vec2, local_px: vec2, valid: vec2) -> vec3 { let step = max_r_px / max(qpx, vec2(1.0)); + let chroma = max(vec2(textureDimensions(texU)), vec2(1.0)); + let hi = max(valid - 0.5 / chroma, vec2(0.0)); let noise = fract(52.9829189 * fract(0.06711056 * local_px.x + 0.00583715 * local_px.y)); let angle = noise * 6.2831853; let s = sin(angle); @@ -573,12 +578,16 @@ fn blur_webcam_bg(uv: vec2, intensity: f32, qpx: vec2, local_px: vec2< let p = VOGEL_TAPS[k].xy; let w = VOGEL_TAPS[k].z; let rot_p = vec2(p.x * c - p.y * s, p.x * s + p.y * c); - sum = sum + sample_yuv(clamp(uv + rot_p * step, vec2(0.0), vec2(1.0))) * w; + sum = sum + sample_yuv(clamp(uv + rot_p * step, vec2(0.0), hi)) * w; total = total + w; } return sum / max(total, 1e-4); } +fn blur_webcam_bg(uv: vec2, intensity: f32, qpx: vec2, local_px: vec2, valid: vec2) -> vec3 { + return blur_webcam_radius(uv, max(intensity, 0.0) * 22.0 + 1.5, qpx, local_px, valid); +} + // ---- Curseur MODELISE (mode 15) ---- // Port ligne pour ligne de `cursor_model` (HLSL), dont les commentaires font foi : un curseur // SCULPTE (`trail_a.x` > 0, la fleche ou la main d'un des cinq themes d'origine modelee en @@ -2308,11 +2317,18 @@ fn fs_main(i: VsOut) -> @location(0) vec4 { if effect > 2.5 { rgb = mix(layer.color.rgb, rgb, person); } else if effect > 1.5 { - rgb = mix(blur_webcam_bg(i.uv, layer.fx.w, layer.quad_px, i.local), rgb, person); + rgb = mix(blur_webcam_bg(i.uv, layer.fx.w, layer.quad_px, i.local, layer.fx.xy), rgb, person); } else { alpha_mask = person; } } + + // Desk-view cover: the camera is being tilted, so the whole picture is blurred and + // dimmed (cover.x = strength, cover.y = radius in quad px, cover.z = dim at full cover). + if layer.cover.x > 0.001 { + let hidden = blur_webcam_radius(i.uv, layer.cover.y, layer.quad_px, i.local, layer.fx.xy); + rgb = mix(rgb, hidden, layer.cover.x) * (1.0 - layer.cover.z * layer.cover.x); + } } else if layer.mode < 1.5 { // Mode 1 — couleur pleine. rgb = layer.color.rgb; diff --git a/src/components/ai-edition/NativeCompositorOverlay.test.tsx b/src/components/ai-edition/NativeCompositorOverlay.test.tsx index ff3522665..7c4c0f397 100644 --- a/src/components/ai-edition/NativeCompositorOverlay.test.tsx +++ b/src/components/ai-edition/NativeCompositorOverlay.test.tsx @@ -18,20 +18,23 @@ import { publishNativePosition } from "@/native/nativeSync"; const native = vi.hoisted(() => ({ setActiveClip: vi.fn(async () => ({ ok: true })), setNativePlaying: vi.fn(), + setNativeScene: vi.fn(), })); +const i18n = vi.hoisted(() => ({ locale: "en" })); vi.mock("@/native", () => ({ pushAllNativeParams: vi.fn(), setActiveClip: native.setActiveClip, setCurrentNativeViewId: vi.fn(), setNativePlaying: native.setNativePlaying, - setNativeScene: vi.fn(), + setNativeScene: native.setNativeScene, subscribeNativeCompositor: () => () => undefined, useIsCpuCompositor: () => false, useNativeCompositorView: () => ({ viewId: 7, error: null }), })); vi.mock("@/contexts/I18nContext", () => ({ + useI18n: () => ({ locale: i18n.locale }), useScopedT: () => (key: string) => key, })); @@ -249,3 +252,43 @@ describe("NativeCompositorOverlay while playing", () => { expect(native.setActiveClip).toHaveBeenCalledTimes(1); }); }); + +// The desk-view label is generated text inside the scene, so the scene is rebuilt when the +// user switches language — otherwise the preview keeps the old label until the next edit. +describe("NativeCompositorOverlay on a language change", () => { + beforeEach(() => { + vi.clearAllMocks(); + i18n.locale = "en"; + useProjectStore.setState({ + projectId: "proj_sync", + document: makeDocument(), + revision: 1, + status: "ready", + error: null, + sourceDurationSec: 12, + currentTimeSec: 1, + playing: false, + dirty: false, + lastSavedAt: new Date(), + }); + }); + + afterEach(() => { + cleanup(); + i18n.locale = "en"; + useProjectStore.getState().clear(); + }); + + it("pushes the scene again when the locale changes", async () => { + const { rerender } = render(); + await act(async () => { + await Promise.resolve(); + }); + native.setNativeScene.mockClear(); + + i18n.locale = "de"; + rerender(); + + expect(native.setNativeScene).toHaveBeenCalledTimes(1); + }); +}); diff --git a/src/components/ai-edition/NativeCompositorOverlay.tsx b/src/components/ai-edition/NativeCompositorOverlay.tsx index d60bbcd20..f3a90605b 100644 --- a/src/components/ai-edition/NativeCompositorOverlay.tsx +++ b/src/components/ai-edition/NativeCompositorOverlay.tsx @@ -1,5 +1,5 @@ import { useEffect, useMemo, useRef, useSyncExternalStore } from "react"; -import { useScopedT } from "@/contexts/I18nContext"; +import { useI18n, useScopedT } from "@/contexts/I18nContext"; import { readSpeedRegions } from "@/lib/ai-edition/document/timeline"; import { noteUiProbeClipSwitch } from "@/lib/ai-edition/perf/uiFrameProbe"; import { getEditorSettings } from "@/lib/ai-edition/store/editorSettings"; @@ -129,6 +129,8 @@ export function NativeCompositorOverlay() { sources: sources ?? undefined, }); const t = useScopedT("editor"); + // The scene carries translated text (the desk-view label): a language switch rebuilds it. + const { locale } = useI18n(); // No usable GPU: the preview still renders every effect, just slowly (~8 fps with // everything on). Nothing is disabled — the output stays identical to the GPU path — // so this is a notice, not a degradation warning. @@ -147,7 +149,7 @@ export function NativeCompositorOverlay() { // `_webcamSizeRevision` ci-dessus) : le layout preset et cie pilotent le rendu (remplace le // layout fixture). Effet APRÈS celui du viewId ci-dessus → currentViewId est déjà publié // quand on pousse. - // biome-ignore lint/correctness/useExhaustiveDependencies: size revision + // biome-ignore lint/correctness/useExhaustiveDependencies: size revision, locale useEffect(() => { if (viewId === null || !document) { return; @@ -163,8 +165,9 @@ export function NativeCompositorOverlay() { // re-trigger this effect when the probed-size cache mutates; the actual value is // re-read fresh via getWebcamNativeSize() above on every run (biome flags this as // an "unnecessary" dependency, but removing it would mean a probed webcam size - // arriving after mount never gets pushed to native). - }, [viewId, document, sources, _webcamSizeRevision]); + // arriving after mount never gets pushed to native). `locale` is the same kind of + // trigger: the label text is read inside `buildSceneDescription`. + }, [viewId, document, sources, _webcamSizeRevision, locale]); // SYNCHRO COMPLETE DES PARAMS, en un seul endroit. // diff --git a/src/components/ai-edition/v4/EditorShellV4.module.css b/src/components/ai-edition/v4/EditorShellV4.module.css index b4a1449cc..d4bc74c96 100644 --- a/src/components/ai-edition/v4/EditorShellV4.module.css +++ b/src/components/ai-edition/v4/EditorShellV4.module.css @@ -599,6 +599,13 @@ flex-direction: column; min-height: 0; } +/* A pane button that toggles (Desk view) lights up like the timeline's toggle tools, off + aria-pressed. The hover rule keeps it lit: the secondary button's own hover would win. */ +.paneToggle[aria-pressed="true"], +.paneToggle[aria-pressed="true"]:hover:not(:disabled) { + background: var(--accent-soft); + color: var(--accent); +} .facetRail { pointer-events: auto; display: flex; diff --git a/src/components/ai-edition/v4/FloatingInspector.test.tsx b/src/components/ai-edition/v4/FloatingInspector.test.tsx index 512b2ab18..275b93cbc 100644 --- a/src/components/ai-edition/v4/FloatingInspector.test.tsx +++ b/src/components/ai-edition/v4/FloatingInspector.test.tsx @@ -1,5 +1,7 @@ // @vitest-environment jsdom import "@testing-library/jest-dom"; +import { readFileSync } from "node:fs"; +import path from "node:path"; import { act, fireEvent, @@ -24,6 +26,7 @@ vi.mock("../RightPanes", async (importOriginal) => ({ AudioPane: () =>
AudioPane
, // The real row: the zoom pane's choices are read and pressed below. ChoiceRow: (await importOriginal()).ChoiceRow, + Toggle: (await importOriginal()).Toggle, AudioTrackPane: ({ onClose }: { onClose?: () => void }) => (
AudioTrackPane @@ -59,6 +62,7 @@ vi.mock("../CaptionsPane", () => ({ CaptionsPane: () =>
CaptionsPane
, })); +import styles from "./EditorShellV4.module.css"; import { AnnotationSizeControl, AnnotationSizeField, FloatingInspector } from "./FloatingInspector"; // The rail's buttons have tooltips, and the app's root provides the provider they need. @@ -516,6 +520,85 @@ describe("FloatingInspector", () => { vi.unstubAllGlobals(); }); }); + + describe("full camera pane", () => { + const camTl = (region: Record) => { + const updateCameraFullscreenOrientation = vi.fn(); + const updateCameraFullscreenDeskLabel = vi.fn(); + const tl = { + ...defaultProps.tl, + selection: { kind: "cameraFullscreen", id: "cf" }, + cameraFullscreenRegions: [{ id: "cf", startMs: 0, endMs: 2000, ...region }], + updateCameraFullscreenOrientation, + updateCameraFullscreenDeskLabel, + removeRegion: vi.fn(), + } as unknown as React.ComponentProps["tl"]; + return { tl, updateCameraFullscreenOrientation, updateCameraFullscreenDeskLabel }; + }; + + it("desk view sets both fields in one call", () => { + const { tl, updateCameraFullscreenOrientation } = camTl({}); + render(); + const desk = screen.getByRole("button", { name: "settings.cameraFullscreen.deskView" }); + expect(desk).toHaveAttribute("aria-pressed", "false"); + fireEvent.click(desk); + expect(updateCameraFullscreenOrientation).toHaveBeenCalledTimes(1); + expect(updateCameraFullscreenOrientation).toHaveBeenCalledWith("cf", { + rotation: 180, + mirror: "auto", + }); + }); + + it("turns desk view off again in one call", () => { + const { tl, updateCameraFullscreenOrientation } = camTl({ rotation: 180 }); + render(); + const desk = screen.getByRole("button", { name: "settings.cameraFullscreen.deskView" }); + expect(desk).toHaveAttribute("aria-pressed", "true"); + fireEvent.click(desk); + expect(updateCameraFullscreenOrientation).toHaveBeenCalledWith("cf", { + rotation: 0, + mirror: "auto", + }); + }); + + // The pane button has no pressed look of its own, so desk view wears a toggle class + // whose lit state is styled off aria-pressed — and only that button wears it. + it("lights the desk view button off its pressed state", () => { + const { tl } = camTl({ rotation: 180 }); + render(); + const desk = screen.getByRole("button", { name: "settings.cameraFullscreen.deskView" }); + expect(styles.paneToggle).toBeTruthy(); + expect(desk.classList).toContain(styles.paneToggle); + expect(document.querySelectorAll(`.${styles.paneToggle}`)).toHaveLength(1); + const css = readFileSync(path.join(__dirname, "EditorShellV4.module.css"), "utf8"); + expect(css).toMatch(/\.paneToggle\[aria-pressed="true"\]\s*[,{]/); + }); + + it("offers the label switch only on a turned section", () => { + const plain = camTl({}); + const { unmount } = render(); + expect( + screen.queryByRole("button", { name: "settings.cameraFullscreen.showLabel" }), + ).toBeNull(); + unmount(); + const turned = camTl({ rotation: 180 }); + render(); + const toggle = screen.getByRole("button", { name: "settings.cameraFullscreen.showLabel" }); + expect(toggle).toHaveAttribute("aria-pressed", "true"); + fireEvent.click(toggle); + expect(turned.updateCameraFullscreenDeskLabel).toHaveBeenCalledWith("cf", false); + }); + + it("changes the mirror without touching the rotation", () => { + const { tl, updateCameraFullscreenOrientation } = camTl({ rotation: 180 }); + render(); + fireEvent.click(screen.getByRole("button", { name: "settings.cameraFullscreen.mirror.on" })); + expect(updateCameraFullscreenOrientation).toHaveBeenCalledWith("cf", { + rotation: 180, + mirror: "on", + }); + }); + }); }); describe("AnnotationSizeField", () => { diff --git a/src/components/ai-edition/v4/FloatingInspector.tsx b/src/components/ai-edition/v4/FloatingInspector.tsx index 8b0bac86e..db594f517 100644 --- a/src/components/ai-edition/v4/FloatingInspector.tsx +++ b/src/components/ai-edition/v4/FloatingInspector.tsx @@ -17,6 +17,7 @@ import { Maximize2, MousePointer2, Pencil, + RotateCw, Scissors, SlidersHorizontal, Trash2, @@ -60,6 +61,14 @@ import { useEditorSettings } from "@/lib/ai-edition/store/useEditorSettings"; import type { useTimeline } from "@/lib/ai-edition/store/useTimeline"; import { formatSeconds } from "@/lib/ai-edition/timeline/format"; import { coalescedTrimGroups } from "@/lib/ai-edition/timeline/trim-mapping"; +import { + type CameraMirrorMode, + type CameraRotation, + isDeskView, + normalizeCameraMirror, + normalizeCameraRotation, + showsDeskLabel, +} from "@/lib/cameraOrientation"; import { clampToBound } from "@/lib/projectDefaults"; import { annotationFootageRect, zoomScaleLimit } from "@/native/sceneDescription"; import { ColorField } from "../ColorField"; @@ -71,6 +80,7 @@ import { CursorPane, LayoutPane, SliderCell, + Toggle, TranscriptPane, VideoEffectsPane, } from "../RightPanes"; @@ -1267,6 +1277,11 @@ function SelectionPane({ tl, onClose }: { tl: TimelineApi; onClose: () => void } if (selection.kind === "cameraFullscreen") { const region = tl.cameraFullscreenRegions.find((c) => c.id === selection.id); if (!region) return null; + const rotation = normalizeCameraRotation(region.rotation); + const mirror = normalizeCameraMirror(region.mirror); + const desk = isDeskView(region); + const setOrientation = (next: { rotation: CameraRotation; mirror: CameraMirrorMode }) => + void tl.updateCameraFullscreenOrientation(region.id, next); return (
{paneHeader( @@ -1276,6 +1291,55 @@ function SelectionPane({ tl, onClose }: { tl: TimelineApi; onClose: () => void } tc("actions.close"), )}
+ {/* One click for the common case: a camera tilted onto the desk is upside down + and must not be mirrored, or the papers' text reads back to front. */} + + {paneStack( + ts("cameraFullscreen.rotation"), + + label={ts("cameraFullscreen.rotation")} + options={[ + { value: 0, label: "0°" }, + { value: 180, label: "180°" }, + ]} + value={rotation} + onChange={(next) => setOrientation({ rotation: next, mirror })} + />, + )} + {paneStack( + ts("cameraFullscreen.mirror.title"), + + label={ts("cameraFullscreen.mirror.title")} + options={[ + { value: "auto", label: ts("cameraFullscreen.mirror.auto") }, + { value: "on", label: ts("cameraFullscreen.mirror.on") }, + { value: "off", label: ts("cameraFullscreen.mirror.off") }, + ]} + value={mirror} + onChange={(next) => setOrientation({ rotation, mirror: next })} + />, + )} + {rotation === 180 && + paneRow( + ts("cameraFullscreen.showLabel"), + void tl.updateCameraFullscreenDeskLabel(region.id, v)} + />, + )}
+ + {isPerspective ? ( +
+ + +

+ {markerResult === "found" + ? t("cameraCalibration.markersFound") + : markerResult === "notFound" + ? t("cameraCalibration.markersNotFound") + : null} +

+
+ ) : null} + + {isPerspective ? ( +
+
+ {t("cameraCalibration.format")} + + label={t("cameraCalibration.format")} + columns={3} + options={[ + { value: "a4Portrait", label: t("cameraCalibration.formats.a4Portrait") }, + { value: "a4Landscape", label: t("cameraCalibration.formats.a4Landscape") }, + { value: "wide", label: "16:9" }, + { value: "standard", label: "4:3" }, + { value: "square", label: "1:1" }, + { value: "free", label: t("cameraCalibration.formats.free") }, + ]} + value={format} + onChange={setFormat} + /> + {format === "free" ? ( + + ) : null} + {format === "free" && aspect === null ? ( + + ) : null} + +
+
+ {t("cameraCalibration.preview")} + +
+
+ ) : null} + + {isPerspective && !quadValid ? ( + + ) : null} + +
+ +
+ + +
+
+ + ); +} diff --git a/src/components/ai-edition/CamerasSection.test.tsx b/src/components/ai-edition/CamerasSection.test.tsx new file mode 100644 index 000000000..a382aaa5f --- /dev/null +++ b/src/components/ai-edition/CamerasSection.test.tsx @@ -0,0 +1,185 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { fireEvent, render, screen } from "@testing-library/react"; +import { describe, expect, it, vi } from "vitest"; +import type { AxcutDocument } from "@/lib/ai-edition/schema"; + +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: (scope: string) => (key: string, vars?: Record) => + vars?.n !== undefined + ? `${scope}.${key}#${vars.n}${vars.label ? `|${vars.label}` : ""}` + : `${scope}.${key}`, +})); +vi.mock("@/lib/ai-edition/timeline/grabFrame", () => ({ + grabFrameDataUrl: vi.fn(() => Promise.resolve("data:image/png;base64,AAAA")), +})); + +import { readFileSync } from "node:fs"; +import { act } from "@testing-library/react"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import { grabFrameDataUrl } from "@/lib/ai-edition/timeline/grabFrame"; +import { CamerasSection, calibrationCameraAt, cameraHasCrop } from "./CamerasSection"; + +const track = (sourcePath: string, extra: Record = {}) => ({ + sourcePath, + visible: true, + startMs: 0, + offsetMs: 0, + ...extra, +}); + +function makeDoc(extraCameras: number): AxcutDocument { + return { + assets: [ + { + id: "a1", + cameraTrack: track("C:/cam1.mp4"), + additionalCameraTracks: Array.from({ length: extraCameras }, (_, i) => + track(`C:/cam${i + 2}.mp4`, { label: i === 0 ? "Desk" : "" }), + ), + }, + ], + timeline: { + clips: [ + { + id: "c1", + assetId: "a1", + sourceStartSec: 0, + sourceEndSec: 10, + timelineStartSec: 0, + timelineEndSec: 10, + }, + ], + }, + } as unknown as AxcutDocument; +} + +const render2 = (extra: number, setCameraSettings = vi.fn(), onOpenCalibration = vi.fn()) => { + render( + , + ); + return { setCameraSettings, onOpenCalibration }; +}; + +describe("CamerasSection", () => { + it("lists every camera of the clip", () => { + render2(2); + expect(screen.getByTestId("camera-row-0")).toHaveTextContent("settings.cameras.cameraN#1"); + expect(screen.getByTestId("camera-row-1")).toHaveTextContent( + "settings.cameras.cameraNamed#2|Desk", + ); + expect(screen.getByTestId("camera-row-2")).toHaveTextContent("settings.cameras.cameraN#3"); + }); + + it("camera 1 offers only the perspective", () => { + const { onOpenCalibration } = render2(1); + const row = screen.getByTestId("camera-row-0"); + expect(row).toHaveTextContent("settings.cameras.camera1Hint"); + expect(row.querySelectorAll("button")).toHaveLength(1); + fireEvent.click(row.querySelector("button") as Element); + expect(onOpenCalibration).toHaveBeenCalledWith(0, "perspective"); + }); + + it("rotating camera 2 calls setCameraSettings", () => { + const { setCameraSettings, onOpenCalibration } = render2(1); + const row = screen.getByTestId("camera-row-1"); + fireEvent.click( + row.querySelector("button[aria-pressed]:not([aria-pressed='true'])") as Element, + ); + expect(setCameraSettings).toHaveBeenCalledWith(1, { rotation: 180 }); + const buttons = Array.from(row.querySelectorAll("button")); + fireEvent.click(buttons.find((b) => b.textContent === "settings.cameras.crop") as Element); + expect(onOpenCalibration).toHaveBeenCalledWith(1, "crop"); + }); + + it("follows the playhead from the store without a prop", async () => { + vi.useFakeTimers(); + useProjectStore.setState({ currentTimeSec: 1 }); + render( + , + ); + act(() => { + useProjectStore.setState({ currentTimeSec: 4 }); + }); + act(() => { + vi.advanceTimersByTime(300); + }); + expect(vi.mocked(grabFrameDataUrl).mock.calls.at(-1)?.[1]).toBe(4); + vi.useRealTimers(); + }); + + it("keeps the playhead subscription out of LayoutPane", () => { + const source = readFileSync("src/components/ai-edition/RightPanes.tsx", "utf8"); + const start = source.indexOf("export function LayoutPane("); + const end = source.indexOf("/** The tightest the frame gets"); + expect(source.slice(start, end)).not.toContain("currentTimeSec"); + }); + + it("finds the calibration still of a camera at the playhead", () => { + const t = (key: string, vars?: Record) => + vars?.n !== undefined ? `${key}#${vars.n}${vars.label ? `|${vars.label}` : ""}` : key; + const doc = makeDoc(1); + const desk = calibrationCameraAt(doc, 2, 1, t); + expect(desk).toMatchObject({ index: 1, label: "cameras.cameraNamed#2|Desk", timeSec: 2 }); + expect(desk?.src).toMatch(/^file:\/\/.*cam2\.mp4$/); + expect(calibrationCameraAt(doc, 2, 0, t)?.label).toBe("cameras.cameraN#1"); + // No such camera, or no document: nothing to calibrate. + expect(calibrationCameraAt(doc, 2, 3, t)).toBeNull(); + expect(calibrationCameraAt(null, 2, 0, t)).toBeNull(); + }); + + it("a perspective disables the crop with a hint", () => { + const onOpenCalibration = vi.fn(); + render( + , + ); + const row = screen.getByTestId("camera-row-1"); + const crop = screen.getByRole("button", { name: "settings.cameras.crop" }); + expect(crop).toBeDisabled(); + expect(row).toHaveTextContent("settings.cameras.cropOffWithPerspective"); + fireEvent.click(crop); + expect(onOpenCalibration).not.toHaveBeenCalled(); + }); + + it("without a perspective the crop stays available and no hint shows", () => { + render2(1); + expect(screen.getByRole("button", { name: "settings.cameras.crop" })).toBeEnabled(); + expect(screen.queryByText("settings.cameras.cropOffWithPerspective")).toBeNull(); + }); + + it("knows whether a camera has a crop", () => { + const crop = { x: 0.1, y: 0.1, width: 0.5, height: 0.5 }; + expect(cameraHasCrop(null, [null, { crop }], 1)).toBe(true); + expect(cameraHasCrop(null, [null, { mirror: true }], 1)).toBe(false); + const withCamera1Crop = (webcamCropRegion: unknown) => + ({ legacyEditor: { webcamCropRegion } }) as unknown as AxcutDocument; + expect(cameraHasCrop(withCamera1Crop(crop), [], 0)).toBe(true); + expect(cameraHasCrop(withCamera1Crop({ x: 0, y: 0, width: 1, height: 1 }), [], 0)).toBe(false); + expect(cameraHasCrop(makeDoc(0), [], 0)).toBe(false); + }); +}); diff --git a/src/components/ai-edition/CamerasSection.tsx b/src/components/ai-edition/CamerasSection.tsx new file mode 100644 index 000000000..92585ae43 --- /dev/null +++ b/src/components/ai-edition/CamerasSection.tsx @@ -0,0 +1,245 @@ +// The "Cameras" section of the layout pane: every camera of the clip under the playhead, with +// a thumbnail and the per-camera settings. Camera 1 keeps its rotation, mirror and crop in +// the controls above (older fields), so its row only offers the perspective correction. + +import { useEffect, useMemo, useState } from "react"; +import { toFileUrl } from "@/components/video-editor/projectPersistence"; +import type { CameraSettings, CropRegion } from "@/components/video-editor/types"; +import { useScopedT } from "@/contexts/I18nContext"; +import type { AxcutDocument } from "@/lib/ai-edition/schema"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { camerasForClipAt, type ProjectCamera } from "@/lib/ai-edition/timeline/cameraList"; +import { grabFrameDataUrl } from "@/lib/ai-edition/timeline/grabFrame"; +import { locateVirtualPosition } from "@/lib/ai-edition/timeline/virtual-preview"; +import type { CameraRotation } from "@/lib/cameraOrientation"; +import styles from "./NewEditorShell.module.css"; +import { ChoiceRow, Toggle } from "./RightPanes"; + +/** What the calibration dialog edits: the picture's crop, or its perspective. */ +export type CalibrationMode = "crop" | "perspective"; + +const THUMBNAIL_SIDE = 96; +/** The playhead moves every frame during playback; a still is only worth grabbing once it rests. */ +const THUMBNAIL_DEBOUNCE_MS = 250; + +export interface CamerasSectionProps { + document: AxcutDocument | null; + /** Overrides the playhead on the ruler; by default it is read from the project store here, so + * only this section re-renders while the playhead moves. */ + playheadSec?: number; + /** `legacyEditor.cameraSettings`, normalized: index 0 = camera 1. */ + cameraSettings: (CameraSettings | null)[]; + setCameraSettings: (index: number, patch: Partial | null) => void | Promise; + /** Opens the calibration dialog for a camera. */ + onOpenCalibration?: (cameraIndex: number, mode: CalibrationMode) => void; +} + +export type CameraStill = { src: string; timeSec: number }; + +/** Where each camera's file is, and which time of it the playhead shows. */ +export function stillsAt( + document: AxcutDocument | null, + playheadSec: number, +): Map { + const stills = new Map(); + if (!document) return stills; + const position = locateVirtualPosition(document.timeline.clips, playheadSec); + if (!position) return stills; + const asset = document.assets.find((a) => a.id === position.clip.assetId); + const sources = [assetCameraSource(asset), ...assetAdditionalCameraSources(asset)]; + sources.forEach((source, index) => { + if (!source.path) return; + const src = /^(https?|blob|data):/.test(source.path) ? source.path : toFileUrl(source.path); + stills.set(index, { src, timeSec: Math.max(0, position.sourceTimeSec - source.offsetSec) }); + }); + return stills; +} + +/** + * Whether a camera has a crop stored. Camera 1 keeps its crop in `webcamCropRegion` (a + * full-frame rect means none); the others in `cameraSettings[k].crop`. + */ +export function cameraHasCrop( + document: AxcutDocument | null, + cameraSettings: (CameraSettings | null)[], + index: number, +): boolean { + if (index !== 0) return cameraSettings[index]?.crop != null; + const legacy = document?.legacyEditor as Record | null | undefined; + const crop = legacy?.webcamCropRegion as Partial | undefined; + if (!crop) return false; + const coversFrame = (v: unknown) => typeof v !== "number" || v >= 1 - 1e-6; + return !(coversFrame(crop.width) && coversFrame(crop.height)); +} + +/** The camera the calibration dialog opens on: its label and its still at the playhead. */ +export function calibrationCameraAt( + document: AxcutDocument | null, + playheadSec: number, + index: number, + t: (key: string, vars?: Record) => string, +): { index: number; label: string; src: string; timeSec: number } | null { + if (!document) return null; + const camera = camerasForClipAt(document, playheadSec, t).find((c) => c.index === index); + const still = stillsAt(document, playheadSec).get(index); + if (!camera?.available || !still) return null; + return { index, label: camera.label, ...still }; +} + +function CameraThumbnail({ still, label }: { still: CameraStill | undefined; label: string }) { + const [url, setUrl] = useState(null); + const src = still?.src; + const timeSec = still?.timeSec; + useEffect(() => { + if (src === undefined || timeSec === undefined) { + setUrl(null); + return; + } + let cancelled = false; + const timer = setTimeout(() => { + grabFrameDataUrl(src, timeSec, THUMBNAIL_SIDE) + .then((dataUrl) => { + if (!cancelled) setUrl(dataUrl); + }) + .catch(() => { + if (!cancelled) setUrl(null); + }); + }, THUMBNAIL_DEBOUNCE_MS); + return () => { + cancelled = true; + clearTimeout(timer); + }; + }, [src, timeSec]); + return ( +
+ {url ? ( + {label} + ) : null} +
+ ); +} + +const BUTTON = `${styles.btn} ${styles.btnSecondary}`; + +export function CamerasSection({ + document, + playheadSec: playheadOverride, + cameraSettings, + setCameraSettings, + onOpenCalibration, +}: CamerasSectionProps) { + const ts = useScopedT("settings"); + const storePlayheadSec = useProjectStore((s) => s.currentTimeSec); + const playheadSec = playheadOverride ?? storePlayheadSec; + const cameras: ProjectCamera[] = useMemo( + () => (document ? camerasForClipAt(document, playheadSec, ts) : []), + [document, playheadSec, ts], + ); + const stills = useMemo(() => stillsAt(document, playheadSec), [document, playheadSec]); + if (cameras.length === 0) return null; + return ( + <> +
{ts("cameras.title")}
+ {cameras.map((camera) => { + const settings = cameraSettings[camera.index] ?? null; + const isFirst = camera.index === 0; + // A perspective replaces the crop at render, so the crop is not offered then. + const hasPerspective = settings?.perspective != null; + return ( +
+
+ +
+
{camera.label}
+ {camera.available ? null : ( +

{ts("cameras.unavailable")}

+ )} +
+
+ {isFirst ?

{ts("cameras.camera1Hint")}

: null} + {isFirst ? null : ( + <> + + label={ts("cameras.rotation")} + options={[ + { value: 0, label: "0°" }, + { value: 180, label: "180°" }, + ]} + value={settings?.rotation ?? 0} + onChange={(rotation) => void setCameraSettings(camera.index, { rotation })} + /> +
+ {ts("cameras.mirror")} + void setCameraSettings(camera.index, { mirror })} + /> +
+ + )} +
+ {isFirst ? null : ( + + )} + + {!isFirst && settings ? ( + + ) : null} +
+ {!isFirst && hasPerspective ? ( +

{ts("cameras.cropOffWithPerspective")}

+ ) : null} +
+ ); + })} + + ); +} diff --git a/src/components/ai-edition/ExportDialog.params.test.tsx b/src/components/ai-edition/ExportDialog.params.test.tsx index d78ad091c..76cc992c2 100644 --- a/src/components/ai-edition/ExportDialog.params.test.tsx +++ b/src/components/ai-edition/ExportDialog.params.test.tsx @@ -228,3 +228,47 @@ describe("ExportDialog format settings", () => { } }); }); + +describe("ExportDialog clip list", () => { + beforeEach(() => { + window.electronAPI = { + pickExportSavePath: vi.fn(async () => ({ path: "/tmp/out.mp4" })), + onNativeExportProgress: vi.fn(() => noop), + } as unknown as ElectronAPI; + }); + + afterEach(() => { + cleanup(); + vi.clearAllMocks(); + }); + + it("hands the native export the asset's extra cameras", async () => { + const withCameras: AxcutDocument = { + ...DOC, + assets: [ + { + ...DOC.assets[0], + additionalCameraTracks: [ + { sourcePath: "/tmp/cam2.mp4", startMs: 500, offsetMs: 250, visible: true, label: "" }, + { sourcePath: "/tmp/cam3.mp4", startMs: 0, offsetMs: 0, visible: false, label: "" }, + ], + }, + ], + }; + renderDialog(withCameras); + await exportMp4(); + const clips = vi.mocked(exportMultiNative).mock.calls.at(-1)?.[0]; + expect(clips?.[0]?.additionalCameras).toEqual([ + { path: "/tmp/cam2.mp4", offsetSec: 0.75 }, + // A hidden track keeps its slot, so the indices stay aligned with the tracks. + { path: "", offsetSec: 0 }, + ]); + }); + + it("sends no extra cameras for a one-camera asset", async () => { + renderDialog(); + await exportMp4(); + const clips = vi.mocked(exportMultiNative).mock.calls.at(-1)?.[0]; + expect(clips?.[0]).not.toHaveProperty("additionalCameras"); + }); +}); diff --git a/src/components/ai-edition/ExportDialog.tsx b/src/components/ai-edition/ExportDialog.tsx index 47b08b08a..be340bb60 100644 --- a/src/components/ai-edition/ExportDialog.tsx +++ b/src/components/ai-edition/ExportDialog.tsx @@ -19,7 +19,7 @@ import { } from "@/lib/ai-edition/document/outputFormat"; import type { AxcutDocument } from "@/lib/ai-edition/schema"; import { getEditorSettings } from "@/lib/ai-edition/store/editorSettings"; -import { assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; import { resolveClipSourceEndSec } from "@/lib/ai-edition/timeline/clipDuration"; import { type ExportFormat, @@ -117,6 +117,8 @@ function buildNativeClipList(document: AxcutDocument): CompositorClipInput[] { return []; } const camera = assetCameraSource(asset); + // Cameras 2-4, only sent when the asset has any (same rule as `buildSceneDescription`). + const additionalCameras = assetAdditionalCameraSources(asset); // sourceEndSec is optional in the schema (unknown until probed) — fall back through // the single canonical precedence used by every consumer (clip.probe → asset.duration // → timeline-length guess). See `resolveClipSourceEndSec` for the full order. @@ -132,6 +134,7 @@ function buildNativeClipList(document: AxcutDocument): CompositorClipInput[] { sourceEndSec, webcamOffsetSec: camera.offsetSec, hasAudio: true, + ...(additionalCameras.length > 0 ? { additionalCameras } : {}), }, ]; }); diff --git a/src/components/ai-edition/NativeCompositorOverlay.test.tsx b/src/components/ai-edition/NativeCompositorOverlay.test.tsx index 7c4c0f397..f776feab7 100644 --- a/src/components/ai-edition/NativeCompositorOverlay.test.tsx +++ b/src/components/ai-edition/NativeCompositorOverlay.test.tsx @@ -157,10 +157,31 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(8, true); - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 10); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 10, []); expect(native.setNativePlaying).not.toHaveBeenCalledWith(false); }); + it("sends the asset's extra cameras with the clip", async () => { + const document = makeDocument(); + document.assets[0] = { + ...document.assets[0], + additionalCameraTracks: [ + { sourcePath: "/cam-2.mp4", startMs: 500, offsetMs: 0, visible: true, label: "" }, + { sourcePath: "/cam-3.mp4", startMs: 0, offsetMs: 0, visible: false, label: "" }, + ], + }; + useProjectStore.setState({ document }); + await mountAtFirstClip(); + publishNativePosition({ clipIndex: 0, sourceTimeSec: 4.9 }, now); + + setPlayhead(8, true); + + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 10, [ + { path: "/cam-2.mp4", offsetSec: 0.5 }, + { path: "", offsetSec: 0 }, + ]); + }); + it("re-anchors a view that stays behind, once the gap holds", async () => { await mountAtFirstClip(); // The view stalled 0.4 s behind the playhead, inside the first clip. @@ -173,7 +194,7 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(2.52, true); expect(native.setActiveClip).toHaveBeenCalledTimes(1); - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 2.52); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 2.52, []); }); // At 16× a frame that takes 30 ms to arrive shows the playhead 0.48 s of programme ago. @@ -228,8 +249,8 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(time, false); } - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 14.4); - expect(native.setActiveClip).toHaveBeenLastCalledWith(7, "/take.mp4", "", 0, 0, 14.4); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 0, 14.4, []); + expect(native.setActiveClip).toHaveBeenLastCalledWith(7, "/take.mp4", "", 0, 0, 14.4, []); }); it("still sends the clip on a change while paused", async () => { @@ -239,7 +260,7 @@ describe("NativeCompositorOverlay while playing", () => { setPlayhead(6, false); - expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 8); + expect(native.setActiveClip).toHaveBeenCalledWith(7, "/take.mp4", "", 0, 1, 8, []); }); // An addon that reports no position cannot be read, so it is driven as before. diff --git a/src/components/ai-edition/NativeCompositorOverlay.tsx b/src/components/ai-edition/NativeCompositorOverlay.tsx index f3a90605b..84a67ce39 100644 --- a/src/components/ai-edition/NativeCompositorOverlay.tsx +++ b/src/components/ai-edition/NativeCompositorOverlay.tsx @@ -4,7 +4,7 @@ import { readSpeedRegions } from "@/lib/ai-edition/document/timeline"; import { noteUiProbeClipSwitch } from "@/lib/ai-edition/perf/uiFrameProbe"; import { getEditorSettings } from "@/lib/ai-edition/store/editorSettings"; import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; -import { assetCameraSource } from "@/lib/ai-edition/timeline/camera"; +import { assetAdditionalCameraSources, assetCameraSource } from "@/lib/ai-edition/timeline/camera"; import { findActiveSpeedRegion, type SpeedRegion } from "@/lib/ai-edition/timeline/speed"; import { resolveNativePosition } from "@/lib/ai-edition/timeline/timelineMap"; import { @@ -279,6 +279,7 @@ export function NativeCompositorOverlay() { camera.offsetSec, activeClipIndex, activeSourceTimeSec, + assetAdditionalCameraSources(asset), ) .then(() => { if (pendingTargetClipIdRef.current !== targetClipId) { @@ -355,6 +356,7 @@ export function NativeCompositorOverlay() { camera.offsetSec, activeClipIndex, activeSourceTimeSec, + assetAdditionalCameraSources(asset), ).catch((error: unknown) => { console.warn("[compositor-view] re-anchoring the preview failed:", error); }); diff --git a/src/components/ai-edition/NewEditorShell.module.css b/src/components/ai-edition/NewEditorShell.module.css index 30b3c0947..162591937 100644 --- a/src/components/ai-edition/NewEditorShell.module.css +++ b/src/components/ai-edition/NewEditorShell.module.css @@ -763,6 +763,26 @@ visibility: hidden; display: block; } +.layoutPlace { + /* Move/resize hitbox of a layout section's camera window, over the native-drawn picture. */ + position: absolute; + z-index: 3; + box-sizing: border-box; + border: 1px dashed var(--accent); + cursor: move; + touch-action: none; +} +.layoutPlaceHandle { + position: absolute; + width: 12px; + height: 12px; + transform: translate(-50%, -50%); + box-sizing: border-box; + border: 2px solid var(--accent); + border-radius: 2px; + background: #fff; + touch-action: none; +} /* ─── transport bar (lives inside .timelineHead, next to the tool icons) ─ */ .transport { display: flex; diff --git a/src/components/ai-edition/NewEditorShell.tsx b/src/components/ai-edition/NewEditorShell.tsx index ee12161bd..032b878cd 100644 --- a/src/components/ai-edition/NewEditorShell.tsx +++ b/src/components/ai-edition/NewEditorShell.tsx @@ -2,6 +2,7 @@ import { useCallback, useEffect, useMemo, useRef, useState } from "react"; import { toast } from "sonner"; import type { EditorProjectData } from "@/components/video-editor/projectPersistence"; import { toFileUrl } from "@/components/video-editor/projectPersistence"; +import type { CameraSettings } from "@/components/video-editor/types"; import { useEditorDialogActions } from "@/contexts/EditorDialogsContext"; import { useScopedT } from "@/contexts/I18nContext"; import { useShortcuts } from "@/contexts/ShortcutsContext"; @@ -33,6 +34,12 @@ import { } from "@/lib/ai-edition/schema"; import { useMcpDocumentHost } from "@/lib/ai-edition/store/mcpDocumentHost"; import { saveWithDeadline, useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import { + copySourceKey, + pasteHitsCameraSection, + pasteIdPrefix, + pasteTarget, +} from "@/lib/ai-edition/store/regionClipboardKinds"; import { useAssetTranscriptions, useAutoTranscription, @@ -44,6 +51,7 @@ import { future as redoStack, past as undoStack } from "@/lib/ai-edition/store/u import { useChatPromptBus } from "@/lib/ai-edition/store/useChatPromptBus"; import { useSequentialTimelineOps } from "@/lib/ai-edition/store/useSequentialTimelineOps"; import { useTimeline } from "@/lib/ai-edition/store/useTimeline"; +import { showCameraSectionOutcome } from "@/lib/ai-edition/timeline/cameraSectionNotice"; import { isGeneratedAssetId } from "@/lib/ai-edition/timeline/clip-parts"; import { mergeCloseCuts } from "@/lib/ai-edition/timeline/cut-breath"; import { newRegionDurationSec } from "@/lib/ai-edition/timeline/newRegionDuration"; @@ -58,6 +66,8 @@ import { nativeBridgeClient } from "@/native"; import type { AiEditionProjectSummary } from "@/native/contracts"; import { resolveVisibleClips } from "@/native/sceneDescription"; import { useNativePlaybackSync } from "@/native/useNativePlaybackSync"; +import { type CalibrationCamera, CameraCalibrationModal } from "./CameraCalibrationModal"; +import { type CalibrationMode, calibrationCameraAt, cameraHasCrop } from "./CamerasSection"; import { ExportDialog } from "./ExportDialog"; import { insertionsEnabled } from "./insertionsEnabled"; import { ChatStripPanel } from "./LeftPanel"; @@ -192,6 +202,8 @@ export async function runLoadedMetadataWrite( export function NewEditorShell() { const te = useScopedT("editor"); + const tt = useScopedT("timeline"); + const ts = useScopedT("settings"); useMcpDocumentHost(); const document = useProjectStore((s) => s.document); const projectId = useProjectStore((s) => s.projectId); @@ -266,6 +278,20 @@ export function NewEditorShell() { // "Edit clip" rail button — a single shell-level instance instead of one // mounted per trigger site. const [editClipTarget, setEditClipTarget] = useState(null); + // The camera calibration dialog (perspective or crop), opened from the layout pane's camera + // list. The still is taken at the playhead as it opens; one instance for the editor. + const [calibration, setCalibration] = useState<{ + camera: CalibrationCamera; + mode: CalibrationMode; + } | null>(null); + const openCalibration = useCallback( + (cameraIndex: number, mode: CalibrationMode) => { + const { document: doc, currentTimeSec } = useProjectStore.getState(); + const camera = calibrationCameraAt(doc, currentTimeSec, cameraIndex, ts); + if (camera) setCalibration({ camera, mode }); + }, + [ts], + ); const [exportOpen, setExportOpen] = useState(false); const [unsavedPrompt, setUnsavedPrompt] = useState<{ action: "close" | "new" | "open" | "record"; @@ -340,6 +366,15 @@ export function NewEditorShell() { saveDocument, }); + // Every per-camera settings write (Cameras section toggles, reset, calibration apply) goes + // through the shared queue, so a toggle cannot race a calibration apply on a stale document. + const setTimelineCameraSettings = tl.setCameraSettings; + const setCameraSettingsQueued = useCallback( + (index: number, patch: Partial | null) => + enqueueTimelineWrite(() => setTimelineCameraSettings(index, patch)), + [enqueueTimelineWrite, setTimelineCameraSettings], + ); + const promptUnsaved = useCallback( (action: "close" | "new" | "open" | "record"): Promise => { if (!dirty) return Promise.resolve("discard"); @@ -1082,7 +1117,9 @@ export function NewEditorShell() { // Land it at the playhead, keeping the copied length. const timeMs = Math.round(useProjectStore.getState().currentTimeSec * 1000); const src = snapshot.region as { startMs: number; endMs: number }; - const prefix = snapshot.kind === "annotation" ? "ann" : snapshot.kind; + const prefix = pasteIdPrefix(snapshot.kind); + const target = pasteTarget(snapshot.kind); + if (!target) return; // Audio re-ventilates through its own anchorer, which advances each // fragment's source offset — the generic one would copy the offset into @@ -1120,32 +1157,24 @@ export function NewEditorShell() { () => createId(prefix), ); - if (snapshot.kind === "zoom") { - await saveDocument( - { - ...doc, - zoomRanges: [...doc.zoomRanges, ...anchored] as typeof doc.zoomRanges, - }, - { history: true }, - ); - } else if (snapshot.kind === "annotation") { - await saveDocument( - { - ...doc, - annotations: [...doc.annotations, ...anchored] as typeof doc.annotations, - }, - { history: true }, - ); + if (target.store === "document") { + const rows = doc[target.key] as unknown[]; + await saveDocument({ ...doc, [target.key]: [...rows, ...anchored] }, { history: true }); } else { - // speed and cameraFullscreen are both plain spans on legacyEditor. - const key = snapshot.kind === "speed" ? "speedRegions" : "cameraFullscreenRegions"; const legacy = (doc.legacyEditor as Record) ?? {}; - const prev = (legacy[key] as unknown[]) ?? []; + // Full Camera and layout sections share one lane: a paste that would land on the + // other list is refused like an add is, instead of being dropped by the scene. A Full + // Camera over Full Camera is not refused: those merge, as they do on add. + if ( + target.key !== "speedRegions" && + pasteHitsCameraSection(legacy, target.key, pasted.startMs, pasted.endMs) + ) { + showCameraSectionOutcome("occupied", tt); + return; + } + const prev = (legacy[target.key] as unknown[]) ?? []; await saveDocument( - { - ...doc, - legacyEditor: { ...legacy, [key]: [...prev, ...anchored] }, - }, + { ...doc, legacyEditor: { ...legacy, [target.key]: [...prev, ...anchored] } }, { history: true }, ); } @@ -1153,7 +1182,7 @@ export function NewEditorShell() { // `tl` belongs here now that the trim branch calls tl.addTrim: useTimeline // returns a fresh object each render, so memoizing on saveDocument alone // would paste through a callback holding a stale document. - }, [saveDocument, tl, te]); + }, [saveDocument, tl, te, tt]); // Copy the SELECTED pill. Reads the same arrays the lanes render, so what gets // copied is what the user is looking at — the old version dug into the raw @@ -1197,14 +1226,9 @@ export function NewEditorShell() { return; } - const source = - sel.kind === "zoom" - ? tl.zoomRegions - : sel.kind === "annotation" - ? tl.annotationRegions - : sel.kind === "speed" - ? tl.speedRegions - : tl.cameraFullscreenRegions; + const sourceKey = copySourceKey(sel.kind); + if (!sourceKey) return; + const source = tl[sourceKey]; const region = (source as Array<{ id: string }>).find((r) => r.id === sel.id); if (!region) return; copyRegion({ kind: sel.kind, region: region as unknown as Record }); @@ -1386,7 +1410,9 @@ export function NewEditorShell() { } if (matchesShortcut(e, shortcuts.addCameraFullscreen, isMac)) { e.preventDefault(); - void tl.addCameraFullscreen(newRegionDurationSec()); + void tl.addCameraFullscreen(newRegionDurationSec()).then((outcome) => { + showCameraSectionOutcome(outcome, tt); + }); return; } @@ -1433,6 +1459,7 @@ export function NewEditorShell() { isMac, togglePlay, handleSeek, + tt, ]); const showTimeline = mode !== "rec"; @@ -1622,6 +1649,12 @@ export function NewEditorShell() { selectedZoomRegionId={tl.selection?.kind === "zoom" ? tl.selection.id : null} onZoomFocusChange={tl.updateZoomFocusLive} onZoomFocusCommit={() => void tl.commitZoomFocus()} + cameraLayoutRegions={tl.cameraLayoutRegions} + selectedLayoutRegionId={ + tl.selection?.kind === "cameraLayout" ? tl.selection.id : null + } + onLayoutSlotRectLive={tl.updateLayoutSlotRectLive} + onLayoutSlotRectCommit={() => void tl.commitLayoutSlotRect()} annotationRegions={tl.annotationRegions} selectedAnnotationId={ tl.selection?.kind === "annotation" ? tl.selection.id : null @@ -1655,6 +1688,8 @@ export function NewEditorShell() { clips={tl.clips} onEditClip={setEditClipTarget} transcriptProps={transcriptProps} + onOpenCalibration={openCalibration} + setCameraSettings={setCameraSettingsQueued} /> ) : mode === "media" ? ( @@ -1746,6 +1781,18 @@ export function NewEditorShell() { setEditClipTarget(null); }} /> + {calibration ? ( + void setCameraSettingsQueued(calibration.camera.index, patch)} + onClose={() => setCalibration(null)} + /> + ) : null} { diff --git a/src/components/ai-edition/Preview.tsx b/src/components/ai-edition/Preview.tsx index f4ad0a5c5..23f032c00 100644 --- a/src/components/ai-edition/Preview.tsx +++ b/src/components/ai-edition/Preview.tsx @@ -1,5 +1,9 @@ import { useCallback, useEffect, useMemo, useState } from "react"; -import type { CameraFullscreenRegion, ZoomFocus } from "@/components/video-editor/types"; +import type { + CameraFullscreenRegion, + NormalizedRect, + ZoomFocus, +} from "@/components/video-editor/types"; import { useScopedT } from "@/contexts/I18nContext"; import type { AxcutAnnotationRegion, @@ -11,6 +15,7 @@ import type { } from "@/lib/ai-edition/schema"; import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; import type { SpeedRegion } from "@/lib/ai-edition/timeline/speed"; +import type { AnchoredCameraLayoutRegion } from "@/lib/cameraLayouts"; import { EditorEmptyState } from "./EditorEmptyState"; import styles from "./NewEditorShell.module.css"; import { PreviewCanvas } from "./PreviewCanvas"; @@ -43,6 +48,10 @@ interface PreviewProps { selectedZoomRegionId?: string | null; onZoomFocusChange?: (id: string, focus: ZoomFocus) => void; onZoomFocusCommit?: () => void; + cameraLayoutRegions?: AnchoredCameraLayoutRegion[]; + selectedLayoutRegionId?: string | null; + onLayoutSlotRectLive?: (id: string, slotIndex: number, rect: NormalizedRect) => void; + onLayoutSlotRectCommit?: () => void; annotationRegions?: AxcutAnnotationRegion[]; selectedAnnotationId?: string | null; onSelectAnnotation?: (id: string) => void; @@ -78,6 +87,10 @@ export function Preview({ selectedZoomRegionId, onZoomFocusChange, onZoomFocusCommit, + cameraLayoutRegions, + selectedLayoutRegionId, + onLayoutSlotRectLive, + onLayoutSlotRectCommit, annotationRegions, selectedAnnotationId, onSelectAnnotation, @@ -221,6 +234,10 @@ export function Preview({ selectedZoomRegionId={selectedZoomRegionId} onZoomFocusChange={onZoomFocusChange} onZoomFocusCommit={onZoomFocusCommit} + cameraLayoutRegions={cameraLayoutRegions} + selectedLayoutRegionId={selectedLayoutRegionId} + onLayoutSlotRectLive={onLayoutSlotRectLive} + onLayoutSlotRectCommit={onLayoutSlotRectCommit} annotationRegions={annotationRegions} selectedAnnotationId={selectedAnnotationId} onSelectAnnotation={onSelectAnnotation} diff --git a/src/components/ai-edition/PreviewCanvas.layoutPlaces.test.tsx b/src/components/ai-edition/PreviewCanvas.layoutPlaces.test.tsx new file mode 100644 index 000000000..3baa78fa3 --- /dev/null +++ b/src/components/ai-edition/PreviewCanvas.layoutPlaces.test.tsx @@ -0,0 +1,177 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { cleanup, fireEvent, render, screen } from "@testing-library/react"; +import type { ComponentProps } from "react"; +import { afterEach, beforeAll, describe, expect, it, vi } from "vitest"; +import type { AxcutAsset, AxcutClip } from "@/lib/ai-edition/schema"; +import { createEmptyDocument } from "@/lib/ai-edition/schema"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import type { AnchoredCameraLayoutRegion } from "@/lib/cameraLayouts"; +import { PreviewCanvas } from "./PreviewCanvas"; + +vi.mock("@/native/client", () => ({ nativeBridgeClient: { aiEdition: {} } })); +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: (scope: string) => (key: string) => `${scope}.${key}`, +})); +// The pixel and media layers are not under test: only the DOM hitboxes are. +vi.mock("./NativeCompositorOverlay", () => ({ NativeCompositorOverlay: () => null })); +vi.mock("./VirtualPreview", () => ({ VirtualPreview: () => null })); +vi.mock("./WebcamOverlay", () => ({ WebcamOverlay: () => null })); +vi.mock("./ZoomFocusOverlay", () => ({ ZoomFocusOverlay: () => null })); +vi.mock("./AnnotationLayer", () => ({ AnnotationLayer: () => null })); + +const track = (path: string) => ({ + sourcePath: path, + startMs: 0, + offsetMs: 0, + visible: true, + width: 1920, + height: 1080, +}); + +const asset = { + id: "a1", + kind: "video", + label: "a", + originalPath: "/screen.mp4", + cameraTrack: track("/c1.mp4"), + additionalCameraTracks: [{ ...track("/c2.mp4"), label: "" }], +} as unknown as AxcutAsset; + +const clip = { + id: "c1", + assetId: "a1", + sourceStartSec: 0, + sourceEndSec: 10, + timelineStartSec: 0, + timelineEndSec: 10, + wordRefs: [], + origin: "user", + reason: "", +} as AxcutClip; + +const region = ( + template: AnchoredCameraLayoutRegion["template"], + cameras: number[], +): AnchoredCameraLayoutRegion => ({ + id: "l1", + startMs: 0, + endMs: 10_000, + template, + slots: cameras.map((camera) => ({ camera })), + clipId: "c1", + assetId: "a1", + sourceStartSec: 0, + sourceEndSec: 10, +}); + +type Props = ComponentProps; + +function renderCanvas(over: Partial) { + const document = createEmptyDocument({ projectId: "p", title: "t" }); + useProjectStore.setState({ + projectId: "p", + document: { ...document, assets: [asset] }, + }); + const props: Props = { + videoSources: [], + clips: [clip], + seekTarget: null, + onTimeChange: vi.fn(), + onSeek: vi.fn(), + onLoadedMetadata: vi.fn(), + onVideoElement: vi.fn(), + currentTimeSec: 5, + onLayoutSlotRectLive: vi.fn(), + onLayoutSlotRectCommit: vi.fn(), + ...over, + }; + render( +
+ +
, + ); + return props; +} + +beforeAll(() => { + // jsdom lays nothing out: give every element a 1000×500 box and a pointer-capture API. + vi.spyOn(HTMLElement.prototype, "getBoundingClientRect").mockReturnValue({ + left: 0, + top: 0, + width: 1000, + height: 500, + right: 1000, + bottom: 500, + x: 0, + y: 0, + toJSON: () => ({}), + }); + HTMLElement.prototype.setPointerCapture = vi.fn(); + HTMLElement.prototype.releasePointerCapture = vi.fn(); +}); + +afterEach(() => { + cleanup(); + useProjectStore.getState().clear(); +}); + +describe("PreviewCanvas layout places", () => { + it("a selected layout section shows a handle per pip place", () => { + const regions = [region("screen-pip", [0, 1])]; + renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: "l1" }); + expect(screen.getAllByTestId("layout-place")).toHaveLength(2); + expect(screen.getAllByTestId("layout-place-handle")).toHaveLength(2); + }); + + it("gives a frame-filling place no handle", () => { + const regions = [region("camera-full-pip", [1, 0])]; + renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: "l1" }); + expect(screen.getAllByTestId("layout-place-handle")).toHaveLength(1); + }); + + it("shows nothing when the section is not selected or the playhead is outside it", () => { + const regions = [region("screen-pip", [0, 1])]; + renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: null }); + expect(screen.queryByTestId("layout-place")).toBeNull(); + cleanup(); + renderCanvas({ + cameraLayoutRegions: [{ ...regions[0], endMs: 2000 }], + selectedLayoutRegionId: "l1", + }); + expect(screen.queryByTestId("layout-place")).toBeNull(); + }); + + it("moves a place live and commits once on release", () => { + const regions = [region("camera-full-pip", [1, 0])]; + const onLive = vi.fn>(); + const props = renderCanvas({ + cameraLayoutRegions: regions, + selectedLayoutRegionId: "l1", + onLayoutSlotRectLive: onLive, + }); + const place = screen.getByTestId("layout-place"); + fireEvent.pointerDown(place, { pointerId: 1, clientX: 500, clientY: 250 }); + fireEvent.pointerMove(place, { pointerId: 1, clientX: 400, clientY: 200 }); + fireEvent.pointerMove(place, { pointerId: 1, clientX: 300, clientY: 150 }); + fireEvent.pointerUp(place, { pointerId: 1 }); + expect(onLive).toHaveBeenCalledTimes(2); + const [[id, slotIndex, first], [, , last]] = onLive.mock.calls; + // Slot 1 is the PiP of camera-full-pip (slot 0 fills the frame). + expect([id, slotIndex]).toEqual(["l1", 1]); + // Both moves are measured from the grab: the second one went twice as far. + expect(first.x - last.x).toBeCloseTo(0.1); + expect(first.y - last.y).toBeCloseTo(0.1); + expect(last.width).toBe(first.width); + expect(props.onLayoutSlotRectCommit).toHaveBeenCalledTimes(1); + }); + + it("leaves no undo step for a click without a move", () => { + const regions = [region("screen-pip", [0])]; + const props = renderCanvas({ cameraLayoutRegions: regions, selectedLayoutRegionId: "l1" }); + const place = screen.getByTestId("layout-place"); + fireEvent.pointerDown(place, { pointerId: 1, clientX: 500, clientY: 250 }); + fireEvent.pointerUp(place, { pointerId: 1 }); + expect(props.onLayoutSlotRectCommit).not.toHaveBeenCalled(); + }); +}); diff --git a/src/components/ai-edition/PreviewCanvas.tsx b/src/components/ai-edition/PreviewCanvas.tsx index c817fde3c..8ed2c57b0 100644 --- a/src/components/ai-edition/PreviewCanvas.tsx +++ b/src/components/ai-edition/PreviewCanvas.tsx @@ -28,6 +28,7 @@ import { type CameraFullscreenRegion, type CropRegion, DEFAULT_CROP_REGION, + type NormalizedRect, type WebcamLayoutPreset, type WebcamMaskShape, type ZoomFocus, @@ -47,9 +48,23 @@ import type { } from "@/lib/ai-edition/schema"; import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; import { useEditorSettings } from "@/lib/ai-edition/store/useEditorSettings"; -import { resolveActiveCameraTrack } from "@/lib/ai-edition/timeline/camera"; +import { + assetAdditionalCameraSources, + assetCameraSource, + resolveActiveCameraTrack, +} from "@/lib/ai-edition/timeline/camera"; +import { + inwardCorner, + moveSlotRect, + type PipPlace, + pipPlacesOf, + resizeSlotRect, + type SlotCorner, +} from "@/lib/ai-edition/timeline/layoutSlotDrag"; import type { SpeedRegion } from "@/lib/ai-edition/timeline/speed"; +import { resolvePillIds } from "@/lib/ai-edition/timeline/timelineMap"; import { locateVirtualPosition } from "@/lib/ai-edition/timeline/virtual-preview"; +import { type AnchoredCameraLayoutRegion, normalizeCameraSettings } from "@/lib/cameraLayouts"; import { computeCameraFullscreenRect, computeCompositeLayout, @@ -61,7 +76,12 @@ import { webcamAnchorAt } from "@/lib/projectDefaults"; import { wallpaperStyle } from "@/lib/wallpaper"; import { getCssClipPath } from "@/lib/webcamMaskShapes"; import { computeCameraFullscreenProgress } from "@/lib/zoomMath/cameraFullscreenUtils"; -import { webcamBoxSourceSize } from "@/native/sceneDescription"; +import { + camera0PerspectiveOf, + cameraLayoutContextOf, + clipPipLayoutOf, + webcamBoxSourceSize, +} from "@/native/sceneDescription"; import { getWebcamNativeSize, getWebcamNativeSizeRevision, @@ -92,6 +112,13 @@ interface PreviewCanvasProps { selectedZoomRegionId?: string | null; onZoomFocusChange?: (id: string, focus: ZoomFocus) => void; onZoomFocusCommit?: () => void; + /** Layout sections; the selected one's PiP places get move/resize hitboxes. */ + cameraLayoutRegions?: AnchoredCameraLayoutRegion[]; + selectedLayoutRegionId?: string | null; + /** Live edit of a place's rect during a gesture (no undo step). */ + onLayoutSlotRectLive?: (id: string, slotIndex: number, rect: NormalizedRect) => void; + /** End of the gesture: one undo step. */ + onLayoutSlotRectCommit?: () => void; annotationRegions?: AxcutAnnotationRegion[]; selectedAnnotationId?: string | null; onSelectAnnotation?: (id: string) => void; @@ -243,6 +270,8 @@ export function PreviewCanvas(props: PreviewCanvasProps) { getWebcamNativeSizeRevision, () => 0, ); + // A perspective on camera 1 gives the box its corrected ratio, as in the scene. + const camera0Perspective = useMemo(() => camera0PerspectiveOf(document), [document]); // biome-ignore lint/correctness/useExhaustiveDependencies: the revision re-reads the probed-size cache const webcamSourceSize = useMemo( () => @@ -250,8 +279,9 @@ export function PreviewCanvas(props: PreviewCanvasProps) { activeCameraTrack, activeCameraTrack?.sourcePath ? getWebcamNativeSize(activeCameraTrack.sourcePath) : null, settings.webcamCropRegion, + camera0Perspective, ), - [activeCameraTrack, settings.webcamCropRegion, webcamSizeRevision], + [activeCameraTrack, settings.webcamCropRegion, camera0Perspective, webcamSizeRevision], ); const formatFill = useMemo(() => (document ? isFormatFillActive(document) : false), [document]); @@ -351,6 +381,60 @@ export function PreviewCanvas(props: PreviewCanvasProps) { // camera-less clip, so this is belt-and-braces rather than the only guard. const showWebcamSlot = Boolean(layout?.webcamRect && activeClipHasCamera); const [isPlaying, setIsPlaying] = useState(false); + + // The selected layout section's row under the playhead: one pill can span several clip- + // anchored rows, and the one under the playhead names the asset whose cameras are drawn. + const layoutRow = useMemo(() => { + const id = props.selectedLayoutRegionId; + const regions = props.cameraLayoutRegions; + if (!id || !regions) return null; + const pill = new Set(resolvePillIds(regions, id)); + const nowMs = props.currentTimeSec * 1000; + return regions.find((r) => pill.has(r.id) && nowMs >= r.startMs && nowMs < r.endMs) ?? null; + }, [props.selectedLayoutRegionId, props.cameraLayoutRegions, props.currentTimeSec]); + // Its PiP places, through the context the scene builds, so each hitbox sits on the window + // the compositor draws. Frame-filling places get none. + // biome-ignore lint/correctness/useExhaustiveDependencies: the revision re-reads the probed-size cache + const layoutPlaces = useMemo((): PipPlace[] => { + if (!layoutRow || frameSize.width <= 0 || frameSize.height <= 0) return []; + const asset = assets.find((a) => a.id === (layoutRow.assetId ?? activeClip?.assetId)); + if (!asset) return []; + const sources = [assetCameraSource(asset), ...assetAdditionalCameraSources(asset)]; + const legacy = document?.legacyEditor as Record | null | undefined; + const camera0Path = asset.cameraTrack?.sourcePath; + const maskShape = settings.webcamMaskShape as WebcamMaskShape; + const ctx = cameraLayoutContextOf({ + frame: frameSize, + asset, + cameraSettings: normalizeCameraSettings(legacy?.cameraSettings), + webcamCropRegion: settings.webcamCropRegion, + probedCamera0Size: camera0Path ? getWebcamNativeSize(camera0Path) : null, + clipLayout: clipPipLayoutOf(layout, frameSize, maskShape), + webcamMaskShape: maskShape, + webcamRoundness: settings.webcamRoundness, + pipPreset: + resolveWebcamLayoutPreset( + settings.webcamLayoutPreset as WebcamLayoutPreset, + activeClipHasCamera, + ) === "picture-in-picture", + }); + return pipPlacesOf(layoutRow, ctx).filter((p) => (sources[p.camera]?.path ?? "") !== ""); + }, [ + layoutRow, + frameSize, + assets, + activeClip, + document, + layout, + activeClipHasCamera, + settings.webcamCropRegion, + settings.webcamMaskShape, + settings.webcamRoundness, + settings.webcamLayoutPreset, + webcamSizeRevision, + ]); + const editsLayoutPlaces = + layoutPlaces.length > 0 && !isPlaying && props.onLayoutSlotRectLive !== undefined; const handleVideoElement = useMemo(() => props.onVideoElement, [props.onVideoElement]); // L'élément `
) : null} + {editsLayoutPlaces + ? layoutPlaces.map((place) => { + const corner = inwardCorner(place.rect); + return ( +
handleLayoutPlacePointerDown(e, place, "move")} + > +
handleLayoutPlacePointerDown(e, place, corner)} + /> +
+ ); + }) + : null} {/* Last, so a selected annotation over the camera takes the pointer before the camera's drag hitbox does. It spans the frame: text, images and arrows move anywhere in it, over the padding too. */} @@ -583,3 +738,22 @@ function buildWebcamStyle( }; return clipPath ? { ...base, clipPath } : base; } + +// A layout place's hitbox: its rect in percent of the frame. No clip-path, unlike the webcam +// slot: the resize handle sits on the rect's corner, outside a circle's disc. +function layoutPlaceStyle(rect: NormalizedRect): React.CSSProperties { + return { + left: `${rect.x * 100}%`, + top: `${rect.y * 100}%`, + width: `${rect.width * 100}%`, + height: `${rect.height * 100}%`, + }; +} + +function layoutHandleStyle(corner: SlotCorner): React.CSSProperties { + return { + left: corner === "nw" || corner === "sw" ? 0 : "100%", + top: corner === "nw" || corner === "ne" ? 0 : "100%", + cursor: corner === "nw" || corner === "se" ? "nwse-resize" : "nesw-resize", + }; +} diff --git a/src/components/ai-edition/RightPanes.tsx b/src/components/ai-edition/RightPanes.tsx index fe1f00bf6..616529940 100644 --- a/src/components/ai-edition/RightPanes.tsx +++ b/src/components/ai-edition/RightPanes.tsx @@ -149,6 +149,7 @@ import { ROUNDNESS_SLIDER_MAX_PX } from "@/native/paramUnits"; import { wallpaperAcceptsMotion } from "@/native/sceneDescription"; import { ASPECT_RATIO_PRESETS, type AspectRatio } from "@/utils/aspectRatioUtils"; import { useCanSegmentCamera } from "../../native/hooks/useSegmentationSupport"; +import { CamerasSection, type CamerasSectionProps } from "./CamerasSection"; import { CaptionsPane } from "./CaptionsPane"; import { ColorField } from "./ColorField"; import { insertionsEnabled } from "./insertionsEnabled"; @@ -2779,7 +2780,12 @@ const CAMERA_BACKGROUND_MODES: Array<{ }, ]; -export function LayoutPane() { +export function LayoutPane({ + cameras, +}: { + /** The per-camera list; absent where the pane has no timeline store at hand. */ + cameras?: Pick; +} = {}) { const canSegmentCamera = useCanSegmentCamera(); const ts = useScopedT("settings"); const { settings, set, setLive, commit, hasDocument } = useEditorSettings(); @@ -3114,6 +3120,7 @@ export function LayoutPane() { onFrameLive={setCropFrame} onCommit={() => void commit()} /> + {cameras ? : null} ); } diff --git a/src/components/ai-edition/v4/FloatingInspector.test.tsx b/src/components/ai-edition/v4/FloatingInspector.test.tsx index 275b93cbc..5e85d7283 100644 --- a/src/components/ai-edition/v4/FloatingInspector.test.tsx +++ b/src/components/ai-edition/v4/FloatingInspector.test.tsx @@ -38,7 +38,16 @@ vi.mock("../RightPanes", async (importOriginal) => ({
), CursorPane: () =>
CursorPane
, - LayoutPane: () =>
LayoutPane
, + // Hands its `cameras` writer out, so a test can tell which one the inspector passed down. + LayoutPane: ({ cameras }: { cameras?: { setCameraSettings?: unknown } }) => ( + + ), SliderCell: () =>
SliderCell
, TranscriptPane: () =>
TranscriptPane
, VideoEffectsPane: () =>
VideoEffectsPane
, @@ -84,6 +93,7 @@ describe("FloatingInspector", () => { onToggleOpen: vi.fn(), clips: [], onEditClip: vi.fn(), + setCameraSettings: vi.fn(), transcriptProps: {} as unknown as React.ComponentProps< typeof FloatingInspector >["transcriptProps"], @@ -95,6 +105,18 @@ describe("FloatingInspector", () => { } as unknown as React.ComponentProps["tl"], }; + it("the layout pane writes camera settings through the writer it was given", () => { + const setCameraSettings = vi.fn(); + const tl = { + ...defaultProps.tl, + setCameraSettings: vi.fn(), + } as unknown as React.ComponentProps["tl"]; + render(); + fireEvent.click(screen.getByTestId("layout-pane")); + expect(setCameraSettings).toHaveBeenCalledWith(1, null); + expect(tl.setCameraSettings).not.toHaveBeenCalled(); + }); + it("renders layout facet button on rail with camera icon and settings.layout.title", () => { render(); const layoutBtn = screen.getByRole("button", { name: "settings.layout.title" }); @@ -536,6 +558,31 @@ describe("FloatingInspector", () => { return { tl, updateCameraFullscreenOrientation, updateCameraFullscreenDeskLabel }; }; + it("the full camera pane offers the template choice", async () => { + const { tl } = camTl({}); + const setLayoutTemplate = vi.fn(async () => ({ kind: "cameraLayout" as const, id: "L9" })); + const selectRegion = vi.fn(); + Object.assign(tl, { setLayoutTemplate, selectRegion }); + render(); + expect( + screen.getByRole("group", { name: "settings.cameraLayout.template" }), + ).toBeInTheDocument(); + const current = screen.getByRole("button", { name: /timeline.labels.layoutCameraFull$/ }); + expect(current).toHaveAttribute("aria-pressed", "true"); + // Only one camera is known here, so the two-camera templates are off. + expect( + screen.getByRole("button", { name: /timeline.labels.layoutSideBySide/ }), + ).toBeDisabled(); + fireEvent.click(screen.getByRole("button", { name: /timeline.labels.layoutScreenPip/ })); + expect(setLayoutTemplate).toHaveBeenCalledWith( + { kind: "cameraFullscreen", id: "cf" }, + "screen-pip", + [], + ); + // The section moved to the layout list: the selection follows it. + await waitFor(() => expect(selectRegion).toHaveBeenCalledWith("cameraLayout", "L9")); + }); + it("desk view sets both fields in one call", () => { const { tl, updateCameraFullscreenOrientation } = camTl({}); render(); diff --git a/src/components/ai-edition/v4/FloatingInspector.tsx b/src/components/ai-edition/v4/FloatingInspector.tsx index db594f517..f77bfb0d4 100644 --- a/src/components/ai-edition/v4/FloatingInspector.tsx +++ b/src/components/ai-edition/v4/FloatingInspector.tsx @@ -23,7 +23,6 @@ import { Trash2, Type, Undo2, - X, ZoomIn, } from "lucide-react"; import type { ComponentProps } from "react"; @@ -59,6 +58,7 @@ import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; import { rafCoalesce } from "@/lib/ai-edition/store/rafCoalesce"; import { useEditorSettings } from "@/lib/ai-edition/store/useEditorSettings"; import type { useTimeline } from "@/lib/ai-edition/store/useTimeline"; +import { camerasOfSection } from "@/lib/ai-edition/timeline/cameraList"; import { formatSeconds } from "@/lib/ai-edition/timeline/format"; import { coalescedTrimGroups } from "@/lib/ai-edition/timeline/trim-mapping"; import { @@ -69,8 +69,10 @@ import { normalizeCameraRotation, showsDeskLabel, } from "@/lib/cameraOrientation"; +import { isWebcamBlockLayout } from "@/lib/compositeLayout"; import { clampToBound } from "@/lib/projectDefaults"; import { annotationFootageRect, zoomScaleLimit } from "@/native/sceneDescription"; +import type { CamerasSectionProps } from "../CamerasSection"; import { ColorField } from "../ColorField"; import shell from "../NewEditorShell.module.css"; import { @@ -87,6 +89,8 @@ import { import { useHasRecordedCursor } from "../recordedCursorTypes"; import { TextColorField } from "../TextColorField"; import styles from "./EditorShellV4.module.css"; +import { LayoutSectionPane, LayoutTemplateChoice } from "./LayoutSectionPane"; +import { PANE_BODY_STYLE, PANE_BUTTON, paneHeader, paneRow, paneStack } from "./paneParts"; type TimelineApi = ReturnType; @@ -130,6 +134,10 @@ interface FloatingInspectorProps { * selected. Clicking elsewhere on the timeline clears the selection * (see V4Timeline's empty-area click handler) which closes this pane. */ tl: TimelineApi; + /** Opens the camera calibration dialog (the shell holds its one instance). */ + onOpenCalibration?: CamerasSectionProps["onOpenCalibration"]; + /** Per-camera settings writer; the shell's queued one, so it cannot race other writes. */ + setCameraSettings: CamerasSectionProps["setCameraSettings"]; } export function FloatingInspector({ @@ -141,6 +149,8 @@ export function FloatingInspector({ onEditClip, transcriptProps, tl, + onOpenCalibration, + setCameraSettings, }: FloatingInspectorProps) { const ts = useScopedT("settings"); const te = useScopedT("editor"); @@ -185,7 +195,14 @@ export function FloatingInspector({ ) : audioTrackSelected ? ( tl.clearSelection()} /> ) : ( - + )}
) : null} @@ -307,80 +324,6 @@ export function FloatingInspector({ ); } -function paneHeader(icon: React.ReactNode, title: string, onClose: () => void, closeLabel: string) { - return ( -
- {icon} -

- {title} -

- -
- ); -} - -function paneRow(label: string, control: React.ReactNode) { - return ( -
- {label} - {control} -
- ); -} - -/** Un libellé au-dessus de son contrôle, pour ceux qui prennent toute la largeur du panneau - * (une `ChoiceRow`) : à côté d'un libellé, ils n'auraient plus la place de montrer leurs choix. - * `value` nomme le choix courant quand les boutons ne font que le dessiner. */ -function paneStack(label: string, control: React.ReactNode, value?: string) { - return ( -
- - {label} - {value ? {value} : null} - - {control} -
- ); -} - /** What each camera does to the screen (viewBox 0 0 32 22, as the camera layout tiles). A fixed * angle is the outline of its pose: `ROTATION_3D_PRESETS` projected with the shipped * perspective and centred. The moving camera leaves the screen flat and circles it. */ @@ -812,23 +755,7 @@ function SelectionPane({ tl, onClose }: { tl: TimelineApi; onClose: () => void } onClose(); }; - // Le panneau découpe son contenu (coins arrondis + flou), donc un corps sans ascenseur perd - // silencieusement ce qui dépasse — c'est ce qui arrivait au pane d'annotation, le plus haut de - // tous, dès qu'on réduisait la fenêtre. L'en-tête reste fixe, le corps défile, comme les - // panneaux de facette (cf. `.paneBody` de NewEditorShell). - const bodyStyle: React.CSSProperties = { - padding: "16px", - display: "flex", - flexDirection: "column", - gap: 16, - flex: "1 1 auto", - minHeight: 0, - overflowY: "auto", - overflowX: "hidden", - overscrollBehavior: "contain", - scrollbarWidth: "thin", - scrollbarColor: "var(--border) transparent", - }; + const bodyStyle = PANE_BODY_STYLE; if (selection.kind === "zoom") { const region = tl.zoomRegions.find((z) => z.id === selection.id); @@ -1274,12 +1201,34 @@ function SelectionPane({ tl, onClose }: { tl: TimelineApi; onClose: () => void } ); } + // The cameras of the asset a camera section is anchored to (its row's `assetId`). + const camerasUnder = (region: { assetId?: string; startMs: number; endMs: number }) => + doc ? camerasOfSection(doc, region, ts) : []; + const blockPreset = isWebcamBlockLayout(settings.webcamLayoutPreset); + + if (selection.kind === "cameraLayout") { + const region = tl.cameraLayoutRegions.find((r) => r.id === selection.id); + if (!region) return null; + return ( + + ); + } + if (selection.kind === "cameraFullscreen") { const region = tl.cameraFullscreenRegions.find((c) => c.id === selection.id); if (!region) return null; const rotation = normalizeCameraRotation(region.rotation); const mirror = normalizeCameraMirror(region.mirror); const desk = isDeskView(region); + const available = camerasUnder(region) + .filter((c) => c.available) + .map((c) => c.index); const setOrientation = (next: { rotation: CameraRotation; mirror: CameraMirrorMode }) => void tl.updateCameraFullscreenOrientation(region.id, next); return ( @@ -1291,6 +1240,19 @@ function SelectionPane({ tl, onClose }: { tl: TimelineApi; onClose: () => void } tc("actions.close"), )}
+ + void tl + .setLayoutTemplate({ kind: "cameraFullscreen", id: region.id }, template, available) + .then((next) => { + if (next.kind !== "cameraFullscreen" || next.id !== region.id) + tl.selectRegion(next.kind, next.id); + }) + } + /> {/* One click for the common case: a camera tilted onto the desk is upside down and must not be mirrored, or the papers' text reads back to front. */} ); - if (facet === "layout") return wrap(collapse, ); + if (facet === "layout") + return wrap( + collapse, + , + ); if (facet === "audio") return wrap(collapse, ); if (facet === "cursor") return wrap(collapse, ); if (facet === "transcript") return wrap(collapse, ); diff --git a/src/components/ai-edition/v4/LayoutSectionPane.test.tsx b/src/components/ai-edition/v4/LayoutSectionPane.test.tsx new file mode 100644 index 000000000..9aea690bb --- /dev/null +++ b/src/components/ai-edition/v4/LayoutSectionPane.test.tsx @@ -0,0 +1,129 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { fireEvent, render, screen, waitFor } from "@testing-library/react"; +import { describe, expect, it, vi } from "vitest"; +import type { CameraLayoutRegion } from "@/components/video-editor/types"; +import type { CameraSectionHandle } from "@/lib/ai-edition/store/useTimeline"; +import type { ProjectCamera } from "@/lib/ai-edition/timeline/cameraList"; + +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: (scope: string) => (key: string) => `${scope}.${key}`, +})); + +vi.mock("../RightPanes", async (importOriginal) => ({ + ChoiceRow: (await importOriginal()).ChoiceRow, +})); + +import { LayoutSectionPane } from "./LayoutSectionPane"; + +const camera = (index: number, available = true): ProjectCamera => ({ + index, + label: `Cam ${index + 1}`, + path: available ? `/c${index}.mp4` : "", + available, +}); + +function setup(region: Partial, cameras: ProjectCamera[], blockPreset = false) { + const tl = { + setLayoutTemplate: vi.fn(async (handle: CameraSectionHandle) => handle), + setLayoutSlotCamera: vi.fn(async (handle: CameraSectionHandle) => handle), + resetLayoutSlotRects: vi.fn(async () => undefined), + removeRegion: vi.fn(async () => undefined), + selectRegion: vi.fn(), + }; + const full: CameraLayoutRegion = { + id: "L1", + startMs: 0, + endMs: 2000, + template: "side-by-side", + slots: [{ camera: 0 }, { camera: 1 }], + ...region, + }; + render( + , + ); + return tl; +} + +describe("LayoutSectionPane", () => { + it("shows one camera select per place", () => { + setup({}, [camera(0), camera(1), camera(2)]); + expect(screen.getAllByRole("combobox")).toHaveLength(2); + expect(screen.getByRole("combobox", { name: "settings.cameraLayout.placeLeft" })).toHaveValue( + "0", + ); + expect(screen.getByRole("combobox", { name: "settings.cameraLayout.placeRight" })).toHaveValue( + "1", + ); + }); + + it("choosing a template calls setLayoutTemplate", async () => { + const tl = setup({}, [camera(0), camera(1), camera(2)]); + fireEvent.click(screen.getByRole("button", { name: /timeline\.labels\.layoutCameraFullPip/ })); + expect(tl.setLayoutTemplate).toHaveBeenCalledWith( + { kind: "cameraLayout", id: "L1" }, + "camera-full-pip", + [0, 1, 2], + ); + // The returned handle is the same section: the selection stays. + await waitFor(() => expect(tl.setLayoutTemplate).toHaveBeenCalledTimes(1)); + expect(tl.selectRegion).not.toHaveBeenCalled(); + }); + + it("moves the selection to the handle a change returns", async () => { + const tl = setup({}, [camera(0), camera(1)]); + tl.setLayoutSlotCamera.mockResolvedValueOnce({ kind: "cameraFullscreen", id: "F1" }); + fireEvent.change(screen.getByRole("combobox", { name: "settings.cameraLayout.placeLeft" }), { + target: { value: "1" }, + }); + expect(tl.setLayoutSlotCamera).toHaveBeenCalledWith({ kind: "cameraLayout", id: "L1" }, 0, 1); + await waitFor(() => expect(tl.selectRegion).toHaveBeenCalledWith("cameraFullscreen", "F1")); + }); + + it("a slot whose camera is gone shows as unavailable", () => { + setup({ slots: [{ camera: 0 }, { camera: 1 }] }, [camera(0), camera(1, false)]); + const select = screen.getByRole("combobox", { name: "settings.cameraLayout.placeRight" }); + expect(select).toHaveValue("1"); + expect( + screen.getAllByRole("option", { name: "settings.cameraLayout.cameraUnavailable" }), + ).not.toHaveLength(0); + }); + + it("side-by-side is disabled with one camera", () => { + setup({ template: "camera-full", slots: [{ camera: 0 }] }, [camera(0)]); + const button = screen.getByRole("button", { name: /timeline\.labels\.layoutSideBySide/ }); + expect(button).toBeDisabled(); + expect(screen.getByText("timeline.layoutMenu.needsCamerasHint")).toBeInTheDocument(); + }); + + it("pip templates are disabled under dual-frame", () => { + setup({}, [camera(0), camera(1)], true); + expect( + screen.getByRole("button", { name: /timeline\.labels\.layoutScreenPip/ }), + ).toBeDisabled(); + expect( + screen.getByRole("button", { name: /timeline\.labels\.layoutCameraFullPip/ }), + ).toBeDisabled(); + expect( + screen.getByRole("button", { name: /timeline\.labels\.layoutCameraFull$/ }), + ).toBeEnabled(); + expect(screen.getByText("timeline.layoutMenu.blockLayoutHint")).toBeInTheDocument(); + }); + + it("resets the windows only when a place has its own rect", () => { + const tl = setup( + { slots: [{ camera: 0 }, { camera: 1, rect: { x: 0, y: 0, width: 0.3, height: 0.3 } }] }, + [camera(0), camera(1)], + ); + const reset = screen.getByRole("button", { name: "settings.cameraLayout.resetWindows" }); + expect(reset).toBeEnabled(); + fireEvent.click(reset); + expect(tl.resetLayoutSlotRects).toHaveBeenCalledWith("L1"); + }); +}); diff --git a/src/components/ai-edition/v4/LayoutSectionPane.tsx b/src/components/ai-edition/v4/LayoutSectionPane.tsx new file mode 100644 index 000000000..2b5126733 --- /dev/null +++ b/src/components/ai-edition/v4/LayoutSectionPane.tsx @@ -0,0 +1,205 @@ +// The inspector pane of a layout section: its template, one camera per place, and the +// section's own actions. The Full Camera pane reuses `LayoutTemplateChoice`. + +import { RotateCcw, Trash2 } from "lucide-react"; +import { useId } from "react"; +import type { CameraLayoutRegion, CameraLayoutTemplate } from "@/components/video-editor/types"; +import { useScopedT } from "@/contexts/I18nContext"; +import type { useTimeline } from "@/lib/ai-edition/store/useTimeline"; +import type { ProjectCamera } from "@/lib/ai-edition/timeline/cameraList"; +import { + LAYOUT_TEMPLATES, + type LayoutTemplateBlock, + layoutTemplateBlock, +} from "@/lib/ai-edition/timeline/layoutMenu"; +import shell from "../NewEditorShell.module.css"; +import { ChoiceRow } from "../RightPanes"; +import { layoutTemplateIcon, layoutTemplateLabel } from "./layoutTemplateUi"; +import { PANE_BODY_STYLE, PANE_BUTTON, paneHeader, paneStack } from "./paneParts"; + +type TimelineApi = ReturnType; + +/** The four templates as one choice row; the ones that cannot apply are disabled, with the reason. */ +export function LayoutTemplateChoice({ + current, + cameraCount, + blockPreset, + onPick, +}: { + current: CameraLayoutTemplate; + /** How many cameras the section could show: its own plus the clip's available ones. */ + cameraCount: number; + blockPreset: boolean; + onPick: (template: CameraLayoutTemplate) => void; +}) { + const ts = useScopedT("settings"); + const tt = useScopedT("timeline"); + const hintId = useId(); + const blocks = new Map(); + for (const template of LAYOUT_TEMPLATES) { + // The template in use stays pickable: it is where the section already is. + const block = + template === current ? null : layoutTemplateBlock(template, { cameraCount, blockPreset }); + if (block) blocks.set(template, block); + } + const hintFor = (block: LayoutTemplateBlock) => + block === "block-layout" ? tt("layoutMenu.blockLayoutHint") : tt("layoutMenu.needsCamerasHint"); + const reasons = [...new Set(blocks.values())]; + return paneStack( + ts("cameraLayout.template"), + <> + + label={ts("cameraLayout.template")} + columns={2} + display="both" + describedBy={reasons.length > 0 ? hintId : undefined} + options={LAYOUT_TEMPLATES.map((template) => { + const block = blocks.get(template); + return { + value: template, + label: layoutTemplateLabel(tt, template), + icon: layoutTemplateIcon(template, 14), + disabled: block !== undefined, + title: block ? hintFor(block) : null, + }; + })} + value={current} + onChange={onPick} + /> + {reasons.length > 0 ? ( + + {reasons.map(hintFor).join(" · ")} + + ) : null} + , + ); +} + +/** "Place 1 · large": the template says which places are big and which are small. */ +function placeLabel( + ts: (key: string, vars?: Record) => string, + template: CameraLayoutTemplate, + index: number, +): string { + const n = index + 1; + if (template === "side-by-side") { + return ts(index === 0 ? "cameraLayout.placeLeft" : "cameraLayout.placeRight", { n }); + } + const large = template === "camera-full" || (template === "camera-full-pip" && index === 0); + return ts(large ? "cameraLayout.placeLarge" : "cameraLayout.placeSmall", { n }); +} + +export function LayoutSectionPane({ + tl, + region, + cameras, + blockPreset, + onClose, +}: { + tl: Pick< + TimelineApi, + | "setLayoutTemplate" + | "setLayoutSlotCamera" + | "resetLayoutSlotRects" + | "removeRegion" + | "selectRegion" + >; + region: CameraLayoutRegion; + /** The cameras of the clip the section sits on. */ + cameras: ProjectCamera[]; + blockPreset: boolean; + onClose: () => void; +}) { + const ts = useScopedT("settings"); + const tt = useScopedT("timeline"); + const tc = useScopedT("common"); + const te = useScopedT("editor"); + const handle = { kind: "cameraLayout" as const, id: region.id }; + const own = region.slots.map((slot) => slot.camera); + const available = cameras.filter((c) => c.available).map((c) => c.index); + const cameraCount = new Set([...own, ...available]).size; + const followHandle = (next: { kind: "cameraFullscreen" | "cameraLayout"; id: string }) => { + if (next.kind !== handle.kind || next.id !== handle.id) tl.selectRegion(next.kind, next.id); + }; + return ( +
+ {paneHeader( + layoutTemplateIcon(region.template, 16), + layoutTemplateLabel(tt, region.template), + onClose, + tc("actions.close"), + )} +
+ + void tl.setLayoutTemplate(handle, template, available).then(followHandle) + } + /> + {region.slots.map((slot, index) => { + const label = placeLabel(ts, region.template, index); + const known = cameras.some((c) => c.index === slot.camera); + return ( + // A place has no id of its own; its position is its identity. +
+ {paneStack( + label, + , + )} +
+ ); + })} + + +
+
+ ); +} diff --git a/src/components/ai-edition/v4/V4Timeline.geometry.test.tsx b/src/components/ai-edition/v4/V4Timeline.geometry.test.tsx index c3f98ca09..5a2af39aa 100644 --- a/src/components/ai-edition/v4/V4Timeline.geometry.test.tsx +++ b/src/components/ai-edition/v4/V4Timeline.geometry.test.tsx @@ -119,6 +119,7 @@ function renderTimeline( annotationRegions: [annotation], speedRegions: [], cameraFullscreenRegions: [], + cameraLayoutRegions: [], zoomRegions: [], trimRanges: [], hasEditRegions: true, @@ -454,7 +455,7 @@ describe("V4Timeline create-from-toolbar", () => { expect(tl.clearTimeline).toHaveBeenCalledTimes(1); }); - it("puts Clear timeline last, behind a divider, after the Add Full Camera button", () => { + it("puts Clear timeline last, behind a divider, after the Add Full Camera and Add layout buttons", () => { renderTimeline(undefined, undefined, [CAMERA_ASSET]); const toolbar = toolbarOf(); const buttons = Array.from(toolbar.querySelectorAll("button")); @@ -463,7 +464,8 @@ describe("V4Timeline create-from-toolbar", () => { const divider = clear.previousElementSibling; expect(divider?.className).toContain("tlToolSep"); - expect(divider?.previousElementSibling).toBe( + expect(divider?.previousElementSibling).toBe(screen.getByLabelText("buttons.addLayout")); + expect(divider?.previousElementSibling?.previousElementSibling).toBe( screen.getByLabelText("buttons.addCameraFullscreen"), ); expect(dividersIn(toolbar)).toHaveLength(2); @@ -658,6 +660,7 @@ describe("V4Timeline audio lane drag", () => { annotationRegions: [], speedRegions: [], cameraFullscreenRegions: [], + cameraLayoutRegions: [], zoomRegions: [], trimRanges: [], selection: null, diff --git a/src/components/ai-edition/v4/V4Timeline.layout.test.tsx b/src/components/ai-edition/v4/V4Timeline.layout.test.tsx new file mode 100644 index 000000000..82d4ecc70 --- /dev/null +++ b/src/components/ai-edition/v4/V4Timeline.layout.test.tsx @@ -0,0 +1,221 @@ +// @vitest-environment jsdom +import "@testing-library/jest-dom"; +import { cleanup, fireEvent, render, screen, waitFor } from "@testing-library/react"; +import { afterEach, beforeAll, beforeEach, describe, expect, it, vi } from "vitest"; + +const VIEWPORT_PX = 900; +const TOTAL_SEC = 100; // 9 px per second + +// A bare key, with the variables appended, so a label built from keys stays readable. +vi.mock("@/contexts/I18nContext", () => ({ + useScopedT: () => (key: string, vars?: Record) => + vars ? `${key}:${Object.values(vars).join(",")}` : key, +})); +const toastError = vi.hoisted(() => vi.fn()); +vi.mock("sonner", () => ({ toast: { error: toastError, info: vi.fn(), success: vi.fn() } })); +vi.mock("@/hooks/useAudioPeaks", () => ({ useAudioPeaks: () => null })); + +import { ShortcutsProvider } from "@/contexts/ShortcutsContext"; +import { useProjectStore } from "@/lib/ai-edition/store/projectStore"; +import type { useTimeline } from "@/lib/ai-edition/store/useTimeline"; +import { V4Timeline } from "./V4Timeline"; + +beforeAll(() => { + globalThis.ResizeObserver = class { + observe() { + /* noop */ + } + unobserve() { + /* noop */ + } + disconnect() { + /* noop */ + } + } as unknown as typeof ResizeObserver; + Object.defineProperty(HTMLElement.prototype, "clientWidth", { + configurable: true, + get: () => VIEWPORT_PX, + }); + Object.defineProperty(HTMLElement.prototype, "getBoundingClientRect", { + configurable: true, + value: () => ({ + x: 0, + y: 0, + left: 0, + top: 0, + right: VIEWPORT_PX, + bottom: 100, + width: VIEWPORT_PX, + height: 100, + toJSON() { + /* unused */ + }, + }), + }); + vi.spyOn(HTMLCanvasElement.prototype, "getContext").mockImplementation( + () => + ({ + measureText: (text: string) => ({ width: text.length * 6 }), + }) as unknown as CanvasRenderingContext2D, + ); +}); + +beforeEach(() => { + useProjectStore.setState({ currentTimeSec: 30 }); +}); + +afterEach(() => { + cleanup(); + toastError.mockClear(); +}); + +const CLIP = { + id: "c1", + assetId: "a1", + timelineStartSec: 0, + timelineEndSec: TOTAL_SEC, + sourceStartSec: 0, + sourceEndSec: TOTAL_SEC, +}; +const track = (path: string, extra: Record = {}) => ({ + sourcePath: path, + startMs: 0, + offsetMs: 0, + visible: true, + ...extra, +}); +const ONE_CAMERA = { + id: "a1", + label: "rec", + durationSec: TOTAL_SEC, + cameraTrack: track("/tmp/cam1.webm"), +}; +const TWO_CAMERAS = { + ...ONE_CAMERA, + additionalCameraTracks: [track("/tmp/cam2.webm", { label: "Desk" })], +}; +const LAYOUT = { + id: "cl1", + startMs: 20_000, + endMs: 60_000, + template: "camera-full-pip", + slots: [{ camera: 1 }, { camera: 0 }], + assetId: "a1", +}; + +function renderTimeline( + assets: Array>, + layouts: unknown[] = [], + fullCameras: unknown[] = [], +) { + const tl = { + clips: [CLIP], + transcripts: [], + assets, + annotationRegions: [], + speedRegions: [], + cameraFullscreenRegions: fullCameras, + cameraLayoutRegions: layouts, + zoomRegions: [], + trimRanges: [], + hasEditRegions: false, + selection: null, + multiSelection: [], + clipSelection: null, + audioTracks: [], + selectedAudioTrackId: null, + selectAudioTrack: vi.fn(), + clearSelection: vi.fn(), + selectRegion: vi.fn(), + selectClip: vi.fn(), + addCameraFullscreen: vi.fn(async () => "added"), + addCameraLayout: vi.fn(async () => "added"), + updateCameraLayoutSpan: vi.fn(async () => undefined), + }; + render( + + } + setCurrentTime={vi.fn()} + playing={false} + onTogglePlay={vi.fn()} + onEditClip={vi.fn()} + onAddVoiceover={vi.fn()} + /> + , + ); + return tl; +} + +describe("V4Timeline layout lane", () => { + it("renders layout pills in the full camera lane", () => { + renderTimeline([TWO_CAMERAS], [LAYOUT], [{ id: "cf1", startMs: 70_000, endMs: 90_000 }]); + // The Full Camera pill keeps its (now translated) label next to the layout pill. + expect(screen.getByText("labels.cameraFullscreen")).toBeInTheDocument(); + expect(screen.getByText(/labels\.layoutCameraFullPip/)).toBeInTheDocument(); + }); + + it("names the pill after its template and the cameras of its places", () => { + renderTimeline([TWO_CAMERAS], [LAYOUT]); + const pill = screen.getByText( + "labels.layoutCameraFullPip · cameras.cameraNamed:2,Desk, cameras.cameraN:1", + ); + expect(pill).toBeInTheDocument(); + }); + + it("selects a layout pill as a cameraLayout region", () => { + const tl = renderTimeline([TWO_CAMERAS], [LAYOUT]); + const pill = screen.getByText(/labels\.layoutCameraFullPip/).closest("[role='button']"); + fireEvent.pointerDown(pill as Element, { clientX: 0 }); + window.dispatchEvent(new MouseEvent("pointerup", { clientX: 0 })); + expect(tl.selectRegion).toHaveBeenCalledWith("cameraLayout", "cl1", { additive: false }); + }); + + it("a layout pill drag calls updateCameraLayoutSpan", () => { + const tl = renderTimeline([TWO_CAMERAS], [LAYOUT]); + const pill = screen.getByText(/labels\.layoutCameraFullPip/).closest("[role='button']"); + fireEvent.pointerDown(pill as Element, { clientX: 0 }); + window.dispatchEvent(new MouseEvent("pointermove", { clientX: 90 })); + window.dispatchEvent(new MouseEvent("pointerup", { clientX: 90 })); + expect(tl.updateCameraLayoutSpan).toHaveBeenCalledTimes(1); + const [id, startMs, endMs] = tl.updateCameraLayoutSpan.mock.calls[0] as unknown as [ + string, + number, + number, + ]; + expect(id).toBe("cl1"); + expect(startMs).toBeCloseTo(30_000, -2); + expect(endMs - startMs).toBeCloseTo(40_000, -2); + }); + + it("the add-layout menu disables multi-camera templates for a one-camera project", async () => { + renderTimeline([ONE_CAMERA]); + fireEvent.click(screen.getByLabelText("buttons.addLayout")); + const sideBySide = await screen.findByRole("button", { name: /labels\.layoutSideBySide/ }); + expect(sideBySide).toBeDisabled(); + expect(screen.getByRole("button", { name: /labels\.layoutCameraFullPip/ })).toBeDisabled(); + expect(screen.getByRole("button", { name: /labels\.layoutScreenPip/ })).toBeEnabled(); + expect(screen.getByRole("button", { name: /labels\.layoutCameraFull$/ })).toBeEnabled(); + }); + + it("adds a camera-full-pip layout desk camera first and reports an occupied spot", async () => { + const tl = renderTimeline([TWO_CAMERAS]); + tl.addCameraLayout.mockResolvedValueOnce("occupied"); + fireEvent.click(screen.getByLabelText("buttons.addLayout")); + fireEvent.click(await screen.findByRole("button", { name: /labels\.layoutCameraFullPip/ })); + expect(tl.addCameraLayout).toHaveBeenCalledWith("camera-full-pip", [1, 0], expect.any(Number)); + await waitFor(() => + expect(toastError).toHaveBeenCalledWith( + "errors.cannotPlaceCameraSection", + expect.objectContaining({ description: "errors.cameraSectionExistsAtLocation" }), + ), + ); + }); + + it("shows a notice when the Add Full Camera button is refused", async () => { + const tl = renderTimeline([ONE_CAMERA]); + tl.addCameraFullscreen.mockResolvedValueOnce("occupied" as never); + fireEvent.click(screen.getByLabelText("buttons.addCameraFullscreen")); + await waitFor(() => expect(toastError).toHaveBeenCalled()); + }); +}); diff --git a/src/components/ai-edition/v4/V4Timeline.tsx b/src/components/ai-edition/v4/V4Timeline.tsx index f50942b3d..861d53d63 100644 --- a/src/components/ai-edition/v4/V4Timeline.tsx +++ b/src/components/ai-edition/v4/V4Timeline.tsx @@ -3,6 +3,7 @@ import { Clock, Crosshair, Eraser, + LayoutTemplate, Loader2, Maximize2, MessageSquare, @@ -31,7 +32,7 @@ import { toast } from "sonner"; import { Popover, PopoverContent, PopoverTrigger } from "@/components/ui/popover"; import { Tooltip, TooltipProvider } from "@/components/ui/tooltip"; import { toFileUrl } from "@/components/video-editor/projectPersistence"; -import { ZOOM_DEPTH_SCALES } from "@/components/video-editor/types"; +import { type CameraLayoutTemplate, ZOOM_DEPTH_SCALES } from "@/components/video-editor/types"; import { useScopedT } from "@/contexts/I18nContext"; import { useShortcuts } from "@/contexts/ShortcutsContext"; import { useAudioPeaks } from "@/hooks/useAudioPeaks"; @@ -56,7 +57,15 @@ import { useEditorSettings } from "@/lib/ai-edition/store/useEditorSettings"; import type { useTimeline } from "@/lib/ai-edition/store/useTimeline"; import { collectAutoZoomSuggestionsForLatestDocument } from "@/lib/ai-edition/timeline/apply-auto-zooms"; import { hasAnyClipWithCamera } from "@/lib/ai-edition/timeline/camera"; +import { type ProjectCamera, projectCameras } from "@/lib/ai-edition/timeline/cameraList"; +import { showCameraSectionOutcome } from "@/lib/ai-edition/timeline/cameraSectionNotice"; import { formatSec } from "@/lib/ai-edition/timeline/format"; +import { + camerasForLayoutMenu, + defaultLayoutCameras, + LAYOUT_TEMPLATES, + layoutTemplateBlock, +} from "@/lib/ai-edition/timeline/layoutMenu"; import { newRegionDurationSec, setTimelineScale, @@ -68,12 +77,15 @@ import { resolveTimelineSpanToTrim, ventilateTimelineSpanToTrims, } from "@/lib/ai-edition/timeline/trim-mapping"; +import type { AnchoredCameraLayoutRegion } from "@/lib/cameraLayouts"; import { normalizeCameraRotation } from "@/lib/cameraOrientation"; +import { isWebcamBlockLayout } from "@/lib/compositeLayout"; import { formatBinding } from "@/lib/shortcuts"; import { nativeBridgeClient } from "@/native/client"; import { TransportBar } from "../TransportBar"; import type { VideoSource } from "../VirtualPreview"; import styles from "./EditorShellV4.module.css"; +import { layoutTemplateIcon, layoutTemplateLabel } from "./layoutTemplateUi"; // The AI option's prompt — sent straight to the chat agent via the prompt-bus. // @@ -592,7 +604,7 @@ const AudioLanePill = memo(function AudioLanePill({ interface LanePill { id: string; - kind: "annotation" | "speed" | "trim" | "zoom" | "cameraFullscreen"; + kind: "annotation" | "speed" | "trim" | "zoom" | "cameraFullscreen" | "cameraLayout"; start: number; end: number; label: string; @@ -600,6 +612,8 @@ interface LanePill { sourceIds: string[]; /** Desk-view section: the camera is turned 180 degrees. */ rotated?: boolean; + /** Layout pills: which template, for the icon. */ + template?: CameraLayoutTemplate; } export function V4Timeline({ @@ -671,6 +685,32 @@ export function V4Timeline({ } | null>(null); const { settings, set: setSettings } = useEditorSettings(); + // The "Add layout" menu. The cameras of the clip under the playhead (else the nearest with some) are read when the + // menu opens and again on a pick, so no playback frame re-renders the timeline for it. + const [layoutMenuOpen, setLayoutMenuOpen] = useState(false); + const [layoutMenuCameras, setLayoutMenuCameras] = useState([]); + const availableCamerasAtPlayhead = useCallback((): ProjectCamera[] => { + return camerasForLayoutMenu(tl.clips, tl.assets, useProjectStore.getState().currentTimeSec, ts); + }, [tl.clips, tl.assets, ts]); + const openLayoutMenu = useCallback( + (open: boolean) => { + if (open) setLayoutMenuCameras(availableCamerasAtPlayhead()); + setLayoutMenuOpen(open); + }, + [availableCamerasAtPlayhead], + ); + const templateLabel = (template: CameraLayoutTemplate): string => + layoutTemplateLabel(t, template); + const addLayout = async (template: CameraLayoutTemplate) => { + setLayoutMenuOpen(false); + const cameras = defaultLayoutCameras( + template, + availableCamerasAtPlayhead().map((c) => c.index), + ); + const outcome = await tl.addCameraLayout(template, cameras, newRegionDurationSec()); + showCameraSectionOutcome(outcome, t); + }; + const [autoEnhanceOpen, setAutoEnhanceOpen] = useState(false); const [audioMenuOpen, setAudioMenuOpen] = useState(false); const [autoBusy, setAutoBusy] = useState(false); @@ -760,11 +800,35 @@ export function V4Timeline({ kind: "cameraFullscreen", start: p.start, end: p.end, - label: "Full Camera", + label: t("labels.cameraFullscreen"), sourceIds: p.ids, rotated: normalizeCameraRotation(p.member.rotation) === 180, }), ); + // Layout sections: "