From b137992d90b46d1056fb9122cfe1a1bd6a5d353b Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:22:39 +0200 Subject: [PATCH 1/8] feat(recording): capture the webcam at its real resolution and frame rate A Logitech BRIO was recorded at 640x480 and 4 Mbit/s, then upscaled into the picture-in-picture, which is what "the camera looks pixelated" actually was. Nothing in the stack ever asked for a resolution. `getUserMedia({ video: { deviceId } })` and a Media Foundation source reader handed a type with no MF_MT_FRAME_SIZE both fall back to the device's DEFAULT media type, and for a UVC camera that is the first format it enumerates. On a BRIO offering MJPEG up to 3840x2160 that default is 640x480. The DirectShow fallback had the same hole from the other side: RenderStream's intelligent connect takes the capture pin's default because nothing ever touched IAMStreamConfig. Measured with ffmpeg on the same device: unconstrained gives `rawvideo (YUY2), 640x480`, while `-video_size 1920x1080` gives 1080p. Naming the size on the reader's OUTPUT type is not enough on its own. With MF_SOURCE_READER_ENABLE_VIDEO_PROCESSING the reader answers S_OK by inserting a converter in front of whichever native type is already selected, so the camera keeps running at its default and the size asked for is quietly dropped. Selecting the native type first is what reconfigures the camera. `chooseWebcamFormat` decides which of the advertised modes to drive, shared by both backends and unit-tested against the BRIO's real capability list: only modes fitting inside the target in both dimensions, most pixels wins, ties go to the lowest frame rate that still reaches the target, and a camera whose every mode is oversized scales down from its smallest rather than pushing the largest through the encoder. The capture pixel format is now NV12 rather than RGB32 wherever the frame only has to reach the webcam encoder. RGB32 made the source reader decode AND convert every frame: instrumenting the capture loop put ReadSample at 92ms per 3840x2160 frame -- a ceiling near 11 fps -- against 32ms for NV12. The file still claimed 30 fps because the encoder padded the gap with duplicates, so the loss was invisible from the outside. The encoder wanted NV12 anyway and was converting RGB32 back to it, so the old path converted twice to arrive where it started. BGRA is kept for the inline composite, which needs that layout. Capture resolution becomes a user setting (1080p / 1440p / 2160p), persisted beside the camera device and validated at the write boundary. The encoder bitrate ladder gains 1080p, 1440p and 2160p tiers; the old one stopped at 8 Mbit/s for anything 720p or larger, which starved the frames the higher resolutions now produce. Measured on a BRIO (Snapdragon X, alongside a screen capture), 8s takes, frames the camera actually delivered out of 240, four runs each: 1080p 244 14.2 Mbit/s 4/4 1440p 242-244 21.9 Mbit/s 4/4 2160p 243-244 38.4 Mbit/s 4/4 Under heavy competing CPU load a 2160p take can still drop frames; one run overlapping a lint pass produced 174. The capture loop now counts what it did with everything it read -- delivered, empty samples, short buffers, buffer failures, read failures -- and reports the tally when it ends. A camera that produced nothing used to surface only as MF_E_SINK_NO_SAMPLES_PROCESSED from the encoder's Finalize, the one place that cannot say which of the loop's five silent `continue` paths swallowed the frames. The DirectShow fallback keeps delivering BGRA and is unchanged by construction, but it could not be exercised here: the test camera is visible to Media Foundation, so that path never runs. --- electron/app-settings.test.ts | 36 +++ electron/app-settings.ts | 18 ++ electron/ipc/handlers.ts | 4 + electron/ipc/recordingPrefs.test.ts | 1 + electron/native/wgc-capture/CMakeLists.txt | 16 ++ .../wgc-capture/src/dshow_webcam_capture.cpp | 109 +++++++- .../wgc-capture/src/dshow_webcam_capture.h | 16 +- electron/native/wgc-capture/src/main.cpp | 73 ++++- .../native/wgc-capture/src/mf_encoder.cpp | 88 +++++- electron/native/wgc-capture/src/mf_encoder.h | 28 ++ .../native/wgc-capture/src/webcam_capture.cpp | 255 ++++++++++++++++-- .../native/wgc-capture/src/webcam_capture.h | 34 ++- .../native/wgc-capture/src/webcam_format.cpp | 82 ++++++ .../native/wgc-capture/src/webcam_format.h | 44 +++ .../wgc-capture/src/webcam_format_test.cpp | 131 +++++++++ scripts/build-windows-wgc-helper.mjs | 10 + scripts/test-windows-wgc-helper.mjs | 30 ++- .../launch/HudDeviceSettings.quality.test.tsx | 96 +++++++ src/components/launch/HudDeviceSettings.tsx | 30 +++ src/components/launch/LaunchWindow.tsx | 20 ++ src/hooks/useCameraPreviewStream.ts | 9 +- src/hooks/useScreenRecorder.ts | 47 ++-- src/hooks/webcamCaptureTarget.test.ts | 137 ++++++++++ src/hooks/webcamCaptureTarget.ts | 104 +++++++ src/i18n/locales/ar/launch.json | 6 +- src/i18n/locales/cs/launch.json | 6 +- src/i18n/locales/de/launch.json | 6 +- src/i18n/locales/en/launch.json | 6 +- src/i18n/locales/es/launch.json | 6 +- src/i18n/locales/fr/launch.json | 6 +- src/i18n/locales/it/launch.json | 6 +- src/i18n/locales/ja-JP/launch.json | 6 +- src/i18n/locales/ko-KR/launch.json | 6 +- src/i18n/locales/pt-BR/launch.json | 6 +- src/i18n/locales/ru/launch.json | 6 +- src/i18n/locales/tr/launch.json | 6 +- src/i18n/locales/vi/launch.json | 6 +- src/i18n/locales/zh-CN/launch.json | 6 +- src/i18n/locales/zh-TW/launch.json | 6 +- 39 files changed, 1432 insertions(+), 76 deletions(-) create mode 100644 electron/native/wgc-capture/src/webcam_format.cpp create mode 100644 electron/native/wgc-capture/src/webcam_format.h create mode 100644 electron/native/wgc-capture/src/webcam_format_test.cpp create mode 100644 src/components/launch/HudDeviceSettings.quality.test.tsx create mode 100644 src/hooks/webcamCaptureTarget.test.ts create mode 100644 src/hooks/webcamCaptureTarget.ts diff --git a/electron/app-settings.test.ts b/electron/app-settings.test.ts index c6c9e74f3..f3781734f 100644 --- a/electron/app-settings.test.ts +++ b/electron/app-settings.test.ts @@ -129,4 +129,40 @@ describe("app settings store", () => { store.dismissStarPrompt(); expect(store.getSnapshot().recording).toMatchObject({ micEnabled: true, micDeviceId: "mic" }); }); + + it("stores a camera quality and rejects one it cannot capture at", () => { + const dir = temp(); + const store = new AppSettingsStore(dir); + + expect(store.setRecordingPreferences({ camQuality: "1080p" }).recording.camQuality).toBe( + "1080p", + ); + expect(() => store.setRecordingPreferences({ camQuality: "4320p" as never })).toThrow( + TypeError, + ); + expect(store.getSnapshot().recording.camQuality).toBe("1080p"); + }); + + it("leaves the camera quality alone when a patch does not mention it", () => { + // Every other window writes narrow patches; a camera toggle from the HUD + // must not quietly reset the resolution the user picked. + const dir = temp(); + const store = new AppSettingsStore(dir); + store.setRecordingPreferences({ camQuality: "1440p" }); + + store.setRecordingPreferences({ camEnabled: true }); + + expect(store.getSnapshot().recording.camQuality).toBe("1440p"); + }); + + it("reads a settings file written before the camera had a quality", () => { + const dir = temp(); + writeFileSync( + path.join(dir, "recording-settings.json"), + JSON.stringify({ micEnabled: true }), + "utf8", + ); + + expect(new AppSettingsStore(dir).getSnapshot().recording.camQuality).toBe("2160p"); + }); }); diff --git a/electron/app-settings.ts b/electron/app-settings.ts index 2cc1cee25..bef025bf3 100644 --- a/electron/app-settings.ts +++ b/electron/app-settings.ts @@ -1,5 +1,11 @@ import { readFileSync, renameSync, rmSync, writeFileSync } from "node:fs"; import path from "node:path"; +import { + DEFAULT_WEBCAM_QUALITY, + WEBCAM_QUALITY_IDS, + type WebcamQualityId, + webcamQualityFrom, +} from "../src/hooks/webcamCaptureTarget"; import type { CursorCaptureMode } from "../src/lib/recordingSession"; export interface RecordingPreferences { @@ -9,6 +15,8 @@ export interface RecordingPreferences { camEnabled: boolean; camDeviceId: string | null; camDeviceName: string | null; + /** Capture resolution for the camera. See WEBCAM_QUALITY_PRESETS. */ + camQuality: WebcamQualityId; systemAudioEnabled: boolean; cursorCaptureMode: CursorCaptureMode; /** Display captures on macOS and Windows. Opt-in: on Windows the icons also leave the real desktop while recording. */ @@ -30,6 +38,7 @@ export const DEFAULT_RECORDING_PREFERENCES: RecordingPreferences = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camQuality: DEFAULT_WEBCAM_QUALITY, systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", hideDesktopIcons: false, @@ -100,6 +109,9 @@ function parseRecording(raw: RawSettings): RecordingPreferences { camEnabled: bool(raw.camEnabled, DEFAULT_RECORDING_PREFERENCES.camEnabled), camDeviceId: nullableString(raw.camDeviceId, DEFAULT_RECORDING_PREFERENCES.camDeviceId), camDeviceName: nullableString(raw.camDeviceName, DEFAULT_RECORDING_PREFERENCES.camDeviceName), + // Unset in every settings file written before the camera had a quality + // setting, and `webcamQualityFrom` answers those with the default. + camQuality: webcamQualityFrom(raw.camQuality), systemAudioEnabled: bool( raw.systemAudioEnabled, DEFAULT_RECORDING_PREFERENCES.systemAudioEnabled, @@ -167,6 +179,12 @@ function validateRecordingPatch(patch: Partial): void { if (key === "cursorCaptureMode" && value !== "system" && value !== "editable-overlay") { throw new TypeError("cursorCaptureMode is invalid"); } + // Rejected here rather than coerced on read, so a bad write is a visible + // error at its source instead of a resolution that silently is not the + // one the caller asked for. + if (key === "camQuality" && !WEBCAM_QUALITY_IDS.includes(value as WebcamQualityId)) { + throw new TypeError("camQuality is invalid"); + } } } diff --git a/electron/ipc/handlers.ts b/electron/ipc/handlers.ts index bd19f1059..58b41b86a 100644 --- a/electron/ipc/handlers.ts +++ b/electron/ipc/handlers.ts @@ -17,6 +17,7 @@ import { shell, systemPreferences, } from "electron"; +import { DEFAULT_WEBCAM_QUALITY, type WebcamQualityId } from "../../src/hooks/webcamCaptureTarget"; import { type AxcutDocument, isAxcutDocumentFile, @@ -649,6 +650,8 @@ export interface RecordingPrefs { camDeviceId: string | null; /** Camera label paired with the preferred id for restart-safe resolution. */ camDeviceName: string | null; + /** Capture resolution for the camera. See WEBCAM_QUALITY_PRESETS. */ + camQuality: WebcamQualityId; systemAudioEnabled: boolean; cursorCaptureMode: CursorCaptureMode; hideDesktopIcons: boolean; @@ -662,6 +665,7 @@ const defaultRecordingPrefs: RecordingPrefs = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camQuality: DEFAULT_WEBCAM_QUALITY, systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", hideDesktopIcons: false, diff --git a/electron/ipc/recordingPrefs.test.ts b/electron/ipc/recordingPrefs.test.ts index fe8cdce9d..2c3317af9 100644 --- a/electron/ipc/recordingPrefs.test.ts +++ b/electron/ipc/recordingPrefs.test.ts @@ -19,6 +19,7 @@ const defaults: RecordingPrefs = { camEnabled: false, camDeviceId: null, camDeviceName: null, + camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", hideDesktopIcons: false, diff --git a/electron/native/wgc-capture/CMakeLists.txt b/electron/native/wgc-capture/CMakeLists.txt index c0921486f..efa82295d 100644 --- a/electron/native/wgc-capture/CMakeLists.txt +++ b/electron/native/wgc-capture/CMakeLists.txt @@ -55,6 +55,8 @@ add_executable(wgc-capture src/wasapi_render_keepalive.cpp src/wasapi_render_keepalive.h src/webcam_capture.cpp + src/webcam_format.cpp + src/webcam_format.h src/webcam_capture.h src/wgc_session.cpp src/wgc_session.h @@ -117,3 +119,17 @@ target_link_libraries(audio_sample_utils_test PRIVATE mfuuid ole32 ) + +add_executable(webcam_format_test + src/webcam_format.cpp + src/webcam_format.h + src/webcam_format_test.cpp +) + +target_compile_definitions(webcam_format_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(webcam_format_test PRIVATE /EHsc /W4 /utf-8) diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp index f39e4b065..d421f9cf3 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp @@ -1,5 +1,7 @@ #include "dshow_webcam_capture.h" +#include "webcam_format.h" + #include #include #include @@ -29,6 +31,19 @@ ISampleGrabber : public IUnknown { virtual HRESULT STDMETHODCALLTYPE SetCallback(IUnknown* callback, long whichMethodToCallback) = 0; }; +/** + * Is this subtype something the graph must DECODE before this class can unpack it? + * + * Listed positively -- an unknown subtype counts as compressed -- because + * guessing wrong this way only costs resolution, while guessing wrong the other + * way costs the camera track entirely. + */ +bool isCompressedDshowSubtype(const GUID& subtype) { + return !(subtype == MEDIASUBTYPE_YUY2 || subtype == MEDIASUBTYPE_NV12 || + subtype == MEDIASUBTYPE_RGB32 || subtype == MEDIASUBTYPE_RGB24 || + subtype == MEDIASUBTYPE_UYVY || subtype == MEDIASUBTYPE_YV12); +} + bool succeeded(HRESULT hr, const char* label) { if (SUCCEEDED(hr)) { return true; @@ -112,7 +127,93 @@ DirectShowWebcamCapture::~DirectShowWebcamCapture() { delete impl_; } -bool DirectShowWebcamCapture::buildGraph(const CLSID& sourceClsid, const GUID* preferredSubtype) { +void DirectShowWebcamCapture::applyPreferredFormat(int requestedWidth, int requestedHeight, int requestedFps) { + Microsoft::WRL::ComPtr streamConfig; + if (FAILED(impl_->captureGraph->FindInterface( + &PIN_CATEGORY_CAPTURE, + &MEDIATYPE_Video, + impl_->captureFilter.Get(), + IID_PPV_ARGS(&streamConfig)))) { + // Virtual cameras commonly expose a single fixed format and no + // IAMStreamConfig at all. Nothing to choose from; let the graph connect. + return; + } + + int capCount = 0; + int capSize = 0; + if (FAILED(streamConfig->GetNumberOfCapabilities(&capCount, &capSize)) || + capSize != sizeof(VIDEO_STREAM_CONFIG_CAPS)) { + return; + } + + std::vector formats; + formats.reserve(static_cast(std::max(0, capCount))); + for (int index = 0; index < capCount; ++index) { + VIDEO_STREAM_CONFIG_CAPS caps{}; + AM_MEDIA_TYPE* mediaType = nullptr; + if (FAILED(streamConfig->GetStreamCaps(index, &mediaType, reinterpret_cast(&caps))) || !mediaType) { + continue; + } + if (mediaType->formattype == FORMAT_VideoInfo && mediaType->pbFormat) { + const auto* videoInfo = reinterpret_cast(mediaType->pbFormat); + // AvgTimePerFrame is in 100ns units; 333333 is 30 fps. + const int fps = videoInfo->AvgTimePerFrame > 0 + ? static_cast((10'000'000LL + videoInfo->AvgTimePerFrame / 2) / videoInfo->AvgTimePerFrame) + : 0; + formats.push_back(WebcamFormat{ + std::abs(videoInfo->bmiHeader.biWidth), + std::abs(videoInfo->bmiHeader.biHeight), + fps, + isCompressedDshowSubtype(mediaType->subtype)}); + } + freeMediaType(*mediaType); + CoTaskMemFree(mediaType); + } + + const WebcamFormat chosen = chooseWebcamFormat( + formats, + requestedWidth > 0 ? requestedWidth : 1920, + requestedHeight > 0 ? requestedHeight : 1080, + std::max(1, requestedFps)); + + // Second pass: set the first capability that matches the choice. GetStreamCaps + // hands back the media type we must give to SetFormat, so it has to be + // re-read rather than cached across the loop above. + for (int index = 0; index < capCount; ++index) { + VIDEO_STREAM_CONFIG_CAPS caps{}; + AM_MEDIA_TYPE* mediaType = nullptr; + if (FAILED(streamConfig->GetStreamCaps(index, &mediaType, reinterpret_cast(&caps))) || !mediaType) { + continue; + } + bool matched = false; + if (mediaType->formattype == FORMAT_VideoInfo && mediaType->pbFormat) { + auto* videoInfo = reinterpret_cast(mediaType->pbFormat); + if (std::abs(videoInfo->bmiHeader.biWidth) == chosen.width && + std::abs(videoInfo->bmiHeader.biHeight) == chosen.height && + isCompressedDshowSubtype(mediaType->subtype) == chosen.compressed) { + if (chosen.fps > 0) { + videoInfo->AvgTimePerFrame = 10'000'000LL / chosen.fps; + } + matched = SUCCEEDED(streamConfig->SetFormat(mediaType)); + if (matched) { + std::cerr << "INFO: DirectShow webcam format " << chosen.width << "x" << chosen.height + << "@" << chosen.fps << std::endl; + } + } + } + freeMediaType(*mediaType); + CoTaskMemFree(mediaType); + if (matched) { + return; + } + } +} + +bool DirectShowWebcamCapture::buildGraph( + const CLSID& sourceClsid, + const GUID* preferredSubtype, + int preferredWidth, + int preferredHeight) { // Every attempt starts from empty filters. A RenderStream that fails can // leave pins connected behind it, and retrying on top of that half-built // graph is how you get a second failure that says nothing about the format. @@ -175,6 +276,8 @@ bool DirectShowWebcamCapture::buildGraph(const CLSID& sourceClsid, const GUID* p return false; } + applyPreferredFormat(preferredWidth, preferredHeight, fps_); + return succeeded(impl_->captureGraph->RenderStream( &PIN_CATEGORY_CAPTURE, &MEDIATYPE_Video, @@ -231,14 +334,14 @@ bool DirectShowWebcamCapture::initialize( // it connects with a subtype this file cannot read // (getopenscreen/openscreen#387). Naming a concrete subtype on the retry is // what makes DirectShow insert a colour converter for it. - if (!buildGraph(selectedClsid, nullptr)) { + if (!buildGraph(selectedClsid, nullptr, requestedWidth, requestedHeight)) { return false; } if (!resolveConnectedFormat(requestedWidth, requestedHeight, false)) { std::cerr << "WARNING: DirectShow webcam speaks a format this build cannot unpack; " "asking for RGB32 so the graph converts it" << std::endl; - if (!buildGraph(selectedClsid, &MEDIASUBTYPE_RGB32)) { + if (!buildGraph(selectedClsid, &MEDIASUBTYPE_RGB32, requestedWidth, requestedHeight)) { return false; } if (!resolveConnectedFormat(requestedWidth, requestedHeight, true)) { diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.h b/electron/native/wgc-capture/src/dshow_webcam_capture.h index cf510e2df..c0ba3503d 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.h +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.h @@ -59,7 +59,21 @@ class DirectShowWebcamCapture { * without leaving the graph half-built, so the caller can retry with a * different constraint. */ - bool buildGraph(const CLSID& sourceClsid, const GUID* preferredSubtype); + bool buildGraph( + const CLSID& sourceClsid, + const GUID* preferredSubtype, + int preferredWidth, + int preferredHeight); + /** + * Pins the capture pin to the best format the device offers, before the + * graph is rendered. + * + * Skipped, RenderStream's intelligent connect takes the pin's default + * format, which on a UVC camera is the first one it enumerates -- 640x480 + * on hardware that can do far better. Best-effort: a device without + * IAMStreamConfig, or one that rejects the format, still gets a graph. + */ + void applyPreferredFormat(int requestedWidth, int requestedHeight, int requestedFps); /** * Reads back what the graph actually negotiated and records how to unpack it. * diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index c86524d6d..08fe11aa0 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -393,6 +393,42 @@ void reportCaptureAdapters(ID3D11Device* device, HMONITOR targetMonitor) { } } +/** + * The NV12 twin of `hasVisibleBgraContent`: luma is the Y plane, one byte per + * pixel, so no colour conversion is needed to judge whether a frame has any + * picture in it. The Y plane is the first two thirds of an NV12 buffer. + */ +bool hasVisibleNv12Content(const std::vector& frame) { + if (frame.size() < 6) { + return false; + } + + const size_t lumaCount = frame.size() * 2 / 3; + const size_t step = std::max(1, lumaCount / 4096); + uint64_t lumaTotal = 0; + BYTE maxLuma = 0; + size_t sampled = 0; + for (size_t offset = 0; offset < lumaCount; offset += step) { + const BYTE luma = frame[offset]; + lumaTotal += luma; + maxLuma = std::max(maxLuma, luma); + sampled += 1; + } + + // The same thresholds the BGRA probe uses. NV12 from a camera is + // studio-range, so a black frame sits at 16 rather than 0 -- which the + // maxLuma > 24 test already tolerates. + const uint64_t averageLuma = sampled > 0 ? lumaTotal / sampled : 0; + return maxLuma > 24 || averageLuma > 4; +} + +bool hasVisibleBgraContent(const std::vector& frame); + +/** Dispatches to the probe matching the frame's layout. */ +bool hasVisibleWebcamContent(const std::vector& frame, bool isNv12) { + return isNv12 ? hasVisibleNv12Content(frame) : hasVisibleBgraContent(frame); +} + bool hasVisibleBgraContent(const std::vector& frame) { if (frame.size() < 4) { return false; @@ -777,7 +813,10 @@ int wmain(int argc, wchar_t* argv[]) { WebcamCapture webcamCapture; bool webcamActive = false; - bool writeSeparateWebcam = false; + // Decided before initialize(), not after: it selects the capture pixel + // format, and only a camera going to its own file can use NV12 -- an inline + // picture-in-picture composite needs the frame as BGRA. + bool writeSeparateWebcam = config.webcamEnabled && !config.webcamOutputPath.empty(); if (config.webcamEnabled) { if (!webcamCapture.initialize( utf8ToWide(config.webcamDeviceId), @@ -785,7 +824,8 @@ int wmain(int argc, wchar_t* argv[]) { utf8ToWide(config.webcamDirectShowClsid), config.webcamWidth, config.webcamHeight, - config.webcamFps > 0 ? config.webcamFps : config.fps)) { + config.webcamFps > 0 ? config.webcamFps : config.fps, + writeSeparateWebcam)) { // Non-fatal: a screen+audio recording the user can still use is far // better than losing the whole recording because one camera device // didn't match. Report it so the renderer can inform the user (and, @@ -797,13 +837,14 @@ int wmain(int argc, wchar_t* argv[]) { "\"Failed to initialize native webcam capture\"}" << std::endl; config.webcamEnabled = false; + writeSeparateWebcam = false; } else { std::cout << "{\"event\":\"webcam-format\",\"schemaVersion\":2,\"width\":" << webcamCapture.width() << ",\"height\":" << webcamCapture.height() << ",\"fps\":" << webcamCapture.fps() << ",\"deviceName\":\"" << jsonEscape(wideToUtf8(webcamCapture.selectedDeviceName())) << "\"}" << std::endl; - writeSeparateWebcam = !config.webcamOutputPath.empty(); + // writeSeparateWebcam was decided above, before the pixel format. } } @@ -980,8 +1021,18 @@ int wmain(int argc, wchar_t* argv[]) { MFEncoderOptions webcamEncoderOptions = encoderOptions; webcamEncoderOptions.injectDefaultSinkWriterFailureOnce = false; webcamEncoderOptions.useDxgiInput = false; + webcamEncoderOptions.cpuInputIsNv12 = webcamCapture.deliversNv12(); + // The two-step ladder this replaces topped out at 8 Mbit/s for anything + // 720p or larger. That was sized for a camera nobody had configured + // above 640x480; now that the capture runs at the camera's real + // resolution, 8 Mbit/s starves a 1440p or 2160p frame badly enough to + // undo the extra pixels. The tiers mirror the screen ladder above. const int webcamPixels = std::max(1, webcamCapture.width()) * std::max(1, webcamCapture.height()); - const int webcamBitrate = webcamPixels >= 1280 * 720 ? 8'000'000 : 4'000'000; + const int webcamBitrate = webcamPixels >= 3840 * 2160 ? 40'000'000 + : webcamPixels >= 2560 * 1440 ? 24'000'000 + : webcamPixels >= 1920 * 1080 ? 16'000'000 + : webcamPixels >= 1280 * 720 ? 8'000'000 + : 4'000'000; if (!webcamEncoder.initialize( utf8ToWide(config.webcamOutputPath), webcamCapture.width(), @@ -1175,7 +1226,7 @@ int wmain(int argc, wchar_t* argv[]) { WebcamFrameSnapshot candidateWebcamFrame; if (webcamCapture.copyLatestFrame(candidateWebcamFrame) && candidateWebcamFrame.sequence != latestWebcamSequence && - hasVisibleBgraContent(candidateWebcamFrame.data)) { + hasVisibleWebcamContent(candidateWebcamFrame.data, webcamCapture.deliversNv12())) { latestWebcamFrame = std::move(candidateWebcamFrame.data); latestWebcamWidth = candidateWebcamFrame.width; latestWebcamHeight = candidateWebcamFrame.height; @@ -1230,7 +1281,15 @@ int wmain(int argc, wchar_t* argv[]) { // Capture the sample here, but submit it to the sink // writer OUTSIDE this block below (issue #115) so a // slow WriteSample can't hold up the next frame pull. - hasWebcamSample = webcamEncoder.captureBgraSample(webcamFrame, webcamTimestampHns, webcamSample); + hasWebcamSample = + webcamCapture.deliversNv12() + ? webcamEncoder.captureNv12Sample( + Nv12FrameView{ + webcamFrame.data, webcamFrame.width, webcamFrame.height}, + webcamTimestampHns, + webcamSample) + : webcamEncoder.captureBgraSample( + webcamFrame, webcamTimestampHns, webcamSample); if (!hasWebcamSample) { encodeFailed = true; control.requestStop(); @@ -1461,7 +1520,7 @@ int wmain(int argc, wchar_t* argv[]) { while (std::chrono::steady_clock::now() < webcamDeadline && !hasVisibleWebcamFrame) { WebcamFrameSnapshot candidateWebcamFrame; if (webcamCapture.copyLatestFrame(candidateWebcamFrame) && - hasVisibleBgraContent(candidateWebcamFrame.data)) { + hasVisibleWebcamContent(candidateWebcamFrame.data, webcamCapture.deliversNv12())) { latestWebcamFrame = std::move(candidateWebcamFrame.data); latestWebcamWidth = candidateWebcamFrame.width; latestWebcamHeight = candidateWebcamFrame.height; diff --git a/electron/native/wgc-capture/src/mf_encoder.cpp b/electron/native/wgc-capture/src/mf_encoder.cpp index b4dabcc54..786881e00 100644 --- a/electron/native/wgc-capture/src/mf_encoder.cpp +++ b/electron/native/wgc-capture/src/mf_encoder.cpp @@ -690,6 +690,7 @@ bool MFEncoder::initialize( // attempt would eat the injection and the run would land on the plain CPU // encoder, never reaching the software encoder the knob is aimed at. useDxgiInput_ = options.useDxgiInput && !options.injectDefaultSinkWriterFailureOnce; + cpuInputIsNv12_ = options.cpuInputIsNv12; videoEncoderSelection_ = kVideoEncoderSelectionDefault; videoEncoderRuntime_ = kVideoEncoderRuntimeUnknown; @@ -725,9 +726,26 @@ bool MFEncoder::initialize( // type the RGB32 path would have produced from scratch. Every attribute // one mode sets is deleted by the other; nothing carries over. auto configureVideoInputType = [&](bool dxgi) { + // Three input shapes, not two: GPU NV12, system-memory NV12 (the + // webcam, whose camera hands us NV12 already) and system-memory RGB32. + const bool nv12 = dxgi || cpuInputIsNv12_; inputType->SetGUID(MF_MT_MAJOR_TYPE, MFMediaType_Video); - inputType->SetGUID(MF_MT_SUBTYPE, dxgi ? MFVideoFormat_NV12 : MFVideoFormat_RGB32); + inputType->SetGUID(MF_MT_SUBTYPE, nv12 ? MFVideoFormat_NV12 : MFVideoFormat_RGB32); inputType->SetUINT32(MF_MT_INTERLACE_MODE, MFVideoInterlace_Progressive); + if (!dxgi && cpuInputIsNv12_) { + // NV12's declared stride is the Y plane's, which is one byte per + // pixel -- not the four an RGB32 row needs. + inputType->SetUINT32(MF_MT_DEFAULT_STRIDE, static_cast(width_)); + // A camera's NV12 is studio-range, and at these sizes BT.709. + // Untagged, the encoder and the player each fall back to their own + // default and the recording comes back with shifted colours. + inputType->SetUINT32(MF_MT_VIDEO_NOMINAL_RANGE, MFNominalRange_16_235); + inputType->SetUINT32(MF_MT_YUV_MATRIX, MFVideoTransferMatrix_BT709); + setFrameSize(inputType.Get(), static_cast(width_), static_cast(height_)); + setFrameRate(inputType.Get(), static_cast(fps_)); + setPixelAspectRatio(inputType.Get()); + return; + } if (dxgi) { inputType->DeleteItem(MF_MT_DEFAULT_STRIDE); // The video processor below converts full-range BGRA into @@ -750,7 +768,7 @@ bool MFEncoder::initialize( // Carried on the H.264 type as well so the MP4 sink writes the matching // colour tags instead of leaving players to guess from the frame size. auto configureOutputColorTags = [&](bool dxgi) { - if (dxgi) { + if (dxgi || cpuInputIsNv12_) { outputType->SetUINT32(MF_MT_VIDEO_NOMINAL_RANGE, MFNominalRange_16_235); outputType->SetUINT32(MF_MT_YUV_MATRIX, MFVideoTransferMatrix_BT709); } else { @@ -1110,12 +1128,16 @@ bool MFEncoder::copyBgraFrameToBuffer(const BgraFrameView& frame, BYTE* destinat } if (frame.width == width_ && frame.height == height_) { - for (DWORD i = 0; i < requiredBytes; i += 4) { - destination[i] = frame.data[i]; - destination[i + 1] = frame.data[i + 1]; - destination[i + 2] = frame.data[i + 2]; - destination[i + 3] = 255; - } + // One memcpy, not a per-pixel loop forcing alpha to 255. + // + // The loop this replaces ran once per BYTE: at 3840x2160 that is 8.3 + // million iterations per frame, which measured out at ~12 fps of real + // camera motion inside a file whose container claimed 30 -- the encoder + // padded the gap with duplicates. The alpha it was writing is dead + // weight anyway: this buffer feeds an H.264 encoder through + // MFVideoFormat_RGB32, and RGB-to-YUV conversion ignores the alpha + // channel entirely. + std::memcpy(destination, frame.data, requiredBytes); return true; } @@ -1649,6 +1671,56 @@ bool MFEncoder::captureVideoSample( return true; } +bool MFEncoder::captureNv12Sample( + const Nv12FrameView& frame, + int64_t timestampHns, + Microsoft::WRL::ComPtr& outSample) { + outSample.Reset(); + + if (!frame.data || frame.width != width_ || frame.height != height_) { + // No rescaler here on purpose: the webcam encoder is created with the + // capture's own dimensions, so a mismatch means a bug upstream rather + // than a frame worth stretching. + std::cerr << "ERROR: NV12 webcam frame does not match the encoder's size" << std::endl; + return false; + } + + const int64_t sampleDuration = 10'000'000LL / fps_; + const int64_t sampleTime = nextSampleTime(timestampHns, sampleDuration); + const DWORD frameBytes = static_cast(width_ * height_ * 3 / 2); + + Microsoft::WRL::ComPtr buffer; + if (!succeeded(MFCreateMemoryBuffer(frameBytes, &buffer), "MFCreateMemoryBuffer(webcam NV12)")) { + return false; + } + + BYTE* data = nullptr; + DWORD maxLength = 0; + DWORD currentLength = 0; + if (!succeeded(buffer->Lock(&data, &maxLength, ¤tLength), "IMFMediaBuffer::Lock(webcam NV12)")) { + return false; + } + if (maxLength < frameBytes) { + buffer->Unlock(); + std::cerr << "ERROR: Media Foundation webcam NV12 buffer is too small" << std::endl; + return false; + } + std::memcpy(data, frame.data, frameBytes); + buffer->Unlock(); + buffer->SetCurrentLength(frameBytes); + + Microsoft::WRL::ComPtr sample; + if (!succeeded(MFCreateSample(&sample), "MFCreateSample(webcam NV12)")) { + return false; + } + sample->AddBuffer(buffer.Get()); + sample->SetSampleTime(sampleTime); + sample->SetSampleDuration(sampleDuration); + + outSample = sample; + return true; +} + bool MFEncoder::captureBgraSample( const BgraFrameView& frame, int64_t timestampHns, diff --git a/electron/native/wgc-capture/src/mf_encoder.h b/electron/native/wgc-capture/src/mf_encoder.h index 8d1d6ae6e..148693408 100644 --- a/electron/native/wgc-capture/src/mf_encoder.h +++ b/electron/native/wgc-capture/src/mf_encoder.h @@ -18,6 +18,21 @@ struct BgraFrameView { int height = 0; }; +/** + * A planar NV12 frame: a width*height Y plane followed by an interleaved + * width/2 * height/2 UV plane, so width*height*3/2 bytes in total. + * + * Used for the webcam, whose camera can hand Media Foundation NV12 directly. + * Asking that same camera for RGB32 made the source reader decode and convert + * every frame, which measured at 92ms per 4K frame -- a hard ceiling near 11 + * fps -- against 32ms for NV12. + */ +struct Nv12FrameView { + const BYTE* data = nullptr; + int width = 0; + int height = 0; +}; + struct AudioInputFormat { GUID subtype = MFAudioFormat_PCM; UINT32 sampleRate = 0; @@ -47,6 +62,14 @@ struct MFEncoderOptions { // driver that refuses shared keyed-mutex textures records exactly as it did // before the path existed. Ask usesDxgiInput() for what actually happened. bool useDxgiInput = false; + /** + * Feed this encoder NV12 from system memory instead of RGB32. + * + * Only meaningful when `useDxgiInput` is false. The webcam encoder sets it + * when the camera itself delivers NV12; the screen encoder's CPU path + * still produces BGRA and leaves it alone. + */ + bool cpuInputIsNv12 = false; }; constexpr const char* kVideoEncoderSelectionDefault = "default"; @@ -122,6 +145,10 @@ class MFEncoder { ID3D11Texture2D* texture, int64_t timestampHns, Microsoft::WRL::ComPtr& outSample); + bool captureNv12Sample( + const Nv12FrameView& frame, + int64_t timestampHns, + Microsoft::WRL::ComPtr& outSample); bool captureBgraSample( const BgraFrameView& frame, int64_t timestampHns, @@ -237,6 +264,7 @@ class MFEncoder { int64_t lastTimestampHns_ = -1; bool finalized_ = false; bool useDxgiInput_ = false; + bool cpuInputIsNv12_ = false; const char* videoEncoderSelection_ = kVideoEncoderSelectionDefault; const char* videoEncoderRuntime_ = kVideoEncoderRuntimeUnknown; const char* containerFormat_ = kContainerFormatMp4; diff --git a/electron/native/wgc-capture/src/webcam_capture.cpp b/electron/native/wgc-capture/src/webcam_capture.cpp index 6377a799b..0a3666482 100644 --- a/electron/native/wgc-capture/src/webcam_capture.cpp +++ b/electron/native/wgc-capture/src/webcam_capture.cpp @@ -1,5 +1,7 @@ #include "webcam_capture.h" +#include "webcam_format.h" + #include #include #include @@ -153,7 +155,8 @@ bool WebcamCapture::initialize( const std::wstring& directShowClsid, int requestedWidth, int requestedHeight, - int requestedFps) { + int requestedFps, + bool preferNv12) { fps_ = std::clamp(requestedFps > 0 ? requestedFps : 30, 1, 60); usingDirectShow_ = false; selectedMatchScore_ = 0; @@ -196,7 +199,7 @@ bool WebcamCapture::initialize( return false; } - return configureReader(requestedWidth, requestedHeight, fps_); + return configureReader(requestedWidth, requestedHeight, fps_, preferNv12); } bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstring& deviceName) { @@ -252,7 +255,133 @@ bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstrin return succeeded(hr, "ActivateObject(webcam)"); } -bool WebcamCapture::configureReader(int requestedWidth, int requestedHeight, int requestedFps) { +namespace { + +/** + * Is this subtype something the source reader must DECODE before it can convert? + * + * Listed positively -- an unknown subtype counts as compressed -- because the + * cost of guessing wrong the other way is a camera that silently records + * nothing, while guessing wrong this way only costs resolution. + */ +bool isCompressedVideoSubtype(const GUID& subtype) { + static const GUID* const kUncompressed[] = { + &MFVideoFormat_NV12, + &MFVideoFormat_YUY2, + &MFVideoFormat_RGB32, + &MFVideoFormat_RGB24, + &MFVideoFormat_ARGB32, + &MFVideoFormat_UYVY, + &MFVideoFormat_YV12, + &MFVideoFormat_I420, + &MFVideoFormat_IYUV, + }; + for (const GUID* candidate : kUncompressed) { + if (subtype == *candidate) { + return false; + } + } + return true; +} + +/** + * Every video format the camera advertises, read off the source reader. + * + * MF_MT_FRAME_RATE is a ratio; a 30000/1001 entry is NTSC 29.97 and rounds to + * 30, which is what we want to compare against the requested rate. A type + * missing either attribute is skipped rather than guessed at -- it cannot be + * requested by size anyway. + */ +std::vector readNativeFormats(IMFSourceReader* reader) { + std::vector formats; + for (DWORD index = 0;; ++index) { + Microsoft::WRL::ComPtr nativeType; + const HRESULT hr = + reader->GetNativeMediaType(MF_SOURCE_READER_FIRST_VIDEO_STREAM, index, &nativeType); + if (hr == MF_E_NO_MORE_TYPES || FAILED(hr)) { + break; + } + + UINT32 width = 0; + UINT32 height = 0; + if (FAILED(MFGetAttributeSize(nativeType.Get(), MF_MT_FRAME_SIZE, &width, &height))) { + continue; + } + + UINT32 numerator = 0; + UINT32 denominator = 0; + int fps = 0; + if (SUCCEEDED(MFGetAttributeRatio(nativeType.Get(), MF_MT_FRAME_RATE, &numerator, &denominator)) && + denominator > 0) { + fps = static_cast((numerator + denominator / 2) / denominator); + } + + GUID subtype{}; + const bool compressed = SUCCEEDED(nativeType->GetGUID(MF_MT_SUBTYPE, &subtype)) + ? isCompressedVideoSubtype(subtype) + : true; + + formats.push_back( + WebcamFormat{static_cast(width), static_cast(height), fps, compressed}); + } + return formats; +} + + +/** + * Puts the camera itself on `wanted` by selecting the matching native type. + * + * Best-effort: a driver that refuses leaves the reader on whatever it had, and + * the RGB32 request that follows still produces a usable -- if smaller -- frame. + */ +void selectNativeFormat(IMFSourceReader* reader, const WebcamFormat& wanted) { + for (DWORD index = 0;; ++index) { + Microsoft::WRL::ComPtr nativeType; + const HRESULT hr = + reader->GetNativeMediaType(MF_SOURCE_READER_FIRST_VIDEO_STREAM, index, &nativeType); + if (hr == MF_E_NO_MORE_TYPES || FAILED(hr)) { + return; + } + + UINT32 width = 0; + UINT32 height = 0; + if (FAILED(MFGetAttributeSize(nativeType.Get(), MF_MT_FRAME_SIZE, &width, &height))) { + continue; + } + if (static_cast(width) != wanted.width || static_cast(height) != wanted.height) { + continue; + } + + GUID subtype{}; + if (FAILED(nativeType->GetGUID(MF_MT_SUBTYPE, &subtype)) || + isCompressedVideoSubtype(subtype) != wanted.compressed) { + continue; + } + + UINT32 numerator = 0; + UINT32 denominator = 0; + if (wanted.fps > 0 && + SUCCEEDED(MFGetAttributeRatio(nativeType.Get(), MF_MT_FRAME_RATE, &numerator, &denominator)) && + denominator > 0) { + const int fps = static_cast((numerator + denominator / 2) / denominator); + if (fps != wanted.fps) { + continue; + } + } + + if (SUCCEEDED(reader->SetCurrentMediaType(MF_SOURCE_READER_FIRST_VIDEO_STREAM, nullptr, nativeType.Get()))) { + return; + } + } +} + +} // namespace + +bool WebcamCapture::configureReader( + int requestedWidth, + int requestedHeight, + int requestedFps, + bool preferNv12) { Microsoft::WRL::ComPtr attributes; if (!succeeded(MFCreateAttributes(&attributes, 3), "MFCreateAttributes(webcam reader)")) { return false; @@ -274,15 +403,57 @@ bool WebcamCapture::configureReader(int requestedWidth, int requestedHeight, int return false; } mediaType->SetGUID(MF_MT_MAJOR_TYPE, MFMediaType_Video); - mediaType->SetGUID(MF_MT_SUBTYPE, MFVideoFormat_RGB32); - if (requestedWidth > 0 && requestedHeight > 0) { - MFSetAttributeSize(mediaType.Get(), MF_MT_FRAME_SIZE, static_cast(requestedWidth), static_cast(requestedHeight)); - } + // NV12 when the frame only has to reach the webcam encoder, which wants + // NV12 anyway. Asking for RGB32 instead makes the source reader decode AND + // convert every frame: measured at 92ms per 3840x2160 frame against 32ms + // for NV12, which is the difference between 11 fps and the camera's full + // 30. BGRA is still used when the frame is composited into the screen + // recording inline, which needs it in that layout. + deliversNv12_ = preferNv12; + mediaType->SetGUID(MF_MT_SUBTYPE, preferNv12 ? MFVideoFormat_NV12 : MFVideoFormat_RGB32); + + // Pick the capture format, then SELECT it on the source before asking for + // RGB32 output. + // + // Both halves matter. Handing Media Foundation an output type with no + // MF_MT_FRAME_SIZE leaves the device on its default format -- 640x480 on a + // BRIO that offers 3840x2160 -- and that default is what the overlay was + // upscaling. But naming the size on the *output* type alone is not enough + // either: with MF_SOURCE_READER_ENABLE_VIDEO_PROCESSING the reader answers + // S_OK by inserting a converter in front of whichever native type is + // already selected, so the camera keeps running at 640x480 and the size we + // asked for is quietly dropped. Selecting the native type first is what + // actually reconfigures the camera. + const WebcamFormat chosen = chooseWebcamFormat( + readNativeFormats(sourceReader_.Get()), + requestedWidth > 0 ? requestedWidth : 1920, + requestedHeight > 0 ? requestedHeight : 1080, + std::max(1, requestedFps)); + std::cerr << "INFO: Native webcam format " << chosen.width << "x" << chosen.height << "@" + << chosen.fps << (chosen.compressed ? " (compressed)" : " (uncompressed)") << std::endl; + selectNativeFormat(sourceReader_.Get(), chosen); + + MFSetAttributeSize( + mediaType.Get(), MF_MT_FRAME_SIZE, static_cast(chosen.width), static_cast(chosen.height)); MFSetAttributeRatio(mediaType.Get(), MF_MT_FRAME_RATE, static_cast(std::max(1, requestedFps)), 1); - if (!succeeded(sourceReader_->SetCurrentMediaType(MF_SOURCE_READER_FIRST_VIDEO_STREAM, nullptr, mediaType.Get()), - "SetCurrentMediaType(webcam RGB32)")) { - return false; + if (FAILED(sourceReader_->SetCurrentMediaType(MF_SOURCE_READER_FIRST_VIDEO_STREAM, nullptr, mediaType.Get()))) { + // Some drivers refuse a size they nonetheless enumerate. A slightly soft + // camera beats no camera, so drop the size and let the driver choose -- + // exactly what this code did before it asked for anything. + std::cerr << "WARNING: Webcam rejected " << chosen.width << "x" << chosen.height + << "; falling back to the device default format" << std::endl; + Microsoft::WRL::ComPtr fallbackType; + if (!succeeded(MFCreateMediaType(&fallbackType), "MFCreateMediaType(webcam fallback)")) { + return false; + } + fallbackType->SetGUID(MF_MT_MAJOR_TYPE, MFMediaType_Video); + fallbackType->SetGUID(MF_MT_SUBTYPE, preferNv12 ? MFVideoFormat_NV12 : MFVideoFormat_RGB32); + if (!succeeded( + sourceReader_->SetCurrentMediaType(MF_SOURCE_READER_FIRST_VIDEO_STREAM, nullptr, fallbackType.Get()), + "SetCurrentMediaType(webcam RGB32)")) { + return false; + } } sourceReader_->SetStreamSelection(MF_SOURCE_READER_ALL_STREAMS, FALSE); sourceReader_->SetStreamSelection(MF_SOURCE_READER_FIRST_VIDEO_STREAM, TRUE); @@ -336,12 +507,14 @@ void WebcamCapture::stop() { void WebcamCapture::captureLoop() { CoInitializeEx(nullptr, COINIT_MULTITHREADED); + const auto loopStartedAt = std::chrono::steady_clock::now(); while (!stopRequested_) { DWORD streamIndex = 0; DWORD flags = 0; LONGLONG timestamp = 0; Microsoft::WRL::ComPtr sample; + const auto readStartedAt = std::chrono::steady_clock::now(); HRESULT hr = sourceReader_->ReadSample( MF_SOURCE_READER_FIRST_VIDEO_STREAM, 0, @@ -349,24 +522,35 @@ void WebcamCapture::captureLoop() { &flags, ×tamp, &sample); + readSampleUs_ += static_cast( + std::chrono::duration_cast( + std::chrono::steady_clock::now() - readStartedAt) + .count()); (void)streamIndex; (void)timestamp; if (FAILED(hr)) { - std::cerr << "WARNING: Failed to read webcam sample (hr=0x" << std::hex << hr << std::dec << ")" - << std::endl; + // Counted, not printed: a reader that fails every call would other- + // wise write thousands of identical lines over a long take. + readFailures_ += 1; + lastReadFailure_ = hr; std::this_thread::sleep_for(std::chrono::milliseconds(20)); continue; } if ((flags & MF_SOURCE_READERF_ENDOFSTREAM) != 0) { + sawEndOfStream_ = true; break; } if (!sample) { + // Normal in small numbers (MF_SOURCE_READERF_STREAMTICK); a run of + // nothing else is a camera that never started producing. + emptySamples_ += 1; continue; } Microsoft::WRL::ComPtr buffer; if (FAILED(sample->ConvertToContiguousBuffer(&buffer)) || !buffer) { + bufferFailures_ += 1; continue; } @@ -374,19 +558,54 @@ void WebcamCapture::captureLoop() { DWORD maxLength = 0; DWORD currentLength = 0; if (FAILED(buffer->Lock(&data, &maxLength, ¤tLength)) || !data) { + bufferFailures_ += 1; continue; } - const DWORD expectedLength = static_cast(std::max(0, width_) * std::max(0, height_) * 4); + const int pixels = std::max(0, width_) * std::max(0, height_); + const DWORD expectedLength = + static_cast(deliversNv12_ ? pixels * 3 / 2 : pixels * 4); if (currentLength >= expectedLength && expectedLength > 0) { - std::scoped_lock lock(frameMutex_); - latestFrame_.assign(data, data + expectedLength); - latestFrameSequence_ += 1; + if (framesDelivered_ == 0) { + const auto waitedMs = std::chrono::duration_cast( + std::chrono::steady_clock::now() - loopStartedAt) + .count(); + std::cerr << "INFO: First webcam frame after " << waitedMs << "ms" << std::endl; + } + framesDelivered_ += 1; + const auto storeStartedAt = std::chrono::steady_clock::now(); + { + std::scoped_lock lock(frameMutex_); + latestFrame_.assign(data, data + expectedLength); + latestFrameSequence_ += 1; + } + storeUs_ += static_cast( + std::chrono::duration_cast( + std::chrono::steady_clock::now() - storeStartedAt) + .count()); + } else if (shortBuffers_++, !reportedShortBuffer_) { + // Every frame arriving short means the reader is handing us something + // other than the RGB32 we asked for -- a compressed frame it never + // converted, most often. Silently dropped, that reads downstream as a + // camera that produced nothing at all, with no clue why. Said once, + // because it is true for every frame that follows. + reportedShortBuffer_ = true; + std::cerr << "WARNING: Webcam frame is " << currentLength << " bytes, expected " + << expectedLength << " for " << width_ << "x" << height_ << " " + << (deliversNv12_ ? "NV12" : "RGB32") << "; dropping frames" << std::endl; } buffer->Unlock(); } + std::cerr << "INFO: Webcam capture loop ended: delivered=" << framesDelivered_ + << " emptySamples=" << emptySamples_ << " shortBuffers=" << shortBuffers_ + << " bufferFailures=" << bufferFailures_ << " readFailures=" << readFailures_ + << " lastReadHr=0x" << std::hex << lastReadFailure_ << std::dec + << " endOfStream=" << (sawEndOfStream_ ? "yes" : "no") + << " readSampleMs=" << (readSampleUs_ / 1000) << " storeMs=" << (storeUs_ / 1000) + << std::endl; + CoUninitialize(); } @@ -406,6 +625,10 @@ bool WebcamCapture::copyLatestFrame(WebcamFrameSnapshot& destination) { return true; } +bool WebcamCapture::deliversNv12() const { + return !usingDirectShow_ && deliversNv12_; +} + int WebcamCapture::width() const { if (usingDirectShow_) { return directShowCapture_.width(); diff --git a/electron/native/wgc-capture/src/webcam_capture.h b/electron/native/wgc-capture/src/webcam_capture.h index 5b61aa6b9..612e36603 100644 --- a/electron/native/wgc-capture/src/webcam_capture.h +++ b/electron/native/wgc-capture/src/webcam_capture.h @@ -28,7 +28,8 @@ class WebcamCapture { const std::wstring& directShowClsid, int requestedWidth, int requestedHeight, - int requestedFps); + int requestedFps, + bool preferNv12); bool start(); void stop(); bool copyLatestFrame(WebcamFrameSnapshot& destination); @@ -36,11 +37,19 @@ class WebcamCapture { int width() const; int height() const; int fps() const; + /** + * Do the frames from `copyLatestFrame` carry NV12 rather than BGRA? + * + * Always false on the DirectShow fallback, which unpacks whatever the + * camera speaks into BGRA itself. Callers must ask rather than assume: + * the two layouts differ in size as well as in meaning. + */ + bool deliversNv12() const; const std::wstring& selectedDeviceName() const; private: bool selectDevice(const std::wstring& deviceId, const std::wstring& deviceName); - bool configureReader(int requestedWidth, int requestedHeight, int requestedFps); + bool configureReader(int requestedWidth, int requestedHeight, int requestedFps, bool preferNv12); void captureLoop(); Microsoft::WRL::ComPtr mediaSource_; @@ -56,6 +65,27 @@ class WebcamCapture { int fps_ = 30; bool mfStarted_ = false; bool usingDirectShow_ = false; + bool deliversNv12_ = false; + /** Latches the short-buffer warning so it is said once, not 30x a second. */ + bool reportedShortBuffer_ = false; + /** + * Why the capture loop did or did not produce frames. + * + * A camera that delivers nothing used to surface only as + * MF_E_SINK_NO_SAMPLES_PROCESSED from the encoder's Finalize -- the one + * place that cannot say which of the loop's five silent `continue` paths + * swallowed the frames. These are reported once when the loop ends. + */ + uint64_t framesDelivered_ = 0; + uint64_t emptySamples_ = 0; + uint64_t readFailures_ = 0; + uint64_t shortBuffers_ = 0; + uint64_t bufferFailures_ = 0; + bool sawEndOfStream_ = false; + HRESULT lastReadFailure_ = S_OK; + /** Where the loop's wall time goes, in microseconds. */ + uint64_t readSampleUs_ = 0; + uint64_t storeUs_ = 0; int selectedMatchScore_ = 0; std::wstring selectedDeviceName_; }; diff --git a/electron/native/wgc-capture/src/webcam_format.cpp b/electron/native/wgc-capture/src/webcam_format.cpp new file mode 100644 index 000000000..32e2020ed --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_format.cpp @@ -0,0 +1,82 @@ +#include "webcam_format.h" + +#include + +namespace { + +bool isUsable(const WebcamFormat& format) { + return format.width > 0 && format.height > 0; +} + +bool fitsTarget(const WebcamFormat& format, int targetWidth, int targetHeight) { + return format.width <= targetWidth && format.height <= targetHeight; +} + +long long pixels(const WebcamFormat& format) { + return static_cast(format.width) * static_cast(format.height); +} + +// True when `candidate` is a better frame rate than `best` for a target of +// `targetFps`: the lowest rate that still reaches the target, or -- when no +// mode reaches it -- the highest one on offer. +bool betterFps(int candidate, int best, int targetFps) { + const bool candidateReaches = candidate >= targetFps; + const bool bestReaches = best >= targetFps; + if (candidateReaches != bestReaches) { + return candidateReaches; + } + return candidateReaches ? candidate < best : candidate > best; +} + +} // namespace + +WebcamFormat chooseWebcamFormat( + const std::vector& available, + int targetWidth, + int targetHeight, + int targetFps) { + const WebcamFormat target{targetWidth, targetHeight, targetFps, false}; + + // Compressed modes are a last resort, so they are only considered when the + // camera offers nothing else at all. + const bool anyUncompressed = std::any_of(available.begin(), available.end(), [](const WebcamFormat& format) { + return isUsable(format) && !format.compressed; + }); + + const auto eligible = [&](const WebcamFormat& format) { + return isUsable(format) && !(anyUncompressed && format.compressed); + }; + + const bool anyFits = std::any_of(available.begin(), available.end(), [&](const WebcamFormat& format) { + return eligible(format) && fitsTarget(format, targetWidth, targetHeight); + }); + + const WebcamFormat* best = nullptr; + for (const WebcamFormat& format : available) { + if (!eligible(format)) { + continue; + } + // Once anything fits, oversized modes are out of the running entirely. + // With nothing fitting we rank the oversized modes instead, smallest + // first, so the scaler has the least work to do. + if (anyFits != fitsTarget(format, targetWidth, targetHeight)) { + continue; + } + if (best == nullptr) { + best = &format; + continue; + } + if (pixels(format) != pixels(*best)) { + const bool wins = anyFits ? pixels(format) > pixels(*best) : pixels(format) < pixels(*best); + if (wins) { + best = &format; + } + continue; + } + if (betterFps(format.fps, best->fps, targetFps)) { + best = &format; + } + } + + return best ? *best : target; +} diff --git a/electron/native/wgc-capture/src/webcam_format.h b/electron/native/wgc-capture/src/webcam_format.h new file mode 100644 index 000000000..7cb329af7 --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_format.h @@ -0,0 +1,44 @@ +#pragma once + +#include + +// One capture format a camera advertises. +struct WebcamFormat { + int width = 0; + int height = 0; + int fps = 0; + /** + * True for MJPEG/H264 modes, which reach higher resolutions than a camera's + * uncompressed ones but need a decoder MFT in front of the RGB32 conversion. + */ + bool compressed = false; +}; + +// Picks the format to drive the camera at, given everything it advertises. +// +// Both capture backends used to leave this to the driver: Media Foundation was +// handed a media type with no MF_MT_FRAME_SIZE, and the DirectShow graph was +// rendered without touching IAMStreamConfig. In both cases the device answers +// with its *default* type, and for a UVC camera that is the first format it +// enumerates -- 640x480 even on a BRIO that offers 3840x2160. The recording was +// then upscaled into the overlay, which is the pixelation users reported. +// +// Rules, in order: +// * uncompressed modes beat compressed ones outright, whatever their size. +// Pinning a camera's MJPEG mode and asking the source reader for RGB32 +// leaves it delivering no samples at all on some hosts -- a recording with +// no camera in it, which is far worse than a smaller one that works; +// * only formats that fit inside the target in BOTH dimensions are eligible, +// so an ultrawide mode cannot sneak in on pixel count alone; +// * among those, the most pixels wins; +// * ties go to the lowest frame rate that still reaches the target, else the +// highest available -- a 1080p60 mode is no help when we encode at 30; +// * if nothing fits (a camera that only does 4K), the smallest mode wins, so +// we scale down rather than push 1 GB/s of RGB32 through the encoder; +// * an empty list returns the target unchanged, letting the caller fall back +// to whatever the driver would have picked on its own. +WebcamFormat chooseWebcamFormat( + const std::vector& available, + int targetWidth, + int targetHeight, + int targetFps); diff --git a/electron/native/wgc-capture/src/webcam_format_test.cpp b/electron/native/wgc-capture/src/webcam_format_test.cpp new file mode 100644 index 000000000..9e40f5cbf --- /dev/null +++ b/electron/native/wgc-capture/src/webcam_format_test.cpp @@ -0,0 +1,131 @@ +#include "webcam_format.h" + +#include +#include + +namespace { + +int failures = 0; + +void expectFormat(const std::string& label, WebcamFormat actual, int width, int height, int fps) { + if (actual.width == width && actual.height == height && actual.fps == fps) { + return; + } + std::printf( + "FAIL %s: expected %dx%d@%d, got %dx%d@%d\n", + label.c_str(), + width, + height, + fps, + actual.width, + actual.height, + actual.fps); + ++failures; +} + +// Every mode a Logitech BRIO advertises that this helper could drive, as read +// off the device with `ffmpeg -f dshow -list_options true`. +const std::vector kBrio = { + {640, 480, 30, false}, {640, 360, 30, false}, {1280, 720, 30, false}, + {1920, 1080, 30, false}, {1280, 720, 90, true}, {1920, 1080, 60, true}, + {2560, 1440, 30, true}, {3840, 2160, 30, true}, {160, 120, 30, false}, + {340, 340, 30, false}, {1600, 896, 30, false}, {1024, 576, 30, false}, +}; + +} // namespace + +int main() { + // The regression this file exists for: a BRIO asked for 1080p must not come + // back with the 640x480 default it enumerates first. + expectFormat("brio targets 1080p", chooseWebcamFormat(kBrio, 1920, 1080, 30), 1920, 1080, 30); + + // Asked for more than the camera can do uncompressed, it must stop at the + // best uncompressed mode rather than pin an MJPEG one. Pinning MJPEG and + // asking the source reader for RGB32 produced a reader that delivered no + // samples at all, so the take had no camera in it. + expectFormat( + "4K request settles for the best uncompressed mode", + chooseWebcamFormat(kBrio, 3840, 2160, 30), + 1920, + 1080, + 30); + if (chooseWebcamFormat(kBrio, 3840, 2160, 30).compressed) { + std::printf("FAIL 4K request must not pin a compressed mode\n"); + ++failures; + } + + // A camera with nothing but compressed modes still has to be usable. + expectFormat( + "compressed-only camera is still driven", + chooseWebcamFormat({{1920, 1080, 30, true}, {1280, 720, 30, true}}, 1920, 1080, 30), + 1920, + 1080, + 30); + + // Uncompressed wins even when it is much smaller than a compressed option. + expectFormat( + "uncompressed beats a larger compressed mode", + chooseWebcamFormat({{3840, 2160, 30, true}, {640, 480, 30, false}}, 3840, 2160, 30), + 640, + 480, + 30); + + // 1080p60 exists too; at a 30 fps encode the extra frames only cost bandwidth. + expectFormat( + "prefers the frame rate we encode at", + chooseWebcamFormat({{1920, 1080, 60}, {1920, 1080, 30}}, 1920, 1080, 30), + 1920, + 1080, + 30); + + // Nothing reaches 30, so take the best the camera can do rather than fail. + expectFormat( + "falls back to the highest rate offered", + chooseWebcamFormat({{1920, 1080, 15}, {1920, 1080, 24}}, 1920, 1080, 30), + 1920, + 1080, + 24); + + // A 720p-only camera keeps working, just smaller. + expectFormat( + "720p camera stays 720p", + chooseWebcamFormat({{1280, 720, 30}, {640, 480, 30}}, 1920, 1080, 30), + 1280, + 720, + 30); + + // 2560x800 has fewer pixels than 1920x1080 but is far wider; picking it + // would letterbox the overlay and cost a needless downscale. + expectFormat( + "rejects an oversized dimension even when pixel count fits", + chooseWebcamFormat({{2560, 800, 30}, {1280, 720, 30}}, 1920, 1080, 30), + 1280, + 720, + 30); + + // Only modes above the target: scale down from the cheapest of them. + expectFormat( + "oversized-only camera takes the smallest mode", + chooseWebcamFormat({{3840, 2160, 30}, {2560, 1440, 30}}, 1920, 1080, 30), + 2560, + 1440, + 30); + + // Nothing to choose from: hand the target back and let the driver decide. + expectFormat("empty list returns the target", chooseWebcamFormat({}, 1920, 1080, 30), 1920, 1080, 30); + + // Drivers do report junk; it must not win. + expectFormat( + "ignores degenerate entries", + chooseWebcamFormat({{0, 0, 30}, {-1920, 1080, 30}, {1280, 720, 30}}, 1920, 1080, 30), + 1280, + 720, + 30); + + if (failures == 0) { + std::printf("webcam_format_test: all assertions passed\n"); + return 0; + } + std::printf("webcam_format_test: %d assertion(s) failed\n", failures); + return 1; +} diff --git a/scripts/build-windows-wgc-helper.mjs b/scripts/build-windows-wgc-helper.mjs index 8d4cfc018..b4e4a59bd 100644 --- a/scripts/build-windows-wgc-helper.mjs +++ b/scripts/build-windows-wgc-helper.mjs @@ -105,3 +105,13 @@ if (!fs.existsSync(audioUtilsTestPath)) { // Pack) instead of failing this packaging command. await run(audioUtilsTestPath, [], { cwd: BUILD_DIR }); console.log(`Passed ${audioUtilsTestPath}`); + +const webcamFormatTestPath = path.join(BUILD_DIR, "webcam_format_test.exe"); +if (!fs.existsSync(webcamFormatTestPath)) { + throw new Error(`WGC helper build completed but ${webcamFormatTestPath} was not found.`); +} +// Guards the capture resolution the camera is driven at. Left unpinned, both +// backends fall back to the device default -- 640x480 on hardware that offers +// far more -- and the overlay upscales it. +await run(webcamFormatTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${webcamFormatTestPath}`); diff --git a/scripts/test-windows-wgc-helper.mjs b/scripts/test-windows-wgc-helper.mjs index 785e48c1e..2bca55bd7 100644 --- a/scripts/test-windows-wgc-helper.mjs +++ b/scripts/test-windows-wgc-helper.mjs @@ -490,9 +490,13 @@ const config = { webcamDirectShowClsid: resolveDirectShowWebcamClsid( process.env.OPENSCREEN_WGC_TEST_WEBCAM_DEVICE_NAME ?? "", ), - webcamWidth: 640, - webcamHeight: 360, - webcamFps: 30, + // DEFAULT_WEBCAM_QUALITY from src/hooks/webcamCaptureTarget.ts -- the target the + // app actually sends, so this exercises the format negotiation rather than a + // size no shipped pipeline ever asks for. A camera that cannot reach it is + // driven at its own best format, which is the case worth covering anyway. + webcamWidth: Number(process.env.OPENSCREEN_WGC_TEST_WEBCAM_WIDTH ?? 3840), + webcamHeight: Number(process.env.OPENSCREEN_WGC_TEST_WEBCAM_HEIGHT ?? 2160), + webcamFps: Number(process.env.OPENSCREEN_WGC_TEST_WEBCAM_FPS ?? 30), outputs: { screenPath: outputPath, ...(webcamOutputPath ? { webcamPath: webcamOutputPath } : {}), @@ -631,9 +635,23 @@ const encoderSelectionLine = result.stdout .split(/\r?\n/) .find((line) => line.includes('"event":"encoder-selection"')); const encoderSelection = encoderSelectionLine ? JSON.parse(encoderSelectionLine) : null; -const nativeWebcamDiagnostics = result.stderr - .split(/\r?\n/) - .filter((line) => line.includes("Native webcam candidate")); +const nativeWebcamDiagnostics = result.stderr.split(/\r?\n/).filter( + (line) => + line.includes("Native webcam candidate") || + // Which capture format the camera was actually driven at. Without this + // the smoke test could pass on a 640x480 take from a 4K camera and say + // nothing about it. + line.includes("Native webcam format") || + line.includes("DirectShow webcam format") || + line.includes("DirectShow webcam connected") || + line.includes("falling back to the device default") || + // How long the camera took to produce its first frame, and the tally of what + // the capture loop did with everything it read. Without these, a camera that + // delivers nothing is only visible as a failed Finalize, which cannot say why. + line.includes("First webcam frame") || + line.includes("Webcam capture loop ended") || + line.includes("Webcam frame is"), +); const nativeMicrophoneDiagnostics = result.stderr .split(/\r?\n/) .filter( diff --git a/src/components/launch/HudDeviceSettings.quality.test.tsx b/src/components/launch/HudDeviceSettings.quality.test.tsx new file mode 100644 index 000000000..a44cf19b4 --- /dev/null +++ b/src/components/launch/HudDeviceSettings.quality.test.tsx @@ -0,0 +1,96 @@ +// @vitest-environment jsdom +import { fireEvent, render, screen } from "@testing-library/react"; +import { describe, expect, it, vi } from "vitest"; +import { WEBCAM_QUALITY_IDS } from "../../hooks/webcamCaptureTarget"; +import { HudDeviceSettings, type HudDeviceSettingsLabels } from "./HudDeviceSettings"; + +vi.mock("../../hooks/useAudioLevelMeter", () => ({ + useAudioLevelMeter: () => ({ level: 0 }), +})); +vi.mock("../../hooks/useCameraPreviewStream", () => ({ + useCameraPreviewStream: () => ({ stream: null, error: null }), +})); + +const labels: HudDeviceSettingsLabels = { + title: "Device settings", + done: "Done", + microphone: "Microphone", + camera: "Camera", + micLevel: "Input level", + micHint: "Speak to check", + noMicrophones: "No microphone found", + searching: "Searching...", + noCameras: "No camera found", + cameraUnavailable: "Camera unavailable", + preview: "Preview", + previewUnavailable: "Preview unavailable", + about: "About", + checkForUpdates: "Check for updates", + checkingForUpdates: "Checking…", + cameraQuality: "Camera quality", + cameraQualityOptions: { + "1080p": "1080p", + "1440p": "1440p", + "2160p": "4K", + }, +}; + +function renderPanel(overrides: Partial[0]> = {}) { + const onSelectCameraQuality = vi.fn(); + render( + undefined} + {...overrides} + />, + ); + return { onSelectCameraQuality }; +} + +describe("HudDeviceSettings camera quality", () => { + it("offers every preset the capture pipeline knows about", () => { + renderPanel(); + + for (const id of WEBCAM_QUALITY_IDS) { + expect(screen.getByTestId(`camera-quality-${id}`)).toBeTruthy(); + } + }); + + it("marks the stored choice as the checked one", () => { + renderPanel(); + + expect(screen.getByTestId("camera-quality-1440p").getAttribute("aria-checked")).toBe("true"); + expect(screen.getByTestId("camera-quality-2160p").getAttribute("aria-checked")).toBe("false"); + }); + + it("reports the picked preset so it can be persisted", () => { + const { onSelectCameraQuality } = renderPanel(); + + fireEvent.click(screen.getByTestId("camera-quality-2160p")); + + expect(onSelectCameraQuality).toHaveBeenCalledWith("2160p"); + }); + + it("hides the choice when there is no camera to apply it to", () => { + // A resolution picker above "No camera found" is a control that cannot do + // anything, and it pushes the real message out of view. + renderPanel({ cameraDevices: [], activeCameraId: undefined }); + + expect(screen.queryByTestId("camera-quality-2160p")).toBeNull(); + }); +}); diff --git a/src/components/launch/HudDeviceSettings.tsx b/src/components/launch/HudDeviceSettings.tsx index d807f752b..3bf43df74 100644 --- a/src/components/launch/HudDeviceSettings.tsx +++ b/src/components/launch/HudDeviceSettings.tsx @@ -4,6 +4,7 @@ import { useAudioLevelMeter } from "../../hooks/useAudioLevelMeter"; import type { CameraDevice } from "../../hooks/useCameraDevices"; import { useCameraPreviewStream } from "../../hooks/useCameraPreviewStream"; import type { MicrophoneDevice } from "../../hooks/useMicrophoneDevices"; +import { WEBCAM_QUALITY_IDS, type WebcamQualityId } from "../../hooks/webcamCaptureTarget"; import styles from "./LaunchWindow.module.css"; const LEVEL_SEGMENTS = 12; @@ -25,6 +26,8 @@ export interface HudDeviceSettingsLabels { about: string; checkForUpdates: string; checkingForUpdates: string; + cameraQuality: string; + cameraQualityOptions: Record; } /** Segmented input-level bar, driven by the live analyser. */ @@ -96,6 +99,8 @@ export const HudDeviceSettings = memo(function HudDeviceSettings({ versionLabel, canCheckForUpdates, checkingForUpdates, + cameraQuality, + onSelectCameraQuality, onSelectMic, onSelectCamera, onCheckForUpdates, @@ -116,6 +121,8 @@ export const HudDeviceSettings = memo(function HudDeviceSettings({ versionLabel: string | null; canCheckForUpdates: boolean; checkingForUpdates: boolean; + cameraQuality: WebcamQualityId; + onSelectCameraQuality: (quality: WebcamQualityId) => void; onSelectMic: (device: MicrophoneDevice) => void; onSelectCamera: (device: CameraDevice) => void; onCheckForUpdates: () => void; @@ -214,6 +221,29 @@ export const HudDeviceSettings = memo(function HudDeviceSettings({ )} {hasCamera ? ( <> + {/* Below the device list, because it qualifies the camera picked + above. Hidden with no camera present -- a resolution control + over "No camera found" cannot do anything. */} +
+ {labels.cameraQuality} +
+ {WEBCAM_QUALITY_IDS.map((quality) => { + const isActive = quality === cameraQuality; + return ( + + ); + })}
{labels.preview}
diff --git a/src/components/launch/LaunchWindow.tsx b/src/components/launch/LaunchWindow.tsx index b7fcc9715..5ddffd89e 100644 --- a/src/components/launch/LaunchWindow.tsx +++ b/src/components/launch/LaunchWindow.tsx @@ -13,6 +13,7 @@ import { } from "../../hooks/useMicrophoneDevices"; import { usePortalOwnsSource } from "../../hooks/usePortalOwnsSource"; import { useScreenRecorder } from "../../hooks/useScreenRecorder"; +import type { WebcamQualityId } from "../../hooks/webcamCaptureTarget"; import { requestCameraAccess } from "../../lib/requestCameraAccess"; import { HudCameraButton, @@ -111,6 +112,8 @@ export function LaunchWindow() { setWebcamEnabled, webcamDeviceId, setWebcamDeviceId, + webcamQuality, + setWebcamQuality, webcamDeviceName, setWebcamDeviceName, cursorCaptureMode, @@ -862,6 +865,7 @@ export function LaunchWindow() { camEnabled?: boolean; camDeviceId?: string; camDeviceName?: string; + camQuality?: WebcamQualityId; micEnabled?: boolean; micDeviceId?: string; micDeviceName?: string; @@ -927,6 +931,14 @@ export function LaunchWindow() { [persistRecordingPrefs, setSelectedCameraId, setWebcamDeviceId, setWebcamDeviceName], ); + const handleSelectCameraQuality = useCallback( + (quality: WebcamQualityId) => { + setWebcamQuality(quality); + persistRecordingPrefs({ camQuality: quality }); + }, + [persistRecordingPrefs, setWebcamQuality], + ); + const toggleDeviceSettings = useCallback(() => { if (controlsLocked) return; setIsLanguageMenuOpen(false); @@ -1068,6 +1080,12 @@ export function LaunchWindow() { about: t("deviceSettings.about"), checkForUpdates: tCommon("actions.checkForUpdates"), checkingForUpdates: t("deviceSettings.checkingForUpdates"), + cameraQuality: t("webcam.quality"), + cameraQualityOptions: { + "1080p": t("webcam.quality1080p"), + "1440p": t("webcam.quality1440p"), + "2160p": t("webcam.quality2160p"), + }, }), [t, tCommon], ); @@ -1282,6 +1300,8 @@ export function LaunchWindow() { // main process refuses the check then — an offered button would be dead. canCheckForUpdates={(appInfo?.canCheckForUpdates ?? false) && !recording} checkingForUpdates={isCheckingForUpdates} + cameraQuality={webcamQuality} + onSelectCameraQuality={handleSelectCameraQuality} onSelectMic={handleSelectMicDevice} onSelectCamera={handleSelectCameraDevice} onCheckForUpdates={handleCheckForUpdates} diff --git a/src/hooks/useCameraPreviewStream.ts b/src/hooks/useCameraPreviewStream.ts index 9080e7c8d..9e04c0558 100644 --- a/src/hooks/useCameraPreviewStream.ts +++ b/src/hooks/useCameraPreviewStream.ts @@ -1,4 +1,5 @@ import { useEffect, useRef, useState } from "react"; +import { webcamVideoConstraints } from "./webcamCaptureTarget"; export interface CameraPreviewStreamOptions { enabled: boolean; @@ -28,7 +29,13 @@ export function useCameraPreviewStream({ enabled, deviceId }: CameraPreviewStrea let cancelled = false; navigator.mediaDevices .getUserMedia({ - video: deviceId ? { deviceId: { exact: deviceId } } : true, + // Pinned to the smallest preset, not to whatever the recording is set + // to. Left unconstrained this opened at 640x480 and made the camera + // look as soft in the HUD as it did in the take; driving it at the + // user's 4K choice would be the opposite mistake, since this renders + // into a thumbnail a couple of hundred pixels wide and the recorder + // opens its own stream anyway. + video: webcamVideoConstraints(deviceId, "1080p"), audio: false, }) .then((s) => { diff --git a/src/hooks/useScreenRecorder.ts b/src/hooks/useScreenRecorder.ts index 0d1460412..b3b3ed9c3 100644 --- a/src/hooks/useScreenRecorder.ts +++ b/src/hooks/useScreenRecorder.ts @@ -21,6 +21,15 @@ import { requestCameraAccess } from "@/lib/requestCameraAccess"; import { loadUserPreferences, saveUserPreferences } from "@/lib/userPreferences"; import { canRecordMicrophone } from "@/utils/platformUtils"; import { createRecorderHandle, type RecorderHandle } from "./recorderHandle"; +import { + DEFAULT_WEBCAM_QUALITY, + WEBCAM_TARGET_FRAME_RATE, + type WebcamQualityId, + webcamBitrateForStream, + webcamPresetFor, + webcamQualityFrom, + webcamVideoConstraints, +} from "./webcamCaptureTarget"; import { webcamDeviceIdentityFrom } from "./webcamDeviceIdentity"; const TARGET_FRAME_RATE = 60; @@ -77,8 +86,6 @@ function effectiveBrowserCursorMode( const AUDIO_BITRATE_VOICE = 128_000; const AUDIO_BITRATE_SYSTEM = 192_000; -const WEBCAM_TARGET_FRAME_RATE = 30; - type UseScreenRecorderReturn = { recording: boolean; paused: boolean; @@ -99,6 +106,8 @@ type UseScreenRecorderReturn = { setMicrophoneDeviceName: (deviceName: string | undefined) => void; webcamDeviceId: string | undefined; setWebcamDeviceId: (deviceId: string | undefined) => void; + webcamQuality: WebcamQualityId; + setWebcamQuality: (quality: WebcamQualityId) => void; webcamDeviceName: string | undefined; setWebcamDeviceName: (deviceName: string | undefined) => void; systemAudioEnabled: boolean; @@ -251,6 +260,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { const [microphoneDeviceId, setMicrophoneDeviceId] = useState(undefined); const [microphoneDeviceName, setMicrophoneDeviceName] = useState(undefined); const [webcamDeviceId, setWebcamDeviceId] = useState(undefined); + const [webcamQuality, setWebcamQuality] = useState(DEFAULT_WEBCAM_QUALITY); const [webcamDeviceName, setWebcamDeviceName] = useState(undefined); const [systemAudioEnabled, setSystemAudioEnabled] = useState(false); const [webcamEnabled, setWebcamEnabledState] = useState(false); @@ -275,6 +285,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { camEnabled: boolean; camDeviceId?: string | null; camDeviceName?: string | null; + camQuality?: WebcamQualityId | null; systemAudioEnabled: boolean; cursorCaptureMode: CursorCaptureMode; }) => { @@ -292,6 +303,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { setWebcamDeviceId(prefs.camDeviceId ?? undefined); setWebcamDeviceName(prefs.camDeviceName ?? undefined); } + setWebcamQuality(webcamQualityFrom(prefs.camQuality)); setSystemAudioEnabled(prefs.systemAudioEnabled); setCursorCaptureMode(prefs.cursorCaptureMode); setRecordingPrefsLoaded(true); @@ -473,14 +485,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { try { const stream = await navigator.mediaDevices.getUserMedia({ audio: false, - video: webcamDeviceId - ? { - deviceId: { exact: webcamDeviceId }, - frameRate: { ideal: WEBCAM_TARGET_FRAME_RATE, max: WEBCAM_TARGET_FRAME_RATE }, - } - : { - frameRate: { ideal: WEBCAM_TARGET_FRAME_RATE, max: WEBCAM_TARGET_FRAME_RATE }, - }, + video: webcamVideoConstraints(webcamDeviceId, webcamQuality), }); if (cancelled || thisAcquireId !== webcamAcquireId.current) { @@ -538,7 +543,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { webcamStream.current = null; } }; - }, [webcamEnabled, webcamDeviceId, webcamDeviceName, t]); + }, [webcamEnabled, webcamDeviceId, webcamDeviceName, webcamQuality, t]); const finalizeRecording = useCallback( ( @@ -1233,8 +1238,8 @@ export function useScreenRecorder(): UseScreenRecorderReturn { enabled: webcamEnabled, deviceId: webcamIdentity.deviceId, deviceName: webcamIdentity.deviceName, - width: 0, - height: 0, + width: webcamPresetFor(webcamQuality).width, + height: webcamPresetFor(webcamQuality).height, fps: WEBCAM_TARGET_FRAME_RATE, }, cursor: { @@ -1351,7 +1356,10 @@ export function useScreenRecorder(): UseScreenRecorderReturn { webcamStream.current, { mimeType: selectMimeType(), - videoBitsPerSecond: BITRATE_BASE, + // Sized from the track, not from BITRATE_BASE: that was the + // screen's rate and it starves a 1440p or 2160p camera frame + // badly enough to undo the resolution we just asked for. + videoBitsPerSecond: webcamBitrateForStream(webcamStream.current), }, `${RECORDING_FILE_PREFIX}${activeRecordingId}${WEBCAM_FILE_SUFFIX}${VIDEO_FILE_EXTENSION}`, ); @@ -1396,8 +1404,8 @@ export function useScreenRecorder(): UseScreenRecorderReturn { // Same pairing rule as the Windows path; here the stream is still // open, so the identity can be read at the point of use. ...readWebcamDeviceIdentity(), - width: 0, - height: 0, + width: webcamPresetFor(webcamQuality).width, + height: webcamPresetFor(webcamQuality).height, fps: WEBCAM_TARGET_FRAME_RATE, }, cursor: { @@ -1550,7 +1558,10 @@ export function useScreenRecorder(): UseScreenRecorderReturn { webcamStream.current, { mimeType: selectMimeType(), - videoBitsPerSecond: BITRATE_BASE, + // Sized from the track, not from BITRATE_BASE: that was the + // screen's rate and it starves a 1440p or 2160p camera frame + // badly enough to undo the resolution we just asked for. + videoBitsPerSecond: webcamBitrateForStream(webcamStream.current), }, `${RECORDING_FILE_PREFIX}${activeRecordingId}${WEBCAM_FILE_SUFFIX}${VIDEO_FILE_EXTENSION}`, ); @@ -2396,6 +2407,8 @@ export function useScreenRecorder(): UseScreenRecorderReturn { setMicrophoneDeviceName, webcamDeviceId, setWebcamDeviceId, + webcamQuality, + setWebcamQuality, webcamDeviceName, setWebcamDeviceName, systemAudioEnabled, diff --git a/src/hooks/webcamCaptureTarget.test.ts b/src/hooks/webcamCaptureTarget.test.ts new file mode 100644 index 000000000..25f83efdd --- /dev/null +++ b/src/hooks/webcamCaptureTarget.test.ts @@ -0,0 +1,137 @@ +import { describe, expect, it } from "vitest"; +import { + DEFAULT_WEBCAM_QUALITY, + WEBCAM_QUALITY_IDS, + WEBCAM_QUALITY_PRESETS, + WEBCAM_TARGET_FRAME_RATE, + webcamBitrateFor, + webcamBitrateForStream, + webcamPresetFor, + webcamQualityFrom, + webcamVideoConstraints, +} from "./webcamCaptureTarget"; + +describe("webcamVideoConstraints", () => { + it("asks for the chosen preset, not the browser's 640x480 default", () => { + expect(webcamVideoConstraints(undefined, "2160p")).toMatchObject({ + width: { ideal: 3840 }, + height: { ideal: 2160 }, + }); + expect(webcamVideoConstraints(undefined, "1080p")).toMatchObject({ + width: { ideal: 1920 }, + height: { ideal: 1080 }, + }); + }); + + it("keeps the size negotiable so a 720p-only camera still opens", () => { + // `exact` would make getUserMedia throw OverconstrainedError on every + // camera that cannot hit the target, losing the webcam entirely rather + // than recording it a little smaller. + const constraints = webcamVideoConstraints(undefined, "2160p"); + + expect(constraints.width).not.toHaveProperty("exact"); + expect(constraints.height).not.toHaveProperty("exact"); + expect(constraints.width).not.toHaveProperty("min"); + expect(constraints.height).not.toHaveProperty("min"); + }); + + it("still pins the chosen device and caps the frame rate", () => { + const constraints = webcamVideoConstraints("camera-7", "1440p"); + + expect(constraints.deviceId).toEqual({ exact: "camera-7" }); + expect(constraints.frameRate).toEqual({ + ideal: WEBCAM_TARGET_FRAME_RATE, + max: WEBCAM_TARGET_FRAME_RATE, + }); + }); + + it("omits deviceId entirely when no camera was chosen", () => { + expect(webcamVideoConstraints(undefined, "1080p")).not.toHaveProperty("deviceId"); + }); + + it("falls back to the default preset rather than an unconstrained request", () => { + // An unrecognised stored value must not reopen the 640x480 hole. + expect(webcamVideoConstraints(undefined, undefined)).toEqual( + webcamVideoConstraints(undefined, DEFAULT_WEBCAM_QUALITY), + ); + }); +}); + +describe("webcamQualityFrom", () => { + it("accepts every advertised preset", () => { + for (const id of WEBCAM_QUALITY_IDS) { + expect(webcamQualityFrom(id)).toBe(id); + } + }); + + it("rejects anything else, including settings files from a future build", () => { + expect(webcamQualityFrom("4320p")).toBe(DEFAULT_WEBCAM_QUALITY); + expect(webcamQualityFrom(undefined)).toBe(DEFAULT_WEBCAM_QUALITY); + expect(webcamQualityFrom(null)).toBe(DEFAULT_WEBCAM_QUALITY); + expect(webcamQualityFrom(1080)).toBe(DEFAULT_WEBCAM_QUALITY); + }); +}); + +describe("webcamPresetFor", () => { + it("never returns a size below 1080p", () => { + for (const id of WEBCAM_QUALITY_IDS) { + const preset = webcamPresetFor(id); + expect(preset.width).toBeGreaterThanOrEqual(1920); + expect(preset.height).toBeGreaterThanOrEqual(1080); + } + expect(Object.keys(WEBCAM_QUALITY_PRESETS)).toEqual([...WEBCAM_QUALITY_IDS]); + }); +}); + +describe("webcamBitrateFor", () => { + it("scales with the frame the camera actually delivers", () => { + // A flat rate sized for 640x480 is what made the extra pixels pointless: + // starve a 2160p frame and it blocks up worse than a clean 1080p one. + expect(webcamBitrateFor(3840, 2160)).toBeGreaterThan(webcamBitrateFor(2560, 1440)); + expect(webcamBitrateFor(2560, 1440)).toBeGreaterThan(webcamBitrateFor(1920, 1080)); + expect(webcamBitrateFor(1920, 1080)).toBeGreaterThan(webcamBitrateFor(1280, 720)); + expect(webcamBitrateFor(1280, 720)).toBeGreaterThan(webcamBitrateFor(640, 480)); + }); + + it("gives 4K enough headroom to be worth capturing", () => { + expect(webcamBitrateFor(3840, 2160)).toBe(40_000_000); + }); + + it("matches the tier boundaries the native helper uses", () => { + // One pixel under a tier must not claim that tier's rate; the ladder in + // wgc-capture/src/main.cpp draws the lines at the same places. + expect(webcamBitrateFor(2560, 1439)).toBe(webcamBitrateFor(1920, 1080)); + expect(webcamBitrateFor(1920, 1079)).toBe(webcamBitrateFor(1280, 720)); + }); + + it("rates a small camera by what it delivered, not by what we asked for", () => { + // The preset is a request. A 720p camera asked for 2160p must not be + // handed a 4K bitrate it cannot fill. + expect(webcamBitrateFor(1280, 720)).toBe(8_000_000); + }); + + it("survives a track that reports nothing", () => { + // getSettings() may omit width/height before the first frame arrives. + const preset = WEBCAM_QUALITY_PRESETS[DEFAULT_WEBCAM_QUALITY]; + expect(webcamBitrateFor(undefined, undefined)).toBe( + webcamBitrateFor(preset.width, preset.height), + ); + expect(webcamBitrateFor(0, 0)).toBeGreaterThan(0); + }); +}); + +describe("webcamBitrateForStream", () => { + const streamWith = (settings: MediaTrackSettings) => + ({ getVideoTracks: () => [{ getSettings: () => settings }] }) as unknown as MediaStream; + + it("reads the size off the live track", () => { + expect(webcamBitrateForStream(streamWith({ width: 2560, height: 1440 }))).toBe(24_000_000); + }); + + it("falls back when there is no stream or no video track", () => { + expect(webcamBitrateForStream(null)).toBe(webcamBitrateFor(undefined, undefined)); + expect(webcamBitrateForStream({ getVideoTracks: () => [] } as unknown as MediaStream)).toBe( + webcamBitrateFor(undefined, undefined), + ); + }); +}); diff --git a/src/hooks/webcamCaptureTarget.ts b/src/hooks/webcamCaptureTarget.ts new file mode 100644 index 000000000..906412b19 --- /dev/null +++ b/src/hooks/webcamCaptureTarget.ts @@ -0,0 +1,104 @@ +/** + * The resolution the webcam is captured at, and the rate it is encoded at. + * + * Nothing here used to name a size. `getUserMedia({ video: { deviceId } })` and + * a Media Foundation source reader left without MF_MT_FRAME_SIZE both fall back + * to the device's *default* media type, and for a UVC camera that is the first + * format it enumerates -- 640x480. A Logitech BRIO that offers MJPEG up to + * 3840x2160 was therefore recorded at 640x480 and then scaled up into the + * picture-in-picture, which is what "the camera looks pixelated" actually was. + */ + +export const WEBCAM_TARGET_FRAME_RATE = 30; + +export const WEBCAM_QUALITY_IDS = ["1080p", "1440p", "2160p"] as const; +export type WebcamQualityId = (typeof WEBCAM_QUALITY_IDS)[number]; + +export interface WebcamQualityPreset { + id: WebcamQualityId; + width: number; + height: number; +} + +/** + * Sizes are a *request*, never a floor. A camera that tops out below the chosen + * preset is driven at the best format it has; the native helper picks that with + * `chooseWebcamFormat`, and the browser does it through `ideal` constraints. + * + * Measured on a BRIO (Snapdragon X, alongside a screen capture), 8s takes, + * frames the camera actually delivered out of 240, over four runs each: + * 1080p 244 14.2 Mbit/s 4/4 + * 1440p 242-244 21.9 Mbit/s 4/4 + * 2160p 243-244 38.4 Mbit/s 4/4 + * + * All three run at the camera's full rate. That is recent: driving the capture + * as RGB32 made Media Foundation decode and convert every frame, costing 92ms + * per 2160p frame -- a ceiling near 11 fps, inside a file whose container still + * claimed 30 because the encoder padded the gap with duplicates. The capture + * now asks for NV12, which the camera produces directly and the encoder + * consumes directly; see webcam_capture.cpp. Higher presets cost file size, not + * frames, which is why this is a user choice. + */ +export const WEBCAM_QUALITY_PRESETS: Record = { + "1080p": { id: "1080p", width: 1920, height: 1080 }, + "1440p": { id: "1440p", width: 2560, height: 1440 }, + "2160p": { id: "2160p", width: 3840, height: 2160 }, +}; + +export const DEFAULT_WEBCAM_QUALITY: WebcamQualityId = "2160p"; + +/** Narrows a persisted or IPC-delivered value to a preset we can act on. */ +export function webcamQualityFrom(value: unknown): WebcamQualityId { + return WEBCAM_QUALITY_IDS.includes(value as WebcamQualityId) + ? (value as WebcamQualityId) + : DEFAULT_WEBCAM_QUALITY; +} + +export function webcamPresetFor(quality: WebcamQualityId | undefined): WebcamQualityPreset { + return WEBCAM_QUALITY_PRESETS[webcamQualityFrom(quality)]; +} + +/** + * Video constraints for a webcam stream, optionally pinned to one device. + * + * Sizes are `ideal`, never `exact` or `min`: a camera that tops out at 720p has + * to keep working. `exact` turns "record it a bit smaller" into + * OverconstrainedError, and the recording then has no camera at all. + */ +export function webcamVideoConstraints( + deviceId: string | undefined, + quality?: WebcamQualityId, +): MediaTrackConstraints { + const preset = webcamPresetFor(quality); + return { + ...(deviceId ? { deviceId: { exact: deviceId } } : {}), + width: { ideal: preset.width }, + height: { ideal: preset.height }, + frameRate: { ideal: WEBCAM_TARGET_FRAME_RATE, max: WEBCAM_TARGET_FRAME_RATE }, + }; +} + +/** + * Encoder bitrate for a frame the camera actually delivered. + * + * Keyed off the delivered size rather than the requested preset: a 720p camera + * asked for 2160p must not be encoded as if it were 4K. The tiers mirror the + * native ladder in electron/native/wgc-capture/src/main.cpp -- keep them in step. + */ +export function webcamBitrateFor(width: number | undefined, height: number | undefined): number { + const preset = WEBCAM_QUALITY_PRESETS[DEFAULT_WEBCAM_QUALITY]; + const pixels = + width && height && width > 0 && height > 0 ? width * height : preset.width * preset.height; + + if (pixels >= 3840 * 2160) return 40_000_000; + if (pixels >= 2560 * 1440) return 24_000_000; + if (pixels >= 1920 * 1080) return 16_000_000; + if (pixels >= 1280 * 720) return 8_000_000; + return 4_000_000; +} + +/** The bitrate for whatever the stream's first video track ended up delivering. */ +export function webcamBitrateForStream(stream: MediaStream | null): number { + const settings = stream?.getVideoTracks()[0]?.getSettings(); + return webcamBitrateFor(settings?.width, settings?.height); +} diff --git a/src/i18n/locales/ar/launch.json b/src/i18n/locales/ar/launch.json index 2b3db0f7e..b1dc21aff 100644 --- a/src/i18n/locales/ar/launch.json +++ b/src/i18n/locales/ar/launch.json @@ -50,7 +50,11 @@ "noneFound": "لم يتم العثور على كاميرا", "unavailable": "الكاميرا غير متوفرة", "camera": "الكاميرا", - "cameraDevice": "جهاز الكاميرا" + "cameraDevice": "جهاز الكاميرا", + "quality": "جودة الكاميرا", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "sourceSelector": { "loading": "جاري تحميل المصادر...", diff --git a/src/i18n/locales/cs/launch.json b/src/i18n/locales/cs/launch.json index b3ae9a56f..d4d3f3104 100644 --- a/src/i18n/locales/cs/launch.json +++ b/src/i18n/locales/cs/launch.json @@ -50,7 +50,11 @@ "noneFound": "Kamera nenalezena", "unavailable": "Kamera nedostupná", "camera": "Kamera", - "cameraDevice": "Zařízení kamery" + "cameraDevice": "Zařízení kamery", + "quality": "Kvalita kamery", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Použít upravitelný kurzor: znovu zapne automatické přiblížení a efekty kurzoru", diff --git a/src/i18n/locales/de/launch.json b/src/i18n/locales/de/launch.json index 81c6eeec0..8d199afaa 100644 --- a/src/i18n/locales/de/launch.json +++ b/src/i18n/locales/de/launch.json @@ -50,7 +50,11 @@ "noneFound": "Keine Kamera gefunden", "unavailable": "Kamera nicht verfügbar", "camera": "Kamera", - "cameraDevice": "Kameragerät" + "cameraDevice": "Kameragerät", + "quality": "Kameraqualität", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Bearbeitbaren Cursor verwenden: schaltet Auto-Zoom und Cursor-Effekte wieder ein", diff --git a/src/i18n/locales/en/launch.json b/src/i18n/locales/en/launch.json index 3963513c3..e02d96f8e 100644 --- a/src/i18n/locales/en/launch.json +++ b/src/i18n/locales/en/launch.json @@ -50,7 +50,11 @@ "noneFound": "No camera found", "unavailable": "Camera unavailable", "camera": "Camera", - "cameraDevice": "Camera device" + "cameraDevice": "Camera device", + "quality": "Camera quality", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Use editable cursor: turns auto zoom and cursor effects back on", diff --git a/src/i18n/locales/es/launch.json b/src/i18n/locales/es/launch.json index 348f42b5e..34965561c 100644 --- a/src/i18n/locales/es/launch.json +++ b/src/i18n/locales/es/launch.json @@ -50,7 +50,11 @@ "noneFound": "No se encontró cámara", "unavailable": "Cámara no disponible", "camera": "Cámara", - "cameraDevice": "Dispositivo de cámara" + "cameraDevice": "Dispositivo de cámara", + "quality": "Calidad de la cámara", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Usar cursor editable: vuelve a activar el zoom automático y los efectos del cursor", diff --git a/src/i18n/locales/fr/launch.json b/src/i18n/locales/fr/launch.json index 5782d07e6..3a8d35690 100644 --- a/src/i18n/locales/fr/launch.json +++ b/src/i18n/locales/fr/launch.json @@ -50,7 +50,11 @@ "noneFound": "Aucune caméra trouvée", "unavailable": "Caméra non disponible", "camera": "Caméra", - "cameraDevice": "Périphérique caméra" + "cameraDevice": "Périphérique caméra", + "quality": "Qualité de la caméra", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Utiliser le curseur éditable : réactive le zoom automatique et les effets de curseur", diff --git a/src/i18n/locales/it/launch.json b/src/i18n/locales/it/launch.json index 2c863de0f..fae376c8d 100644 --- a/src/i18n/locales/it/launch.json +++ b/src/i18n/locales/it/launch.json @@ -50,7 +50,11 @@ "noneFound": "Nessuna fotocamera trovata", "unavailable": "Fotocamera non disponibile", "camera": "Fotocamera", - "cameraDevice": "Dispositivo fotocamera" + "cameraDevice": "Dispositivo fotocamera", + "quality": "Qualità della fotocamera", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Usa cursore modificabile: riattiva lo zoom automatico e gli effetti del cursore", diff --git a/src/i18n/locales/ja-JP/launch.json b/src/i18n/locales/ja-JP/launch.json index 20c7e0530..0429770d8 100644 --- a/src/i18n/locales/ja-JP/launch.json +++ b/src/i18n/locales/ja-JP/launch.json @@ -50,7 +50,11 @@ "noneFound": "カメラが見つかりません", "unavailable": "カメラが利用できません", "camera": "カメラ", - "cameraDevice": "カメラデバイス" + "cameraDevice": "カメラデバイス", + "quality": "カメラの画質", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "編集可能なカーソルを使う:自動ズームとカーソル効果が再びオンになります", diff --git a/src/i18n/locales/ko-KR/launch.json b/src/i18n/locales/ko-KR/launch.json index 1aef23e07..70f0886dc 100644 --- a/src/i18n/locales/ko-KR/launch.json +++ b/src/i18n/locales/ko-KR/launch.json @@ -50,7 +50,11 @@ "noneFound": "카메라를 찾을 수 없음", "unavailable": "카메라를 사용할 수 없음", "camera": "카메라", - "cameraDevice": "카메라 장치" + "cameraDevice": "카메라 장치", + "quality": "카메라 화질", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "편집 가능한 커서 사용: 자동 줌과 커서 효과가 다시 켜집니다", diff --git a/src/i18n/locales/pt-BR/launch.json b/src/i18n/locales/pt-BR/launch.json index 35aedb23a..0ed727214 100644 --- a/src/i18n/locales/pt-BR/launch.json +++ b/src/i18n/locales/pt-BR/launch.json @@ -50,7 +50,11 @@ "noneFound": "Nenhuma câmera encontrada", "unavailable": "Câmera indisponível", "camera": "Câmera", - "cameraDevice": "Dispositivo de câmera" + "cameraDevice": "Dispositivo de câmera", + "quality": "Qualidade da câmera", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Usar cursor editável: reativa o zoom automático e os efeitos do cursor", diff --git a/src/i18n/locales/ru/launch.json b/src/i18n/locales/ru/launch.json index 48ca06b7e..67ebf3da8 100644 --- a/src/i18n/locales/ru/launch.json +++ b/src/i18n/locales/ru/launch.json @@ -50,7 +50,11 @@ "noneFound": "Камера не найдена", "unavailable": "Камера недоступна", "camera": "Камера", - "cameraDevice": "Устройство камеры" + "cameraDevice": "Устройство камеры", + "quality": "Качество камеры", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "sourceSelector": { "loading": "Загрузка источников...", diff --git a/src/i18n/locales/tr/launch.json b/src/i18n/locales/tr/launch.json index e919ba448..7df10beb9 100644 --- a/src/i18n/locales/tr/launch.json +++ b/src/i18n/locales/tr/launch.json @@ -50,7 +50,11 @@ "noneFound": "Kamera bulunamadı", "unavailable": "Kamera kullanılamıyor", "camera": "Kamera", - "cameraDevice": "Kamera cihazı" + "cameraDevice": "Kamera cihazı", + "quality": "Kamera kalitesi", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "Düzenlenebilir imleci kullan: otomatik yakınlaştırmayı ve imleç efektlerini yeniden açar", diff --git a/src/i18n/locales/vi/launch.json b/src/i18n/locales/vi/launch.json index 4bc27b80a..a76f0db84 100644 --- a/src/i18n/locales/vi/launch.json +++ b/src/i18n/locales/vi/launch.json @@ -50,7 +50,11 @@ "noneFound": "Không tìm thấy máy ảnh", "unavailable": "Máy ảnh không khả dụng", "camera": "Máy ảnh", - "cameraDevice": "Thiết bị máy ảnh" + "cameraDevice": "Thiết bị máy ảnh", + "quality": "Chất lượng camera", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "sourceSelector": { "loading": "Đang tải nguồn...", diff --git a/src/i18n/locales/zh-CN/launch.json b/src/i18n/locales/zh-CN/launch.json index 4ee356acf..7a0b26cb9 100644 --- a/src/i18n/locales/zh-CN/launch.json +++ b/src/i18n/locales/zh-CN/launch.json @@ -50,7 +50,11 @@ "noneFound": "未找到摄像头", "unavailable": "摄像头不可用", "camera": "摄像头", - "cameraDevice": "摄像头设备" + "cameraDevice": "摄像头设备", + "quality": "摄像头画质", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "使用可编辑光标:将重新开启自动缩放和光标效果", diff --git a/src/i18n/locales/zh-TW/launch.json b/src/i18n/locales/zh-TW/launch.json index a899f50e0..423d585d6 100644 --- a/src/i18n/locales/zh-TW/launch.json +++ b/src/i18n/locales/zh-TW/launch.json @@ -50,7 +50,11 @@ "noneFound": "未找到攝影機", "unavailable": "攝影機不可用", "camera": "攝影機", - "cameraDevice": "攝影機裝置" + "cameraDevice": "攝影機裝置", + "quality": "攝影機畫質", + "quality1080p": "1080p", + "quality1440p": "1440p", + "quality2160p": "4K" }, "cursor": { "useEditableCursorHint": "使用可編輯游標:將重新開啟自動縮放與游標效果", From 597fe5e30ca2cc4c2cd5e9bba4e42c451d1387bf Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:39:59 +0200 Subject: [PATCH 2/8] fix(recording): ignore a camera quality change made during a take The gear is disabled while recording, but a settings panel already open stays mounted, so its resolution options stayed clickable. `webcamQuality` is a dependency of the webcam acquisition effect: changing it re-runs that effect's cleanup, which stops every track of the live stream -- the same stream the browser, macOS and Linux paths hand to the webcam MediaRecorder. The camera ended partway through the take, without a toast or any other word to the user. Guarded with `controlsLocked`, as every other control in this component already is. The test mock for `useCameraDevices` gains a configurable device list: the panel only renders its camera controls when a camera exists, so a test reaching for them could not previously do so. Reported by CodeRabbit on #875. --- src/components/launch/LaunchWindow.test.tsx | 32 ++++++++++++++++++++- src/components/launch/LaunchWindow.tsx | 10 ++++++- 2 files changed, 40 insertions(+), 2 deletions(-) diff --git a/src/components/launch/LaunchWindow.test.tsx b/src/components/launch/LaunchWindow.test.tsx index ae4484864..6af724a8e 100644 --- a/src/components/launch/LaunchWindow.test.tsx +++ b/src/components/launch/LaunchWindow.test.tsx @@ -59,6 +59,8 @@ const recorderState = vi.hoisted(() => ({ webcamDeviceId: undefined, setWebcamDeviceId: vi.fn(), setWebcamDeviceName: vi.fn(), + webcamQuality: "2160p", + setWebcamQuality: vi.fn(), systemAudioEnabled: false, setSystemAudioEnabled: vi.fn(), cursorCaptureMode: "editable-overlay", @@ -96,11 +98,15 @@ vi.mock("../../hooks/useMicrophoneDevices", () => ({ const cameraDevicesState = vi.hoisted(() => ({ isReady: true, + // Empty by default, as before. The device-settings panel only renders its + // camera controls when a camera exists, so tests that reach for them fill + // this in. + devices: [] as Array<{ deviceId: string; label: string }>, })); vi.mock("../../hooks/useCameraDevices", () => ({ useCameraDevices: () => ({ - devices: [], + devices: cameraDevicesState.devices, selectedDeviceId: "", setSelectedDeviceId: vi.fn(), isLoading: false, @@ -320,6 +326,7 @@ function resetLaunchMocks() { recorderState.value.setWebcamEnabled.mockClear(); recorderState.value.recordingPrefsLoaded = true; cameraDevicesState.isReady = true; + cameraDevicesState.devices = []; micDevicesState.value = []; micDevicesState.enabled = undefined; systemVersionState.value = "15.5"; @@ -1437,6 +1444,29 @@ describe("LaunchWindow device settings", () => { expect(within(panel).getByText("Version 1.9.6")).toBeInTheDocument(); }); + // Changing the capture resolution re-runs the webcam acquisition effect, whose cleanup stops + // every track of the live stream — the same stream the browser, macOS and Linux paths hand to + // the webcam MediaRecorder. Mid-take that ends the camera partway through, without a word. + it("ignores a camera quality change made in a panel left open by a recording", async () => { + cameraDevicesState.devices = [{ deviceId: "cam-1", label: "Logitech BRIO" }]; + const { rerender } = renderLaunchWindow(); + + fireEvent.click(await screen.findByTestId("launch-device-settings-button")); + const panel = await screen.findByTestId("hud-device-settings"); + + recorderState.value.recording = true; + rerender( + + + , + ); + + fireEvent.click(within(panel).getByTestId("camera-quality-1080p")); + + expect(recorderState.value.setWebcamQuality).not.toHaveBeenCalled(); + expect(window.electronAPI.setRecordingPrefs).not.toHaveBeenCalled(); + }); + it("is unavailable while recording, when devices can't be changed anyway", async () => { recorderState.value.recording = true; diff --git a/src/components/launch/LaunchWindow.tsx b/src/components/launch/LaunchWindow.tsx index 5ddffd89e..e72ea2b25 100644 --- a/src/components/launch/LaunchWindow.tsx +++ b/src/components/launch/LaunchWindow.tsx @@ -933,10 +933,18 @@ export function LaunchWindow() { const handleSelectCameraQuality = useCallback( (quality: WebcamQualityId) => { + // The same guard the other controls carry. The gear is disabled mid-take, + // but a settings panel already open stays mounted, so this stays + // clickable. `webcamQuality` is a dependency of the webcam acquisition + // effect: changing it re-runs that effect's cleanup, which stops every + // track of the live stream -- the very stream the browser, macOS and + // Linux paths are recording. The camera would end partway through the + // take, silently. + if (controlsLocked) return; setWebcamQuality(quality); persistRecordingPrefs({ camQuality: quality }); }, - [persistRecordingPrefs, setWebcamQuality], + [controlsLocked, persistRecordingPrefs, setWebcamQuality], ); const toggleDeviceSettings = useCallback(() => { From 847cc4f256e22ca001531530bb9e29e8a7f29db2 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:45:04 +0200 Subject: [PATCH 3/8] fix(recording): do not read studio-range black as a visible webcam frame The warm-up probe exists to tell "the camera is running" from "the camera is open but has not produced anything yet". Its NV12 variant judged the Y plane raw, and a camera's NV12 is studio-range: an all-black frame is a plane of 16, whose average clears the average threshold on its own. Every black frame therefore counted as a picture, and the no-visible-frame warning could not fire for the case it was written for. The comment claiming `maxLuma > 24` covered this was wrong -- it overlooked the `||`. Y is now normalised from studio range onto 0..255, so the thresholds mean the same thing here as they do for BGRA. Both probes move out of main.cpp into frame_visibility.{h,cpp} with a unit test, alongside audio_sample_utils and webcam_format. Reverting the normalisation makes that test fail on exactly the reported case, and the packaging build runs it. Reported by CodeRabbit on #875. --- electron/native/wgc-capture/CMakeLists.txt | 16 ++++ .../wgc-capture/src/frame_visibility.cpp | 78 +++++++++++++++++++ .../native/wgc-capture/src/frame_visibility.h | 27 +++++++ .../wgc-capture/src/frame_visibility_test.cpp | 66 ++++++++++++++++ electron/native/wgc-capture/src/main.cpp | 61 +-------------- scripts/build-windows-wgc-helper.mjs | 10 +++ 6 files changed, 198 insertions(+), 60 deletions(-) create mode 100644 electron/native/wgc-capture/src/frame_visibility.cpp create mode 100644 electron/native/wgc-capture/src/frame_visibility.h create mode 100644 electron/native/wgc-capture/src/frame_visibility_test.cpp diff --git a/electron/native/wgc-capture/CMakeLists.txt b/electron/native/wgc-capture/CMakeLists.txt index efa82295d..7a80f3214 100644 --- a/electron/native/wgc-capture/CMakeLists.txt +++ b/electron/native/wgc-capture/CMakeLists.txt @@ -41,6 +41,8 @@ add_executable(wgc-capture src/desktop_icon_cover.cpp src/desktop_icon_cover.h src/dpi_awareness.h + src/frame_visibility.cpp + src/frame_visibility.h src/dshow_webcam_capture.cpp src/dshow_webcam_capture.h src/main.cpp @@ -133,3 +135,17 @@ target_compile_definitions(webcam_format_test PRIVATE ) target_compile_options(webcam_format_test PRIVATE /EHsc /W4 /utf-8) + +add_executable(frame_visibility_test + src/frame_visibility.cpp + src/frame_visibility.h + src/frame_visibility_test.cpp +) + +target_compile_definitions(frame_visibility_test PRIVATE + NOMINMAX + WIN32_LEAN_AND_MEAN + _WIN32_WINNT=0x0A00 +) + +target_compile_options(frame_visibility_test PRIVATE /EHsc /W4 /utf-8) diff --git a/electron/native/wgc-capture/src/frame_visibility.cpp b/electron/native/wgc-capture/src/frame_visibility.cpp new file mode 100644 index 000000000..d296b5f0e --- /dev/null +++ b/electron/native/wgc-capture/src/frame_visibility.cpp @@ -0,0 +1,78 @@ +#include "frame_visibility.h" + +#include + +namespace { + +// Sampling at most this many pixels keeps the probe cheap on a 4K frame while +// still covering it evenly. +constexpr size_t kMaxSamples = 4096; + +// Luma is judged on a full-range 0..255 scale. A frame counts as visible when +// any sample is clearly above black, or when the picture is dim but not empty. +constexpr int kMaxLumaThreshold = 24; +constexpr int kAverageLumaThreshold = 4; + +// BT.709/BT.601 studio range: black sits at 16, white at 235. +constexpr int kStudioBlack = 16; +constexpr int kStudioSpan = 235 - kStudioBlack; + +bool isVisible(uint64_t lumaTotal, int maxLuma, size_t sampled) { + const uint64_t averageLuma = sampled > 0 ? lumaTotal / sampled : 0; + return maxLuma > kMaxLumaThreshold || averageLuma > kAverageLumaThreshold; +} + +} // namespace + +bool hasVisibleBgraContent(const std::vector& frame) { + if (frame.size() < 4) { + return false; + } + + uint64_t lumaTotal = 0; + int maxLuma = 0; + const size_t pixelCount = frame.size() / 4; + const size_t step = std::max(1, pixelCount / kMaxSamples); + size_t sampled = 0; + for (size_t pixel = 0; pixel < pixelCount; pixel += step) { + const size_t offset = pixel * 4; + const int b = frame[offset + 0]; + const int g = frame[offset + 1]; + const int r = frame[offset + 2]; + const int luma = (r * 54 + g * 183 + b * 19) >> 8; + lumaTotal += static_cast(luma); + maxLuma = std::max(maxLuma, luma); + sampled += 1; + } + + return isVisible(lumaTotal, maxLuma, sampled); +} + +bool hasVisibleNv12Content(const std::vector& frame) { + if (frame.size() < 6) { + return false; + } + + const size_t lumaCount = frame.size() * 2 / 3; + const size_t step = std::max(1, lumaCount / kMaxSamples); + uint64_t lumaTotal = 0; + int maxLuma = 0; + size_t sampled = 0; + for (size_t offset = 0; offset < lumaCount; offset += step) { + // Studio range mapped onto 0..255, so the thresholds mean the same here + // as they do for BGRA. Skip this and an all-black frame averages 16 -- + // past the average threshold on its own, so every black frame would + // count as a picture. + const int studio = static_cast(frame[offset]); + const int luma = std::clamp(((studio - kStudioBlack) * 255) / kStudioSpan, 0, 255); + lumaTotal += static_cast(luma); + maxLuma = std::max(maxLuma, luma); + sampled += 1; + } + + return isVisible(lumaTotal, maxLuma, sampled); +} + +bool hasVisibleWebcamContent(const std::vector& frame, bool isNv12) { + return isNv12 ? hasVisibleNv12Content(frame) : hasVisibleBgraContent(frame); +} diff --git a/electron/native/wgc-capture/src/frame_visibility.h b/electron/native/wgc-capture/src/frame_visibility.h new file mode 100644 index 000000000..9f62c2298 --- /dev/null +++ b/electron/native/wgc-capture/src/frame_visibility.h @@ -0,0 +1,27 @@ +#pragma once + +#include +#include + +/** + * Does this frame contain any picture, or is it the blank one a camera emits + * while it is still warming up? + * + * Used before recording starts, to tell "the camera is running" from "the + * camera is open but has not produced anything yet". + */ +bool hasVisibleBgraContent(const std::vector& frame); + +/** + * The NV12 twin: luma is the Y plane, one byte per pixel, so no colour + * conversion is needed. The Y plane is the first two thirds of the buffer. + * + * A camera's NV12 is studio-range, where black is 16 rather than 0. Those + * values are normalised to full range before the thresholds are applied -- + * without that, the average of an all-black frame is 16, clears the average + * threshold on its own, and every black frame counts as visible. + */ +bool hasVisibleNv12Content(const std::vector& frame); + +/** Dispatches to the probe matching the frame's layout. */ +bool hasVisibleWebcamContent(const std::vector& frame, bool isNv12); diff --git a/electron/native/wgc-capture/src/frame_visibility_test.cpp b/electron/native/wgc-capture/src/frame_visibility_test.cpp new file mode 100644 index 000000000..51e9f7531 --- /dev/null +++ b/electron/native/wgc-capture/src/frame_visibility_test.cpp @@ -0,0 +1,66 @@ +#include "frame_visibility.h" + +#include +#include +#include + +namespace { + +int failures = 0; + +void expectVisible(const std::string& label, bool actual, bool expected) { + if (actual == expected) { + return; + } + std::printf("FAIL %s: expected %s, got %s\n", label.c_str(), expected ? "visible" : "blank", + actual ? "visible" : "blank"); + ++failures; +} + +std::vector nv12(int width, int height, std::uint8_t luma) { + std::vector frame(static_cast(width) * height * 3 / 2, 128); + std::fill(frame.begin(), frame.begin() + static_cast(width) * height, luma); + return frame; +} + +std::vector bgra(int width, int height, std::uint8_t value) { + return std::vector(static_cast(width) * height * 4, value); +} + +} // namespace + +int main() { + // The regression this file exists for. A camera's NV12 is studio-range, so an + // all-black frame is a plane of 16, not 0. Averaged raw that is 16 -- past the + // average threshold on its own -- and every black frame counted as a picture, + // which is exactly what the warm-up probe exists to rule out. + expectVisible("studio-range black is blank", hasVisibleNv12Content(nv12(640, 480, 16)), false); + expectVisible("true zero is blank", hasVisibleNv12Content(nv12(640, 480, 0)), false); + + // Just above black stays blank; a clearly lit frame does not. + expectVisible("near-black is blank", hasVisibleNv12Content(nv12(640, 480, 20)), false); + expectVisible("mid grey is visible", hasVisibleNv12Content(nv12(640, 480, 128)), true); + expectVisible("studio white is visible", hasVisibleNv12Content(nv12(640, 480, 235)), true); + + // BGRA keeps its own behaviour: full-range, black is 0. + expectVisible("bgra black is blank", hasVisibleBgraContent(bgra(640, 480, 0)), false); + expectVisible("bgra grey is visible", hasVisibleBgraContent(bgra(640, 480, 128)), true); + + // Buffers too small to carry a frame must not be read past their end. + expectVisible("empty nv12", hasVisibleNv12Content({}), false); + expectVisible("empty bgra", hasVisibleBgraContent({}), false); + expectVisible("runt nv12", hasVisibleNv12Content({16, 16, 16}), false); + + // The dispatcher has to pick the layout it is told, not guess. The same bytes + // read as BGRA are bright; read as studio-range NV12 they are black. + const std::vector blackNv12 = nv12(320, 240, 16); + expectVisible("dispatch nv12", hasVisibleWebcamContent(blackNv12, true), false); + expectVisible("dispatch bgra", hasVisibleWebcamContent(bgra(320, 240, 200), false), true); + + if (failures == 0) { + std::printf("frame_visibility_test: all assertions passed\n"); + return 0; + } + std::printf("frame_visibility_test: %d assertion(s) failed\n", failures); + return 1; +} diff --git a/electron/native/wgc-capture/src/main.cpp b/electron/native/wgc-capture/src/main.cpp index 08fe11aa0..996540de7 100644 --- a/electron/native/wgc-capture/src/main.cpp +++ b/electron/native/wgc-capture/src/main.cpp @@ -6,6 +6,7 @@ #include "wasapi_device_watcher.h" #include "wasapi_loopback_capture.h" #include "wasapi_render_keepalive.h" +#include "frame_visibility.h" #include "webcam_capture.h" #include "wgc_session.h" @@ -393,66 +394,6 @@ void reportCaptureAdapters(ID3D11Device* device, HMONITOR targetMonitor) { } } -/** - * The NV12 twin of `hasVisibleBgraContent`: luma is the Y plane, one byte per - * pixel, so no colour conversion is needed to judge whether a frame has any - * picture in it. The Y plane is the first two thirds of an NV12 buffer. - */ -bool hasVisibleNv12Content(const std::vector& frame) { - if (frame.size() < 6) { - return false; - } - - const size_t lumaCount = frame.size() * 2 / 3; - const size_t step = std::max(1, lumaCount / 4096); - uint64_t lumaTotal = 0; - BYTE maxLuma = 0; - size_t sampled = 0; - for (size_t offset = 0; offset < lumaCount; offset += step) { - const BYTE luma = frame[offset]; - lumaTotal += luma; - maxLuma = std::max(maxLuma, luma); - sampled += 1; - } - - // The same thresholds the BGRA probe uses. NV12 from a camera is - // studio-range, so a black frame sits at 16 rather than 0 -- which the - // maxLuma > 24 test already tolerates. - const uint64_t averageLuma = sampled > 0 ? lumaTotal / sampled : 0; - return maxLuma > 24 || averageLuma > 4; -} - -bool hasVisibleBgraContent(const std::vector& frame); - -/** Dispatches to the probe matching the frame's layout. */ -bool hasVisibleWebcamContent(const std::vector& frame, bool isNv12) { - return isNv12 ? hasVisibleNv12Content(frame) : hasVisibleBgraContent(frame); -} - -bool hasVisibleBgraContent(const std::vector& frame) { - if (frame.size() < 4) { - return false; - } - - uint64_t lumaTotal = 0; - BYTE maxLuma = 0; - const size_t pixelCount = frame.size() / 4; - const size_t step = std::max(1, pixelCount / 4096); - size_t sampledPixels = 0; - for (size_t pixel = 0; pixel < pixelCount; pixel += step) { - const size_t offset = pixel * 4; - const BYTE b = frame[offset + 0]; - const BYTE g = frame[offset + 1]; - const BYTE r = frame[offset + 2]; - const BYTE luma = static_cast((static_cast(r) * 54 + static_cast(g) * 183 + static_cast(b) * 19) >> 8); - lumaTotal += luma; - maxLuma = std::max(maxLuma, luma); - sampledPixels += 1; - } - - const uint64_t averageLuma = sampledPixels > 0 ? lumaTotal / sampledPixels : 0; - return maxLuma > 24 || averageLuma > 4; -} bool findBool(const std::string& json, const std::string& key, bool fallback) { auto pos = json.find("\"" + key + "\""); diff --git a/scripts/build-windows-wgc-helper.mjs b/scripts/build-windows-wgc-helper.mjs index b4e4a59bd..219d03b08 100644 --- a/scripts/build-windows-wgc-helper.mjs +++ b/scripts/build-windows-wgc-helper.mjs @@ -115,3 +115,13 @@ if (!fs.existsSync(webcamFormatTestPath)) { // far more -- and the overlay upscales it. await run(webcamFormatTestPath, [], { cwd: BUILD_DIR }); console.log(`Passed ${webcamFormatTestPath}`); + +const frameVisibilityTestPath = path.join(BUILD_DIR, "frame_visibility_test.exe"); +if (!fs.existsSync(frameVisibilityTestPath)) { + throw new Error(`WGC helper build completed but ${frameVisibilityTestPath} was not found.`); +} +// Guards the warm-up probe that decides whether the camera has produced a +// picture yet. Studio-range black is 16, not 0, so an unnormalised average +// reads every black frame as content. +await run(frameVisibilityTestPath, [], { cwd: BUILD_DIR }); +console.log(`Passed ${frameVisibilityTestPath}`); From 49198e032492336fae830a232c5d112ad720237f Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:59:02 +0200 Subject: [PATCH 4/8] fix(recording): size the browser webcam sidecar from its own camera stream MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two of the three webcam recorders were switched to a camera-derived bitrate; the browser pipeline kept `Math.min(videoBitsPerSecond, BITRATE_BASE)`. That value is the SCREEN recording's rate, so the camera's quality tracked whichever monitor was being captured and was capped at 18 Mbit/s — well under what a 1440p or 2160p frame needs now that the capture resolution is a choice. The rule now lives in one `webcamRecorderOptions` helper that all three call sites share, because repeating it inline is exactly how one of them was missed. --- src/hooks/useScreenRecorder.ts | 31 ++++++++++++++++--------------- 1 file changed, 16 insertions(+), 15 deletions(-) diff --git a/src/hooks/useScreenRecorder.ts b/src/hooks/useScreenRecorder.ts index b3b3ed9c3..90644df39 100644 --- a/src/hooks/useScreenRecorder.ts +++ b/src/hooks/useScreenRecorder.ts @@ -371,6 +371,19 @@ export function useScreenRecorder(): UseScreenRecorderReturn { return accumulatedDurationMs.current + segmentDuration; }, []); + /** + * Recorder options for a webcam sidecar. + * + * The bitrate comes from the camera's own frame, never from the screen + * recording. Repeated inline at each call site, that rule held in two of the + * three and left the browser pipeline encoding a 2160p camera at the + * monitor's rate, capped well below what the frame needs. + */ + const webcamRecorderOptions = (stream: MediaStream | null): MediaRecorderOptions => ({ + mimeType: selectMimeType(), + videoBitsPerSecond: webcamBitrateForStream(stream), + }); + const selectMimeType = () => { // H.264 first: hardware-accelerated, so sharp real-time output. AV1/VP9 are // better for distribution but too CPU-heavy for live 60 fps capture (software @@ -1354,13 +1367,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { // recordingId we send here, so this name is the one finalize rebuilds. nativeWebcamRecorder = createRecorderHandle( webcamStream.current, - { - mimeType: selectMimeType(), - // Sized from the track, not from BITRATE_BASE: that was the - // screen's rate and it starves a 1440p or 2160p camera frame - // badly enough to undo the resolution we just asked for. - videoBitsPerSecond: webcamBitrateForStream(webcamStream.current), - }, + webcamRecorderOptions(webcamStream.current), `${RECORDING_FILE_PREFIX}${activeRecordingId}${WEBCAM_FILE_SUFFIX}${VIDEO_FILE_EXTENSION}`, ); } else { @@ -1556,13 +1563,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { // take is never flattened into one ArrayBuffer at finalize (#253). nativeWebcamRecorder = createRecorderHandle( webcamStream.current, - { - mimeType: selectMimeType(), - // Sized from the track, not from BITRATE_BASE: that was the - // screen's rate and it starves a 1440p or 2160p camera frame - // badly enough to undo the resolution we just asked for. - videoBitsPerSecond: webcamBitrateForStream(webcamStream.current), - }, + webcamRecorderOptions(webcamStream.current), `${RECORDING_FILE_PREFIX}${activeRecordingId}${WEBCAM_FILE_SUFFIX}${VIDEO_FILE_EXTENSION}`, ); } else { @@ -2025,7 +2026,7 @@ export function useScreenRecorder(): UseScreenRecorderReturn { if (webcamStream.current) { webcamRecorder.current = createRecorderHandle( webcamStream.current, - { mimeType, videoBitsPerSecond: Math.min(videoBitsPerSecond, BITRATE_BASE) }, + webcamRecorderOptions(webcamStream.current), `${RECORDING_FILE_PREFIX}${activeRecordingId}${WEBCAM_FILE_SUFFIX}${VIDEO_FILE_EXTENSION}`, ); } From 27a7ba9a4c2d983acbf4d93918a26077e9712b23 Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Mon, 28 Sep 2026 11:59:30 +0200 Subject: [PATCH 5/8] fix(recording): copy padded NV12 rows a row at a time Media Foundation may pad every row out to a stride wider than the frame. `ConvertToContiguousBuffer` joins multiple buffers but leaves that padding in place, and the flat `Lock()` view cannot express it -- so taking the first width*height*3/2 bytes folds the padding into the picture and shears it. The length check did not catch it either: a padded buffer is larger than the tight size it is compared against, so it passed. `IMF2DBuffer` reports the real pitch, so each Y and UV row is now copied on its own. A buffer offering no usable 2D view, or a pitch narrower than the frame, falls back to the flat copy -- which is correct for every tightly packed driver and is what this code did before. RGB32 keeps the flat path unchanged; only NV12, added in this branch, goes through the 2D view. Verified against a Logitech BRIO at 3840x2160: 245 frames delivered over 8s at 37 Mbit/s, and an extracted frame shows no shear or colour shift. This camera reports a tight pitch, so the padded case itself remains unexercised here. Reported by CodeRabbit on #875. --- .../native/wgc-capture/src/webcam_capture.cpp | 81 +++++++++++++++++++ 1 file changed, 81 insertions(+) diff --git a/electron/native/wgc-capture/src/webcam_capture.cpp b/electron/native/wgc-capture/src/webcam_capture.cpp index 0a3666482..e002c1771 100644 --- a/electron/native/wgc-capture/src/webcam_capture.cpp +++ b/electron/native/wgc-capture/src/webcam_capture.cpp @@ -257,6 +257,59 @@ bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstrin namespace { +/** + * Copies an NV12 sample into `destination`, tightly packed. + * + * Media Foundation may pad every row out to a stride wider than the frame. + * `ConvertToContiguousBuffer` joins multiple buffers but leaves that padding + * in place, and the flat `Lock()` view cannot express it -- taking the first + * width*height*3/2 bytes of a padded buffer folds the padding into the picture + * and shears it. `IMF2DBuffer` reports the real pitch, so each row is copied on + * its own. + * + * Returns false when the buffer offers no usable 2D view, leaving the caller on + * its flat copy, which is what every tightly packed driver needs anyway. + */ +bool copyNv12Tightly(IMFMediaBuffer* buffer, int width, int height, std::vector& destination) { + if (width <= 0 || height <= 0) { + return false; + } + + Microsoft::WRL::ComPtr twoD; + if (FAILED(buffer->QueryInterface(IID_PPV_ARGS(&twoD)))) { + return false; + } + + BYTE* scanline0 = nullptr; + LONG pitch = 0; + if (FAILED(twoD->Lock2D(&scanline0, &pitch)) || !scanline0) { + return false; + } + // A negative pitch means bottom-up, which NV12 never is here. Rather than + // guess at a layout we have never seen, hand the frame back to the flat path. + if (pitch < width) { + twoD->Unlock2D(); + return false; + } + + const size_t rowBytes = static_cast(width); + destination.resize(rowBytes * static_cast(height) * 3 / 2); + BYTE* out = destination.data(); + for (int row = 0; row < height; ++row) { + std::memcpy(out, scanline0 + static_cast(pitch) * row, rowBytes); + out += rowBytes; + } + // The interleaved UV plane follows the Y plane at the same pitch, half as tall. + const BYTE* uvPlane = scanline0 + static_cast(pitch) * height; + for (int row = 0; row < height / 2; ++row) { + std::memcpy(out, uvPlane + static_cast(pitch) * row, rowBytes); + out += rowBytes; + } + + twoD->Unlock2D(); + return true; +} + /** * Is this subtype something the source reader must DECODE before it can convert? * @@ -554,6 +607,34 @@ void WebcamCapture::captureLoop() { continue; } + // Padded rows are only possible to detect through the 2D view, and only + // NV12 is driven through it -- RGB32 keeps the flat copy it has always had. + if (deliversNv12_) { + const auto storeStartedAt = std::chrono::steady_clock::now(); + bool copied = false; + { + std::scoped_lock lock(frameMutex_); + copied = copyNv12Tightly(buffer.Get(), width_, height_, latestFrame_); + if (copied) { + latestFrameSequence_ += 1; + } + } + if (copied) { + if (framesDelivered_ == 0) { + const auto waitedMs = std::chrono::duration_cast( + std::chrono::steady_clock::now() - loopStartedAt) + .count(); + std::cerr << "INFO: First webcam frame after " << waitedMs << "ms" << std::endl; + } + framesDelivered_ += 1; + storeUs_ += static_cast( + std::chrono::duration_cast( + std::chrono::steady_clock::now() - storeStartedAt) + .count()); + continue; + } + } + BYTE* data = nullptr; DWORD maxLength = 0; DWORD currentLength = 0; From 29907a748f049e07692d6937aef21cba4d7e846f Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:03:40 +0200 Subject: [PATCH 6/8] revert: drop the NV12 2-D copy path, which fixed a bug that cannot happen `IMFMediaBuffer::Lock` already guarantees what the reverted commit went to IMF2DBuffer for. From the IMF2DBuffer remarks: "The Lock method returns a buffer that is guaranteed to be contiguous. If the underlying buffer is not contiguous, the method copies the data into a new buffer, and the Unlock method copies it back into the original buffer." Row padding therefore never reaches the first width*height*3/2 bytes, and the flat copy was correct all along. The reverted commit added a second native-layout path, and its bounds and lifetime surface, for a corruption mode the documented contract excludes -- and said so itself by recording the padded case as unexercised on hardware. Lock2D would still avoid the copy Lock may perform, but that is an optimisation needing a measurement, and there is none: the capture loop's store step measured 2.3ms per 2160p frame against 92ms in ReadSample. The same page also advises against mixing IMF2DBuffer and IMFMediaBuffer calls on one buffer, which the reverted code did on its fallback. The contract is now quoted at the Lock site so the next reader does not have to rediscover it. Verified after the revert: 3840x2160 at 30 fps, 246 frames delivered over 8s. Raised by @My-Denia on #875. --- .../native/wgc-capture/src/webcam_capture.cpp | 91 ++----------------- 1 file changed, 10 insertions(+), 81 deletions(-) diff --git a/electron/native/wgc-capture/src/webcam_capture.cpp b/electron/native/wgc-capture/src/webcam_capture.cpp index e002c1771..27145e227 100644 --- a/electron/native/wgc-capture/src/webcam_capture.cpp +++ b/electron/native/wgc-capture/src/webcam_capture.cpp @@ -257,59 +257,6 @@ bool WebcamCapture::selectDevice(const std::wstring& deviceId, const std::wstrin namespace { -/** - * Copies an NV12 sample into `destination`, tightly packed. - * - * Media Foundation may pad every row out to a stride wider than the frame. - * `ConvertToContiguousBuffer` joins multiple buffers but leaves that padding - * in place, and the flat `Lock()` view cannot express it -- taking the first - * width*height*3/2 bytes of a padded buffer folds the padding into the picture - * and shears it. `IMF2DBuffer` reports the real pitch, so each row is copied on - * its own. - * - * Returns false when the buffer offers no usable 2D view, leaving the caller on - * its flat copy, which is what every tightly packed driver needs anyway. - */ -bool copyNv12Tightly(IMFMediaBuffer* buffer, int width, int height, std::vector& destination) { - if (width <= 0 || height <= 0) { - return false; - } - - Microsoft::WRL::ComPtr twoD; - if (FAILED(buffer->QueryInterface(IID_PPV_ARGS(&twoD)))) { - return false; - } - - BYTE* scanline0 = nullptr; - LONG pitch = 0; - if (FAILED(twoD->Lock2D(&scanline0, &pitch)) || !scanline0) { - return false; - } - // A negative pitch means bottom-up, which NV12 never is here. Rather than - // guess at a layout we have never seen, hand the frame back to the flat path. - if (pitch < width) { - twoD->Unlock2D(); - return false; - } - - const size_t rowBytes = static_cast(width); - destination.resize(rowBytes * static_cast(height) * 3 / 2); - BYTE* out = destination.data(); - for (int row = 0; row < height; ++row) { - std::memcpy(out, scanline0 + static_cast(pitch) * row, rowBytes); - out += rowBytes; - } - // The interleaved UV plane follows the Y plane at the same pitch, half as tall. - const BYTE* uvPlane = scanline0 + static_cast(pitch) * height; - for (int row = 0; row < height / 2; ++row) { - std::memcpy(out, uvPlane + static_cast(pitch) * row, rowBytes); - out += rowBytes; - } - - twoD->Unlock2D(); - return true; -} - /** * Is this subtype something the source reader must DECODE before it can convert? * @@ -601,40 +548,22 @@ void WebcamCapture::captureLoop() { continue; } + // `Lock()` below is enough; row padding cannot reach `latestFrame_`. + // + // A 2-D media buffer may well pad each row out to a stride wider than + // the frame, but that padding is never what `Lock()` hands back: + // "The Lock method returns a buffer that is guaranteed to be + // contiguous. If the underlying buffer is not contiguous, the method + // copies the data into a new buffer" (IMF2DBuffer, Remarks). Reading + // the native pitch through IMF2DBuffer::Lock2D would only be worth it + // as a measured optimisation -- it saves that copy -- and mixing the + // two interfaces on one buffer is advised against in the same page. Microsoft::WRL::ComPtr buffer; if (FAILED(sample->ConvertToContiguousBuffer(&buffer)) || !buffer) { bufferFailures_ += 1; continue; } - // Padded rows are only possible to detect through the 2D view, and only - // NV12 is driven through it -- RGB32 keeps the flat copy it has always had. - if (deliversNv12_) { - const auto storeStartedAt = std::chrono::steady_clock::now(); - bool copied = false; - { - std::scoped_lock lock(frameMutex_); - copied = copyNv12Tightly(buffer.Get(), width_, height_, latestFrame_); - if (copied) { - latestFrameSequence_ += 1; - } - } - if (copied) { - if (framesDelivered_ == 0) { - const auto waitedMs = std::chrono::duration_cast( - std::chrono::steady_clock::now() - loopStartedAt) - .count(); - std::cerr << "INFO: First webcam frame after " << waitedMs << "ms" << std::endl; - } - framesDelivered_ += 1; - storeUs_ += static_cast( - std::chrono::duration_cast( - std::chrono::steady_clock::now() - storeStartedAt) - .count()); - continue; - } - } - BYTE* data = nullptr; DWORD maxLength = 0; DWORD currentLength = 0; From cf6580555090ca25a47c4454c172283f763749cb Mon Sep 17 00:00:00 2001 From: Christian Schmittel <90287914+christian-wr@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:03:44 +0200 Subject: [PATCH 7/8] fix(recording): report the frame rate the DirectShow graph settled on `fps_` held the rate that was asked for, never the one negotiated. `chooseWebcamFormat` deliberately settles for less when a camera offers nothing faster -- its own tests cover a 30 fps target picking a 24 fps mode -- and a driver may choose its nearest supported rate after SetFormat regardless. `fps()` feeds three things that all have to agree with the frames actually arriving: the `webcam-format` event, the webcam encoder's nominal rate, and the constant-rate write interval in main.cpp. A 24 fps camera asked for 30 was therefore encoded and paced as 30, and reported as 30 to the app. `resolveConnectedFormat` now reads AvgTimePerFrame back off the connected type -- the format the graph actually holds, after any adjustment the driver made -- and logs it when it differs from the request. Rewriting AvgTimePerFrame in the capability pass stays: naming a rate inside the range a capability advertises is the documented way to ask for one. What was missing was reading the answer. Raised by @My-Denia on #875. --- .../wgc-capture/src/dshow_webcam_capture.cpp | 21 ++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp index d421f9cf3..fcbb04d58 100644 --- a/electron/native/wgc-capture/src/dshow_webcam_capture.cpp +++ b/electron/native/wgc-capture/src/dshow_webcam_capture.cpp @@ -394,9 +394,28 @@ bool DirectShowWebcamCapture::resolveConnectedFormat( sourceStride_ = ((width_ * bitsPerPixel + 31) / 32) * 4; } sourceTopDown_ = pixelFormat_ != PixelFormat::Bgra || videoInfo->bmiHeader.biHeight < 0; + // The rate the graph settled on, not the one that was asked for. + // + // `chooseWebcamFormat` deliberately settles for less than the target + // when a camera offers nothing faster, and a driver may pick its own + // nearest rate after SetFormat regardless. `fps()` feeds three things + // that all have to agree with the frames actually arriving: the + // `webcam-format` event, the webcam encoder's nominal rate, and the + // constant-rate write interval in main.cpp. Left at the requested + // value, a 24 fps camera asked for 30 is encoded and paced as 30. + if (videoInfo->AvgTimePerFrame > 0) { + const int negotiated = static_cast( + (10'000'000LL + videoInfo->AvgTimePerFrame / 2) / videoInfo->AvgTimePerFrame); + if (negotiated > 0 && negotiated != fps_) { + std::cerr << "INFO: DirectShow webcam negotiated " << negotiated << " fps (asked for " + << fps_ << ")" << std::endl; + fps_ = std::clamp(negotiated, 1, 60); + } + } } std::cerr << "INFO: DirectShow webcam connected subtype " << guidToString(connectedType.subtype) - << " " << width_ << "x" << height_ << " stride=" << sourceStride_ << std::endl; + << " " << width_ << "x" << height_ << "@" << fps_ << " stride=" << sourceStride_ + << std::endl; freeMediaType(connectedType); if (width_ <= 0 || height_ <= 0) { width_ = requestedWidth > 0 ? requestedWidth : 1280; From 27af4a98aa4d67962efa8ccbe8362954149c4800 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?=E8=B6=85=E8=B6=85=E8=B6=85=E8=B6=85=E8=B6=85=E7=BA=A7?= =?UTF-8?q?=E5=96=9C=E6=AC=A2=E4=BD=A0=E7=9A=84=E8=BE=BE=E5=A6=AE=E5=A8=85?= <176143450+My-Denia@users.noreply.github.com> Date: Tue, 29 Sep 2026 15:17:32 +0800 Subject: [PATCH 8/8] test(recording): update camera quality fixtures The webcam quality PR added camQuality as a required RecordingPreferences member, so the complete typed preference fixtures in RecStage and the prefsRace tests need the product default (DEFAULT_WEBCAM_QUALITY, 2160p). The new HudDeviceSettings quality test needed groupId on its CameraDevice fixture and showMicrophone on the panel, whose overrides spread over a partial can only typecheck when the base covers every required prop. Fixes the failing Typecheck (tests) CI job. --- src/components/ai-edition/v4/RecStage.test.tsx | 2 ++ src/components/launch/HudDeviceSettings.quality.test.tsx | 3 ++- src/hooks/useScreenRecorder.prefsRace.test.tsx | 1 + 3 files changed, 5 insertions(+), 1 deletion(-) diff --git a/src/components/ai-edition/v4/RecStage.test.tsx b/src/components/ai-edition/v4/RecStage.test.tsx index f931783aa..0f7a5696c 100644 --- a/src/components/ai-edition/v4/RecStage.test.tsx +++ b/src/components/ai-edition/v4/RecStage.test.tsx @@ -224,6 +224,7 @@ describe("RecStage controls", () => { camEnabled: false, camDeviceId: null, camDeviceName: null, + camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", hideDesktopIcons: false, @@ -424,6 +425,7 @@ describe("RecStage controls", () => { camEnabled: false, camDeviceId: null, camDeviceName: null, + camQuality: "2160p", systemAudioEnabled: false, cursorCaptureMode: "editable-overlay", hideDesktopIcons: false, diff --git a/src/components/launch/HudDeviceSettings.quality.test.tsx b/src/components/launch/HudDeviceSettings.quality.test.tsx index a44cf19b4..67db4a267 100644 --- a/src/components/launch/HudDeviceSettings.quality.test.tsx +++ b/src/components/launch/HudDeviceSettings.quality.test.tsx @@ -39,8 +39,9 @@ function renderPanel(overrides: Partial[0]> const onSelectCameraQuality = vi.fn(); render(