From 53db6cd9dcc6229b9f1992149a12dd6c757bc646 Mon Sep 17 00:00:00 2001 From: RobVanProd Date: Sun, 26 Jul 2026 02:37:48 -0400 Subject: [PATCH] End capture when the speaker stops, and think until speech actually starts Two things reported from live use: replies were cut off at 4.8 s, and the face did not look like it was thinking while a reply was being prepared. Capture length: - The voice-activity endpoint's maximumCaptureMs was 4800, so any sentence past roughly four seconds was truncated mid-word regardless of whether the speaker had finished. It is now 12000, which is a backstop rather than the normal path: capture still ends on 550 ms of trailing silence as soon as the speaker stops. 12 s of 16 kHz mono PCM is 384 KB, inside the 512 KB uplink limit. - The dedicated capture chunk ceiling moves from 96 (9.6 s) to 130 (13 s) so it sits behind the endpoint rather than in front of it. - The endpoint now runs for the initial wake-gated utterance too, not only for conversation replies. Previously the first capture after a wake phrase always ran its full fixed length, so a two-second question still cost the whole window before anything reached the bridge. Measured, capture length against how long the speaker actually talks: spoke 1.0s before 1.6s after 1.6s spoke 4.0s before 4.6s after 4.6s spoke 6.0s before 4.8s after 6.6s spoke 8.0s before 4.8s after 8.6s spoke 11.0s before 4.8s after 11.6s Before, anything longer than about four seconds ended with reason max_duration and lost the remainder. After, every case ends with trailing_silence and a constant 0.6 s tail. Thinking face: - response_start only means the reply text is ready; TTS still has to render. That gap was spent in Speak, with a speaking mouth and no sound. He now holds the thinking face through it and moves to Speak on the first audio of the response, whether that arrives as an audio stream or a mouth frame. - Adds IntentEngine::setMode for state changes that are a continuation of something already accounted for, so the emotional event is not applied twice. It refreshes the event clock so the mode-decay timers do not unwind it, and it deliberately cannot wake a sleeping robot; waking stays event-driven. Co-Authored-By: Claude Opus 5 --- src/io/VoiceActivityEndpoint.hpp | 7 +- src/main.cpp | 41 ++++++++++-- src/persona/IntentEngine.cpp | 8 +++ src/persona/IntentEngine.hpp | 7 ++ test/test_native_logic/test_main.cpp | 95 ++++++++++++++++++++++++++++ 5 files changed, 151 insertions(+), 7 deletions(-) diff --git a/src/io/VoiceActivityEndpoint.hpp b/src/io/VoiceActivityEndpoint.hpp index 842452e9..2ce8af20 100644 --- a/src/io/VoiceActivityEndpoint.hpp +++ b/src/io/VoiceActivityEndpoint.hpp @@ -21,7 +21,12 @@ struct VoiceActivityEndpointConfig { uint32_t minimumCaptureMs = 600; uint32_t minimumSpeechMs = 150; uint32_t trailingSilenceMs = 550; - uint32_t maximumCaptureMs = 4800; + // Ceiling, not the normal path. Capture ends on trailing silence as soon as + // the speaker stops; this only catches the case where silence is never + // detected. It used to be 4800 ms, which truncated any sentence longer than + // about five seconds mid-word. Bounded by the 512 KB uplink limit: 12 s of + // 16 kHz mono PCM is 384 KB. + uint32_t maximumCaptureMs = 12000; float initialNoiseFloor = 0.015f; float minimumSpeechLevel = 0.040f; float speechNoiseMultiplier = 2.6f; diff --git a/src/main.cpp b/src/main.cpp index 68388a91..b16c9d94 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -984,7 +984,12 @@ volatile bool gWakeMwwStereoDirectionPendingReady = false; #endif #if STACKCHAN_ENABLE_BRIDGE_AUDIO_UPLINK && STACKCHAN_MWW_DEDICATED_WAKE_CAPTURE constexpr uint32_t kWakeMwwCueCompletionTimeoutMs = 120; -constexpr uint8_t kWakeMwwDedicatedCaptureChunks = 96; +// Hard ceiling on one capture, in 100 ms chunks. This is the backstop behind the +// voice-activity endpoint, which normally ends capture as soon as the speaker +// stops. 130 chunks is 13 s, or 416 KB of 16 kHz mono PCM, inside the 512 KB +// uplink limit. Was 96 (9.6 s), and the endpoint's own 4.8 s cap fired first, +// which is what truncated longer sentences. +constexpr uint16_t kWakeMwwDedicatedCaptureChunks = 130; RobotEvent gWakeMwwPendingCaptureEvent {}; bool gWakeMwwPendingCaptureEventReady = false; bool gWakeMwwPendingCaptureIsConversationReply = false; @@ -1974,6 +1979,9 @@ char gBridgeEndpointResponse[kBridgeEndpointControlResponseMax] = {}; SpeechCue gPendingBridgeSpeechCue {}; bool gBridgeSpeechCuePending = false; bool gBridgeResponseHadAudioStream = false; +// True between response_start and the first audio of that response, while the +// thinking face is held. +bool gBridgeAwaitingFirstSpeech = false; uint32_t gBridgeLocalSpeechSuppressedUntilMs = 0; enum class BridgeAudioSafetyStopReason : uint8_t { @@ -2018,6 +2026,7 @@ bool stopBridgeAudioRuntime(uint32_t nowMs, BridgeAudioSafetyStopReason reason) gSpeakerSink.stop(nowMs); gBridgeSpeechCuePending = false; gBridgeResponseHadAudioStream = false; + gBridgeAwaitingFirstSpeech = false; gBridgeLocalSpeechSuppressedUntilMs = 0; if (!held) { return false; @@ -4699,8 +4708,10 @@ bool beginDedicatedWakeCaptureAfterCue(const RobotEvent& wakeEvent) { gWakeMwwDedicatedCapture.chunksAttempted = 0; gWakeMwwDedicatedCapture.chunksSubmitted = 0; VoiceActivityEndpointConfig endpointConfig; - endpointConfig.enabled = endpointConfig.enabled && - gWakeMwwDedicatedCapture.conversationReplyCapture; + // Endpoint the first wake-gated utterance too, not just conversation replies. + // Previously the initial capture ran a fixed length regardless of when the + // speaker stopped, so a two-second question still cost the full window before + // anything was sent. Both paths now end on trailing silence. if (!gWakeMwwDedicatedCapture.endpoint.begin(endpointConfig, captureStartMs)) { return false; } @@ -4768,7 +4779,7 @@ void serviceDedicatedWakeCaptureChunk() { gWakeSrProbe.samplesFed += kMonoSamples; gWakeSrProbe.lastRecordMs = millis(); - if (gWakeMwwDedicatedCapture.conversationReplyCapture) { + if (gWakeMwwDedicatedCapture.endpoint.telemetry().enabled) { endpointReason = gWakeMwwDedicatedCapture.endpoint.process( monoBuf, kMonoSamples, gWakeSrProbe.lastRecordMs); } @@ -4788,7 +4799,7 @@ void serviceDedicatedWakeCaptureChunk() { } const bool chunkLimitReached = gWakeMwwDedicatedCapture.chunksAttempted >= kWakeMwwDedicatedCaptureChunks; - if (chunkLimitReached && gWakeMwwDedicatedCapture.conversationReplyCapture && + if (chunkLimitReached && gWakeMwwDedicatedCapture.endpoint.telemetry().enabled && endpointReason == VoiceActivityEndpointReason::None) { endpointReason = gWakeMwwDedicatedCapture.endpoint.forceMaximum(millis()); } @@ -7092,6 +7103,16 @@ void publishAudioOutSpeechFrame(uint32_t nowMs) { xQueueOverwrite(gSpeechQueue, &input); } +// Move from the thinking face into speaking, on the first real audio of a +// response rather than on the response frame itself. +void beginBridgeSpeechIfPending(uint32_t nowMs) { + if (!gBridgeAwaitingFirstSpeech) { + return; + } + gBridgeAwaitingFirstSpeech = false; + gIntent.setMode(CharacterMode::Speak, nowMs); +} + void publishBridgeSpeechFrame(const BridgeAudioChunk& audio, uint32_t nowMs) { if (gSpeechQueue == nullptr) { return; @@ -7178,7 +7199,12 @@ void handleBridgeOutput(const BridgeClientOutput& output, uint32_t nowMs) { } if (output.type == BridgeClientOutputType::ResponseStart) { - gIntent.applyEvent(output.event, CharacterMode::Speak); + // Keep the thinking face until speech actually starts. response_start only + // means the text is ready; TTS still has to render, and that gap used to be + // spent sitting in Speak with a speaking mouth and no sound. Flipping to + // Speak is deferred to the first audio below. + gIntent.applyEvent(output.event, CharacterMode::Think); + gBridgeAwaitingFirstSpeech = true; gIntent.startResponseGesture(output.response.gesture, output.response.seq, nowMs); gBridgeWakeGate.applyEvent(output.event, nowMs); gAudioOut.cancel(); @@ -7196,10 +7222,12 @@ void handleBridgeOutput(const BridgeClientOutput& output, uint32_t nowMs) { } if (output.type == BridgeClientOutputType::AudioFrame) { + beginBridgeSpeechIfPending(nowMs); publishBridgeSpeechFrame(output.audio, nowMs); } if (output.type == BridgeClientOutputType::AudioStreamStart) { + beginBridgeSpeechIfPending(nowMs); gBridgeResponseHadAudioStream = true; gAudioOut.cancel(); gBridgeLocalSpeechSuppressedUntilMs = nowMs + 120000u; @@ -7228,6 +7256,7 @@ void handleBridgeOutput(const BridgeClientOutput& output, uint32_t nowMs) { } gBridgeSpeechCuePending = false; gBridgeResponseHadAudioStream = false; + gBridgeAwaitingFirstSpeech = false; gBridgeAudioDownlink.abort(nowMs); } } diff --git a/src/persona/IntentEngine.cpp b/src/persona/IntentEngine.cpp index d609770e..17cbc772 100644 --- a/src/persona/IntentEngine.cpp +++ b/src/persona/IntentEngine.cpp @@ -48,6 +48,14 @@ void IntentEngine::begin() { sleepEnteredAtMs_ = 0; } +void IntentEngine::setMode(CharacterMode mode, uint32_t nowMs) { + if (mode_ == CharacterMode::Sleep) { + return; + } + mode_ = mode; + lastEventAtMs_ = nowMs; +} + void IntentEngine::applyEvent(const RobotEvent& event, CharacterMode mode) { // Being asleep is a state you have to be roused out of. Touch, a wake phrase, // being picked up, or a loud noise wakes him; his own bookkeeping does not. diff --git a/src/persona/IntentEngine.hpp b/src/persona/IntentEngine.hpp index f3ee5dd4..a78429f8 100644 --- a/src/persona/IntentEngine.hpp +++ b/src/persona/IntentEngine.hpp @@ -15,6 +15,13 @@ class IntentEngine { public: void begin(); void applyEvent(const RobotEvent& event, CharacterMode mode); + + // Move to a mode without applying an emotional event. Used when a state change + // is a continuation of something already accounted for, such as a reply + // finally starting to produce audio after its response frame arrived. Refreshes + // the event clock so the mode-decay timers do not immediately unwind it. Has no + // effect while asleep; waking requires a rousing event. + void setMode(CharacterMode mode, uint32_t nowMs); void queueSpeechCue(const SpeechCue& cue, uint32_t nowMs); void startResponseGesture(ResponseGesture gesture, uint32_t seed, uint32_t nowMs); void applyCircadian(uint8_t hourOfDay); diff --git a/test/test_native_logic/test_main.cpp b/test/test_native_logic/test_main.cpp index 5150ff40..93e94bb6 100644 --- a/test/test_native_logic/test_main.cpp +++ b/test/test_native_logic/test_main.cpp @@ -1308,6 +1308,36 @@ void test_sleep_pressure_builds_only_when_left_alone() { TEST_ASSERT_GREATER_THAN_FLOAT(busy.profile().fatigue, alone.profile().fatigue); } +void test_set_mode_moves_without_an_event_and_respects_sleep() { + IntentEngine engine; + engine.begin(); + engine.setDemoEnabled(false, 0); + + // A reply that has produced its first audio should move from the thinking face + // into speaking without re-applying an emotional event for it. + RobotEvent started; + started.type = EventType::ResponseStarted; + started.timestampMs = 1000; + started.strength = 1.0f; + engine.applyEvent(started, CharacterMode::Think); + TEST_ASSERT_EQUAL(static_cast(CharacterMode::Think), + static_cast(engine.update(1050).mode)); + + engine.setMode(CharacterMode::Speak, 1200); + TEST_ASSERT_EQUAL(static_cast(CharacterMode::Speak), + static_cast(engine.update(1250).mode)); + + // Waking must stay an event-driven decision, so setMode cannot do it. + IntentEngine sleeper; + sleeper.begin(); + sleeper.setDemoEnabled(false, 0); + uint32_t asleepAtMs = 0; + runUntilAsleep(sleeper, 900000u, &asleepAtMs); + sleeper.setMode(CharacterMode::Speak, asleepAtMs + 1000); + TEST_ASSERT_EQUAL(static_cast(CharacterMode::Sleep), + static_cast(sleeper.update(asleepAtMs + 1050).mode)); +} + void test_lingering_attention_decays_back_to_idle() { IntentEngine engine; engine.begin(); @@ -1643,6 +1673,68 @@ void test_voice_activity_endpoint_rejects_short_noise_and_uses_maximum_fallback( TEST_ASSERT_EQUAL_UINT32(1, endpoint.telemetry().maxDurationFallbacks); } +// Drive one capture with a given amount of speech, then silence, and report how +// many milliseconds of audio were taken. +uint32_t voiceEndpointCaptureMs(uint32_t speakMs, uint32_t ceilingMs, + VoiceActivityEndpointReason* reasonOut) { + VoiceActivityEndpointConfig config; + config.enabled = true; + VoiceActivityEndpoint endpoint; + endpoint.begin(config, 0); + + int16_t speech[1600]; + int16_t silence[1600] = {}; + fillVoiceEndpointSpeech(speech, 1600, 9000); + + VoiceActivityEndpointReason reason = VoiceActivityEndpointReason::None; + uint32_t nowMs = 0; + while (nowMs < ceilingMs && reason == VoiceActivityEndpointReason::None) { + nowMs += 100; + const bool speaking = nowMs <= speakMs; + reason = endpoint.process(speaking ? speech : silence, 1600, nowMs); + } + if (reason == VoiceActivityEndpointReason::None) { + reason = endpoint.forceMaximum(nowMs); + } + if (reasonOut != nullptr) { + *reasonOut = reason; + } + return nowMs; +} + +void test_voice_activity_endpoint_capture_tracks_speech_length() { + // Capture must follow how long the speaker actually talks. The default used to + // cap at 4800 ms, which truncated any sentence past about five seconds + // mid-word regardless of whether the speaker had finished. + const uint32_t ceilingMs = 13000; + const uint32_t speakLengths[] = {1000, 2000, 4000, 6000, 9000}; + uint32_t previousCaptureMs = 0; + + for (size_t i = 0; i < sizeof(speakLengths) / sizeof(speakLengths[0]); ++i) { + VoiceActivityEndpointReason reason = VoiceActivityEndpointReason::None; + const uint32_t captureMs = voiceEndpointCaptureMs(speakLengths[i], ceilingMs, &reason); + + // Ended because the speaker stopped, not because time ran out. + TEST_ASSERT_EQUAL(static_cast(VoiceActivityEndpointReason::TrailingSilence), + static_cast(reason)); + // Kept everything that was said, and did not linger long after. + TEST_ASSERT_GREATER_OR_EQUAL_UINT32(speakLengths[i], captureMs); + TEST_ASSERT_LESS_THAN_UINT32(speakLengths[i] + 1500u, captureMs); + // Longer utterances yield longer captures. + TEST_ASSERT_GREATER_THAN_UINT32(previousCaptureMs, captureMs); + previousCaptureMs = captureMs; + } +} + +void test_voice_activity_endpoint_default_ceiling_fits_a_long_sentence() { + VoiceActivityEndpointConfig config; + // Generous enough that the ceiling is a backstop rather than the normal path. + TEST_ASSERT_GREATER_THAN_UINT32(8000u, config.maximumCaptureMs); + // Still inside the 512 KB uplink limit for 16 kHz mono PCM. + const uint32_t bytes = (config.maximumCaptureMs / 1000u) * 16000u * 2u; + TEST_ASSERT_LESS_THAN_UINT32(512u * 1024u, bytes); +} + void test_voice_activity_endpoint_disabled_path_preserves_fixed_capture() { VoiceActivityEndpointConfig config; config.enabled = false; @@ -7943,6 +8035,8 @@ int main() { RUN_TEST(test_audio_reflex_loud_noise_preempts_speech_events); RUN_TEST(test_voice_activity_endpoint_ends_after_sustained_speech_and_trailing_silence); RUN_TEST(test_voice_activity_endpoint_rejects_short_noise_and_uses_maximum_fallback); + RUN_TEST(test_voice_activity_endpoint_capture_tracks_speech_length); + RUN_TEST(test_voice_activity_endpoint_default_ceiling_fits_a_long_sentence); RUN_TEST(test_voice_activity_endpoint_disabled_path_preserves_fixed_capture); RUN_TEST(test_audio_capture_adapter_disabled_default_is_ready_without_source); RUN_TEST(test_audio_capture_adapter_rejects_oversized_window); @@ -7983,6 +8077,7 @@ int main() { RUN_TEST(test_breathing_produces_occasional_deeper_sigh); RUN_TEST(test_breathing_survives_a_stalled_frame); RUN_TEST(test_sleep_pressure_builds_only_when_left_alone); + RUN_TEST(test_set_mode_moves_without_an_event_and_respects_sleep); RUN_TEST(test_lingering_attention_decays_back_to_idle); RUN_TEST(test_stalled_conversation_mode_recovers_to_idle); RUN_TEST(test_engine_falls_asleep_when_left_alone);