diff --git a/bridge/conversation_session.py b/bridge/conversation_session.py index 0c4be3d..fde2e14 100644 --- a/bridge/conversation_session.py +++ b/bridge/conversation_session.py @@ -22,6 +22,15 @@ class ConversationConfig: reply_window_ms: int = 10_000 reply_window_min_ms: int = 1_000 reply_window_step_ms: int = 0 + # How long a capture already in progress may outlive the reply window. + # + # The reply window bounds how long we wait for someone to *start* speaking. + # Once they have started, the device owns the ending: firmware's dedicated + # capture ceiling is 12 s and the terminal plus final chunks still have to + # reach us afterwards. Sizing this from reply_window_ms tied an unrelated + # number to that ceiling and left a long but perfectly valid utterance able + # to be closed out from under itself. + capture_commit_ms: int = 13_500 acoustic_tail_ms: int = 250 cooldown_ms: int = 300 max_turns: int = 24 @@ -46,6 +55,12 @@ def __post_init__(self) -> None: raise ValueError("reply_window_min_ms must be between 1000 and reply_window_ms") if self.reply_window_step_ms < 0: raise ValueError("reply_window_step_ms cannot be negative") + # Must outlast the firmware's 12 s dedicated capture ceiling so a capture + # that runs to that ceiling can still deliver its terminal, and must stay + # within the host's 14.5 s absolute capture lease so this can never be the + # control that keeps an abandoned capture alive. + if not 12_000 < self.capture_commit_ms <= 14_500: + raise ValueError("capture_commit_ms must be above 12000 and at most 14500") if not 0 <= self.acoustic_tail_ms <= 2_000: raise ValueError("acoustic_tail_ms must be between 0 and 2000") if self.cooldown_ms < 0: @@ -237,7 +252,7 @@ def utterance_started(self, now_ms: int) -> ConversationTransition: self.capture_in_progress = True # The device starts its bounded capture inside the reply window, but the # final audio chunks can arrive after that listening lease expires. - self.capture_commit_until_ms = now + self.config.reply_window_ms + self.capture_commit_until_ms = now + self.config.capture_commit_ms return self._transition("utterance_accepted", reason="listening") def utterance_committed(self, now_ms: int, text: str) -> ConversationTransition: diff --git a/bridge/test_conversation_session.py b/bridge/test_conversation_session.py index ea707dd..078d97c 100644 --- a/bridge/test_conversation_session.py +++ b/bridge/test_conversation_session.py @@ -309,12 +309,48 @@ def test_started_capture_gets_bounded_time_to_finish_after_short_window(self) -> self.assertEqual("capture_in_progress", session.tick(1_060).reason) snapshot = session.snapshot(1_100) self.assertEqual(0, snapshot["conversation_reply_window_remaining_ms"]) - self.assertEqual(1_800, snapshot["conversation_capture_commit_remaining_ms"]) + # The commitment is sized from the firmware capture ceiling, not from the + # reply window, so a two-second window does not cap how long a capture + # that already started is allowed to finish. + self.assertEqual(13_300, snapshot["conversation_capture_commit_remaining_ms"]) committed = session.utterance_committed(2_000, "third") self.assertEqual(("close_capture", "begin_generation"), committed.actions) self.assertEqual(ConversationPhase.THINKING, session.phase) + def test_capture_running_to_the_firmware_ceiling_still_commits(self) -> None: + # Firmware's dedicated capture ceiling is 12 s and its terminal plus final + # chunks arrive after that. With the commitment sized from reply_window_ms + # the session closed the capture out from under a valid long utterance. + session = ConversationSession(ConversationConfig(acoustic_tail_ms=0)) + session.wake(0) + session.utterance_started(0) + + # Reply window is long gone; the capture is still running. + self.assertEqual("capture_in_progress", session.tick(10_500).reason) + self.assertEqual("capture_in_progress", session.tick(12_400).reason) + + committed = session.utterance_committed(12_600, "a genuinely long question") + self.assertEqual(("close_capture", "begin_generation"), committed.actions) + self.assertEqual(ConversationPhase.THINKING, session.phase) + + def test_capture_commitment_still_expires_so_it_cannot_hold_a_session_open(self) -> None: + session = ConversationSession(ConversationConfig(acoustic_tail_ms=0)) + session.wake(0) + session.utterance_started(0) + + # Past the commitment with nothing delivered, the session must close + # rather than wait on a capture that is never going to finish. + self.assertEqual("reply_timeout", session.tick(13_600).reason) + self.assertEqual(ConversationPhase.COOLDOWN, session.phase) + + def test_capture_commitment_must_outlast_the_firmware_capture_ceiling(self) -> None: + with self.assertRaises(ValueError): + ConversationConfig(capture_commit_ms=12_000) + with self.assertRaises(ValueError): + ConversationConfig(capture_commit_ms=14_501) + self.assertEqual(13_500, ConversationConfig().capture_commit_ms) + def test_invalid_config_is_rejected(self) -> None: with self.assertRaises(ValueError): ConversationConfig(reply_window_ms=0) diff --git a/docs/BRIDGE_AI_HANDOFF.md b/docs/BRIDGE_AI_HANDOFF.md index 469c6ce..7dca748 100644 --- a/docs/BRIDGE_AI_HANDOFF.md +++ b/docs/BRIDGE_AI_HANDOFF.md @@ -63,8 +63,12 @@ when behaviour looks wrong. `CharacterMode` values are `0 Boot, 1 Idle, 2 Attend Completed turns no longer make the listener progressively less patient. The unchanged main firmware rejects out-of-range values rather than silently clamping them. The feature remains explicit and still needs exact-image hardware qualification before promotion. The host's - 10-second capture commitment is currently shorter than the firmware's 12-second endpoint ceiling - and can reject a valid long utterance; this is an open source-level blocker. + capture commitment is no longer derived from the reply window: it is its own `capture_commit_ms`, + defaulting to 13.5 s and validated to sit above the firmware's 12-second endpoint ceiling and at or + below the host's 14.5 s absolute capture lease. The reply window bounds how long to wait for + someone to start speaking; once they have started, the device owns the ending. This closes the + source-level blocker but changes live conversation timing, so it needs a supervised run with a + deliberately long utterance before promotion. - `bridge/initiative_policy.py` implements the ten-minute hard floor, intended fresh-person requirement, circadian suppression, busy/safety gates, curiosity decay, and two-ignored-opener backoff. Initiative generation uses the normal Character Lock and TTS path but never opens a microphone