diff --git a/.agents/skills/deepgram-python-conversational-stt/SKILL.md b/.agents/skills/deepgram-python-conversational-stt/SKILL.md index 4d9bfac6..83fff585 100644 --- a/.agents/skills/deepgram-python-conversational-stt/SKILL.md +++ b/.agents/skills/deepgram-python-conversational-stt/SKILL.md @@ -91,7 +91,7 @@ with client.listen.v2.connect( **No `language` parameter** on v2 — language is implied by model (`flux-general-en`) or hinted via `language_hint` on multi. -For application-controlled turns, use `eot_threshold="1.0"` with a sufficiently large `eot_timeout_ms`, then call `conn.send_force_end_turn()` for the active turn. ForceEndTurn requires deployment enablement; see `examples/16-transcription-force-end-turn.py`. +For application-controlled turns, use `eot_threshold="1.0"` with a sufficiently large `eot_timeout_ms`, then call `conn.send_force_end_turn()` for the active turn. ForceEndTurn is available on Deepgram-hosted deployments, including EU and AU. On self-hosted deployments, it requires the 2026-09-15 release or later with `listen_v2 = true` and `listen_v2_force_end_turn = true` under `[features]`; see `examples/16-transcription-force-end-turn.py`. ## Events (server → client) diff --git a/.fernignore b/.fernignore index 74cc0cfc..4885d798 100644 --- a/.fernignore +++ b/.fernignore @@ -65,7 +65,7 @@ src/deepgram/speak/v2/socket_client.py # Agent TTS provider expressivity must reject fractional values rather than # allowing Pydantic v1 to truncate them before the API sees the request. -# [temporarily frozen — remove when Fern emits a strict integer] +# [temporarily frozen — remove when Pydantic v1 support is dropped] src/deepgram/types/speak_settings_v1provider.py src/deepgram/types/deepgram.py @@ -268,6 +268,7 @@ tests/custom/test_flux_tts_controls.py tests/custom/test_topics_intents_v7_compat.py tests/custom/test_transport.py tests/custom/test_websocket_streaming_coverage.py +tests/custom/test_voice_agent_force_end_turn_example.py tests/typecheck/compat_aliases.py # Wire test with restored compatibility coverage for legacy create-key request alias diff --git a/AGENTS.md b/AGENTS.md index e93ae152..9a9d2ea0 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -43,6 +43,7 @@ Current permanently frozen files: - `tests/custom/test_regen_response_fields.py` — hand-written regression test guarding the response fields added by the 2026-08-18 regen: `ListenV1Response/ResultsMetadata.diarize_info` (`model_uuid`, `arch`), per-word `speaker` / `speaker_confidence`, and `ListenV2TurnInfo.trigger`. Same rationale as `test_latency_report_stt_compat.py`: `UncheckedBaseModel` + `skip_validation` makes response parsing lenient, so a future spec removal would silently turn these into `None` rather than failing. Fixtures are real captured payloads; the `trigger` cases pin the parsing contract for the values the server sends (`model` / `manual` / `timeout`) plus the absent case, which is what a deployment without the server-gated `ForceEndTurn` feature returns. Live coverage of the same field lives in `tests/manual/listen/v2/force_end_turn/` - `tests/custom/test_secure_logging.py` — hand-written regression test that the `websockets` Authorization-header DEBUG logs are redacted (API key never logged in clear text) - `tests/custom/test_speak_v2_interrupt_configure.py` — hand-written coverage for the Speak V2 barge-in / mid-stream reconfigure surface (`send_interrupt`, `send_configure`, `SpeechInterrupted`, `ConfigureSuccess`/`ConfigureFailure`, and the `speed`/`expressivity` connect params) +- `tests/custom/test_voice_agent_force_end_turn_example.py` — hand-written regression coverage that the Voice Agent ForceEndTurn example waits for a rejection queued after `AgentAudioDone` - `tests/custom/test_text_builder.py`, `tests/custom/test_transport.py` — hand-written tests - `tests/custom/test_flux_tts_controls.py`, `tests/custom/test_topics_intents_v7_compat.py` — hand-written regression coverage for Flux TTS controls and direct Topics/Intents response compatibility - `tests/typecheck/compat_aliases.py` — hand-written mypy `assert_type` coverage for backward-compatible alias TypedDicts @@ -64,10 +65,10 @@ Current temporarily frozen files: - `.gitignore` — Fern generates a baseline version and has regenerated it before (`fdcce88`, `ad93815`, `4bee463`). We hand-add the local coverage artifact ignores (`.coverage`, `htmlcov/`, `coverage.xml`). Before each regen, unfreeze and re-diff so Fern's own additions are picked up, then re-apply the three coverage lines. - `src/deepgram/speak/v1/socket_client.py` — optional message param defaults, broad exception catch - `src/deepgram/speak/v2/socket_client.py` — same (optional `send_flush`/`send_close`/`send_interrupt` defaults, broad exception catch); new websocket TTS client added in the 2026-07-08 regen. `send_interrupt` carries no required payload so it takes the same optional-default treatment as the other control sends; `send_configure` deliberately keeps its required argument (a Configure with no settings is meaningless) -- `src/deepgram/types/speak_settings_v1provider.py`, `src/deepgram/types/deepgram.py` — validate Agent TTS `expressivity` as `pydantic.StrictInt` so Pydantic v1 rejects fractional values instead of truncating them before they reach the API. Regression coverage in `tests/custom/test_socket_client_shims.py`. Unfreeze when Fern emits a strict integer. - `src/deepgram/listen/v1/socket_client.py` — same - `src/deepgram/listen/v2/socket_client.py` — same (broad except, optional `send_close_stream` default). As of the 2026-08-11 regen the generator properly types `send_configure(ListenV2Configure)` and puts `ListenV2ConfigureSuccess` in the response Union, so those are taken from the generator; the only `send_configure` patch retained is runtime tolerance for a raw dict (sent verbatim) for back-compat with pre-typed-model callers - `src/deepgram/agent/v1/socket_client.py` — same + `_sanitize_numeric_types` +- `src/deepgram/types/speak_settings_v1provider.py`, `src/deepgram/types/deepgram.py` — validate Agent TTS `expressivity` as `pydantic.StrictInt` so Pydantic v1 rejects fractional values instead of truncating them before they reach the API. Regression coverage in `tests/custom/test_socket_client_shims.py`. Unfreeze when Pydantic v1 support is dropped. - `src/deepgram/agent/v1/types/agent_v1settings_agent_context.py`, `src/deepgram/agent/v1/types/agent_v1settings_agent.py`, `src/deepgram/agent/v1/types/agent_v1settings.py`, `src/deepgram/agent/v1/requests/agent_v1settings_agent_context.py`, `src/deepgram/agent/v1/requests/agent_v1settings_agent.py`, `src/deepgram/agent/v1/requests/agent_v1settings.py` — backward-compat patches for the 2026-05-05 Agent Settings schema restructure. These preserve callable `AgentV1SettingsAgent(...)`, keep `AgentV1Settings.agent` accepting both that wrapper and `agent_id` strings, restore the legacy request TypedDict shapes, remap legacy `messages=[...]` / nested `context=AgentV1SettingsAgentContext(messages=[...])` usage into the new `context={"messages": [...]}` wire shape, and keep read-side `obj.messages` access working. - `src/deepgram/core/api_error.py`, `src/deepgram/core/parse_error.py` — credential redaction. Every websocket `connect()` path raises `ApiError(headers=dict(headers), ...)` with the full request headers, and both error types stringify that dict, so an unredacted `Authorization` reached `str(e)`, tracebacks, log aggregators and error trackers (which serialise attributes as well as the message). Both now mask credential values at construction via `_secure_logging.redact_sensitive_headers`, preserving non-sensitive headers (`dg-request-id`) for debugging. This is the same threat `_secure_logging.py` covers for the `websockets` DEBUG handshake logs, via the other path to it. Regression coverage in `tests/custom/test_api_error_redaction.py`. Unfreeze if the generator starts redacting credentials itself. - `src/deepgram/core/query_encoder.py` — coerces Python bools to lowercase `"true"`/`"false"` before they reach `urllib.parse.urlencode` (which would otherwise produce `"True"`/`"False"` via `str()` and break websocket query strings). Only the four `*/connect()` paths call `urlencode`; HTTP raw clients hand params to httpx, which lowercases bools itself, so the patch is a no-op for the HTTP path. Once Fern's websocket codegen normalizes bools (or the spec types these as `boolean` end-to-end), this can be unfrozen. diff --git a/examples/16-transcription-force-end-turn.py b/examples/16-transcription-force-end-turn.py index 31df4f6d..f1825133 100644 --- a/examples/16-transcription-force-end-turn.py +++ b/examples/16-transcription-force-end-turn.py @@ -21,9 +21,15 @@ push-to-talk button being released. The connection stays open after a forced end: the turn index advances and transcription continues. -Note: ForceEndTurn requires server-side enablement and is not available on every -deployment. Where it is not enabled the server replies UNPARSABLE_CLIENT_MESSAGE and -closes the connection; this example reports that and exits. +ForceEndTurn is available on Deepgram-hosted deployments, including EU and AU. On +self-hosted deployments, it requires the 2026-09-15 release or later with +the following [features] settings: + + listen_v2 = true + listen_v2_force_end_turn = true + +Unsupported deployments reply UNPARSABLE_CLIENT_MESSAGE and close the connection; this +example reports that and exits. """ import os @@ -148,7 +154,8 @@ def send_audio() -> None: # before it could print anything. if feature_disabled.is_set(): print("\nForceEndTurn is not enabled on this deployment.") - print("Set DEEPGRAM_BASE_URL to a deployment that has the feature.") + print("Self-hosted deployments need the 2026-09-15 release or later with") + print("listen_v2 = true and listen_v2_force_end_turn = true under [features].") except Exception as e: print(f"Error: {type(e).__name__}: {e}") diff --git a/examples/32-voice-agent-force-end-turn.py b/examples/32-voice-agent-force-end-turn.py index a5861462..5f094fba 100644 --- a/examples/32-voice-agent-force-end-turn.py +++ b/examples/32-voice-agent-force-end-turn.py @@ -31,11 +31,36 @@ load_dotenv() AUDIO_PATH = Path(__file__).parent / "fixtures" / "audio.wav" +FORCE_END_TURN_REJECTION_GRACE_SECONDS = 0.5 + + +def wait_for_force_end_turn_outcome( + agent_finished: threading.Event, + force_end_turn_rejected: threading.Event, + *, + timeout_seconds: float, +) -> bool: + """Return whether ForceEndTurn completed without a queued rejection.""" + deadline = time.monotonic() + timeout_seconds + while True: + if force_end_turn_rejected.is_set(): + return False + + remaining = deadline - time.monotonic() + if remaining <= 0: + raise TimeoutError("Timed out waiting for the agent response") + if agent_finished.wait(min(0.1, remaining)): + # AgentAudioDone can precede FORCE_END_TURN_UNSUPPORTED in the socket reader. + return not force_end_turn_rejected.wait( + min(FORCE_END_TURN_REJECTION_GRACE_SECONDS, remaining) + ) def main() -> None: user_started = threading.Event() agent_finished = threading.Event() + force_end_turn_rejected = threading.Event() + force_end_turn_error: str | None = None settings = AgentV1Settings( audio=AgentV1SettingsAudio(input=AgentV1SettingsAudioInput(encoding="linear16", sample_rate=44100)), @@ -56,6 +81,7 @@ def main() -> None: with DeepgramClient().agent.v1.connect() as agent: def on_message(message: object) -> None: + nonlocal force_end_turn_error message_type = getattr(message, "type", None) if message_type == "UserStartedSpeaking": user_started.set() @@ -67,6 +93,9 @@ def on_message(message: object) -> None: print("AgentAudioDone received") elif message_type in {"Warning", "Error"}: print(f"{message_type}: {message.code} - {message.description}") + if message.code == "FORCE_END_TURN_UNSUPPORTED": + force_end_turn_error = f"{message.code}: {message.description}" + force_end_turn_rejected.set() agent.on(EventType.MESSAGE, on_message) agent.on(EventType.ERROR, lambda error: print(f"Connection error: {error}")) @@ -88,8 +117,12 @@ def on_message(message: object) -> None: print("Sending ForceEndTurn") agent.send_force_end_turn() - if not agent_finished.wait(15): - raise TimeoutError("Timed out waiting for the agent response") + if not wait_for_force_end_turn_outcome( + agent_finished, + force_end_turn_rejected, + timeout_seconds=15, + ): + raise RuntimeError(f"ForceEndTurn failed: {force_end_turn_error}") if __name__ == "__main__": diff --git a/tests/custom/test_voice_agent_force_end_turn_example.py b/tests/custom/test_voice_agent_force_end_turn_example.py new file mode 100644 index 00000000..a8c9b3db --- /dev/null +++ b/tests/custom/test_voice_agent_force_end_turn_example.py @@ -0,0 +1,45 @@ +import importlib.util +import sys +import threading +import types +from pathlib import Path + +EXAMPLE_PATH = Path(__file__).parents[2] / "examples" / "32-voice-agent-force-end-turn.py" + + +class RejectionDuringGraceWait(threading.Event): + def __init__(self): + super().__init__() + self.wait_calls = 0 + + def wait(self, _timeout): + self.wait_calls += 1 + self.set() + return True + + +def load_example(monkeypatch): + dotenv = types.ModuleType("dotenv") + dotenv.load_dotenv = lambda: None + monkeypatch.setitem(sys.modules, "dotenv", dotenv) + + spec = importlib.util.spec_from_file_location("voice_agent_force_end_turn_example", EXAMPLE_PATH) + assert spec is not None + assert spec.loader is not None + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_agent_audio_done_waits_for_late_force_end_turn_rejection(monkeypatch): + example = load_example(monkeypatch) + agent_finished = threading.Event() + force_end_turn_rejected = RejectionDuringGraceWait() + agent_finished.set() + + assert not example.wait_for_force_end_turn_outcome( + agent_finished, + force_end_turn_rejected, + timeout_seconds=1, + ) + assert force_end_turn_rejected.wait_calls == 1 diff --git a/tests/manual/listen/v2/force_end_turn/async.py b/tests/manual/listen/v2/force_end_turn/async.py index d38af3d0..99335285 100644 --- a/tests/manual/listen/v2/force_end_turn/async.py +++ b/tests/manual/listen/v2/force_end_turn/async.py @@ -5,8 +5,10 @@ step 1 of `main.py`: force an in-progress turn to end and confirm the resulting EndOfTurn reports trigger="manual". See `main.py` for the full end-of-turn control matrix. -ForceEndTurn is gated per deployment; where it is not enabled this reports SKIP rather -than failing. Point it at a deployment that has the feature with DEEPGRAM_BASE_URL. +ForceEndTurn is available on Deepgram-hosted deployments, including EU and AU. On +self-hosted deployments, it requires the 2026-09-15 release or later with +`listen_v2 = true` and `listen_v2_force_end_turn = true` under [features]. Unsupported +deployments report SKIP. Requires DEEPGRAM_API_KEY. Run with: @@ -114,7 +116,8 @@ def on_message(message: Any) -> None: if gated: print(" SKIP: ForceEndTurn is not enabled on this deployment") print(f" ({errors[0]})") - print(" Set DEEPGRAM_BASE_URL to a deployment that has the feature.") + print(" Self-hosted deployments need the 2026-09-15 release or later with") + print(" listen_v2 = true and listen_v2_force_end_turn = true under [features].") return if errors: print(f" FAIL: {errors[0]}") diff --git a/tests/manual/listen/v2/force_end_turn/main.py b/tests/manual/listen/v2/force_end_turn/main.py index d8d546b5..66f10fcb 100644 --- a/tests/manual/listen/v2/force_end_turn/main.py +++ b/tests/manual/listen/v2/force_end_turn/main.py @@ -21,10 +21,15 @@ ForceEndTurn becomes the only way to close a turn. This is the combination to reach for when the application owns turn boundaries. -ForceEndTurn is gated per deployment. Where it is not enabled the server replies -UNPARSABLE_CLIENT_MESSAGE ("not enabled on this deployment") and closes the connection; -this script reports SKIP rather than failing. Point it at a deployment that has the -feature with DEEPGRAM_BASE_URL. +ForceEndTurn is available on Deepgram-hosted deployments, including EU and AU. On +self-hosted deployments, it requires the 2026-09-15 release or later with +the following [features] settings: + + listen_v2 = true + listen_v2_force_end_turn = true + +Unsupported deployments reply UNPARSABLE_CLIENT_MESSAGE ("not enabled on this deployment") +and close the connection; this script reports SKIP rather than failing. Requires DEEPGRAM_API_KEY. Run with: @@ -173,7 +178,8 @@ def main() -> None: if gated: print(" SKIP: ForceEndTurn is not enabled on this deployment") print(f" ({errors[0]})") - print(" Set DEEPGRAM_BASE_URL to a deployment that has the feature.") + print(" Self-hosted deployments need the 2026-09-15 release or later with") + print(" listen_v2 = true and listen_v2_force_end_turn = true under [features].") return if errors: raise AssertionError(errors[0])