From 8a700e068554000473ce3e3a653cbda749bda248 Mon Sep 17 00:00:00 2001 From: Joao Date: Fri, 25 Sep 2026 10:55:29 -0500 Subject: [PATCH 1/3] fix(th): a dropped connection to ElevenLabs during synthesis is an outage too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit synthesize_speech's own TTS post was the one ElevenLabs call in the repo still unwrapped: every other caller (transcribe_audio, platform/stt, platform/tts, project_health's elevenlabs_client) already turns a dropped connection or a read timeout into UpstreamServiceError, but this one let httpx.HTTPError rise raw past the status-code check that only fires once a response exists — so a network failure to ElevenLabs answered the client as a generic 500 instead of the 502 every other path here already gives. ENG-1093 (#542) scoped this gap out on purpose: it wrapped project_health's two calls and widened the status split everywhere but named synthesize_speech's TTS post as the one still open, tracked for a follow-up. This closes it, with the same try/except httpx.HTTPError shape the other four already use. Co-Authored-By: Claude Sonnet 5 --- app/services/translation_helper/synthesize_speech.py | 10 +++++++--- tests/test_th_audio_service.py | 12 ++++++++++++ 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/app/services/translation_helper/synthesize_speech.py b/app/services/translation_helper/synthesize_speech.py index f78cc6aec..e8a64c2ce 100644 --- a/app/services/translation_helper/synthesize_speech.py +++ b/app/services/translation_helper/synthesize_speech.py @@ -367,9 +367,13 @@ async def synthesize_speech( } http = client or _make_client() - response = await http.post( - url, json=body, params={"output_format": cfg.elevenlabs_output_format}, headers=headers - ) + try: + response = await http.post( + url, json=body, params={"output_format": cfg.elevenlabs_output_format}, headers=headers + ) + except httpx.HTTPError as error: + logger.warning("ElevenLabs TTS unreachable: %s", error) + raise UpstreamServiceError(f"Speech request could not reach ElevenLabs: {error}") from error if response.status_code >= 400: logger.warning( "ElevenLabs TTS failed: status=%s body=%s", diff --git a/tests/test_th_audio_service.py b/tests/test_th_audio_service.py index 14563700c..56c6f6359 100644 --- a/tests/test_th_audio_service.py +++ b/tests/test_th_audio_service.py @@ -329,6 +329,18 @@ async def test_synthesize_speech_raises_when_api_error(status: int) -> None: await synthesize_speech("hello", client=client, settings=_settings()) +@pytest.mark.parametrize( + "failure", [httpx.ConnectError("boom"), httpx.ReadTimeout("boom")], ids=["connect", "timeout"] +) +async def test_synthesize_speech_treats_a_dropped_connection_as_upstream_too( + failure: Exception, +) -> None: + audio_cache.clear() + client = SimpleNamespace(post=AsyncMock(side_effect=failure)) + with pytest.raises(UpstreamServiceError): + await synthesize_speech("hello", client=client, settings=_settings()) + + async def test_synthesize_speech_requires_api_key() -> None: audio_cache.clear() s = Settings(database_url="sqlite+aiosqlite:///./test.db", elevenlabs_api_key="") From 4f7512a2b34a8028acc40b33ea0bd8e675471041 Mon Sep 17 00:00:00 2001 From: Joao Date: Fri, 25 Sep 2026 10:58:14 -0500 Subject: [PATCH 2/3] refactor(core,ph,th): the ElevenLabs status split lives in one place, not three MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit translation_helper/synthesize_speech.py, translation_helper/transcribe_audio.py and project_health/voice/elevenlabs_client.py each carried their own copy of the same status-to-exception mapping (401/403/429 or >=500 -> UpstreamServiceError, other 4xx -> ValidationError), widened to agree by ENG-1093/#542 but never merged. app/core/exceptions.py already owns both classes and every one of these callers already imports from it, so the shared function goes there instead of a new module, with the docstring transcribe_audio.py carried since it names the rationale (401/403 is not silence any more than a rate limit is). The shared helper takes the message explicitly rather than building one string for every caller, so each site keeps the exact wording it had ("TTS request failed with status..." vs "Transcription request failed..."). No behaviour changes here: every caller's own tests, unmodified, still pass, and each still raises the same exception type with the same message for the same status code as before. platform/stt.py and platform/tts.py still keep their own narrower copy (429 or >=500 only, no 401/403) — collapsing those into the same split is a real behaviour change and lands as its own commit. Co-Authored-By: Claude Sonnet 5 --- app/core/exceptions.py | 11 +++++++++++ .../project_health/voice/elevenlabs_client.py | 17 +++-------------- .../translation_helper/synthesize_speech.py | 18 ++++-------------- .../translation_helper/transcribe_audio.py | 18 ++++-------------- 4 files changed, 22 insertions(+), 42 deletions(-) diff --git a/app/core/exceptions.py b/app/core/exceptions.py index 24c4f259b..5ace0b124 100644 --- a/app/core/exceptions.py +++ b/app/core/exceptions.py @@ -167,6 +167,17 @@ class UpstreamServiceError(Exception): """ +def upstream_or_validation_error(status_code: int, message: str) -> Exception: + """Their outage is not our client's bad request. + + A revoked key or an exhausted quota (401, 403) is not silence any more than a rate + limit is: both mean ElevenLabs refused the request, not that the room said nothing. + """ + if status_code in (401, 403, 429) or status_code >= 500: + return UpstreamServiceError(message) + return ValidationError(message) + + class UnreadableReply(Exception): """A model answered, and the answer could not be read — not a provider that is down. diff --git a/app/services/project_health/voice/elevenlabs_client.py b/app/services/project_health/voice/elevenlabs_client.py index 14161f451..ddaecee12 100644 --- a/app/services/project_health/voice/elevenlabs_client.py +++ b/app/services/project_health/voice/elevenlabs_client.py @@ -5,7 +5,7 @@ import httpx from app.core.config import Settings, get_settings -from app.core.exceptions import UpstreamServiceError, ValidationError +from app.core.exceptions import UpstreamServiceError, ValidationError, upstream_or_validation_error from app.services.project_health.voice.cache import CachedAudio, audio_cache from app.services.project_health.voice.voice_map import ( MULTILINGUAL_VOICE_ID, @@ -31,17 +31,6 @@ def _require_api_key(cfg: Settings) -> str: return cfg.ph_elevenlabs_api_key -def _upstream_or_validation_error(status_code: int, message: str) -> Exception: - """Their outage is not our client's bad request. - - A revoked key or an exhausted quota (401, 403) is not silence any more than a rate - limit is — same split as translation_helper/transcribe_audio.py. - """ - if status_code in (401, 403, 429) or status_code >= 500: - return UpstreamServiceError(message) - return ValidationError(message) - - async def synthesize_speech( text: str, *, @@ -96,7 +85,7 @@ async def synthesize_speech( response.status_code, response.text[:500], ) - raise _upstream_or_validation_error( + raise upstream_or_validation_error( response.status_code, f"TTS request failed with status {response.status_code}" ) @@ -197,7 +186,7 @@ async def transcribe_audio( response.status_code, response.text[:500], ) - raise _upstream_or_validation_error( + raise upstream_or_validation_error( response.status_code, f"Transcription request failed with status {response.status_code}", ) diff --git a/app/services/translation_helper/synthesize_speech.py b/app/services/translation_helper/synthesize_speech.py index e8a64c2ce..d2b332e43 100644 --- a/app/services/translation_helper/synthesize_speech.py +++ b/app/services/translation_helper/synthesize_speech.py @@ -8,7 +8,7 @@ import httpx from app.core.config import Settings, get_settings -from app.core.exceptions import UpstreamServiceError, ValidationError +from app.core.exceptions import UpstreamServiceError, ValidationError, upstream_or_validation_error from app.services.platform.tts import SpeechStore from app.services.translation_helper.audio_cache import CachedAudio, audio_cache from app.services.translation_helper.detect_language import detect_language_code @@ -380,7 +380,9 @@ async def synthesize_speech( response.status_code, response.text[:500], ) - raise _upstream_or_validation_error(response.status_code) + raise upstream_or_validation_error( + response.status_code, f"TTS request failed with status {response.status_code}" + ) payload = response.json() audio_b64 = payload.get("audio_base64") or "" @@ -397,15 +399,3 @@ async def synthesize_speech( if speech_store is not None: await _write_durable(speech_store, cache_key, audio_bytes, timepoints) return entry, False - - -def _upstream_or_validation_error(status_code: int) -> Exception: - """Their outage is not our client's bad request. - - A revoked key or an exhausted quota (401, 403) is not silence any more than a rate - limit is — same split as translation_helper/transcribe_audio.py. - """ - message = f"TTS request failed with status {status_code}" - if status_code in (401, 403, 429) or status_code >= 500: - return UpstreamServiceError(message) - return ValidationError(message) diff --git a/app/services/translation_helper/transcribe_audio.py b/app/services/translation_helper/transcribe_audio.py index 7d6ddf24e..429fc4e3e 100644 --- a/app/services/translation_helper/transcribe_audio.py +++ b/app/services/translation_helper/transcribe_audio.py @@ -6,7 +6,7 @@ import httpx from app.core.config import Settings, get_settings -from app.core.exceptions import UpstreamServiceError, ValidationError +from app.core.exceptions import UpstreamServiceError, ValidationError, upstream_or_validation_error logger = logging.getLogger(__name__) @@ -47,18 +47,6 @@ def _guess_mime_type(filename: str | None, fallback: str | None) -> str: return "audio/mpeg" -def _upstream_or_validation_error(status_code: int) -> Exception: - """Their outage is not our client's bad request. - - A revoked key or an exhausted quota (401, 403) is not silence any more than a rate - limit is: both mean ElevenLabs refused the request, not that the room said nothing. - """ - message = f"Transcription request failed with status {status_code}" - if status_code in (401, 403, 429) or status_code >= 500: - return UpstreamServiceError(message) - return ValidationError(message) - - def _filename_for_upload(filename: str | None, mime_type: str) -> str: if filename: return filename @@ -167,7 +155,9 @@ async def transcribe_audio_detailed( response.status_code, response.text[:500], ) - raise _upstream_or_validation_error(response.status_code) + raise upstream_or_validation_error( + response.status_code, f"Transcription request failed with status {response.status_code}" + ) payload = response.json() text = (payload.get("text") or "").strip() From 58a64f2665a7f7d86583902c439aa50e6bc15bbc Mon Sep 17 00:00:00 2001 From: Joao Date: Fri, 25 Sep 2026 11:01:07 -0500 Subject: [PATCH 3/3] fix(platform): a revoked key or an exhausted quota is an upstream failure too MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit platform/stt.py and platform/tts.py were the two ElevenLabs callers still splitting status codes the old way (429 or >=500 only): 401 and 403 read as ValidationError there, a bad-request 400 for a key ElevenLabs itself revoked or a quota it says is spent. The other three callers already widened this to 401/403/429/>=500 in ENG-1093/#542, and bed7becc's own body named platform/tts as the copy left untouched. Both now call the shared app.core.exceptions.upstream_or_validation_error, closing the last two copies of this split — one definition, five callers. Falsified per file: narrowing the shared helper back to 429-and-above made each file's own new 401/403 cases fail with ValidationError again; restoring it turned both files, and the whole suite, green. Co-Authored-By: Claude Sonnet 5 --- app/services/platform/stt.py | 14 ++++---------- app/services/platform/tts.py | 19 ++++--------------- tests/test_platform/test_stt.py | 2 +- tests/test_platform/test_tts_service.py | 5 +++-- 4 files changed, 12 insertions(+), 28 deletions(-) diff --git a/app/services/platform/stt.py b/app/services/platform/stt.py index 26c3c7aab..e6fd1986f 100644 --- a/app/services/platform/stt.py +++ b/app/services/platform/stt.py @@ -22,7 +22,7 @@ import httpx from app.core.config import Settings, get_settings -from app.core.exceptions import UpstreamServiceError, ValidationError +from app.core.exceptions import UpstreamServiceError, ValidationError, upstream_or_validation_error from app.services.platform.voices import language_hint logger = logging.getLogger(__name__) @@ -83,7 +83,9 @@ async def transcribe_speech( logger.warning( "ElevenLabs STT failed: status=%s body=%s", response.status_code, response.text[:500] ) - raise _upstream_or_validation_error(response.status_code) + raise upstream_or_validation_error( + response.status_code, f"Transcription request failed with status {response.status_code}" + ) text = str(response.json().get("text") or "").strip() logger.info( @@ -96,14 +98,6 @@ async def transcribe_speech( return text -def _upstream_or_validation_error(status_code: int) -> Exception: - """Their outage is not our client's bad request — same split as the TTS service.""" - message = f"Transcription request failed with status {status_code}" - if status_code == 429 or status_code >= 500: - return UpstreamServiceError(message) - return ValidationError(message) - - def _make_client() -> httpx.AsyncClient: global _DEFAULT_CLIENT if _DEFAULT_CLIENT is None: diff --git a/app/services/platform/tts.py b/app/services/platform/tts.py index b1bfb04e0..aa8b663c7 100644 --- a/app/services/platform/tts.py +++ b/app/services/platform/tts.py @@ -28,7 +28,7 @@ import httpx from app.core.config import Settings, get_settings -from app.core.exceptions import UpstreamServiceError, ValidationError +from app.core.exceptions import UpstreamServiceError, ValidationError, upstream_or_validation_error from app.services.platform.voices import language_hint, resolve_voice logger = logging.getLogger(__name__) @@ -358,24 +358,13 @@ async def _synthesize( response.status_code, response.text[:500], ) - raise _upstream_or_validation_error(response.status_code) + raise upstream_or_validation_error( + response.status_code, f"TTS request failed with status {response.status_code}" + ) return bytes(response.content) -def _upstream_or_validation_error(status_code: int) -> Exception: - """Their outage is not our client's bad request. - - 429 and 5xx mean ElevenLabs is rate limiting or down: that is an upstream failure (502), - and dressing it as a 400 means the right alert never fires. Other 4xx really are a - malformed request we sent, so they stay a business error. - """ - message = f"TTS request failed with status {status_code}" - if status_code == 429 or status_code >= 500: - return UpstreamServiceError(message) - return ValidationError(message) - - def etag_of(audio: bytes) -> str: return hashlib.sha256(audio).hexdigest()[:32] diff --git a/tests/test_platform/test_stt.py b/tests/test_platform/test_stt.py index 806761ad7..ad774accf 100644 --- a/tests/test_platform/test_stt.py +++ b/tests/test_platform/test_stt.py @@ -93,7 +93,7 @@ async def test_without_an_api_key_it_is_a_configuration_error() -> None: ) -@pytest.mark.parametrize("status", [429, 500, 503]) +@pytest.mark.parametrize("status", [401, 403, 429, 500, 503]) async def test_elevenlabs_unavailability_is_an_upstream_failure_not_a_client_error( status: int, ) -> None: diff --git a/tests/test_platform/test_tts_service.py b/tests/test_platform/test_tts_service.py index 349454c68..5f1eaba13 100644 --- a/tests/test_platform/test_tts_service.py +++ b/tests/test_platform/test_tts_service.py @@ -251,11 +251,12 @@ async def test_empty_text_is_an_error() -> None: ) -@pytest.mark.parametrize("status", [429, 500, 503]) +@pytest.mark.parametrize("status", [401, 403, 429, 500, 503]) async def test_elevenlabs_unavailability_is_an_upstream_failure_not_a_client_error( status: int, ) -> None: - # Their 429/5xx is not a bad request from the SPA: as a 400, the right alert never fires. + # A revoked key, a spent quota, or 429/5xx is not a bad request from the SPA: as a 400, + # the right alert never fires. store = MemoryStore() with pytest.raises(UpstreamServiceError):