diff --git a/CLAUDE.md b/CLAUDE.md index 7e055617..24a53c06 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -51,7 +51,7 @@ aai.settings.api_key = "your-key" - `aai.DictationTranscriber` — Dictation API: audio uploaded as it is spoken over one live request, transcript (and optional LLM pass) out. Methods: `open_live()` (push-style `DictationLiveSession`: `write()`/`close()`/`result()`/`abort()`), `transcribe_live()` (iterable, file object, bytes or path), `warm()`. There is no buffered `transcribe()`: every entry point posts to `/v1/transcribe/live` - `aai.AsyncDictationTranscriber` — Asyncio counterpart of `DictationTranscriber`; `open_live()` returns an `AsyncDictationLiveSession`. Owns an HTTP pool: use `async with` or `await aclose()` - `aai.DictationConfig` — Dictation options: `sample_rate`, `channels`, `language_codes`, `keyterms_prompt`, `llm_instruction`. Rejects unknown fields -- `aai.DictationResponse` — Dictation result: `.text`, `.words` (`DictationWord` with `text` + `confidence`), `.confidence`, `.llm_response`, `.llm_error`, `.audio_duration_ms`, `.session_id`, `.request_time_ms`, `.sync_time_ms`, and `.final_text` (the LLM rewrite, falling back to `.text`) +- `aai.DictationResponse` — Dictation result: `.text`, `.words` (`DictationWord` with `text` + `confidence`), `.confidence`, `.llm_response`, `.llm_error`, `.audio_duration_ms`, `.session_id`, `.request_time_ms`, `.sync_time_ms`, `.auth_time_ms`, and `.final_text` (the LLM rewrite, falling back to `.text`) - `assemblyai.streaming.v3.RealTimeTranscriber` — Real-time streaming with event-based API (threaded) - `assemblyai.streaming.v3.AsyncRealTimeTranscriber` — Asyncio-native counterpart; same options/events - `aai.LLMGateway` — LLM Gateway client. Resources: `models.list()`, `chat.completions.create()` (incl. `stream=True`), `understanding.create()`/`.validate()` diff --git a/assemblyai/__version__.py b/assemblyai/__version__.py index f49459c7..51bbb3f2 100644 --- a/assemblyai/__version__.py +++ b/assemblyai/__version__.py @@ -1 +1 @@ -__version__ = "1.6.1" +__version__ = "1.6.2" diff --git a/assemblyai/dictation/v1/models.py b/assemblyai/dictation/v1/models.py index a0fc3886..8b4d609e 100644 --- a/assemblyai/dictation/v1/models.py +++ b/assemblyai/dictation/v1/models.py @@ -192,6 +192,10 @@ class DictationResponse(BaseModel): sync_time_ms: Optional[float] = None "Time in milliseconds spent transcribing, excluding the LLM pass." + auth_time_ms: Optional[float] = None + """Authentication and key-validation time in milliseconds, as a portion + of request_time_ms. ``None`` when the server predates the field.""" + @property def final_text(self) -> str: """ diff --git a/tests/unit/test_dictation.py b/tests/unit/test_dictation.py index 0258f983..b9c91696 100644 --- a/tests/unit/test_dictation.py +++ b/tests/unit/test_dictation.py @@ -43,6 +43,7 @@ "session_id": "eb92c4ff-4bbb-429f-9b99-7279d7fe738f", "request_time_ms": 243.7, "sync_time_ms": 180.2, + "auth_time_ms": 24.6, } @@ -288,6 +289,24 @@ def test_transcribe_live_parses_response(httpx_mock: HTTPXMock): assert result.audio_duration_ms == 400 assert result.request_time_ms == 243.7 assert result.sync_time_ms == 180.2 + assert result.auth_time_ms == 24.6 + + +def test_transcribe_live_parses_response_without_auth_time(httpx_mock: HTTPXMock): + # Given a server response that predates the auth_time_ms field + response = {k: v for k, v in _OK_RESPONSE.items() if k != "auth_time_ms"} + httpx_mock.add_response( + url=LIVE_URL, + method="POST", + status_code=httpx.codes.OK, + json=response, + ) + + # When streaming audio chunks + result = aai.DictationTranscriber().transcribe_live(_chunks(b"RIFF", b"fake")) + + # Then auth_time_ms is None instead of a parse failure + assert result.auth_time_ms is None def test_transcribe_live_sends_raw_api_key_to_the_dictation_host(httpx_mock: HTTPXMock): diff --git a/tests/unit/test_dictation_async.py b/tests/unit/test_dictation_async.py index ae839a9a..a47c1dfc 100644 --- a/tests/unit/test_dictation_async.py +++ b/tests/unit/test_dictation_async.py @@ -39,6 +39,7 @@ "session_id": "eb92c4ff-4bbb-429f-9b99-7279d7fe738f", "request_time_ms": 243.7, "sync_time_ms": 180.2, + "auth_time_ms": 24.6, } @@ -195,6 +196,25 @@ async def test_transcribe_live_parses_response(httpx_mock: HTTPXMock): assert result.words[1].text == "reports" assert result.request_time_ms == 243.7 assert result.sync_time_ms == 180.2 + assert result.auth_time_ms == 24.6 + + +async def test_transcribe_live_parses_response_without_auth_time(httpx_mock: HTTPXMock): + # Given a server response that predates the auth_time_ms field + response = {k: v for k, v in _OK_RESPONSE.items() if k != "auth_time_ms"} + httpx_mock.add_response( + url=LIVE_URL, + method="POST", + status_code=httpx.codes.OK, + json=response, + ) + + # When streaming from an async producer + async with aai.AsyncDictationTranscriber() as transcriber: + result = await transcriber.transcribe_live(_achunks(b"RIFF", b"fake")) + + # Then auth_time_ms is None instead of a parse failure + assert result.auth_time_ms is None async def test_transcribe_live_posts_to_the_dictation_host(httpx_mock: HTTPXMock):