realTime streaming of tts api of openAi
10:19 09 May 2025
def stream_tts(text: str, ws, main_loop, voice="nova"):
    printLogs(f"[TTS] {text}")

    if not ws.connOpen:
        printLogs("[ERROR] WebSocket is not open. Skipping TTS")
        return

    try:
        wait_start = time.perf_counter()
        
        with client.audio.speech.with_streaming_response.create(
            model=ttsModelId,
            voice=voice,
            input=text,
            instructions="act as an advicer",
            response_format="mp3"
        ) as response:
            wait_end = time.perf_counter()
            wait_time = wait_end - wait_start
            printLogs(f"[LATENCY][Consumer thread] waited {wait_time:.2f}s")
            printLogs("[TTS] Streaming audio...")
            for chunk in response.iter_bytes(chunk_size=8192):
                if not chunk:
                    printLogs("[ERROR] chunk is empty. skipping.")
                    continue  

                if not ws.connOpen:
                    printLogs("[ERROR] WebSocket closed during streaming. Stopping.")
                    break

                try:
                    start_time = time.perf_counter()
                    fut = asyncio.run_coroutine_threadsafe(ws.send(chunk), main_loop)
                    fut.result()
                    end_time = time.perf_counter()
                    latency = end_time - start_time
                    printLogs(f"[LATENCY][TTS STREAM] Sent chunk in {latency:.4f}s")
                except Exception as send_error:
                    printLogs(f"[ERROR] WebSocket send failed: {str(send_error)}")
                    break  # Stop if WebSocket sending fails

    except Exception as e:
        printLogs(f"[ERROR] TTS streaming failed: {str(e)}")
        printLogs(traceback.format_exc())

    finally:
        printLogs("[TTS] Streaming complete.")

In this i am trying to send audio in chunks, but the problem is when chunks are sent then that plays on frontend is broken because of incomplete chunks, increasing chunk size can decrease the realtime latency

python text-to-speech openai-api