From 263d6dc3d96ff985f74b006e99c3ce6677b7a316 Mon Sep 17 00:00:00 2001 From: Nitish-Kumar-Bote Date: Mon, 21 Sep 2026 15:57:04 +0530 Subject: [PATCH] 60db TTS provider integration --- .env.template | 18 ++++- README.md | 35 ++++++++-- config.py | 23 +++++-- config_ws.py | 23 +++++-- main.py | 95 +++++++++++++++++++++++++-- main_ws.py | 168 +++++++++++++++++++++++++++++++++++++++++++++-- requirements.txt | 4 +- 7 files changed, 336 insertions(+), 30 deletions(-) diff --git a/.env.template b/.env.template index db34ed4..20becf4 100644 --- a/.env.template +++ b/.env.template @@ -17,9 +17,23 @@ AZURE_OPENAI_MODEL= AZUREAI_API_KEY= AZUREAI_REGION= ################################################################################ -### ELEVENLABS API +### TTS PROVIDER +## Which text-to-speech provider to use: elevenlabs | sixtydb +## Keys below are only required for the selected provider. +################################################################################ +TTS_PROVIDER=elevenlabs +################################################################################ +### ELEVENLABS API (used when TTS_PROVIDER=elevenlabs) ## Eleven Labs Default Voice IDs ## https://elevenlabs.io/docs/voicelab/pre-made-voices ################################################################################ ELEVENLABS_API_KEY= -ELEVENLABS_VOICE_ID= \ No newline at end of file +ELEVENLABS_VOICE_ID= +################################################################################ +### 60DB API (used when TTS_PROVIDER=sixtydb) +## Dashboard: https://app.60db.ai | Docs: https://docs.60db.ai +## Voices: GET https://api.60db.ai/voices (Bearer auth) +## 60db default voice: fbb75ed2-975a-40c7-9e06-38e30524a9a1 +################################################################################ +SIXTYDB_API_KEY= +SIXTYDB_VOICE_ID= \ No newline at end of file diff --git a/README.md b/README.md index 0cb88b6..83d0268 100644 --- a/README.md +++ b/README.md @@ -1,14 +1,14 @@ # AI Assistant with fast Speech-to-Speech -This project is a Text or Voice Input AI Assistant that uses the OpenAI API to generate responses and the ElevenLabs Text-to-Speech API to convert the responses into audio quickly. +This project is a Text or Voice Input AI Assistant that uses the OpenAI API to generate responses and the ElevenLabs or 60db Text-to-Speech APIs to convert the responses into audio quickly. ## Features - Real-time speech-to-speech conversation - Customizable voice selection +- Switchable TTS provider: ElevenLabs or 60db (`TTS_PROVIDER` in `.env`) - Easy-to-use command-line interface -- Powered by OpenAI and ElevenLabs -- **Paid ElevenLabs subscription required** +- Powered by OpenAI, ElevenLabs and 60db # Getting Started @@ -18,7 +18,9 @@ This project is a Text or Voice Input AI Assistant that uses the OpenAI API to g - Get your [OpenAI API Key](https://platform.openai.com/api-keys) -- Get your [ElevenLabs API Key](https://elevenlabs.io/app/subscription) +- Get your [ElevenLabs API Key](https://elevenlabs.io/app/subscription) **or** your [60db API Key](https://app.60db.ai) + +- The 60db TTS playback uses [mpv](https://mpv.io/installation/); install it and make sure it is on your PATH - To run ```main_ws.py``` you will need a [Azure](https://portal.azure.com/) OpenAI Deployment and Speech Service Resource @@ -50,7 +52,7 @@ pip install -r requirements.txt cp .env.template .env ``` -- Edit the `.env` file and add your OpenAI API key, ElevenLabs API key, and desired ElevenLabs voice ID. +- Edit the `.env` file and add your OpenAI API key plus the keys of the TTS provider you want to use (ElevenLabs or 60db). ## Usage @@ -74,9 +76,27 @@ cp .env.template .env ## Customization -You can customize the voice used by the AI Assistant by changing the `ELEVENLABS_VOICE_ID` in the `.env` file. +### Choosing the TTS provider + +Set `TTS_PROVIDER` in the `.env` file to switch between providers: + +```bash +TTS_PROVIDER=elevenlabs # ElevenLabs (default) +TTS_PROVIDER=sixtydb # 60db +``` + +Only the API keys of the selected provider are required: + +- `elevenlabs`: `ELEVENLABS_API_KEY`, `ELEVENLABS_VOICE_ID` +- `sixtydb`: `SIXTYDB_API_KEY`, `SIXTYDB_VOICE_ID` + +`main.py` uses the 60db TTS stream API (`POST /tts-stream`), `main_ws.py` uses the 60db TTS WebSocket (`wss://api.60db.ai/ws/tts`). + +### Changing the voice + +You can customize the voice used by the AI Assistant by changing the `ELEVENLABS_VOICE_ID` or `SIXTYDB_VOICE_ID` in the `.env` file. -A link to the pre-made voices is in the `.env.template` file. +Links to the pre-made voices are in the `.env.template` file. ## License @@ -97,3 +117,4 @@ If you encounter any issues or have questions, please open an issue on GitHub. - [OpenAI](https://www.openai.com/) - [ElevenLabs](https://www.elevenlabs.io/) +- [60db](https://60db.ai/) diff --git a/config.py b/config.py index 4a16c32..0a8255c 100644 --- a/config.py +++ b/config.py @@ -32,10 +32,23 @@ if OPENAI_SYSTEM_PROMPT is None: raise ValueError("OPENAI_SYSTEM_PROMPT not set") -ELEVENLABS_API_KEY = os.getenv("ELEVENLABS_API_KEY") -if ELEVENLABS_API_KEY is None: - raise ValueError("ELEVENLABS_API_KEY not set") +TTS_PROVIDER = os.getenv("TTS_PROVIDER", "elevenlabs").strip().lower() +if TTS_PROVIDER not in ("elevenlabs", "sixtydb"): + raise ValueError("TTS_PROVIDER must be 'elevenlabs' or 'sixtydb'") +ELEVENLABS_API_KEY = os.getenv("ELEVENLABS_API_KEY") ELEVENLABS_VOICE_ID = os.getenv("ELEVENLABS_VOICE_ID") -if ELEVENLABS_VOICE_ID is None: - raise ValueError("ELEVENLABS_VOICE_ID not set") + +SIXTYDB_API_KEY = os.getenv("SIXTYDB_API_KEY") +SIXTYDB_VOICE_ID = os.getenv("SIXTYDB_VOICE_ID") + +if TTS_PROVIDER == "elevenlabs": + if ELEVENLABS_API_KEY is None: + raise ValueError("ELEVENLABS_API_KEY not set") + if ELEVENLABS_VOICE_ID is None: + raise ValueError("ELEVENLABS_VOICE_ID not set") +else: + if SIXTYDB_API_KEY is None: + raise ValueError("SIXTYDB_API_KEY not set") + if SIXTYDB_VOICE_ID is None: + raise ValueError("SIXTYDB_VOICE_ID not set") diff --git a/config_ws.py b/config_ws.py index 4c839ae..4e415cd 100644 --- a/config_ws.py +++ b/config_ws.py @@ -40,10 +40,23 @@ if AZUREAI_REGION is None: raise ValueError("AZUREAI_REGION not set") -ELEVENLABS_API_KEY = os.getenv("ELEVENLABS_API_KEY") -if ELEVENLABS_API_KEY is None: - raise ValueError("ELEVENLABS_API_KEY not set") +TTS_PROVIDER = os.getenv("TTS_PROVIDER", "elevenlabs").strip().lower() +if TTS_PROVIDER not in ("elevenlabs", "sixtydb"): + raise ValueError("TTS_PROVIDER must be 'elevenlabs' or 'sixtydb'") +ELEVENLABS_API_KEY = os.getenv("ELEVENLABS_API_KEY") ELEVENLABS_VOICE_ID = os.getenv("ELEVENLABS_VOICE_ID") -if ELEVENLABS_VOICE_ID is None: - raise ValueError("ELEVENLABS_VOICE_ID not set") + +SIXTYDB_API_KEY = os.getenv("SIXTYDB_API_KEY") +SIXTYDB_VOICE_ID = os.getenv("SIXTYDB_VOICE_ID") + +if TTS_PROVIDER == "elevenlabs": + if ELEVENLABS_API_KEY is None: + raise ValueError("ELEVENLABS_API_KEY not set") + if ELEVENLABS_VOICE_ID is None: + raise ValueError("ELEVENLABS_VOICE_ID not set") +else: + if SIXTYDB_API_KEY is None: + raise ValueError("SIXTYDB_API_KEY not set") + if SIXTYDB_VOICE_ID is None: + raise ValueError("SIXTYDB_VOICE_ID not set") diff --git a/main.py b/main.py index 0c65e38..cef04ee 100644 --- a/main.py +++ b/main.py @@ -1,7 +1,14 @@ """ -This script uses the OpenAI API and ElevenLab's REST API +This script uses the OpenAI API and the ElevenLabs REST API +or the 60db TTS stream API, selected via TTS_PROVIDER. """ +import base64 +import json +import shutil +import subprocess import sys + +import requests import speech_recognition as sr # type: ignore from openai import OpenAI from elevenlabs import stream @@ -14,8 +21,11 @@ OPENAI_PROJECT_ID, OPENAI_MODEL, OPENAI_SYSTEM_PROMPT, + TTS_PROVIDER, ELEVENLABS_API_KEY, ELEVENLABS_VOICE_ID, + SIXTYDB_API_KEY, + SIXTYDB_VOICE_ID, ) console = Console() @@ -25,14 +35,15 @@ organization=OPENAI_ORG_ID, project=OPENAI_PROJECT_ID, ) -e_client = ElevenLabs(api_key=ELEVENLABS_API_KEY) +if TTS_PROVIDER == "elevenlabs": + e_client = ElevenLabs(api_key=ELEVENLABS_API_KEY) voice_ident = ELEVENLABS_VOICE_ID def generate_and_play_response(user_input, conversation_history): """ - Generates response using the OpenAI API and - plays it using the ElevenLabs TTS API. + Generates response using the OpenAI API and plays it using + the selected TTS provider (ElevenLabs or 60db). Args: user_input (str): The user's input. @@ -72,6 +83,10 @@ def generate_and_play_response(user_input, conversation_history): def text_stream(): yield response_text + if TTS_PROVIDER == "sixtydb": + tts_stream_sixtydb(response_text) + return + audio_stream = e_client.generate( text=text_stream(), voice=Voice( @@ -91,6 +106,78 @@ def text_stream(): stream(audio_stream) +def tts_stream_sixtydb(text): + """ + Converts text to speech using the 60db TTS stream API + (POST https://api.60db.ai/tts-stream) and plays the streamed + MP3 chunks through the mpv player. + + The response is newline-delimited JSON (NDJSON): each "chunk" + message contains a base64-encoded piece of audio, followed by a + final "complete" message. + + Args: + text (str): The text to convert to speech. + + Returns: + None + + Raises: + ValueError: If mpv is not installed on the system. + RuntimeError: If the 60db API returns an error message. + """ + if shutil.which("mpv") is None: + raise ValueError( + "mpv not found, necessary to stream audio. " + "Install instructions: https://mpv.io/installation/" + ) + + response = requests.post( + "https://api.60db.ai/tts-stream", + headers={ + "Authorization": f"Bearer {SIXTYDB_API_KEY}", + "Content-Type": "application/json", + }, + json={ + "text": text, + "voice_id": SIXTYDB_VOICE_ID, + "speed": 1, + "stability": 50, + "similarity": 75, + }, + stream=True, + timeout=60, + ) + response.raise_for_status() + + mpv_process = subprocess.Popen( + ["mpv", "--no-cache", "--no-terminal", "--", "fd://0"], + stdin=subprocess.PIPE, + stdout=subprocess.DEVNULL, + stderr=subprocess.DEVNULL, + ) + + try: + for line in response.iter_lines(): + if not line: + continue + data = json.loads(line) + if data.get("type") == "chunk": + audio_chunk = base64.b64decode( + data["result"]["audioContent"] + ) + mpv_process.stdin.write(audio_chunk) + mpv_process.stdin.flush() + elif data.get("type") == "error": + raise RuntimeError( + f"60db TTS error: {data.get('message')}" + ) + finally: + if mpv_process.stdin: + mpv_process.stdin.close() + mpv_process.wait() + + def recognize_speech(timeout=20): """ Recognizes speech from the microphone input. diff --git a/main_ws.py b/main_ws.py index e03c1da..a8bb5d3 100644 --- a/main_ws.py +++ b/main_ws.py @@ -1,5 +1,6 @@ """ -This script uses the AzureOpenAI Service and ElevenLabs websockets +This script uses the AzureOpenAI Service and the ElevenLabs +or 60db TTS websockets, selected via TTS_PROVIDER. """ import asyncio import base64 @@ -9,6 +10,7 @@ import subprocess import sys import time +import uuid import azure.cognitiveservices.speech as speechsdk import websockets @@ -19,8 +21,9 @@ from config_ws import (AZUREAI_API_KEY, AZUREAI_REGION, AZURE_API_VERSION, AZURE_OPENAI_ENDPOINT, AZURE_OPENAI_KEY, AZURE_OPENAI_MODEL, - AZURE_SYSTEM_PROMPT, ELEVENLABS_API_KEY, - ELEVENLABS_VOICE_ID) + AZURE_SYSTEM_PROMPT, TTS_PROVIDER, + ELEVENLABS_API_KEY, ELEVENLABS_VOICE_ID, + SIXTYDB_API_KEY, SIXTYDB_VOICE_ID) console = Console() @@ -32,6 +35,10 @@ voice_id = ELEVENLABS_VOICE_ID +# 60db WebSocket returns raw PCM (LINEAR16, 16-bit signed +# little-endian, mono) so mpv needs matching rawaudio arguments. +SIXTYDB_SAMPLE_RATE = 24000 + def is_installed(lib_name): """ @@ -77,12 +84,14 @@ async def text_chunker(chunks): yield buffer + " " -async def stream(audio_stream): +async def stream(audio_stream, extra_args=None): """ Stream audio data using the mpv player. Args: audio_stream: An asynchronous generator that yields audio chunks. + extra_args (list, optional): Additional mpv arguments, e.g. raw + PCM demuxer settings for the 60db WebSocket audio format. Raises: ValueError: If mpv is not installed on the system. @@ -95,7 +104,8 @@ async def stream(audio_stream): ) mpv_process = subprocess.Popen( - ["mpv", "--no-cache", "--no-terminal", "--", "fd://0"], + ["mpv", "--no-cache", "--no-terminal", + *(extra_args or []), "--", "fd://0"], stdin=subprocess.PIPE, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, @@ -113,7 +123,24 @@ async def stream(audio_stream): async def text_to_speech_input_streaming(text_iterator): """ - Sends text chunks to a WebSocket server for TTS conversion + Sends text chunks to the selected TTS provider's WebSocket + and receives audio chunks in response. + + Args: + text_iterator (iterator): Yields text chunks to be converted to speech. + + Returns: + None + """ + if TTS_PROVIDER == "sixtydb": + await text_to_speech_input_streaming_sixtydb(text_iterator) + else: + await text_to_speech_input_streaming_elevenlabs(text_iterator) + + +async def text_to_speech_input_streaming_elevenlabs(text_iterator): + """ + Sends text chunks to the ElevenLabs WebSocket for TTS conversion and receives audio chunks in response. Args: @@ -178,6 +205,135 @@ async def listen(): await listen_task +async def text_to_speech_input_streaming_sixtydb(text_iterator): + """ + Sends text chunks to the 60db TTS WebSocket for TTS conversion + and receives raw PCM audio chunks in response. + + Follows the 60db context protocol: + connection_established -> create_context -> context_created -> + send_text (repeatable) -> flush_context -> audio_chunk ... -> + flush_completed -> close_context -> context_closed. + + Args: + text_iterator (iterator): Yields text chunks to be converted to speech. + + Returns: + None + + Raises: + RuntimeError: If the 60db WebSocket returns an error message. + websockets.exceptions.ConnectionClosed: If the WebSocket connection + is closed unexpectedly. + """ + uri = f"wss://api.60db.ai/ws/tts?apiKey={SIXTYDB_API_KEY}" + context_id = str(uuid.uuid4()) + + async with websockets.connect(uri) as websocket: + async def await_message(key): + """ + Waits until a message containing the given key arrives. + + Args: + key (str): The message key to wait for. + + Returns: + dict: The full message that contains the key. + """ + while True: + data = json.loads(await websocket.recv()) + if key in data: + return data + if "error" in data: + raise RuntimeError( + f"60db TTS error: {data['error'].get('message')}" + ) + + await await_message("connection_established") + + await websocket.send( + json.dumps( + { + "create_context": { + "context_id": context_id, + "voice_id": SIXTYDB_VOICE_ID, + "audio_config": { + "audio_encoding": "LINEAR16", + "sample_rate_hertz": SIXTYDB_SAMPLE_RATE, + }, + "speed": 1, + "stability": 50, + "similarity": 75, + } + } + ) + ) + await await_message("context_created") + + async def listen(): + """ + Listens for audio and control messages from the websocket + connection. + + Yields: + bytes: Decoded audio data received from the websocket. + + Raises: + websockets.exceptions.ConnectionClosed: + If the websocket connection is closed unexpectedly. + """ + while True: + try: + message = await websocket.recv() + data = json.loads(message) + if data.get("audio_chunk"): + yield base64.b64decode( + data["audio_chunk"]["audioContent"] + ) + elif data.get("flush_completed"): + await websocket.send( + json.dumps( + {"close_context": {"context_id": context_id}} + ) + ) + elif data.get("context_closed"): + break + elif data.get("error"): + console.print( + f"60db TTS error: " + f"{data['error'].get('message')}" + ) + break + except websockets.exceptions.ConnectionClosed: + print("Connection closed") + break + + listen_task = asyncio.create_task( + stream( + listen(), + extra_args=[ + "--demuxer=rawaudio", + f"--audio-samplerate={SIXTYDB_SAMPLE_RATE}", + "--audio-channels=1", + "--audio-format=s16p", + ], + ) + ) + + async for text in text_chunker(text_iterator): + await websocket.send( + json.dumps( + {"send_text": {"context_id": context_id, "text": text}} + ) + ) + + await websocket.send( + json.dumps({"flush_context": {"context_id": context_id}}) + ) + + await listen_task + + async def generate_and_play_response(user_input, conversation_history): """ Generates response using Azure OpenAI model and plays using TTS streaming. diff --git a/requirements.txt b/requirements.txt index 94a7352..7966b6c 100644 --- a/requirements.txt +++ b/requirements.txt @@ -3,6 +3,8 @@ elevenlabs==1.1.2 openai==1.21.2 PyAudio==0.2.14 python-dotenv>=1.0.1 +requests>=2.31.0 rich>=13.7.1 setuptools==69.5.1 -SpeechRecognition==3.10.3 \ No newline at end of file +SpeechRecognition==3.10.3 +websockets>=12.0 \ No newline at end of file