From 85eb2e3ac999a7b4e6588afbdaea9686bf981475 Mon Sep 17 00:00:00 2001 From: Claas Date: Mon, 16 Jun 2025 18:14:54 +0000 Subject: [PATCH] Create conversation audio --- .devcontainer/Dockerfile | 5 +++ .gitignore | 3 +- agent.py | 27 +++++++++++-- constants.py | 2 +- main.py | 85 +++++++++++++++++++++++++++++----------- pyproject.toml | 2 + uv.lock | 22 +++++++++++ 7 files changed, 119 insertions(+), 27 deletions(-) diff --git a/.devcontainer/Dockerfile b/.devcontainer/Dockerfile index 70109f0..d73271a 100644 --- a/.devcontainer/Dockerfile +++ b/.devcontainer/Dockerfile @@ -1,3 +1,8 @@ FROM mcr.microsoft.com/devcontainers/python:1-3.12-bullseye +# Install ffmpeg to merge audio files using pydub +RUN apt-get update && apt-get install -y --no-install-recommends \ + ffmpeg \ + && rm -rf /var/lib/apt/lists/* + RUN pipx install uv \ No newline at end of file diff --git a/.gitignore b/.gitignore index 63b3298..af4b905 100644 --- a/.gitignore +++ b/.gitignore @@ -1,3 +1,4 @@ .env __pycache__/ -.db \ No newline at end of file +.db +.DS_Store \ No newline at end of file diff --git a/agent.py b/agent.py index 536becc..2cb459a 100644 --- a/agent.py +++ b/agent.py @@ -1,5 +1,7 @@ +import random from sqlite3 import Connection +import typing from openai import OpenAI from openai.types.responses import ResponseInputParam from pydantic import BaseModel, Field @@ -7,6 +9,16 @@ from pydantic import BaseModel, Field from constants import MODEL import database +type OpenAiTextToSpeechVoices = typing.Literal['alloy', 'ash', + 'coral', 'echo', 'fable', 'onyx', 'nova', 'sage', 'shimmer'] + + +voice_options = list(typing.get_args( + # Duplicating type because get_args does not work with type alias of OpenAiTextToSpeechVoices and Idl anymore + typing.Literal['alloy', 'ash', 'coral', 'echo', 'fable', 'onyx', 'nova', 'sage', 'shimmer'])) +print("VOICE OPTIONS", voice_options) +random.shuffle(voice_options) + class Response(BaseModel): inner_thoughts: str = Field( @@ -19,16 +31,25 @@ class Agent: client: OpenAI history: ResponseInputParam connection: Connection + role_description: str + text_to_speech_voice: OpenAiTextToSpeechVoices - def __init__(self, client: OpenAI, connection: Connection, message: str, initial_message: str | None = None): - assert isinstance(message, str) + def __init__(self, client: OpenAI, connection: Connection, role_description: str, initial_message: str | None = None): + assert isinstance(role_description, str) self.client = client self.connection = connection + self.role_description = role_description + + if len(voice_options) == 0: + raise "All voices for the conversation have been taken. There is no more voice available to differentiate agents in the conversation" + + self.text_to_speech_voice = voice_options.pop() + # Instruct the character self.history = [ { "role": "system", - "content": message + "content": role_description }, ] if initial_message != None: diff --git a/constants.py b/constants.py index 1a0a925..d7e0696 100644 --- a/constants.py +++ b/constants.py @@ -1,3 +1,3 @@ # Project wide constants. Mainly to avoid circular dependencies MODEL = "gpt-4o-mini" -CONVERSATION_TURNS = 20 +CONVERSATION_TURNS = 5 diff --git a/main.py b/main.py index 13223b5..2498e60 100644 --- a/main.py +++ b/main.py @@ -1,7 +1,11 @@ import os +from pathlib import Path +import time from dotenv import load_dotenv +import nanoid from openai import OpenAI from pydantic import BaseModel, Field +from pydub import AudioSegment from agent import Agent from constants import CONVERSATION_TURNS, MODEL @@ -17,6 +21,11 @@ class RoleSeed(BaseModel): "The initial message role 1 should start the conversation with") +def ensure_directory_exists(path: Path): + if not os.path.exists(path): + os.makedirs(path) + + def main(): print("Starting AI conversation!") load_dotenv() @@ -28,42 +37,74 @@ def main(): response = client.responses.parse( model=MODEL, - input=[{"role": "user", "content": "Create two role instructions for two opposing that have conflicting interests in an awkward confrontation. The descriptions are handed to actors to act out the roles in an improvised conversation. The opposing interests should create a funny ongoing conversation. Write it in a way that each role is aware of the other without revealing too much detail. One of the roles tries to manipulate and cheat the other. Instruct each role to be short and concise."}], + input=[{"role": "user", "content": "Create two role instructions for two opposing people that have conflicting interests in an awkward confrontation. The descriptions are handed to actors to act out the roles in an improvised conversation. The opposing interests should create a funny ongoing conversation. Write it in a way that each role is aware of the other without revealing too much detail. One of the roles tries to manipulate and cheat the other. Instruct each role to be short and concise."}], text_format=RoleSeed, ) database.create_usage( connection, response.usage.input_tokens, response.usage.output_tokens) - role_1 = response.output_parsed.role_1_description - role_1_initial = response.output_parsed.role_1_initial_message - role_2 = response.output_parsed.role_2_description + role_1_description = response.output_parsed.role_1_description + role_1_initial_message = response.output_parsed.role_1_initial_message + role_2_description = response.output_parsed.role_2_description - print(f"Role 1: {role_1}") - print(f"Role 2: {role_2}") + print(f"Role 1: {role_1_description}") + print(f"Role 2: {role_2_description}") agent_1 = Agent( - client, connection, role_1, role_1_initial) + client, connection, role_1_description, role_1_initial_message) agent_2 = Agent( - client, connection, role_2) + client, connection, role_2_description) agents = [agent_1, agent_2] - # Iterable that switches between 0 and 1 to switch between agent 0 and 1 - # to switch for each chat turn - agent_turn_indices = map(lambda turn_index: turn_index % - len(agents), range(CONVERSATION_TURNS)) - # Ask initial question to get the chat rolling - last_message = role_1_initial + last_message = role_1_initial_message print(f"[Agent 1 Initial Message]:\t{last_message}") - for agent_index in agent_turn_indices: - current_agent = agents[agent_index] - (last_message, inner_thoughts) = current_agent.message(last_message) - print(f"\n[Agent {agent_index + 1} inner thoughts]: {inner_thoughts}") - print( - f"\n[Agent {agent_index + 1}]: {last_message}") - - print("Conversation completed!") + conversation_id = nanoid.generate() + timestamp = time.strftime("%Y%m%d-%H%M%S") + conversations_directory = Path.cwd() / "conversations" / \ + f"{timestamp}-{conversation_id}" + turns_directory = conversations_directory / "turns" + ensure_directory_exists(turns_directory) + print(f"Writing audio files to: {conversations_directory}") + + def generate_audio(agent: Agent, turn: int, message: str) -> Path: + turn_path = turns_directory / f"{turn}.mp3" + with client.audio.speech.with_streaming_response.create( + model="gpt-4o-mini-tts", + voice=agent.text_to_speech_voice, + input=message, + instructions=agent.role_description, + ) as response: + response.stream_to_file(turn_path) + + return turn_path + + conversation_audio = AudioSegment.empty() + try: + turn_files = [] + + turn_path = generate_audio(agent_1, 0, last_message) + turn_files.append(turn_path) + segment = AudioSegment.from_mp3(turn_path) + conversation_audio += segment + + for turn in range(1, CONVERSATION_TURNS): + agent_index = turn % len(agents) + current_agent = agents[agent_index] + (last_message, inner_thoughts) = current_agent.message(last_message) + print( + f"\n[Agent {agent_index + 1} inner thoughts]:\t{inner_thoughts}") + print( + f"\n[Agent {agent_index + 1}]:\t{last_message}") + turn_path = generate_audio(current_agent, turn, last_message) + turn_files.append(turn_path) + segment = AudioSegment.from_mp3(turn_path) + conversation_audio += segment + + print("Conversation completed!") + finally: + conversation_audio.export(conversations_directory / "full.mp3") if __name__ == "__main__": diff --git a/pyproject.toml b/pyproject.toml index 3acc80c..f16d9bf 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -6,7 +6,9 @@ readme = "README.md" requires-python = ">=3.12" dependencies = [ "instructor>=1.8.2", + "nanoid>=2.0.0", "openai>=1.68.2", "pydantic>=2.10.6", + "pydub>=0.25.1", "python-dotenv>=1.0.1", ] diff --git a/uv.lock b/uv.lock index 02c1821..dccb125 100644 --- a/uv.lock +++ b/uv.lock @@ -193,16 +193,20 @@ version = "0.1.0" source = { virtual = "." } dependencies = [ { name = "instructor" }, + { name = "nanoid" }, { name = "openai" }, { name = "pydantic" }, + { name = "pydub" }, { name = "python-dotenv" }, ] [package.metadata] requires-dist = [ { name = "instructor", specifier = ">=1.8.2" }, + { name = "nanoid", specifier = ">=2.0.0" }, { name = "openai", specifier = ">=1.68.2" }, { name = "pydantic", specifier = ">=2.10.6" }, + { name = "pydub", specifier = ">=0.25.1" }, { name = "python-dotenv", specifier = ">=1.0.1" }, ] @@ -500,6 +504,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/84/5d/e17845bb0fa76334477d5de38654d27946d5b5d3695443987a094a71b440/multidict-6.4.4-py3-none-any.whl", hash = "sha256:bd4557071b561a8b3b6075c3ce93cf9bfb6182cb241805c3d66ced3b75eff4ac", size = 10481 }, ] +[[package]] +name = "nanoid" +version = "2.0.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/b7/9d/0250bf5935d88e214df469d35eccc0f6ff7e9db046fc8a9aeb4b2a192775/nanoid-2.0.0.tar.gz", hash = "sha256:5a80cad5e9c6e9ae3a41fa2fb34ae189f7cb420b2a5d8f82bd9d23466e4efa68", size = 3290 } +wheels = [ + { url = "https://files.pythonhosted.org/packages/2e/0d/8630f13998638dc01e187fadd2e5c6d42d127d08aeb4943d231664d6e539/nanoid-2.0.0-py3-none-any.whl", hash = "sha256:90aefa650e328cffb0893bbd4c236cfd44c48bc1f2d0b525ecc53c3187b653bb", size = 5844 }, +] + [[package]] name = "openai" version = "1.68.2" @@ -629,6 +642,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/51/b2/b2b50d5ecf21acf870190ae5d093602d95f66c9c31f9d5de6062eb329ad1/pydantic_core-2.27.2-cp313-cp313-win_arm64.whl", hash = "sha256:ac4dbfd1691affb8f48c2c13241a2e3b60ff23247cbcf981759c768b6633cf8b", size = 1885186 }, ] +[[package]] +name = "pydub" +version = "0.25.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/fe/9a/e6bca0eed82db26562c73b5076539a4a08d3cffd19c3cc5913a3e61145fd/pydub-0.25.1.tar.gz", hash = "sha256:980a33ce9949cab2a569606b65674d748ecbca4f0796887fd6f46173a7b0d30f", size = 38326 } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a6/53/d78dc063216e62fc55f6b2eebb447f6a4b0a59f55c8406376f76bf959b08/pydub-0.25.1-py2.py3-none-any.whl", hash = "sha256:65617e33033874b59d87db603aa1ed450633288aefead953b30bded59cb599a6", size = 32327 }, +] + [[package]] name = "pygments" version = "2.19.1" -- 2.51.2