실시간 트레이스 통합하기
이 섹션에서는 모든 애플리케이션에서 OpenAI Realtime API에 대한 Weave 트레이싱을 활성화하는 데 필요한 최소한의 코드를 소개합니다.- Python
- TypeScript
Weave는 OpenAI Realtime API를 자동으로 패치하므로, 코드 몇 줄만 추가하면 애플리케이션의 오디오 상호작용을 캡처할 수 있습니다. 다음 코드는 Weave와 Realtime API 인테그레이션을 임포트합니다.이 코드를 임포트하여 실행하면 Weave가 사용자와 OpenAI Realtime API 간의 상호작용을 자동으로 트레이스합니다.
import weave
from weave.integrations import patch_openai_realtime
weave.init("your-team-name/your-project-name")
patch_openai_realtime()
# 애플리케이션 로직
TypeScript에서는 Weave가 OpenAI Agents SDK를 통해 Realtime API를 트레이스합니다. Weave를 임포트하면 동적 임포트나 일부 번들러처럼 Weave가 자동 계측에 사용하는 Node.js 모듈 로더 훅을 우회하는 설정이라면, 애플리케이션이 세션을 생성하기 전 시작 시점에
@openai/agents-realtime의 모든 RealtimeSession이 자동으로 계측되므로 패치 함수를 따로 호출할 필요가 없습니다. Weave 프로젝트만 초기화하면 됩니다.import * as weave from 'weave';
await weave.init('your-team-name/your-project-name');
// 애플리케이션 로직
patchRealtimeSession()을 한 번 호출하세요.import { patchRealtimeSession } from 'weave';
await patchRealtimeSession();
OpenAI Agents SDK로 실시간 에이전트 실행하기
- Python
- TypeScript
이 예제는 마이크 오디오를 OpenAI의 Realtime API로 스트리밍하고, AI의 음성 응답을 로컬 머신의 스피커로 재생하는 실시간 음성 어시스턴트를 실행합니다. 이 애플리케이션은 OpenAI Agents SDK의
RealtimeAgent와 RealtimeRunner를 사용하며, patch_openai_realtime()으로 패치를 적용해 Weave 트레이싱을 활성화합니다. Realtime 세션 관리를 상위 수준의 Agents SDK에 맡기려면 이 방법을 사용하세요.예제를 실행하려면 다음 단계를 완료하세요.-
Python 환경을 시작하고 다음 라이브러리를 설치하세요.
uv add weave openai-agents websockets pyaudio numpypip install weave openai-agents websockets pyaudio numpy -
weave_voice_assistant.py파일을 만들고 다음 코드를 추가하세요. 강조 표시된 줄은 애플리케이션에 Weave를 인테그레이션하는 부분입니다. 나머지 코드는 기본 음성 어시스턴트 앱을 구성합니다.weave_voice_assistant.py
import argparse import asyncio import queue import sys import termios import threading import tty import weave import pyaudio import numpy as np from weave.integrations import patch_openai_realtime from agents.realtime import RealtimeAgent, RealtimeRunner DEFAULT_WEAVE_PROJECT = "<your-team-name/your-project-name>" FORMAT = pyaudio.paInt16 RATE = 24000 # OpenAI Realtime API에서 요구하는 값입니다. CHUNK = 1024 MAX_INPUT_CHANNELS = 1 MAX_OUTPUT_CHANNELS = 1 INP_DEV_IDX = None OUT_DEV_IDX = None # Weave 프로젝트 이름과 오디오 장치 선택에 사용할 CLI 인수를 파싱합니다. def parse_args(): parser = argparse.ArgumentParser(description="Realtime agent with Weave logging") parser.add_argument( "--weave-project", default=DEFAULT_WEAVE_PROJECT, help=f"Weave project name (default: {DEFAULT_WEAVE_PROJECT})", dest="weave_project" ) parser.add_argument( "--input-device", type=int, default=None, help="PyAudio input (mic) device index. Defaults to system default. Run mic_detect.py to list devices.", dest="input_device" ) parser.add_argument( "--output-device", type=int, default=None, help="PyAudio output (speaker) device index. Defaults to system default. Run mic_detect.py to list devices.", dest="output_device" ) return parser.parse_args() # Weave를 초기화하고, 트레이싱을 위해 OpenAI Realtime API를 패치합니다. def init_weave(project_name: str | None = None) -> None: name = project_name or DEFAULT_WEAVE_PROJECT weave.init(name) patch_openai_realtime() # Realtime API 세션을 자동으로 트레이싱합니다. mic_enabled = True # 't' 키 입력을 감지해 마이크를 켜거나 끕니다. 데몬 스레드에서 실행됩니다. def start_keylistener(): global mic_enabled fd = sys.stdin.fileno() old_settings = termios.tcgetattr(fd) try: tty.setcbreak(fd) while True: ch = sys.stdin.read(1) if ch.lower() == 't': mic_enabled = not mic_enabled state = "ON" if mic_enabled else "OFF" print(f"\n🎙 Mic {state} (press t to toggle)") elif ch == '\x03': # Ctrl-C break finally: termios.tcsetattr(fd, termios.TCSADRAIN, old_settings) # 백그라운드 스레드에서 오디오 큐의 데이터를 꺼내 스피커로 출력합니다. def play_audio(output_stream: pyaudio.Stream, audio_output_queue: queue.Queue): while True: data = audio_output_queue.get() if data is None: break output_stream.write(data) # 오디오 스트림을 열고 Realtime 세션을 시작한 뒤 송수신 루프를 실행합니다. async def main(*, input_device_index: int | None = None, output_device_index: int | None = None): p = pyaudio.PyAudio() if input_device_index is None: input_device_index = int(p.get_default_input_device_info()['index']) if output_device_index is None: output_device_index = int(p.get_default_output_device_info()['index']) # pyaudio 오류를 방지하기 위해 채널 수를 장치가 지원하는 범위로 제한합니다. input_info = p.get_device_info_by_index(input_device_index) output_info = p.get_device_info_by_index(output_device_index) input_channels = min(int(input_info['maxInputChannels']), MAX_INPUT_CHANNELS) output_channels = min(int(output_info['maxOutputChannels']), MAX_OUTPUT_CHANNELS) mic = p.open( format=FORMAT, channels=input_channels, rate=RATE, input=True, output=False, frames_per_buffer=CHUNK, input_device_index=input_device_index, start=False, ) speaker = p.open( format=FORMAT, channels=output_channels, rate=RATE, input=False, output=True, frames_per_buffer=CHUNK, output_device_index=output_device_index, start=False, ) mic.start_stream() speaker.start_stream() # 중단 시 재생 대기 중인 오디오를 플러시할 수 있도록 큐로 오디오를 버퍼링합니다. audio_output_queue = queue.Queue() threading.Thread( target=play_audio, args=(speaker, audio_output_queue), daemon=True ).start() s_agent = RealtimeAgent( name="Speech Assistant", instructions="You are a tool using AI. Use tools to accomplish a task whenever possible" ) s_runner = RealtimeRunner(s_agent, config={ "model_settings": { "model_name": "gpt-realtime", "modalities": ["audio"], "output_modalities": ["audio"], "input_audio_format": "pcm16", "output_audio_format": "pcm16", "speed": 1.2, "turn_detection": { "prefix_padding_ms": 100, "silence_duration_ms": 100, "type": "server_vad", "interrupt_response": True, "create_response": True, }, } }) print("--- Session Active (Speak into mic) ---") print("🎙 Mic ON (press t to toggle)") threading.Thread(target=start_keylistener, daemon=True).start() async with await s_runner.run() as session: # 마이크 입력을 Realtime API로 스트리밍하고, 음소거 상태에서는 무음을 전송합니다. async def send_mic_audio(): silence = b'\x00' * CHUNK * 2 # 샘플당 2바이트(16비트 PCM)입니다. try: while True: raw_data = mic.read(CHUNK, exception_on_overflow=False) if mic_enabled: audio_data = np.frombuffer(raw_data, dtype=np.int16).astype(np.float64) rms = np.sqrt(np.mean(audio_data**2)) meter = int(min(rms / 50, 50)) print(f"Mic Level: {'█' * meter}{' ' * (50-meter)} | 🎙 ON ", end="\r") await session.send_audio(raw_data) else: print(f"Mic Level: {' ' * 50} | 🎙 OFF", end="\r") await session.send_audio(silence) await asyncio.sleep(0) # 읽기 작업 사이에 이벤트 루프에 제어권을 넘깁니다. except Exception: pass # 세션에서 이벤트를 수신하고 오디오를 스피커로 전달합니다. async def handle_events(): async for event in session: if event.type == "audio": audio_output_queue.put(event.audio.data) elif event.type == "audio_interrupted": # AI가 사용자의 말과 겹쳐 말하지 않도록 큐에 쌓인 AI 오디오를 플러시합니다. while not audio_output_queue.empty(): try: audio_output_queue.get_nowait() except queue.Empty: break mic_task = asyncio.create_task(send_mic_audio()) try: await handle_events() finally: mic_task.cancel() # 정리 audio_output_queue.put(None) # 재생 스레드에 종료 신호를 보냅니다. mic.close() speaker.close() p.terminate() if __name__ == "__main__": args = parse_args() init_weave(args.weave_project) fd = sys.stdin.fileno() old_settings = termios.tcgetattr(fd) try: asyncio.run(main(input_device_index=args.input_device, output_device_index=args.output_device)) finally: termios.tcsetattr(fd, termios.TCSADRAIN, old_settings) -
DEFAULT_WEAVE_PROJECT값을 사용 중인 팀 이름과 프로젝트 이름으로 변경하세요. -
OPENAI_API_KEY환경 변수를 설정하세요. -
음성 어시스턴트를 시작하세요:
python weave_voice_assistant.py
T를 눌러 마이크 음소거를 켜거나 끄세요. 어시스턴트는 서버 측 음성 활동 감지를 사용해 차례 전환과 끼어들기를 처리합니다.어시스턴트와 대화하는 동안 Weave가 세션 오디오를 포함한 트레이스를 캡처하며, 이 트레이스는 Weights & Biases UI에서 살펴볼 수 있습니다.이 예제에서는
RealtimeAgent와 RealtimeSession으로 실시간 에이전트를 만들고, 날씨 도구를 제공한 다음, 텍스트 메시지를 보냅니다. 세션을 임포트하면 Weave가 자동으로 계측하여 세션, 생성, 도구 Call, 오디오 출력을 트레이스로 캡처합니다.예제를 실행하려면 다음 단계를 완료하세요.-
새 디렉터리에서 TypeScript 기반 Node.js 프로젝트를 설정하고 다음 라이브러리를 설치하세요.
npm init -y npm install weave @openai/agents zod npm install --save-dev typescript @types/node npx tsc --init npm pkg set type=module -
weave_realtime_agent.ts파일을 만들고 다음 코드를 추가하세요. 강조 표시된 줄은 애플리케이션에 Weave를 통합하는 부분이며, 나머지 코드는 기본 실시간 에이전트를 구성합니다.weave_realtime_agent.ts
import * as weave from 'weave'; import {RealtimeAgent, RealtimeSession, tool} from '@openai/agents/realtime'; import {z} from 'zod'; const WEAVE_PROJECT = '<your-team-name/your-project-name>'; // 에이전트가 호출할 시뮬레이션용 날씨 조회 도구입니다. const getWeatherTool = tool({ name: 'get_weather', description: 'Get the current weather for a given city.', parameters: z.object({ city: z.string().describe('The name of the city'), }), async execute({city}) { const weather: Record<string, string> = { 'San Francisco': '65°F, foggy', 'New York': '72°F, sunny', London: '55°F, cloudy', }; return weather[city] ?? '70°F, clear'; }, }); const agent = new RealtimeAgent({ name: 'Assistant', instructions: 'You are a helpful voice assistant. Keep responses concise because this is a voice conversation.', tools: [getWeatherTool], }); async function main() { await weave.init(WEAVE_PROJECT); // `@openai/agents-realtime`을 임포트하면 Weave가 자동으로 계측합니다. // 설정에서 모듈 로더 훅을 우회하는 경우 다음 줄의 주석을 해제하세요. // await weave.patchRealtimeSession(); const apiKey = process.env.OPENAI_API_KEY; if (!apiKey) throw new Error('OPENAI_API_KEY is required'); const session = new RealtimeSession(agent); // 대화 이력이 업데이트될 때마다 assistant의 텍스트 응답을 출력합니다. session.on('history_updated', history => { const last = history[history.length - 1]; if (last?.type === 'message' && last.role === 'assistant') { const text = (last.content ?? []) .filter((c: any) => c.type === 'output_text') .map((c: any) => c.text) .join(''); if (text) console.log(`\nAssistant: ${text}`); } }); session.on('error', event => { console.error('[error]', event.error); }); console.log('Connecting...'); await session.connect({apiKey}); console.log('Connected.'); session.sendMessage('What is the weather like in San Francisco?'); // 응답을 기다린 후 연결을 끊습니다. await new Promise<void>(resolve => setTimeout(resolve, 10_000)); session.close(); console.log('\nDone.'); } main().catch(console.error); -
WEAVE_PROJECT값을 본인의 팀 이름과 프로젝트 이름으로 변경하세요. -
OPENAI_API_KEY및WANDB_API_KEY환경 변수를 설정하세요. -
에이전트를 컴파일하고 실행하세요.
npx tsc node --import=weave/instrument --import=tsx weave_realtime_agent.js
--import 플래그는 Weave 계측을 미리 로드하고, 두 번째 --import 플래그는 Node가 TypeScript를 실행할 수 있도록 tsx를 미리 로드합니다.에이전트는 Realtime API에 연결해 San Francisco의 날씨를 묻고 날씨 도구를 호출한 뒤 연결을 끊습니다. Weights & Biases UI에서 세션, 생성, 도구 Call, assistant의 오디오 출력 등 생성된 트레이스를 살펴볼 수 있습니다.WebSocket으로 실시간 음성 어시스턴트 실행하기
다음 Python 예시는 WebSocket을 통해 OpenAI Realtime API에 직접 연결합니다. 이 예시는 마이크 오디오를 API로 스트리밍하고, 음성 응답을 재생하며, 도구 호출(날씨 조회, 수식 계산, 코드 실행, 파일 쓰기)을 지원합니다. Weave는weave.init()과 patch_openai_realtime()을 사용해 세션을 트레이스합니다. Agents SDK에 의존하지 않고 Realtime 세션을 완전히 제어하려면 이 방식을 사용하세요.
예시를 실행하려면 다음 단계를 완료하세요.
-
Python 환경을 실행하고 다음 라이브러리를 설치하세요:
uv add weave websockets pyaudio numpypip install weave websockets pyaudio numpy -
tool_definitions.py라는 이름의 파일을 만들고 다음 도구 정의를 추가하세요. 메인 애플리케이션은 이 모듈에서 필요한 항목을 임포트합니다.tool_definitions.py
import json import subprocess import tempfile from pathlib import Path import weave # @function_tool @weave.op def get_weather(city: str) -> str: """Get the current weather for a city. Args: city: The city name to get weather for. """ return json.dumps({"city": city, "temperature": "72°F", "condition": "sunny"}) @weave.op def calculate(expression: str) -> str: """Evaluate a math expression and return the result. Args: expression: A math expression to evaluate, e.g. '2 + 2'. """ try: result = eval(expression) return str(result) except Exception as e: return f"Error: {e}" @weave.op def run_python_code(code: str) -> str: """Write and execute a Python script, returning its stdout/stderr. Args: code: The Python source code to execute. """ with tempfile.NamedTemporaryFile( mode="w", suffix=".py", dir=tempfile.gettempdir(), delete=False ) as f: f.write(code) script_path = Path(f.name) try: result = subprocess.run( ["python", str(script_path)], capture_output=True, text=True, timeout=30, ) output = result.stdout if result.stderr: output += f"\nSTDERR:\n{result.stderr}" if result.returncode != 0: output += f"\n(exit code {result.returncode})" return output or "(no output)" except subprocess.TimeoutExpired: return "Error: script timed out after 30 seconds." finally: script_path.unlink(missing_ok=True) @weave.op async def write_file(file_path: str, content: str) -> str: """Write content to a file on disk. Args: file_path: The path to write the file to. content: The content to write into the file. """ try: path = Path(file_path) path.parent.mkdir(parents=True, exist_ok=True) path.write_text(content) return f"Wrote {len(content)} bytes to {file_path}" except Exception as e: return f"Error writing file: {e}" -
같은 디렉터리에
weave_ws_voice_assistant.py파일을 만들고 다음 코드를 추가하세요.weave_ws_voice_assistant.py
import asyncio import base64 import json import os import queue import threading from typing import Any, Callable import numpy as np import pyaudio import websockets import weave weave.init("<your-team-name/your-project-name>") from weave.integrations import patch_openai_realtime patch_openai_realtime() from tool_definitions import ( calculate, get_weather, run_python_code, write_file, ) # 오디오 형식(Realtime API에서는 반드시 PCM16이어야 함). FORMAT = pyaudio.paInt16 RATE = 24000 CHUNK = 1024 MAX_INPUT_CHANNELS = 2 MAX_OUTPUT_CHANNELS = 2 OPENAI_API_KEY = os.environ.get("OPENAI_API_KEY", "") REALTIME_URL = "wss://api.openai.com/v1/realtime?model=gpt-realtime" DEBUG_WRITE_LOG = False # 함수 호출을 디스패치할 수 있도록 도구 이름을 callable에 매핑합니다. TOOL_REGISTRY: dict[str, Callable[..., Any]] = { "get_weather": get_weather, "calculate": calculate, "run_python_code": run_python_code, "write_file": write_file, } # Realtime API 세션 설정에 사용할 원시 도구 정의입니다. TOOL_DEFINITIONS = [ { "type": "function", "name": "get_weather", "description": "Get the current weather for a city.", "parameters": { "type": "object", "properties": { "city": { "type": "string", "description": "The city name to get weather for.", } }, "required": ["city"], }, }, { "type": "function", "name": "calculate", "description": "Evaluate a math expression and return the result.", "parameters": { "type": "object", "properties": { "expression": { "type": "string", "description": "A math expression to evaluate, e.g. '2 + 2'.", } }, "required": ["expression"], }, }, { "type": "function", "name": "run_python_code", "description": "Write and execute a Python script, returning its stdout/stderr.", "parameters": { "type": "object", "properties": { "code": { "type": "string", "description": "The Python source code to execute.", } }, "required": ["code"], }, }, { "type": "function", "name": "write_file", "description": "Write content to a file on disk.", "parameters": { "type": "object", "properties": { "file_path": { "type": "string", "description": "The path to write the file to.", }, "content": { "type": "string", "description": "The content to write into the file.", }, }, "required": ["file_path", "content"], }, }, ] async def send_event(ws, event: dict) -> None: await ws.send(json.dumps(event)) async def configure_session(ws) -> None: event = { "type": "session.update", "session": { "type": "realtime", "model": "gpt-realtime", "output_modalities": ["audio"], "instructions": ( "You are a helpful AI assistant with access to tools. " "Use tools to accomplish tasks whenever possible. " "Speak clearly and briefly." ), "tools": TOOL_DEFINITIONS, "tool_choice": "auto", "audio": { "input": { "format": {"type": "audio/pcm", "rate": 24000}, "transcription": {"model": "gpt-4o-transcribe"}, "turn_detection": { "type": "server_vad", "threshold": 0.5, "prefix_padding_ms": 300, "silence_duration_ms": 500, }, }, "output": { "format": {"type": "audio/pcm", "rate": 24000}, }, }, }, } await send_event(ws, event) print("Session configured.") async def handle_function_call(ws, call_id: str, name: str, arguments: str) -> None: if not name: raise Exception("Did not get a function name") print(f"\n[Function Call] {name}({arguments})") tool_fn = TOOL_REGISTRY.get(name) if tool_fn is None: result = json.dumps({"error": f"Unknown function: {name}"}) else: try: args = json.loads(arguments) result = tool_fn(**args) if asyncio.iscoroutine(result): result = await result except Exception as e: result = json.dumps({"error": str(e)}) print(f"[Function Result] {result}") # 함수 호출 출력을 모델에 다시 전달합니다. await send_event(ws, { "type": "conversation.item.create", "item": { "type": "function_call_output", "call_id": call_id, "output": result if isinstance(result, str) else json.dumps(result), }, }) # 모델이 함수 결과를 반영하도록 새 응답을 트리거합니다. await send_event(ws, {"type": "response.create"}) def play_audio(output_stream: pyaudio.Stream, audio_output_queue: queue.Queue): """Runs in a separate thread because pyaudio's write() blocks until the sound card consumes the samples. Decoupling playback from the async event loop lets us flush the queue on interrupt without waiting for in-flight writes to finish.""" while True: data = audio_output_queue.get() if data is None: break output_stream.write(data) async def send_mic_audio(ws, mic) -> None: try: while True: raw_data = mic.read(CHUNK, exception_on_overflow=False) # 볼륨 미터를 화면에 표시합니다. audio_data = np.frombuffer(raw_data, dtype=np.int16).astype(np.float64) rms = np.sqrt(np.mean(audio_data**2)) meter = int(min(rms / 50, 50)) print(f"Mic Level: {'█' * meter}{' ' * (50 - meter)} |", end="\r") # 오디오 청크를 Base64로 인코딩하여 전송합니다. b64_audio = base64.b64encode(raw_data).decode("utf-8") await send_event(ws, { "type": "input_audio_buffer.append", "audio": b64_audio, }) await asyncio.sleep(0) except asyncio.CancelledError: pass async def receive_events(ws, audio_output_queue: queue.Queue) -> None: # 여러 delta 이벤트에 걸쳐 들어오는 함수 호출 인수를 누적합니다. pending_calls: dict[str, dict] = {} async for raw_message in ws: if DEBUG_WRITE_LOG: with open("data.jsonl", "a", encoding="utf-8") as f: f.write(json.dumps(raw_message) + "\n") event = json.loads(raw_message) event_type = event.get("type", "") if event_type == "session.created": print(raw_message) elif event_type == "session.updated": print(raw_message) elif event_type == "error": print(f"\n[Error] {event}") elif event_type == "input_audio_buffer.speech_started": # AI가 사용자의 말과 겹쳐 말하지 않도록 큐에 쌓인 AI 오디오를 플러시합니다. while not audio_output_queue.empty(): try: audio_output_queue.get_nowait() except queue.Empty: break elif event_type == "input_audio_buffer.speech_stopped": pass elif event_type == "input_audio_buffer.committed": pass elif event_type == "response.created": pass elif event_type == "response.output_text.delta": pass elif event_type == "response.output_text.done": pass # 오디오 출력 delta - 재생할 수 있도록 큐에 추가합니다. elif event_type == "response.output_audio.delta": audio_bytes = base64.b64decode(event.get("delta", "")) audio_output_queue.put(audio_bytes) elif event_type == "response.output_audio_transcript.delta": pass elif event_type == "response.output_audio_transcript.done": pass # 함수 호출 시작 - 대기 중인 호출을 초기화합니다. elif event_type == "response.output_item.added": item = event.get("item", {}) if item.get("type") == "function_call" and item.get("status") == "in_progress": item_id = item.get("id", "") pending_calls[item_id] = { "call_id": item.get("call_id", ""), "name": item.get("name", ""), "arguments": "", } print(f"\n[Function Call Started] {item.get('name', '')}") # 함수 호출 인수 delta - 누적합니다. elif event_type == "response.function_call_arguments.delta": item_id = event.get("item_id", "") if item_id in pending_calls: pending_calls[item_id]["arguments"] += event.get("delta", "") elif event_type == "response.function_call_arguments.done": item_id = event.get("item_id", "") call_info = pending_calls.pop(item_id, None) if call_info is None: # 폴백: done 이벤트의 데이터를 직접 사용합니다. call_info = { "call_id": event.get("call_id"), "name": event.get("name"), "arguments": event.get("arguments"), } try: await handle_function_call( ws, call_info["call_id"], call_info["name"], call_info["arguments"], ) except Exception as e: print(f"Failed to call function for message {call_info}: error - {e}") elif event_type == "response.done": pass elif event_type == "rate_limits.updated": pass else: print(f"\n[Event: {event_type}]") async def main(): if not OPENAI_API_KEY: print("Error: OPENAI_API_KEY environment variable not set") return p = pyaudio.PyAudio() input_device_index = int(p.get_default_input_device_info()['index']) output_device_index = int(p.get_default_output_device_info()['index']) # 채널 수는 장치가 지원하는 범위와 일치해야 합니다. 그렇지 않으면 open 시 pyaudio에서 오류가 발생합니다. input_info = p.get_device_info_by_index(input_device_index) output_info = p.get_device_info_by_index(output_device_index) input_channels = min(int(input_info['maxInputChannels']), 1) output_channels = min(int(output_info['maxOutputChannels']), 1) mic = p.open( format=FORMAT, channels=input_channels, rate=RATE, input=True, output=False, frames_per_buffer=CHUNK, input_device_index=input_device_index, start=False, ) speaker = p.open( format=FORMAT, channels=output_channels, rate=RATE, input=False, output=True, frames_per_buffer=CHUNK, output_device_index=output_device_index, start=False, ) mic.start_stream() speaker.start_stream() # 사용자가 말을 끊을 때 오디오를 플러시할 수 있도록 오디오를 큐를 거쳐 전달합니다. # 스피커에 직접 쓰면 이미 재생 중인 오디오를 취소할 수 없습니다. audio_output_queue = queue.Queue() threading.Thread( target=play_audio, args=(speaker, audio_output_queue), daemon=True ).start() headers = { "Authorization": f"Bearer {OPENAI_API_KEY}", } print("Connecting to OpenAI Realtime API...") async with websockets.connect( REALTIME_URL, additional_headers=headers, ) as ws: print("Connected! Configuring session...") await configure_session(ws) print("--- Session Active (Speak into mic) ---") mic_task = asyncio.create_task(send_mic_audio(ws, mic)) try: await receive_events(ws, audio_output_queue) finally: mic_task.cancel() try: await mic_task except asyncio.CancelledError: pass # 정리 audio_output_queue.put(None) # 재생 스레드에 종료 신호를 보냅니다. mic.close() speaker.close() p.terminate() print("\nSession ended.") if __name__ == "__main__": asyncio.run(main()) -
weave.init()호출에 사용자의 팀 이름과 프로젝트 이름을 입력하세요. -
OPENAI_API_KEY환경 변수를 설정하세요. -
음성 어시스턴트를 실행하세요:
python weave_ws_voice_assistant.py
T를 눌러 마이크 음소거를 켜거나 끌 수 있습니다. 어시스턴트는 서버 측 음성 활동 감지를 사용해 차례 전환과 끼어들기를 처리합니다.
어시스턴트와 대화하는 동안 Weave가 세션 오디오를 포함한 트레이스를 캡처하며, 캡처된 트레이스는 Weights & Biases UI에서 살펴볼 수 있습니다.