Live audio and video chat
How to use
1. Establish a connection
Qwen-Omni-Realtime supports two protocols: WebSocket and WebRTC. WebSocket is suitable for server-side integration and quick setup. WebRTC is designed for browser-based and low-latency voice scenarios, with audio transmitted directly over UDP and built-in echo cancellation and noise reduction. For session duration and conversation history limits, see Limitations.- WebSocket
- WebRTC
- Native WebSocket connection
- DashScope SDK
The following configuration items are required:
| Configuration item | Description |
|---|---|
| Endpoint | wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime |
| Query parameter | The query parameter is model. Set it to the name of the model you want to access. See Model selection. Example: ?model=qwen3.5-omni-plus-realtime |
| Request header | Use Bearer Token for authentication: Authorization: Bearer DASHSCOPE_API_KEY. DASHSCOPE_API_KEY is the API Key that you requested on QwenCloud. |
Copy
# pip install websocket-client
import json
import websocket
import os
API_KEY=os.getenv("DASHSCOPE_API_KEY")
API_URL = "wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime?model=qwen3.5-omni-plus-realtime"
headers = [
"Authorization: Bearer " + API_KEY
]
def on_open(ws):
print(f"Connected to server: {API_URL}")
def on_message(ws, message):
data = json.loads(message)
print("Received event:", json.dumps(data, indent=2))
def on_error(ws, error):
print("Error:", error)
ws = websocket.WebSocketApp(
API_URL,
header=headers,
on_open=on_open,
on_message=on_message,
on_error=on_error
)
ws.run_forever()
Copy
# SDK version 1.23.9 or later
import os
import json
from dashscope.audio.qwen_omni import OmniRealtimeConversation,OmniRealtimeCallback
import dashscope
# If you have not configured an API key, change the following line to dashscope.api_key = "sk-xxx"
dashscope.api_key = os.getenv("DASHSCOPE_API_KEY")
class PrintCallback(OmniRealtimeCallback):
def on_open(self) -> None:
print("Connected Successfully")
def on_event(self, response: dict) -> None:
print("Received event:")
print(json.dumps(response, indent=2, ensure_ascii=False))
def on_close(self, close_status_code: int, close_msg: str) -> None:
print(f"Connection closed (code={close_status_code}, msg={close_msg}).")
callback = PrintCallback()
conversation = OmniRealtimeConversation(
model="qwen3.5-omni-plus-realtime",
callback=callback,
url="wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime"
)
try:
conversation.connect()
print("Conversation started. Press Ctrl+C to exit.")
conversation.thread.join()
except KeyboardInterrupt:
conversation.close()
Establishing a WebRTC connection involves two stages:
Connection code examples:
- SDP exchange (HTTP): The client sends its media capabilities and network addresses (Offer SDP) to the server via HTTP POST. The server returns its information (Answer SDP) to complete the capability negotiation.
- Connection (automatic): After the negotiation, the WebRTC layer automatically establishes the audio transport channel.
| Parameter | Description |
|---|---|
| Request URL | POST https://{endpoint}/api/v1/webrtc/realtime The WebRTC feature is currently available by allowlist only. Contact your sales manager to get the endpoint. |
| Query parameter | Use the model query parameter to specify the model. Example: ?model=qwen3.5-omni-plus-realtime |
| Content-Type | application/sdp |
| Request header | Authorization: Bearer DASHSCOPE_API_KEY |
| Request body | The client-generated Offer SDP string |
| Response | Success: HTTP 200 with the server Answer SDP string. Failure: HTTP 4xx with a JSON error message. |
Copy
# pip install aiortc aiohttp certifi
import asyncio, aiohttp, ssl, certifi
from aiortc import RTCPeerConnection, RTCConfiguration, RTCSessionDescription
from aiortc.mediastreams import AudioStreamTrack
API_KEY = "your-api-key"
MODEL = "qwen3.5-omni-plus-realtime"
SIGNALING_URL = f"https://{{endpoint}}/api/v1/webrtc/realtime?model={MODEL}"
async def connect():
pc = RTCPeerConnection(RTCConfiguration(iceServers=[]))
# Add an audio track to ensure the Offer SDP contains m=audio (required by the server)
pc.addTrack(AudioStreamTrack())
# Create a DataChannel to trigger SDP negotiation (name is customizable; the server pushes events through a channel named "txt")
pc.createDataChannel("oai-events")
# SDP exchange: create an Offer and send it to the server
offer = await pc.createOffer()
await pc.setLocalDescription(offer)
async with aiohttp.ClientSession() as session:
async with session.post(
SIGNALING_URL,
ssl=ssl.create_default_context(cafile=certifi.where()),
data=offer.sdp.encode("utf-8"),
headers={
"Content-Type": "application/sdp",
"Authorization": f"Bearer {API_KEY}",
},
) as resp:
if not resp.ok:
raise Exception(f"SDP exchange failed: {resp.status} {await resp.text()}")
answer_sdp = await resp.text()
print("=== Offer SDP ===")
print(offer.sdp)
print("=== Answer SDP ===")
print(answer_sdp)
# ICE connection is established automatically
await pc.setRemoteDescription(RTCSessionDescription(sdp=answer_sdp, type="answer"))
print("WebRTC connection established")
return pc
2. Configure the session
Send the session.update client event:Copy
{
// The ID of this event, generated by the client.
"event_id": "event_ToPZqeobitzUJnt3QqtWg",
// The event type. This is fixed to session.update.
"type": "session.update",
// Session configuration.
"session": {
// The output modalities. Supported values are ["text"] (text only) or ["text", "audio"] (text and audio).
"modalities": [
"text",
"audio"
],
// The voice for the output audio.
"voice": "Tina",
// The input audio format. Only "pcm" is supported. The input audio must be a PCM audio stream at a 16 kHz sample rate.
"input_audio_format": "pcm",
// The output audio format. Only "pcm" is supported. The output audio is a PCM audio stream at a 24 kHz sample rate.
"output_audio_format": "pcm",
// The system message. It is used to set the model's goal or role.
"instructions": "You are an AI customer service agent for a five-star hotel. Answer customer inquiries about room types, facilities, prices, and booking policies accurately and friendly. Always respond with a professional and helpful attitude. Do not provide unconfirmed information or information beyond the scope of the hotel's services.",
// Specifies whether to enable voice activity detection. To enable it, pass a configuration object. The server will automatically detect the start and end of speech based on this object.
// Set to null to let the client decide when to initiate a model response.
"turn_detection": {
// The VAD type. Valid values: server_vad and semantic_vad. We recommend that you set this parameter to semantic_vad for models like qwen3.5-omni-realtime.
"type": "semantic_vad",
// The VAD detection threshold. Increase this value in noisy environments and decrease it in quiet environments.
"threshold": 0.5,
// The duration of silence to detect the end of speech. If this value is exceeded, a model response is triggered.
"silence_duration_ms": 800
}
}
}
Qwen3.5-Omni-Realtime models support cloned voices in addition to preset voices. See Voice cloning to create a custom voice from an audio sample.
3. Input audio and images
- WebSocket
- WebRTC
The client sends Base64-encoded audio and image data to the server buffer using the input_audio_buffer.append and input_image_buffer.append events. Audio input is required. Image input is optional and can come from local files or a real-time video stream.
When server-side Voice Activity Detection (VAD) is enabled, the server automatically submits the data and triggers a response when it detects the end of speech. When VAD is disabled (manual mode), the client must call the input_audio_buffer.commit event to submit the data.
The audio and video tracks (RTP media channels) added during connection establishment automatically transmit data to the server.
- Audio: Transmitted directly through the audio track (RTP). No need to send
input_audio_buffer.appendevents. - Images: Sent as video frames through the video track (RTP). The
input_image_buffer.appendevent is not supported.
WebRTC only supports server-side VAD mode (
server_vad or semantic_vad). Manual mode is not supported.4. Receive model responses
The format of the model response depends on the configured output modalities.- WebSocket
- WebRTC
- Text only Receive streaming text through the response.text.delta event. Retrieve the full text with the response.text.done event.
-
Text and audio
- Text: Receive streaming text through the response.audio_transcript.delta event. Retrieve the full text with the response.audio_transcript.done event.
- Audio: Retrieve Base64-encoded streaming audio output data through the response.audio.delta event. The response.audio.done event indicates that audio generation is complete.
- Text only Same as WebSocket. Receive streaming text events through the DataChannel.
-
Text and audio
- Text: Received through the DataChannel as streaming text events, same as WebSocket.
- Audio: Received and played in real time through RTP tracks. No need to retrieve audio data through
response.audio.deltaevents.
Limitations
- Mutually exclusive features: Web search and tool calling cannot be enabled at the same time.
- Session duration: A single WebSocket session can last up to 120 minutes. The connection closes automatically at this limit.
Conversation history limits
The model retains conversation history up to the following turn and duration limits. When exceeded, the oldest history is discarded. Max duration is the cumulative audio or video (image frame) duration retained in context.| Model | Audio max turns | Video max turns | Audio max duration | Video max duration |
|---|---|---|---|---|
| qwen3.5-omni-plus-realtime | 100 turns | 50 turns | 600 seconds | 240 seconds |
| qwen3.5-omni-flash-realtime | 80 turns | 50 turns | 480 seconds | 120 seconds |
| qwen3-omni-flash-realtime | 8 turns | 8 turns | — | — |
- Video is input as extracted frames (recommended: 1 fps). Video max duration is the cumulative frame duration retained — for example, 240 s means only frames from the last 240 seconds are kept.
- The
qwen3-omni-flash-realtimemodel has a limit of 8 dialog turns (typically reached first). Its duration limit depends on the model's context length and is not listed separately.
Getting started
Get an API key and set it as an environment variable.- DashScope Python SDK
- DashScope Java SDK
- WebSocket (Python)
Prepare the runtime environmentYour Python version must be 3.10 or later.First, install PyAudio based on your operating system.Then, you can install the package using pip in the activated virtual environment.After installation completes, install the remaining dependencies using pip:Choose an interaction modeSample code on GitHubDownload complete sample code from GitHub, which includes:
- macOS
- Debian/Ubuntu
- CentOS
- Windows
Copy
brew install portaudio && pip install pyaudio
- If you are not using a virtual environment, install it directly using the system package manager:
Copy
sudo apt-get install python3-pyaudio
- If you are in a virtual environment, first install the compilation dependencies:
Copy
sudo apt update
sudo apt install -y python3-dev portaudio19-dev
Copy
pip install pyaudio
Copy
sudo yum install -y portaudio portaudio-devel && pip install pyaudio
Copy
pip install pyaudio
Copy
pip install websocket-client dashscope
- VAD mode (automatically detects the start and end of speech) The server automatically determines when the user starts and stops speaking and responds accordingly.
- Manual mode (press to talk, release to send) The client controls the start and end of speech. After the user finishes speaking, the client must actively send a message to the server.
- VAD mode
- Manual mode
Create a new Python file named vad_dash.py and copy the following code into the file:
Run
vad_dash.py
vad_dash.py
Copy
# Dependencies: dashscope >= 1.23.9, pyaudio
import os
import base64
import time
import pyaudio
from dashscope.audio.qwen_omni import MultiModality, AudioFormat,OmniRealtimeCallback,OmniRealtimeConversation
import dashscope
# Configuration parameters: URL, API key, voice, model, model role
url = 'wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime'
# Configure the API key. If you have not set an environment variable, replace the following line with dashscope.api_key = "sk-xxx"
dashscope.api_key = os.getenv('DASHSCOPE_API_KEY')
# Specify the voice
voice = 'Tina'
# Specify the model
model = 'qwen3.5-omni-plus-realtime'
# Specify the model role
instructions = "You are Xiaoyun, a personal assistant. Please answer the user's questions in a humorous and witty way."
class SimpleCallback(OmniRealtimeCallback):
def __init__(self, pya):
self.pya = pya
self.out = None
def on_open(self):
# Initialize the audio output stream
self.out = self.pya.open(
format=pyaudio.paInt16,
channels=1,
rate=24000,
output=True
)
def on_event(self, response):
if response['type'] == 'response.audio.delta':
# Play the audio
self.out.write(base64.b64decode(response['delta']))
elif response['type'] == 'conversation.item.input_audio_transcription.delta':
# Streaming preview: text is the confirmed prefix, stash is the unconfirmed suffix.
preview = response.get('text', '') + response.get('stash', '')
print(f"\r[User] {preview}", end='', flush=True)
elif response['type'] == 'conversation.item.input_audio_transcription.completed':
# Print the transcribed text
print(f"\r[User] {response['transcript']}")
elif response['type'] == 'response.audio_transcript.done':
# Print the assistant's reply text
print(f"[LLM] {response['transcript']}")
# 1. Initialize the audio device
pya = pyaudio.PyAudio()
# 2. Create the callback function and session
callback = SimpleCallback(pya)
conv = OmniRealtimeConversation(model=model, callback=callback, url=url)
# 3. Establish the connection and configure the session
conv.connect()
conv.update_session(output_modalities=[MultiModality.AUDIO, MultiModality.TEXT], voice=voice, instructions=instructions)
# 4. Initialize the audio input stream
mic = pya.open(format=pyaudio.paInt16, channels=1, rate=16000, input=True)
# 5. Main loop to process audio input
print("Conversation started. Speak into the microphone (Ctrl+C to exit)...")
try:
while True:
audio_data = mic.read(3200, exception_on_overflow=False)
conv.append_audio(base64.b64encode(audio_data).decode())
time.sleep(0.01)
except KeyboardInterrupt:
# Clean up resources
conv.close()
mic.close()
callback.out.close()
pya.terminate()
print("\nConversation ended")
vad_dash.py to have a real-time conversation with Qwen-Omni-Realtime through your microphone. The system detects the start and end of your speech and automatically sends it to the server without manual intervention.Create a new Python file named
Run
manual_dash.py and copy the following code into the file:manual_dash.py
manual_dash.py
Copy
# Dependencies: dashscope >= 1.23.9, pyaudio.
import os
import base64
import sys
import threading
import pyaudio
from dashscope.audio.qwen_omni import *
import dashscope
# If you have not set an environment variable, replace the following line with your API key: dashscope.api_key = "sk-xxx"
dashscope.api_key = os.getenv('DASHSCOPE_API_KEY')
voice = 'Tina'
class MyCallback(OmniRealtimeCallback):
"""Minimal callback: Initializes the speaker upon connection and plays the returned audio directly in the event."""
def __init__(self, ctx):
super().__init__()
self.ctx = ctx
def on_open(self) -> None:
# Initialize PyAudio and the speaker (24k/mono/16bit) after connection is established.
print('connection opened')
try:
self.ctx['pya'] = pyaudio.PyAudio()
self.ctx['out'] = self.ctx['pya'].open(
format=pyaudio.paInt16,
channels=1,
rate=24000,
output=True
)
print('audio output initialized')
except Exception as e:
print('[Error] audio init failed: {}'.format(e))
def on_close(self, close_status_code, close_msg) -> None:
print('connection closed with code: {}, msg: {}'.format(close_status_code, close_msg))
sys.exit(0)
def on_event(self, response: str) -> None:
try:
t = response['type']
handlers = {
'session.created': lambda r: print('start session: {}'.format(r['session']['id'])),
'conversation.item.input_audio_transcription.delta': self._transcription_delta,
'conversation.item.input_audio_transcription.completed': self._transcription_completed,
'response.audio_transcript.delta': lambda r: print('llm text: {}'.format(r['delta'])),
'response.audio.delta': self._play_audio,
'response.done': self._response_done,
}
h = handlers.get(t)
if h:
h(response)
except Exception as e:
print('[Error] {}'.format(e))
def _transcription_delta(self, response):
# Streaming preview: text is the confirmed prefix, stash is the unconfirmed suffix.
preview = response.get('text', '') + response.get('stash', '')
print(f"\r[User] {preview}", end='', flush=True)
def _transcription_completed(self, response):
print()
self.ctx['transcription_done'].set()
def _play_audio(self, response):
# Directly decode base64 and write to the output stream for playback.
if self.ctx['out'] is None:
return
try:
data = base64.b64decode(response['delta'])
self.ctx['out'].write(data)
except Exception as e:
print('[Error] audio playback failed: {}'.format(e))
def _response_done(self, response):
# Mark the current conversation turn as complete for the main loop to wait.
if self.ctx['conv'] is not None:
print('[Metric] response: {}, first text delay: {}, first audio delay: {}'.format(
self.ctx['conv'].get_last_response_id(),
self.ctx['conv'].get_last_first_text_delay(),
self.ctx['conv'].get_last_first_audio_delay(),
))
if self.ctx['resp_done'] is not None:
self.ctx['resp_done'].set()
def shutdown_ctx(ctx):
"""Safely release audio and PyAudio resources."""
try:
if ctx['out'] is not None:
ctx['out'].close()
ctx['out'] = None
except Exception:
pass
try:
if ctx['pya'] is not None:
ctx['pya'].terminate()
ctx['pya'] = None
except Exception:
pass
def stream_record_and_send(pya_inst: pyaudio.PyAudio, conversation, sample_rate=16000, chunk_size=3200):
"""Press Enter to stop recording; each chunk is streamed to the server immediately."""
stop_evt = threading.Event()
stream = pya_inst.open(
format=pyaudio.paInt16,
channels=1,
rate=sample_rate,
input=True,
frames_per_buffer=chunk_size
)
def _reader():
while not stop_evt.is_set():
try:
data = stream.read(chunk_size, exception_on_overflow=False)
conversation.append_audio(base64.b64encode(data).decode())
except Exception:
break
t = threading.Thread(target=_reader, daemon=True)
t.start()
input() # User presses Enter again to stop recording.
stop_evt.set()
t.join(timeout=1.0)
stream.close()
if __name__ == '__main__':
print('Initializing ...')
# Runtime context: Stores audio and session handles.
ctx = {'pya': None, 'out': None, 'conv': None, 'resp_done': threading.Event(), 'transcription_done': threading.Event()}
callback = MyCallback(ctx)
conversation = OmniRealtimeConversation(
model='qwen3.5-omni-plus-realtime',
callback=callback,
url="wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime",
)
try:
conversation.connect()
except Exception as e:
print('[Error] connect failed: {}'.format(e))
sys.exit(1)
ctx['conv'] = conversation
# Session configuration: Enable text and audio output (disable server-side VAD, switch to manual recording).
conversation.update_session(
output_modalities=[MultiModality.AUDIO, MultiModality.TEXT],
voice=voice,
enable_input_audio_transcription=True,
# The model for transcribing input audio.
input_audio_transcription_model='qwen3-asr-flash-realtime',
enable_turn_detection=False,
instructions="You are Xiaoyun, a personal assistant. Please answer the user's questions accurately and friendly, always responding with a helpful attitude."
)
try:
turn = 1
while True:
print(f"\n--- Turn {turn} ---")
print("Press Enter to start recording (enter q to exit)...")
user_input = input()
if user_input.strip().lower() in ['q', 'quit']:
print("User requested to exit...")
break
print("Recording... Press Enter again to stop.")
if ctx['pya'] is None:
ctx['pya'] = pyaudio.PyAudio()
stream_record_and_send(ctx['pya'], conversation)
ctx['resp_done'].clear()
ctx['transcription_done'].clear()
conversation.commit()
ctx['transcription_done'].wait(timeout=10)
print("Waiting for model response...")
conversation.create_response()
ctx['resp_done'].wait()
turn += 1
except KeyboardInterrupt:
print("\nProgram interrupted by user.")
finally:
shutdown_ctx(ctx)
print("Program exited.")
manual_dash.py. Press Enter to start speaking. Press Enter again to receive the model's audio response.- Audio conversation: Captures real-time audio from a microphone with VAD mode (
enable_turn_detection= True) and supports voice interruption. - Audio and video conversation: Captures real-time audio and video from a microphone and camera with VAD mode and supports voice interruption.
- Local call: Uses local audio and images as input with Manual mode (
enable_turn_detection= False).
Use headphones for audio playback to prevent echoes from triggering voice interruption.
DashScope Java SDK version 2.20.9 or later is required. See Install DashScope SDK for Maven and Gradle setup.Choose an interaction mode
You can run the
Run the Sample code on GitHubDownload complete sample code from GitHub, which includes:
- VAD mode (automatically detects the start and end of speech) The Realtime API automatically determines when the user starts and stops speaking and responds accordingly.
- Manual mode (press to talk, release to send) The client controls the start and end of speech. After the user finishes speaking, the client must actively send a message to the server.
- VAD mode
- Manual mode
OmniServerVad.java
OmniServerVad.java
Copy
import com.alibaba.dashscope.audio.omni.*;
import com.alibaba.dashscope.exception.NoApiKeyException;
import com.google.gson.JsonObject;
import javax.sound.sampled.*;
import java.nio.ByteBuffer;
import java.util.Arrays;
import java.util.Base64;
import java.util.Map;
import java.util.Queue;
import java.util.concurrent.ConcurrentLinkedQueue;
import java.util.concurrent.atomic.AtomicBoolean;
public class OmniServerVad {
static class SequentialAudioPlayer {
private final SourceDataLine line;
private final Queue<byte[]> audioQueue = new ConcurrentLinkedQueue<>();
private final Thread playerThread;
private final AtomicBoolean shouldStop = new AtomicBoolean(false);
public SequentialAudioPlayer() throws LineUnavailableException {
AudioFormat format = new AudioFormat(24000, 16, 1, true, false);
line = AudioSystem.getSourceDataLine(format);
line.open(format);
line.start();
playerThread = new Thread(() -> {
while (!shouldStop.get()) {
byte[] audio = audioQueue.poll();
if (audio != null) {
line.write(audio, 0, audio.length);
} else {
try { Thread.sleep(10); } catch (InterruptedException ignored) {}
}
}
}, "AudioPlayer");
playerThread.start();
}
public void play(String base64Audio) {
try {
byte[] audio = Base64.getDecoder().decode(base64Audio);
audioQueue.add(audio);
} catch (Exception e) {
System.err.println("Audio decoding failed: " + e.getMessage());
}
}
public void cancel() {
audioQueue.clear();
line.flush();
}
public void close() {
shouldStop.set(true);
try { playerThread.join(1000); } catch (InterruptedException ignored) {}
line.drain();
line.close();
}
}
public static void main(String[] args) {
try {
SequentialAudioPlayer player = new SequentialAudioPlayer();
AtomicBoolean userIsSpeaking = new AtomicBoolean(false);
AtomicBoolean shouldStop = new AtomicBoolean(false);
OmniRealtimeParam param = OmniRealtimeParam.builder()
.model("qwen3.5-omni-plus-realtime")
.apikey(System.getenv("DASHSCOPE_API_KEY"))
.url("wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime")
.build();
OmniRealtimeConversation conversation = new OmniRealtimeConversation(param, new OmniRealtimeCallback() {
@Override public void onOpen() {
System.out.println("Connection established");
}
@Override public void onClose(int code, String reason) {
System.out.println("Connection closed (" + code + "): " + reason);
shouldStop.set(true);
}
@Override public void onEvent(JsonObject event) {
handleEvent(event, player, userIsSpeaking);
}
});
conversation.connect();
conversation.updateSession(OmniRealtimeConfig.builder()
.modalities(Arrays.asList(OmniRealtimeModality.AUDIO, OmniRealtimeModality.TEXT))
.voice("Tina")
.enableTurnDetection(true)
.enableInputAudioTranscription(true)
.parameters(Map.of("instructions",
"You are an AI customer service agent for a five-star hotel. Answer customer inquiries about room types, facilities, prices, and booking policies accurately and friendly. Always respond with a professional and helpful attitude. Do not provide unconfirmed information or information beyond the scope of the hotel's services."))
.build()
);
System.out.println("Please start speaking (automatic detection of speech start/end, press Ctrl+C to exit)...");
AudioFormat format = new AudioFormat(16000, 16, 1, true, false);
TargetDataLine mic = AudioSystem.getTargetDataLine(format);
mic.open(format);
mic.start();
ByteBuffer buffer = ByteBuffer.allocate(3200);
while (!shouldStop.get()) {
int bytesRead = mic.read(buffer.array(), 0, buffer.capacity());
if (bytesRead > 0) {
try {
conversation.appendAudio(Base64.getEncoder().encodeToString(buffer.array()));
} catch (Exception e) {
if (e.getMessage() != null && e.getMessage().contains("closed")) {
System.out.println("Conversation closed. Stopping recording.");
break;
}
}
}
Thread.sleep(20);
}
conversation.close(1000, "Normal exit");
player.close();
mic.close();
System.out.println("\nProgram exited.");
} catch (NoApiKeyException e) {
System.err.println("API KEY not found: Please set the DASHSCOPE_API_KEY environment variable.");
System.exit(1);
} catch (Exception e) {
e.printStackTrace();
}
}
private static void handleEvent(JsonObject event, SequentialAudioPlayer player, AtomicBoolean userIsSpeaking) {
String type = event.get("type").getAsString();
switch (type) {
case "input_audio_buffer.speech_started":
System.out.println("\n[User started speaking]");
player.cancel();
userIsSpeaking.set(true);
break;
case "input_audio_buffer.speech_stopped":
System.out.println("[User stopped speaking]");
userIsSpeaking.set(false);
break;
case "response.audio.delta":
if (!userIsSpeaking.get()) {
player.play(event.get("delta").getAsString());
}
break;
case "conversation.item.input_audio_transcription.delta":
// Streaming preview: text is confirmed prefix, stash is unconfirmed suffix
String preview = event.get("text").getAsString() + event.get("stash").getAsString();
System.out.print("\rUser: " + preview);
break;
case "conversation.item.input_audio_transcription.completed":
System.out.println("\rUser: " + event.get("transcript").getAsString());
break;
case "response.audio_transcript.delta":
System.out.print(event.get("delta").getAsString());
break;
case "response.done":
System.out.println("Response complete");
break;
}
}
}
OmniServerVad.main() method to have a real-time conversation with the Realtime model using your microphone. The system detects the start of your audio and automatically sends it to the server, so you do not need to send it manually.OmniWithoutServerVad.java
OmniWithoutServerVad.java
Copy
// DashScope Java SDK version 2.20.9 or later
import com.alibaba.dashscope.audio.omni.*;
import com.alibaba.dashscope.exception.NoApiKeyException;
import com.google.gson.JsonObject;
import javax.sound.sampled.*;
import java.io.IOException;
import java.util.Arrays;
import java.util.Base64;
import java.util.HashMap;
import java.util.Queue;
import java.util.concurrent.ConcurrentLinkedQueue;
import java.util.concurrent.CountDownLatch;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.atomic.AtomicBoolean;
import java.util.concurrent.atomic.AtomicReference;
public class Main {
// RealtimePcmPlayer class definition starts
public static class RealtimePcmPlayer {
private int sampleRate;
private SourceDataLine line;
private AudioFormat audioFormat;
private Thread decoderThread;
private Thread playerThread;
private AtomicBoolean stopped = new AtomicBoolean(false);
private Queue<String> b64AudioBuffer = new ConcurrentLinkedQueue<>();
private Queue<byte[]> RawAudioBuffer = new ConcurrentLinkedQueue<>();
// The constructor initializes the audio format and audio line.
public RealtimePcmPlayer(int sampleRate) throws LineUnavailableException {
this.sampleRate = sampleRate;
this.audioFormat = new AudioFormat(this.sampleRate, 16, 1, true, false);
DataLine.Info info = new DataLine.Info(SourceDataLine.class, audioFormat);
line = (SourceDataLine) AudioSystem.getLine(info);
line.open(audioFormat);
line.start();
decoderThread = new Thread(new Runnable() {
@Override
public void run() {
while (!stopped.get()) {
String b64Audio = b64AudioBuffer.poll();
if (b64Audio != null) {
byte[] rawAudio = Base64.getDecoder().decode(b64Audio);
RawAudioBuffer.add(rawAudio);
} else {
try {
Thread.sleep(100);
} catch (InterruptedException e) {
throw new RuntimeException(e);
}
}
}
}
});
playerThread = new Thread(new Runnable() {
@Override
public void run() {
while (!stopped.get()) {
byte[] rawAudio = RawAudioBuffer.poll();
if (rawAudio != null) {
try {
playChunk(rawAudio);
} catch (IOException e) {
throw new RuntimeException(e);
} catch (InterruptedException e) {
throw new RuntimeException(e);
}
} else {
try {
Thread.sleep(100);
} catch (InterruptedException e) {
throw new RuntimeException(e);
}
}
}
}
});
decoderThread.start();
playerThread.start();
}
// Play an audio chunk and block until playback is complete.
private void playChunk(byte[] chunk) throws IOException, InterruptedException {
if (chunk == null || chunk.length == 0) return;
int bytesWritten = 0;
while (bytesWritten < chunk.length) {
bytesWritten += line.write(chunk, bytesWritten, chunk.length - bytesWritten);
}
int audioLength = chunk.length / (this.sampleRate*2/1000);
// Wait for the audio in the buffer to finish playing.
Thread.sleep(audioLength - 10);
}
public void write(String b64Audio) {
b64AudioBuffer.add(b64Audio);
}
public void cancel() {
b64AudioBuffer.clear();
RawAudioBuffer.clear();
}
public void waitForComplete() throws InterruptedException {
while (!b64AudioBuffer.isEmpty() || !RawAudioBuffer.isEmpty()) {
Thread.sleep(100);
}
line.drain();
}
public void shutdown() throws InterruptedException {
stopped.set(true);
decoderThread.join();
playerThread.join();
if (line != null && line.isRunning()) {
line.drain();
line.close();
}
}
} // RealtimePcmPlayer class definition ends
// Add a recording method
private static void recordAndSend(TargetDataLine line, OmniRealtimeConversation conversation) {
byte[] buffer = new byte[3200];
AtomicBoolean stopRecording = new AtomicBoolean(false);
// Start a thread to listen for the Enter key.
Thread enterKeyListener = new Thread(() -> {
try {
System.in.read();
stopRecording.set(true);
} catch (IOException e) {
e.printStackTrace();
}
});
enterKeyListener.start();
// Recording loop
while (!stopRecording.get()) {
int count = line.read(buffer, 0, buffer.length);
if (count > 0) {
byte[] chunk = Arrays.copyOf(buffer, count);
conversation.appendAudio(Base64.getEncoder().encodeToString(chunk));
}
}
}
public static void main(String[] args) throws InterruptedException, LineUnavailableException {
OmniRealtimeParam param = OmniRealtimeParam.builder()
.model("qwen3.5-omni-plus-realtime")
// If you have not configured an environment variable, replace the following line with your API key: .apikey("sk-xxx")
.apikey(System.getenv("DASHSCOPE_API_KEY"))
.url("wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime")
.build();
AtomicReference<CountDownLatch> responseDoneLatch = new AtomicReference<>(null);
responseDoneLatch.set(new CountDownLatch(1));
AtomicReference<CountDownLatch> transcriptionDoneLatch = new AtomicReference<>(null);
transcriptionDoneLatch.set(new CountDownLatch(1));
RealtimePcmPlayer audioPlayer = new RealtimePcmPlayer(24000);
final AtomicReference<OmniRealtimeConversation> conversationRef = new AtomicReference<>(null);
OmniRealtimeConversation conversation = new OmniRealtimeConversation(param, new OmniRealtimeCallback() {
@Override
public void onOpen() {
System.out.println("connection opened");
}
@Override
public void onEvent(JsonObject message) {
String type = message.get("type").getAsString();
switch(type) {
case "session.created":
System.out.println("start session: " + message.get("session").getAsJsonObject().get("id").getAsString());
break;
case "conversation.item.input_audio_transcription.delta":
String preview = message.get("text").getAsString() + message.get("stash").getAsString();
System.out.print("\rquestion: " + preview);
break;
case "conversation.item.input_audio_transcription.completed":
System.out.println("\rquestion: " + message.get("transcript").getAsString());
transcriptionDoneLatch.get().countDown();
break;
case "response.audio_transcript.delta":
System.out.println("got llm response delta: " + message.get("delta").getAsString());
break;
case "response.audio.delta":
String recvAudioB64 = message.get("delta").getAsString();
audioPlayer.write(recvAudioB64);
break;
case "response.done":
System.out.println("======RESPONSE DONE======");
if (conversationRef.get() != null) {
System.out.println("[Metric] response: " + conversationRef.get().getResponseId() +
", first text delay: " + conversationRef.get().getFirstTextDelay() +
" ms, first audio delay: " + conversationRef.get().getFirstAudioDelay() + " ms");
}
responseDoneLatch.get().countDown();
break;
default:
break;
}
}
@Override
public void onClose(int code, String reason) {
System.out.println("connection closed code: " + code + ", reason: " + reason);
}
});
conversationRef.set(conversation);
try {
conversation.connect();
} catch (NoApiKeyException e) {
throw new RuntimeException(e);
}
OmniRealtimeConfig config = OmniRealtimeConfig.builder()
.modalities(Arrays.asList(OmniRealtimeModality.AUDIO, OmniRealtimeModality.TEXT))
.voice("Tina")
.enableTurnDetection(false)
// Set the model role.
.parameters(new HashMap<String, Object>() {{
put("instructions","You are Xiaoyun, a personal assistant. Please answer the user's questions accurately and friendly, always responding with a helpful attitude.");
}})
.build();
conversation.updateSession(config);
// Add microphone recording functionality.
AudioFormat format = new AudioFormat(16000, 16, 1, true, false);
DataLine.Info info = new DataLine.Info(TargetDataLine.class, format);
if (!AudioSystem.isLineSupported(info)) {
System.out.println("Line not supported");
return;
}
TargetDataLine line = null;
try {
line = (TargetDataLine) AudioSystem.getLine(info);
line.open(format);
line.start();
while (true) {
System.out.println("Press Enter to start recording...");
System.in.read();
System.out.println("Recording started. Please speak... Press Enter again to stop recording and send.");
recordAndSend(line, conversation);
conversation.commit();
transcriptionDoneLatch.get().await(10, TimeUnit.SECONDS);
System.out.println("Waiting for model response...");
conversation.createResponse(null, null);
// Reset the latch for the next wait.
responseDoneLatch.set(new CountDownLatch(1));
transcriptionDoneLatch.set(new CountDownLatch(1));
}
} catch (LineUnavailableException | IOException | InterruptedException e) {
e.printStackTrace();
} finally {
if (line != null) {
line.stop();
line.close();
}
}
}
}
OmniWithoutServerVad.main() method. Press Enter to start recording. During recording, press Enter again to stop recording and send the audio. Then receive and play the model's response.- Audio conversation: Captures real-time audio from a microphone with VAD mode (
enableTurnDetection= true) and supports voice interruption. - Audio and video conversation: Captures real-time audio and video from a microphone and camera with VAD mode and supports voice interruption.
- Local call: Uses local audio and images as input with Manual mode (
enableTurnDetection= false).
Set the
enableTurnDetection parameter to true for VAD mode or false for Manual mode. Use headphones for audio playback to prevent echoes from triggering voice interruption.1
Prepare the runtime environment
Your Python version must be 3.10 or later.First, install PyAudio based on your operating system.After installation completes, install the WebSocket-related dependencies using pip:
- macOS
- Debian/Ubuntu
- CentOS
- Windows
Copy
brew install portaudio && pip install pyaudio
Copy
sudo apt-get install python3-pyaudio
or
pip install pyaudio
We recommend running
pip install pyaudio. If the installation fails, first install the portaudio dependency for your operating system.Copy
sudo yum install -y portaudio portaudio-devel && pip install pyaudio
Copy
pip install pyaudio
Copy
pip install websockets==15.0.1
2
Create the client
Create a new Python file named
omni_realtime_client.py in your local directory and copy the following code into the file:omni_realtime_client.py
omni_realtime_client.py
Copy
import asyncio
import websockets
import json
import base64
import time
from typing import Optional, Callable, List, Dict, Any
from enum import Enum
class TurnDetectionMode(Enum):
SERVER_VAD = "server_vad"
SEMANTIC_VAD = "semantic_vad" # Recommended for models like qwen3.5-omni-realtime
MANUAL = "manual"
class OmniRealtimeClient:
def __init__(
self,
base_url,
api_key: str,
model: str = "",
voice: str = "Tina",
instructions: str = "You are a helpful assistant.",
turn_detection_mode: TurnDetectionMode = TurnDetectionMode.SERVER_VAD,
on_text_delta: Optional[Callable[[str], None]] = None,
on_audio_delta: Optional[Callable[[bytes], None]] = None,
on_input_transcript: Optional[Callable[[str], None]] = None,
on_output_transcript: Optional[Callable[[str], None]] = None,
extra_event_handlers: Optional[Dict[str, Callable[[Dict[str, Any]], None]]] = None
):
self.base_url = base_url
self.api_key = api_key
self.model = model
self.voice = voice
self.instructions = instructions
self.ws = None
self.on_text_delta = on_text_delta
self.on_audio_delta = on_audio_delta
self.on_input_transcript = on_input_transcript
self.on_output_transcript = on_output_transcript
self.turn_detection_mode = turn_detection_mode
self.extra_event_handlers = extra_event_handlers or {}
# Current response status
self._current_response_id = None
self._current_item_id = None
self._is_responding = False
# Input/output transcript printing status
self._print_input_transcript = True
self._output_transcript_buffer = ""
async def connect(self) -> None:
"""Establish a WebSocket connection with the Realtime API."""
url = f"{self.base_url}?model={self.model}"
headers = {
"Authorization": f"Bearer {self.api_key}"
}
self.ws = await websockets.connect(url, additional_headers=headers)
# Session configuration
session_config = {
"modalities": ["text", "audio"],
"voice": self.voice,
"instructions": self.instructions,
"input_audio_format": "pcm",
"output_audio_format": "pcm",
"input_audio_transcription": {
"model": "qwen3-asr-flash-realtime"
}
}
if self.turn_detection_mode == TurnDetectionMode.MANUAL:
session_config['turn_detection'] = None
await self.update_session(session_config)
elif self.turn_detection_mode == TurnDetectionMode.SERVER_VAD:
session_config['turn_detection'] = {
"type": "server_vad",
"threshold": 0.1,
"prefix_padding_ms": 500,
"silence_duration_ms": 900
}
await self.update_session(session_config)
elif self.turn_detection_mode == TurnDetectionMode.SEMANTIC_VAD:
session_config['turn_detection'] = {
"type": "semantic_vad",
"threshold": 0.1,
"prefix_padding_ms": 500,
"silence_duration_ms": 900
}
await self.update_session(session_config)
else:
raise ValueError(f"Invalid turn detection mode: {self.turn_detection_mode}")
async def send_event(self, event) -> None:
event['event_id'] = "event_" + str(int(time.time() * 1000))
await self.ws.send(json.dumps(event))
async def update_session(self, config: Dict[str, Any]) -> None:
"""Update the session configuration."""
event = {
"type": "session.update",
"session": config
}
await self.send_event(event)
async def stream_audio(self, audio_chunk: bytes) -> None:
"""Stream raw audio data to the API."""
# Only 16-bit, 16 kHz, mono PCM is supported.
audio_b64 = base64.b64encode(audio_chunk).decode()
append_event = {
"type": "input_audio_buffer.append",
"audio": audio_b64
}
await self.send_event(append_event)
async def commit_audio_buffer(self) -> None:
"""Commit the audio buffer to trigger processing."""
event = {
"type": "input_audio_buffer.commit"
}
await self.send_event(event)
async def append_image(self, image_chunk: bytes) -> None:
"""Append image data to the image buffer.
Image data can come from local files or a real-time video stream.
Note:
- The image format must be JPG or JPEG. A resolution of 480p or 720p is recommended. The maximum supported resolution is 1080p.
- A single image after Base64 encoding must not exceed 256 KB. We recommend keeping the raw image size below 190 KB before encoding.
- Encode the image data to Base64 before sending.
- We recommend sending images to the server at a rate of no more than 1 frame per second.
- You must send audio data at least once before sending image data.
"""
image_b64 = base64.b64encode(image_chunk).decode()
event = {
"type": "input_image_buffer.append",
"image": image_b64
}
await self.send_event(event)
async def create_response(self) -> None:
"""Request the API to generate a response (only needs to be called in manual mode)."""
event = {
"type": "response.create"
}
await self.send_event(event)
async def cancel_response(self) -> None:
"""Cancel the current response."""
event = {
"type": "response.cancel"
}
await self.send_event(event)
async def handle_interruption(self):
"""Handle user interruption of the current response."""
if not self._is_responding:
return
# 1. Cancel the current response.
if self._current_response_id:
await self.cancel_response()
self._is_responding = False
self._current_response_id = None
self._current_item_id = None
async def handle_messages(self) -> None:
try:
async for message in self.ws:
event = json.loads(message)
event_type = event.get("type")
if event_type == "error":
print(" Error: ", event['error'])
continue
elif event_type == "response.created":
self._current_response_id = event.get("response", {}).get("id")
self._is_responding = True
elif event_type == "response.output_item.added":
self._current_item_id = event.get("item", {}).get("id")
elif event_type == "response.done":
self._is_responding = False
self._current_response_id = None
self._current_item_id = None
elif event_type == "input_audio_buffer.speech_started":
print("Speech start detected")
if self._is_responding:
print("Handling interruption")
await self.handle_interruption()
elif event_type == "input_audio_buffer.speech_stopped":
print("Speech end detected")
elif event_type == "response.text.delta":
if self.on_text_delta:
self.on_text_delta(event["delta"])
elif event_type == "response.audio.delta":
if self.on_audio_delta:
audio_bytes = base64.b64decode(event["delta"])
self.on_audio_delta(audio_bytes)
elif event_type == "conversation.item.input_audio_transcription.delta":
# Streaming preview: text is confirmed prefix, stash is unconfirmed suffix
preview = event.get("text", "") + event.get("stash", "")
print(f"\rUser: {preview}", end="", flush=True)
elif event_type == "conversation.item.input_audio_transcription.completed":
transcript = event.get("transcript", "")
print(f"\rUser: {transcript}")
if self.on_input_transcript:
await asyncio.to_thread(self.on_input_transcript, transcript)
self._print_input_transcript = True
elif event_type == "response.audio_transcript.delta":
if self.on_output_transcript:
delta = event.get("delta", "")
if not self._print_input_transcript:
self._output_transcript_buffer += delta
else:
if self._output_transcript_buffer:
await asyncio.to_thread(self.on_output_transcript, self._output_transcript_buffer)
self._output_transcript_buffer = ""
await asyncio.to_thread(self.on_output_transcript, delta)
elif event_type == "response.audio_transcript.done":
print(f"LLM: {event.get('transcript', '')}")
self._print_input_transcript = False
elif event_type in self.extra_event_handlers:
self.extra_event_handlers[event_type](event)
except websockets.exceptions.ConnectionClosed:
print(" Connection closed")
except Exception as e:
print(" Error in message handling: ", str(e))
async def close(self) -> None:
"""Close the WebSocket connection."""
if self.ws:
await self.ws.close()
3
Choose an interaction mode
- VAD mode (automatically detects the start and end of speech) The Realtime API automatically determines when the user starts and stops speaking and responds accordingly.
- Manual mode (press to talk, release to send) The client controls the start and end of speech. After the user finishes speaking, the client must actively send a message to the server.
- VAD mode
- Manual mode
In the same directory as
Run
omni_realtime_client.py, create another Python file named vad_mode.py and copy the following code into the file:vad_mode.py
vad_mode.py
Copy
# -- coding: utf-8 --
import os, asyncio, pyaudio, queue, threading
from omni_realtime_client import OmniRealtimeClient, TurnDetectionMode
# Audio player class (handles interruptions)
class AudioPlayer:
def __init__(self, pyaudio_instance, rate=24000):
self.stream = pyaudio_instance.open(format=pyaudio.paInt16, channels=1, rate=rate, output=True)
self.queue = queue.Queue()
self.stop_evt = threading.Event()
self.interrupt_evt = threading.Event()
threading.Thread(target=self._run, daemon=True).start()
def _run(self):
while not self.stop_evt.is_set():
try:
data = self.queue.get(timeout=0.5)
if data is None: break
if not self.interrupt_evt.is_set(): self.stream.write(data)
self.queue.task_done()
except queue.Empty: continue
def add_audio(self, data): self.queue.put(data)
def handle_interrupt(self): self.interrupt_evt.set(); self.queue.queue.clear()
def stop(self): self.stop_evt.set(); self.queue.put(None); self.stream.stop_stream(); self.stream.close()
# Record from microphone and send
async def record_and_send(client):
p = pyaudio.PyAudio()
stream = p.open(format=pyaudio.paInt16, channels=1, rate=16000, input=True, frames_per_buffer=3200)
print("Recording started. Please speak...")
try:
while True:
audio_data = stream.read(3200)
await client.stream_audio(audio_data)
await asyncio.sleep(0.02)
finally:
stream.stop_stream(); stream.close(); p.terminate()
async def main():
p = pyaudio.PyAudio()
player = AudioPlayer(pyaudio_instance=p)
client = OmniRealtimeClient(
base_url="wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime",
api_key=os.environ.get("DASHSCOPE_API_KEY"),
model="qwen3.5-omni-plus-realtime",
voice="Tina",
instructions="You are Xiaoyun, a witty and humorous assistant.",
# We recommend that you set this parameter to SEMANTIC_VAD when you use the qwen3.5-omni-realtime model.
turn_detection_mode=TurnDetectionMode.SEMANTIC_VAD,
on_text_delta=lambda t: print(f"\nAssistant: {t}", end="", flush=True),
on_audio_delta=player.add_audio,
)
await client.connect()
print("Connection successful. Starting real-time conversation...")
# Run concurrently
await asyncio.gather(client.handle_messages(), record_and_send(client))
if __name__ == "__main__":
try:
asyncio.run(main())
except KeyboardInterrupt:
print("\nProgram exited.")
vad_mode.py to have a real-time conversation with Qwen-Omni-Realtime through your microphone. The system detects the start and end of your speech and automatically sends it to the server without manual intervention.In the same directory as
Run
omni_realtime_client.py, create another Python file named manual_mode.py and copy the following code into the file:manual_mode.py
manual_mode.py
Copy
# -- coding: utf-8 --
import os
import asyncio
import time
import threading
import queue
import pyaudio
from omni_realtime_client import OmniRealtimeClient, TurnDetectionMode
class AudioPlayer:
"""Real-time audio player class"""
def __init__(self, sample_rate=24000, channels=1, sample_width=2):
self.sample_rate = sample_rate
self.channels = channels
self.sample_width = sample_width # 2 bytes for 16-bit
self.audio_queue = queue.Queue()
self.is_playing = False
self.play_thread = None
self.pyaudio_instance = None
self.stream = None
self._lock = threading.Lock() # Add a lock for synchronized access
self._last_data_time = time.time() # Record the time the last data was received
self._response_done = False # Add a flag to indicate response completion
self._waiting_for_response = False # Flag to indicate if waiting for a server response
# Record the time the last data was written to the audio stream and the duration of the most recent audio chunk for more accurate playback end detection
self._last_play_time = time.time()
self._last_chunk_duration = 0.0
def start(self):
"""Start the audio player"""
with self._lock:
if self.is_playing:
return
self.is_playing = True
try:
self.pyaudio_instance = pyaudio.PyAudio()
# Create an audio output stream
self.stream = self.pyaudio_instance.open(
format=pyaudio.paInt16, # 16-bit
channels=self.channels,
rate=self.sample_rate,
output=True,
frames_per_buffer=1024
)
# Start the playback thread
self.play_thread = threading.Thread(target=self._play_audio)
self.play_thread.daemon = True
self.play_thread.start()
print("Audio player started")
except Exception as e:
print(f"Failed to start audio player: {e}")
self._cleanup_resources()
raise
def stop(self):
"""Stop the audio player"""
with self._lock:
if not self.is_playing:
return
self.is_playing = False
# Clear the queue
while not self.audio_queue.empty():
try:
self.audio_queue.get_nowait()
except queue.Empty:
break
# Wait for the playback thread to finish (wait outside the lock to avoid deadlock)
if self.play_thread and self.play_thread.is_alive():
self.play_thread.join(timeout=2.0)
# Acquire the lock again to clean up resources
with self._lock:
self._cleanup_resources()
print("Audio player stopped")
def _cleanup_resources(self):
"""Clean up audio resources (must be called within the lock)"""
try:
# Close the audio stream
if self.stream:
if not self.stream.is_stopped():
self.stream.stop_stream()
self.stream.close()
self.stream = None
except Exception as e:
print(f"Error closing audio stream: {e}")
try:
if self.pyaudio_instance:
self.pyaudio_instance.terminate()
self.pyaudio_instance = None
except Exception as e:
print(f"Error terminating PyAudio: {e}")
def add_audio_data(self, audio_data):
"""Add audio data to the playback queue"""
if self.is_playing and audio_data:
self.audio_queue.put(audio_data)
with self._lock:
self._last_data_time = time.time() # Update the time the last data was received
self._waiting_for_response = False # Data received, no longer waiting
def stop_receiving_data(self):
"""Mark that no more new audio data will be received"""
with self._lock:
self._response_done = True
self._waiting_for_response = False # Response ended, no longer waiting
def prepare_for_next_turn(self):
"""Reset the player state for the next conversation turn."""
with self._lock:
self._response_done = False
self._last_data_time = time.time()
self._last_play_time = time.time()
self._last_chunk_duration = 0.0
self._waiting_for_response = True # Start waiting for the next response
# Clear any remaining audio data from the previous turn
while not self.audio_queue.empty():
try:
self.audio_queue.get_nowait()
except queue.Empty:
break
def is_finished_playing(self):
"""Check if all audio data has been played"""
with self._lock:
queue_size = self.audio_queue.qsize()
time_since_last_data = time.time() - self._last_data_time
time_since_last_play = time.time() - self._last_play_time
# ---------------------- Smart end detection ----------------------
# 1. Preferred: If the server has marked completion and the playback queue is empty.
# Wait for the most recent audio chunk to finish playing (chunk duration + 0.1s tolerance).
if self._response_done and queue_size == 0:
min_wait = max(self._last_chunk_duration + 0.1, 0.5) # Wait at least 0.5s
if time_since_last_play >= min_wait:
return True
# 2. Fallback: If no new data has been received for a long time and the playback queue is empty.
# This logic serves as a safeguard if the server does not explicitly send `response.done`.
if not self._waiting_for_response and queue_size == 0 and time_since_last_data > 1.0:
print("\n(No new audio received for a while, assuming playback is finished)")
return True
return False
def _play_audio(self):
"""Worker thread for playing audio data"""
while True:
# Check if it should stop
with self._lock:
if not self.is_playing:
break
stream_ref = self.stream # Get a reference to the stream
try:
# Get audio data from the queue, with a timeout of 0.1 seconds
audio_data = self.audio_queue.get(timeout=0.1)
# Check the status and stream validity again
with self._lock:
if self.is_playing and stream_ref and not stream_ref.is_stopped():
try:
# Play the audio data
stream_ref.write(audio_data)
# Update the latest playback information
self._last_play_time = time.time()
self._last_chunk_duration = len(audio_data) / (
self.channels * self.sample_width) / self.sample_rate
except Exception as e:
print(f"Error writing to audio stream: {e}")
break
# Mark this data block as processed
self.audio_queue.task_done()
except queue.Empty:
# Continue waiting if the queue is empty
continue
except Exception as e:
print(f"Error playing audio: {e}")
break
class MicrophoneRecorder:
"""Real-time microphone recorder"""
def __init__(self, sample_rate=16000, channels=1, chunk_size=3200):
self.sample_rate = sample_rate
self.channels = channels
self.chunk_size = chunk_size
self.pyaudio_instance = None
self.stream = None
self.frames = []
self._is_recording = False
self._record_thread = None
def _recording_thread(self):
"""Recording worker thread"""
# Continuously read data from the audio stream while _is_recording is True
while self._is_recording:
try:
# Use exception_on_overflow=False to avoid crashing due to buffer overflow
data = self.stream.read(self.chunk_size, exception_on_overflow=False)
self.frames.append(data)
except (IOError, OSError) as e:
# Reading from the stream might raise an error when it's closed
print(f"Error reading from recording stream, it might be closed: {e}")
break
def start(self):
"""Start recording"""
if self._is_recording:
print("Recording is already in progress.")
return
self.frames = []
self._is_recording = True
try:
self.pyaudio_instance = pyaudio.PyAudio()
self.stream = self.pyaudio_instance.open(
format=pyaudio.paInt16,
channels=self.channels,
rate=self.sample_rate,
input=True,
frames_per_buffer=self.chunk_size
)
self._record_thread = threading.Thread(target=self._recording_thread)
self._record_thread.daemon = True
self._record_thread.start()
print("Microphone recording started...")
except Exception as e:
print(f"Failed to start microphone: {e}")
self._is_recording = False
self._cleanup()
raise
def stop(self):
"""Stop recording and return the audio data"""
if not self._is_recording:
return None
self._is_recording = False
# Wait for the recording thread to exit safely
if self._record_thread:
self._record_thread.join(timeout=1.0)
self._cleanup()
print("Microphone recording stopped.")
return b''.join(self.frames)
def _cleanup(self):
"""Safely clean up PyAudio resources"""
if self.stream:
try:
if self.stream.is_active():
self.stream.stop_stream()
self.stream.close()
except Exception as e:
print(f"Error closing audio stream: {e}")
if self.pyaudio_instance:
try:
self.pyaudio_instance.terminate()
except Exception as e:
print(f"Error terminating PyAudio instance: {e}")
self.stream = None
self.pyaudio_instance = None
async def interactive_test():
"""
Interactive test script: Allows for multi-turn conversations, with audio and images sent in each turn.
"""
# ------------------- 1. Initialization and connection (one-time) -------------------
api_key = os.environ.get("DASHSCOPE_API_KEY")
if not api_key:
print("Please set the DASHSCOPE_API_KEY environment variable.")
return
print("--- Real-time Multimodal Audio/Video Chat Client ---")
print("Initializing audio player and client...")
audio_player = AudioPlayer()
audio_player.start()
def on_audio_received(audio_data):
audio_player.add_audio_data(audio_data)
def on_response_done(event):
print("\n(Received response end marker)")
audio_player.stop_receiving_data()
realtime_client = OmniRealtimeClient(
base_url="wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime",
api_key=api_key,
model="qwen3.5-omni-plus-realtime",
voice="Tina",
instructions="You are Xiaoyun, a personal assistant. Please answer the user's questions accurately and friendly, always responding with a helpful attitude.", # Set the model role
on_text_delta=lambda text: print(f"Assistant reply: {text}", end="", flush=True),
on_audio_delta=on_audio_received,
turn_detection_mode=TurnDetectionMode.MANUAL,
extra_event_handlers={"response.done": on_response_done}
)
message_handler_task = None
try:
await realtime_client.connect()
print("Connected to the server. Enter 'q' or 'quit' to exit at any time.")
message_handler_task = asyncio.create_task(realtime_client.handle_messages())
await asyncio.sleep(0.5)
turn_counter = 1
# ------------------- 2. Multi-turn conversation loop -------------------
while True:
print(f"\n--- Turn {turn_counter} ---")
audio_player.prepare_for_next_turn()
recorded_audio = None
image_paths = []
# --- Get user input: Record from microphone ---
loop = asyncio.get_event_loop()
recorder = MicrophoneRecorder(sample_rate=16000) # 16k sample rate is recommended for speech recognition
print("Ready to record. Press Enter to start recording (or enter 'q' to exit)...")
user_input = await loop.run_in_executor(None, input)
if user_input.strip().lower() in ['q', 'quit']:
print("User requested to exit...")
return
try:
recorder.start()
except Exception:
print("Could not start recording. Please check your microphone permissions and device. Skipping this turn.")
continue
print("Recording... Press Enter again to stop.")
await loop.run_in_executor(None, input)
recorded_audio = recorder.stop()
if not recorded_audio or len(recorded_audio) == 0:
print("No valid audio was recorded. Please start this turn again.")
continue
# --- 3. Send data and get response ---
print("\n--- Input Confirmation ---")
print(f"Audio to process: 1 (from microphone), Images: {len(image_paths)}")
print("------------------")
# 3.1 Send the recorded audio
try:
print(f"Sending microphone recording ({len(recorded_audio)} bytes)")
await realtime_client.stream_audio(recorded_audio)
await asyncio.sleep(0.1)
except Exception as e:
print(f"Failed to send microphone recording: {e}")
continue
# 3.3 Submit and wait for response
print("Submitting all inputs, requesting server response...")
await realtime_client.commit_audio_buffer()
await realtime_client.create_response()
print("Waiting for and playing server response audio...")
start_time = time.time()
max_wait_time = 60
while not audio_player.is_finished_playing():
if time.time() - start_time > max_wait_time:
print(f"\nWait timed out ({max_wait_time} seconds). Moving to the next turn.")
break
await asyncio.sleep(0.2)
print("\nAudio playback for this turn is complete!")
turn_counter += 1
except (asyncio.CancelledError, KeyboardInterrupt):
print("\nProgram was interrupted.")
except Exception as e:
print(f"An unhandled error occurred: {e}")
finally:
# ------------------- 4. Clean up resources -------------------
print("\nClosing connection and cleaning up resources...")
if message_handler_task and not message_handler_task.done():
message_handler_task.cancel()
if 'realtime_client' in locals() and realtime_client.ws and not realtime_client.ws.close:
await realtime_client.close()
print("Connection closed.")
audio_player.stop()
print("Program exited.")
if __name__ == "__main__":
try:
asyncio.run(interactive_test())
except KeyboardInterrupt:
print("\nProgram was forcibly exited by the user.")
manual_mode.py. Press Enter to start speaking. Press Enter again to receive the model's audio response.Interaction flow
- VAD mode
- Manual mode
session.turn_detection.type in the session.update event to "server_vad" or "semantic_vad" to enable VAD mode, which is suitable for voice call scenarios.The interaction flow is as follows:- The server detects the start of speech and sends the input_audio_buffer.speech_started event.
- The client can send input_audio_buffer.append and input_image_buffer.append events at any time to append audio and images to the buffer.
Before sending an input_image_buffer.append event, send at least one input_audio_buffer.append event.
- The server detects the end of speech and sends the input_audio_buffer.speech_stopped event.
- The server sends the input_audio_buffer.committed event to commit the audio buffer.
- The server sends a conversation.item.created event. This event contains the user message item created from the buffer.
| Lifecycle | Client events | Server-side events |
|---|---|---|
| Session initialization | session.update - Session configuration | session.created - Session created. session.updated - Session configuration updated |
| User audio input | input_audio_buffer.append - Add audio to the buffer. input_image_buffer.append - Add an image to the buffer | input_audio_buffer.speech_started - Speech start detected. input_audio_buffer.speech_stopped - Speech end detected. input_audio_buffer.committed - Server received the submitted audio |
| Server audio output | None | response.created - Server starts generating a response. response.output_item.added - New output content during response. conversation.item.created - Conversation item created. response.content_part.added - New output content added to the assistant message. response.audio_transcript.delta - Incrementally generated transcribed text. response.audio.delta - Incrementally generated audio from the model. response.audio_transcript.done - Text transcription complete. response.audio.done - Audio generation complete. response.content_part.done - Streaming output of text or audio content for the assistant message is complete. response.output_item.done - Streaming of the entire output item for the assistant message is complete. response.done - Response complete |
session.turn_detection in the session.update event to null to enable Manual mode. In this mode, the client requests a server response by explicitly sending the input_audio_buffer.commit and response.create events. This mode is suitable for push-to-talk scenarios, such as sending voice messages in chat applications.The interaction flow is as follows:- The client can send input_audio_buffer.append and input_image_buffer.append events at any time to append audio and images to the buffer.
Before sending an input_image_buffer.append event, send at least one input_audio_buffer.append event.
- The client sends the input_audio_buffer.commit event to commit the audio and image buffers. This informs the server that all user input, including audio and images, for the current turn has been sent.
- The server responds with an input_audio_buffer.committed event.
- The client sends a response.create event and waits for the model's output from the server.
- The server responds with a conversation.item.created event.
| Lifecycle | Client events | Server-side events |
|---|---|---|
| Session initialization | session.update - Session configuration | session.created - Session created. session.updated - Session configuration updated |
| User audio input | input_audio_buffer.append - Add audio to the buffer. input_image_buffer.append - Add an image to the buffer. input_audio_buffer.commit - Submit audio and images to the server. response.create - Create a model response | input_audio_buffer.committed - Server received the submitted audio |
| Server audio output | input_audio_buffer.clear - Clear the audio from the buffer | response.created - Server starts generating a response. response.output_item.added - New output content during response. conversation.item.created - Conversation item created. response.content_part.added - New output content added to the assistant message item. response.audio_transcript.delta - Incrementally generated transcribed text. response.audio.delta - Incrementally generated audio from the model. response.audio_transcript.done - Text transcription complete. response.audio.done - Audio generation complete. response.content_part.done - Streaming output of text or audio content for the assistant message is complete. response.output_item.done - Streaming of the entire output item for the assistant message is complete. response.done - Response complete |
Web search
Web search lets the model reply using real-time retrieved data for scenarios that need up-to-date information, such as stock prices or weather forecasts. The model autonomously decides whether to search.Only the
Qwen3.5-Omni-Realtime model supports web search. It is disabled by default. Enable it using the session.update event.For billing details, see the agent policy in the Billing details.How to enable
In thesession.update event, add these parameters:
enable_search: Set totrueto enable web search.search_options.enable_source: Set totrueto return a list of search result sources.
Response format
After you enable web search, theresponse.done event includes a new plugins field in the usage object. This field records search usage metrics:
Copy
{
"usage": {
"total_tokens": 2937,
"input_tokens": 2554,
"output_tokens": 383,
"input_tokens_details": {
"text_tokens": 2512,
"audio_tokens": 42
},
"output_tokens_details": {
"text_tokens": 90,
"audio_tokens": 293
},
"plugins": {
"search": {
"count": 1,
"strategy": "agent"
}
}
}
}
Code examples
The following examples show how to enable web search.- DashScope Python SDK
- DashScope Java SDK
- WebSocket (Python)
In the
update_session call, pass the enable_search and search_options parameters:Copy
import os
import base64
import time
import json
import pyaudio
from dashscope.audio.qwen_omni import MultiModality, AudioFormat, OmniRealtimeCallback, OmniRealtimeConversation
import dashscope
dashscope.api_key = os.getenv('DASHSCOPE_API_KEY')
url = 'wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime'
model = 'qwen3.5-omni-plus-realtime'
voice = 'Tina'
class SearchCallback(OmniRealtimeCallback):
def __init__(self, pya):
self.pya = pya
self.out = None
def on_open(self):
self.out = self.pya.open(format=pyaudio.paInt16, channels=1, rate=24000, output=True)
def on_event(self, response):
if response['type'] == 'response.audio.delta':
self.out.write(base64.b64decode(response['delta']))
elif response['type'] == 'conversation.item.input_audio_transcription.delta':
# Streaming preview: text is the confirmed prefix, stash is the unconfirmed suffix.
preview = response.get('text', '') + response.get('stash', '')
print(f"\r[User] {preview}", end='', flush=True)
elif response['type'] == 'conversation.item.input_audio_transcription.completed':
print(f"\r[User] {response['transcript']}")
elif response['type'] == 'response.audio_transcript.done':
print(f"[LLM] {response['transcript']}")
elif response['type'] == 'response.done':
usage = response.get('response', {}).get('usage', {})
plugins = usage.get('plugins', {})
if plugins.get('search'):
print(f"[Search] count={plugins['search']['count']}, strategy={plugins['search']['strategy']}")
pya = pyaudio.PyAudio()
callback = SearchCallback(pya)
conv = OmniRealtimeConversation(model=model, callback=callback, url=url)
conv.connect()
conv.update_session(
output_modalities=[MultiModality.AUDIO, MultiModality.TEXT],
voice=voice,
instructions="You are Xiao Yun, a personal assistant",
enable_search=True,
search_options={'enable_source': True}
)
mic = pya.open(format=pyaudio.paInt16, channels=1, rate=16000, input=True)
print("Web search is enabled. Speak into the microphone (press Ctrl+C to exit)...")
try:
while True:
audio_data = mic.read(3200, exception_on_overflow=False)
conv.append_audio(base64.b64encode(audio_data).decode())
time.sleep(0.01)
except KeyboardInterrupt:
conv.close()
mic.close()
callback.out.close()
pya.terminate()
print("\nConversation ended")
In
updateSession, pass web search settings through the parameters map:Copy
import com.alibaba.dashscope.audio.omni.*;
import com.alibaba.dashscope.exception.NoApiKeyException;
import com.google.gson.JsonObject;
import javax.sound.sampled.*;
import java.nio.ByteBuffer;
import java.util.*;
import java.util.concurrent.ConcurrentLinkedQueue;
import java.util.concurrent.atomic.AtomicBoolean;
public class OmniSearch {
static class SequentialAudioPlayer {
private final SourceDataLine line;
private final Queue<byte[]> audioQueue = new ConcurrentLinkedQueue<>();
private final Thread playerThread;
private final AtomicBoolean shouldStop = new AtomicBoolean(false);
public SequentialAudioPlayer() throws LineUnavailableException {
AudioFormat format = new AudioFormat(24000, 16, 1, true, false);
line = AudioSystem.getSourceDataLine(format);
line.open(format);
line.start();
playerThread = new Thread(() -> {
while (!shouldStop.get()) {
byte[] audio = audioQueue.poll();
if (audio != null) {
line.write(audio, 0, audio.length);
} else {
try { Thread.sleep(10); } catch (InterruptedException ignored) {}
}
}
}, "AudioPlayer");
playerThread.start();
}
public void play(String base64Audio) {
audioQueue.add(Base64.getDecoder().decode(base64Audio));
}
public void close() {
shouldStop.set(true);
try { playerThread.join(1000); } catch (InterruptedException ignored) {}
line.drain();
line.close();
}
}
public static void main(String[] args) {
try {
SequentialAudioPlayer player = new SequentialAudioPlayer();
AtomicBoolean shouldStop = new AtomicBoolean(false);
OmniRealtimeParam param = OmniRealtimeParam.builder()
.model("qwen3.5-omni-plus-realtime")
.apikey(System.getenv("DASHSCOPE_API_KEY"))
.url("wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime")
.build();
OmniRealtimeConversation conversation = new OmniRealtimeConversation(param, new OmniRealtimeCallback() {
@Override public void onOpen() {
System.out.println("Connection established");
}
@Override public void onClose(int code, String reason) {
System.out.println("Connection closed");
shouldStop.set(true);
}
@Override public void onEvent(JsonObject event) {
String type = event.get("type").getAsString();
if ("response.audio.delta".equals(type)) {
player.play(event.get("delta").getAsString());
} else if ("response.audio_transcript.done".equals(type)) {
System.out.println("[LLM] " + event.get("transcript").getAsString());
} else if ("response.done".equals(type)) {
JsonObject response = event.getAsJsonObject("response");
if (response != null && response.has("usage")) {
JsonObject usage = response.getAsJsonObject("usage");
if (usage.has("plugins")) {
JsonObject plugins = usage.getAsJsonObject("plugins");
if (plugins.has("search")) {
JsonObject search = plugins.getAsJsonObject("search");
System.out.println("[Search] count=" + search.get("count").getAsInt()
+ ", strategy=" + search.get("strategy").getAsString());
}
}
}
}
}
});
conversation.connect();
conversation.updateSession(OmniRealtimeConfig.builder()
.modalities(Arrays.asList(OmniRealtimeModality.AUDIO, OmniRealtimeModality.TEXT))
.voice("Tina")
.enableTurnDetection(true)
.enableInputAudioTranscription(true)
.parameters(Map.of(
"instructions", "You are Xiao Yun, a personal assistant",
"enable_search", true,
"search_options", Map.of("enable_source", true)
))
.build()
);
System.out.println("Web search is enabled. Start speaking (press Ctrl+C to exit)...");
AudioFormat format = new AudioFormat(16000, 16, 1, true, false);
TargetDataLine mic = AudioSystem.getTargetDataLine(format);
mic.open(format);
mic.start();
ByteBuffer buffer = ByteBuffer.allocate(3200);
while (!shouldStop.get()) {
int bytesRead = mic.read(buffer.array(), 0, buffer.capacity());
if (bytesRead > 0) {
conversation.appendAudio(Base64.getEncoder().encodeToString(buffer.array()));
}
Thread.sleep(20);
}
conversation.close(1000, "Normal end");
player.close();
mic.close();
} catch (NoApiKeyException e) {
System.err.println("API key not found: Set the DASHSCOPE_API_KEY environment variable");
} catch (Exception e) {
e.printStackTrace();
}
}
}
In the JSON payload for
session.update, add the enable_search and search_options fields:Copy
import json
import os
import websocket
import base64
import pyaudio
import threading
API_KEY = os.getenv("DASHSCOPE_API_KEY")
API_URL = "wss://dashscope-intl.aliyuncs.com/api-ws/v1/realtime?model=qwen3.5-omni-plus-realtime"
pya = pyaudio.PyAudio()
out_stream = pya.open(format=pyaudio.paInt16, channels=1, rate=24000, output=True)
def on_open(ws):
ws.send(json.dumps({
"type": "session.update",
"session": {
"modalities": ["text", "audio"],
"voice": "Tina",
"instructions": "You are Xiao Yun, a personal assistant",
"input_audio_format": "pcm",
"output_audio_format": "pcm",
"enable_search": True,
"search_options": {
"enable_source": True
}
}
}))
print("Web search is enabled. Speak into the microphone...")
def send_audio():
mic = pya.open(format=pyaudio.paInt16, channels=1, rate=16000, input=True)
try:
while True:
audio = mic.read(3200, exception_on_overflow=False)
ws.send(json.dumps({
"type": "input_audio_buffer.append",
"audio": base64.b64encode(audio).decode()
}))
except Exception:
mic.close()
threading.Thread(target=send_audio, daemon=True).start()
def on_message(ws, message):
event = json.loads(message)
if event["type"] == "response.audio.delta":
out_stream.write(base64.b64decode(event["delta"]))
elif event["type"] == "response.audio_transcript.done":
print(f"[LLM] {event['transcript']}")
elif event["type"] == "response.done":
usage = event.get("response", {}).get("usage", {})
plugins = usage.get("plugins", {})
if plugins.get("search"):
print(f"[Search] count={plugins['search']['count']}, strategy={plugins['search']['strategy']}")
def on_error(ws, error):
print(f"Error: {error}")
headers = ["Authorization: Bearer " + API_KEY]
ws = websocket.WebSocketApp(API_URL, header=headers, on_open=on_open, on_message=on_message, on_error=on_error)
ws.run_forever()
API reference
Billing and rate limits
Billing rules
Qwen-Omni-Realtime is billed based on the number of tokens used for different input modalities, such as audio and images. For more information about billing, see Pricing.In a multi-turn real-time conversation, each time the model generates a response, it processes all historical conversation content within the context window — including audio, images, and text from previous turns — together with the new input of the current turn as input tokens. As a result, input tokens accumulate with each turn rather than being counted only for the new input in the current turn.For example, if a 10-second audio input converts to 70 tokens (Qwen3.5-Omni-Realtime), and that audio is still within the context window at turn 3, it will still be counted toward the input tokens for turn 3. The actual billed input tokens = tokens from all historical turns within the context window + tokens from the new input in the current turn.
Rules for converting audio and images to tokens
Rules for converting audio and images to tokens
- Audio
- Image
- Qwen3.5-Omni-Realtime: Input audio total tokens = Audio duration (in seconds) x 7; Output audio total tokens = Audio duration (in seconds) x 12.5
- Qwen3-Omni-Flash-Realtime: Total tokens for both input and output audio = Audio duration (in seconds) x 12.5
- Qwen-Omni-Turbo-Realtime: Total tokens for both input and output audio = Audio duration (in seconds) x 25
Qwen3.5-Omni-Plus-Realtimemodel: 1 token per32x32pixelsQwen3-Omni-Flash-Realtimemodel: 1 token per32x32pixelsQwen-Omni-Turbo-Realtimemodel: 1 token per28x28pixels
Copy
# Install the Pillow library using the following command: pip install Pillow
from PIL import Image
import math
# For the Qwen-Omni-Turbo-Realtime model, the zoom factor is 28.
# factor = 28
# For the Qwen3-Omni-Flash-Realtime and Qwen3.5-Omni-Realtime models, the zoom factor is 32.
factor = 32
def token_calculate(image_path='', duration=10):
"""
:param image_path: The path of the image.
:param duration: The duration of the session connection.
:return: The number of tokens for the image.
"""
if len(image_path) > 0:
# Open the specified PNG image file.
image = Image.open(image_path)
# Get the original dimensions of the image.
height = image.height
width = image.width
print(f"Image dimensions before scaling: height={height}, width={width}")
# Adjust the height to be an integer multiple of the factor.
h_bar = round(height / factor) * factor
# Adjust the width to be an integer multiple of the factor.
w_bar = round(width / factor) * factor
# Lower limit for image tokens: 4 tokens.
min_pixels = factor * factor * 4
# Upper limit for image tokens: 1,280 tokens.
max_pixels = 1280 * factor * factor
# Scale the image to ensure the total number of pixels is within the range [min_pixels, max_pixels].
if h_bar * w_bar > max_pixels:
# Calculate the scaling factor beta so that the total number of pixels of the scaled image does not exceed max_pixels.
beta = math.sqrt((height * width) / max_pixels)
# Recalculate the adjusted height to ensure it is an integer multiple of the factor.
h_bar = math.floor(height / beta / factor) * factor
# Recalculate the adjusted width to ensure it is an integer multiple of the factor.
w_bar = math.floor(width / beta / factor) * factor
elif h_bar * w_bar < min_pixels:
# Calculate the scaling factor beta so that the total number of pixels of the scaled image is not less than min_pixels.
beta = math.sqrt(min_pixels / (height * width))
# Recalculate the adjusted height to ensure it is an integer multiple of the factor.
h_bar = math.ceil(height * beta / factor) * factor
# Recalculate the adjusted width to ensure it is an integer multiple of the factor.
w_bar = math.ceil(width * beta / factor) * factor
print(f"Image dimensions after scaling: height={h_bar}, width={w_bar}")
# Calculate the number of tokens for the image: total pixels divided by (factor x factor).
token = int((h_bar * w_bar) / (factor * factor))
print(f"Number of tokens after scaling: {token}")
total_token = token * math.ceil(duration / 2)
print(f"Total number of tokens: {total_token}")
return total_token
else:
print("Error: image_path is empty. Cannot calculate tokens")
return 0
if __name__ == "__main__":
total_token = token_calculate(image_path="xxx/test.jpg", duration=10)
Rate limits
For more information about model rate limit rules, see Rate limits.Error codes
If a call fails, see Error codes.Voice list
For the voices each omni-modal model supports, including samples andvoice parameter values, see Omni-modal voice list.