The AI_AUDIO_TRANSCRIBE function transcribes audio to text. Use this function for scenarios such as customer service call recording indexing, meeting minutes generation, voice message processing, and podcast content archiving.
Before transcribing your own recordings, confirm that you have fulfilled the recording notification obligations and obtained the necessary authorization. We recommend that you use short-lived media URLs with least privilege, and limit the storage duration of both the original audio and the transcription text according to your customer service data retention policy.
Command format
REST API
REST API
{
"model_name": "qwen3-asr-flash",
"texts": ["<audio_url>"],
"params": {
"language": "zh",
"enable_itn": true,
"max_concurrency": 1
}
}
Python
Python
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
schema.add_field("id", DataType.INT64, is_primary=True)
schema.add_field("audio_url", DataType.VARCHAR, max_length=4096)
schema.add_field("transcript", DataType.VARCHAR, max_length=4096)
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=2)
schema.add_function(
Function(
name="transcribe_audio",
function_type=texttransform_function_type(),
input_field_names=["audio_url"],
output_field_names=["transcript"],
params={
"provider": "aliyun_milvus",
"model_name": "qwen3-asr-flash",
"task": "ai_audio_transcribe",
"language": "zh",
"enable_itn": "true",
},
)
)
Parameters
| Parameter | Description |
model_name |
Required. Use qwen3-asr-flash. |
texts |
Required for REST. Array of audio URLs. Cannot contain empty strings. Results correspond to the input order. |
language |
Optional. Source language code, for example zh, en. |
enable_itn |
Optional. Boolean. When enabled, normalizes spoken numbers and similar expressions to written form. |
timeout_sec / max_concurrency |
Optional. Controls call timeout and concurrent processing count for multiple audio files, respectively. |
media_type |
Can be omitted or set to audio. Other values cause an error. |
| Others | prompt, temperature, and other custom model parameters are not supported. Whether the audio URL is accessible, and the supported formats and duration, are subject to the model service limitations. |
Return values
data.output.outputs returns the transcription text corresponding to each audio input in order. usage.audio_tokens indicates audio token usage, and usage.seconds indicates the processed audio duration.
{
"code": 0,
"data": {
"output": {"outputs": ["<transcript>"]},
"usage": {"audio_tokens": 256, "total_tokens": 256, "seconds": 45}
}
}
Example: Transcribe a customer service recording
The following example converts a hotline welcome message to text for version archiving and manual review of whether service hours, recording notifications, and other content are clearly expressed. Fields that do not appear in the transcription should be marked as "unconfirmed" and cannot be supplemented by the model.
The default public welcome.mp3 is a Chinese welcome message; the example uses language=zh. When replacing AIFUNC_AUDIO_URL, also update AIFUNC_AUDIO_LANGUAGE to the actual source language. The cURL example requires jq.
REST API
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<yourUsername>:<yourPassword>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
MODEL_NAME="qwen3-asr-flash"
AUDIO_URL="${AIFUNC_AUDIO_URL:-https://dashscope.oss-cn-beijing.aliyuncs.com/audios/welcome.mp3}"
BODY=$(cat <<JSON
{
"model_name": "$MODEL_NAME",
"texts": ["$AUDIO_URL"],
"params": {"language": "zh", "enable_itn": true, "max_concurrency": 1}
}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/audio_transcribe" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fi
Python
Python
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus provides TEXTTRANSFORM as a hosted extension (function type value 9);
# some pymilvus versions do not yet have this enum member built in, while Function(...) validates via FunctionType(...).
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
MODEL_NAME = "qwen3-asr-flash"
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_audio_transcribe_schema",
input_fields=[("audio_url", DataType.VARCHAR, 4096)],
output_field=("transcript", DataType.VARCHAR, 4096),
function_name="transcribe_audio",
function_params={"provider": "aliyun_milvus", "model_name": MODEL_NAME, "task": "ai_audio_transcribe", "language": "zh", "enable_itn": "true"},
rows=[{"audio_url": "https://dashscope.oss-cn-beijing.aliyuncs.com/audios/welcome.mp3", "dummy_vector": [0.1, 0.2]}],
)
Expected result: data.output.outputs[0] is a non-empty welcome message transcription of no more than 10,000 characters, which can be written to the transcript field of the hotline version archive (actual test return, rendered in English: Welcome to Alibaba Cloud.; the API returns the transcription in the source language of the audio).
{"code": 0, "data": {"output": {"outputs": ["Welcome to Alibaba Cloud."]}}}
For the Python scenario, you can name the input field audio_url and the output field transcript. The Function performs transcription automatically on insert.