AI_EXTRACT 関数は、指定されたラベルに基づいて、テキスト、画像、または動画から構造化情報を抽出します。この関数を使用すると、非構造化コンテンツをフィルタリング可能な属性に変換でき、レビューのアーカイブ、契約要素の入力、製品属性の補完などのシナリオで活用できます。
コマンド構文
REST API
POST /v2/vectordb/ai/extract
Content-Type: application/json
{
"model_name": "<model-name>",
"texts": ["<text-or-media-URL>"],
"params": {"labels": ["<label>"]}
}Python
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
schema.add_field("id", DataType.INT64, is_primary=True)
schema.add_field("content", DataType.VARCHAR, max_length=4096)
schema.add_field("attributes", DataType.JSON)
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=2)
schema.add_function(
Function(
name="extract_attributes",
function_type=texttransform_function_type(),
input_field_names=["content"],
output_field_names=["attributes"],
params={
"provider": "aliyun_milvus",
"model_name": "<model-name>",
"task": "ai_extract",
"labels": "brand,model,color,price,satisfaction",
"temperature": "0",
"enable_thinking": "false",
},
)
)パラメーター
| パラメーター | 説明 |
model_name | 必須。モデル名。画像や動画の場合は、qwen3.7-plus などのマルチモーダルモデルを指定します。 |
texts | 必須 (REST のみ)。抽出元のテキスト、または画像か動画の URL。 |
labels | 必須。抽出するフィールド。JSON 文字列配列またはコンマ区切りの文字列を受け入れます。1〜20 個のラベルを指定できます。出力される JSON キーはラベルの term と一致します。 |
prompt | 任意。補足的な抽出ルール。最大 5000 文字。${...} はサポートされていません。 |
media_type | 任意。入力がメディア URL であることを示すために image または video に設定します。 |
temperature / enable_thinking / max_concurrency / timeout_sec | 任意。モデルの安定性と呼び出し制御のパラメーター。 |
provider / task | コレクション関数でのみ必須。固定値は aliyun_milvus と ai_extract です。 |
レスポンス
data.output.outputs 内の各項目は、リクエストされたすべてのラベルを含む JSON 文字列です。コンテンツ内で値が識別できないラベルは null に設定されます。コレクション関数として使用する場合、有効な JSON が自動的にターゲットフィールドに書き込まれます。
例 1:製品レビュー (テキスト) からフィルタリング可能な属性を抽出
フリーテキストの製品レビューを、統一されたフィルタリング可能な属性フィールドに変換します。
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<yourUsername>:<yourPassword>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
BODY=$(cat <<JSON
{
"model_name": "qwen3.7-max",
"texts": ["iPhone 15 Pro のナチュラルチタンを 7999 元で購入しました。とても満足しています。"],
"params": {"labels": "brand,model,color,price,satisfaction", "temperature": 0, "enable_thinking": false}
}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/extract" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fiPython
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus は、マネージド拡張機能 (関数タイプ値 9) として TEXTTRANSFORM を提供します。
# 一部の pymilvus バージョンにはこの enum メンバーが含まれていないため、Function(...) は FunctionType(...) を通じて検証します。
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_extract_text",
input_fields=[("content", DataType.VARCHAR, 4096)],
output_field=("attributes", DataType.JSON, None),
function_name="extract_attributes",
function_params={"provider": "aliyun_milvus", "model_name": "qwen3.7-max", "task": "ai_extract", "labels": "brand,model,color,price,satisfaction", "temperature": "0", "enable_thinking": "false"},
rows=[{"content": "iPhone 15 Pro のナチュラルチタンを 7999 元で購入しました。とても満足しています。", "dummy_vector": [0.1, 0.2]}],
)期待される結果: attributes フィールドには {"brand":"Apple","model":"iPhone 15 Pro","color":"natural titanium","price":"7999 yuan","satisfaction":"very satisfied"} が含まれます (キーは英語のラベルに従います)。
例 2:製品画像から属性を抽出
コンテンツレビュー中に、製品画像から主題、色、シーンのフィールドを補完します。
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<yourUsername>:<yourPassword>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
BODY=$(cat <<JSON
{"model_name":"qwen3.7-plus","texts":["https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/20260415/hynnff/wan-video-edit-clothes.webp"],"params":{"media_type":"image","labels":["subject","color","scene"],"temperature":0}}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/extract" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fiPython
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus は、マネージド拡張機能 (関数タイプ値 9) として TEXTTRANSFORM を提供します。
# 一部の pymilvus バージョンにはこの enum メンバーが含まれていないため、Function(...) は FunctionType(...) を通じて検証します。
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_extract_image",
input_fields=[("image_url", DataType.VARCHAR, 4096)],
output_field=("attributes", DataType.JSON, None),
function_name="extract_image_attributes",
function_params={"provider": "aliyun_milvus", "model_name": "qwen3.7-plus", "task": "ai_extract", "media_type": "image", "labels": "subject,color,scene", "temperature": "0"},
rows=[{"image_url": "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/20260415/hynnff/wan-video-edit-clothes.webp", "dummy_vector": [0.1, 0.2]}],
)期待される結果: attributes フィールドは、subject、color、scene を含む JSON オブジェクトになります。識別できない属性は null に設定されます。
例 3:マーケティング動画から情報を抽出
動画アセットから主題、アクション、シーンの属性をインデックスフィールドに書き込み、検索フィルタリングに利用します。
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<お使いのユーザー名>:<お使いのパスワード>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
BODY=$(cat <<JSON
{"model_name":"qwen3.7-plus","texts":["https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20260409/dozxak/Wan_Video_Edit_33_1.mp4"],"params":{"media_type":"video","labels":["主題","アクション","シーン"],"temperature":0}}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/extract" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fiPython
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus は、マネージド拡張機能 (関数タイプ値 9) として TEXTTRANSFORM を提供します。
# 一部の pymilvus バージョンにはこの enum メンバーが含まれていないため、Function(...) は FunctionType(...) を通じて検証します。
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_extract_video",
input_fields=[("video_url", DataType.VARCHAR, 4096)],
output_field=("attributes", DataType.JSON, None),
function_name="extract_video_attributes",
function_params={"provider": "aliyun_milvus", "model_name": "qwen3.7-plus", "task": "ai_extract", "media_type": "video", "labels": "subject,action,scene", "temperature": "0"},
rows=[{"video_url": "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/20260409/dozxak/Wan_Video_Edit_33_1.mp4", "dummy_vector": [0.1, 0.2]}],
)期待される結果: attributes フィールドは、subject、action、scene を含む JSON オブジェクトになります。これらの属性を動画検索のフィルター条件として使用できます。