`AI_ENTITY_EXTRACT` 関数は、テキスト、イメージ、またはビデオから、人物、組織、地名、日付、金額などの固有表現を抽出します。この関数は、ニュースのインデックス作成、メディアアセットのアノテーション、イベントの取得に使用します。
コマンドフォーマット
REST API または Python コレクション関数を介して AI_ENTITY_EXTRACT を呼び出すことができます。
REST API
POST /v2/vectordb/ai/entity_extract
Content-Type: application/json
{
"model_name": "<model name>",
"texts": ["<text or media URL>"],
"params": {"entity_types": ["PERSON", "ORGANIZATION"]}
}Python
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
schema.add_field("id", DataType.INT64, is_primary=True)
schema.add_field("content", DataType.VARCHAR, max_length=4096)
schema.add_field("entities", DataType.JSON)
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=2)
schema.add_function(
Function(
name="extract_entities",
function_type=texttransform_function_type(),
input_field_names=["content"],
output_field_names=["entities"],
params={
"provider": "aliyun_milvus",
"model_name": "<model name>",
"task": "ai_entity_extract",
"entity_types": "PERSON,ORGANIZATION,LOCATION,DATE,PRODUCT",
"temperature": "0",
},
)
)パラメーター
| パラメーター | 説明 |
model_name | 必須。モデル名。テキスト抽出には、テキストモデル (例: qwen3.7-max) を使用します。イメージおよびビデオには、設定済みのマルチモーダルモデル (例: qwen3.7-plus) を使用します。 |
texts | REST API で必須。認識対象のテキスト、またはモデルがアクセス可能なイメージおよびビデオの URL。 |
entity_types | オプション。ターゲットエンティティタイプ。配列またはカンマ区切りの文字列で、1〜30 個のエントリを指定します。省略した場合、この関数はデフォルトで人物、組織、地名、プロダクト、イベント、日付、時間、金額、パーセンテージ、URL、メールアドレス、電話番号、IP アドレスを認識します。 |
prompt | オプション。追加の認識ルール。最大 5,000 文字。${...} はサポートされていません。(推奨) メディアシナリオでは、「外観に基づいて ID、ブランド、または場所を推測しないこと」などの指示を含めてください。 |
media_type | 入力がメディア URL であることを示すには、任意で image または video に設定します。 |
temperature/max_concurrency/timeout_sec | オプション。モデルの安定性、同時実行数、およびタイムアウトの設定。 |
プロバイダー/タスク | コレクション機能でのみ必須です。値は aliyun_milvus と ai_entity_extract に固定されています。 |
戻り値
各出力は、{"entities":[{"text":"entity text","type":"entity type"}]} 形式の JSON オブジェクト、または同じ構造に解析される JSON 文字列です。 entities の各アイテムには、空でない text と、リクエストの type が含まれます。該当する名前付きエンティティが認識されない場合、{"entities":[]} は有効な結果です。スキーマによって、有効な JSON が出力フィールドに書き込まれます。
例 1:ニュースのエンティティインデックスの構築 (テキスト)
あるコンテンツプラットフォームが、ニュース記事から明確に言及されている人物、組織、地名、日付、プロダクトを抽出し、検索フィルター条件として使用するシナリオです。この例では、JSON の構造とタイプを検証します。モデルが全く同じエンティティリストを返す必要はありません。
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<yourUsername>:<yourPassword>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
BODY=$(cat <<JSON
{
"model_name": "qwen3.7-max",
"texts": [
"2024年3月、田中太郎は Milvus の開発に携わるため、杭州の Alibaba に入社しました。"
],
"params": {
"entity_types": ["PERSON", "ORGANIZATION", "LOCATION", "DATE", "PRODUCT"],
"temperature": 0
}
}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/entity_extract" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fiPython
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus provides TEXTTRANSFORM as a managed extension (function type value 9);
# some pymilvus versions do not have this enum member built-in, while Function(...) validates via FunctionType(...).
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_entity_extract_text",
input_fields=[("content", DataType.VARCHAR, 4096)],
output_field=("entities", DataType.JSON, None),
function_name="extract_entities",
function_params={"provider": "aliyun_milvus", "model_name": "qwen3.7-max", "task": "ai_entity_extract", "entity_types": "PERSON,ORGANIZATION,LOCATION,DATE,PRODUCT", "temperature": "0"},
rows=[{"content": "2024年3月、田中太郎は Milvus の開発に携わるため、杭州の Alibaba に入社しました。", "dummy_vector": [0.1, 0.2]}],
)検証
entities の各アイテムには、空でない text フィールドと、リクエストされたセットの type 値が含まれています。以下は、実際のテストで得られたサンプル結果です。
{"entities":[{"text":"2024年3月","type":"DATE"},{"text":"張三","type":"PERSON"},{"text":"アリババ","type":"ORGANIZATION"},{"text":"杭州","type":"LOCATION"},{"text":"Milvus","type":"PRODUCT"}]}実際のエンティティの境界は、モデルとエンティティルールに依存します。
例 2:ファッションプロモーションイメージ内の固有表現のレビュー (イメージ)
メディアライブラリが、白黒のストライプのトップスを着用し、黒いバッグを持った人物のイメージを受け取るシナリオです。これらの目に見えるオブジェクトだけでは、確認可能な人物名、ブランド、プロダクト名、または地名とは断定できません。この例では、推測を明示的に禁止し、空のエンティティを有効な結果として受け入れます。
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<yourUsername>:<yourPassword>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
BODY=$(cat <<JSON
{"model_name":"qwen3.7-plus","texts":["https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/20260415/hynnff/wan-video-edit-clothes.webp"],"params":{"media_type":"image","entity_types":["PERSON","PRODUCT","LOCATION"],"temperature":0}}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/entity_extract" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fiPython
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus は、マネージド拡張機能 (関数タイプの値は 9) として TEXTTRANSFORM を提供します。
# 一部の pymilvus バージョンにはこの enum メンバーが組み込まれていませんが、Function(...) は FunctionType(...) を介して検証を行います。
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_entity_extract_image",
input_fields=[("image_url", DataType.VARCHAR, 4096)],
output_field=("entities", DataType.JSON, None),
function_name="extract_image_entities",
function_params={"provider": "aliyun_milvus", "model_name": "qwen3.7-plus", "task": "ai_entity_extract", "media_type": "image", "entity_types": "PERSON,PRODUCT,LOCATION", "temperature": "0"},
rows=[{"image_url": "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20260415/hynnff/wan-video-edit-clothes.webp", "dummy_vector": [0.1, 0.2]}],
)検証
出力は、各アイテムの type がリクエストされたセットに属する、有効なエンティティオブジェクトです。イメージに名前付きエンティティを確認するための十分な証拠が含まれていない場合、{"entities":[]} は有効な結果です。
例 3:人間のような馬のキャラクターのビデオ内の固有表現のレビュー (ビデオ)
ビデオアセットに、スーツを着た人間のような馬のキャラクターのクローズアップが映っているシナリオです。キャラクターの外見は、実在の人物の ID、作品のタイトル、ブランド、または地名を証明するものではありません。この例では、「明確な証拠がある場合のみ抽出する」というルールを適用するため、空のエンティティ配列は有効な結果となります。
REST API
#!/usr/bin/env bash
set -euo pipefail
MILVUS_REST_BASE_URL="http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_AUTH_TOKEN="<yourUsername>:<yourPassword>"
post_json() {
local path="$1"
local body="$2"
curl -X POST \
"$MILVUS_REST_BASE_URL$path" \
-H "Authorization: Bearer $MILVUS_AUTH_TOKEN" \
-H "Content-Type: application/json" \
-d "$body"
}
BODY=$(cat <<JSON
{"model_name":"qwen3.7-plus","texts":["https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/20260409/dozxak/Wan_Video_Edit_33_1.mp4"],"params":{"media_type":"video","entity_types":["PERSON","PRODUCT","LOCATION"],"temperature":0}}
JSON
)
RESPONSE_BODY="$(post_json "/v2/vectordb/ai/entity_extract" "$BODY")"
if command -v jq >/dev/null 2>&1; then
echo "$RESPONSE_BODY" | jq .
[ "$(echo "$RESPONSE_BODY" | jq -r '.code // -1')" = "0" ] || exit 1
else
echo "$RESPONSE_BODY"
fiPython
from __future__ import annotations
from typing import Any
from pymilvus import DataType, Function, FunctionType, MilvusClient
MILVUS_URI = "http://c-xxxx.milvus.aliyuncs.com:19530"
MILVUS_TOKEN = "<yourUsername>:<yourPassword>"
DUMMY_VECTOR_DIM = 2
TEXTTRANSFORM_FUNCTION_TYPE = 9
def texttransform_function_type() -> Any:
for type_name in ("TEXTTRANSFORM", "TEXT_TRANSFORM", "TextTransform"):
function_type = getattr(FunctionType, type_name, None)
if function_type is not None:
return function_type
# Alibaba Cloud Milvus provides TEXTTRANSFORM as a managed extension (function type value 9);
# some pymilvus versions do not have this enum member built-in, while Function(...) validates via FunctionType(...).
existing = getattr(FunctionType, "_value2member_map_", {}).get(TEXTTRANSFORM_FUNCTION_TYPE)
if existing is not None:
return existing
extension = int.__new__(FunctionType, TEXTTRANSFORM_FUNCTION_TYPE)
extension._name_ = "TEXTTRANSFORM"
extension._value_ = TEXTTRANSFORM_FUNCTION_TYPE
FunctionType._value2member_map_[TEXTTRANSFORM_FUNCTION_TYPE] = extension
FunctionType._member_map_["TEXTTRANSFORM"] = extension
return extension
def add_id(schema: Any) -> None:
schema.add_field("id", DataType.INT64, is_primary=True)
def add_dummy_vector(schema: Any) -> None:
schema.add_field("dummy_vector", DataType.FLOAT_VECTOR, dim=DUMMY_VECTOR_DIM)
def run_texttransform_example(*, client, collection_name, input_fields, output_field, function_name, function_params, rows) -> None:
if client.has_collection(collection_name):
client.drop_collection(collection_name)
schema = MilvusClient.create_schema(auto_id=True, enable_dynamic_field=False)
add_id(schema)
for name, data_type, max_length in input_fields:
field_params = {"max_length": max_length} if max_length is not None else {}
schema.add_field(name, data_type, **field_params)
output_name, output_data_type, output_max_length = output_field
output_params = {"max_length": output_max_length} if output_max_length is not None else {}
schema.add_field(output_name, output_data_type, **output_params)
add_dummy_vector(schema)
schema.add_function(
Function(
name=function_name,
function_type=texttransform_function_type(),
input_field_names=[name for name, _, _ in input_fields],
output_field_names=[output_name],
params=function_params,
)
)
index_params = client.prepare_index_params()
index_params.add_index(field_name="dummy_vector", index_type="AUTOINDEX", metric_type="COSINE")
client.create_collection(collection_name=collection_name, schema=schema, index_params=index_params)
client.insert(collection_name, rows)
client.flush(collection_name)
fields = [name for name, _, _ in input_fields] + [output_name]
for row in client.query(collection_name, filter="", output_fields=fields, limit=len(rows)):
print(row)
client = MilvusClient(uri=MILVUS_URI, token=MILVUS_TOKEN)
run_texttransform_example(
client=client,
collection_name="simple_ai_entity_extract_video",
input_fields=[("video_url", DataType.VARCHAR, 4096)],
output_field=("entities", DataType.JSON, None),
function_name="extract_video_entities",
function_params={"provider": "aliyun_milvus", "model_name": "qwen3.7-plus", "task": "ai_entity_extract", "media_type": "video", "entity_types": "PERSON,PRODUCT,LOCATION", "temperature": "0"},
rows=[{"video_url": "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/20260409/dozxak/Wan_Video_Edit_33_1.mp4", "dummy_vector": [0.1, 0.2]}],
)検証
出力構造は正しく、すべてのエンティティタイプはリクエストされたセットに属しています。確認可能な名前付きエンティティが存在しない場合、空の entities 配列は有効です。実際のテストでは {"entities":[]} が返されました。
キャラクターのタイプ、服装、または動きを記述するには、エンティティ抽出の代わりに、コンテンツの要約または分類機能を使用してください。