30 lines
1.4 KiB
Python
30 lines
1.4 KiB
Python
from pydantic import BaseModel, Field
|
|
|
|
# Default instruction sent to the vision model when the agent does not provide a
|
|
# more specific question. Kept in Chinese to match the product's primary locale.
|
|
DEFAULT_VISION_PROMPT = "请仔细观察这张图片,尽可能详细、准确地描述其中的全部内容,包括文字、图表、物体、场景与布局等关键信息。"
|
|
|
|
|
|
class VisionConfig(BaseModel):
|
|
"""Configuration for image recognition (识图).
|
|
|
|
When ``model_name`` names a model from ``models[]``, the lead agent's
|
|
``view_image`` tool delegates image understanding to that dedicated model
|
|
and returns a text description — so a text-only chat model can still "see"
|
|
uploaded images. When ``model_name`` is unset, vision falls back to the main
|
|
chat model, which only works when that model has ``supports_vision: true``.
|
|
"""
|
|
|
|
model_name: str | None = Field(
|
|
default=None,
|
|
description="Name of the model (from models[]) used for image recognition. When unset, falls back to the main chat model if it supports vision.",
|
|
)
|
|
prompt: str = Field(
|
|
default=DEFAULT_VISION_PROMPT,
|
|
description="Default instruction for the vision model when no per-call question is given.",
|
|
)
|
|
max_tokens: int | None = Field(
|
|
default=None,
|
|
description="Optional override for the vision model's max output tokens; falls back to the model's own config when unset.",
|
|
)
|