feat(video): DashScope video specialist with correct i2v request bodies

- Add video_generation_client: async video-synthesis, t2v vs i2v (Wan 2.7 input.media first_frame vs legacy img_url).

- Coerce *-t2v* to *-i2v* when first frame present; AIA_VIDEO_I2V_INPUT_STYLE / I2V_MODEL overrides.

- Gateway/direct_loop early return for video specialist; workspace video + factory allowlist.

- Normalize WS and admin chat attachments (image_ref parity); mp4 attachment store roundtrip tests.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-05-10 16:20:43 +08:00
parent 398d85ba06
commit 4335040a27
21 changed files with 1152 additions and 56 deletions

View file

@ -276,7 +276,7 @@ def build_ops_agent(
# `default_registry` treats empty allow_tags + empty allow_tools as "no filter". Use an impossible
# tool name so the image specialist gets an empty tool surface (vision-only turns).
# tool name so image/video specialists get an empty tool surface (dedicated HTTP lanes).
_IMAGE_SPECIALIST_TOOL_ALLOWLIST: tuple[str, ...] = ("__oclaw_image_specialist_no_tools__",)
@ -334,7 +334,7 @@ def build_gateway_executor(
"path_policy_user_id": path_policy_user_id,
"store": store,
}
if prof.name == "image":
if prof.name in {"image", "video"}:
reg_kw["allow_tools"] = list(_IMAGE_SPECIALIST_TOOL_ALLOWLIST)
tools = default_registry(**reg_kw)
return Agent(

View file

@ -16,10 +16,17 @@ from oclaw.platform.persistence.sqlite_store import SqliteStore
from oclaw.platform.llm.image_legacy_client import (
IMAGE_SPECIALIST_DEFAULT_PROMPT_ZH,
collect_legacy_lane_images_from_attachments,
collect_legacy_lane_images_with_session_fallback,
legacy_image_assistant_body_with_placeholder,
legacy_image_turn_bundle,
send_legacy_image_messages,
)
from oclaw.platform.llm.video_generation_client import (
VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH,
legacy_video_assistant_body_with_placeholder,
legacy_video_turn_bundle,
send_video_generation_request,
)
from oclaw.runtime.tools import default_registry
from oclaw.runtime.agents.specialists import expert_name_for_specialist
@ -301,6 +308,51 @@ class SpecialistAgentRunner:
tool_traces=(),
notes="image_pipeline",
)
elif step.specialist == "video":
image_protocol = "video_generation.http"
_, chosen_model, _ = self._resolve_profile_and_model(step.specialist)
text_parts = [
str(x).strip()
for x in (step.objective, step.input_text, parent_task.user_text)
if str(x or "").strip()
]
user_text_v = "\n".join(text_parts) if text_parts else VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH
policy_sid = str(policy_session_id or parent_task.session_id or session_id or "").strip()
v_frames, _v_src = collect_legacy_lane_images_with_session_fallback(
store=self.store,
session_id=policy_sid,
attachments=list(parent_task.attachments or []),
max_images=1,
)
v_frame = str(v_frames[0]).strip() if v_frames else None
resp_v = send_video_generation_request(
prompt=user_text_v,
model=str(getattr(chosen_model, "model", "") or "").strip() or None,
api_key=str(getattr(chosen_model, "api_key", "") or "").strip() or None,
base_url=str(getattr(chosen_model, "base_url", "") or "").strip() or None,
img_url=v_frame,
on_progress=on_progress,
should_stop=should_stop,
)
ok, output, produced_attachments = legacy_video_turn_bundle(resp_v)
output = legacy_video_assistant_body_with_placeholder(
lang=self.lang,
body_text=output,
produced=produced_attachments if ok else None,
)
self.store.add_message(
session_id=session_id,
role="assistant",
content=output,
attachments=(produced_attachments or None) if ok else None,
)
specialist_delivery = SpecialistDelivery(
specialist=step.specialist,
step_id=step.step_id,
answer_text=str(output or ""),
tool_traces=(),
notes="video_pipeline",
)
else:
agent = self._build_agent_for(
step.specialist,

View file

@ -43,10 +43,15 @@ SPECIALISTS: dict[SpecialistId, SpecialistConfig] = {
expert_name="image",
default_tool_tags=None,
),
"video": SpecialistConfig(
specialist_id="video",
expert_name="video",
default_tool_tags=None,
),
}
def discover_specialist_ids() -> tuple[SpecialistId, ...]:
rows = specialist_registry_snapshot(base_order=("generalist", "ops", "memory", "image"))
rows = specialist_registry_snapshot(base_order=("generalist", "ops", "memory", "image", "video"))
return tuple(str(x.get("id") or "").strip().lower() for x in rows if str(x.get("id") or "").strip())

View file

@ -1206,6 +1206,97 @@ def _maybe_image_specialist_legacy_gateway_turn(
)
def _maybe_video_specialist_legacy_gateway_turn(
*,
store: Any,
session_id: str,
turn_uuid: str,
lang: str,
model: ChatModel,
user_text: str,
attachments: list[dict[str, Any]] | None,
skill_binding_role: str | None,
on_token: Optional[Callable[[str], None]],
on_progress: Optional[Callable[[str], None]],
should_stop: Optional[Callable[[], bool]] = None,
) -> TurnRunOutcome | None:
"""When the UI selects **video** specialist, skip Responses/chat-model transports.
Uses DashScope async ``video-synthesis`` (see :mod:`oclaw.platform.llm.video_generation_client`).
With a user image (or session image fallback), sends ``input.img_url`` for **image-to-video**;
otherwise **text-to-video**. Disable with ``AIA_VIDEO_SPECIALIST_DISABLE_LEGACY_GATEWAY_LANE=1``.
"""
if str(os.getenv("AIA_VIDEO_SPECIALIST_DISABLE_LEGACY_GATEWAY_LANE") or "").strip().lower() in (
"1",
"true",
"yes",
"on",
):
return None
if str(skill_binding_role or "").strip().lower() != "video":
return None
from oclaw.platform.llm.image_legacy_client import collect_legacy_lane_images_with_session_fallback
from oclaw.platform.llm.video_generation_client import (
VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH,
legacy_video_assistant_body_with_placeholder,
legacy_video_turn_bundle,
send_video_generation_request,
)
frames, frame_src = collect_legacy_lane_images_with_session_fallback(
store=store,
session_id=session_id,
attachments=attachments,
max_images=1,
)
frame_url = str(frames[0]).strip() if frames else None
if on_progress:
if frame_url:
if frame_src.endswith("_history"):
if str(lang or "").startswith("en"):
on_progress("oclaw: reusing an earlier session image as first frame…")
else:
on_progress("oclaw: 使用会话中较早的图片作为图生视频首帧…")
on_progress("oclaw: video specialist (DashScope image-to-video)…")
else:
on_progress("oclaw: video specialist (DashScope text-to-video)…")
prompt_plain = str(user_text or "").strip() or VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH
resp = send_video_generation_request(
prompt=prompt_plain,
model=str(getattr(model, "model", "") or "").strip() or None,
api_key=str(getattr(model, "api_key", "") or "").strip() or None,
base_url=str(getattr(model, "base_url", "") or "").strip() or None,
img_url=frame_url,
on_progress=on_progress,
should_stop=should_stop,
)
ok, body_text, produced = legacy_video_turn_bundle(resp)
body_text = legacy_video_assistant_body_with_placeholder(
lang=lang,
body_text=body_text,
produced=produced if ok else None,
)
store.add_message(
session_id=session_id,
role="assistant",
content=body_text,
turn_uuid=turn_uuid,
event_type="assistant_text",
attachments=(produced or None) if ok else None,
)
if ok and on_token and body_text:
on_token(body_text)
return TurnRunOutcome(
final_text=body_text,
tool_traces=tuple(),
handoff_note="video_specialist_legacy_http" if ok else "video_specialist_legacy_upstream_failed",
turn_uuid=turn_uuid,
)
def run_oclaw_direct_loop(
*,
store: Any,
@ -1267,6 +1358,21 @@ def run_oclaw_direct_loop(
)
if legacy_early is not None:
return legacy_early
video_early = _maybe_video_specialist_legacy_gateway_turn(
store=store,
session_id=session_id,
turn_uuid=turn_uuid,
lang=lang,
model=model,
user_text=str(user_text or ""),
attachments=attachments,
skill_binding_role=skill_binding_role,
on_token=on_token,
on_progress=on_progress,
should_stop=should_stop,
)
if video_early is not None:
return video_early
skill_exec = SkillExecutor(config=ToolExecutionConfig(max_workers=max(1, min(int(max_tool_workers or 8), 32))))
tool_traces: list[dict[str, Any]] = []

View file

@ -919,7 +919,7 @@ class OclawGateway:
attachments=list(msg.attachments or []),
metadata=dict(base_metadata),
)
if manager_specialist in {"ops", "generalist", "image", "memory"}:
if manager_specialist in {"ops", "generalist", "image", "memory", "video"}:
if callable(specialist_executor_factory):
try:
selected_executor = specialist_executor_factory(manager_specialist)

View file

@ -17,7 +17,7 @@ def _truthy(v: str | None) -> bool:
def ordered_specialist_ids() -> list[str]:
base = [str(k).strip().lower() for k in discover_specialist_ids() if str(k).strip()]
preferred = [x for x in ("generalist", "ops", "memory", "image") if x in set(base)]
preferred = [x for x in ("generalist", "ops", "memory", "image", "video") if x in set(base)]
return preferred + [x for x in base if x not in set(preferred)]

View file

@ -252,7 +252,7 @@ def build_expert_catalog_block(*, include_main: bool = False, per_field_limit: i
def discover_specialist_ids_from_workspaces(
*,
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image"),
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image", "video"),
) -> tuple[str, ...]:
cache_key = (expert_workspace_signature_token(), tuple(str(x).strip().lower() for x in base_order if str(x).strip()))
with _CACHE_LOCK:
@ -291,7 +291,7 @@ def warm_expert_workspace_cache() -> None:
def specialist_registry_snapshot(
*,
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image"),
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image", "video"),
) -> tuple[dict[str, Any], ...]:
"""Single source of truth for runtime specialist discovery and metadata."""
ordered = discover_specialist_ids_from_workspaces(base_order=base_order)

View file

@ -0,0 +1,5 @@
{
"display_name_en": "Video generation",
"display_name_zh": "视频生成专家",
"role": "expert"
}

View file

@ -0,0 +1 @@
你是视频生成专家:将用户自然语言 prompt 交给 Wan / 百炼 text-to-video API,返回可下载或可播放的成片附件。不要编造已生成视频的 URL;仅展示接口真实返回结果或明确错误。

View file

@ -0,0 +1,9 @@
你是**文生视频 / 图生视频**方向的专家助手:根据用户给出的画面与镜头描述(及可选的**首帧参考图**),调用百炼 / DashScope **异步视频合成**接口生成短视频结果,并将产出以会话附件(`video_ref`)形式返回。用户上传图片时,首帧会作为 `img_url` 提交(需使用支持图生视频的 i2v 模型)。
回答要求:
- 若用户描述含糊,可基于常识补全合理的镜头语言,但避免与用户明确约束相矛盾。
- 生成失败时给出可读的上游错误或参数提示(如模型与区域、时长、分辨率不匹配)。
边界:
- 本专家链路**不调用**通用工具循环;仅走专用 HTTP 视频合成与轮询。
- 不承诺具体成片内容符合版权素材或真人肖像等合规要求;用户需自行确保 prompt 合规。