mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 11:50:45 +08:00
feat(video): DashScope video specialist with correct i2v request bodies
- Add video_generation_client: async video-synthesis, t2v vs i2v (Wan 2.7 input.media first_frame vs legacy img_url). - Coerce *-t2v* to *-i2v* when first frame present; AIA_VIDEO_I2V_INPUT_STYLE / I2V_MODEL overrides. - Gateway/direct_loop early return for video specialist; workspace video + factory allowlist. - Normalize WS and admin chat attachments (image_ref parity); mp4 attachment store roundtrip tests. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
398d85ba06
commit
4335040a27
21 changed files with 1152 additions and 56 deletions
|
|
@ -276,7 +276,7 @@ def build_ops_agent(
|
|||
|
||||
|
||||
# `default_registry` treats empty allow_tags + empty allow_tools as "no filter". Use an impossible
|
||||
# tool name so the image specialist gets an empty tool surface (vision-only turns).
|
||||
# tool name so image/video specialists get an empty tool surface (dedicated HTTP lanes).
|
||||
_IMAGE_SPECIALIST_TOOL_ALLOWLIST: tuple[str, ...] = ("__oclaw_image_specialist_no_tools__",)
|
||||
|
||||
|
||||
|
|
@ -334,7 +334,7 @@ def build_gateway_executor(
|
|||
"path_policy_user_id": path_policy_user_id,
|
||||
"store": store,
|
||||
}
|
||||
if prof.name == "image":
|
||||
if prof.name in {"image", "video"}:
|
||||
reg_kw["allow_tools"] = list(_IMAGE_SPECIALIST_TOOL_ALLOWLIST)
|
||||
tools = default_registry(**reg_kw)
|
||||
return Agent(
|
||||
|
|
|
|||
|
|
@ -16,10 +16,17 @@ from oclaw.platform.persistence.sqlite_store import SqliteStore
|
|||
from oclaw.platform.llm.image_legacy_client import (
|
||||
IMAGE_SPECIALIST_DEFAULT_PROMPT_ZH,
|
||||
collect_legacy_lane_images_from_attachments,
|
||||
collect_legacy_lane_images_with_session_fallback,
|
||||
legacy_image_assistant_body_with_placeholder,
|
||||
legacy_image_turn_bundle,
|
||||
send_legacy_image_messages,
|
||||
)
|
||||
from oclaw.platform.llm.video_generation_client import (
|
||||
VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH,
|
||||
legacy_video_assistant_body_with_placeholder,
|
||||
legacy_video_turn_bundle,
|
||||
send_video_generation_request,
|
||||
)
|
||||
from oclaw.runtime.tools import default_registry
|
||||
from oclaw.runtime.agents.specialists import expert_name_for_specialist
|
||||
|
||||
|
|
@ -301,6 +308,51 @@ class SpecialistAgentRunner:
|
|||
tool_traces=(),
|
||||
notes="image_pipeline",
|
||||
)
|
||||
elif step.specialist == "video":
|
||||
image_protocol = "video_generation.http"
|
||||
_, chosen_model, _ = self._resolve_profile_and_model(step.specialist)
|
||||
text_parts = [
|
||||
str(x).strip()
|
||||
for x in (step.objective, step.input_text, parent_task.user_text)
|
||||
if str(x or "").strip()
|
||||
]
|
||||
user_text_v = "\n".join(text_parts) if text_parts else VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH
|
||||
policy_sid = str(policy_session_id or parent_task.session_id or session_id or "").strip()
|
||||
v_frames, _v_src = collect_legacy_lane_images_with_session_fallback(
|
||||
store=self.store,
|
||||
session_id=policy_sid,
|
||||
attachments=list(parent_task.attachments or []),
|
||||
max_images=1,
|
||||
)
|
||||
v_frame = str(v_frames[0]).strip() if v_frames else None
|
||||
resp_v = send_video_generation_request(
|
||||
prompt=user_text_v,
|
||||
model=str(getattr(chosen_model, "model", "") or "").strip() or None,
|
||||
api_key=str(getattr(chosen_model, "api_key", "") or "").strip() or None,
|
||||
base_url=str(getattr(chosen_model, "base_url", "") or "").strip() or None,
|
||||
img_url=v_frame,
|
||||
on_progress=on_progress,
|
||||
should_stop=should_stop,
|
||||
)
|
||||
ok, output, produced_attachments = legacy_video_turn_bundle(resp_v)
|
||||
output = legacy_video_assistant_body_with_placeholder(
|
||||
lang=self.lang,
|
||||
body_text=output,
|
||||
produced=produced_attachments if ok else None,
|
||||
)
|
||||
self.store.add_message(
|
||||
session_id=session_id,
|
||||
role="assistant",
|
||||
content=output,
|
||||
attachments=(produced_attachments or None) if ok else None,
|
||||
)
|
||||
specialist_delivery = SpecialistDelivery(
|
||||
specialist=step.specialist,
|
||||
step_id=step.step_id,
|
||||
answer_text=str(output or ""),
|
||||
tool_traces=(),
|
||||
notes="video_pipeline",
|
||||
)
|
||||
else:
|
||||
agent = self._build_agent_for(
|
||||
step.specialist,
|
||||
|
|
|
|||
|
|
@ -43,10 +43,15 @@ SPECIALISTS: dict[SpecialistId, SpecialistConfig] = {
|
|||
expert_name="image",
|
||||
default_tool_tags=None,
|
||||
),
|
||||
"video": SpecialistConfig(
|
||||
specialist_id="video",
|
||||
expert_name="video",
|
||||
default_tool_tags=None,
|
||||
),
|
||||
}
|
||||
|
||||
def discover_specialist_ids() -> tuple[SpecialistId, ...]:
|
||||
rows = specialist_registry_snapshot(base_order=("generalist", "ops", "memory", "image"))
|
||||
rows = specialist_registry_snapshot(base_order=("generalist", "ops", "memory", "image", "video"))
|
||||
return tuple(str(x.get("id") or "").strip().lower() for x in rows if str(x.get("id") or "").strip())
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1206,6 +1206,97 @@ def _maybe_image_specialist_legacy_gateway_turn(
|
|||
)
|
||||
|
||||
|
||||
def _maybe_video_specialist_legacy_gateway_turn(
|
||||
*,
|
||||
store: Any,
|
||||
session_id: str,
|
||||
turn_uuid: str,
|
||||
lang: str,
|
||||
model: ChatModel,
|
||||
user_text: str,
|
||||
attachments: list[dict[str, Any]] | None,
|
||||
skill_binding_role: str | None,
|
||||
on_token: Optional[Callable[[str], None]],
|
||||
on_progress: Optional[Callable[[str], None]],
|
||||
should_stop: Optional[Callable[[], bool]] = None,
|
||||
) -> TurnRunOutcome | None:
|
||||
"""When the UI selects **video** specialist, skip Responses/chat-model transports.
|
||||
|
||||
Uses DashScope async ``video-synthesis`` (see :mod:`oclaw.platform.llm.video_generation_client`).
|
||||
With a user image (or session image fallback), sends ``input.img_url`` for **image-to-video**;
|
||||
otherwise **text-to-video**. Disable with ``AIA_VIDEO_SPECIALIST_DISABLE_LEGACY_GATEWAY_LANE=1``.
|
||||
"""
|
||||
if str(os.getenv("AIA_VIDEO_SPECIALIST_DISABLE_LEGACY_GATEWAY_LANE") or "").strip().lower() in (
|
||||
"1",
|
||||
"true",
|
||||
"yes",
|
||||
"on",
|
||||
):
|
||||
return None
|
||||
if str(skill_binding_role or "").strip().lower() != "video":
|
||||
return None
|
||||
|
||||
from oclaw.platform.llm.image_legacy_client import collect_legacy_lane_images_with_session_fallback
|
||||
from oclaw.platform.llm.video_generation_client import (
|
||||
VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH,
|
||||
legacy_video_assistant_body_with_placeholder,
|
||||
legacy_video_turn_bundle,
|
||||
send_video_generation_request,
|
||||
)
|
||||
|
||||
frames, frame_src = collect_legacy_lane_images_with_session_fallback(
|
||||
store=store,
|
||||
session_id=session_id,
|
||||
attachments=attachments,
|
||||
max_images=1,
|
||||
)
|
||||
frame_url = str(frames[0]).strip() if frames else None
|
||||
|
||||
if on_progress:
|
||||
if frame_url:
|
||||
if frame_src.endswith("_history"):
|
||||
if str(lang or "").startswith("en"):
|
||||
on_progress("oclaw: reusing an earlier session image as first frame…")
|
||||
else:
|
||||
on_progress("oclaw: 使用会话中较早的图片作为图生视频首帧…")
|
||||
on_progress("oclaw: video specialist (DashScope image-to-video)…")
|
||||
else:
|
||||
on_progress("oclaw: video specialist (DashScope text-to-video)…")
|
||||
|
||||
prompt_plain = str(user_text or "").strip() or VIDEO_SPECIALIST_DEFAULT_PROMPT_ZH
|
||||
resp = send_video_generation_request(
|
||||
prompt=prompt_plain,
|
||||
model=str(getattr(model, "model", "") or "").strip() or None,
|
||||
api_key=str(getattr(model, "api_key", "") or "").strip() or None,
|
||||
base_url=str(getattr(model, "base_url", "") or "").strip() or None,
|
||||
img_url=frame_url,
|
||||
on_progress=on_progress,
|
||||
should_stop=should_stop,
|
||||
)
|
||||
ok, body_text, produced = legacy_video_turn_bundle(resp)
|
||||
body_text = legacy_video_assistant_body_with_placeholder(
|
||||
lang=lang,
|
||||
body_text=body_text,
|
||||
produced=produced if ok else None,
|
||||
)
|
||||
store.add_message(
|
||||
session_id=session_id,
|
||||
role="assistant",
|
||||
content=body_text,
|
||||
turn_uuid=turn_uuid,
|
||||
event_type="assistant_text",
|
||||
attachments=(produced or None) if ok else None,
|
||||
)
|
||||
if ok and on_token and body_text:
|
||||
on_token(body_text)
|
||||
return TurnRunOutcome(
|
||||
final_text=body_text,
|
||||
tool_traces=tuple(),
|
||||
handoff_note="video_specialist_legacy_http" if ok else "video_specialist_legacy_upstream_failed",
|
||||
turn_uuid=turn_uuid,
|
||||
)
|
||||
|
||||
|
||||
def run_oclaw_direct_loop(
|
||||
*,
|
||||
store: Any,
|
||||
|
|
@ -1267,6 +1358,21 @@ def run_oclaw_direct_loop(
|
|||
)
|
||||
if legacy_early is not None:
|
||||
return legacy_early
|
||||
video_early = _maybe_video_specialist_legacy_gateway_turn(
|
||||
store=store,
|
||||
session_id=session_id,
|
||||
turn_uuid=turn_uuid,
|
||||
lang=lang,
|
||||
model=model,
|
||||
user_text=str(user_text or ""),
|
||||
attachments=attachments,
|
||||
skill_binding_role=skill_binding_role,
|
||||
on_token=on_token,
|
||||
on_progress=on_progress,
|
||||
should_stop=should_stop,
|
||||
)
|
||||
if video_early is not None:
|
||||
return video_early
|
||||
|
||||
skill_exec = SkillExecutor(config=ToolExecutionConfig(max_workers=max(1, min(int(max_tool_workers or 8), 32))))
|
||||
tool_traces: list[dict[str, Any]] = []
|
||||
|
|
|
|||
|
|
@ -919,7 +919,7 @@ class OclawGateway:
|
|||
attachments=list(msg.attachments or []),
|
||||
metadata=dict(base_metadata),
|
||||
)
|
||||
if manager_specialist in {"ops", "generalist", "image", "memory"}:
|
||||
if manager_specialist in {"ops", "generalist", "image", "memory", "video"}:
|
||||
if callable(specialist_executor_factory):
|
||||
try:
|
||||
selected_executor = specialist_executor_factory(manager_specialist)
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ def _truthy(v: str | None) -> bool:
|
|||
|
||||
def ordered_specialist_ids() -> list[str]:
|
||||
base = [str(k).strip().lower() for k in discover_specialist_ids() if str(k).strip()]
|
||||
preferred = [x for x in ("generalist", "ops", "memory", "image") if x in set(base)]
|
||||
preferred = [x for x in ("generalist", "ops", "memory", "image", "video") if x in set(base)]
|
||||
return preferred + [x for x in base if x not in set(preferred)]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -252,7 +252,7 @@ def build_expert_catalog_block(*, include_main: bool = False, per_field_limit: i
|
|||
|
||||
def discover_specialist_ids_from_workspaces(
|
||||
*,
|
||||
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image"),
|
||||
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image", "video"),
|
||||
) -> tuple[str, ...]:
|
||||
cache_key = (expert_workspace_signature_token(), tuple(str(x).strip().lower() for x in base_order if str(x).strip()))
|
||||
with _CACHE_LOCK:
|
||||
|
|
@ -291,7 +291,7 @@ def warm_expert_workspace_cache() -> None:
|
|||
|
||||
def specialist_registry_snapshot(
|
||||
*,
|
||||
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image"),
|
||||
base_order: tuple[str, ...] = ("generalist", "ops", "memory", "image", "video"),
|
||||
) -> tuple[dict[str, Any], ...]:
|
||||
"""Single source of truth for runtime specialist discovery and metadata."""
|
||||
ordered = discover_specialist_ids_from_workspaces(base_order=base_order)
|
||||
|
|
|
|||
5
runtime/workspaces/video/META.json
Normal file
5
runtime/workspaces/video/META.json
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
{
|
||||
"display_name_en": "Video generation",
|
||||
"display_name_zh": "视频生成专家",
|
||||
"role": "expert"
|
||||
}
|
||||
1
runtime/workspaces/video/ROLE_SYSTEM.md
Normal file
1
runtime/workspaces/video/ROLE_SYSTEM.md
Normal file
|
|
@ -0,0 +1 @@
|
|||
你是视频生成专家:将用户自然语言 prompt 交给 Wan / 百炼 text-to-video API,返回可下载或可播放的成片附件。不要编造已生成视频的 URL;仅展示接口真实返回结果或明确错误。
|
||||
9
runtime/workspaces/video/SOUL.md
Normal file
9
runtime/workspaces/video/SOUL.md
Normal file
|
|
@ -0,0 +1,9 @@
|
|||
你是**文生视频 / 图生视频**方向的专家助手:根据用户给出的画面与镜头描述(及可选的**首帧参考图**),调用百炼 / DashScope **异步视频合成**接口生成短视频结果,并将产出以会话附件(`video_ref`)形式返回。用户上传图片时,首帧会作为 `img_url` 提交(需使用支持图生视频的 i2v 模型)。
|
||||
|
||||
回答要求:
|
||||
- 若用户描述含糊,可基于常识补全合理的镜头语言,但避免与用户明确约束相矛盾。
|
||||
- 生成失败时给出可读的上游错误或参数提示(如模型与区域、时长、分辨率不匹配)。
|
||||
|
||||
边界:
|
||||
- 本专家链路**不调用**通用工具循环;仅走专用 HTTP 视频合成与轮询。
|
||||
- 不承诺具体成片内容符合版权素材或真人肖像等合规要求;用户需自行确保 prompt 合规。
|
||||
Loading…
Add table
Add a link
Reference in a new issue