feat(chat): image specialist legacy lane, Responses fixes, ACL + UI attachments

- Image expert: DashScope-style /chat/completions via image_legacy_client; early exit in
  direct_loop when skill_binding_role is image; shared placeholder helper; docs/IMAGE_SPECIALIST_LANE.md.
- Strict attachment ACL: link_attachment_acl on assistant chat_message rows (sqlite_store);
  chat attachment rate limit when user_id empty; admin chat tests updated.
- Admin chat UI: aggregate bubbles render assistant_text attachments (image_ref); WS expand path.
- turn_runner: persisted_chat_attachments_nonempty for final_msg selection.
- OpenAI Responses transport + agent_messages/agent_core_attempt adjustments; env docs and tests.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-05-10 07:06:05 +08:00
parent d1bcc4debe
commit faeb067856
21 changed files with 2516 additions and 331 deletions

View file

@ -1,5 +1,6 @@
from __future__ import annotations
import json
from dataclasses import dataclass
from typing import Any, Callable, Optional
@ -74,36 +75,67 @@ class AttemptRunnerOutput:
outcome: TurnRunOutcome
def _exception_reason_for_chat(exc: BaseException, *, max_chars: int = 12000) -> str:
"""Human-visible attempt failure text (WS/chat reads ``AttemptState.reason``).
OpenAI Python SDK's ``APIError`` often carries a decoded JSON ``body``; include it so oclaw /chat
can show the full upstream validation payload instead of only ``str(exc)``.
"""
head = f"{type(exc).__name__}:{exc}"
blob = ""
cur: BaseException | None = exc
seen_ids: set[int] = set()
depth = 0
while cur is not None and depth < 16:
if id(cur) in seen_ids:
break
seen_ids.add(id(cur))
depth += 1
body = getattr(cur, "body", None)
if body is not None:
try:
blob = json.dumps(body, ensure_ascii=False, indent=2) if isinstance(body, (dict, list)) else str(body)
except Exception:
blob = str(body)
break
cur = cur.__cause__
out = head if not blob else f"{head}\n\nupstream_json_body:\n{blob}"
if len(out) > max_chars:
out = out[: max_chars - 24] + "\n...[truncated]"
return out
def _classify_attempt_error(exc: Exception) -> tuple[str, str, bool]:
raw = f"{type(exc).__name__}:{exc}"
low = raw.lower()
reason = _exception_reason_for_chat(exc)
if "relay_envelope_unsupported_version" in low:
return ("relay_envelope_unsupported_version", raw[:500], False)
return ("relay_envelope_unsupported_version", reason, False)
if "relay_envelope_invalid" in low:
return ("relay_envelope_invalid", raw[:500], False)
return ("relay_envelope_invalid", reason, False)
if "interrupted" in low or "cancel" in low or "stopped" in low:
return ("control_interrupted", raw[:500], False)
return ("control_interrupted", reason, False)
if "api_key" in low or "invalid api key" in low or "unauthorized" in low or "401" in low or "forbidden" in low or "403" in low:
return ("auth_invalid_credentials", raw[:500], False)
return ("auth_invalid_credentials", reason, False)
if "invalid tool_result sequence" in low or "unexpected tool_use_id" in low:
return ("tool_replay_protocol_mismatch", raw[:500], True)
return ("tool_replay_protocol_mismatch", reason, True)
if "invalid_request" in low or "bad_request" in low or "400" in low:
return ("input_invalid_request", raw[:500], False)
return ("input_invalid_request", reason, False)
if "context_length" in low or "token limit" in low or "max context" in low:
return ("context_overflow", raw[:500], True)
return ("context_overflow", reason, True)
if "tool_loop_guard" in low:
return ("tool_loop_guard", raw[:500], False)
return ("tool_loop_guard", reason, False)
if "tool_timeout_or_failed" in low or "tool_execution_error" in low:
return ("tool_execution_failed", raw[:500], True)
return ("tool_execution_failed", reason, True)
if "timeout" in low:
return ("provider_timeout", raw[:500], True)
return ("provider_timeout", reason, True)
if "rate" in low and "limit" in low:
return ("provider_rate_limited", raw[:500], True)
return ("provider_rate_limited", reason, True)
if "temporary" in low or "temporar" in low:
return ("provider_temporary_error", raw[:500], True)
return ("provider_temporary_error", reason, True)
if "connection" in low or "network" in low or "503" in low or "502" in low:
return ("provider_unavailable", raw[:500], True)
return ("runtime_unknown_error", raw[:500], False)
return ("provider_unavailable", reason, True)
return ("runtime_unknown_error", reason, False)
def run_attempt(*, store: Any, data: AttemptRunnerInput) -> AttemptRunnerOutput:

View file

@ -2,10 +2,9 @@ from __future__ import annotations
import json
import os
import sys
import time
import base64
import hashlib
import httpx
from collections.abc import Callable
from dataclasses import dataclass, field
from typing import Any, Optional
@ -14,8 +13,13 @@ from oclaw.runtime.chat.agent import Agent
from oclaw.runtime.chat.agent import GenerationInterrupted
from oclaw.runtime.agents.network_ops_agent import NetworkOpsAgent
from oclaw.platform.persistence.sqlite_store import SqliteStore
from oclaw.platform.files.attachment_assets import AttachmentAssetStore, attachment_id_to_data_url
from oclaw.platform.llm.image_ocr_client import send_ocr_image_messages
from oclaw.platform.llm.image_legacy_client import (
IMAGE_SPECIALIST_DEFAULT_PROMPT_ZH,
collect_legacy_lane_images_from_attachments,
legacy_image_assistant_body_with_placeholder,
legacy_image_turn_bundle,
send_legacy_image_messages,
)
from oclaw.runtime.tools import default_registry
from oclaw.runtime.agents.specialists import expert_name_for_specialist
@ -233,56 +237,44 @@ class SpecialistAgentRunner:
try:
if step.specialist == "image":
image_protocol = "messages.content.image"
selected_images: list[str] = []
for att in parent_task.attachments or []:
if not isinstance(att, dict):
continue
t = str(att.get("type") or "").strip().lower()
if t == "image_ref":
aid = str(att.get("attachment_id") or "").strip()
if not aid:
continue
data_url = attachment_id_to_data_url(aid, mime=str(att.get("mime") or ""))
if data_url:
selected_images.append(data_url)
elif t in ("input_image", "image"):
raw = str(att.get("image_base64") or att.get("data") or "").strip()
if raw:
mime = str(att.get("mime") or "image/jpeg")
if raw.startswith("data:"):
selected_images.append(raw)
else:
selected_images.append(f"data:{mime};base64,{raw}")
elif t == "image_url":
u = str(att.get("url") or "").strip()
if u:
selected_images.append(u)
if len(selected_images) >= 3:
break
selected_images = collect_legacy_lane_images_from_attachments(
list(parent_task.attachments or []),
max_images=3,
)
image_input_count = len(selected_images)
image_input_kind = ["data_url" if s.startswith("data:") else "url" for s in selected_images]
if not selected_images:
output = "Image specialist received no image input."
ok = False
else:
# Use the user's chosen model/session profile (same as specialist routing UI). Wrong model ⇒ upstream HTTP error as-is (no OCR lane, no alternate payload).
_, chosen_model, _ = self._resolve_profile_and_model(step.specialist)
model_name = str(
os.getenv("AIA_OCR_MODEL") or getattr(chosen_model, "model", None) or ""
).strip() or None
api_key = str(getattr(chosen_model, "api_key", "") or "").strip() or None
base_url = str(getattr(chosen_model, "base_url", "") or "").strip() or None
text_parts = [
str(x).strip()
for x in (step.objective, step.input_text, parent_task.user_text)
if str(x or "").strip()
]
user_text = "\n".join(text_parts) if text_parts else "请根据用户上传的图片作答:描述可见场景、物体与文字;不确定处请标明。"
resp = send_ocr_image_messages(
user_text = "\n".join(text_parts) if text_parts else IMAGE_SPECIALIST_DEFAULT_PROMPT_ZH
if str(os.getenv("AIA_IMAGE_EXPERT_DEBUG_PRINT_PAYLOAD") or "").strip().lower() in (
"1",
"true",
"yes",
"on",
):
try:
sys.stderr.write(
"[oclaw specialist:image] lane=legacy_http → send_legacy_image_messages "
"(NOT OpenAIResponsesModel).\n"
)
sys.stderr.flush()
except Exception:
pass
resp = send_legacy_image_messages(
images=selected_images,
prompt=user_text,
model=model_name,
api_key=api_key,
base_url=base_url,
model=str(getattr(chosen_model, "model", "") or "").strip() or None,
api_key=str(getattr(chosen_model, "api_key", "") or "").strip() or None,
base_url=str(getattr(chosen_model, "base_url", "") or "").strip() or None,
)
image_debug_schema = str(resp.get("debug_used_schema") or "").strip()
dbg = resp.get("debug_used_debug")
@ -290,91 +282,12 @@ class SpecialistAgentRunner:
image_debug_payload = dbg
elif dbg is not None:
image_debug_payload = str(dbg)
ok = bool(resp.get("ok"))
output = str(resp.get("text") or "").strip()
if not ok:
err = str(resp.get("error") or "").strip()
output = f"Image generation failed: {err or 'unknown error'}"
elif not output:
output = "Image processed."
# persist output images as attachment assets for UI rendering
produced_attachments: list[dict[str, Any]] = []
if ok:
out_images = resp.get("images")
if isinstance(out_images, list):
store = AttachmentAssetStore()
for idx, item in enumerate(out_images[:3], start=1):
s = str(item or "").strip()
if not s:
continue
if s.startswith("data:") and ";base64," in s:
head, b64 = s.split(";base64,", 1)
mime = head.replace("data:", "", 1) or "image/png"
try:
blob = base64.b64decode(b64.encode("ascii"))
except Exception:
continue
meta = store.save_bytes(
blob,
filename=f"image-output-{idx}.png",
mime=mime,
)
produced_attachments.append(
{
"type": "image_ref",
"attachment_id": meta.attachment_id,
"name": meta.name,
"mime": meta.mime,
"bytes": meta.bytes,
"width": meta.width,
"height": meta.height,
}
)
elif s.startswith("http://") or s.startswith("https://"):
try:
with httpx.Client(timeout=20.0, follow_redirects=True) as client:
r = client.get(s)
if r.status_code < 400 and r.content:
mime = str(r.headers.get("content-type") or "image/png").split(";", 1)[0].strip() or "image/png"
ext = ".png"
if mime == "image/jpeg":
ext = ".jpg"
elif mime == "image/webp":
ext = ".webp"
elif mime == "image/gif":
ext = ".gif"
meta = store.save_bytes(
r.content,
filename=f"image-output-{idx}{ext}",
mime=mime,
)
produced_attachments.append(
{
"type": "image_ref",
"attachment_id": meta.attachment_id,
"name": meta.name,
"mime": meta.mime,
"bytes": meta.bytes,
"width": meta.width,
"height": meta.height,
}
)
continue
except Exception:
pass
produced_attachments.append(
{
"type": "image_url",
"url": s,
"name": f"image-output-{idx}.png",
}
)
# Treat missing image outputs as failure to avoid false "generated" state.
if not produced_attachments:
ok = False
output = (
"Image generation failed: response succeeded but no image output was returned."
)
ok, output, produced_attachments = legacy_image_turn_bundle(resp)
output = legacy_image_assistant_body_with_placeholder(
lang=self.lang,
body_text=output,
produced=produced_attachments if ok else None,
)
self.store.add_message(
session_id=session_id,
role="assistant",

View file

@ -210,8 +210,10 @@ def build_llm_messages(
:func:`~oclaw.runtime.direct_loop._guard_tool_results_for_llm_context`). Omit or leave empty
to apply image-blob stripping for every tool row (safe default for callers without turn context).
Only the **last** user message may expand attachments into native multimodal ``input_image``;
older user attachments are replayed as text metadata only.
Only the **last** user message may expand attachments into native multimodal ``input_image``
blocks shaped like ``{"type":"input_image","image_base64","mime"}``; ``OpenAIChatModel`` and
``OpenAIResponsesModel`` both accept this shape (see transport normalization code). Older user
attachments are replayed as text metadata only.
"""
out: list[dict[str, Any]] = [{"role": "system", "content": (system_prompt or "").strip()}]
dropped_unpaired_tool_rows = 0

View file

@ -1102,6 +1102,101 @@ def _execute_tool_step(
return int((time.perf_counter() - t0) * 1000), results_by_id
def _maybe_image_specialist_legacy_gateway_turn(
*,
store: Any,
session_id: str,
turn_uuid: str,
lang: str,
model: ChatModel,
user_text: str,
attachments: list[dict[str, Any]] | None,
skill_binding_role: str | None,
on_token: Optional[Callable[[str], None]],
on_progress: Optional[Callable[[str], None]],
) -> TurnRunOutcome | None:
"""When the UI selects **image** specialist, skip Responses/chat-model transports.
Vision/gen HTTP goes through :func:`oclaw.platform.llm.image_legacy_client.send_legacy_image_messages`
(``/chat/completions`` lane). Disable with ``AIA_IMAGE_SPECIALIST_DISABLE_LEGACY_GATEWAY_LANE=1``.
End-to-end notes and safe edit boundaries: ``docs/IMAGE_SPECIALIST_LANE.md``.
"""
if str(os.getenv("AIA_IMAGE_SPECIALIST_DISABLE_LEGACY_GATEWAY_LANE") or "").strip().lower() in (
"1",
"true",
"yes",
"on",
):
return None
if str(skill_binding_role or "").strip().lower() != "image":
return None
from oclaw.platform.llm.image_legacy_client import (
IMAGE_SPECIALIST_DEFAULT_PROMPT_ZH,
collect_legacy_lane_images_from_attachments,
legacy_image_assistant_body_with_placeholder,
legacy_image_turn_bundle,
send_legacy_image_messages,
)
imgs = collect_legacy_lane_images_from_attachments(attachments)
if not imgs:
hint_en = "Image specialist received no image input. Attach an image and try again."
hint_zh = "图片专家未收到可用的图片输入;请先上传或附上图片后再试。"
hint = hint_en if str(lang or "").startswith("en") else hint_zh
store.add_message(
session_id=session_id,
role="assistant",
content=hint,
turn_uuid=turn_uuid,
event_type="assistant_text",
)
return TurnRunOutcome(
final_text=hint,
tool_traces=tuple(),
handoff_note="image_specialist_legacy_missing_attachment",
turn_uuid=turn_uuid,
)
if on_progress:
on_progress("oclaw: image specialist (legacy multimodal HTTP)…")
prompt_plain = str(user_text or "").strip()
if not prompt_plain:
prompt_plain = IMAGE_SPECIALIST_DEFAULT_PROMPT_ZH
resp = send_legacy_image_messages(
images=imgs,
prompt=prompt_plain,
model=str(getattr(model, "model", "") or "").strip() or None,
api_key=str(getattr(model, "api_key", "") or "").strip() or None,
base_url=str(getattr(model, "base_url", "") or "").strip() or None,
)
ok, body_text, produced = legacy_image_turn_bundle(resp)
body_text = legacy_image_assistant_body_with_placeholder(
lang=lang,
body_text=body_text,
produced=produced if ok else None,
)
store.add_message(
session_id=session_id,
role="assistant",
content=body_text,
turn_uuid=turn_uuid,
event_type="assistant_text",
attachments=(produced or None) if ok else None,
)
if ok and on_token and body_text:
on_token(body_text)
return TurnRunOutcome(
final_text=body_text,
tool_traces=tuple(),
handoff_note="image_specialist_legacy_http" if ok else "image_specialist_legacy_upstream_failed",
turn_uuid=turn_uuid,
)
def run_oclaw_direct_loop(
*,
store: Any,
@ -1149,6 +1244,20 @@ def run_oclaw_direct_loop(
turn_uuid=turn_uuid,
event_type="user_text",
)
legacy_early = _maybe_image_specialist_legacy_gateway_turn(
store=store,
session_id=session_id,
turn_uuid=turn_uuid,
lang=lang,
model=model,
user_text=str(user_text or ""),
attachments=attachments,
skill_binding_role=skill_binding_role,
on_token=on_token,
on_progress=on_progress,
)
if legacy_early is not None:
return legacy_early
skill_exec = SkillExecutor(config=ToolExecutionConfig(max_workers=max(1, min(int(max_tool_workers or 8), 32))))
tool_traces: list[dict[str, Any]] = []

View file

@ -0,0 +1,106 @@
#!/usr/bin/env python3
"""POST minimal vision payloads to an OpenAI-compatible ``/responses`` endpoint.
Run locally: loads ``oclaw/_local/system.env`` the same way the gateway does, then POSTs variants.
Example::
set OPENAI_BASE_URL=https://...
set OPENAI_API_KEY=sk-...
set OPENAI_MODEL=qwen-vl-plus
python runtime/operations/scripts/probe_openai_responses_image.py
"""
from __future__ import annotations
import json
import os
import sys
from pathlib import Path
_REPO_ROOT = Path(__file__).resolve().parents[3]
if str(_REPO_ROOT) not in sys.path:
sys.path.insert(0, str(_REPO_ROOT))
# Same bootstrap as gateway: ``interfaces/http/fastapi_app.py`` calls ``load_system_env()`` so
# ``oclaw/_local/system.env`` is merged before reading ``OPENAI_*``.
try:
from oclaw.platform.config.bootstrap_env import load_system_env
load_system_env()
except ImportError:
pass
try:
import httpx
except ImportError:
print("install httpx: pip install httpx", file=sys.stderr)
raise SystemExit(2)
# 1x1 transparent PNG
_PNG_B64 = (
"iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8BQDwAEhQGAhKmMIQAAAABJRU5ErkJggg=="
)
def _variants(model: str, data_uri: str) -> list[tuple[str, dict]]:
user_block_resp = {
"type": "message",
"role": "user",
"content": [
{"type": "input_text", "text": "What color is this pixel image? One word."},
{"type": "input_image", "image_url": data_uri, "detail": "auto"},
],
}
user_plain_resp = {
"role": "user",
"content": [
{"type": "input_text", "text": "What color is this pixel image? One word."},
{"type": "input_image", "image_url": data_uri, "detail": "auto"},
],
}
user_chat = {
"role": "user",
"content": [
{"type": "text", "text": "What color is this pixel image? One word."},
{"type": "image_url", "image_url": {"url": data_uri}},
],
}
return [
("flat_resp_envelope", {"model": model, "input": [user_block_resp]}),
("flat_resp_plain", {"model": model, "input": [user_plain_resp]}),
("nested_messages_resp_env", {"model": model, "input": {"messages": [user_block_resp]}}),
("nested_messages_resp_plain", {"model": model, "input": {"messages": [user_plain_resp]}}),
("nested_messages_chat", {"model": model, "input": {"messages": [user_chat]}}),
]
def main() -> None:
base = (os.getenv("OPENAI_BASE_URL") or "https://api.openai.com/v1").strip().rstrip("/")
key = (os.getenv("OPENAI_API_KEY") or "").strip()
model = (os.getenv("OPENAI_MODEL") or "gpt-4o-mini").strip()
if not key:
print("OPENAI_API_KEY is required", file=sys.stderr)
raise SystemExit(2)
url = f"{base}/responses"
data_uri = f"data:image/png;base64,{_PNG_B64}"
headers = {"Authorization": f"Bearer {key}", "Content-Type": "application/json"}
print(f"POST {url}", flush=True)
for label, body in _variants(model, data_uri):
try:
r = httpx.post(url, headers=headers, json=body, timeout=120.0)
except httpx.HTTPError as exc:
print(f"\n=== {label} transport_error {exc}", flush=True)
continue
tail = (r.text or "")[:2400]
print(f"\n=== {label} status={r.status_code}", flush=True)
print(tail, flush=True)
if r.status_code < 400:
print("(first successful variant — use this shape for your gateway)", flush=True)
break
if __name__ == "__main__":
main()