feat(chat): image specialist legacy lane, Responses fixes, ACL + UI attachments

- Image expert: DashScope-style /chat/completions via image_legacy_client; early exit in
  direct_loop when skill_binding_role is image; shared placeholder helper; docs/IMAGE_SPECIALIST_LANE.md.
- Strict attachment ACL: link_attachment_acl on assistant chat_message rows (sqlite_store);
  chat attachment rate limit when user_id empty; admin chat tests updated.
- Admin chat UI: aggregate bubbles render assistant_text attachments (image_ref); WS expand path.
- turn_runner: persisted_chat_attachments_nonempty for final_msg selection.
- OpenAI Responses transport + agent_messages/agent_core_attempt adjustments; env docs and tests.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-05-10 07:06:05 +08:00
parent d1bcc4debe
commit faeb067856
21 changed files with 2516 additions and 331 deletions

View file

@ -58,7 +58,7 @@ class AttachmentAclBackfillTests(unittest.TestCase):
def _h(self) -> dict[str, str]:
return {"authorization": f"Bearer {self.token}"}
def test_backfill_enables_strict_acl_download(self) -> None:
def test_add_message_links_acl_for_strict_download(self) -> None:
store = SqliteStore(str(self.db))
sess = store.create_session_for_user(title="t", tenant_id=self.tenant_id, user_id=self.alice_id)
ast = AttachmentAssetStore()
@ -75,11 +75,12 @@ class AttachmentAclBackfillTests(unittest.TestCase):
prev = os.environ.get("AIA_ATTACHMENT_ACL_STRICT")
os.environ["AIA_ATTACHMENT_ACL_STRICT"] = "1"
try:
# Strict mode: without backfill, this should be forbidden (not avatar).
# Strict mode: add_message should have written attachment_acl (not only tool_result rows).
r0 = self.client.get(f"/admin/api/chat/attachments/{aid}", headers=self._h())
self.assertEqual(r0.status_code, 403, r0.text)
self.assertEqual(r0.status_code, 200, r0.text)
self.assertTrue(len(r0.content) > 10)
# Backfill ACL for tenant, then strict download should succeed.
# Backfill remains idempotent.
res = store.backfill_attachment_acl_from_messages(tenant_id=self.tenant_id, limit_messages=5000)
self.assertTrue(res.get("ok"), res)
r1 = self.client.get(f"/admin/api/chat/attachments/{aid}", headers=self._h())

View file

@ -0,0 +1,60 @@
"""Image specialist lane must use ``AIA_IMAGE_EXPERT_*``, not ``AIA_OCR_*``."""
from __future__ import annotations
from oclaw.platform.llm.image_legacy_client import send_legacy_image_messages
def test_legacy_fails_when_only_ocr_env_configured(monkeypatch: object) -> None:
monkeypatch.setenv("AIA_OCR_API_KEY", "k-ocr")
monkeypatch.setenv("AIA_OCR_BASE_URL", "https://ocr.example/v1")
monkeypatch.setenv("AIA_OCR_MODEL", "ocr-model")
for k in ("AIA_IMAGE_EXPERT_API_KEY", "AIA_IMAGE_EXPERT_BASE_URL", "AIA_IMAGE_EXPERT_MODEL"):
monkeypatch.delenv(k, raising=False)
out = send_legacy_image_messages(
images=["https://example.com/x.png"],
prompt="hi",
model=None,
api_key=None,
base_url=None,
)
assert out.get("ok") is False
err = str(out.get("error") or "")
assert "AIA_IMAGE_EXPERT" in err
def test_legacy_accepts_explicit_kwargs_without_expert_env(monkeypatch: object) -> None:
"""Explicit api_key/base_url/model override env (for scripts/tests only)."""
for k in (
"AIA_OCR_API_KEY",
"AIA_OCR_BASE_URL",
"AIA_IMAGE_EXPERT_API_KEY",
"AIA_IMAGE_EXPERT_BASE_URL",
):
monkeypatch.delenv(k, raising=False)
recorded: dict[str, object] = {}
def fake_post(client, *, url: str, headers: dict[str, str], payload: dict[str, object]) -> object:
recorded["url"] = url
class R:
status_code = 200
def json(_self): # noqa: PLR6301
return {"choices": [{"message": {"content": [{"text": "ok"}]}}]}
return R()
import oclaw.platform.llm.image_legacy_client as mod
monkeypatch.setattr(mod, "post_with_retry", fake_post)
out = send_legacy_image_messages(
images=["https://example.com/x.png"],
prompt="hello",
model="m-special",
api_key="explicit-key",
base_url="https://gw.example/expert/v1",
)
assert out.get("ok") is True
assert "/chat/completions" in str(recorded.get("url") or "")

View file

@ -0,0 +1,248 @@
from __future__ import annotations
from oclaw.platform.llm.image_http_common import (
dashscope_multimodal_http_ok,
dashscope_native_multimodal_url_from_compatible_base,
extract_text_and_images,
)
from oclaw.platform.llm.image_http_common import build_extract_diag_empty
from oclaw.platform.llm.image_legacy_client import (
collect_legacy_lane_images_from_attachments,
legacy_image_assistant_body_with_placeholder,
legacy_image_turn_bundle,
normalize_legacy_output_image_urls,
)
def test_normalize_legacy_output_dict_image_parts() -> None:
u = "https://dashscope-result-sz.oss-cn-shenzhen.aliyuncs.com/x.png?Expires=1"
assert normalize_legacy_output_image_urls([{"image": u}]) == [u]
assert normalize_legacy_output_image_urls([{"image_url": {"url": u}}]) == [u]
def test_legacy_turn_bundle_coerces_dict_images_to_attachments() -> None:
ok, text, att = legacy_image_turn_bundle(
{"ok": True, "text": "", "images": [{"image": "https://example.invalid/a.png"}]}
)
assert ok is True
assert len(att) == 1
assert att[0]["type"] == "image_url"
assert att[0]["url"] == "https://example.invalid/a.png"
def test_dashscope_native_url_from_compatible_base() -> None:
assert (
dashscope_native_multimodal_url_from_compatible_base(
"https://dashscope.aliyuncs.com/compatible-mode/v1"
)
== "https://dashscope.aliyuncs.com/api/v1/services/aigc/multimodal-generation/generation"
)
assert (
dashscope_native_multimodal_url_from_compatible_base("https://example.com/openai/v1") is None
)
def test_collect_legacy_lane_images_image_url() -> None:
atts = [{"type": "image_url", "url": "https://example.invalid/x.png"}]
assert collect_legacy_lane_images_from_attachments(atts) == ["https://example.invalid/x.png"]
def test_collect_legacy_lane_images_raw_base64() -> None:
atts = [{"type": "input_image", "mime": "image/png", "image_base64": "SGVsbG8="}]
got = collect_legacy_lane_images_from_attachments(atts)
assert len(got) == 1
assert got[0].startswith("data:image/png;base64,")
def test_legacy_turn_bundle_text_only_success() -> None:
ok, text, att = legacy_image_turn_bundle({"ok": True, "text": "caption only", "images": []})
assert ok is True
assert text == "caption only"
assert att == []
def test_legacy_image_assistant_placeholder_zh_en() -> None:
produced = [{"type": "image_ref", "attachment_id": "a" * 64}]
assert "附件" in legacy_image_assistant_body_with_placeholder(
lang="zh", body_text="", produced=produced
)
assert "attachment" in legacy_image_assistant_body_with_placeholder(
lang="en", body_text="", produced=produced
).lower()
assert legacy_image_assistant_body_with_placeholder(lang="zh", body_text="x", produced=produced) == "x"
def test_legacy_turn_bundle_upstream_error() -> None:
ok, text, att = legacy_image_turn_bundle({"ok": False, "error": "rate"})
assert ok is False
assert "rate" in text
assert att == []
def test_extract_harvests_nested_https_under_message() -> None:
url = "https://cdn.example.invalid/generated.png"
text, images = extract_text_and_images(
{
"choices": [
{
"message": {
"role": "assistant",
"content": None,
"metadata": {"preview_image": url},
}
}
]
}
)
assert text == ""
assert images == [url]
def test_build_extract_diag_top_level_openai_choices() -> None:
d = build_extract_diag_empty(
{
"choices": [],
"model": "x",
}
)
assert d.get("choices_len") == 0
def test_legacy_turn_bundle_includes_provider_redacted() -> None:
ok, msg, att = legacy_image_turn_bundle(
{
"ok": True,
"text": "",
"images": [],
"extract_diag": {"choices_len": 0},
"provider_response_redacted": '{"choices":[]}',
}
)
assert ok is False
assert "provider_json=" in msg
assert att == []
def test_legacy_turn_bundle_empty_ok_response_fails() -> None:
ok, text, att = legacy_image_turn_bundle({"ok": True, "text": "", "images": []})
assert ok is False
assert att == []
def test_extract_text_and_images_dashscope_output_wrapper() -> None:
"""Native multimodal HTTP wraps ``choices`` under ``output`` (not top-level OpenAI shape)."""
url = "https://dashscope-result-hz.oss-cn-hangzhou.aliyuncs.com/x.png?Expires=1"
payload = {
"status_code": 200,
"request_id": "959afba6-544e-487e-b58a-6bd9fea97xxx",
"code": "",
"message": "",
"output": {
"text": None,
"finish_reason": None,
"choices": [
{
"finish_reason": "stop",
"message": {
"role": "assistant",
"content": [{"image": url}],
},
}
],
"audio": None,
},
"usage": {
"input_tokens": 0,
"output_tokens": 0,
"image_count": 1,
"width": 2048,
"height": 2048,
},
}
text, images = extract_text_and_images(payload)
assert text == ""
assert images == [url]
assert dashscope_multimodal_http_ok(payload)[0] is True
def test_extract_text_and_images_content_dict_not_list() -> None:
"""Some gateways return a single object for ``message.content`` instead of an array."""
url = "https://dashscope-result-hz.oss-cn-hangzhou.aliyuncs.com/out.png"
text, images = extract_text_and_images(
{
"output": {
"choices": [
{
"message": {
"role": "assistant",
"content": {"image": url},
}
}
]
}
}
)
assert text == ""
assert images == [url]
def test_extract_text_and_images_messages_fallback() -> None:
text, images = extract_text_and_images(
{
"output": {
"messages": [
{"role": "user", "content": "x"},
{
"role": "assistant",
"content": [{"image": "https://example.invalid/gen.png"}],
},
]
}
}
)
assert images == ["https://example.invalid/gen.png"]
def test_extract_text_and_images_typed_image_url_part() -> None:
text, images = extract_text_and_images(
{
"choices": [
{
"message": {
"content": [
{
"type": "image_url",
"image_url": {"url": "https://example.invalid/v.png"},
}
]
}
}
]
}
)
assert images == ["https://example.invalid/v.png"]
def test_dashscope_envelope_rejects_non_success_code() -> None:
ok, msg = dashscope_multimodal_http_ok({"code": "InvalidParameter", "message": "bad"})
assert ok is False
assert "bad" in msg
def test_extract_text_and_images_openai_top_level_unchanged() -> None:
text, images = extract_text_and_images(
{
"choices": [
{
"message": {
"content": [
{"type": "text", "text": "hi"},
{"image": "https://example.invalid/a.jpg"},
]
}
}
]
}
)
assert "hi" in text
assert images == ["https://example.invalid/a.jpg"]

View file

@ -0,0 +1,160 @@
from __future__ import annotations
import httpx
from openai import BadRequestError
from oclaw.platform.llm.transports.openai_responses import OpenAIResponsesModel, _is_input_messages_validation_error
def test_input_messages_validation_detects_body_not_str_exc() -> None:
"""OpenAI SDK ``str(exc)`` is usually only ``Error code: 400``; gateway detail is in ``body``."""
req = httpx.Request("POST", "http://example.invalid/v1/responses")
resp = httpx.Response(400, request=req)
body = {
"message": (
"Input should be 'user': input.messages.0.role & Input should be a valid list: "
"input.messages.0.content"
),
"type": "invalid_request_error",
"code": "invalid_parameter_error",
"param": None,
}
exc = BadRequestError("Error code: 400", response=resp, body=body)
assert "input.messages" not in str(exc).lower()
assert _is_input_messages_validation_error(exc) is True
def test_strip_leading_system_to_instructions_kw() -> None:
msgs = [
{"role": "system", "content": "You are helpful."},
{"role": "user", "content": "Hi"},
]
txt, tail = OpenAIResponsesModel._strip_leading_system_messages(msgs)
assert txt == "You are helpful."
assert tail == [{"role": "user", "content": "Hi"}]
def test_normalize_default_responses_parts_and_openai_envelope() -> None:
msgs = [
{
"role": "user",
"content": [
{"type": "image_url", "image_url": {"url": "https://example.invalid/x.png"}},
{"type": "text", "text": "what?"},
],
}
]
out = OpenAIResponsesModel._normalize_messages(msgs)
assert len(out) == 1
assert out[0]["type"] == "message"
assert out[0]["role"] == "user"
cc = out[0]["content"]
assert isinstance(cc, list)
assert any(
x.get("type") == "input_image"
and isinstance(x.get("image_url"), str)
and x.get("detail") == "auto"
for x in cc
)
assert any(x.get("type") == "input_text" and x.get("text") == "what?" for x in cc)
def test_normalize_nested_chat_parts_opt_in() -> None:
msgs = [
{
"role": "user",
"content": [{"type": "image_url", "image_url": {"url": "https://example.invalid/x.png"}}],
}
]
out = OpenAIResponsesModel._normalize_messages(
msgs, envelope_openai_message=False, content_chat_completions_parts=True
)
assert "type" not in out[0]
assert out[0]["role"] == "user"
assert any(x.get("type") == "image_url" for x in out[0]["content"])
def test_normalize_dashscope_shorthand_image_text_blocks() -> None:
msgs = [
{"role": "user", "content": [{"image": "https://example.invalid/y.png"}, {"text": "caption"}]},
]
out = OpenAIResponsesModel._normalize_messages(msgs)
assert out[0]["type"] == "message"
assert out[0]["role"] == "user"
parts = out[0]["content"]
assert isinstance(parts, list)
imgs = [p for p in parts if isinstance(p, dict) and p.get("type") == "input_image"]
assert len(imgs) == 1 and imgs[0].get("detail") == "auto"
txts = [p for p in parts if isinstance(p, dict) and p.get("type") == "input_text"]
assert any("caption" in str(p.get("text")) for p in txts)
def test_responses_input_candidates_cover_flat_and_nested() -> None:
msgs = [{"role": "user", "content": "hi"}]
flat = OpenAIResponsesModel._responses_input_candidates(
msgs, flat_responses=True, prefer_envelope=True, prefer_chat_parts=False
)
assert len(flat) >= 1
nested = OpenAIResponsesModel._responses_input_candidates(
msgs, flat_responses=False, prefer_envelope=True, prefer_chat_parts=False
)
tags = [t for t, _ in nested]
assert any("_messages" in t for t in tags)
assert any("_flat_input" in t for t in tags)
def test_responses_input_candidates_primary_combo_first() -> None:
msgs = [{"role": "user", "content": "hi"}]
nested = OpenAIResponsesModel._responses_input_candidates(
msgs, flat_responses=False, prefer_envelope=True, prefer_chat_parts=False
)
assert nested[0][0] == "e1c0_messages"
assert nested[1][0] == "e1c0_flat_input"
nested2 = OpenAIResponsesModel._responses_input_candidates(
msgs, flat_responses=False, prefer_envelope=False, prefer_chat_parts=True
)
assert nested2[0][0] == "e0c1_messages"
assert nested2[1][0] == "e0c1_flat_input"
def test_normalize_agent_messages_style_input_image() -> None:
"""Matches ``build_llm_messages`` last-turn multimodal blocks (``input_image`` + ``image_base64``)."""
msgs = [
{
"role": "user",
"content": [
{"type": "input_image", "image_base64": "SGk=", "mime": "image/png"},
{"type": "text", "text": "what is this"},
],
}
]
out = OpenAIResponsesModel._normalize_messages(msgs)
assert len(out) == 1
assert out[0]["type"] == "message"
parts = out[0]["content"]
imgs = [p for p in parts if isinstance(p, dict) and p.get("type") == "input_image"]
assert len(imgs) == 1
assert imgs[0]["image_url"].startswith("data:image/png;base64,")
assert imgs[0].get("detail") == "auto"
txts = [p for p in parts if isinstance(p, dict) and p.get("type") == "input_text"]
assert any(p.get("text") == "what is this" for p in txts)
chat_parts = OpenAIResponsesModel._normalize_messages(
msgs, envelope_openai_message=False, content_chat_completions_parts=True
)
cp = chat_parts[0]["content"]
assert any(
isinstance(p, dict) and p.get("type") == "image_url" and "base64" in str(p.get("image_url", {}).get("url"))
for p in cp
)
assert any(p.get("type") == "text" and p.get("text") == "what is this" for p in cp)
def test_image_legacy_compatible_mode_uses_image_url_parts() -> None:
from oclaw.platform.llm.image_legacy_client import _openai_compatible_vision_content
cc = _openai_compatible_vision_content(["data:image/jpeg;base64,SGk="], "go")
assert cc[-1]["type"] == "text"
assert cc[-1]["text"] == "go"
assert cc[0]["type"] == "image_url"