mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 13:00:45 +08:00
feat: multimodal image clients, workspaces, Tushare skill, test fixes
- Replace monolithic image_message_client with HTTP/OCR/legacy modules; tighten OpenAI transport + tool schemas for multimodal downgrade to OCR specialist path. - Add image/stock workspace prompts (META/SOUL/ROLE_SYSTEM); register experts; tweak specialist agent/direct loop/query_image_attachment. - Add bundled runtime/skills/tushare-finance (references, api_client, SKILL metadata). - Document OCR-related env vars; admin chat tweaks; README; weixin_install Ensure-OfficialPluginRuntimeDeps helper. - Tests: multimodal downgrade + OCR coverage, strict tool pairing in attachment replay guard, workspace contract skips _internal/_system dirs, router/trace/prompt guards. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
6307ade480
commit
d1bcc4debe
266 changed files with 57024 additions and 839 deletions
148
platform/llm/image_http_common.py
Normal file
148
platform/llm/image_http_common.py
Normal file
|
|
@ -0,0 +1,148 @@
|
|||
"""Shared HTTP helpers for OCR and legacy image lanes (internal)."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
import os
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def join_url(base: str, path: str) -> str:
|
||||
b = (base or "").rstrip("/")
|
||||
p = (path or "").lstrip("/")
|
||||
return f"{b}/{p}"
|
||||
|
||||
|
||||
def env_ocr_lane_api_key() -> str:
|
||||
return (os.getenv("AIA_OCR_API_KEY") or "").strip()
|
||||
|
||||
|
||||
def env_ocr_lane_base_url() -> str:
|
||||
return (os.getenv("AIA_OCR_BASE_URL") or "").strip()
|
||||
|
||||
|
||||
def env_ocr_lane_model() -> str:
|
||||
return (os.getenv("AIA_OCR_MODEL") or "").strip()
|
||||
|
||||
|
||||
def env_ocr_lane_chat_endpoint() -> str:
|
||||
raw = (os.getenv("AIA_OCR_CHAT_ENDPOINT") or "").strip()
|
||||
return raw or "/chat/completions"
|
||||
|
||||
|
||||
def is_data_url(s: str) -> bool:
|
||||
return s.startswith("data:") and ";base64," in s
|
||||
|
||||
|
||||
def compress_data_url_image(
|
||||
data_url: str,
|
||||
*,
|
||||
max_side: int = 1600,
|
||||
max_bytes: int = 2 * 1024 * 1024,
|
||||
jpeg_quality: int = 82,
|
||||
) -> str:
|
||||
if not is_data_url(data_url):
|
||||
return data_url
|
||||
try:
|
||||
header, b64 = data_url.split(";base64,", 1)
|
||||
mime = header.replace("data:", "", 1).strip() or "image/jpeg"
|
||||
raw = base64.b64decode(b64.encode("ascii"))
|
||||
if len(raw) <= max_bytes:
|
||||
return data_url
|
||||
|
||||
with Image.open(io.BytesIO(raw)) as im:
|
||||
im = im.convert("RGB")
|
||||
w, h = im.size
|
||||
longest = max(w, h)
|
||||
if longest > max_side:
|
||||
scale = max_side / float(longest)
|
||||
nw, nh = max(1, int(w * scale)), max(1, int(h * scale))
|
||||
im = im.resize((nw, nh), Image.Resampling.LANCZOS)
|
||||
|
||||
out = io.BytesIO()
|
||||
im.save(out, format="JPEG", quality=max(30, min(95, int(jpeg_quality))), optimize=True)
|
||||
blob = out.getvalue()
|
||||
if not blob:
|
||||
return data_url
|
||||
b64_new = base64.b64encode(blob).decode("ascii")
|
||||
return f"data:{'image/jpeg' if mime.startswith('image/') else mime};base64,{b64_new}"
|
||||
except Exception:
|
||||
return data_url
|
||||
|
||||
|
||||
def post_with_retry(
|
||||
client: httpx.Client,
|
||||
*,
|
||||
url: str,
|
||||
headers: dict[str, str],
|
||||
payload: dict[str, Any],
|
||||
retries: int = 3,
|
||||
backoff_sec: float = 0.8,
|
||||
) -> httpx.Response:
|
||||
last_exc: Exception | None = None
|
||||
for attempt in range(1, max(1, retries) + 1):
|
||||
try:
|
||||
return client.post(url, headers=headers, json=payload)
|
||||
except (httpx.ReadError, httpx.ConnectError, httpx.TimeoutException) as e:
|
||||
last_exc = e
|
||||
if attempt >= retries:
|
||||
break
|
||||
time.sleep(backoff_sec * (2 ** (attempt - 1)))
|
||||
assert last_exc is not None
|
||||
raise last_exc
|
||||
|
||||
|
||||
def extract_text_and_images(resp_json: dict[str, Any]) -> tuple[str, list[str]]:
|
||||
text_parts: list[str] = []
|
||||
images: list[str] = []
|
||||
choices = resp_json.get("choices")
|
||||
if isinstance(choices, list) and choices:
|
||||
msg = choices[0].get("message") if isinstance(choices[0], dict) else None
|
||||
if isinstance(msg, dict):
|
||||
c = msg.get("content")
|
||||
if isinstance(c, str):
|
||||
text_parts.append(c)
|
||||
elif isinstance(c, list):
|
||||
for it in c:
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
if isinstance(it.get("text"), str):
|
||||
text_parts.append(str(it.get("text")))
|
||||
elif isinstance(it.get("image"), str):
|
||||
images.append(str(it.get("image")))
|
||||
elif isinstance(it.get("image_url"), str):
|
||||
images.append(str(it.get("image_url")))
|
||||
elif isinstance(it.get("image_url"), dict) and isinstance(it["image_url"].get("url"), str):
|
||||
images.append(str(it["image_url"]["url"]))
|
||||
elif isinstance(it.get("b64_json"), str):
|
||||
images.append(f"data:image/png;base64,{it['b64_json']}")
|
||||
|
||||
if not images:
|
||||
data = resp_json.get("data")
|
||||
if isinstance(data, list):
|
||||
for it in data:
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
if isinstance(it.get("url"), str):
|
||||
images.append(str(it.get("url")))
|
||||
if isinstance(it.get("b64_json"), str):
|
||||
images.append(f"data:image/png;base64,{it['b64_json']}")
|
||||
return "\n".join([x for x in text_parts if x]).strip(), images
|
||||
|
||||
|
||||
__all__ = [
|
||||
"compress_data_url_image",
|
||||
"env_ocr_lane_api_key",
|
||||
"env_ocr_lane_base_url",
|
||||
"env_ocr_lane_chat_endpoint",
|
||||
"env_ocr_lane_model",
|
||||
"extract_text_and_images",
|
||||
"is_data_url",
|
||||
"join_url",
|
||||
"post_with_retry",
|
||||
]
|
||||
146
platform/llm/image_legacy_client.py
Normal file
146
platform/llm/image_legacy_client.py
Normal file
|
|
@ -0,0 +1,146 @@
|
|||
"""Legacy image+text payloads for DashScope-style gateways.
|
||||
|
||||
Uses ``{"image":...}/{"text":...}`` or typed compatible-mode blocks on ``/chat/completions`` only.
|
||||
Prefer :mod:`oclaw.platform.llm.image_ocr_client` for OpenAI-compatible vision (图片专家已改用该路径).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from oclaw.platform.llm.image_http_common import (
|
||||
compress_data_url_image,
|
||||
env_ocr_lane_api_key,
|
||||
env_ocr_lane_base_url,
|
||||
env_ocr_lane_model,
|
||||
env_ocr_lane_chat_endpoint,
|
||||
extract_text_and_images,
|
||||
is_data_url,
|
||||
join_url,
|
||||
post_with_retry,
|
||||
)
|
||||
from oclaw.runtime.prompt_templates import render_prompt
|
||||
|
||||
|
||||
def _http_content_blocks(images: list[str], prompt: str, *, typed: bool) -> list[dict[str, Any]]:
|
||||
prompt_text = str(prompt or "").strip() or render_prompt("image/default_edit_prompt.zh.md", strict=True)
|
||||
if not typed:
|
||||
blocks: list[dict[str, Any]] = [{"image": img} for img in images]
|
||||
blocks.append({"text": prompt_text})
|
||||
return blocks
|
||||
blocks_typed: list[dict[str, Any]] = [{"type": "image", "image": img} for img in images]
|
||||
blocks_typed.append({"type": "text", "text": prompt_text})
|
||||
return blocks_typed
|
||||
|
||||
|
||||
def send_legacy_image_messages(
|
||||
*,
|
||||
images: list[str],
|
||||
prompt: str,
|
||||
model: str | None = None,
|
||||
timeout_sec: int = 60,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Legacy multimodal HTTP (non--OpenAI-``image_url`` schema). Optional; specialists use OCR client."""
|
||||
resolved_base_url = (base_url or env_ocr_lane_base_url()).strip()
|
||||
resolved_api_key = (api_key or env_ocr_lane_api_key()).strip()
|
||||
model_name = ((model or "").strip() or env_ocr_lane_model())
|
||||
endpoint = env_ocr_lane_chat_endpoint()
|
||||
url = join_url(resolved_base_url, endpoint)
|
||||
if not resolved_api_key or not resolved_base_url:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": "missing AIA_OCR_API_KEY or AIA_OCR_BASE_URL (or pass api_key and base_url)",
|
||||
}
|
||||
if not model_name:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": "missing AIA_OCR_MODEL (or pass model=...) — no default model id",
|
||||
}
|
||||
if not images:
|
||||
return {"ok": False, "error": "at least one image input is required"}
|
||||
|
||||
raw_selected = [str(x).strip() for x in images if str(x).strip()][:3]
|
||||
selected: list[str] = []
|
||||
input_kind: list[str] = []
|
||||
for img in raw_selected:
|
||||
if is_data_url(img):
|
||||
selected.append(compress_data_url_image(img))
|
||||
input_kind.append("data_url")
|
||||
continue
|
||||
if img.startswith("http://") or img.startswith("https://"):
|
||||
selected.append(img)
|
||||
input_kind.append("url")
|
||||
continue
|
||||
if not selected:
|
||||
return {"ok": False, "error": "no usable image input (expected URL or data URL)"}
|
||||
|
||||
prefer_typed_http = "compatible-mode" in resolved_base_url.lower()
|
||||
content_multi = _http_content_blocks(selected, prompt, typed=prefer_typed_http)
|
||||
content_multi_fallback = _http_content_blocks(selected, prompt, typed=not prefer_typed_http)
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {resolved_api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload_multi = {
|
||||
"model": model_name,
|
||||
"messages": [{"role": "user", "content": content_multi}],
|
||||
}
|
||||
payload_single = {
|
||||
"model": model_name,
|
||||
"messages": [{"role": "user", "content": content_multi_fallback[-2:] if len(content_multi_fallback) >= 2 else content_multi_fallback}],
|
||||
}
|
||||
|
||||
with httpx.Client(timeout=float(timeout_sec)) as client:
|
||||
try:
|
||||
r = post_with_retry(client, url=url, headers=headers, payload=payload_multi)
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"http request failed: {type(e).__name__}: {e}", "backend_shape": "multi"}
|
||||
|
||||
if r.status_code >= 400:
|
||||
try:
|
||||
r2 = post_with_retry(client, url=url, headers=headers, payload=payload_single)
|
||||
except Exception as e:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": f"http fallback request failed: {type(e).__name__}: {e}",
|
||||
"backend_shape": "single-fallback-failed",
|
||||
}
|
||||
if r2.status_code >= 400:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": f"http {r2.status_code}: {r2.text[:500]}",
|
||||
"backend_shape": "single-fallback-failed",
|
||||
}
|
||||
try:
|
||||
obj2 = r2.json()
|
||||
except Exception:
|
||||
return {"ok": False, "error": f"non-json response: {r2.text[:500]}", "backend_shape": "single"}
|
||||
text, out_images = extract_text_and_images(obj2 if isinstance(obj2, dict) else {})
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text,
|
||||
"images": out_images,
|
||||
"backend_shape": "single",
|
||||
"input_kind": input_kind[-1:] if input_kind else ["data_url"],
|
||||
}
|
||||
|
||||
try:
|
||||
obj = r.json()
|
||||
except Exception:
|
||||
return {"ok": False, "error": f"non-json response: {r.text[:500]}", "backend_shape": "multi"}
|
||||
text, out_images = extract_text_and_images(obj if isinstance(obj, dict) else {})
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text,
|
||||
"images": out_images,
|
||||
"backend_shape": "multi",
|
||||
"input_kind": input_kind if input_kind else ["data_url"],
|
||||
}
|
||||
|
||||
|
||||
__all__ = ["send_legacy_image_messages"]
|
||||
|
|
@ -1,685 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
import os
|
||||
import time
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
from PIL import Image
|
||||
from oclaw.runtime.prompt_templates import render_prompt
|
||||
|
||||
|
||||
def _summarize_data_url(url: str) -> dict[str, Any]:
|
||||
"""
|
||||
Summarize data URLs without logging raw base64 payload.
|
||||
"""
|
||||
s = str(url or "")
|
||||
if not s.startswith("data:"):
|
||||
return {"kind": "url", "prefix": s[:30]}
|
||||
mime = "image/jpeg"
|
||||
try:
|
||||
# data:{mime};base64,{b64}
|
||||
head = s.split(";", 1)[0] # data:image/xxx
|
||||
if head.startswith("data:"):
|
||||
mime = head[len("data:") :] or mime
|
||||
except Exception:
|
||||
pass
|
||||
b64_part = s.split(",", 1)[1] if "," in s else ""
|
||||
has_b64 = ";base64," in s
|
||||
return {"kind": "data_url", "mime": mime, "has_base64": has_b64, "b64_len": len(b64_part)}
|
||||
|
||||
|
||||
def _summarize_dashscope_messages(messages: list[dict[str, Any]], *, max_items: int = 3) -> dict[str, Any]:
|
||||
"""
|
||||
Debug-only summary of messages[0].content structure.
|
||||
"""
|
||||
if not messages or not isinstance(messages, list):
|
||||
return {"content_items": []}
|
||||
first = messages[0] if messages else {}
|
||||
if not isinstance(first, dict):
|
||||
return {"content_items": []}
|
||||
content = first.get("content")
|
||||
if not isinstance(content, list):
|
||||
return {"content_items": []}
|
||||
out: list[dict[str, Any]] = []
|
||||
for it in content[:max_items]:
|
||||
if not isinstance(it, dict):
|
||||
out.append({"item_type": "non_dict"})
|
||||
continue
|
||||
item: dict[str, Any] = {"keys": sorted([str(k) for k in it.keys()])}
|
||||
if "type" in it:
|
||||
item["type"] = it.get("type")
|
||||
if "image" in it:
|
||||
item["image"] = _summarize_data_url(str(it.get("image") or ""))
|
||||
if "image_url" in it:
|
||||
iu = it.get("image_url")
|
||||
if isinstance(iu, dict) and "url" in iu:
|
||||
item["image_url"] = _summarize_data_url(str(iu.get("url") or ""))
|
||||
else:
|
||||
item["image_url"] = {"kind": "image_url_non_dict"}
|
||||
if "text" in it:
|
||||
item["text_len"] = len(str(it.get("text") or ""))
|
||||
out.append(item)
|
||||
return {"content_items": out}
|
||||
|
||||
|
||||
def _join_url(base: str, path: str) -> str:
|
||||
b = (base or "").rstrip("/")
|
||||
p = (path or "").lstrip("/")
|
||||
return f"{b}/{p}"
|
||||
|
||||
|
||||
def _is_data_url(s: str) -> bool:
|
||||
return s.startswith("data:") and ";base64," in s
|
||||
|
||||
|
||||
def _compress_data_url_image(
|
||||
data_url: str,
|
||||
*,
|
||||
max_side: int = 1600,
|
||||
max_bytes: int = 2 * 1024 * 1024,
|
||||
jpeg_quality: int = 82,
|
||||
) -> str:
|
||||
"""Best-effort compress image data URLs to reduce transport failures."""
|
||||
if not _is_data_url(data_url):
|
||||
return data_url
|
||||
try:
|
||||
header, b64 = data_url.split(";base64,", 1)
|
||||
mime = header.replace("data:", "", 1).strip() or "image/jpeg"
|
||||
raw = base64.b64decode(b64.encode("ascii"))
|
||||
if len(raw) <= max_bytes:
|
||||
return data_url
|
||||
|
||||
with Image.open(io.BytesIO(raw)) as im:
|
||||
im = im.convert("RGB")
|
||||
w, h = im.size
|
||||
longest = max(w, h)
|
||||
if longest > max_side:
|
||||
scale = max_side / float(longest)
|
||||
nw, nh = max(1, int(w * scale)), max(1, int(h * scale))
|
||||
im = im.resize((nw, nh), Image.Resampling.LANCZOS)
|
||||
|
||||
out = io.BytesIO()
|
||||
# Use JPEG for broad compatibility and smaller payload.
|
||||
im.save(out, format="JPEG", quality=max(30, min(95, int(jpeg_quality))), optimize=True)
|
||||
blob = out.getvalue()
|
||||
if not blob:
|
||||
return data_url
|
||||
b64_new = base64.b64encode(blob).decode("ascii")
|
||||
return f"data:{'image/jpeg' if mime.startswith('image/') else mime};base64,{b64_new}"
|
||||
except Exception:
|
||||
return data_url
|
||||
|
||||
|
||||
def _post_with_retry(
|
||||
client: httpx.Client,
|
||||
*,
|
||||
url: str,
|
||||
headers: dict[str, str],
|
||||
payload: dict[str, Any],
|
||||
retries: int = 3,
|
||||
backoff_sec: float = 0.8,
|
||||
) -> httpx.Response:
|
||||
last_exc: Exception | None = None
|
||||
for attempt in range(1, max(1, retries) + 1):
|
||||
try:
|
||||
return client.post(url, headers=headers, json=payload)
|
||||
except (httpx.ReadError, httpx.ConnectError, httpx.TimeoutException) as e:
|
||||
last_exc = e
|
||||
if attempt >= retries:
|
||||
break
|
||||
time.sleep(backoff_sec * (2 ** (attempt - 1)))
|
||||
assert last_exc is not None
|
||||
raise last_exc
|
||||
|
||||
|
||||
def _extract_text_and_images(resp_json: dict[str, Any]) -> tuple[str, list[str]]:
|
||||
text_parts: list[str] = []
|
||||
images: list[str] = []
|
||||
choices = resp_json.get("choices")
|
||||
if isinstance(choices, list) and choices:
|
||||
msg = choices[0].get("message") if isinstance(choices[0], dict) else None
|
||||
if isinstance(msg, dict):
|
||||
c = msg.get("content")
|
||||
if isinstance(c, str):
|
||||
text_parts.append(c)
|
||||
elif isinstance(c, list):
|
||||
for it in c:
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
if isinstance(it.get("text"), str):
|
||||
text_parts.append(str(it.get("text")))
|
||||
elif isinstance(it.get("image"), str):
|
||||
images.append(str(it.get("image")))
|
||||
elif isinstance(it.get("image_url"), str):
|
||||
images.append(str(it.get("image_url")))
|
||||
elif isinstance(it.get("image_url"), dict) and isinstance(it["image_url"].get("url"), str):
|
||||
images.append(str(it["image_url"]["url"]))
|
||||
elif isinstance(it.get("b64_json"), str):
|
||||
images.append(f"data:image/png;base64,{it['b64_json']}")
|
||||
|
||||
# fallback for image APIs that return data[0].b64_json/url
|
||||
if not images:
|
||||
data = resp_json.get("data")
|
||||
if isinstance(data, list):
|
||||
for it in data:
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
if isinstance(it.get("url"), str):
|
||||
images.append(str(it.get("url")))
|
||||
if isinstance(it.get("b64_json"), str):
|
||||
images.append(f"data:image/png;base64,{it['b64_json']}")
|
||||
return "\n".join([x for x in text_parts if x]).strip(), images
|
||||
|
||||
|
||||
def _extract_from_dashscope_response(resp: Any) -> tuple[str, list[str], str]:
|
||||
"""Parse dashscope SDK response object (or dict-like) into text/images."""
|
||||
text_parts: list[str] = []
|
||||
images: list[str] = []
|
||||
|
||||
def _as_dict(x: Any) -> dict[str, Any]:
|
||||
if isinstance(x, dict):
|
||||
return x
|
||||
if hasattr(x, "__dict__"):
|
||||
try:
|
||||
return dict(vars(x))
|
||||
except Exception:
|
||||
return {}
|
||||
return {}
|
||||
|
||||
if isinstance(resp, dict):
|
||||
obj = resp
|
||||
else:
|
||||
obj = _as_dict(resp)
|
||||
if not obj and hasattr(resp, "to_dict"):
|
||||
try:
|
||||
obj = resp.to_dict()
|
||||
except Exception:
|
||||
obj = {}
|
||||
|
||||
status_code = int(obj.get("status_code") or getattr(resp, "status_code", 0) or 0)
|
||||
output = obj.get("output") if isinstance(obj.get("output"), dict) else _as_dict(getattr(resp, "output", {}))
|
||||
choices = output.get("choices") if isinstance(output, dict) else None
|
||||
if isinstance(choices, list) and choices:
|
||||
msg = choices[0].get("message") if isinstance(choices[0], dict) else _as_dict(getattr(choices[0], "message", {}))
|
||||
content = msg.get("content")
|
||||
if isinstance(content, list):
|
||||
for it in content:
|
||||
# DashScope SDK sometimes returns content items as objects, not plain dicts.
|
||||
# Convert best-effort so we don't silently drop images.
|
||||
it_d: dict[str, Any]
|
||||
if isinstance(it, dict):
|
||||
it_d = it
|
||||
else:
|
||||
it_d = _as_dict(it)
|
||||
if not it_d:
|
||||
continue
|
||||
if isinstance(it_d.get("text"), str):
|
||||
text_parts.append(str(it_d["text"]))
|
||||
# image can be str/dict/list depending on SDK variant
|
||||
img_val = it_d.get("image")
|
||||
if isinstance(img_val, str):
|
||||
images.append(img_val)
|
||||
elif isinstance(img_val, dict):
|
||||
if isinstance(img_val.get("url"), str):
|
||||
images.append(str(img_val["url"]))
|
||||
elif isinstance(img_val, list):
|
||||
for elem in img_val:
|
||||
if isinstance(elem, str):
|
||||
images.append(elem)
|
||||
elif isinstance(elem, dict) and isinstance(elem.get("url"), str):
|
||||
images.append(str(elem["url"]))
|
||||
|
||||
if isinstance(it_d.get("image_url"), str):
|
||||
images.append(str(it_d["image_url"]))
|
||||
elif isinstance(it_d.get("image_url"), dict) and isinstance(it_d["image_url"].get("url"), str):
|
||||
images.append(str(it_d["image_url"]["url"]))
|
||||
elif isinstance(it_d.get("image_url"), list):
|
||||
for elem in it_d["image_url"]:
|
||||
if isinstance(elem, str):
|
||||
images.append(elem)
|
||||
elif isinstance(elem, dict) and isinstance(elem.get("url"), str):
|
||||
images.append(str(elem["url"]))
|
||||
|
||||
b64_val = it_d.get("b64_json")
|
||||
if isinstance(b64_val, str):
|
||||
images.append(f"data:image/png;base64,{b64_val}")
|
||||
elif isinstance(b64_val, list):
|
||||
for elem in b64_val:
|
||||
if isinstance(elem, str):
|
||||
images.append(f"data:image/png;base64,{elem}")
|
||||
|
||||
# fallback for variants returning output.data[{url|b64_json}]
|
||||
if not images and isinstance(output, dict):
|
||||
data = output.get("data")
|
||||
if isinstance(data, list):
|
||||
for it in data:
|
||||
it_d: dict[str, Any]
|
||||
if isinstance(it, dict):
|
||||
it_d = it
|
||||
else:
|
||||
it_d = _as_dict(it)
|
||||
if not it_d:
|
||||
continue
|
||||
url_val = it_d.get("url")
|
||||
if isinstance(url_val, str):
|
||||
images.append(url_val)
|
||||
elif isinstance(url_val, list):
|
||||
for elem in url_val:
|
||||
if isinstance(elem, str):
|
||||
images.append(elem)
|
||||
|
||||
b64_val = it_d.get("b64_json")
|
||||
if isinstance(b64_val, str):
|
||||
images.append(f"data:image/png;base64,{b64_val}")
|
||||
elif isinstance(b64_val, list):
|
||||
for elem in b64_val:
|
||||
if isinstance(elem, str):
|
||||
images.append(f"data:image/png;base64,{elem}")
|
||||
|
||||
return "\n".join([x for x in text_parts if x]).strip(), images, str(status_code)
|
||||
|
||||
|
||||
def _dashscope_messages(images: list[str], prompt: str) -> list[dict[str, Any]]:
|
||||
prompt_text = str(prompt or "").strip() or render_prompt("image/default_edit_prompt.zh.md", strict=True)
|
||||
content: list[dict[str, Any]] = [{"image": img} for img in images]
|
||||
content.append({"text": prompt_text})
|
||||
return [{"role": "user", "content": content}]
|
||||
|
||||
|
||||
def _http_content_blocks(images: list[str], prompt: str, *, typed: bool) -> list[dict[str, Any]]:
|
||||
prompt_text = str(prompt or "").strip() or render_prompt("image/default_edit_prompt.zh.md", strict=True)
|
||||
if not typed:
|
||||
blocks: list[dict[str, Any]] = [{"image": img} for img in images]
|
||||
blocks.append({"text": prompt_text})
|
||||
return blocks
|
||||
# DashScope compatible-mode error is very explicit about `content[*].type` missing.
|
||||
# To align with DashScope's simple schema keys ({"image": ...}, {"text": ...}),
|
||||
# we keep those keys and only add the `type` discriminator.
|
||||
blocks_typed: list[dict[str, Any]] = [{"type": "image", "image": img} for img in images]
|
||||
blocks_typed.append({"type": "text", "text": prompt_text})
|
||||
return blocks_typed
|
||||
|
||||
|
||||
def _responses_input_messages(images: list[str], prompt: str) -> list[dict[str, Any]]:
|
||||
prompt_text = str(prompt or "").strip() or render_prompt("image/default_edit_prompt.zh.md", strict=True)
|
||||
content: list[dict[str, Any]] = []
|
||||
for img in images:
|
||||
s = str(img or "").strip()
|
||||
if not s:
|
||||
continue
|
||||
if s.startswith("data:") or s.startswith("http://") or s.startswith("https://"):
|
||||
content.append({"type": "input_image", "image_url": s})
|
||||
if prompt_text:
|
||||
content.append({"type": "input_text", "text": prompt_text})
|
||||
return [{"role": "user", "content": content}]
|
||||
|
||||
|
||||
def _dashscope_typed_messages(images: list[str], prompt: str) -> list[dict[str, Any]]:
|
||||
"""
|
||||
DashScope compatible-mode often requires a typed schema, where each content item has `type`.
|
||||
Example:
|
||||
{"type":"image_url","image_url":{"url":"..."}}, {"type":"text","text":"..."}
|
||||
"""
|
||||
typed_blocks = _http_content_blocks(images, prompt, typed=True)
|
||||
return [{"role": "user", "content": typed_blocks}]
|
||||
|
||||
|
||||
def _send_via_dashscope(
|
||||
*,
|
||||
images: list[str],
|
||||
prompt: str,
|
||||
model_name: str,
|
||||
api_key: str | None = None,
|
||||
base_http_api_url: str | None = None,
|
||||
force_typed_schema: bool = False,
|
||||
) -> dict[str, Any]:
|
||||
try:
|
||||
import dashscope # type: ignore
|
||||
from dashscope import MultiModalConversation # type: ignore
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"dashscope package unavailable: {type(e).__name__}: {e}"}
|
||||
|
||||
resolved_key = (api_key or os.getenv("DASHSCOPE_API_KEY") or "").strip()
|
||||
if not resolved_key:
|
||||
return {"ok": False, "error": "missing DASHSCOPE_API_KEY"}
|
||||
|
||||
ds_base = (base_http_api_url or os.getenv("DASHSCOPE_BASE_HTTP_API_URL") or "").strip()
|
||||
if ds_base:
|
||||
dashscope.base_http_api_url = ds_base
|
||||
|
||||
call_kwargs_base: dict[str, Any] = {
|
||||
"api_key": resolved_key,
|
||||
"model": model_name,
|
||||
"stream": False,
|
||||
}
|
||||
# Optional params compatible with user's sample
|
||||
if os.getenv("DASHSCOPE_IMAGE_N"):
|
||||
try:
|
||||
call_kwargs_base["n"] = int(os.getenv("DASHSCOPE_IMAGE_N") or "1")
|
||||
except Exception:
|
||||
pass
|
||||
if os.getenv("DASHSCOPE_IMAGE_WATERMARK"):
|
||||
call_kwargs_base["watermark"] = os.getenv("DASHSCOPE_IMAGE_WATERMARK", "false").lower() in ("1", "true", "yes")
|
||||
if os.getenv("DASHSCOPE_IMAGE_NEGATIVE_PROMPT"):
|
||||
call_kwargs_base["negative_prompt"] = os.getenv("DASHSCOPE_IMAGE_NEGATIVE_PROMPT")
|
||||
if os.getenv("DASHSCOPE_IMAGE_PROMPT_EXTEND"):
|
||||
call_kwargs_base["prompt_extend"] = os.getenv("DASHSCOPE_IMAGE_PROMPT_EXTEND", "true").lower() in ("1", "true", "yes")
|
||||
if os.getenv("DASHSCOPE_IMAGE_SIZE"):
|
||||
call_kwargs_base["size"] = os.getenv("DASHSCOPE_IMAGE_SIZE")
|
||||
|
||||
def _call_with_retry(messages: list[dict[str, Any]]) -> tuple[Any | None, Exception | None]:
|
||||
retries = max(1, int(os.getenv("AIA_IMAGE_RETRIES", "3") or "3"))
|
||||
backoff_sec = float(os.getenv("AIA_IMAGE_RETRY_BACKOFF_SEC", "0.8") or "0.8")
|
||||
last_exc: Exception | None = None
|
||||
resp_obj: Any | None = None
|
||||
for attempt in range(1, retries + 1):
|
||||
try:
|
||||
kwargs = dict(call_kwargs_base)
|
||||
kwargs["messages"] = messages
|
||||
resp_obj = MultiModalConversation.call(**kwargs)
|
||||
return resp_obj, None
|
||||
except Exception as e:
|
||||
last_exc = e
|
||||
if attempt >= retries:
|
||||
break
|
||||
time.sleep(backoff_sec * (2 ** (attempt - 1)))
|
||||
return None, last_exc
|
||||
|
||||
# DashScope SDK path:
|
||||
# Compatible-mode may require typed content schema (`content[*].type`),
|
||||
# so if we detect compatible-mode, prefer typed schema directly.
|
||||
typed_msgs = _dashscope_typed_messages(images, prompt)
|
||||
simple_msgs = _dashscope_messages(images, prompt)
|
||||
|
||||
first_msgs = typed_msgs if force_typed_schema else simple_msgs
|
||||
second_msgs = simple_msgs if force_typed_schema else typed_msgs
|
||||
|
||||
first_schema = "typed" if first_msgs is typed_msgs else "simple"
|
||||
second_schema = "simple" if first_schema == "typed" else "typed"
|
||||
debug_first = _summarize_dashscope_messages(first_msgs)
|
||||
debug_second = _summarize_dashscope_messages(second_msgs)
|
||||
|
||||
used_schema = first_schema
|
||||
used_debug = debug_first
|
||||
|
||||
resp, err = _call_with_retry(first_msgs)
|
||||
if resp is None:
|
||||
# If the first schema fails without a response, try the other schema once.
|
||||
resp2, err2 = _call_with_retry(second_msgs)
|
||||
if resp2 is None:
|
||||
assert err2 is not None
|
||||
return {
|
||||
"ok": False,
|
||||
"error": (
|
||||
f"dashscope call failed: {type(err2).__name__}: {err2} "
|
||||
f"(debug={{first_schema={first_schema},second_schema={second_schema}}})"
|
||||
)[:900],
|
||||
}
|
||||
resp = resp2
|
||||
used_schema = second_schema
|
||||
used_debug = debug_second
|
||||
|
||||
text, out_images, status = _extract_from_dashscope_response(resp)
|
||||
try:
|
||||
status_code = int(status)
|
||||
except Exception:
|
||||
status_code = 0
|
||||
if status_code and status_code != 200:
|
||||
# DashScope may return HTTP 429 (rate limit) as a non-exception status.
|
||||
# _call_with_retry only retries on exceptions, so handle 429 explicitly here.
|
||||
if status_code == 429:
|
||||
status_retries = max(1, int(os.getenv("AIA_IMAGE_STATUS_RETRIES", "4") or "4"))
|
||||
status_backoff_sec = float(os.getenv("AIA_IMAGE_STATUS_RETRY_BACKOFF_SEC", "5.0") or "5.0")
|
||||
for attempt in range(1, status_retries):
|
||||
time.sleep(status_backoff_sec * (2 ** (attempt - 1)))
|
||||
resp_retry, _err_retry = _call_with_retry(simple_msgs)
|
||||
if resp_retry is None:
|
||||
continue
|
||||
text_r, out_images_r, status_r = _extract_from_dashscope_response(resp_retry)
|
||||
try:
|
||||
status_r_code = int(status_r)
|
||||
except Exception:
|
||||
status_r_code = 0
|
||||
if status_r_code == 200 or not status_r_code:
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text_r,
|
||||
"images": out_images_r,
|
||||
"backend_shape": "dashscope",
|
||||
}
|
||||
if status_r_code == 400:
|
||||
# If we hit schema issues after rate limit, fall back to typed once.
|
||||
typed_msgs = _dashscope_typed_messages(images, prompt)
|
||||
resp_t, _err_t = _call_with_retry(typed_msgs)
|
||||
if resp_t is not None:
|
||||
text_t, out_images_t, status_t = _extract_from_dashscope_response(resp_t)
|
||||
try:
|
||||
status_t_code = int(status_t)
|
||||
except Exception:
|
||||
status_t_code = 0
|
||||
if status_t_code == 200 or not status_t_code:
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text_t,
|
||||
"images": out_images_t,
|
||||
"backend_shape": "dashscope",
|
||||
}
|
||||
# continue loop for more 429 tries
|
||||
code = getattr(resp, "code", "") or ""
|
||||
msg = getattr(resp, "message", "") or ""
|
||||
return {
|
||||
"ok": False,
|
||||
"error": (
|
||||
f"dashscope status={status_code} code={code} debug={{used_schema={used_schema},used_debug={used_debug}}} "
|
||||
f"message={msg[:400]} (after 429 retries)"
|
||||
),
|
||||
"backend_shape": "dashscope",
|
||||
}
|
||||
|
||||
# Compatible-mode may reject "simple" schema with HTTP 400 and a message like:
|
||||
# messages[0].content[1].type missing required parameter.
|
||||
# To make this robust, retry typed schema whenever we see 400.
|
||||
if status_code == 400:
|
||||
resp2, _err2 = _call_with_retry(typed_msgs)
|
||||
if resp2 is not None:
|
||||
text2, out_images2, status2 = _extract_from_dashscope_response(resp2)
|
||||
try:
|
||||
status2_code = int(status2)
|
||||
except Exception:
|
||||
status2_code = 0
|
||||
if status2_code == 200 or not status2_code:
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text2,
|
||||
"images": out_images2,
|
||||
"backend_shape": "dashscope",
|
||||
}
|
||||
|
||||
code = getattr(resp, "code", "") or ""
|
||||
msg = getattr(resp, "message", "") or ""
|
||||
return {
|
||||
"ok": False,
|
||||
"error": (
|
||||
f"dashscope status={status_code} code={code} debug={{used_schema={used_schema},used_debug={used_debug}}} "
|
||||
f"message={msg[:400]}"
|
||||
),
|
||||
"backend_shape": "dashscope",
|
||||
}
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text,
|
||||
"images": out_images,
|
||||
"backend_shape": "dashscope",
|
||||
"debug_used_schema": used_schema,
|
||||
"debug_used_debug": used_debug,
|
||||
}
|
||||
|
||||
|
||||
def send_image_messages(
|
||||
*,
|
||||
images: list[str],
|
||||
prompt: str,
|
||||
model: str | None = None,
|
||||
timeout_sec: int = 60,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
dashscope_api_key: str | None = None,
|
||||
dashscope_base_http_api_url: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Send multimodal messages payload with image/text content blocks.
|
||||
|
||||
Preferred payload: {"messages":[{"role":"user","content":[{"image":"..."},{"text":"..."}]}]}
|
||||
Auto-compat: try multi-image first, then fallback to single `image` field style.
|
||||
"""
|
||||
model_name = (
|
||||
model
|
||||
or os.getenv("AIA_IMAGE_MODEL")
|
||||
or os.getenv("DASHSCOPE_IMAGE_MODEL")
|
||||
or os.getenv("OPENAI_MODEL")
|
||||
or "qwen-image-2.0-pro"
|
||||
).strip()
|
||||
resolved_base_url = (base_url or os.getenv("AIA_IMAGE_BASE_URL") or os.getenv("OPENAI_BASE_URL") or "https://api.openai.com/v1").strip()
|
||||
resolved_api_key = (api_key or os.getenv("AIA_IMAGE_API_KEY") or os.getenv("OPENAI_API_KEY") or "").strip()
|
||||
resolved_ds_key = (dashscope_api_key or os.getenv("DASHSCOPE_API_KEY") or "").strip()
|
||||
resolved_ds_base = (dashscope_base_http_api_url or os.getenv("DASHSCOPE_BASE_HTTP_API_URL") or "").strip()
|
||||
endpoint = (os.getenv("AIA_IMAGE_CHAT_ENDPOINT") or "/chat/completions").strip()
|
||||
url = _join_url(resolved_base_url, endpoint)
|
||||
if not images:
|
||||
return {"ok": False, "error": "at least one image input is required"}
|
||||
|
||||
raw_selected = [str(x).strip() for x in images if str(x).strip()][:3]
|
||||
selected: list[str] = []
|
||||
input_kind: list[str] = []
|
||||
for img in raw_selected:
|
||||
if _is_data_url(img):
|
||||
selected.append(_compress_data_url_image(img))
|
||||
input_kind.append("data_url")
|
||||
continue
|
||||
if img.startswith("http://") or img.startswith("https://"):
|
||||
# Keep URL as-is when caller provides URL input.
|
||||
selected.append(img)
|
||||
input_kind.append("url")
|
||||
continue
|
||||
if not selected:
|
||||
return {"ok": False, "error": "no usable image input (expected URL or data URL)"}
|
||||
|
||||
if resolved_ds_key:
|
||||
# SpecialistAgent already normalizes DashScope SDK calls to native `/api/v1`.
|
||||
# For the SDK path, prefer the official simple schema for both single and
|
||||
# multi-image requests: {"image": ...}, {"text": ...}.
|
||||
#
|
||||
# This avoids a split where multi-image succeeds but single-image still
|
||||
# trips over `messages[0].content[*].type` handling.
|
||||
ds = _send_via_dashscope(
|
||||
images=selected,
|
||||
prompt=prompt,
|
||||
model_name=model_name,
|
||||
api_key=resolved_ds_key,
|
||||
base_http_api_url=resolved_ds_base,
|
||||
force_typed_schema=False,
|
||||
)
|
||||
ds["input_kind"] = input_kind if input_kind else ["data_url"]
|
||||
# When DashScope credentials are present, always return the DashScope path result.
|
||||
# Do not silently fall through to the OpenAI-compatible HTTP path, because that can
|
||||
# mask the real SDK error with a secondary `content[*].type` error from another backend path.
|
||||
return ds
|
||||
|
||||
if not resolved_api_key:
|
||||
return {"ok": False, "error": "missing api key (AIA_IMAGE_API_KEY/OPENAI_API_KEY or DASHSCOPE_API_KEY)"}
|
||||
prefer_typed_http = "compatible-mode" in resolved_base_url.lower()
|
||||
content_multi = _http_content_blocks(selected, prompt, typed=prefer_typed_http)
|
||||
content_multi_fallback = _http_content_blocks(selected, prompt, typed=not prefer_typed_http)
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {resolved_api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload_multi = {
|
||||
"model": model_name,
|
||||
"messages": [{"role": "user", "content": content_multi}],
|
||||
}
|
||||
payload_single = {
|
||||
"model": model_name,
|
||||
"messages": [{"role": "user", "content": content_multi_fallback[-2:] if len(content_multi_fallback) >= 2 else content_multi_fallback}],
|
||||
}
|
||||
responses_payload = {
|
||||
"model": model_name,
|
||||
"input": {"messages": _responses_input_messages(selected, prompt)},
|
||||
}
|
||||
|
||||
with httpx.Client(timeout=float(timeout_sec)) as client:
|
||||
# try multi-image payload first
|
||||
try:
|
||||
r = _post_with_retry(client, url=url, headers=headers, payload=payload_multi)
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"http request failed: {type(e).__name__}: {e}", "backend_shape": "multi"}
|
||||
if r.status_code >= 400:
|
||||
# fallback single-image shape for strict backends
|
||||
try:
|
||||
r2 = _post_with_retry(client, url=url, headers=headers, payload=payload_single)
|
||||
except Exception as e:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": f"http fallback request failed: {type(e).__name__}: {e}",
|
||||
"backend_shape": "single-fallback-failed",
|
||||
}
|
||||
if r2.status_code >= 400:
|
||||
body2 = str(r2.text or "")
|
||||
if ("input.messages" in body2) or ("Input should be 'user'" in body2):
|
||||
# OpenAI-compatible Responses schema fallback.
|
||||
try:
|
||||
r3 = _post_with_retry(
|
||||
client,
|
||||
url=_join_url(resolved_base_url, "/responses"),
|
||||
headers=headers,
|
||||
payload=responses_payload,
|
||||
)
|
||||
if r3.status_code < 400:
|
||||
obj3 = r3.json()
|
||||
text3, out_images3 = _extract_text_and_images(obj3 if isinstance(obj3, dict) else {})
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text3,
|
||||
"images": out_images3,
|
||||
"backend_shape": "responses-fallback",
|
||||
"input_kind": input_kind if input_kind else ["data_url"],
|
||||
}
|
||||
except Exception:
|
||||
pass
|
||||
return {
|
||||
"ok": False,
|
||||
"error": f"http {r2.status_code}: {r2.text[:500]}",
|
||||
"backend_shape": "single-fallback-failed",
|
||||
}
|
||||
try:
|
||||
obj2 = r2.json()
|
||||
except Exception:
|
||||
return {"ok": False, "error": f"non-json response: {r2.text[:500]}", "backend_shape": "single"}
|
||||
text, out_images = _extract_text_and_images(obj2 if isinstance(obj2, dict) else {})
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text,
|
||||
"images": out_images,
|
||||
"backend_shape": "single",
|
||||
"input_kind": input_kind[-1:] if input_kind else ["data_url"],
|
||||
}
|
||||
|
||||
try:
|
||||
obj = r.json()
|
||||
except Exception:
|
||||
return {"ok": False, "error": f"non-json response: {r.text[:500]}", "backend_shape": "multi"}
|
||||
text, out_images = _extract_text_and_images(obj if isinstance(obj, dict) else {})
|
||||
return {
|
||||
"ok": True,
|
||||
"text": text,
|
||||
"images": out_images,
|
||||
"backend_shape": "multi",
|
||||
"input_kind": input_kind if input_kind else ["data_url"],
|
||||
}
|
||||
|
||||
|
||||
__all__ = ["send_image_messages"]
|
||||
|
||||
200
platform/llm/image_ocr_client.py
Normal file
200
platform/llm/image_ocr_client.py
Normal file
|
|
@ -0,0 +1,200 @@
|
|||
"""OCR / vision lane: OpenAI-compatible multimodal ``image_url`` + ``text`` only."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
from oclaw.platform.llm.image_http_common import (
|
||||
compress_data_url_image,
|
||||
env_ocr_lane_api_key,
|
||||
env_ocr_lane_base_url,
|
||||
env_ocr_lane_model,
|
||||
env_ocr_lane_chat_endpoint,
|
||||
extract_text_and_images,
|
||||
is_data_url,
|
||||
join_url,
|
||||
post_with_retry,
|
||||
)
|
||||
from oclaw.runtime.prompt_templates import render_prompt
|
||||
|
||||
# Shared with query_image_attachment and OpenAI-compat multimodal→text downgrade.
|
||||
VISION_OCR_EXTRACT_PROMPT_ZH = (
|
||||
"请只提取图片中可见文字并按阅读顺序输出。"
|
||||
"如果有表格,保持行列结构;不确定的内容标注为[unclear]。"
|
||||
)
|
||||
VISION_DESCRIBE_PROMPT_ZH = (
|
||||
"请详细描述这张图片的主要内容、对象、场景和可见文字。"
|
||||
"回答请使用要点列表,避免臆测。"
|
||||
)
|
||||
|
||||
|
||||
def _http_content_blocks_openai_vision(images: list[str], prompt: str) -> list[dict[str, Any]]:
|
||||
prompt_text = str(prompt or "").strip() or render_prompt("image/default_edit_prompt.zh.md", strict=True)
|
||||
blocks: list[dict[str, Any]] = []
|
||||
for img in images:
|
||||
s = str(img or "").strip()
|
||||
if not s:
|
||||
continue
|
||||
blocks.append({"type": "image_url", "image_url": {"url": s}})
|
||||
if prompt_text:
|
||||
blocks.append({"type": "text", "text": prompt_text})
|
||||
return blocks
|
||||
|
||||
|
||||
def vision_llm_backend_status(
|
||||
*,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
model: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
ak = (api_key or env_ocr_lane_api_key()).strip()
|
||||
bu = (base_url or env_ocr_lane_base_url()).strip()
|
||||
md = ((model or "").strip() or env_ocr_lane_model())
|
||||
if ak and bu and md:
|
||||
return {"ok": True, "backend": "aia_ocr_http"}
|
||||
return {
|
||||
"ok": False,
|
||||
"backend": None,
|
||||
"hint_zh": (
|
||||
"未配置 OCR/看图通道:请同时设置 AIA_OCR_BASE_URL、AIA_OCR_API_KEY、AIA_OCR_MODEL"
|
||||
"(OpenAI-compatible 多模态 /chat/completions)。不设默认模型 id。"
|
||||
),
|
||||
"hint_en": (
|
||||
"OCR/vision lane not configured: set AIA_OCR_BASE_URL, AIA_OCR_API_KEY, and AIA_OCR_MODEL. "
|
||||
"No default model id."
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def send_ocr_image_messages(
|
||||
*,
|
||||
images: list[str],
|
||||
prompt: str,
|
||||
model: str | None = None,
|
||||
timeout_sec: int = 60,
|
||||
api_key: str | None = None,
|
||||
base_url: str | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""OpenAI Chat Completions multimodal only (``/chat/completions``): multi-image then single-image."""
|
||||
resolved_base_url = (base_url or env_ocr_lane_base_url()).strip()
|
||||
resolved_api_key = (api_key or env_ocr_lane_api_key()).strip()
|
||||
model_name = ((model or "").strip() or env_ocr_lane_model())
|
||||
endpoint = env_ocr_lane_chat_endpoint()
|
||||
url = join_url(resolved_base_url, endpoint)
|
||||
if not resolved_api_key or not resolved_base_url:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": "missing AIA_OCR_API_KEY or AIA_OCR_BASE_URL (or pass api_key and base_url)",
|
||||
}
|
||||
if not model_name:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": "missing AIA_OCR_MODEL (or pass model=...) — no default model id",
|
||||
}
|
||||
if not images:
|
||||
return {"ok": False, "error": "at least one image input is required"}
|
||||
|
||||
raw_selected = [str(x).strip() for x in images if str(x).strip()][:3]
|
||||
selected: list[str] = []
|
||||
input_kind: list[str] = []
|
||||
for img in raw_selected:
|
||||
if is_data_url(img):
|
||||
selected.append(compress_data_url_image(img))
|
||||
input_kind.append("data_url")
|
||||
continue
|
||||
if img.startswith("http://") or img.startswith("https://"):
|
||||
selected.append(img)
|
||||
input_kind.append("url")
|
||||
continue
|
||||
if not selected:
|
||||
return {"ok": False, "error": "no usable image input (expected URL or data URL)"}
|
||||
|
||||
content_openai = _http_content_blocks_openai_vision(selected, prompt)
|
||||
content_openai_single = _http_content_blocks_openai_vision(selected[-1:], prompt) if selected else []
|
||||
|
||||
headers = {
|
||||
"Authorization": f"Bearer {resolved_api_key}",
|
||||
"Content-Type": "application/json",
|
||||
}
|
||||
payload_openai_multi = {
|
||||
"model": model_name,
|
||||
"messages": [{"role": "user", "content": content_openai}],
|
||||
}
|
||||
payload_openai_single = {
|
||||
"model": model_name,
|
||||
"messages": [{"role": "user", "content": content_openai_single}],
|
||||
}
|
||||
|
||||
def _ok_response(resp: httpx.Response) -> tuple[dict[str, Any] | None, bool]:
|
||||
if resp.status_code >= 400:
|
||||
return None, False
|
||||
try:
|
||||
obj = resp.json()
|
||||
except Exception:
|
||||
return None, False
|
||||
if not isinstance(obj, dict):
|
||||
return None, False
|
||||
text, out_imgs = extract_text_and_images(obj)
|
||||
if not str(text or "").strip():
|
||||
return obj, False
|
||||
return {"text": text, "images": out_imgs}, True
|
||||
|
||||
err_last = ""
|
||||
with httpx.Client(timeout=float(timeout_sec)) as client:
|
||||
try:
|
||||
r0 = post_with_retry(client, url=url, headers=headers, payload=payload_openai_multi)
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"http request failed: {type(e).__name__}: {e}", "backend_shape": "openai-multi"}
|
||||
if r0 is not None:
|
||||
if r0.status_code >= 400:
|
||||
err_last = f"http {r0.status_code}: {r0.text[:500]}"
|
||||
else:
|
||||
parsed, ok = _ok_response(r0)
|
||||
if ok and isinstance(parsed, dict):
|
||||
return {
|
||||
"ok": True,
|
||||
"text": str(parsed.get("text") or ""),
|
||||
"images": list(parsed.get("images") or []),
|
||||
"backend_shape": "openai-multi",
|
||||
"input_kind": input_kind if input_kind else ["data_url"],
|
||||
}
|
||||
|
||||
try:
|
||||
r1 = post_with_retry(client, url=url, headers=headers, payload=payload_openai_single)
|
||||
except Exception as e:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": err_last or f"openai-single request failed: {type(e).__name__}: {e}",
|
||||
"backend_shape": "openai-single-failed",
|
||||
}
|
||||
if r1.status_code >= 400:
|
||||
return {
|
||||
"ok": False,
|
||||
"error": err_last or f"http {r1.status_code}: {r1.text[:500]}",
|
||||
"backend_shape": "openai-single-failed",
|
||||
}
|
||||
try:
|
||||
obj1 = r1.json()
|
||||
except Exception:
|
||||
return {"ok": False, "error": f"non-json response: {r1.text[:500]}", "backend_shape": "openai-single"}
|
||||
text, out_images = extract_text_and_images(obj1 if isinstance(obj1, dict) else {})
|
||||
out: dict[str, Any] = {
|
||||
"ok": bool(str(text or "").strip()),
|
||||
"text": text,
|
||||
"images": out_images,
|
||||
"backend_shape": "openai-single",
|
||||
"input_kind": input_kind[-1:] if input_kind else ["data_url"],
|
||||
}
|
||||
if not str(text or "").strip():
|
||||
out["error"] = "empty assistant text after openai-single"
|
||||
return out
|
||||
|
||||
|
||||
__all__ = [
|
||||
"VISION_DESCRIBE_PROMPT_ZH",
|
||||
"VISION_OCR_EXTRACT_PROMPT_ZH",
|
||||
"send_ocr_image_messages",
|
||||
"vision_llm_backend_status",
|
||||
]
|
||||
|
|
@ -7,7 +7,105 @@ from typing import Any
|
|||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_MIN_TOOL_PARAMETERS: dict[str, Any] = {"type": "object", "additionalProperties": True}
|
||||
# Full minimal object schema for ``function.parameters`` (wire / shrink / tier_minimal).
|
||||
# A bare ``{type, additionalProperties}`` alone has led strict gateways to emit or validate
|
||||
# ``required: null`` / missing array fields — always send explicit ``properties`` + ``required``.
|
||||
_MIN_TOOL_PARAMETERS: dict[str, Any] = {
|
||||
"type": "object",
|
||||
"properties": {},
|
||||
"required": [],
|
||||
"additionalProperties": True,
|
||||
}
|
||||
|
||||
_COMBO_KEYS = frozenset({"allOf", "anyOf", "oneOf", "prefixItems"})
|
||||
|
||||
|
||||
def _schema_type_tags(t: Any) -> set[str]:
|
||||
if t is None:
|
||||
return set()
|
||||
if isinstance(t, list):
|
||||
return {str(x).strip().lower() for x in t if x is not None and str(x).strip()}
|
||||
s = str(t).strip().lower()
|
||||
return {s} if s else set()
|
||||
|
||||
|
||||
def complete_json_schema_for_openai_tools(obj: Any) -> Any:
|
||||
"""Recursively ensure JSON-schema fragments match what strict OpenAI-compat gateways expect.
|
||||
|
||||
This is **structural completion** for our wire path (tier/shrink/SDK), not a general MCP fix-all:
|
||||
every ``type: object`` node gets explicit ``properties`` + ``required`` (list); every
|
||||
``type: array`` gets ``items`` (object). Combo keywords that must be arrays and are ``null``
|
||||
are dropped; list elements that are ``null`` are removed.
|
||||
"""
|
||||
|
||||
if isinstance(obj, dict):
|
||||
out: dict[str, Any] = {}
|
||||
for k, v in obj.items():
|
||||
key = str(k)
|
||||
if key in _COMBO_KEYS and v is None:
|
||||
if key in ("allOf", "prefixItems"):
|
||||
out[key] = []
|
||||
continue
|
||||
out[key] = complete_json_schema_for_openai_tools(v)
|
||||
|
||||
for combo in _COMBO_KEYS:
|
||||
if combo not in out:
|
||||
continue
|
||||
val = out[combo]
|
||||
if not isinstance(val, list):
|
||||
continue
|
||||
cleaned = [complete_json_schema_for_openai_tools(x) for x in val if x is not None]
|
||||
if combo in ("anyOf", "oneOf") and not cleaned:
|
||||
out.pop(combo, None)
|
||||
else:
|
||||
out[combo] = cleaned
|
||||
|
||||
if isinstance(out.get("properties"), dict):
|
||||
t0 = out.get("type")
|
||||
if t0 is None or (isinstance(t0, str) and not str(t0).strip()):
|
||||
out["type"] = "object"
|
||||
|
||||
tags = _schema_type_tags(out.get("type"))
|
||||
if "object" in tags:
|
||||
if not isinstance(out.get("properties"), dict):
|
||||
out["properties"] = {}
|
||||
r = out.get("required")
|
||||
if r is None or not isinstance(r, list):
|
||||
out["required"] = []
|
||||
if "array" in tags:
|
||||
if out.get("items") is None:
|
||||
out["items"] = {}
|
||||
elif isinstance(out.get("items"), dict):
|
||||
out["items"] = complete_json_schema_for_openai_tools(out["items"])
|
||||
return out
|
||||
if isinstance(obj, list):
|
||||
return [complete_json_schema_for_openai_tools(x) for x in obj]
|
||||
return obj
|
||||
|
||||
|
||||
def complete_openai_tools_wire_parameters(tools: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""Apply :func:`complete_json_schema_for_openai_tools` to each function's ``parameters``."""
|
||||
out: list[dict[str, Any]] = []
|
||||
for ent in tools or []:
|
||||
if not isinstance(ent, dict):
|
||||
continue
|
||||
if str(ent.get("type") or "") != "function":
|
||||
out.append(ent)
|
||||
continue
|
||||
fn = ent.get("function")
|
||||
if not isinstance(fn, dict):
|
||||
out.append(ent)
|
||||
continue
|
||||
params = fn.get("parameters")
|
||||
if not isinstance(params, dict):
|
||||
row = dict(ent)
|
||||
row["function"] = {**dict(fn), "parameters": dict(_MIN_TOOL_PARAMETERS)}
|
||||
out.append(row)
|
||||
continue
|
||||
row = dict(ent)
|
||||
row["function"] = {**dict(fn), "parameters": complete_json_schema_for_openai_tools(dict(params))}
|
||||
out.append(row)
|
||||
return out
|
||||
|
||||
|
||||
def openai_tools_json_byte_length(tools: list[dict[str, Any]]) -> int:
|
||||
|
|
@ -96,8 +194,12 @@ def default_max_openai_tools_json_bytes(base_url: str | None) -> int | None:
|
|||
|
||||
|
||||
__all__ = [
|
||||
"MIN_OPENAI_FUNCTION_PARAMETERS",
|
||||
"complete_json_schema_for_openai_tools",
|
||||
"complete_openai_tools_wire_parameters",
|
||||
"default_max_openai_tools_json_bytes",
|
||||
"openai_tools_json_byte_length",
|
||||
"shrink_openai_tools_payload_for_api",
|
||||
]
|
||||
|
||||
MIN_OPENAI_FUNCTION_PARAMETERS = _MIN_TOOL_PARAMETERS
|
||||
|
|
|
|||
|
|
@ -12,6 +12,8 @@ import os
|
|||
from datetime import datetime, timedelta, timezone
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.llm.tool_schema import MIN_OPENAI_FUNCTION_PARAMETERS, complete_openai_tools_wire_parameters
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SETTINGS_KEY_PENALTY_STATE = "mcp_tool_wire_penalty_state"
|
||||
|
|
@ -25,7 +27,7 @@ SETTINGS_KEY_PENALTY_STATE_BY_ROLE = "mcp_tool_wire_penalty_state_by_role"
|
|||
SETTINGS_KEY_TOOL_POLICIES_BY_ROLE = "mcp_tool_wire_tool_policies_by_role"
|
||||
SETTINGS_KEY_ROLE_MODE_BY_ROLE = "mcp_tool_wire_role_mode_by_role"
|
||||
|
||||
_MIN_PARAMETERS: dict[str, Any] = {"type": "object", "additionalProperties": True}
|
||||
_MIN_PARAMETERS: dict[str, Any] = dict(MIN_OPENAI_FUNCTION_PARAMETERS)
|
||||
|
||||
|
||||
def _env_prefixed(name_suffix: str, default: str = "") -> str:
|
||||
|
|
@ -413,6 +415,12 @@ def prepare_openai_tools_for_llm_api(
|
|||
if not tools:
|
||||
return tools
|
||||
|
||||
tools = complete_openai_tools_wire_parameters(tools)
|
||||
|
||||
def _finalize(result: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
"""Re-apply completion after tier/shrink/squeeze (those paths emit fresh ``parameters`` dicts)."""
|
||||
return complete_openai_tools_wire_parameters(result)
|
||||
|
||||
policies: dict[str, int] = {}
|
||||
admin: dict[str, Any] = _admin_defaults_from_env()
|
||||
role_mode = "restricted"
|
||||
|
|
@ -450,17 +458,17 @@ def prepare_openai_tools_for_llm_api(
|
|||
if role_mode == "forbidden":
|
||||
out = _filter_mcp_out(tools)
|
||||
if max_json_bytes is not None and max_json_bytes > 0:
|
||||
return _fallback_shrink(out, max_json_bytes)
|
||||
return out
|
||||
return _finalize(_fallback_shrink(out, max_json_bytes))
|
||||
return _finalize(out)
|
||||
if role_mode == "unrestricted":
|
||||
if max_json_bytes is not None and max_json_bytes > 0:
|
||||
return _fallback_shrink(tools_pass1, max_json_bytes)
|
||||
return tools_pass1
|
||||
return _finalize(_fallback_shrink(tools_pass1, max_json_bytes))
|
||||
return _finalize(tools_pass1)
|
||||
|
||||
if not wire_graduation_effective(base_url, admin):
|
||||
if max_json_bytes is not None and max_json_bytes > 0:
|
||||
return _fallback_shrink(tools_pass1, max_json_bytes)
|
||||
return tools_pass1
|
||||
return _finalize(_fallback_shrink(tools_pass1, max_json_bytes))
|
||||
return _finalize(tools_pass1)
|
||||
|
||||
top_n = int(admin.get("top_n_full") or 20)
|
||||
top_n = max(3, min(top_n, 80))
|
||||
|
|
@ -481,8 +489,8 @@ def prepare_openai_tools_for_llm_api(
|
|||
except Exception as exc:
|
||||
logger.warning("tool_wire_policy: usage store unavailable (%s); fallback shrink only", exc)
|
||||
if max_json_bytes is not None and max_json_bytes > 0:
|
||||
return _fallback_shrink(tools_pass1, max_json_bytes)
|
||||
return tools_pass1
|
||||
return _finalize(_fallback_shrink(tools_pass1, max_json_bytes))
|
||||
return _finalize(tools_pass1)
|
||||
|
||||
ranked = sorted(
|
||||
usage_map.items(),
|
||||
|
|
@ -601,9 +609,9 @@ def prepare_openai_tools_for_llm_api(
|
|||
)
|
||||
|
||||
if max_json_bytes is not None and max_json_bytes > 0 and _wire_json_size(built) > max_json_bytes:
|
||||
return _squeeze_to_budget(built, max_json_bytes, admin=admin)
|
||||
return _finalize(_squeeze_to_budget(built, max_json_bytes, admin=admin))
|
||||
|
||||
return built
|
||||
return _finalize(built)
|
||||
|
||||
|
||||
def _squeeze_to_budget(
|
||||
|
|
|
|||
|
|
@ -8,6 +8,7 @@ import uuid
|
|||
from typing import Any, Optional
|
||||
from collections.abc import Callable
|
||||
|
||||
from oclaw.platform.llm.tool_schema import complete_openai_tools_wire_parameters
|
||||
from oclaw.platform.llm.transports.base import ChatModel, LLMResponse, LLMToolCall, normalize_image_b64_payload, coerce_thought_signature_for_storage
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
@ -71,6 +72,272 @@ def _should_enable_thinking(base_url: str | None, *, thinking_mode_enabled: bool
|
|||
return True
|
||||
|
||||
|
||||
def _should_force_text_only_messages(base_url: str | None) -> bool:
|
||||
del base_url
|
||||
# Opt-in only. Avoid hard-coding vendor/domain specific heuristics.
|
||||
return _truthy_env("AIA_OPENAI_FORCE_TEXT_CONTENT", "0")
|
||||
|
||||
|
||||
def _multimodal_proactive_downgrade_enabled() -> bool:
|
||||
raw = str(os.getenv("AIA_OPENAI_MULTIMODAL_PROACTIVE_DOWNGRADE") or "1").strip().lower()
|
||||
return raw not in {"0", "false", "no", "off"}
|
||||
|
||||
|
||||
def _text_only_multimodal_substrings() -> tuple[str, ...]:
|
||||
raw = os.getenv("AIA_OPENAI_TEXT_ONLY_MULTIMODAL_SUBSTRINGS")
|
||||
if raw is None:
|
||||
return ("deepseek",)
|
||||
s = str(raw).strip().lower()
|
||||
if s in {"", "-", "none", "off"}:
|
||||
return ()
|
||||
return tuple(part.strip().lower() for part in str(raw).split(",") if part.strip())
|
||||
|
||||
|
||||
def _messages_contain_list_with_image(messages: list[dict[str, Any]]) -> bool:
|
||||
for m in messages or []:
|
||||
if not isinstance(m, dict):
|
||||
continue
|
||||
c = m.get("content")
|
||||
if not isinstance(c, list):
|
||||
continue
|
||||
for item in c:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
typ = str(item.get("type") or "").strip().lower()
|
||||
if typ in ("image_url", "input_image"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _should_proactively_downgrade_multimodal_messages(
|
||||
messages: list[dict[str, Any]],
|
||||
*,
|
||||
model: str | None,
|
||||
base_url: str | None,
|
||||
) -> bool:
|
||||
if _should_force_text_only_messages(base_url):
|
||||
return _messages_contain_list_with_image(messages)
|
||||
if not _multimodal_proactive_downgrade_enabled():
|
||||
return False
|
||||
if not _messages_contain_list_with_image(messages):
|
||||
return False
|
||||
parts = _text_only_multimodal_substrings()
|
||||
if not parts:
|
||||
return False
|
||||
hay = f"{model or ''} {base_url or ''}".lower()
|
||||
return any(p and p in hay for p in parts)
|
||||
|
||||
|
||||
def _wire_error_message(exc: BaseException) -> str:
|
||||
chunks: list[str] = [str(exc)]
|
||||
em = getattr(exc, "message", None)
|
||||
if isinstance(em, str) and em.strip():
|
||||
chunks.append(em)
|
||||
body = getattr(exc, "body", None)
|
||||
if isinstance(body, dict):
|
||||
try:
|
||||
chunks.append(json.dumps(body, ensure_ascii=False, default=str))
|
||||
except Exception:
|
||||
chunks.append(repr(body))
|
||||
elif isinstance(body, str) and body.strip():
|
||||
chunks.append(body)
|
||||
resp = getattr(exc, "response", None)
|
||||
if resp is not None:
|
||||
t = ""
|
||||
try:
|
||||
rt = getattr(resp, "text", "")
|
||||
if isinstance(rt, str):
|
||||
t = rt
|
||||
except Exception:
|
||||
pass
|
||||
if not t.strip():
|
||||
try:
|
||||
ct = getattr(resp, "content", None)
|
||||
if isinstance(ct, (bytes, bytearray)):
|
||||
t = bytes(ct).decode("utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
if t.strip():
|
||||
chunks.append(t)
|
||||
inner = getattr(exc, "__cause__", None)
|
||||
if inner is not None:
|
||||
chunks.append(str(inner))
|
||||
return "\n".join(chunks)
|
||||
|
||||
|
||||
def _is_text_only_gateway_error(msg: str) -> bool:
|
||||
m = str(msg or "").strip().lower()
|
||||
if not m:
|
||||
return False
|
||||
if "unknown variant image_url" in m:
|
||||
return True
|
||||
if "image_url" in m and "expected text" in m:
|
||||
return True
|
||||
if "failed to deserialize" in m and ("expected text" in m or "unknown variant image_url" in m):
|
||||
return True
|
||||
if "deserialize" in m and "image_url" in m and ("expected text" in m or "messages[" in m):
|
||||
return True
|
||||
if "invalid_request_error" in m and ("image_url" in m or "expected text" in m):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _flatten_message_content_to_text(content: Any) -> str:
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
if not isinstance(content, list):
|
||||
return str(content or "")
|
||||
parts: list[str] = []
|
||||
for item in content:
|
||||
if isinstance(item, str):
|
||||
if item.strip():
|
||||
parts.append(item)
|
||||
continue
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
typ = str(item.get("type") or "").strip().lower()
|
||||
if typ in ("text", "input_text"):
|
||||
txt = str(item.get("text") or item.get("content") or "").strip()
|
||||
if txt:
|
||||
parts.append(txt)
|
||||
continue
|
||||
if typ in ("image_url", "input_image"):
|
||||
# Proactive / forced text-only path without OCR injection (or OCR disabled).
|
||||
parts.append(
|
||||
"【图片】本消息含图片,但当前以纯文本发往上游模型,像素未传入;"
|
||||
"若需要图中细节,请改用支持多模态的模型或开启看图 OCR 降级通道。"
|
||||
)
|
||||
continue
|
||||
return "\n".join([p for p in parts if p]).strip()
|
||||
|
||||
|
||||
def _multimodal_downgrade_ocr_enabled() -> bool:
|
||||
"""When True, text-only downgrade replaces image blocks with OCR text via AIA_OCR_* lane."""
|
||||
return str(os.getenv("AIA_MULTIMODAL_DOWNGRADE_OCR") or "1").strip().lower() not in {
|
||||
"0",
|
||||
"false",
|
||||
"no",
|
||||
"off",
|
||||
}
|
||||
|
||||
|
||||
def _image_block_to_url(item: dict[str, Any]) -> str | None:
|
||||
typ = str(item.get("type") or "").strip().lower()
|
||||
if typ == "image_url":
|
||||
u = item.get("image_url")
|
||||
if isinstance(u, dict):
|
||||
s = str(u.get("url") or "").strip()
|
||||
return s or None
|
||||
if isinstance(u, str) and u.strip():
|
||||
return u.strip()
|
||||
if typ == "input_image":
|
||||
b64 = normalize_image_b64_payload(item.get("image_base64") or item.get("data"))
|
||||
mime = str(item.get("mime") or "image/jpeg").strip() or "image/jpeg"
|
||||
if b64:
|
||||
return f"data:{mime};base64,{b64}"
|
||||
return None
|
||||
|
||||
|
||||
def _message_list_contains_image_block(content: Any) -> bool:
|
||||
if not isinstance(content, list):
|
||||
return False
|
||||
for item in content:
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
typ = str(item.get("type") or "").strip().lower()
|
||||
if typ in ("image_url", "input_image"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _downgrade_ocr_wrap_for_llm(ocr_text: str) -> str:
|
||||
"""User-facing copy injected into the main model when multimodal was OCR-downgraded to text."""
|
||||
head = (
|
||||
"【图片内容·OCR】当前主对话模型仅接收纯文本,无法直接读图;"
|
||||
"下面是由独立看图通道从用户图片中转写提取的正文。请把它和用户的文字问题一起当作依据来回答;"
|
||||
"转写可能有漏字、错行或与截图不完全一致,重要结论可请用户核对原图或补充说明。"
|
||||
)
|
||||
return f"{head}\n\n{ocr_text.strip()}"
|
||||
|
||||
|
||||
def _flatten_message_content_for_text_gateway(content: Any) -> str:
|
||||
"""Flatten list content to a string; image blocks may become OCR text when configured."""
|
||||
if isinstance(content, str):
|
||||
return content
|
||||
if not isinstance(content, list):
|
||||
return str(content or "")
|
||||
if _multimodal_downgrade_ocr_enabled() and _message_list_contains_image_block(content):
|
||||
try:
|
||||
from oclaw.platform.llm.image_ocr_client import (
|
||||
VISION_OCR_EXTRACT_PROMPT_ZH,
|
||||
send_ocr_image_messages,
|
||||
vision_llm_backend_status,
|
||||
)
|
||||
except Exception:
|
||||
return _flatten_message_content_to_text(content)
|
||||
if vision_llm_backend_status().get("ok"):
|
||||
parts: list[str] = []
|
||||
for item in content:
|
||||
if isinstance(item, str):
|
||||
if item.strip():
|
||||
parts.append(item)
|
||||
continue
|
||||
if not isinstance(item, dict):
|
||||
continue
|
||||
typ = str(item.get("type") or "").strip().lower()
|
||||
if typ in ("image_url", "input_image"):
|
||||
url = _image_block_to_url(item)
|
||||
if not url:
|
||||
parts.append(
|
||||
"【图片】无法从本消息中还原图片数据,图中细节未传入本轮模型;"
|
||||
"若作答需要像素级信息,请提示用户重发或改用支持多模态的模型配置。"
|
||||
)
|
||||
continue
|
||||
try:
|
||||
out = send_ocr_image_messages(images=[url], prompt=VISION_OCR_EXTRACT_PROMPT_ZH)
|
||||
except Exception as exc:
|
||||
parts.append(
|
||||
f"【图片】独立看图通道 OCR 调用异常({type(exc).__name__}),"
|
||||
"本张图的正文未能写入对话;可请用户改述图中要点或稍后重试。"
|
||||
)
|
||||
continue
|
||||
if not out.get("ok"):
|
||||
err = str(out.get("error") or "unknown")
|
||||
parts.append(
|
||||
f"【图片】看图通道返回失败({err}),本张图未转成文字;"
|
||||
"请据用户文字说明作答,或提示检查 OCR 配置/重发图片。"
|
||||
)
|
||||
continue
|
||||
t = str(out.get("text") or "").strip()
|
||||
if not t:
|
||||
parts.append(
|
||||
"【图片】看图通道未返回有效文字(可能模型拒识或图不清晰)。"
|
||||
"请提示用户补充文字说明或重发更清晰的截图。"
|
||||
)
|
||||
else:
|
||||
parts.append(_downgrade_ocr_wrap_for_llm(t))
|
||||
continue
|
||||
if typ in ("text", "input_text"):
|
||||
txt = str(item.get("text") or item.get("content") or "").strip()
|
||||
if txt:
|
||||
parts.append(txt)
|
||||
continue
|
||||
return "\n\n".join([p for p in parts if p]).strip()
|
||||
return _flatten_message_content_to_text(content)
|
||||
|
||||
|
||||
def _normalize_messages_for_text_only_gateway(messages: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||
out: list[dict[str, Any]] = []
|
||||
for m in messages or []:
|
||||
if not isinstance(m, dict):
|
||||
continue
|
||||
mm = dict(m)
|
||||
if isinstance(mm.get("content"), list):
|
||||
mm["content"] = _flatten_message_content_for_text_gateway(mm.get("content"))
|
||||
out.append(mm)
|
||||
return out
|
||||
|
||||
|
||||
def _find_thought_signature_in_obj(o: Any) -> str | None:
|
||||
if isinstance(o, dict):
|
||||
for k in ("thought_signature", "thoughtSignature"):
|
||||
|
|
@ -222,7 +489,14 @@ class OpenAIChatModel(ChatModel):
|
|||
else:
|
||||
cleaned_msgs.append(m)
|
||||
|
||||
kwargs: dict[str, Any] = {"model": self.model, "messages": cleaned_msgs, "stream": stream}
|
||||
msgs_wire = cleaned_msgs
|
||||
if _should_force_text_only_messages(self.base_url) or _should_proactively_downgrade_multimodal_messages(
|
||||
cleaned_msgs,
|
||||
model=self.model,
|
||||
base_url=self.base_url,
|
||||
):
|
||||
msgs_wire = _normalize_messages_for_text_only_gateway(cleaned_msgs)
|
||||
kwargs: dict[str, Any] = {"model": self.model, "messages": msgs_wire, "stream": stream}
|
||||
if _should_enable_thinking(self.base_url, thinking_mode_enabled=bool(getattr(self, "thinking_mode_enabled", False))):
|
||||
extra_body = kwargs.get("extra_body") if isinstance(kwargs.get("extra_body"), dict) else {}
|
||||
extra_body = dict(extra_body)
|
||||
|
|
@ -253,11 +527,14 @@ class OpenAIChatModel(ChatModel):
|
|||
)
|
||||
kwargs["tools"] = plan.tools_wired
|
||||
except Exception:
|
||||
kwargs["tools"] = tools
|
||||
kwargs["tools"] = complete_openai_tools_wire_parameters(tools)
|
||||
try:
|
||||
return self._client.chat.completions.create(**kwargs)
|
||||
except Exception as exc:
|
||||
msg = str(exc)
|
||||
msg = _wire_error_message(exc)
|
||||
if _is_text_only_gateway_error(msg) and isinstance(kwargs.get("messages"), list):
|
||||
kwargs["messages"] = _normalize_messages_for_text_only_gateway(kwargs["messages"])
|
||||
return self._client.chat.completions.create(**kwargs)
|
||||
if "reasoning_content" in msg and "thinking mode" in msg and "must be passed back" in msg:
|
||||
# Provider requires replaying assistant.reasoning_content in thinking mode.
|
||||
# As a safety fallback, force-disable thinking and retry once.
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue