Refactor media payload handling to persist base64 blobs as attachment refs and keep non-turn model paths safe by degrading risky payloads.

This preserves multimodal fidelity for the latest user turn while preventing historical/tool replay bloat, and adds admin UI support for referenced attachment preview/download flows.

Made-with: Cursor
This commit is contained in:
oliver 2026-04-28 10:30:27 +08:00
parent 3d27a01879
commit 6cfaff06f6
20 changed files with 1626 additions and 101 deletions

View file

@ -14,6 +14,7 @@ import re
from typing import Any
from oclaw.platform.llm.chat_models import _normalize_image_b64_payload, gemini_openai_compat_client, ChatModel
from oclaw.runtime.chat.media_redact import redact_embedded_image_blobs
from oclaw.runtime.chat.tool_runtime import tool_llm_message_max_chars, truncate_tool_result_for_llm_messages
from oclaw.prompts import render_prompt
from oclaw.platform.files.attachment_assets import attachment_id_to_data_url
@ -196,8 +197,18 @@ def build_llm_messages(
model: ChatModel,
lang: str,
tool_context_truncate_enabled: bool = True,
active_turn_uuid: str | None = None,
) -> list[dict[str, Any]]:
"""把 DB 中的消息序列转换为 LLM messages。"""
"""把 DB 中的消息序列转换为 LLM messages。
When ``active_turn_uuid`` matches a tool/user row ``turn_uuid``, that turn is treated as the
in-flight MCP turn: tool JSON is not stripped of nested image payloads here (see also
:func:`~oclaw.runtime.direct_loop._guard_tool_results_for_llm_context`). Omit or leave empty
to apply image-blob stripping for every tool row (safe default for callers without turn context).
Only the **last** user message may expand attachments into native multimodal ``input_image``;
older user attachments are replayed as text metadata only.
"""
out: list[dict[str, Any]] = [{"role": "system", "content": (system_prompt or "").strip()}]
thinking_mode_enabled = bool(getattr(model, "thinking_mode_enabled", False))
allow_signature_replay = _allow_reasoning_signature_replay(model)
@ -239,6 +250,15 @@ def build_llm_messages(
seen_tool_ids.add(tcid)
tool_ids_after.append(set(seen_tool_ids))
tool_ids_after.reverse()
last_user_msg_idx = -1
for _ui, _um in enumerate(store_messages or []):
if str(getattr(_um, "role", "") or "") != "user":
continue
if str(getattr(_um, "event_type", "") or "").strip().lower() == "reasoning":
continue
last_user_msg_idx = _ui
def _attach_reasoning_content(row: dict[str, Any], m: Any) -> dict[str, Any]:
if not thinking_mode_enabled:
return row
@ -291,34 +311,53 @@ def build_llm_messages(
if not isinstance(att, dict):
continue
att_type = att.get("type")
expand_user_image_for_model = bool(i == last_user_msg_idx)
if att_type in ("image", "input_image"):
b64 = _normalize_image_b64_payload(att.get("image_base64") or att.get("data"))
if not b64:
continue
content_list.append(
{
"type": "input_image",
"image_base64": b64,
"mime": att.get("mime") or "image/jpeg",
}
)
if expand_user_image_for_model:
b64 = _normalize_image_b64_payload(att.get("image_base64") or att.get("data"))
if not b64:
continue
content_list.append(
{
"type": "input_image",
"image_base64": b64,
"mime": att.get("mime") or "image/jpeg",
}
)
else:
name = str(att.get("name") or "image")
mime = str(att.get("mime") or "image/jpeg")
hs = "(historical attachment; pixels not replayed into model)"
hs_zh = "(历史附件;不向模型回放像素)"
hint = hs_zh if not str(lang or "").startswith("en") else hs
meta_line = f"- name={name} mime={mime} {hint}"
content_list.append(
{
"type": "text",
"text": render_prompt(
"tools/image_attachment_meta.md",
variables={"meta_line": meta_line},
strict=True,
),
}
)
elif att_type == "image_ref":
# Prefer actual image bytes so multi-agent/image specialist can truly "see" history images.
name = str(att.get("name") or "image")
mime = str(att.get("mime") or "image/jpeg")
aid = str(att.get("attachment_id") or "")
data_url = attachment_id_to_data_url(aid, mime=mime) if aid else ""
if data_url:
if ";base64," in data_url:
b64 = data_url.split(";base64,", 1)[1]
content_list.append(
{
"type": "input_image",
"image_base64": b64,
"mime": mime,
}
)
continue
if expand_user_image_for_model:
data_url = attachment_id_to_data_url(aid, mime=mime) if aid else ""
if data_url:
if ";base64," in data_url:
b64 = data_url.split(";base64,", 1)[1]
content_list.append(
{
"type": "input_image",
"image_base64": b64,
"mime": mime,
}
)
continue
w = att.get("width")
h = att.get("height")
sz = att.get("bytes")
@ -423,7 +462,7 @@ def build_llm_messages(
aid = str(_fid or "").strip()
except Exception:
aid = ""
if aid and mime.startswith("image/"):
if expand_user_image_for_model and aid and mime.startswith("image/"):
data_url = attachment_id_to_data_url(aid, mime=mime)
if data_url and ";base64," in data_url:
b64 = data_url.split(";base64,", 1)[1]
@ -575,6 +614,15 @@ def build_llm_messages(
)
continue
raw_tc_content = getattr(m, "content", "") or ""
_tun = str(getattr(m, "turn_uuid", "") or "").strip()
_aus = str(active_turn_uuid or "").strip()
if (not _aus) or (_tun != _aus):
try:
_p = json.loads(raw_tc_content)
_p2 = redact_embedded_image_blobs(_p)
raw_tc_content = json.dumps(_p2, ensure_ascii=False, default=str)
except Exception:
pass
tool_content_out = raw_tc_content
cap = tool_llm_message_max_chars()
if str(tool_call_id) in historical_tool_ids:

View file

@ -0,0 +1,198 @@
"""Strip/ingest embedded binary payloads from tool/MCP-shaped JSON.
Persistence is untouched; callers use copies when building model context."""
from __future__ import annotations
import base64
from typing import Any
from oclaw.platform.files.attachment_assets import AttachmentAssetStore
_IMAGE_CONTENT_TYPES = frozenset({"image", "input_image"})
_BASE64_PAYLOAD_KEYS = ("data", "image_base64", "base64", "content_base64", "body_base64")
# Below this length we keep values (tiny icons / markers).
_MIN_B64_CHARS = 200
def redact_embedded_image_blobs(obj: Any) -> Any:
"""Deep-copy-ish transform: replace large base64 payloads with metadata placeholders."""
if isinstance(obj, dict):
return _redact_dict(obj)
if isinstance(obj, list):
return [redact_embedded_image_blobs(x) for x in obj]
return obj
def _looks_like_large_payload(s: str) -> bool:
t = str(s or "").strip()
if len(t) < _MIN_B64_CHARS:
return False
allowed = frozenset("ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r-_")
if not t[: min(512, len(t))]:
return False
noise = sum(1 for ch in t[: min(2000, len(t))] if ch not in allowed)
return noise <= max(2, len(t[: min(2000, len(t))]) // 200)
def _redact_dict(d: dict[str, Any]) -> dict[str, Any]:
typ = str(d.get("type") or "").strip().lower()
payload_keys = [k for k in _BASE64_PAYLOAD_KEYS if isinstance(d.get(k), str) and _looks_like_large_payload(str(d.get(k) or ""))]
if payload_keys:
plen = max(len(str(d.get(k) or "")) for k in payload_keys)
dup: dict[str, Any] = {}
for k, v in d.items():
if k in payload_keys:
continue
if isinstance(v, dict):
dup[k] = _redact_dict(v)
elif isinstance(v, list):
dup[k] = [redact_embedded_image_blobs(x) for x in v]
else:
dup[k] = v
is_image = typ in _IMAGE_CONTENT_TYPES
dup["_image_payload_redacted" if is_image else "_binary_payload_redacted"] = True
dup["_redacted_payload_chars"] = int(plen)
dup["_redacted_payload_keys"] = list(payload_keys)
return dup
out: dict[str, Any] = {}
for k, v in d.items():
if isinstance(v, dict):
out[k] = _redact_dict(v)
elif isinstance(v, list):
out[k] = [redact_embedded_image_blobs(x) for x in v]
else:
out[k] = v
return out
def ingest_embedded_image_blobs_as_refs(
obj: Any,
*,
root_dir: str | None = None,
filename_prefix: str = "tool-image",
) -> tuple[Any, list[dict[str, Any]]]:
"""Persist nested base64 blobs and replace them with attachment refs.
Returns transformed object and newly created attachment refs.
"""
store = AttachmentAssetStore(root_dir=root_dir) if root_dir else AttachmentAssetStore()
refs: list[dict[str, Any]] = []
def _ingest(node: Any, idx_seed: list[int]) -> Any:
if isinstance(node, list):
return [_ingest(x, idx_seed) for x in node]
if not isinstance(node, dict):
return node
typ = str(node.get("type") or "").strip().lower()
raw = _pick_base64_payload(node)
if raw:
blob = _decode_image_bytes(raw)
if blob:
idx_seed[0] += 1
mime = str(node.get("mime") or node.get("mime_type") or "image/png").strip() or "image/png"
ext = _filename_ext_for_mime(mime)
name = str(node.get("name") or f"{filename_prefix}-{idx_seed[0]}{ext}").strip()
meta = store.save_bytes(
blob,
filename=name,
mime=mime,
width=_safe_int(node.get("width")),
height=_safe_int(node.get("height")),
)
ref_type = _ref_type_for_mime(mime, typ)
ref = {
"type": ref_type,
"attachment_id": meta.attachment_id,
"name": meta.name,
"mime": meta.mime,
"bytes": meta.bytes,
"width": meta.width,
"height": meta.height,
}
refs.append(ref)
return ref
redacted = _redact_dict(node)
redacted["type"] = _ref_type_for_mime(
str(node.get("mime") or node.get("mime_type") or "application/octet-stream"),
typ,
)
redacted.setdefault("name", str(node.get("name") or "attachment"))
redacted.setdefault("mime", str(node.get("mime") or node.get("mime_type") or "application/octet-stream"))
return redacted
out: dict[str, Any] = {}
for k, v in node.items():
out[k] = _ingest(v, idx_seed)
return out
transformed = _ingest(obj, [0])
uniq: list[dict[str, Any]] = []
seen: set[str] = set()
for r in refs:
aid = str(r.get("attachment_id") or "").strip()
if not aid or aid in seen:
continue
seen.add(aid)
uniq.append(r)
return transformed, uniq
def _decode_image_bytes(raw: Any) -> bytes:
s = str(raw or "").strip()
if not s:
return b""
if s.startswith("data:") and ";base64," in s:
s = s.split(";base64,", 1)[1]
try:
return base64.b64decode(s.encode("ascii"), validate=False)
except Exception:
return b""
def _pick_base64_payload(node: dict[str, Any]) -> str:
for k in _BASE64_PAYLOAD_KEYS:
v = node.get(k)
if isinstance(v, str) and str(v).strip():
return v
return ""
def _ref_type_for_mime(mime: str, typ: str = "") -> str:
m = str(mime or "").strip().lower()
t = str(typ or "").strip().lower()
if t in _IMAGE_CONTENT_TYPES or m.startswith("image/"):
return "image_ref"
if m.startswith("video/"):
return "video_ref"
if m.startswith("text/"):
return "text_ref"
return "binary_ref"
def _filename_ext_for_mime(mime: str) -> str:
m = str(mime or "").strip().lower()
if m == "image/png":
return ".png"
if m in {"image/jpeg", "image/jpg"}:
return ".jpg"
if m == "image/webp":
return ".webp"
if m == "image/gif":
return ".gif"
if m == "video/mp4":
return ".mp4"
if m == "text/plain":
return ".txt"
return ".bin"
def _safe_int(raw: Any) -> int | None:
try:
if raw is None:
return None
return int(raw)
except Exception:
return None
__all__ = ["redact_embedded_image_blobs", "ingest_embedded_image_blobs_as_refs"]

View file

@ -0,0 +1,87 @@
from __future__ import annotations
from typing import Any
_MIN_B64_CHARS = 200
def ensure_no_tool_or_embedded_image_payload(*, messages: list[dict[str, Any]], path: str) -> None:
"""Guard non-turn model paths and degrade in place instead of raising.
- `role=tool` is downgraded to assistant text summary.
- Embedded image/base64 payloads are replaced with safe text placeholders.
"""
for m in messages or []:
if not isinstance(m, dict):
continue
role = str(m.get("role") or "").strip().lower()
if role == "tool":
m["role"] = "assistant"
m["content"] = f"[model_path_audit:{path}] tool payload omitted"
continue
content = m.get("content")
if _contains_embedded_image_payload(content):
m["content"] = _sanitize_content(content, path=path)
def _contains_embedded_image_payload(obj: Any) -> bool:
if isinstance(obj, str):
return _contains_large_base64_like_text(obj)
if isinstance(obj, list):
return any(_contains_embedded_image_payload(x) for x in obj)
if not isinstance(obj, dict):
return False
typ = str(obj.get("type") or "").strip().lower()
if typ in {"image", "input_image"}:
for k in ("data", "image_base64"):
v = obj.get(k)
if isinstance(v, str) and len(v.strip()) >= _MIN_B64_CHARS:
return True
for v in obj.values():
if _contains_embedded_image_payload(v):
return True
return False
def _contains_large_base64_like_text(text: str) -> bool:
s = str(text or "").strip()
if len(s) < _MIN_B64_CHARS:
return False
if s.startswith("data:") and ";base64," in s:
s = s.split(";base64,", 1)[1]
head = s[: min(4096, len(s))]
if len(head) < _MIN_B64_CHARS:
return False
allowed = frozenset("ABCDEFGHIJKLMNOPQRSTUVWXYZabcdefghijklmnopqrstuvwxyz0123456789+/=\n\r-_")
noise = sum(1 for ch in head if ch not in allowed)
# Similar heuristic to media redaction: mostly base64 alphabet over a long span.
return noise <= max(4, len(head) // 200)
def _sanitize_content(content: Any, *, path: str) -> Any:
if isinstance(content, str):
if _contains_large_base64_like_text(content):
return f"[model_path_audit:{path}] base64 payload omitted"
return content
if isinstance(content, list):
out: list[Any] = []
for item in content:
if isinstance(item, dict):
typ = str(item.get("type") or "").strip().lower()
if typ in {"image", "input_image"}:
out.append({"type": "text", "text": f"[model_path_audit:{path}] image payload omitted"})
continue
out.append(_sanitize_content(item, path=path))
return out
if isinstance(content, dict):
out: dict[str, Any] = {}
for k, v in content.items():
if str(k) in {"data", "image_base64"} and isinstance(v, str) and _contains_large_base64_like_text(v):
out[k] = f"[model_path_audit:{path}] payload omitted"
continue
out[k] = _sanitize_content(v, path=path)
return out
return content
__all__ = ["ensure_no_tool_or_embedded_image_payload"]

View file

@ -21,6 +21,7 @@ from oclaw.platform.persistence.sqlite_store import SqliteStore
from oclaw.runtime.tools.base import ToolRegistry
from oclaw.platform.llm.chat_models import LLMToolCall
from oclaw.runtime.tools.tool_validation import validate_tool_arguments
from oclaw.runtime.chat.media_redact import ingest_embedded_image_blobs_as_refs
from oclaw.runtime.tools.experts.workspace.workspace_base import (
workspace_path_access_scope,
workspace_write_namespace_scope,
@ -58,13 +59,15 @@ def _attachments_from_tool_result(result: Any) -> list[dict[str, Any]]:
return []
out: list[dict[str, Any]] = []
aid = str(result.get("attachment_id") or "").strip()
root_mime = str(result.get("mime") or "").strip()
if aid:
ref_type = _ref_type_for_mime(root_mime)
out.append(
{
"type": "image_ref",
"type": ref_type,
"attachment_id": aid,
"name": str(result.get("name") or "generated-image"),
"mime": str(result.get("mime") or "image/png"),
"mime": root_mime or "application/octet-stream",
"bytes": result.get("bytes"),
"width": result.get("width"),
"height": result.get("height"),
@ -91,12 +94,16 @@ def _attachments_from_tool_result(result: Any) -> list[dict[str, Any]]:
continue
r_aid = str(r.get("attachment_id") or "").strip()
if r_aid:
r_typ = str(r.get("type") or "").strip().lower()
r_mime = str(r.get("mime_type") or r.get("mime") or "").strip()
if r_typ not in {"image_ref", "video_ref", "text_ref", "binary_ref"}:
r_typ = _ref_type_for_mime(r_mime)
out.append(
{
"type": "image_ref",
"type": r_typ,
"attachment_id": r_aid,
"name": str(r.get("name") or "generated-image"),
"mime": str(r.get("mime") or "image/png"),
"mime": r_mime or "application/octet-stream",
"bytes": r.get("bytes"),
"width": r.get("width"),
"height": r.get("height"),
@ -110,15 +117,18 @@ def _attachments_from_tool_result(result: Any) -> list[dict[str, Any]]:
if not isinstance(item, dict):
continue
typ = str(item.get("type") or "").strip().lower()
if typ in {"image", "input_image"}:
b64 = item.get("image_base64") or item.get("data")
if isinstance(b64, str) and b64.strip():
if typ in {"image_ref", "video_ref", "text_ref", "binary_ref"}:
a_id = str(item.get("attachment_id") or "").strip()
if a_id:
out.append(
{
"type": "image",
"data": b64.strip(),
"mime": str(item.get("mime_type") or item.get("mime") or "image/png"),
"name": str(item.get("name") or "tool-image"),
"type": typ,
"attachment_id": a_id,
"mime": str(item.get("mime_type") or item.get("mime") or "application/octet-stream"),
"name": str(item.get("name") or "tool-attachment"),
"bytes": item.get("bytes"),
"width": item.get("width"),
"height": item.get("height"),
}
)
elif typ == "image_url":
@ -132,7 +142,7 @@ def _attachments_from_tool_result(result: Any) -> list[dict[str, Any]]:
a.get("attachment_id")
or a.get("pointer_uri")
or a.get("url")
or (f"b64:{a.get('mime')}:{len(str(a.get('data') or ''))}" if a.get("data") else "")
or ""
).strip()
if not k or k in seen:
continue
@ -828,6 +838,10 @@ class ToolExecutor:
)
result, duration_ms = results_by_id[tc.id]
result = normalize_tool_result(result)
persisted_result, ingested_refs = ingest_embedded_image_blobs_as_refs(
result,
filename_prefix=f"{str(tc.name or 'tool')}-{str(tc.id or '')}",
)
logger.info(
"tool_runtime tool session=%s name=%s duration_ms=%d ok=%s",
ctx.session_id[:12],
@ -840,7 +854,7 @@ class ToolExecutor:
session_id=ctx.session_id,
tool_name=tc.name,
args=tc.arguments,
result=result,
result=persisted_result,
specialist=ctx.specialist,
duration_ms=duration_ms,
)
@ -849,7 +863,7 @@ class ToolExecutor:
# until the turn finishes, so current-round model context remains lossless.
t_trunc = time.perf_counter()
observed_rows_this_call = int(_estimate_observed_rows(result))
result_for_llm = dict(result or {})
result_for_llm = dict(persisted_result or {})
if tc.name in _SQL_REPLAY_COMPACT_TOOL_NAMES:
current = int(local_turn_tool_name_counts.get(tc.name, 0))
current_rows = int(local_turn_tool_observed_rows.get(tc.name, 0))
@ -870,7 +884,7 @@ class ToolExecutor:
role="tool",
content=tool_content,
tool_calls={"tool_call_id": tc.id, "name": tc.name, "assistant_message_id": assistant_msg_id},
attachments=_attachments_from_tool_result(result) or None,
attachments=(_merge_attachments(_attachments_from_tool_result(persisted_result), ingested_refs) or None),
turn_uuid=ctx.turn_uuid,
event_type="tool_result",
event_payload={"tool_name": tc.name, "observed_rows": int(observed_rows_this_call)},
@ -971,3 +985,29 @@ __all__ = [
"truncate_tool_result_for_llm_messages",
"compact_turn_tool_messages_for_storage",
]
def _merge_attachments(*parts: list[dict[str, Any]]) -> list[dict[str, Any]]:
out: list[dict[str, Any]] = []
seen: set[str] = set()
for part in parts:
for a in part or []:
if not isinstance(a, dict):
continue
k = str(a.get("attachment_id") or a.get("pointer_uri") or a.get("url") or "").strip()
if not k or k in seen:
continue
seen.add(k)
out.append(a)
return out
def _ref_type_for_mime(mime: str) -> str:
m = str(mime or "").strip().lower()
if m.startswith("image/"):
return "image_ref"
if m.startswith("video/"):
return "video_ref"
if m.startswith("text/"):
return "text_ref"
return "binary_ref"