mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 01:50:44 +08:00
fix(attachments): parse channel uploads for LLM and stop echoing user files
WhatsApp/channel binary_ref uploads now expand into text_ref/tabular summaries for the model, and outbound replies only auto-attach generated image/video media instead of lookup tool text_ref results. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
fc10560052
commit
61d54b10fd
5 changed files with 215 additions and 10 deletions
|
|
@ -473,6 +473,7 @@ def _channel_attachments_for_gateway(raw: list[dict[str, Any]]) -> list[dict[str
|
|||
from pathlib import Path
|
||||
|
||||
from svc.files.attachment_assets import AttachmentAssetStore
|
||||
from svc.files.file_attachments import expand_attachment_ref, process_file_data
|
||||
|
||||
out: list[dict[str, Any]] = []
|
||||
ast = AttachmentAssetStore()
|
||||
|
|
@ -481,8 +482,7 @@ def _channel_attachments_for_gateway(raw: list[dict[str, Any]]) -> list[dict[str
|
|||
continue
|
||||
aid = str(a.get("attachment_id") or a.get("attachmentId") or "").strip().lower()
|
||||
if aid:
|
||||
t = str(a.get("type") or "").strip().lower() or "binary_ref"
|
||||
out.append({"type": t, "attachment_id": aid})
|
||||
out.extend(expand_attachment_ref(a))
|
||||
continue
|
||||
lp = str(a.get("local_path") or a.get("media_path") or "").strip()
|
||||
if not lp:
|
||||
|
|
@ -495,7 +495,15 @@ def _channel_attachments_for_gateway(raw: list[dict[str, Any]]) -> list[dict[str
|
|||
if not mime:
|
||||
mime = "application/octet-stream"
|
||||
try:
|
||||
meta = ast.save_bytes(p.read_bytes(), filename=p.name, mime=mime)
|
||||
data = p.read_bytes()
|
||||
except Exception:
|
||||
continue
|
||||
got = process_file_data(p.name, data)
|
||||
if got:
|
||||
out.extend(got)
|
||||
continue
|
||||
try:
|
||||
meta = ast.save_bytes(data, filename=p.name, mime=mime)
|
||||
except Exception:
|
||||
continue
|
||||
if kind == "image" or mime.startswith("image/"):
|
||||
|
|
@ -503,7 +511,7 @@ def _channel_attachments_for_gateway(raw: list[dict[str, Any]]) -> list[dict[str
|
|||
elif kind == "video" or mime.startswith("video/"):
|
||||
out.append({"type": "video_ref", "attachment_id": meta.attachment_id})
|
||||
else:
|
||||
out.append({"type": "binary_ref", "attachment_id": meta.attachment_id})
|
||||
out.append({"type": "binary_ref", "attachment_id": meta.attachment_id, "name": meta.name, "mime": meta.mime})
|
||||
return out
|
||||
|
||||
|
||||
|
|
@ -548,10 +556,16 @@ def _rows_since_last_user_message(rows: list[Any]) -> list[Any]:
|
|||
return list(rows[last_user_idx + 1 :])
|
||||
|
||||
|
||||
_CHANNEL_DELIVERABLE_ATTACHMENT_TYPES = frozenset(
|
||||
{"image_ref", "video_ref", "image", "input_image", "image_url"}
|
||||
)
|
||||
|
||||
|
||||
def _collect_recent_tool_attachments(*, store: Any, session_id: str) -> list[dict[str, Any]]:
|
||||
"""Fallback for channel delivery: reuse tool media produced during the current user turn only.
|
||||
|
||||
Avoids re-sending images from earlier conversation turns when the latest assistant row has no attachments.
|
||||
Only visual outbound media is eligible — not text_ref/binary_ref from read/query tools.
|
||||
"""
|
||||
sid = str(session_id or "").strip()
|
||||
if not sid:
|
||||
|
|
@ -569,15 +583,16 @@ def _collect_recent_tool_attachments(*, store: Any, session_id: str) -> list[dic
|
|||
if not atts:
|
||||
continue
|
||||
ok = False
|
||||
deliverable: list[dict[str, Any]] = []
|
||||
for a in atts:
|
||||
if not isinstance(a, dict):
|
||||
continue
|
||||
t = str(a.get("type") or "").strip().lower()
|
||||
if t in {"image_ref", "video_ref", "binary_ref", "text_ref", "image", "input_image", "image_url"}:
|
||||
if t in _CHANNEL_DELIVERABLE_ATTACHMENT_TYPES:
|
||||
ok = True
|
||||
break
|
||||
deliverable.append(a)
|
||||
if ok:
|
||||
return atts
|
||||
return deliverable
|
||||
return []
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -17,7 +17,8 @@ from svc.llm.chat_models import _normalize_image_b64_payload, gemini_openai_comp
|
|||
from runtime.chat.media_redact import redact_embedded_image_blobs
|
||||
from runtime.chat.tool_runtime import tool_llm_message_max_chars, truncate_tool_result_for_llm_messages
|
||||
from runtime.prompt_templates import render_prompt
|
||||
from svc.files.attachment_assets import attachment_id_to_data_url
|
||||
from svc.files.attachment_assets import attachment_id_to_data_url, AttachmentAssetStore
|
||||
from svc.files.file_attachments import expand_attachment_ref
|
||||
from runtime.relay_pointer import parse_pointer_uri
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
@ -316,11 +317,24 @@ def build_llm_messages(
|
|||
except Exception:
|
||||
attachments = []
|
||||
|
||||
expand_user_image_for_model = bool(i == last_user_msg_idx)
|
||||
if expand_user_image_for_model:
|
||||
expanded_atts: list[dict[str, Any]] = []
|
||||
for att in attachments or []:
|
||||
if not isinstance(att, dict):
|
||||
continue
|
||||
if str(att.get("type") or "").strip().lower() == "binary_ref":
|
||||
subs = expand_attachment_ref(att)
|
||||
if subs and any(str(s.get("type") or "") != "binary_ref" for s in subs):
|
||||
expanded_atts.extend(subs)
|
||||
continue
|
||||
expanded_atts.append(att)
|
||||
attachments = expanded_atts
|
||||
|
||||
for att in attachments or []:
|
||||
if not isinstance(att, dict):
|
||||
continue
|
||||
att_type = att.get("type")
|
||||
expand_user_image_for_model = bool(i == last_user_msg_idx)
|
||||
if att_type in ("image", "input_image"):
|
||||
if expand_user_image_for_model:
|
||||
b64 = _normalize_image_b64_payload(att.get("image_base64") or att.get("data"))
|
||||
|
|
@ -459,6 +473,29 @@ def build_llm_messages(
|
|||
meta_line += f"\n- bytes: {sz}"
|
||||
meta_line += "\n- tools: query_video_attachment"
|
||||
content_list.append({"type": "text", "text": meta_line})
|
||||
elif att_type == "binary_ref":
|
||||
aid = str(att.get("attachment_id") or "").strip()
|
||||
name = str(att.get("name") or "file")
|
||||
mime = str(att.get("mime") or "application/octet-stream")
|
||||
sz = att.get("bytes")
|
||||
if aid and (name == "file" or not mime or mime == "application/octet-stream"):
|
||||
try:
|
||||
meta = AttachmentAssetStore().get_meta(aid)
|
||||
if meta:
|
||||
if name == "file":
|
||||
name = str(meta.name or name)
|
||||
if not mime or mime == "application/octet-stream":
|
||||
mime = str(meta.mime or mime)
|
||||
if sz is None:
|
||||
sz = meta.bytes
|
||||
except Exception:
|
||||
pass
|
||||
meta_line = f"[BinaryAttachment]\n- name: {name}\n- mime: {mime}\n- attachment_id: {aid}"
|
||||
if sz:
|
||||
meta_line += f"\n- bytes: {sz}"
|
||||
meta_line += "\n- tools: attachment_local_url"
|
||||
meta_line += "\n- note: user uploaded a file; resolve or analyze it via attachment_id."
|
||||
content_list.append({"type": "text", "text": meta_line})
|
||||
elif att_type == "relay_pointer":
|
||||
p_uri = str(att.get("pointer_uri") or "").strip()
|
||||
if not p_uri:
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue