实现附件处理链路的统一引用化与可检索增强,避免大文件/多模态内容直接撑爆上下文并提升工具可用性。

本次补齐 text/image/video/archive 的标准化处理、会话级工具守卫、回放压缩、配置与文档对齐,并修复表格与流式输出相关体验问题。

Made-with: Cursor
This commit is contained in:
oliver 2026-04-26 15:25:50 +08:00
parent 9d2900db02
commit 37a2ef35f4
27 changed files with 3227 additions and 278 deletions

View file

@ -0,0 +1,75 @@
from __future__ import annotations
from typing import Any
from oclaw.platform.files.attachment_assets import attachment_id_to_data_url
from oclaw.platform.llm.image_message_client import send_image_messages
from oclaw.runtime.tools.base import ToolSpec
def query_image_attachment_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
attachment_id = str(args.get("attachment_id") or "").strip()
if not attachment_id:
return {"ok": False, "error": "attachment_id_required"}
task = str(args.get("task") or "describe").strip().lower()
question = str(args.get("question") or "").strip()
if task not in {"describe", "ocr"}:
return {"ok": False, "error": "invalid_task"}
data_url = attachment_id_to_data_url(attachment_id=attachment_id)
if not data_url:
return {"ok": False, "error": "attachment_not_found"}
prompt = (
(
"请详细描述这张图片的主要内容、对象、场景和可见文字。"
"回答请使用要点列表,避免臆测。"
)
if task == "describe"
else (
"请只提取图片中可见文字并按阅读顺序输出。"
"如果有表格,保持行列结构;不确定的内容标注为[unclear]。"
)
)
if question:
prompt = f"{prompt}\n\n用户问题:{question}"
out = send_image_messages(images=[data_url], prompt=prompt)
if not bool(out.get("ok")):
return {
"ok": False,
"error": str(out.get("error") or "image_query_failed"),
"task": task,
"attachment_id": attachment_id,
}
text = str(out.get("text") or "")
if len(text) > 12_000:
text = text[:12_000] + "\n\n...[truncated image analysis output]"
return {
"ok": True,
"task": task,
"attachment_id": attachment_id,
"text": text,
"input_kind": list(out.get("input_kind") or []),
"backend_shape": str(out.get("backend_shape") or ""),
}
return ToolSpec(
name="query_image_attachment",
description="Analyze an uploaded image by attachment_id (describe or OCR).",
parameters={
"type": "object",
"properties": {
"attachment_id": {"type": "string"},
"task": {"type": "string", "enum": ["describe", "ocr"]},
"question": {"type": "string"},
},
"required": ["attachment_id"],
"additionalProperties": False,
},
handler=handler,
read_only=True,
tags=frozenset({"image", "read"}),
)
__all__ = ["query_image_attachment_tool"]

View file

@ -0,0 +1,41 @@
from __future__ import annotations
from typing import Any
from oclaw.platform.files.text_attachment_store import query_text_document
from oclaw.runtime.tools.base import ToolSpec
def query_text_attachment_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
text_id = str(args.get("text_id") or "").strip()
if not text_id:
return {"ok": False, "error": "text_id_required"}
return query_text_document(
text_id=text_id,
query=str(args.get("query") or "").strip() or None,
top_k=int(args.get("top_k") or 5),
offset=int(args.get("offset") or 0),
)
return ToolSpec(
name="query_text_attachment",
description="Query long text attachment chunks by text_id with optional keyword search.",
parameters={
"type": "object",
"properties": {
"text_id": {"type": "string"},
"query": {"type": "string"},
"top_k": {"type": "integer", "minimum": 1, "maximum": 50},
"offset": {"type": "integer", "minimum": 0},
},
"required": ["text_id"],
"additionalProperties": False,
},
handler=handler,
read_only=True,
)
__all__ = ["query_text_attachment_tool"]

View file

@ -0,0 +1,278 @@
from __future__ import annotations
import os
import subprocess
import tempfile
import io
import json
from pathlib import Path
from typing import Any
from oclaw.platform.files.attachment_assets import AttachmentAssetStore
from oclaw.platform.files.text_attachment_store import (
DEFAULT_TEXT_CHUNK_OVERLAP,
DEFAULT_TEXT_CHUNK_SIZE,
save_text_document,
)
from oclaw.runtime.extensions.openai.api import OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL
from oclaw.runtime.tools.base import ToolSpec
def _ffmpeg_exists() -> bool:
try:
p = subprocess.run(["ffmpeg", "-version"], capture_output=True, text=True, timeout=3)
return p.returncode == 0
except Exception:
return False
def _ffprobe_json(path: Path) -> dict[str, Any] | None:
try:
p = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-print_format",
"json",
"-show_format",
"-show_streams",
str(path),
],
capture_output=True,
text=True,
timeout=8,
)
if p.returncode != 0:
return None
obj = json.loads(p.stdout or "{}")
return obj if isinstance(obj, dict) else None
except Exception:
return None
def _safe_int(raw: Any, default: int, *, min_value: int = 1, max_value: int = 2_000_000) -> int:
try:
value = int(raw)
except Exception:
return default
if value < min_value:
return default
return min(value, max_value)
def _oclaw_config_path() -> Path:
raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
if raw:
p = Path(raw)
return p if p.is_absolute() else p.resolve()
return Path(__file__).resolve().parents[4] / "oclaw.json"
def _video_transcript_chunk_defaults() -> tuple[int, int]:
size = DEFAULT_TEXT_CHUNK_SIZE
overlap = DEFAULT_TEXT_CHUNK_OVERLAP
try:
cfg_path = _oclaw_config_path()
if cfg_path.exists() and cfg_path.is_file():
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
tab = (
(((obj.get("plugins") or {}).get("entries") or {}).get("memory-wiki") or {})
.get("auto", {})
.get("attachments", {})
.get("tabular", {})
)
if isinstance(tab, dict):
size = _safe_int(tab.get("video_transcript_chunk_size"), size, min_value=200, max_value=8_000)
overlap = _safe_int(tab.get("video_transcript_chunk_overlap"), overlap, min_value=0, max_value=4_000)
except Exception:
pass
overlap = max(0, min(overlap, max(0, size - 1)))
return size, overlap
def _normalized_video_meta(ffprobe_obj: dict[str, Any] | None) -> dict[str, Any]:
out: dict[str, Any] = {}
if not isinstance(ffprobe_obj, dict):
return out
fmt = ffprobe_obj.get("format") if isinstance(ffprobe_obj.get("format"), dict) else {}
streams = ffprobe_obj.get("streams") if isinstance(ffprobe_obj.get("streams"), list) else []
if isinstance(fmt, dict) and fmt.get("duration") is not None:
try:
out["duration_sec"] = float(fmt.get("duration"))
except Exception:
pass
for s in streams:
if not isinstance(s, dict):
continue
if str(s.get("codec_type") or "") != "video":
continue
try:
if s.get("width") is not None:
out["width"] = int(s.get("width"))
if s.get("height") is not None:
out["height"] = int(s.get("height"))
except Exception:
pass
fr = str(s.get("avg_frame_rate") or s.get("r_frame_rate") or "").strip()
if fr and fr != "0/0" and "/" in fr:
try:
a, b = fr.split("/", 1)
fa = float(a)
fb = float(b)
if fb:
out["fps"] = fa / fb
except Exception:
pass
break
return out
def query_video_attachment_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
attachment_id = str(args.get("attachment_id") or "").strip()
task = str(args.get("task") or "meta").strip().lower()
lang = str(args.get("lang") or "").strip().lower()
if not attachment_id:
return {"ok": False, "error": "attachment_id_required"}
if task not in {"meta", "transcript"}:
return {"ok": False, "error": "invalid_task"}
store = AttachmentAssetStore()
p = store.get_local_path(attachment_id)
meta = store.get_meta(attachment_id)
if p is None:
return {"ok": False, "error": "attachment_not_found"}
# Basic metadata (no heavy deps). Prefer ffprobe if present.
if task == "meta":
fp = _ffprobe_json(p)
norm = _normalized_video_meta(fp)
return {
"ok": True,
"task": "meta",
"attachment_id": attachment_id,
"name": (meta.name if meta else p.name),
"mime": (meta.mime if meta else "video/*"),
"bytes": int(meta.bytes if meta else (p.stat().st_size if p.exists() else 0)),
"duration_sec": norm.get("duration_sec"),
"width": norm.get("width"),
"height": norm.get("height"),
"fps": norm.get("fps"),
"ffprobe": fp if fp else None,
"note": "Use task=transcript to extract audio transcript (requires ffmpeg + OpenAI key).",
}
# transcript: extract audio then transcribe via OpenAI, store as text chunks.
if task == "transcript":
if not _ffmpeg_exists():
return {
"ok": False,
"error": "ffmpeg_missing",
"hint": "Install ffmpeg (ffmpeg/ffprobe on PATH) to enable transcript extraction.",
}
api_key = str(os.getenv("OPENAI_API_KEY") or "").strip()
if not api_key:
return {"ok": False, "error": "OPENAI_API_KEY_missing"}
try:
with tempfile.TemporaryDirectory() as td:
wav = Path(td) / "audio.wav"
subprocess.run(
["ffmpeg", "-y", "-i", str(p), "-vn", "-ac", "1", "-ar", "16000", str(wav)],
capture_output=True,
text=True,
timeout=60,
)
if not wav.exists() or wav.stat().st_size <= 0:
return {"ok": False, "error": "audio_extract_failed"}
try:
from openai import OpenAI
except Exception as e:
return {"ok": False, "error": f"openai_package_missing: {type(e).__name__}: {e}"}
base_url = str(os.getenv("OPENAI_BASE_URL") or "").strip()
client_kwargs: dict[str, Any] = {"api_key": api_key}
if base_url:
client_kwargs["base_url"] = base_url
client = OpenAI(**client_kwargs)
model = str(args.get("model") or os.getenv("OPENAI_AUDIO_TRANSCRIPTION_MODEL") or OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL).strip()
prompt = str(args.get("prompt") or "").strip()
# OpenAI SDK expects a file-like object with a name.
wav_bytes = wav.read_bytes()
f = io.BytesIO(wav_bytes)
f.name = "audio.wav" # type: ignore[attr-defined]
# Best-effort: different gateways may accept different param names; keep it minimal.
try:
resp = client.audio.transcriptions.create( # type: ignore[attr-defined]
model=model,
file=f,
**({"prompt": prompt} if prompt else {}),
)
text = str(getattr(resp, "text", "") or "")
except Exception as e:
return {"ok": False, "error": f"transcription_failed: {type(e).__name__}: {e}"}
if not text.strip():
return {"ok": False, "error": "empty_transcript"}
# Persist transcript as a long text document so the model can query evidence by text_id.
name = str(meta.name if meta else p.name)
text_name = f"{name}.transcript.txt"
cfg_chunk_size, cfg_chunk_overlap = _video_transcript_chunk_defaults()
chunk_size = int(args.get("chunk_size") or cfg_chunk_size)
chunk_overlap = int(args.get("chunk_overlap") or cfg_chunk_overlap)
text_meta = save_text_document(
attachment_id=str(attachment_id),
name=text_name,
text=text,
source_kind="video_transcript",
chunk_size=chunk_size,
chunk_overlap=chunk_overlap,
)
preview = text[:1200]
note = (
"Use query_text_attachment(text_id=...) to retrieve exact evidence with offsets."
if lang.startswith("en")
else "后续请用 query_text_attachment(text_id=...) 按需检索证据(支持 offset/top_k/关键词)。"
)
return {
"ok": True,
"task": "transcript",
"attachment_id": attachment_id,
"name": text_name,
"text_id": str(text_meta.get("text_id") or ""),
"chars": int(text_meta.get("chars") or 0),
"chunks": int(text_meta.get("chunks") or 0),
"preview": preview,
"note": note,
}
except Exception as e:
return {"ok": False, "error": f"transcript_failed: {type(e).__name__}: {e}"}
return ToolSpec(
name="query_video_attachment",
description="Query a video attachment by attachment_id (meta or transcript).",
parameters={
"type": "object",
"properties": {
"attachment_id": {"type": "string"},
"task": {"type": "string", "enum": ["meta", "transcript"]},
"lang": {"type": "string", "description": "Optional hint: zh/en."},
"model": {"type": "string", "description": "Optional transcription model override."},
"prompt": {"type": "string", "description": "Optional transcription prompt/context."},
"chunk_size": {"type": "integer", "description": "Transcript chunk size (chars)."},
"chunk_overlap": {"type": "integer", "description": "Transcript chunk overlap (chars)."},
},
"required": ["attachment_id"],
"additionalProperties": False,
},
handler=handler,
read_only=True,
tags=frozenset({"video", "read"}),
)
__all__ = ["query_video_attachment_tool"]

View file

@ -0,0 +1,6 @@
from __future__ import annotations
from oclaw.runtime.tools.experts.generalist.image_query import query_image_attachment_tool
__all__ = ["query_image_attachment_tool"]

View file

@ -0,0 +1,6 @@
from __future__ import annotations
from oclaw.runtime.tools.experts.generalist.text_query import query_text_attachment_tool
__all__ = ["query_text_attachment_tool"]

View file

@ -0,0 +1,6 @@
from __future__ import annotations
from oclaw.runtime.tools.experts.generalist.video_query import query_video_attachment_tool
__all__ = ["query_video_attachment_tool"]