实现附件处理链路的统一引用化与可检索增强,避免大文件/多模态内容直接撑爆上下文并提升工具可用性。

本次补齐 text/image/video/archive 的标准化处理、会话级工具守卫、回放压缩、配置与文档对齐,并修复表格与流式输出相关体验问题。

Made-with: Cursor
This commit is contained in:
oliver 2026-04-26 15:25:50 +08:00
parent 9d2900db02
commit 37a2ef35f4
27 changed files with 3227 additions and 278 deletions

View file

@ -177,6 +177,9 @@ def _summarize_unpaired_tool_content(raw: str, *, cap: int) -> str:
def _truncate_tool_context(text: str, *, lang: str) -> str:
# NOTE: this is a history-replay safety clamp only.
# Never apply this to current-turn tool rows; otherwise the model may
# incorrectly infer runtime truncation from UI-facing copy.
s = str(text or "").strip()
if not s:
return s
@ -315,6 +318,39 @@ def build_llm_messages(
),
}
)
elif att_type == "text_ref":
name = str(att.get("name") or "document")
text_id = str(att.get("text_id") or "")
chars = int(att.get("chars") or 0)
chunks = int(att.get("chunks") or 0)
source_kind = str(att.get("source_kind") or "text")
aid = str(att.get("attachment_id") or "")
content_list.append(
{
"type": "text",
"text": (
f"[LongTextAttachment]\n"
f"- name: {name}\n"
f"- text_id: {text_id}\n"
f"- attachment_id: {aid}\n"
f"- source_kind: {source_kind}\n"
f"- chars: {chars}\n"
f"- chunks: {chunks}\n"
f"- tools: query_text_attachment\n"
"- note: for detailed evidence, call `query_text_attachment` with text_id."
),
}
)
elif att_type == "video_ref":
name = str(att.get("name") or "video")
mime = str(att.get("mime") or "video/*")
aid = str(att.get("attachment_id") or "")
sz = att.get("bytes")
meta_line = f"[VideoAttachment]\n- name: {name}\n- mime: {mime}\n- attachment_id: {aid}"
if sz:
meta_line += f"\n- bytes: {sz}"
meta_line += "\n- tools: query_video_attachment"
content_list.append({"type": "text", "text": meta_line})
elif att_type == "relay_pointer":
p_uri = str(att.get("pointer_uri") or "").strip()
if not p_uri:
@ -471,7 +507,7 @@ def build_llm_messages(
tool_content_out = raw_tc_content[: max(1, cap - 80)] + "\n...<truncated>"
except Exception:
tool_content_out = raw_tc_content[: max(1, cap - 80)] + "\n...<truncated>"
if tool_context_truncate_enabled:
if tool_context_truncate_enabled and str(tool_call_id) in historical_tool_ids:
# Preserve explicit guard markers from upstream context guards.
if "_tool_result_guarded" not in str(tool_content_out or ""):
tool_content_out = _truncate_tool_context(tool_content_out, lang=lang)
@ -494,8 +530,7 @@ def build_llm_messages(
r = getattr(m, "content", "") or ""
cap2 = tool_llm_message_max_chars()
pretty2 = _summarize_unpaired_tool_content(r, cap=cap2)
if tool_context_truncate_enabled:
pretty2 = _truncate_tool_context(pretty2, lang=lang)
# Unpaired tool rows are already summarized; avoid extra 50-char clipping.
out.append(
{
"role": "assistant",

View file

@ -11,6 +11,7 @@ import logging
import time
import os
import threading
from pathlib import Path
from concurrent.futures import ThreadPoolExecutor, as_completed
from concurrent.futures import TimeoutError as FuturesTimeoutError
from dataclasses import dataclass
@ -20,7 +21,10 @@ from oclaw.platform.persistence.sqlite_store import SqliteStore
from oclaw.runtime.tools.base import ToolRegistry
from oclaw.platform.llm.chat_models import LLMToolCall
from oclaw.runtime.tools.tool_validation import validate_tool_arguments
from oclaw.runtime.tools.experts.workspace.workspace_base import workspace_path_access_scope
from oclaw.runtime.tools.experts.workspace.workspace_base import (
workspace_path_access_scope,
workspace_write_namespace_scope,
)
logger = logging.getLogger(__name__)
_tool_exec_log = logging.getLogger("oclaw.tool_exec")
@ -33,6 +37,170 @@ _SQL_REPLAY_COMPACT_TOOL_NAMES = {
"run_tabular_sql",
"analyze_tabular_attachment_full_scan",
}
_TABULAR_QUERY_TOOL_NAMES = {
"query_tabular_attachment",
"run_tabular_sql",
"analyze_tabular_attachment_full_scan",
}
_TEXT_QUERY_TOOL_NAMES = {
"query_text_attachment",
}
_IMAGE_QUERY_TOOL_NAMES = {
"query_image_attachment",
}
_VIDEO_QUERY_TOOL_NAMES = {
"query_video_attachment",
}
def _message_has_tabular_ref(raw_attachments: Any) -> bool:
if raw_attachments is None:
return False
obj = raw_attachments
if isinstance(raw_attachments, str):
s = str(raw_attachments or "").strip()
if not s:
return False
try:
obj = json.loads(s)
except Exception:
return False
if isinstance(obj, dict):
items = [obj]
elif isinstance(obj, list):
items = obj
else:
return False
for it in items:
if not isinstance(it, dict):
continue
if str(it.get("type") or "").strip().lower() == "tabular_ref":
return True
return False
def _message_has_text_ref(raw_attachments: Any) -> bool:
if raw_attachments is None:
return False
obj = raw_attachments
if isinstance(raw_attachments, str):
s = str(raw_attachments or "").strip()
if not s:
return False
try:
obj = json.loads(s)
except Exception:
return False
if isinstance(obj, dict):
items = [obj]
elif isinstance(obj, list):
items = obj
else:
return False
for it in items:
if not isinstance(it, dict):
continue
if str(it.get("type") or "").strip().lower() == "text_ref":
return True
return False
def _message_has_image_ref(raw_attachments: Any) -> bool:
if raw_attachments is None:
return False
obj = raw_attachments
if isinstance(raw_attachments, str):
s = str(raw_attachments or "").strip()
if not s:
return False
try:
obj = json.loads(s)
except Exception:
return False
if isinstance(obj, dict):
items = [obj]
elif isinstance(obj, list):
items = obj
else:
return False
for it in items:
if not isinstance(it, dict):
continue
t = str(it.get("type") or "").strip().lower()
if t in {"image_ref", "image", "input_image"}:
return True
return False
def _message_has_video_ref(raw_attachments: Any) -> bool:
if raw_attachments is None:
return False
obj = raw_attachments
if isinstance(raw_attachments, str):
s = str(raw_attachments or "").strip()
if not s:
return False
try:
obj = json.loads(s)
except Exception:
return False
if isinstance(obj, dict):
items = [obj]
elif isinstance(obj, list):
items = obj
else:
return False
for it in items:
if not isinstance(it, dict):
continue
t = str(it.get("type") or "").strip().lower()
if t == "video_ref":
return True
return False
def _session_has_tabular_ref(store: Any, session_id: str, *, limit: int = 300) -> bool:
try:
rows = store.get_messages(session_id=session_id, limit=max(1, int(limit)))
except Exception:
return False
for m in rows or []:
if _message_has_tabular_ref(getattr(m, "attachments", None)):
return True
return False
def _session_has_text_ref(store: Any, session_id: str, *, limit: int = 300) -> bool:
try:
rows = store.get_messages(session_id=session_id, limit=max(1, int(limit)))
except Exception:
return False
for m in rows or []:
if _message_has_text_ref(getattr(m, "attachments", None)):
return True
return False
def _session_has_image_ref(store: Any, session_id: str, *, limit: int = 300) -> bool:
try:
rows = store.get_messages(session_id=session_id, limit=max(1, int(limit)))
except Exception:
return False
for m in rows or []:
if _message_has_image_ref(getattr(m, "attachments", None)):
return True
return False
def _session_has_video_ref(store: Any, session_id: str, *, limit: int = 300) -> bool:
try:
rows = store.get_messages(session_id=session_id, limit=max(1, int(limit)))
except Exception:
return False
for m in rows or []:
if _message_has_video_ref(getattr(m, "attachments", None)):
return True
return False
def normalize_tool_result(result: Any) -> dict[str, Any]:
@ -206,6 +374,7 @@ class ToolExecutionContext:
#: If ``get_ui_session_owner`` fails, load allowlist for this (tenant, user) from the HTTP/gateway request (``metadata``).
path_policy_tenant_id: str | None = None
path_policy_user_id: str | None = None
workspace_dir: str | None = None
turn_uuid: str | None = None
@ -235,13 +404,21 @@ class ToolExecutor:
timeout_s = 30.0
def _call() -> Any:
ws_ns = ""
raw_ws = str(ctx.workspace_dir or "").strip()
if raw_ws:
try:
wp = Path(raw_ws)
ws_ns = str(wp.name or wp.stem or "").strip()
except Exception:
ws_ns = ""
with workspace_path_access_scope(
ctx.store,
ctx.session_id,
owner_fallback_session_id=ctx.workspace_owner_session_id,
allowlist_tenant_id=ctx.path_policy_tenant_id,
allowlist_user_id=ctx.path_policy_user_id,
):
), workspace_write_namespace_scope(ws_ns):
return tool.handler(tc.arguments)
if isinstance(timeout_s, (int, float)) and float(timeout_s) > 0:
@ -400,65 +577,98 @@ class ToolExecutor:
)
return counts, observed_rows
def _compact_tool_result_for_history(
*,
tool_name: str,
result: dict[str, Any],
call_index: int,
threshold: int,
observed_rows_this_call: int,
observed_rows_cumulative_in_turn: int,
) -> dict[str, Any]:
out: dict[str, Any] = {
"ok": bool(result.get("ok")),
"_history_compacted": True,
"_history_compact_reason": "repeated_tool_calls_in_turn",
"tool_name": str(tool_name or ""),
"call_index_in_turn_for_tool": int(call_index),
"compact_threshold": int(threshold),
"_tool_observed_rows_this_call": int(observed_rows_this_call),
"_tool_observed_rows_cumulative_in_turn": int(observed_rows_cumulative_in_turn),
"result_keys": sorted(list(result.keys()))[:30],
"result_bytes": int(_json_blob_size(result)),
"hint": (
"Repeated tool calls in this turn were compacted in chat history to avoid context bloat. "
"Full payload remains in tool logs."
),
"audit_note": (
"History is compacted by system optimization. If more detail is needed, continue querying "
"with the same SQL/tool parameters from this turn."
),
}
for key in ("error_code", "error", "rows_returned", "limit", "table_id", "engine"):
if key in result:
out[key] = result.get(key)
for key in ("input_sql", "executed_sql"):
v = str(result.get(key) or "").strip()
if v:
out[key] = v[:1200]
guard = result.get("sql_guard")
if isinstance(guard, dict):
out["sql_guard"] = {
"readonly_enforced": bool(guard.get("readonly_enforced")),
"auto_limit_applied": bool(guard.get("auto_limit_applied")),
"result_row_cap": int(guard.get("result_row_cap") or 0),
}
return out
_check_stop()
if not tool_uses:
return [], {}
history_summary_threshold = int(tool_history_summary_after_calls())
turn_tool_name_counts, turn_tool_observed_rows = _load_turn_tool_stats()
local_turn_tool_name_counts: dict[str, int] = {}
local_turn_tool_observed_rows: dict[str, int] = {}
local_turn_written_tool_msgs: dict[str, list[dict[str, Any]]] = {}
has_tabular_ref_in_session = _session_has_tabular_ref(ctx.store, ctx.session_id)
has_text_ref_in_session = _session_has_text_ref(ctx.store, ctx.session_id)
has_image_ref_in_session = _session_has_image_ref(ctx.store, ctx.session_id)
has_video_ref_in_session = _session_has_video_ref(ctx.store, ctx.session_id)
results_by_id: dict[str, tuple[dict[str, Any], int]] = {}
runnable_tool_uses: list[LLMToolCall] = []
sig_seen: dict[str, int] = {}
budget = max(1, min(int(signature_budget or 2), 8))
for tc in tool_uses:
if tc.name in _TABULAR_QUERY_TOOL_NAMES and not has_tabular_ref_in_session:
results_by_id[tc.id] = (
{
"ok": False,
"error_code": "tabular_ref_missing",
"error": "tabular_ref_missing",
"hint": "No tabular_ref attachment found in this session. Query tools require table_id from tabular_ref.",
},
0,
)
_trace(
"tabular_query_guard",
{
"tool_name": tc.name,
"blocked": True,
"reason": "tabular_ref_missing",
},
)
continue
if tc.name in _TEXT_QUERY_TOOL_NAMES and not has_text_ref_in_session:
results_by_id[tc.id] = (
{
"ok": False,
"error_code": "text_ref_missing",
"error": "text_ref_missing",
"hint": "No text_ref attachment found in this session. Query tools require text_id from text_ref.",
},
0,
)
_trace(
"text_query_guard",
{
"tool_name": tc.name,
"blocked": True,
"reason": "text_ref_missing",
},
)
continue
if tc.name in _IMAGE_QUERY_TOOL_NAMES and not has_image_ref_in_session:
results_by_id[tc.id] = (
{
"ok": False,
"error_code": "image_ref_missing",
"error": "image_ref_missing",
"hint": "No image_ref attachment found in this session. Query tools require attachment_id from image_ref.",
},
0,
)
_trace(
"image_query_guard",
{
"tool_name": tc.name,
"blocked": True,
"reason": "image_ref_missing",
},
)
continue
if tc.name in _VIDEO_QUERY_TOOL_NAMES and not has_video_ref_in_session:
results_by_id[tc.id] = (
{
"ok": False,
"error_code": "video_ref_missing",
"error": "video_ref_missing",
"hint": "No video_ref attachment found in this session. Query tools require attachment_id from video_ref.",
},
0,
)
_trace(
"video_query_guard",
{
"tool_name": tc.name,
"blocked": True,
"reason": "video_ref_missing",
},
)
continue
sig = f"{tc.name}:{self._json_dumps_safe(dict(tc.arguments or {}))}"
count = int(sig_seen.get(sig, 0))
if count >= budget:
@ -539,27 +749,14 @@ class ToolExecutor:
duration_ms=duration_ms,
)
tool_log_write_ms = int((time.perf_counter() - t_db1) * 1000)
# Full payload stays in tool_log; chat history must stay under provider per-message limits.
# Keep full payload during the active turn. History compaction is deferred
# until the turn finishes, so current-round model context remains lossless.
t_trunc = time.perf_counter()
observed_rows_this_call = int(_estimate_observed_rows(result))
result_for_llm = truncate_tool_result_for_llm_messages(result)
should_compact_history = tc.name in _SQL_REPLAY_COMPACT_TOOL_NAMES
if history_summary_threshold > 0 and should_compact_history:
prior = int(turn_tool_name_counts.get(tc.name, 0))
result_for_llm = dict(result or {})
if tc.name in _SQL_REPLAY_COMPACT_TOOL_NAMES:
current = int(local_turn_tool_name_counts.get(tc.name, 0))
call_index = prior + current + 1
prior_rows = int(turn_tool_observed_rows.get(tc.name, 0))
current_rows = int(local_turn_tool_observed_rows.get(tc.name, 0))
observed_rows_cumulative_in_turn = prior_rows + current_rows + observed_rows_this_call
if call_index >= history_summary_threshold:
result_for_llm = _compact_tool_result_for_history(
tool_name=tc.name,
result=result,
call_index=call_index,
threshold=history_summary_threshold,
observed_rows_this_call=observed_rows_this_call,
observed_rows_cumulative_in_turn=observed_rows_cumulative_in_turn,
)
local_turn_tool_name_counts[tc.name] = current + 1
local_turn_tool_observed_rows[tc.name] = current_rows + observed_rows_this_call
trunc_ms = int((time.perf_counter() - t_trunc) * 1000)
@ -576,47 +773,6 @@ class ToolExecutor:
)
tool_msg_write_ms = int((time.perf_counter() - t_db2) * 1000)
tool_messages.append({"role": "tool", "tool_call_id": tc.id, "content": tool_content, "name": tc.name})
tool_messages_idx = len(tool_messages) - 1
call_index_for_tool = int(turn_tool_name_counts.get(tc.name, 0)) + int(local_turn_tool_name_counts.get(tc.name, 0))
local_turn_written_tool_msgs.setdefault(tc.name, []).append(
{
"message_id": int(getattr(msg_row, "id", 0) or 0),
"tool_messages_idx": int(tool_messages_idx),
"result": dict(result or {}),
"observed_rows": int(observed_rows_this_call),
"call_index": int(call_index_for_tool),
"compacted": bool(isinstance(result_for_llm, dict) and result_for_llm.get("_history_compacted")),
}
)
# When threshold is reached for one SQL tool in the turn, retro-compact earlier same-tool tool messages too.
if history_summary_threshold > 0 and should_compact_history and call_index_for_tool >= history_summary_threshold:
running_rows = int(turn_tool_observed_rows.get(tc.name, 0))
entries = list(local_turn_written_tool_msgs.get(tc.name) or [])
for ent in entries:
running_rows += int(ent.get("observed_rows") or 0)
compacted_payload = _compact_tool_result_for_history(
tool_name=tc.name,
result=dict(ent.get("result") or {}),
call_index=int(ent.get("call_index") or 0),
threshold=history_summary_threshold,
observed_rows_this_call=int(ent.get("observed_rows") or 0),
observed_rows_cumulative_in_turn=int(running_rows),
)
compacted_content = self._json_dumps_safe(compacted_payload)
if not bool(ent.get("compacted")):
try:
ctx.store.update_message_content(
session_id=ctx.session_id,
message_id=int(ent.get("message_id") or 0),
content=compacted_content,
event_payload={"tool_name": tc.name, "observed_rows": int(ent.get("observed_rows") or 0)},
)
except Exception:
pass
ent["compacted"] = True
ti = int(ent.get("tool_messages_idx") or -1)
if 0 <= ti < len(tool_messages):
tool_messages[ti]["content"] = compacted_content
_trace(
"tool_result",
{
@ -650,6 +806,57 @@ class ToolExecutor:
on_tool_ui("tool_use_result", payload)
return tool_messages, results_by_id
def compact_turn_tool_messages_for_storage(
*,
store: Any,
session_id: str,
turn_uuid: str | None,
) -> dict[str, int]:
"""Compact persisted tool messages after turn completion.
This intentionally runs *after* the active turn so current-round model
context is not affected by truncation/compaction.
"""
tid = str(turn_uuid or "").strip()
if not tid:
return {"scanned": 0, "updated": 0}
try:
rows = store.get_messages(session_id=session_id, limit=800)
except Exception:
return {"scanned": 0, "updated": 0}
scanned = 0
updated = 0
for m in rows or []:
if str(getattr(m, "role", "") or "") != "tool":
continue
if str(getattr(m, "turn_uuid", "") or "") != tid:
continue
scanned += 1
raw = str(getattr(m, "content", "") or "")
if not raw.strip():
continue
try:
obj = json.loads(raw)
except Exception:
continue
if not isinstance(obj, dict):
continue
compacted = truncate_tool_result_for_llm_messages(obj)
if compacted == obj:
continue
try:
store.update_message_content(
session_id=session_id,
message_id=int(getattr(m, "id", 0) or 0),
content=json.dumps(compacted, ensure_ascii=False, default=str),
event_payload=getattr(m, "event_payload", None),
)
updated += 1
except Exception:
continue
return {"scanned": int(scanned), "updated": int(updated)}
__all__ = [
"ToolExecutionConfig",
"ToolExecutionContext",
@ -658,4 +865,5 @@ __all__ = [
"partition_tool_use_batches",
"tool_llm_message_max_chars",
"truncate_tool_result_for_llm_messages",
"compact_turn_tool_messages_for_storage",
]

View file

@ -8,6 +8,7 @@ import uuid
import copy
import threading
from dataclasses import dataclass
from pathlib import Path
from types import SimpleNamespace
from typing import Any, Callable, Optional
@ -24,6 +25,8 @@ from oclaw.runtime.tools.base import ToolRegistry
from oclaw.runtime.hooks_runtime import trigger_hook_event
_OCLAW_TOOL_RESULT_HARD_CAP_CHARS = 24_000
_OCLAW_ATTACHMENT_TEXT_REPLAY_CAP_CHARS = 4_000
_OCLAW_IMAGE_TOOL_RESULT_REPLAY_CAP_CHARS = 4_000
_DIRECT_LOOP_OC_STAGE: dict[str, str] = {
"tool_wire_filter": "wire_filter",
@ -39,6 +42,82 @@ _TOOL_WIRE_LAST_WARM_ROLES: tuple[str, ...] = ()
_TOOL_WIRE_LAST_WARM_COUNT: int = 0
def _safe_int(raw: Any, default: int, *, min_value: int = 1, max_value: int = 2_000_000) -> int:
try:
value = int(raw)
except Exception:
return default
if value < min_value:
return default
return min(value, max_value)
def _oclaw_config_path() -> Path:
raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
if raw:
p = Path(raw)
return p if p.is_absolute() else p.resolve()
return Path(__file__).resolve().parents[1] / "oclaw.json"
def _image_tool_result_replay_cap_chars(store: Any) -> int:
default = _OCLAW_IMAGE_TOOL_RESULT_REPLAY_CAP_CHARS
raw_setting = ""
try:
raw_setting = str(store.get_setting("AIA_IMAGE_TOOL_RESULT_REPLAY_CAP_CHARS") or "").strip()
except Exception:
raw_setting = ""
if raw_setting:
return _safe_int(raw_setting, default, min_value=600, max_value=30_000)
raw_env = str(os.getenv("AIA_IMAGE_TOOL_RESULT_REPLAY_CAP_CHARS") or "").strip()
if raw_env:
return _safe_int(raw_env, default, min_value=600, max_value=30_000)
try:
cfg_path = _oclaw_config_path()
if cfg_path.exists() and cfg_path.is_file():
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
tab = (
(((obj.get("plugins") or {}).get("entries") or {}).get("memory-wiki") or {})
.get("auto", {})
.get("attachments", {})
.get("tabular", {})
)
if isinstance(tab, dict):
return _safe_int(tab.get("image_result_replay_cap_chars"), default, min_value=600, max_value=30_000)
except Exception:
pass
return default
def _video_tool_result_replay_cap_chars(store: Any) -> int:
default = 4_000
raw_setting = ""
try:
raw_setting = str(store.get_setting("AIA_VIDEO_TOOL_RESULT_REPLAY_CAP_CHARS") or "").strip()
except Exception:
raw_setting = ""
if raw_setting:
return _safe_int(raw_setting, default, min_value=600, max_value=30_000)
raw_env = str(os.getenv("AIA_VIDEO_TOOL_RESULT_REPLAY_CAP_CHARS") or "").strip()
if raw_env:
return _safe_int(raw_env, default, min_value=600, max_value=30_000)
try:
cfg_path = _oclaw_config_path()
if cfg_path.exists() and cfg_path.is_file():
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
tab = (
(((obj.get("plugins") or {}).get("entries") or {}).get("memory-wiki") or {})
.get("auto", {})
.get("attachments", {})
.get("tabular", {})
)
if isinstance(tab, dict):
return _safe_int(tab.get("video_result_replay_cap_chars"), default, min_value=600, max_value=30_000)
except Exception:
pass
return default
def _tool_wire_freeze_enabled(store: Any) -> bool:
raw = ""
try:
@ -218,6 +297,7 @@ def _guard_tool_results_for_llm_context(
run_id: str | None = None,
attempt_no: int | None = None,
lang: str = "",
active_turn_uuid: str | None = None,
) -> list[Any]:
"""Hard-guard overlarge `role=tool` message contents before sending to model.
@ -225,28 +305,87 @@ def _guard_tool_results_for_llm_context(
in-flight LLM context to prevent provider context overflow spirals.
"""
cap = max(4096, min(int(hard_cap_chars or _OCLAW_TOOL_RESULT_HARD_CAP_CHARS), 500_000))
image_cap = _image_tool_result_replay_cap_chars(store)
video_cap = _video_tool_result_replay_cap_chars(store)
out: list[Any] = []
for m in store_messages or []:
role = str(getattr(m, "role", "") or "")
if role != "tool":
out.append(m)
continue
raw = str(getattr(m, "content", "") or "")
if len(raw) <= cap:
if str(getattr(m, "turn_uuid", "") or "") == str(active_turn_uuid or "") and str(active_turn_uuid or "").strip():
out.append(m)
continue
# Best-effort parse tool JSON for a minimal summary.
raw = str(getattr(m, "content", "") or "")
# Best-effort parse tool JSON for image-query specific guard and overflow metadata.
ok = None
error_code = ""
error = ""
obj: dict[str, Any] | None = None
try:
obj = json.loads(raw)
if isinstance(obj, dict):
parsed = json.loads(raw)
if isinstance(parsed, dict):
obj = parsed
ok = obj.get("ok")
error_code = str(obj.get("error_code") or "").strip()
error = str(obj.get("error") or "").strip()
except Exception:
obj = None
if isinstance(obj, dict):
task = str(obj.get("task") or "").strip().lower()
text = str(obj.get("text") or "")
has_attachment_id = bool(str(obj.get("attachment_id") or "").strip())
# Guard image describe/OCR result replay aggressively to avoid long visual transcripts
# occupying context across future rounds.
if task in {"describe", "ocr"} and has_attachment_id and len(text) > image_cap:
preview = text[:image_cap] + "\n...<image_tool_result_truncated_for_context_replay>"
guarded_obj = dict(obj)
guarded_obj["text"] = preview
guarded_obj["_image_tool_result_guarded"] = True
guarded_obj["image_result_original_chars"] = len(text)
guarded_obj["image_result_replay_cap_chars"] = image_cap
guarded_obj["image_result_hint"] = (
"Image analysis result was truncated for context replay. "
"Refine query_image_attachment(question=...) for narrower evidence. / "
"图片分析结果在上下文回放中已截断,请缩小 query_image_attachment 的问题范围。"
)
guarded = _json_dumps_safe(guarded_obj)
out.append(
SimpleNamespace(
id=getattr(m, "id", 0),
session_id=getattr(m, "session_id", session_id),
role="tool",
content=guarded,
tool_calls=getattr(m, "tool_calls", None),
timestamp=getattr(m, "timestamp", ""),
attachments=getattr(m, "attachments", None),
)
)
continue
# Guard video transcript replay similarly (usually long).
if str(obj.get("task") or "").strip().lower() == "transcript" and has_attachment_id and len(text) > video_cap:
preview = text[:video_cap] + "\n...<video_tool_result_truncated_for_context_replay>"
guarded_obj = dict(obj)
guarded_obj["text"] = preview
guarded_obj["_video_tool_result_guarded"] = True
guarded_obj["video_result_original_chars"] = len(text)
guarded_obj["video_result_replay_cap_chars"] = video_cap
guarded = _json_dumps_safe(guarded_obj)
out.append(
SimpleNamespace(
id=getattr(m, "id", 0),
session_id=getattr(m, "session_id", session_id),
role="tool",
content=guarded,
tool_calls=getattr(m, "tool_calls", None),
timestamp=getattr(m, "timestamp", ""),
attachments=getattr(m, "attachments", None),
)
)
continue
if len(raw) <= cap:
out.append(m)
continue
preview = raw[: max(1, min(4000, cap - 400))] + "\n...<tool_result_guard_truncated>"
guarded_obj = {
"ok": bool(ok) if ok is not None else None,
@ -294,6 +433,118 @@ def _guard_tool_results_for_llm_context(
return out
def _guard_text_attachments_for_llm_context(
*,
store_messages: list[Any],
cap_chars: int,
active_turn_uuid: str | None = None,
) -> list[Any]:
"""Guard overlarge user text attachments for model context replay.
This does NOT rewrite DB history. It only guards the in-flight LLM context to
prevent large attachments from overwhelming context windows.
"""
cap = max(800, min(int(cap_chars or _OCLAW_ATTACHMENT_TEXT_REPLAY_CAP_CHARS), 80_000))
out: list[Any] = []
for m in store_messages or []:
role = str(getattr(m, "role", "") or "")
if role != "user":
out.append(m)
continue
# Never guard the active user turn.
if str(getattr(m, "turn_uuid", "") or "") == str(active_turn_uuid or "") and str(active_turn_uuid or "").strip():
out.append(m)
continue
raw_att = getattr(m, "attachments", None)
if not raw_att:
out.append(m)
continue
try:
att_obj = json.loads(raw_att) if isinstance(raw_att, str) else raw_att
except Exception:
out.append(m)
continue
if isinstance(att_obj, dict):
atts = [att_obj]
elif isinstance(att_obj, list):
atts = att_obj
else:
out.append(m)
continue
has_text_ref = any(
isinstance(a, dict) and str(a.get("type") or "").strip().lower() == "text_ref" for a in atts
)
changed = False
next_atts: list[dict[str, Any]] = []
for a in atts:
if not isinstance(a, dict):
continue
if str(a.get("type") or "").strip().lower() != "text":
next_atts.append(a)
continue
content = str(a.get("content") or "")
# If this user message already has a text_ref, keep inline text very small in replay context.
# The model can retrieve evidence via query_text_attachment(text_id=...).
if has_text_ref and content:
changed = True
name = str(a.get("name") or "attachment")
next_atts.append(
{
**a,
"content": (
"# Attachment (collapsed; text_ref available)\n"
f"- name: {name}\n"
"- note: use `query_text_attachment` with `text_id` from `text_ref` for details.\n"
"...<attachment_collapsed_for_context_replay>"
),
"_attachment_context_guarded": True,
"_attachment_context_collapsed": True,
}
)
continue
if len(content) <= cap:
next_atts.append(a)
continue
changed = True
name = str(a.get("name") or "attachment")
hint_lines = [
"# Attachment (summarized for context replay)",
f"- name: {name}",
f"- original_chars: {len(content)}",
f"- replay_cap_chars: {cap}",
]
if has_text_ref:
hint_lines.append("- note: use `query_text_attachment` with `text_id` from `text_ref` for details.")
else:
hint_lines.append("- note: attachment was large; re-upload or provide a smaller excerpt if needed.")
preview = content[: min(1200, cap)]
next_atts.append(
{
**a,
"content": "\n".join(hint_lines) + "\n\n## Preview\n" + preview + "\n\n...<attachment_truncated_for_context_replay>",
"_attachment_context_guarded": True,
}
)
if not changed:
out.append(m)
continue
out.append(
SimpleNamespace(
id=getattr(m, "id", 0),
session_id=getattr(m, "session_id", ""),
role="user",
content=getattr(m, "content", ""),
tool_calls=getattr(m, "tool_calls", None),
timestamp=getattr(m, "timestamp", ""),
attachments=next_atts,
turn_uuid=getattr(m, "turn_uuid", ""),
event_type=getattr(m, "event_type", ""),
event_payload=getattr(m, "event_payload", None),
)
)
return out
def _check_stop(should_stop: Optional[Callable[[], bool]]) -> None:
if should_stop and should_stop():
raise RuntimeError("generation interrupted by user")
@ -318,6 +569,7 @@ def _build_model_context(
skill_binding_role: str | None = None,
user_text: str = "",
prompt_build_context: dict[str, Any] | None = None,
active_turn_uuid: str | None = None,
) -> list[dict[str, Any]]:
rows = store.get_messages(session_id=session_id, limit=int(max_messages))
rows = _guard_tool_results_for_llm_context(
@ -330,6 +582,12 @@ def _build_model_context(
run_id=run_id,
attempt_no=attempt_no,
lang=lang,
active_turn_uuid=active_turn_uuid,
)
rows = _guard_text_attachments_for_llm_context(
store_messages=rows,
cap_chars=_OCLAW_ATTACHMENT_TEXT_REPLAY_CAP_CHARS,
active_turn_uuid=active_turn_uuid,
)
final_system = build_oclaw_executor_system_prompt(
store=store,
@ -672,10 +930,11 @@ def run_oclaw_direct_loop(
skill_binding_role: str | None = None,
wire_policy_role: str | None = None,
prompt_build_context: dict[str, Any] | None = None,
turn_uuid: str | None = None,
) -> TurnRunOutcome:
"""A minimal oclaw-style loop: model -> tool_uses -> execute -> tool_results -> continue."""
_check_stop(should_stop)
turn_uuid = str(uuid.uuid4())
turn_uuid = str(turn_uuid or "").strip() or str(uuid.uuid4())
if persist_user_message:
store.add_message(
session_id=session_id,
@ -689,10 +948,12 @@ def run_oclaw_direct_loop(
skill_exec = SkillExecutor(config=ToolExecutionConfig(max_workers=max(1, min(int(max_tool_workers or 8), 32))))
tool_traces: list[dict[str, Any]] = []
final_text = ""
hit_tool_round_limit = False
base_url = str(getattr(model, "base_url", "") or "")
for round_idx in range(max(1, int(max_tool_rounds or 1))):
max_rounds = max(1, int(max_tool_rounds or 1))
for round_idx in range(max_rounds):
_check_stop(should_stop)
if on_progress:
on_progress(f"oclaw: think ({round_idx + 1})…")
@ -715,6 +976,7 @@ def run_oclaw_direct_loop(
skill_binding_role=skill_binding_role,
user_text=str(user_text or ""),
prompt_build_context=prompt_build_context,
active_turn_uuid=turn_uuid,
)
llm_tools = _prepare_llm_tools(
store=store,
@ -744,6 +1006,10 @@ def run_oclaw_direct_loop(
final_text = step.assistant_text
if not step.llm_tool_calls:
break
if round_idx == (max_rounds - 1):
# Reached tool-round cap with pending tool calls. Execute this batch, then
# force one no-tool synthesis pass to guarantee a visible assistant body.
hit_tool_round_limit = True
elapsed_ms, results_by_id = _execute_tool_step(
skill_exec=skill_exec,
@ -783,6 +1049,42 @@ def run_oclaw_direct_loop(
if on_progress:
on_progress(f"oclaw: tools done ({elapsed_ms}ms)")
if hit_tool_round_limit:
_check_stop(should_stop)
if on_progress:
on_progress("oclaw: finalize…")
msgs = _build_model_context(
store=store,
session_id=session_id,
max_messages=max_messages,
system_prompt=system_prompt,
model=model,
lang=lang,
memory_context=memory_context,
trace_id=trace_id,
parent_span_id=parent_span_id,
tools=tools,
base_url=base_url,
run_id=run_id,
attempt_no=attempt_no,
workspace_dir=workspace_dir,
skill_binding_role=skill_binding_role,
user_text=str(user_text or ""),
prompt_build_context=prompt_build_context,
active_turn_uuid=turn_uuid,
)
# Final pass forbids extra tool calls; model must synthesize answer.
resp = model.chat(msgs, [], on_token=on_token)
step = _persist_assistant_step(
store=store,
session_id=session_id,
turn_uuid=turn_uuid,
assistant_text=str(getattr(resp, "content", "") or ""),
reasoning_text=str(getattr(resp, "reasoning_content", "") or ""),
llm_tool_calls=[],
)
final_text = step.assistant_text
return TurnRunOutcome(
final_text=str(final_text or ""),
tool_traces=tuple(tool_traces),

View file

@ -34,6 +34,7 @@ from oclaw.runtime.memory_stage import after_turn_memory
from oclaw.runtime.router import decide_route
from oclaw.runtime.worker import ensure_worker_started
from oclaw.runtime.orchestration.trace import new_span_id, new_trace_id
from oclaw.runtime.chat.tool_runtime import compact_turn_tool_messages_for_storage
_OC_STAGE_BY_EVENT: dict[str, str] = {
"gateway_received": "ingress",
@ -91,6 +92,7 @@ class OclawGateway:
start = t.find("{")
if start < 0:
return None
executed_turn_uuid = ""
try:
obj, _end = json.JSONDecoder().raw_decode(t[start:])
except Exception:
@ -414,6 +416,36 @@ class OclawGateway:
return True
return False
@staticmethod
def _has_text_ref_attachments(msg: StandardMessage) -> bool:
atts = msg.attachments if isinstance(msg.attachments, list) else []
for a in atts:
if isinstance(a, dict) and str(a.get("type") or "").strip().lower() == "text_ref":
return True
return False
@staticmethod
def _has_image_ref_attachments(msg: StandardMessage) -> bool:
atts = msg.attachments if isinstance(msg.attachments, list) else []
for a in atts:
if not isinstance(a, dict):
continue
t = str(a.get("type") or "").strip().lower()
if t in {"image_ref", "image", "input_image"}:
return True
return False
@staticmethod
def _has_video_ref_attachments(msg: StandardMessage) -> bool:
atts = msg.attachments if isinstance(msg.attachments, list) else []
for a in atts:
if not isinstance(a, dict):
continue
t = str(a.get("type") or "").strip().lower()
if t == "video_ref":
return True
return False
@staticmethod
def _tabular_query_system_hint(lang: str) -> str:
limits = OclawGateway._tabular_limits_from_config()
@ -431,6 +463,38 @@ class OclawGateway:
"如果需要更多行或更细节,请通过数据库工具(`query_tabular_attachment` / `run_tabular_sql`)结合 table_id 查询。"
)
@staticmethod
def _text_query_system_hint(lang: str) -> str:
if str(lang or "").startswith("en"):
return (
"For long text attachments: context may contain only summary/preview. "
"For detailed evidence, use `query_text_attachment` with `text_id` from `text_ref` attachment."
)
return (
"对于长文本附件:上下文可能只包含摘要/预览。"
"如需细节证据,请使用 `text_ref` 提供的 text_id 调用 `query_text_attachment`。"
)
@staticmethod
def _image_query_system_hint(lang: str) -> str:
if str(lang or "").startswith("en"):
return (
"For image attachments: use `query_image_attachment` with attachment_id "
"for OCR/description when visual evidence is required."
)
return (
"对于图片附件:如需 OCR 或图像细节,请使用 attachment_id 调用 `query_image_attachment`。"
)
@staticmethod
def _video_query_system_hint(lang: str) -> str:
if str(lang or "").startswith("en"):
return (
"For video attachments: use `query_video_attachment` with attachment_id from `video_ref` "
"to get metadata or transcript (if enabled)."
)
return "对于视频附件:请使用 `video_ref` 提供的 attachment_id 调用 `query_video_attachment` 获取元信息/转写。"
@staticmethod
def _tabular_limits_from_config() -> dict[str, int]:
cfg_path_raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
@ -767,6 +831,25 @@ class OclawGateway:
},
started_at=t0,
)
if str(manager_instruction_text or "").strip() and not bool(manager_memory_mode):
try:
assignment_title = "Task assignment" if str(lang or "").startswith("en") else "任务分配"
assignment_text = (
f"{assignment_title}\n"
f"specialist={str(manager_specialist or '')}\n"
f"instruction:\n{str(manager_instruction_text or '').strip()}"
)
self.store.add_message(
session_id=msg.session_id,
tenant_id=msg.tenant_id,
user_id=msg.user_id,
role="assistant",
content=assignment_text,
tool_calls=None,
event_type="reasoning",
)
except Exception:
pass
# Build specialist/dynamic executor and dispatch only manager instruction to it.
specialist_input_msg = StandardMessage(
session_id=msg.session_id,
@ -984,6 +1067,12 @@ class OclawGateway:
sys_prompt = str(getattr(selected_executor, "system_prompt", "") or "")
if self._has_tabular_ref_attachments(msg):
sys_prompt = f"{sys_prompt}\n\n{self._tabular_query_system_hint(lang)}".strip()
if self._has_text_ref_attachments(msg):
sys_prompt = f"{sys_prompt}\n\n{self._text_query_system_hint(lang)}".strip()
if self._has_image_ref_attachments(msg):
sys_prompt = f"{sys_prompt}\n\n{self._image_query_system_hint(lang)}".strip()
if self._has_video_ref_attachments(msg):
sys_prompt = f"{sys_prompt}\n\n{self._video_query_system_hint(lang)}".strip()
def _get_int_setting(key: str, default: int, lo: int, hi: int) -> int:
try:
@ -1042,6 +1131,7 @@ class OclawGateway:
wire_policy_role="manager" if interaction_mode == "comprehensive" else str(requested_specialist),
),
)
executed_turn_uuid = str(getattr(core_out.outcome, "turn_uuid", "") or "")
specialist_reply = str(core_out.outcome.final_text or "")
if manager_memory_mode:
# manager_memory: write memory silently, but keep dialog output independent.
@ -1095,6 +1185,15 @@ class OclawGateway:
)
except Exception:
pass
if str(executed_turn_uuid or "").strip():
try:
compact_turn_tool_messages_for_storage(
store=self.store,
session_id=msg.session_id,
turn_uuid=executed_turn_uuid,
)
except Exception:
pass
_trace_local(
event_type="response_sent",
payload={"ok": bool(str(reply or "").strip()), "elapsed_ms": elapsed_ms, "mode": "sync_direct"},

View file

@ -0,0 +1,75 @@
from __future__ import annotations
from typing import Any
from oclaw.platform.files.attachment_assets import attachment_id_to_data_url
from oclaw.platform.llm.image_message_client import send_image_messages
from oclaw.runtime.tools.base import ToolSpec
def query_image_attachment_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
attachment_id = str(args.get("attachment_id") or "").strip()
if not attachment_id:
return {"ok": False, "error": "attachment_id_required"}
task = str(args.get("task") or "describe").strip().lower()
question = str(args.get("question") or "").strip()
if task not in {"describe", "ocr"}:
return {"ok": False, "error": "invalid_task"}
data_url = attachment_id_to_data_url(attachment_id=attachment_id)
if not data_url:
return {"ok": False, "error": "attachment_not_found"}
prompt = (
(
"请详细描述这张图片的主要内容、对象、场景和可见文字。"
"回答请使用要点列表,避免臆测。"
)
if task == "describe"
else (
"请只提取图片中可见文字并按阅读顺序输出。"
"如果有表格,保持行列结构;不确定的内容标注为[unclear]。"
)
)
if question:
prompt = f"{prompt}\n\n用户问题:{question}"
out = send_image_messages(images=[data_url], prompt=prompt)
if not bool(out.get("ok")):
return {
"ok": False,
"error": str(out.get("error") or "image_query_failed"),
"task": task,
"attachment_id": attachment_id,
}
text = str(out.get("text") or "")
if len(text) > 12_000:
text = text[:12_000] + "\n\n...[truncated image analysis output]"
return {
"ok": True,
"task": task,
"attachment_id": attachment_id,
"text": text,
"input_kind": list(out.get("input_kind") or []),
"backend_shape": str(out.get("backend_shape") or ""),
}
return ToolSpec(
name="query_image_attachment",
description="Analyze an uploaded image by attachment_id (describe or OCR).",
parameters={
"type": "object",
"properties": {
"attachment_id": {"type": "string"},
"task": {"type": "string", "enum": ["describe", "ocr"]},
"question": {"type": "string"},
},
"required": ["attachment_id"],
"additionalProperties": False,
},
handler=handler,
read_only=True,
tags=frozenset({"image", "read"}),
)
__all__ = ["query_image_attachment_tool"]

View file

@ -0,0 +1,41 @@
from __future__ import annotations
from typing import Any
from oclaw.platform.files.text_attachment_store import query_text_document
from oclaw.runtime.tools.base import ToolSpec
def query_text_attachment_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
text_id = str(args.get("text_id") or "").strip()
if not text_id:
return {"ok": False, "error": "text_id_required"}
return query_text_document(
text_id=text_id,
query=str(args.get("query") or "").strip() or None,
top_k=int(args.get("top_k") or 5),
offset=int(args.get("offset") or 0),
)
return ToolSpec(
name="query_text_attachment",
description="Query long text attachment chunks by text_id with optional keyword search.",
parameters={
"type": "object",
"properties": {
"text_id": {"type": "string"},
"query": {"type": "string"},
"top_k": {"type": "integer", "minimum": 1, "maximum": 50},
"offset": {"type": "integer", "minimum": 0},
},
"required": ["text_id"],
"additionalProperties": False,
},
handler=handler,
read_only=True,
)
__all__ = ["query_text_attachment_tool"]

View file

@ -0,0 +1,278 @@
from __future__ import annotations
import os
import subprocess
import tempfile
import io
import json
from pathlib import Path
from typing import Any
from oclaw.platform.files.attachment_assets import AttachmentAssetStore
from oclaw.platform.files.text_attachment_store import (
DEFAULT_TEXT_CHUNK_OVERLAP,
DEFAULT_TEXT_CHUNK_SIZE,
save_text_document,
)
from oclaw.runtime.extensions.openai.api import OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL
from oclaw.runtime.tools.base import ToolSpec
def _ffmpeg_exists() -> bool:
try:
p = subprocess.run(["ffmpeg", "-version"], capture_output=True, text=True, timeout=3)
return p.returncode == 0
except Exception:
return False
def _ffprobe_json(path: Path) -> dict[str, Any] | None:
try:
p = subprocess.run(
[
"ffprobe",
"-v",
"error",
"-print_format",
"json",
"-show_format",
"-show_streams",
str(path),
],
capture_output=True,
text=True,
timeout=8,
)
if p.returncode != 0:
return None
obj = json.loads(p.stdout or "{}")
return obj if isinstance(obj, dict) else None
except Exception:
return None
def _safe_int(raw: Any, default: int, *, min_value: int = 1, max_value: int = 2_000_000) -> int:
try:
value = int(raw)
except Exception:
return default
if value < min_value:
return default
return min(value, max_value)
def _oclaw_config_path() -> Path:
raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
if raw:
p = Path(raw)
return p if p.is_absolute() else p.resolve()
return Path(__file__).resolve().parents[4] / "oclaw.json"
def _video_transcript_chunk_defaults() -> tuple[int, int]:
size = DEFAULT_TEXT_CHUNK_SIZE
overlap = DEFAULT_TEXT_CHUNK_OVERLAP
try:
cfg_path = _oclaw_config_path()
if cfg_path.exists() and cfg_path.is_file():
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
tab = (
(((obj.get("plugins") or {}).get("entries") or {}).get("memory-wiki") or {})
.get("auto", {})
.get("attachments", {})
.get("tabular", {})
)
if isinstance(tab, dict):
size = _safe_int(tab.get("video_transcript_chunk_size"), size, min_value=200, max_value=8_000)
overlap = _safe_int(tab.get("video_transcript_chunk_overlap"), overlap, min_value=0, max_value=4_000)
except Exception:
pass
overlap = max(0, min(overlap, max(0, size - 1)))
return size, overlap
def _normalized_video_meta(ffprobe_obj: dict[str, Any] | None) -> dict[str, Any]:
out: dict[str, Any] = {}
if not isinstance(ffprobe_obj, dict):
return out
fmt = ffprobe_obj.get("format") if isinstance(ffprobe_obj.get("format"), dict) else {}
streams = ffprobe_obj.get("streams") if isinstance(ffprobe_obj.get("streams"), list) else []
if isinstance(fmt, dict) and fmt.get("duration") is not None:
try:
out["duration_sec"] = float(fmt.get("duration"))
except Exception:
pass
for s in streams:
if not isinstance(s, dict):
continue
if str(s.get("codec_type") or "") != "video":
continue
try:
if s.get("width") is not None:
out["width"] = int(s.get("width"))
if s.get("height") is not None:
out["height"] = int(s.get("height"))
except Exception:
pass
fr = str(s.get("avg_frame_rate") or s.get("r_frame_rate") or "").strip()
if fr and fr != "0/0" and "/" in fr:
try:
a, b = fr.split("/", 1)
fa = float(a)
fb = float(b)
if fb:
out["fps"] = fa / fb
except Exception:
pass
break
return out
def query_video_attachment_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
attachment_id = str(args.get("attachment_id") or "").strip()
task = str(args.get("task") or "meta").strip().lower()
lang = str(args.get("lang") or "").strip().lower()
if not attachment_id:
return {"ok": False, "error": "attachment_id_required"}
if task not in {"meta", "transcript"}:
return {"ok": False, "error": "invalid_task"}
store = AttachmentAssetStore()
p = store.get_local_path(attachment_id)
meta = store.get_meta(attachment_id)
if p is None:
return {"ok": False, "error": "attachment_not_found"}
# Basic metadata (no heavy deps). Prefer ffprobe if present.
if task == "meta":
fp = _ffprobe_json(p)
norm = _normalized_video_meta(fp)
return {
"ok": True,
"task": "meta",
"attachment_id": attachment_id,
"name": (meta.name if meta else p.name),
"mime": (meta.mime if meta else "video/*"),
"bytes": int(meta.bytes if meta else (p.stat().st_size if p.exists() else 0)),
"duration_sec": norm.get("duration_sec"),
"width": norm.get("width"),
"height": norm.get("height"),
"fps": norm.get("fps"),
"ffprobe": fp if fp else None,
"note": "Use task=transcript to extract audio transcript (requires ffmpeg + OpenAI key).",
}
# transcript: extract audio then transcribe via OpenAI, store as text chunks.
if task == "transcript":
if not _ffmpeg_exists():
return {
"ok": False,
"error": "ffmpeg_missing",
"hint": "Install ffmpeg (ffmpeg/ffprobe on PATH) to enable transcript extraction.",
}
api_key = str(os.getenv("OPENAI_API_KEY") or "").strip()
if not api_key:
return {"ok": False, "error": "OPENAI_API_KEY_missing"}
try:
with tempfile.TemporaryDirectory() as td:
wav = Path(td) / "audio.wav"
subprocess.run(
["ffmpeg", "-y", "-i", str(p), "-vn", "-ac", "1", "-ar", "16000", str(wav)],
capture_output=True,
text=True,
timeout=60,
)
if not wav.exists() or wav.stat().st_size <= 0:
return {"ok": False, "error": "audio_extract_failed"}
try:
from openai import OpenAI
except Exception as e:
return {"ok": False, "error": f"openai_package_missing: {type(e).__name__}: {e}"}
base_url = str(os.getenv("OPENAI_BASE_URL") or "").strip()
client_kwargs: dict[str, Any] = {"api_key": api_key}
if base_url:
client_kwargs["base_url"] = base_url
client = OpenAI(**client_kwargs)
model = str(args.get("model") or os.getenv("OPENAI_AUDIO_TRANSCRIPTION_MODEL") or OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL).strip()
prompt = str(args.get("prompt") or "").strip()
# OpenAI SDK expects a file-like object with a name.
wav_bytes = wav.read_bytes()
f = io.BytesIO(wav_bytes)
f.name = "audio.wav" # type: ignore[attr-defined]
# Best-effort: different gateways may accept different param names; keep it minimal.
try:
resp = client.audio.transcriptions.create( # type: ignore[attr-defined]
model=model,
file=f,
**({"prompt": prompt} if prompt else {}),
)
text = str(getattr(resp, "text", "") or "")
except Exception as e:
return {"ok": False, "error": f"transcription_failed: {type(e).__name__}: {e}"}
if not text.strip():
return {"ok": False, "error": "empty_transcript"}
# Persist transcript as a long text document so the model can query evidence by text_id.
name = str(meta.name if meta else p.name)
text_name = f"{name}.transcript.txt"
cfg_chunk_size, cfg_chunk_overlap = _video_transcript_chunk_defaults()
chunk_size = int(args.get("chunk_size") or cfg_chunk_size)
chunk_overlap = int(args.get("chunk_overlap") or cfg_chunk_overlap)
text_meta = save_text_document(
attachment_id=str(attachment_id),
name=text_name,
text=text,
source_kind="video_transcript",
chunk_size=chunk_size,
chunk_overlap=chunk_overlap,
)
preview = text[:1200]
note = (
"Use query_text_attachment(text_id=...) to retrieve exact evidence with offsets."
if lang.startswith("en")
else "后续请用 query_text_attachment(text_id=...) 按需检索证据(支持 offset/top_k/关键词)。"
)
return {
"ok": True,
"task": "transcript",
"attachment_id": attachment_id,
"name": text_name,
"text_id": str(text_meta.get("text_id") or ""),
"chars": int(text_meta.get("chars") or 0),
"chunks": int(text_meta.get("chunks") or 0),
"preview": preview,
"note": note,
}
except Exception as e:
return {"ok": False, "error": f"transcript_failed: {type(e).__name__}: {e}"}
return ToolSpec(
name="query_video_attachment",
description="Query a video attachment by attachment_id (meta or transcript).",
parameters={
"type": "object",
"properties": {
"attachment_id": {"type": "string"},
"task": {"type": "string", "enum": ["meta", "transcript"]},
"lang": {"type": "string", "description": "Optional hint: zh/en."},
"model": {"type": "string", "description": "Optional transcription model override."},
"prompt": {"type": "string", "description": "Optional transcription prompt/context."},
"chunk_size": {"type": "integer", "description": "Transcript chunk size (chars)."},
"chunk_overlap": {"type": "integer", "description": "Transcript chunk overlap (chars)."},
},
"required": ["attachment_id"],
"additionalProperties": False,
},
handler=handler,
read_only=True,
tags=frozenset({"video", "read"}),
)
__all__ = ["query_video_attachment_tool"]

View file

@ -0,0 +1,6 @@
from __future__ import annotations
from oclaw.runtime.tools.experts.generalist.image_query import query_image_attachment_tool
__all__ = ["query_image_attachment_tool"]

View file

@ -0,0 +1,6 @@
from __future__ import annotations
from oclaw.runtime.tools.experts.generalist.text_query import query_text_attachment_tool
__all__ = ["query_text_attachment_tool"]

View file

@ -0,0 +1,6 @@
from __future__ import annotations
from oclaw.runtime.tools.experts.generalist.video_query import query_video_attachment_tool
__all__ = ["query_video_attachment_tool"]