mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 22:15:57 +08:00
将附件/多模态查询与编辑工具从 generalist 迁移到 public。
同时移除 network_ops 的转发壳文件,统一工具入口并更新对应测试引用,降低重复维护成本。 Made-with: Cursor
This commit is contained in:
parent
5d67303d22
commit
1c71a29ec1
10 changed files with 84 additions and 126 deletions
|
|
@ -1,129 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import io
|
||||
import os
|
||||
from typing import Any
|
||||
|
||||
from PIL import Image
|
||||
|
||||
from oclaw.platform.files.attachment_assets import AttachmentAssetStore
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def image_edit_tool() -> ToolSpec:
|
||||
"""Edit an uploaded image using OpenAI Images API.
|
||||
|
||||
Input image is referenced by attachment_id (disk-backed asset store).
|
||||
Output is saved back to the asset store and returned as attachment_id.
|
||||
"""
|
||||
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
attachment_id = str(args.get("attachment_id") or "").strip()
|
||||
instruction = str(args.get("instruction") or "").strip()
|
||||
model = str(args.get("model") or os.getenv("OPENAI_IMAGE_MODEL") or "gpt-image-1").strip()
|
||||
if not attachment_id:
|
||||
return {"ok": False, "error": "attachment_id is required"}
|
||||
if not instruction:
|
||||
return {"ok": False, "error": "instruction is required"}
|
||||
|
||||
store = AttachmentAssetStore()
|
||||
blob, meta = store.load_bytes(attachment_id)
|
||||
if not blob:
|
||||
return {"ok": False, "error": f"attachment not found: {attachment_id}"}
|
||||
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"openai package is not available: {type(e).__name__}: {e}"}
|
||||
|
||||
api_key = (os.getenv("OPENAI_API_KEY") or "").strip()
|
||||
base_url = (os.getenv("OPENAI_BASE_URL") or "").strip()
|
||||
if not api_key:
|
||||
return {"ok": False, "error": "OPENAI_API_KEY is not set"}
|
||||
|
||||
client_kwargs: dict[str, Any] = {"api_key": api_key}
|
||||
if base_url:
|
||||
client_kwargs["base_url"] = base_url
|
||||
client = OpenAI(**client_kwargs)
|
||||
|
||||
# OpenAI SDK expects a file-like object for edits.
|
||||
img_file = io.BytesIO(blob)
|
||||
img_file.name = "input.png" # type: ignore[attr-defined]
|
||||
|
||||
b64_out: str | None = None
|
||||
try:
|
||||
# Preferred: image edit endpoint (if supported by the gateway/model).
|
||||
resp = client.images.edit( # type: ignore[attr-defined]
|
||||
model=model,
|
||||
image=img_file,
|
||||
prompt=instruction,
|
||||
response_format="b64_json",
|
||||
)
|
||||
data0 = resp.data[0] if getattr(resp, "data", None) else None
|
||||
b64_out = getattr(data0, "b64_json", None) if data0 is not None else None
|
||||
except Exception:
|
||||
# Fallback: generate a new image from prompt (still returns an image, but not true edit).
|
||||
try:
|
||||
resp = client.images.generate( # type: ignore[attr-defined]
|
||||
model=model,
|
||||
prompt=instruction,
|
||||
response_format="b64_json",
|
||||
)
|
||||
data0 = resp.data[0] if getattr(resp, "data", None) else None
|
||||
b64_out = getattr(data0, "b64_json", None) if data0 is not None else None
|
||||
except Exception as e2:
|
||||
return {"ok": False, "error": f"image api failed: {type(e2).__name__}: {e2}"}
|
||||
|
||||
if not b64_out:
|
||||
return {"ok": False, "error": "image api returned no b64_json payload"}
|
||||
|
||||
try:
|
||||
out_bytes = base64.b64decode(b64_out.encode("ascii"))
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"failed to decode image b64: {type(e).__name__}: {e}"}
|
||||
|
||||
width = None
|
||||
height = None
|
||||
try:
|
||||
with Image.open(io.BytesIO(out_bytes)) as im:
|
||||
width, height = im.size
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
out_meta = store.save_bytes(
|
||||
out_bytes,
|
||||
filename=f"edited-{meta.name if meta else 'image'}.png",
|
||||
mime="image/png",
|
||||
width=width,
|
||||
height=height,
|
||||
)
|
||||
return {
|
||||
"ok": True,
|
||||
"attachment_id": out_meta.attachment_id,
|
||||
"name": out_meta.name,
|
||||
"mime": out_meta.mime,
|
||||
"bytes": out_meta.bytes,
|
||||
"width": out_meta.width,
|
||||
"height": out_meta.height,
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="image_edit",
|
||||
description="Edit an uploaded image referenced by attachment_id, returning a new attachment_id.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"attachment_id": {"type": "string", "description": "Input image attachment id (image_ref)."},
|
||||
"instruction": {"type": "string", "description": "Edit instruction for the image."},
|
||||
"model": {"type": "string", "description": "OpenAI image model name (default: gpt-image-1)."},
|
||||
},
|
||||
"required": ["attachment_id", "instruction"],
|
||||
},
|
||||
handler=handler,
|
||||
tags=frozenset({"image", "edit"}),
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["image_edit_tool"]
|
||||
|
||||
|
|
@ -1,75 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.files.attachment_assets import attachment_id_to_data_url
|
||||
from oclaw.platform.llm.image_message_client import send_image_messages
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def query_image_attachment_tool() -> ToolSpec:
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
attachment_id = str(args.get("attachment_id") or "").strip()
|
||||
if not attachment_id:
|
||||
return {"ok": False, "error": "attachment_id_required"}
|
||||
task = str(args.get("task") or "describe").strip().lower()
|
||||
question = str(args.get("question") or "").strip()
|
||||
if task not in {"describe", "ocr"}:
|
||||
return {"ok": False, "error": "invalid_task"}
|
||||
data_url = attachment_id_to_data_url(attachment_id=attachment_id)
|
||||
if not data_url:
|
||||
return {"ok": False, "error": "attachment_not_found"}
|
||||
prompt = (
|
||||
(
|
||||
"请详细描述这张图片的主要内容、对象、场景和可见文字。"
|
||||
"回答请使用要点列表,避免臆测。"
|
||||
)
|
||||
if task == "describe"
|
||||
else (
|
||||
"请只提取图片中可见文字并按阅读顺序输出。"
|
||||
"如果有表格,保持行列结构;不确定的内容标注为[unclear]。"
|
||||
)
|
||||
)
|
||||
if question:
|
||||
prompt = f"{prompt}\n\n用户问题:{question}"
|
||||
out = send_image_messages(images=[data_url], prompt=prompt)
|
||||
if not bool(out.get("ok")):
|
||||
return {
|
||||
"ok": False,
|
||||
"error": str(out.get("error") or "image_query_failed"),
|
||||
"task": task,
|
||||
"attachment_id": attachment_id,
|
||||
}
|
||||
text = str(out.get("text") or "")
|
||||
if len(text) > 12_000:
|
||||
text = text[:12_000] + "\n\n...[truncated image analysis output]"
|
||||
return {
|
||||
"ok": True,
|
||||
"task": task,
|
||||
"attachment_id": attachment_id,
|
||||
"text": text,
|
||||
"input_kind": list(out.get("input_kind") or []),
|
||||
"backend_shape": str(out.get("backend_shape") or ""),
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="query_image_attachment",
|
||||
description="Analyze an uploaded image by attachment_id (describe or OCR).",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"attachment_id": {"type": "string"},
|
||||
"task": {"type": "string", "enum": ["describe", "ocr"]},
|
||||
"question": {"type": "string"},
|
||||
},
|
||||
"required": ["attachment_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=handler,
|
||||
read_only=True,
|
||||
tags=frozenset({"image", "read"}),
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["query_image_attachment_tool"]
|
||||
|
||||
|
|
@ -1,150 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.files.tabular_attachment_store import (
|
||||
aggregate_table,
|
||||
analyze_table_full_scan,
|
||||
query_table,
|
||||
run_table_sql,
|
||||
)
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def query_tabular_attachment_tool() -> ToolSpec:
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
table_id = str(args.get("table_id") or "").strip()
|
||||
if not table_id:
|
||||
return {"ok": False, "error": "table_id_required"}
|
||||
raw_cols = args.get("columns")
|
||||
cols = [str(x) for x in raw_cols] if isinstance(raw_cols, list) else None
|
||||
sheet = str(args.get("sheet") or "").strip() or None
|
||||
where_contains = args.get("where_contains") if isinstance(args.get("where_contains"), dict) else None
|
||||
aggregate = args.get("aggregate") if isinstance(args.get("aggregate"), dict) else None
|
||||
if aggregate:
|
||||
return aggregate_table(
|
||||
table_id=table_id,
|
||||
metric=str(aggregate.get("metric") or ""),
|
||||
target_column=str(aggregate.get("target_column") or "").strip() or None,
|
||||
group_by=str(aggregate.get("group_by") or "").strip() or None,
|
||||
where_contains=where_contains,
|
||||
top_n=int(aggregate.get("top_n") or 20),
|
||||
sheet=sheet,
|
||||
)
|
||||
return query_table(
|
||||
table_id=table_id,
|
||||
columns=cols,
|
||||
limit=int(args.get("limit") or 50),
|
||||
offset=int(args.get("offset") or 0),
|
||||
where_contains=where_contains, # {"column":"...", "keyword":"..."}
|
||||
sheet=sheet,
|
||||
)
|
||||
|
||||
return ToolSpec(
|
||||
name="query_tabular_attachment",
|
||||
description="Query rows from a large uploaded table by table_id with optional column selection and keyword filter.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"table_id": {"type": "string"},
|
||||
"sheet": {"type": "string"},
|
||||
"columns": {"type": "array", "items": {"type": "string"}},
|
||||
"limit": {"type": "integer", "minimum": 1, "maximum": 200},
|
||||
"offset": {"type": "integer", "minimum": 0},
|
||||
"where_contains": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"column": {"type": "string"},
|
||||
"keyword": {"type": "string"},
|
||||
},
|
||||
"additionalProperties": False,
|
||||
},
|
||||
"aggregate": {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"metric": {"type": "string", "enum": ["count", "sum", "avg"]},
|
||||
"target_column": {"type": "string"},
|
||||
"group_by": {"type": "string"},
|
||||
"top_n": {"type": "integer", "minimum": 1, "maximum": 200},
|
||||
},
|
||||
"required": ["metric"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
},
|
||||
"required": ["table_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=handler,
|
||||
read_only=True,
|
||||
)
|
||||
|
||||
|
||||
def run_tabular_sql_tool() -> ToolSpec:
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
table_id = str(args.get("table_id") or "").strip()
|
||||
sql = str(args.get("sql") or "").strip()
|
||||
sheet = str(args.get("sheet") or "").strip() or None
|
||||
if not table_id:
|
||||
return {"ok": False, "error": "table_id_required"}
|
||||
return run_table_sql(
|
||||
table_id=table_id,
|
||||
sql=sql,
|
||||
limit=int(args.get("limit") or 200),
|
||||
sheet=sheet,
|
||||
)
|
||||
|
||||
return ToolSpec(
|
||||
name="run_tabular_sql",
|
||||
description="Run a READ-ONLY SQL query against uploaded table by table_id. Only SELECT/WITH allowed.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"table_id": {"type": "string"},
|
||||
"sheet": {"type": "string"},
|
||||
"sql": {"type": "string"},
|
||||
"limit": {"type": "integer", "minimum": 1, "maximum": 500},
|
||||
},
|
||||
"required": ["table_id", "sql"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=handler,
|
||||
read_only=True,
|
||||
)
|
||||
|
||||
|
||||
def analyze_tabular_attachment_full_scan_tool() -> ToolSpec:
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
table_id = str(args.get("table_id") or "").strip()
|
||||
if not table_id:
|
||||
return {"ok": False, "error": "table_id_required"}
|
||||
raw_cols = args.get("columns")
|
||||
cols = [str(x) for x in raw_cols] if isinstance(raw_cols, list) else None
|
||||
sheet = str(args.get("sheet") or "").strip() or None
|
||||
return analyze_table_full_scan(
|
||||
table_id=table_id,
|
||||
columns=cols,
|
||||
sheet=sheet,
|
||||
top_values_limit=int(args.get("top_values_limit") or 3),
|
||||
)
|
||||
|
||||
return ToolSpec(
|
||||
name="analyze_tabular_attachment_full_scan",
|
||||
description="Run a full-table scan for selected columns and return concise profiling stats with audit evidence.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"table_id": {"type": "string"},
|
||||
"sheet": {"type": "string"},
|
||||
"columns": {"type": "array", "items": {"type": "string"}},
|
||||
"top_values_limit": {"type": "integer", "minimum": 0, "maximum": 10},
|
||||
},
|
||||
"required": ["table_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=handler,
|
||||
read_only=True,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["query_tabular_attachment_tool", "run_tabular_sql_tool", "analyze_tabular_attachment_full_scan_tool"]
|
||||
|
||||
|
|
@ -1,41 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.files.text_attachment_store import query_text_document
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def query_text_attachment_tool() -> ToolSpec:
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
text_id = str(args.get("text_id") or "").strip()
|
||||
if not text_id:
|
||||
return {"ok": False, "error": "text_id_required"}
|
||||
return query_text_document(
|
||||
text_id=text_id,
|
||||
query=str(args.get("query") or "").strip() or None,
|
||||
top_k=int(args.get("top_k") or 5),
|
||||
offset=int(args.get("offset") or 0),
|
||||
)
|
||||
|
||||
return ToolSpec(
|
||||
name="query_text_attachment",
|
||||
description="Query long text attachment chunks by text_id with optional keyword search.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"text_id": {"type": "string"},
|
||||
"query": {"type": "string"},
|
||||
"top_k": {"type": "integer", "minimum": 1, "maximum": 50},
|
||||
"offset": {"type": "integer", "minimum": 0},
|
||||
},
|
||||
"required": ["text_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=handler,
|
||||
read_only=True,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["query_text_attachment_tool"]
|
||||
|
||||
|
|
@ -1,278 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import tempfile
|
||||
import io
|
||||
import json
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.files.attachment_assets import AttachmentAssetStore
|
||||
from oclaw.platform.files.text_attachment_store import (
|
||||
DEFAULT_TEXT_CHUNK_OVERLAP,
|
||||
DEFAULT_TEXT_CHUNK_SIZE,
|
||||
save_text_document,
|
||||
)
|
||||
from oclaw.runtime.extensions.openai.api import OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def _ffmpeg_exists() -> bool:
|
||||
try:
|
||||
p = subprocess.run(["ffmpeg", "-version"], capture_output=True, text=True, timeout=3)
|
||||
return p.returncode == 0
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def _ffprobe_json(path: Path) -> dict[str, Any] | None:
|
||||
try:
|
||||
p = subprocess.run(
|
||||
[
|
||||
"ffprobe",
|
||||
"-v",
|
||||
"error",
|
||||
"-print_format",
|
||||
"json",
|
||||
"-show_format",
|
||||
"-show_streams",
|
||||
str(path),
|
||||
],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=8,
|
||||
)
|
||||
if p.returncode != 0:
|
||||
return None
|
||||
obj = json.loads(p.stdout or "{}")
|
||||
return obj if isinstance(obj, dict) else None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _safe_int(raw: Any, default: int, *, min_value: int = 1, max_value: int = 2_000_000) -> int:
|
||||
try:
|
||||
value = int(raw)
|
||||
except Exception:
|
||||
return default
|
||||
if value < min_value:
|
||||
return default
|
||||
return min(value, max_value)
|
||||
|
||||
|
||||
def _oclaw_config_path() -> Path:
|
||||
raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
|
||||
if raw:
|
||||
p = Path(raw)
|
||||
return p if p.is_absolute() else p.resolve()
|
||||
return Path(__file__).resolve().parents[4] / "oclaw.json"
|
||||
|
||||
|
||||
def _video_transcript_chunk_defaults() -> tuple[int, int]:
|
||||
size = DEFAULT_TEXT_CHUNK_SIZE
|
||||
overlap = DEFAULT_TEXT_CHUNK_OVERLAP
|
||||
try:
|
||||
cfg_path = _oclaw_config_path()
|
||||
if cfg_path.exists() and cfg_path.is_file():
|
||||
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
|
||||
tab = (
|
||||
(((obj.get("plugins") or {}).get("entries") or {}).get("memory-wiki") or {})
|
||||
.get("auto", {})
|
||||
.get("attachments", {})
|
||||
.get("tabular", {})
|
||||
)
|
||||
if isinstance(tab, dict):
|
||||
size = _safe_int(tab.get("video_transcript_chunk_size"), size, min_value=200, max_value=8_000)
|
||||
overlap = _safe_int(tab.get("video_transcript_chunk_overlap"), overlap, min_value=0, max_value=4_000)
|
||||
except Exception:
|
||||
pass
|
||||
overlap = max(0, min(overlap, max(0, size - 1)))
|
||||
return size, overlap
|
||||
|
||||
|
||||
def _normalized_video_meta(ffprobe_obj: dict[str, Any] | None) -> dict[str, Any]:
|
||||
out: dict[str, Any] = {}
|
||||
if not isinstance(ffprobe_obj, dict):
|
||||
return out
|
||||
fmt = ffprobe_obj.get("format") if isinstance(ffprobe_obj.get("format"), dict) else {}
|
||||
streams = ffprobe_obj.get("streams") if isinstance(ffprobe_obj.get("streams"), list) else []
|
||||
if isinstance(fmt, dict) and fmt.get("duration") is not None:
|
||||
try:
|
||||
out["duration_sec"] = float(fmt.get("duration"))
|
||||
except Exception:
|
||||
pass
|
||||
for s in streams:
|
||||
if not isinstance(s, dict):
|
||||
continue
|
||||
if str(s.get("codec_type") or "") != "video":
|
||||
continue
|
||||
try:
|
||||
if s.get("width") is not None:
|
||||
out["width"] = int(s.get("width"))
|
||||
if s.get("height") is not None:
|
||||
out["height"] = int(s.get("height"))
|
||||
except Exception:
|
||||
pass
|
||||
fr = str(s.get("avg_frame_rate") or s.get("r_frame_rate") or "").strip()
|
||||
if fr and fr != "0/0" and "/" in fr:
|
||||
try:
|
||||
a, b = fr.split("/", 1)
|
||||
fa = float(a)
|
||||
fb = float(b)
|
||||
if fb:
|
||||
out["fps"] = fa / fb
|
||||
except Exception:
|
||||
pass
|
||||
break
|
||||
return out
|
||||
|
||||
|
||||
def query_video_attachment_tool() -> ToolSpec:
|
||||
def handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
attachment_id = str(args.get("attachment_id") or "").strip()
|
||||
task = str(args.get("task") or "meta").strip().lower()
|
||||
lang = str(args.get("lang") or "").strip().lower()
|
||||
if not attachment_id:
|
||||
return {"ok": False, "error": "attachment_id_required"}
|
||||
if task not in {"meta", "transcript"}:
|
||||
return {"ok": False, "error": "invalid_task"}
|
||||
|
||||
store = AttachmentAssetStore()
|
||||
p = store.get_local_path(attachment_id)
|
||||
meta = store.get_meta(attachment_id)
|
||||
if p is None:
|
||||
return {"ok": False, "error": "attachment_not_found"}
|
||||
|
||||
# Basic metadata (no heavy deps). Prefer ffprobe if present.
|
||||
if task == "meta":
|
||||
fp = _ffprobe_json(p)
|
||||
norm = _normalized_video_meta(fp)
|
||||
return {
|
||||
"ok": True,
|
||||
"task": "meta",
|
||||
"attachment_id": attachment_id,
|
||||
"name": (meta.name if meta else p.name),
|
||||
"mime": (meta.mime if meta else "video/*"),
|
||||
"bytes": int(meta.bytes if meta else (p.stat().st_size if p.exists() else 0)),
|
||||
"duration_sec": norm.get("duration_sec"),
|
||||
"width": norm.get("width"),
|
||||
"height": norm.get("height"),
|
||||
"fps": norm.get("fps"),
|
||||
"ffprobe": fp if fp else None,
|
||||
"note": "Use task=transcript to extract audio transcript (requires ffmpeg + OpenAI key).",
|
||||
}
|
||||
|
||||
# transcript: extract audio then transcribe via OpenAI, store as text chunks.
|
||||
if task == "transcript":
|
||||
if not _ffmpeg_exists():
|
||||
return {
|
||||
"ok": False,
|
||||
"error": "ffmpeg_missing",
|
||||
"hint": "Install ffmpeg (ffmpeg/ffprobe on PATH) to enable transcript extraction.",
|
||||
}
|
||||
api_key = str(os.getenv("OPENAI_API_KEY") or "").strip()
|
||||
if not api_key:
|
||||
return {"ok": False, "error": "OPENAI_API_KEY_missing"}
|
||||
|
||||
try:
|
||||
with tempfile.TemporaryDirectory() as td:
|
||||
wav = Path(td) / "audio.wav"
|
||||
subprocess.run(
|
||||
["ffmpeg", "-y", "-i", str(p), "-vn", "-ac", "1", "-ar", "16000", str(wav)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=60,
|
||||
)
|
||||
if not wav.exists() or wav.stat().st_size <= 0:
|
||||
return {"ok": False, "error": "audio_extract_failed"}
|
||||
try:
|
||||
from openai import OpenAI
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"openai_package_missing: {type(e).__name__}: {e}"}
|
||||
|
||||
base_url = str(os.getenv("OPENAI_BASE_URL") or "").strip()
|
||||
client_kwargs: dict[str, Any] = {"api_key": api_key}
|
||||
if base_url:
|
||||
client_kwargs["base_url"] = base_url
|
||||
client = OpenAI(**client_kwargs)
|
||||
|
||||
model = str(args.get("model") or os.getenv("OPENAI_AUDIO_TRANSCRIPTION_MODEL") or OPENAI_DEFAULT_AUDIO_TRANSCRIPTION_MODEL).strip()
|
||||
prompt = str(args.get("prompt") or "").strip()
|
||||
# OpenAI SDK expects a file-like object with a name.
|
||||
wav_bytes = wav.read_bytes()
|
||||
f = io.BytesIO(wav_bytes)
|
||||
f.name = "audio.wav" # type: ignore[attr-defined]
|
||||
# Best-effort: different gateways may accept different param names; keep it minimal.
|
||||
try:
|
||||
resp = client.audio.transcriptions.create( # type: ignore[attr-defined]
|
||||
model=model,
|
||||
file=f,
|
||||
**({"prompt": prompt} if prompt else {}),
|
||||
)
|
||||
text = str(getattr(resp, "text", "") or "")
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"transcription_failed: {type(e).__name__}: {e}"}
|
||||
|
||||
if not text.strip():
|
||||
return {"ok": False, "error": "empty_transcript"}
|
||||
|
||||
# Persist transcript as a long text document so the model can query evidence by text_id.
|
||||
name = str(meta.name if meta else p.name)
|
||||
text_name = f"{name}.transcript.txt"
|
||||
cfg_chunk_size, cfg_chunk_overlap = _video_transcript_chunk_defaults()
|
||||
chunk_size = int(args.get("chunk_size") or cfg_chunk_size)
|
||||
chunk_overlap = int(args.get("chunk_overlap") or cfg_chunk_overlap)
|
||||
text_meta = save_text_document(
|
||||
attachment_id=str(attachment_id),
|
||||
name=text_name,
|
||||
text=text,
|
||||
source_kind="video_transcript",
|
||||
chunk_size=chunk_size,
|
||||
chunk_overlap=chunk_overlap,
|
||||
)
|
||||
preview = text[:1200]
|
||||
note = (
|
||||
"Use query_text_attachment(text_id=...) to retrieve exact evidence with offsets."
|
||||
if lang.startswith("en")
|
||||
else "后续请用 query_text_attachment(text_id=...) 按需检索证据(支持 offset/top_k/关键词)。"
|
||||
)
|
||||
return {
|
||||
"ok": True,
|
||||
"task": "transcript",
|
||||
"attachment_id": attachment_id,
|
||||
"name": text_name,
|
||||
"text_id": str(text_meta.get("text_id") or ""),
|
||||
"chars": int(text_meta.get("chars") or 0),
|
||||
"chunks": int(text_meta.get("chunks") or 0),
|
||||
"preview": preview,
|
||||
"note": note,
|
||||
}
|
||||
except Exception as e:
|
||||
return {"ok": False, "error": f"transcript_failed: {type(e).__name__}: {e}"}
|
||||
|
||||
return ToolSpec(
|
||||
name="query_video_attachment",
|
||||
description="Query a video attachment by attachment_id (meta or transcript).",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"attachment_id": {"type": "string"},
|
||||
"task": {"type": "string", "enum": ["meta", "transcript"]},
|
||||
"lang": {"type": "string", "description": "Optional hint: zh/en."},
|
||||
"model": {"type": "string", "description": "Optional transcription model override."},
|
||||
"prompt": {"type": "string", "description": "Optional transcription prompt/context."},
|
||||
"chunk_size": {"type": "integer", "description": "Transcript chunk size (chars)."},
|
||||
"chunk_overlap": {"type": "integer", "description": "Transcript chunk overlap (chars)."},
|
||||
},
|
||||
"required": ["attachment_id"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=handler,
|
||||
read_only=True,
|
||||
tags=frozenset({"video", "read"}),
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["query_video_attachment_tool"]
|
||||
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from oclaw.runtime.tools.experts.generalist.image_query import query_image_attachment_tool
|
||||
|
||||
__all__ = ["query_image_attachment_tool"]
|
||||
|
||||
|
|
@ -1,10 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from oclaw.runtime.tools.experts.generalist.tabular_query import (
|
||||
analyze_tabular_attachment_full_scan_tool,
|
||||
query_tabular_attachment_tool,
|
||||
run_tabular_sql_tool,
|
||||
)
|
||||
|
||||
__all__ = ["query_tabular_attachment_tool", "run_tabular_sql_tool", "analyze_tabular_attachment_full_scan_tool"]
|
||||
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from oclaw.runtime.tools.experts.generalist.text_query import query_text_attachment_tool
|
||||
|
||||
__all__ = ["query_text_attachment_tool"]
|
||||
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from oclaw.runtime.tools.experts.generalist.video_query import query_video_attachment_tool
|
||||
|
||||
__all__ = ["query_video_attachment_tool"]
|
||||
|
||||
Loading…
Add table
Add a link
Reference in a new issue