oclaw/svc/files/file_attachments.py
oliver 61d54b10fd fix(attachments): parse channel uploads for LLM and stop echoing user files
WhatsApp/channel binary_ref uploads now expand into text_ref/tabular summaries for the model, and outbound replies only auto-attach generated image/video media instead of lookup tool text_ref results.

Co-authored-by: Cursor <cursoragent@cursor.com>
2026-07-03 21:43:49 +08:00

817 lines
33 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

from __future__ import annotations
"""将上传文件解析为附件字典(不依赖 Streamlit)。"""
import io
import json
import os
import re
import subprocess
import zipfile
from functools import lru_cache
from pathlib import Path
from typing import Any
import pandas as pd
from docx import Document
from PIL import Image
from PyPDF2 import PdfReader
from svc.config.paths import PROJECT_ROOT
from svc.files.archive_processor import (
MAX_ARCHIVE_DEPTH,
MAX_ARCHIVE_ENTRY_BYTES,
MAX_ARCHIVE_FILE_COUNT,
MAX_ARCHIVE_MEMBER_NAME_LENGTH,
MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES,
detect_archive_kind,
is_safe_archive_member_name,
process_archive,
)
from svc.files.attachment_assets import AttachmentAssetStore
from svc.files.tabular_attachment_store import save_dataframe, save_workbook
from svc.files.text_attachment_store import (
DEFAULT_TEXT_CHUNK_OVERLAP,
DEFAULT_TEXT_CHUNK_SIZE,
DEFAULT_TEXT_INLINE_MAX_CHARS,
save_text_document,
)
def _sniff_image_mime(data: bytes) -> str | None:
if len(data) < 12:
return None
if data[:8] == b"\x89PNG\r\n\x1a\n":
return "image/png"
if data[:3] == b"\xff\xd8\xff":
return "image/jpeg"
if data[:6] in (b"GIF87a", b"GIF89a"):
return "image/gif"
if len(data) >= 12 and data[:4] == b"RIFF" and data[8:12] == b"WEBP":
return "image/webp"
return None
def _ffprobe_exists() -> bool:
try:
p = subprocess.run(["ffprobe", "-version"], capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=3)
return p.returncode == 0
except Exception:
return False
def _extract_video_meta_from_ffprobe(obj: dict[str, Any]) -> dict[str, Any]:
fmt = obj.get("format") if isinstance(obj.get("format"), dict) else {}
streams = obj.get("streams") if isinstance(obj.get("streams"), list) else []
duration_sec = None
if isinstance(fmt, dict) and fmt.get("duration") is not None:
try:
duration_sec = float(fmt.get("duration"))
except Exception:
duration_sec = None
width = None
height = None
fps = None
for s in streams:
if not isinstance(s, dict):
continue
if str(s.get("codec_type") or "") != "video":
continue
if s.get("width") is not None and s.get("height") is not None:
try:
width = int(s.get("width") or 0) or None
height = int(s.get("height") or 0) or None
except Exception:
width = width
height = height
fr = str(s.get("avg_frame_rate") or s.get("r_frame_rate") or "").strip()
if fr and fr != "0/0" and "/" in fr:
try:
a, b = fr.split("/", 1)
fa = float(a)
fb = float(b)
if fb:
fps = fa / fb
except Exception:
fps = fps
break
out: dict[str, Any] = {}
if duration_sec is not None:
out["duration_sec"] = duration_sec
if width is not None:
out["width"] = width
if height is not None:
out["height"] = height
if fps is not None:
out["fps"] = fps
return out
def _ffprobe_video_meta(path: Path) -> dict[str, Any]:
if not _ffprobe_exists():
return {}
try:
p = subprocess.run(
["ffprobe", "-v", "error", "-print_format", "json", "-show_format", "-show_streams", str(path)],
capture_output=True,
text=True,
encoding="utf-8",
errors="replace",
timeout=8,
)
if p.returncode != 0:
return {}
obj = json.loads(p.stdout or "{}")
if not isinstance(obj, dict):
return {}
return _extract_video_meta_from_ffprobe(obj)
except Exception:
return {}
MAX_TEXT_CHARS = 12_000
EXCEL_PREVIEW_ROWS = 80
LARGE_TABLE_PREVIEW_ROWS = 20
MAX_ZIP_DEPTH = MAX_ARCHIVE_DEPTH
MAX_ZIP_FILE_COUNT = MAX_ARCHIVE_FILE_COUNT
MAX_ZIP_ENTRY_BYTES = MAX_ARCHIVE_ENTRY_BYTES
MAX_ZIP_TOTAL_UNCOMPRESSED_BYTES = MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES
DEFAULT_TABULAR_ROWS_READ = 5000
DEFAULT_TABULAR_COLUMNS = 200
DEFAULT_TABULAR_CELL_CHARS = 500
DEFAULT_TABULAR_TOOL_MODE_ENABLED = True
DEFAULT_TABULAR_TOOL_MODE_MIN_ROWS = 5_000
DEFAULT_TABULAR_TOOL_MODE_MAX_BYTES = 30 * 1024 * 1024
DEFAULT_LARGE_TABLE_PREVIEW_ROWS = LARGE_TABLE_PREVIEW_ROWS
MAX_ZIP_MEMBER_NAME_LENGTH = MAX_ARCHIVE_MEMBER_NAME_LENGTH
MAX_EXCEL_ZIP_FILE_COUNT = 500
MAX_EXCEL_ZIP_ENTRY_BYTES = 20 * 1024 * 1024
MAX_EXCEL_ZIP_TOTAL_UNCOMPRESSED_BYTES = 120 * 1024 * 1024
DEFAULT_MAX_EXCEL_SHEETS = 50
DEFAULT_TEXT_QUERY_TOP_K = 5
def _truncate_text(text: str, limit: int = MAX_TEXT_CHARS) -> str:
t = str(text or "")
if len(t) <= limit:
return t
return t[:limit] + f"\n\n...[truncated to {limit} chars]"
def _text_ref_summary(*, name: str, chars: int, chunks: int, source_kind: str, preview: str, top_k: int) -> str:
return _truncate_text(
(
f"# Text Document Summary\n"
f"- name: {name}\n"
f"- source_kind: {source_kind}\n"
f"- chars: {chars}\n"
f"- chunks: {chunks}\n"
f"- mode: text_ref\n"
f"- query_tool: query_text_attachment\n"
f"- note: use `text_id` from accompanying `text_ref` attachment; recommended top_k <= {top_k}\n\n"
f"## Preview (first {min(len(preview), 1500)} chars)\n"
f"{preview[:1500]}"
),
limit=MAX_TEXT_CHARS,
)
def _decode_text_bytes(data: bytes) -> str:
try:
return data.decode("utf-8")
except UnicodeDecodeError:
try:
return data.decode("gbk")
except Exception:
return "Error decoding text file (unknown encoding)"
def _clean_html_to_text(raw: str) -> str:
text = re.sub(r"(?is)<script[^>]*>.*?</script>", " ", str(raw or ""))
text = re.sub(r"(?is)<style[^>]*>.*?</style>", " ", text)
text = re.sub(r"(?is)<[^>]+>", " ", text)
text = re.sub(r"[ \t\r\f\v]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
return text.strip()
def _safe_int(raw: Any, default: int, *, min_value: int = 1, max_value: int = 2_000_000) -> int:
try:
value = int(raw)
except Exception:
return default
if value < min_value:
return default
return min(value, max_value)
@lru_cache(maxsize=1)
def _attachments_limits() -> dict[str, Any]:
cfg_path_raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
cfg_path = Path(cfg_path_raw).expanduser() if cfg_path_raw else (Path(PROJECT_ROOT) / "oclaw.json").resolve()
if not cfg_path.is_absolute():
cfg_path = (Path(PROJECT_ROOT) / cfg_path).resolve()
tabular_cfg: dict[str, Any] = {}
try:
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
tabular_cfg = (
(((obj or {}).get("plugins") or {}).get("entries") or {}).get("memory-wiki", {}).get("auto", {}).get("attachments", {}).get("tabular", {}) or {}
)
except Exception:
tabular_cfg = {}
return {
"rows_read": _safe_int(tabular_cfg.get("max_rows_read"), DEFAULT_TABULAR_ROWS_READ),
"columns": _safe_int(tabular_cfg.get("max_columns"), DEFAULT_TABULAR_COLUMNS),
"cell_chars": _safe_int(tabular_cfg.get("max_cell_chars"), DEFAULT_TABULAR_CELL_CHARS),
"max_excel_sheets": _safe_int(tabular_cfg.get("max_excel_sheets"), DEFAULT_MAX_EXCEL_SHEETS, max_value=500),
"large_table_preview_rows": _safe_int(
tabular_cfg.get("large_table_preview_rows"), DEFAULT_LARGE_TABLE_PREVIEW_ROWS, max_value=500
),
"tool_mode_enabled": bool(tabular_cfg.get("tool_mode_enabled", DEFAULT_TABULAR_TOOL_MODE_ENABLED)),
"tool_mode_min_rows": _safe_int(tabular_cfg.get("tool_mode_min_rows"), DEFAULT_TABULAR_TOOL_MODE_MIN_ROWS),
"tool_mode_max_bytes": _safe_int(
tabular_cfg.get("tool_mode_max_bytes"), DEFAULT_TABULAR_TOOL_MODE_MAX_BYTES, max_value=500 * 1024 * 1024
),
"text_inline_max_chars": _safe_int(
tabular_cfg.get("text_inline_max_chars"), DEFAULT_TEXT_INLINE_MAX_CHARS, max_value=200_000
),
"text_chunk_size": _safe_int(tabular_cfg.get("text_chunk_size"), DEFAULT_TEXT_CHUNK_SIZE, max_value=8_000),
"text_chunk_overlap": _safe_int(
tabular_cfg.get("text_chunk_overlap"), DEFAULT_TEXT_CHUNK_OVERLAP, max_value=4_000
),
"text_query_top_k": _safe_int(tabular_cfg.get("text_query_top_k"), DEFAULT_TEXT_QUERY_TOP_K, max_value=50),
"archive_max_depth": _safe_int(tabular_cfg.get("archive_max_depth"), MAX_ARCHIVE_DEPTH, max_value=10),
"archive_max_file_count": _safe_int(
tabular_cfg.get("archive_max_file_count"), MAX_ARCHIVE_FILE_COUNT, max_value=20_000
),
"archive_max_entry_bytes": _safe_int(
tabular_cfg.get("archive_max_entry_bytes"), MAX_ARCHIVE_ENTRY_BYTES, max_value=2_000_000_000
),
"archive_max_total_uncompressed_bytes": _safe_int(
tabular_cfg.get("archive_max_total_uncompressed_bytes"),
MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES,
max_value=5_000_000_000,
),
}
def clear_attachment_limits_cache() -> None:
_attachments_limits.cache_clear()
def _clip_tabular_cells(df: pd.DataFrame, *, max_cell_chars: int) -> tuple[pd.DataFrame, bool]:
clipped_any = False
def _clip_cell(value: Any) -> Any:
nonlocal clipped_any
if value is None:
return value
s = str(value)
if len(s) <= max_cell_chars:
return value
clipped_any = True
return s[:max_cell_chars] + "...[cell-truncated]"
out = df.copy()
for col in out.columns:
out[col] = out[col].map(_clip_cell)
return out, clipped_any
def _dataframe_summary_text(
df: pd.DataFrame,
*,
sampled: bool = False,
clipped_columns: bool = False,
clipped_cells: bool = False,
preview_rows: int = EXCEL_PREVIEW_ROWS,
) -> str:
rows = int(len(df.index))
cols = int(len(df.columns))
header = [str(x) for x in list(df.columns)]
dtypes = [f"{str(k)}:{str(v)}" for k, v in df.dtypes.items()]
show_rows = max(1, int(preview_rows or EXCEL_PREVIEW_ROWS))
sample = df.head(show_rows).to_string(index=False)
rows_scope_line = (
f"## Preview (first {min(rows, show_rows)} rows)"
if rows > show_rows
else f"## Full table included ({rows} rows)"
)
body = (
f"# Table Summary\n"
f"- rows: {rows}\n"
f"- cols: {cols}\n"
f"- sampled: {'yes' if sampled else 'no'}\n"
f"- clipped_columns: {'yes' if clipped_columns else 'no'}\n"
f"- clipped_cells: {'yes' if clipped_cells else 'no'}\n"
f"- columns: {', '.join(header)}\n"
f"- dtypes: {', '.join(dtypes)}\n\n"
f"{rows_scope_line}\n"
f"{sample}"
)
return _truncate_text(body)
def _pdf_summary_text(reader: PdfReader) -> str:
pages = list(getattr(reader, "pages", []) or [])
chunks: list[str] = [f"# PDF Summary\n- pages: {len(pages)}\n"]
for idx, page in enumerate(pages, start=1):
txt = ""
try:
txt = str(page.extract_text() or "").strip()
except Exception:
txt = ""
if not txt:
txt = "(empty page text)"
chunks.append(f"## Page {idx}\n{txt}")
return _truncate_text("\n\n".join(chunks))
def _is_safe_zip_member_name(name: str) -> bool:
return is_safe_archive_member_name(name)
def _validate_excel_zip_payload(data: bytes) -> str | None:
try:
with zipfile.ZipFile(io.BytesIO(data)) as zf:
infos = [x for x in zf.infolist() if not x.is_dir()]
if len(infos) > MAX_EXCEL_ZIP_FILE_COUNT:
return f"excel zip has too many entries ({len(infos)} > {MAX_EXCEL_ZIP_FILE_COUNT})"
total_uncompressed = 0
for info in infos:
if not _is_safe_zip_member_name(info.filename):
return f"excel zip contains unsafe path: {info.filename}"
fsize = int(info.file_size or 0)
total_uncompressed += fsize
if fsize > MAX_EXCEL_ZIP_ENTRY_BYTES:
return f"excel zip entry too large: {info.filename} ({fsize} > {MAX_EXCEL_ZIP_ENTRY_BYTES})"
if total_uncompressed > MAX_EXCEL_ZIP_TOTAL_UNCOMPRESSED_BYTES:
return (
"excel zip total uncompressed size too large "
f"({total_uncompressed} > {MAX_EXCEL_ZIP_TOTAL_UNCOMPRESSED_BYTES})"
)
return None
except Exception:
return "invalid excel zip payload"
def process_zip(zip_data: bytes, *, _depth: int = 0) -> list[dict[str, Any]]:
limits = _attachments_limits()
return process_archive(
archive_name="bundle.zip",
archive_data=zip_data,
process_member=lambda n, b, d: process_file_data(n, b, _zip_depth=d),
depth=_depth,
limits={
"max_depth": int(limits.get("archive_max_depth") or MAX_ARCHIVE_DEPTH),
"max_file_count": int(limits.get("archive_max_file_count") or MAX_ARCHIVE_FILE_COUNT),
"max_entry_bytes": int(limits.get("archive_max_entry_bytes") or MAX_ARCHIVE_ENTRY_BYTES),
"max_total_uncompressed_bytes": int(
limits.get("archive_max_total_uncompressed_bytes") or MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES
),
},
)
def expand_attachment_ref(
att: dict[str, Any],
*,
data: bytes | None = None,
name: str | None = None,
) -> list[dict[str, Any]]:
"""Expand ``binary_ref`` (or raw bytes) into parsed attachment dicts for the agent."""
if not isinstance(att, dict):
return []
t = str(att.get("type") or "").strip().lower()
if t and t not in {"binary_ref"}:
return [att]
aid = str(att.get("attachment_id") or att.get("attachmentId") or "").strip().lower()
meta = None
if data is None and aid:
blob, meta = AttachmentAssetStore().load_bytes(aid)
if not blob:
return [att] if t == "binary_ref" else []
data = blob
name = name or str(getattr(meta, "name", "") or "file")
if data is None:
return [att] if t == "binary_ref" else []
fname = str(name or att.get("name") or getattr(meta, "name", "") or "file")
got = process_file_data(fname, data)
if got:
return got
if aid:
if meta is None:
meta = AttachmentAssetStore().get_meta(aid)
return [
{
"type": "binary_ref",
"attachment_id": aid,
"name": fname,
"mime": str(att.get("mime") or getattr(meta, "mime", "") or "application/octet-stream"),
"bytes": att.get("bytes") if att.get("bytes") is not None else getattr(meta, "bytes", None),
}
]
return []
def process_file_data(name: str, data: bytes, *, _zip_depth: int = 0) -> list[dict[str, Any]]:
ext = name.split(".")[-1].lower() if "." in name else ""
attachments: list[dict[str, Any]] = []
if ext in ("png", "jpg", "jpeg", "webp", "gif", "jfif", "pjpeg", "bmp", "tif", "tiff"):
if ext == "png":
mime = "image/png"
elif ext in ("jpg", "jpeg", "jfif", "pjpeg"):
mime = "image/jpeg"
elif ext == "webp":
mime = "image/webp"
elif ext == "gif":
mime = "image/gif"
elif ext == "bmp":
mime = "image/bmp"
elif ext in ("tif", "tiff"):
mime = "image/tiff"
width = None
height = None
try:
with Image.open(io.BytesIO(data)) as im:
width, height = im.size
except Exception:
pass
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=mime, width=width, height=height)
attachments.append(
{
"type": "image_ref",
"name": meta.name,
"mime": meta.mime,
"attachment_id": meta.attachment_id,
"bytes": meta.bytes,
"width": meta.width,
"height": meta.height,
}
)
return attachments
sniff_mime = _sniff_image_mime(data)
if sniff_mime:
width = None
height = None
try:
with Image.open(io.BytesIO(data)) as im:
width, height = im.size
except Exception:
pass
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=sniff_mime, width=width, height=height)
attachments.append(
{
"type": "image_ref",
"name": meta.name,
"mime": meta.mime,
"attachment_id": meta.attachment_id,
"bytes": meta.bytes,
"width": meta.width,
"height": meta.height,
}
)
return attachments
if ext in ("mp4", "mov", "mkv", "webm", "avi", "m4v"):
mime = "video/mp4"
if ext == "webm":
mime = "video/webm"
elif ext == "mov":
mime = "video/quicktime"
elif ext == "mkv":
mime = "video/x-matroska"
elif ext == "avi":
mime = "video/x-msvideo"
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=mime)
probe_meta: dict[str, Any] = {}
try:
p = AttachmentAssetStore().get_local_path(str(meta.attachment_id))
if p is not None:
probe_meta = _ffprobe_video_meta(p)
except Exception:
probe_meta = {}
attachments.append(
{
"type": "video_ref",
"name": meta.name,
"mime": meta.mime,
"attachment_id": meta.attachment_id,
"bytes": meta.bytes,
**probe_meta,
}
)
attachments.append(
{
"type": "text",
"name": name,
"content": _truncate_text(
"".join(
[
"# Video Attachment\n",
f"- name: {meta.name}\n",
f"- mime: {meta.mime}\n",
f"- bytes: {meta.bytes}\n",
(
f"- duration_sec: {probe_meta.get('duration_sec')}\n"
if probe_meta.get("duration_sec") is not None
else ""
),
(
f"- size: {probe_meta.get('width')}x{probe_meta.get('height')}\n"
if probe_meta.get("width") and probe_meta.get("height")
else ""
),
(f"- fps: {probe_meta.get('fps')}\n" if probe_meta.get("fps") is not None else ""),
"- mode: video_ref\n",
"- query_tool: query_video_attachment\n",
"- note: use `attachment_id` from accompanying `video_ref` attachment.\n",
]
),
limit=MAX_TEXT_CHARS,
),
}
)
return attachments
if ext in ("xlsx", "xls", "csv"):
try:
if ext == "xlsx":
zip_err = _validate_excel_zip_payload(data)
if zip_err:
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing excel/csv: {zip_err}")})
return attachments
limits = _attachments_limits()
rows_read = int(limits.get("rows_read") or DEFAULT_TABULAR_ROWS_READ)
max_columns = int(limits.get("columns") or DEFAULT_TABULAR_COLUMNS)
max_cell_chars = int(limits.get("cell_chars") or DEFAULT_TABULAR_CELL_CHARS)
max_excel_sheets = int(limits.get("max_excel_sheets") or DEFAULT_MAX_EXCEL_SHEETS)
large_table_preview_rows = int(
limits.get("large_table_preview_rows") or DEFAULT_LARGE_TABLE_PREVIEW_ROWS
)
tool_mode_enabled = bool(limits.get("tool_mode_enabled", DEFAULT_TABULAR_TOOL_MODE_ENABLED))
tool_mode_min_rows = int(limits.get("tool_mode_min_rows") or DEFAULT_TABULAR_TOOL_MODE_MIN_ROWS)
tool_mode_max_bytes = int(limits.get("tool_mode_max_bytes") or DEFAULT_TABULAR_TOOL_MODE_MAX_BYTES)
if ext == "csv":
df = pd.read_csv(io.BytesIO(data), nrows=rows_read)
else:
df = pd.read_excel(io.BytesIO(data), nrows=rows_read)
sampled = int(len(df.index)) >= rows_read
estimated_rows = 0
if ext == "csv":
estimated_rows = max(0, int(data.count(b"\n")) - 1)
else:
estimated_rows = int(len(df.index))
clipped_columns = int(len(df.columns)) > max_columns
if clipped_columns:
df = df.iloc[:, :max_columns]
df, clipped_cells = _clip_tabular_cells(df, max_cell_chars=max_cell_chars)
tool_mode = bool(
tool_mode_enabled
and int(len(data or b"")) <= tool_mode_max_bytes
and (estimated_rows >= tool_mode_min_rows or (ext in ("xlsx", "xls") and sampled))
)
attachments.append(
{
"type": "text",
"name": name,
"content": _dataframe_summary_text(
df,
sampled=sampled,
clipped_columns=clipped_columns,
clipped_cells=clipped_cells,
preview_rows=(large_table_preview_rows if tool_mode else EXCEL_PREVIEW_ROWS),
)
+ (
(
f"\n\n# Large Table Mode\n- estimated_rows: {estimated_rows}\n- query_tool: query_tabular_attachment\n"
"- note: use the `table_id` from the accompanying `tabular_ref` attachment to query full data."
)
if tool_mode
else ""
),
}
)
if tool_mode:
workbook_sheets: dict[str, pd.DataFrame]
clipped_sheets = False
if ext == "csv":
full_df = pd.read_csv(io.BytesIO(data), dtype=str)
if int(len(full_df.columns)) > max_columns:
full_df = full_df.iloc[:, :max_columns]
full_df, _ = _clip_tabular_cells(full_df, max_cell_chars=max_cell_chars)
workbook_sheets = {"Sheet1": full_df}
else:
excel_obj = pd.read_excel(io.BytesIO(data), dtype=str, sheet_name=None)
workbook_sheets = {}
sheet_items = list((excel_obj or {}).items())
if len(sheet_items) > max_excel_sheets:
clipped_sheets = True
sheet_items = sheet_items[:max_excel_sheets]
for sheet_name, sheet_df in sheet_items:
sdf = sheet_df
if int(len(sdf.columns)) > max_columns:
sdf = sdf.iloc[:, :max_columns]
sdf, _ = _clip_tabular_cells(sdf, max_cell_chars=max_cell_chars)
workbook_sheets[str(sheet_name or f"Sheet{len(workbook_sheets)+1}")] = sdf
mime = "text/csv" if ext == "csv" else "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=mime)
table_meta = (
save_dataframe(attachment_id=str(meta.attachment_id), name=name, df=workbook_sheets.get("Sheet1", pd.DataFrame()))
if ext == "csv"
else save_workbook(attachment_id=str(meta.attachment_id), name=name, sheets=workbook_sheets)
)
attachments.append(
{
"type": "tabular_ref",
"name": str(name),
"attachment_id": str(meta.attachment_id),
"table_id": str(table_meta.get("table_id") or ""),
"rows": int(table_meta.get("rows") or 0),
"cols": int(table_meta.get("cols") or 0),
"columns": list(table_meta.get("columns") or []),
"sheets": list(table_meta.get("sheets") or []),
}
)
if clipped_sheets:
attachments.append(
{
"type": "text",
"name": f"{name}.sheet-limit",
"content": (
f"Excel workbook contains too many sheets; only the first {max_excel_sheets} sheets were stored "
"for tool-mode querying. Adjust max_excel_sheets if full coverage is required."
),
}
)
except Exception as e:
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing excel/csv: {e}")})
elif ext == "docx":
try:
doc = Document(io.BytesIO(data))
text = "\n".join([p.text for p in doc.paragraphs])
limits = _attachments_limits()
inline_cap = int(limits.get("text_inline_max_chars") or DEFAULT_TEXT_INLINE_MAX_CHARS)
if len(str(text or "")) <= inline_cap:
attachments.append({"type": "text", "name": name, "content": _truncate_text(text, limit=inline_cap)})
else:
meta = AttachmentAssetStore().save_bytes(
data,
filename=name,
mime="application/vnd.openxmlformats-officedocument.wordprocessingml.document",
)
text_meta = save_text_document(
attachment_id=str(meta.attachment_id),
name=name,
text=str(text or ""),
source_kind="docx",
chunk_size=int(limits.get("text_chunk_size") or DEFAULT_TEXT_CHUNK_SIZE),
chunk_overlap=int(limits.get("text_chunk_overlap") or DEFAULT_TEXT_CHUNK_OVERLAP),
)
top_k = int(limits.get("text_query_top_k") or DEFAULT_TEXT_QUERY_TOP_K)
attachments.append(
{
"type": "text",
"name": name,
"content": _text_ref_summary(
name=name,
chars=int(text_meta.get("chars") or 0),
chunks=int(text_meta.get("chunks") or 0),
source_kind="docx",
preview=str(text or ""),
top_k=top_k,
),
}
)
attachments.append(
{
"type": "text_ref",
"name": str(name),
"attachment_id": str(meta.attachment_id),
"text_id": str(text_meta.get("text_id") or ""),
"chars": int(text_meta.get("chars") or 0),
"chunks": int(text_meta.get("chunks") or 0),
"source_kind": "docx",
}
)
except Exception as e:
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing docx: {e}")})
elif ext == "pdf":
try:
reader = PdfReader(io.BytesIO(data))
text = _pdf_summary_text(reader)
limits = _attachments_limits()
inline_cap = int(limits.get("text_inline_max_chars") or DEFAULT_TEXT_INLINE_MAX_CHARS)
if len(str(text or "")) <= inline_cap:
attachments.append({"type": "text", "name": name, "content": _truncate_text(text, limit=inline_cap)})
else:
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime="application/pdf")
text_meta = save_text_document(
attachment_id=str(meta.attachment_id),
name=name,
text=str(text or ""),
source_kind="pdf",
chunk_size=int(limits.get("text_chunk_size") or DEFAULT_TEXT_CHUNK_SIZE),
chunk_overlap=int(limits.get("text_chunk_overlap") or DEFAULT_TEXT_CHUNK_OVERLAP),
)
top_k = int(limits.get("text_query_top_k") or DEFAULT_TEXT_QUERY_TOP_K)
attachments.append(
{
"type": "text",
"name": name,
"content": _text_ref_summary(
name=name,
chars=int(text_meta.get("chars") or 0),
chunks=int(text_meta.get("chunks") or 0),
source_kind="pdf",
preview=str(text or ""),
top_k=top_k,
),
}
)
attachments.append(
{
"type": "text_ref",
"name": str(name),
"attachment_id": str(meta.attachment_id),
"text_id": str(text_meta.get("text_id") or ""),
"chars": int(text_meta.get("chars") or 0),
"chunks": int(text_meta.get("chunks") or 0),
"source_kind": "pdf",
}
)
except Exception as e:
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing pdf: {e}")})
elif ext in ("txt", "py", "js", "html", "css", "md", "json", "yaml", "yml", "sh", "log"):
text = _decode_text_bytes(data)
if ext == "html":
text = _clean_html_to_text(text)
limits = _attachments_limits()
inline_cap = int(limits.get("text_inline_max_chars") or DEFAULT_TEXT_INLINE_MAX_CHARS)
if len(str(text or "")) <= inline_cap:
attachments.append({"type": "text", "name": name, "content": _truncate_text(text, limit=inline_cap)})
else:
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime="text/plain")
text_meta = save_text_document(
attachment_id=str(meta.attachment_id),
name=name,
text=str(text or ""),
source_kind=ext or "text",
chunk_size=int(limits.get("text_chunk_size") or DEFAULT_TEXT_CHUNK_SIZE),
chunk_overlap=int(limits.get("text_chunk_overlap") or DEFAULT_TEXT_CHUNK_OVERLAP),
)
top_k = int(limits.get("text_query_top_k") or DEFAULT_TEXT_QUERY_TOP_K)
attachments.append(
{
"type": "text",
"name": name,
"content": _text_ref_summary(
name=name,
chars=int(text_meta.get("chars") or 0),
chunks=int(text_meta.get("chunks") or 0),
source_kind=str(text_meta.get("source_kind") or ext or "text"),
preview=str(text or ""),
top_k=top_k,
),
}
)
attachments.append(
{
"type": "text_ref",
"name": str(name),
"attachment_id": str(meta.attachment_id),
"text_id": str(text_meta.get("text_id") or ""),
"chars": int(text_meta.get("chars") or 0),
"chunks": int(text_meta.get("chunks") or 0),
"source_kind": str(text_meta.get("source_kind") or ext or "text"),
}
)
elif detect_archive_kind(name):
limits = _attachments_limits()
attachments.extend(
process_archive(
archive_name=name,
archive_data=data,
process_member=lambda n, b, d: process_file_data(n, b, _zip_depth=d),
depth=_zip_depth,
limits={
"max_depth": int(limits.get("archive_max_depth") or MAX_ARCHIVE_DEPTH),
"max_file_count": int(limits.get("archive_max_file_count") or MAX_ARCHIVE_FILE_COUNT),
"max_entry_bytes": int(limits.get("archive_max_entry_bytes") or MAX_ARCHIVE_ENTRY_BYTES),
"max_total_uncompressed_bytes": int(
limits.get("archive_max_total_uncompressed_bytes") or MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES
),
},
)
)
return attachments
__all__ = ["process_zip", "process_file_data", "expand_attachment_ref", "clear_attachment_limits_cache"]