mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 07:20:44 +08:00
- Rename platform/ to svc/ to avoid shadowing stdlib platform. - Replace from oclaw.* with from svc/runtime/interfaces; update -m CLI paths. - tests/conftest: prepend repo root to sys.path (no parent-folder package name). - CI: paths and offline_eval script under repo root. - Ops scripts: PYTHONPATH must be repo root for python -m runtime.* (fixes gateway/WhatsApp sidecar startup). - Fix default oclaw.json path in tabular/file attachment limits; stabilize attachment test config. Co-authored-by: Cursor <cursoragent@cursor.com>
776 lines
32 KiB
Python
776 lines
32 KiB
Python
from __future__ import annotations
|
||
|
||
"""将上传文件解析为附件字典(不依赖 Streamlit)。"""
|
||
|
||
import io
|
||
import json
|
||
import os
|
||
import re
|
||
import subprocess
|
||
import zipfile
|
||
from functools import lru_cache
|
||
from pathlib import Path
|
||
from typing import Any
|
||
|
||
import pandas as pd
|
||
from docx import Document
|
||
from PIL import Image
|
||
from PyPDF2 import PdfReader
|
||
|
||
from svc.config.paths import PROJECT_ROOT
|
||
from svc.files.archive_processor import (
|
||
MAX_ARCHIVE_DEPTH,
|
||
MAX_ARCHIVE_ENTRY_BYTES,
|
||
MAX_ARCHIVE_FILE_COUNT,
|
||
MAX_ARCHIVE_MEMBER_NAME_LENGTH,
|
||
MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES,
|
||
detect_archive_kind,
|
||
is_safe_archive_member_name,
|
||
process_archive,
|
||
)
|
||
from svc.files.attachment_assets import AttachmentAssetStore
|
||
from svc.files.tabular_attachment_store import save_dataframe, save_workbook
|
||
from svc.files.text_attachment_store import (
|
||
DEFAULT_TEXT_CHUNK_OVERLAP,
|
||
DEFAULT_TEXT_CHUNK_SIZE,
|
||
DEFAULT_TEXT_INLINE_MAX_CHARS,
|
||
save_text_document,
|
||
)
|
||
|
||
|
||
def _sniff_image_mime(data: bytes) -> str | None:
|
||
if len(data) < 12:
|
||
return None
|
||
if data[:8] == b"\x89PNG\r\n\x1a\n":
|
||
return "image/png"
|
||
if data[:3] == b"\xff\xd8\xff":
|
||
return "image/jpeg"
|
||
if data[:6] in (b"GIF87a", b"GIF89a"):
|
||
return "image/gif"
|
||
if len(data) >= 12 and data[:4] == b"RIFF" and data[8:12] == b"WEBP":
|
||
return "image/webp"
|
||
return None
|
||
|
||
|
||
def _ffprobe_exists() -> bool:
|
||
try:
|
||
p = subprocess.run(["ffprobe", "-version"], capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=3)
|
||
return p.returncode == 0
|
||
except Exception:
|
||
return False
|
||
|
||
|
||
def _extract_video_meta_from_ffprobe(obj: dict[str, Any]) -> dict[str, Any]:
|
||
fmt = obj.get("format") if isinstance(obj.get("format"), dict) else {}
|
||
streams = obj.get("streams") if isinstance(obj.get("streams"), list) else []
|
||
duration_sec = None
|
||
if isinstance(fmt, dict) and fmt.get("duration") is not None:
|
||
try:
|
||
duration_sec = float(fmt.get("duration"))
|
||
except Exception:
|
||
duration_sec = None
|
||
width = None
|
||
height = None
|
||
fps = None
|
||
for s in streams:
|
||
if not isinstance(s, dict):
|
||
continue
|
||
if str(s.get("codec_type") or "") != "video":
|
||
continue
|
||
if s.get("width") is not None and s.get("height") is not None:
|
||
try:
|
||
width = int(s.get("width") or 0) or None
|
||
height = int(s.get("height") or 0) or None
|
||
except Exception:
|
||
width = width
|
||
height = height
|
||
fr = str(s.get("avg_frame_rate") or s.get("r_frame_rate") or "").strip()
|
||
if fr and fr != "0/0" and "/" in fr:
|
||
try:
|
||
a, b = fr.split("/", 1)
|
||
fa = float(a)
|
||
fb = float(b)
|
||
if fb:
|
||
fps = fa / fb
|
||
except Exception:
|
||
fps = fps
|
||
break
|
||
out: dict[str, Any] = {}
|
||
if duration_sec is not None:
|
||
out["duration_sec"] = duration_sec
|
||
if width is not None:
|
||
out["width"] = width
|
||
if height is not None:
|
||
out["height"] = height
|
||
if fps is not None:
|
||
out["fps"] = fps
|
||
return out
|
||
|
||
|
||
def _ffprobe_video_meta(path: Path) -> dict[str, Any]:
|
||
if not _ffprobe_exists():
|
||
return {}
|
||
try:
|
||
p = subprocess.run(
|
||
["ffprobe", "-v", "error", "-print_format", "json", "-show_format", "-show_streams", str(path)],
|
||
capture_output=True,
|
||
text=True,
|
||
encoding="utf-8",
|
||
errors="replace",
|
||
timeout=8,
|
||
)
|
||
if p.returncode != 0:
|
||
return {}
|
||
obj = json.loads(p.stdout or "{}")
|
||
if not isinstance(obj, dict):
|
||
return {}
|
||
return _extract_video_meta_from_ffprobe(obj)
|
||
except Exception:
|
||
return {}
|
||
|
||
|
||
MAX_TEXT_CHARS = 12_000
|
||
EXCEL_PREVIEW_ROWS = 80
|
||
LARGE_TABLE_PREVIEW_ROWS = 20
|
||
MAX_ZIP_DEPTH = MAX_ARCHIVE_DEPTH
|
||
MAX_ZIP_FILE_COUNT = MAX_ARCHIVE_FILE_COUNT
|
||
MAX_ZIP_ENTRY_BYTES = MAX_ARCHIVE_ENTRY_BYTES
|
||
MAX_ZIP_TOTAL_UNCOMPRESSED_BYTES = MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES
|
||
DEFAULT_TABULAR_ROWS_READ = 5000
|
||
DEFAULT_TABULAR_COLUMNS = 200
|
||
DEFAULT_TABULAR_CELL_CHARS = 500
|
||
DEFAULT_TABULAR_TOOL_MODE_ENABLED = True
|
||
DEFAULT_TABULAR_TOOL_MODE_MIN_ROWS = 5_000
|
||
DEFAULT_TABULAR_TOOL_MODE_MAX_BYTES = 30 * 1024 * 1024
|
||
DEFAULT_LARGE_TABLE_PREVIEW_ROWS = LARGE_TABLE_PREVIEW_ROWS
|
||
MAX_ZIP_MEMBER_NAME_LENGTH = MAX_ARCHIVE_MEMBER_NAME_LENGTH
|
||
MAX_EXCEL_ZIP_FILE_COUNT = 500
|
||
MAX_EXCEL_ZIP_ENTRY_BYTES = 20 * 1024 * 1024
|
||
MAX_EXCEL_ZIP_TOTAL_UNCOMPRESSED_BYTES = 120 * 1024 * 1024
|
||
DEFAULT_MAX_EXCEL_SHEETS = 50
|
||
DEFAULT_TEXT_QUERY_TOP_K = 5
|
||
|
||
|
||
def _truncate_text(text: str, limit: int = MAX_TEXT_CHARS) -> str:
|
||
t = str(text or "")
|
||
if len(t) <= limit:
|
||
return t
|
||
return t[:limit] + f"\n\n...[truncated to {limit} chars]"
|
||
|
||
|
||
def _text_ref_summary(*, name: str, chars: int, chunks: int, source_kind: str, preview: str, top_k: int) -> str:
|
||
return _truncate_text(
|
||
(
|
||
f"# Text Document Summary\n"
|
||
f"- name: {name}\n"
|
||
f"- source_kind: {source_kind}\n"
|
||
f"- chars: {chars}\n"
|
||
f"- chunks: {chunks}\n"
|
||
f"- mode: text_ref\n"
|
||
f"- query_tool: query_text_attachment\n"
|
||
f"- note: use `text_id` from accompanying `text_ref` attachment; recommended top_k <= {top_k}\n\n"
|
||
f"## Preview (first {min(len(preview), 1500)} chars)\n"
|
||
f"{preview[:1500]}"
|
||
),
|
||
limit=MAX_TEXT_CHARS,
|
||
)
|
||
|
||
|
||
def _decode_text_bytes(data: bytes) -> str:
|
||
try:
|
||
return data.decode("utf-8")
|
||
except UnicodeDecodeError:
|
||
try:
|
||
return data.decode("gbk")
|
||
except Exception:
|
||
return "Error decoding text file (unknown encoding)"
|
||
|
||
|
||
def _clean_html_to_text(raw: str) -> str:
|
||
text = re.sub(r"(?is)<script[^>]*>.*?</script>", " ", str(raw or ""))
|
||
text = re.sub(r"(?is)<style[^>]*>.*?</style>", " ", text)
|
||
text = re.sub(r"(?is)<[^>]+>", " ", text)
|
||
text = re.sub(r"[ \t\r\f\v]+", " ", text)
|
||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||
return text.strip()
|
||
|
||
|
||
def _safe_int(raw: Any, default: int, *, min_value: int = 1, max_value: int = 2_000_000) -> int:
|
||
try:
|
||
value = int(raw)
|
||
except Exception:
|
||
return default
|
||
if value < min_value:
|
||
return default
|
||
return min(value, max_value)
|
||
|
||
|
||
@lru_cache(maxsize=1)
|
||
def _attachments_limits() -> dict[str, Any]:
|
||
cfg_path_raw = str(os.getenv("AIA_OCLAW_CONFIG_PATH") or "").strip()
|
||
cfg_path = Path(cfg_path_raw).expanduser() if cfg_path_raw else (Path(PROJECT_ROOT) / "oclaw.json").resolve()
|
||
if not cfg_path.is_absolute():
|
||
cfg_path = (Path(PROJECT_ROOT) / cfg_path).resolve()
|
||
tabular_cfg: dict[str, Any] = {}
|
||
try:
|
||
obj = json.loads(cfg_path.read_text(encoding="utf-8"))
|
||
tabular_cfg = (
|
||
(((obj or {}).get("plugins") or {}).get("entries") or {}).get("memory-wiki", {}).get("auto", {}).get("attachments", {}).get("tabular", {}) or {}
|
||
)
|
||
except Exception:
|
||
tabular_cfg = {}
|
||
return {
|
||
"rows_read": _safe_int(tabular_cfg.get("max_rows_read"), DEFAULT_TABULAR_ROWS_READ),
|
||
"columns": _safe_int(tabular_cfg.get("max_columns"), DEFAULT_TABULAR_COLUMNS),
|
||
"cell_chars": _safe_int(tabular_cfg.get("max_cell_chars"), DEFAULT_TABULAR_CELL_CHARS),
|
||
"max_excel_sheets": _safe_int(tabular_cfg.get("max_excel_sheets"), DEFAULT_MAX_EXCEL_SHEETS, max_value=500),
|
||
"large_table_preview_rows": _safe_int(
|
||
tabular_cfg.get("large_table_preview_rows"), DEFAULT_LARGE_TABLE_PREVIEW_ROWS, max_value=500
|
||
),
|
||
"tool_mode_enabled": bool(tabular_cfg.get("tool_mode_enabled", DEFAULT_TABULAR_TOOL_MODE_ENABLED)),
|
||
"tool_mode_min_rows": _safe_int(tabular_cfg.get("tool_mode_min_rows"), DEFAULT_TABULAR_TOOL_MODE_MIN_ROWS),
|
||
"tool_mode_max_bytes": _safe_int(
|
||
tabular_cfg.get("tool_mode_max_bytes"), DEFAULT_TABULAR_TOOL_MODE_MAX_BYTES, max_value=500 * 1024 * 1024
|
||
),
|
||
"text_inline_max_chars": _safe_int(
|
||
tabular_cfg.get("text_inline_max_chars"), DEFAULT_TEXT_INLINE_MAX_CHARS, max_value=200_000
|
||
),
|
||
"text_chunk_size": _safe_int(tabular_cfg.get("text_chunk_size"), DEFAULT_TEXT_CHUNK_SIZE, max_value=8_000),
|
||
"text_chunk_overlap": _safe_int(
|
||
tabular_cfg.get("text_chunk_overlap"), DEFAULT_TEXT_CHUNK_OVERLAP, max_value=4_000
|
||
),
|
||
"text_query_top_k": _safe_int(tabular_cfg.get("text_query_top_k"), DEFAULT_TEXT_QUERY_TOP_K, max_value=50),
|
||
"archive_max_depth": _safe_int(tabular_cfg.get("archive_max_depth"), MAX_ARCHIVE_DEPTH, max_value=10),
|
||
"archive_max_file_count": _safe_int(
|
||
tabular_cfg.get("archive_max_file_count"), MAX_ARCHIVE_FILE_COUNT, max_value=20_000
|
||
),
|
||
"archive_max_entry_bytes": _safe_int(
|
||
tabular_cfg.get("archive_max_entry_bytes"), MAX_ARCHIVE_ENTRY_BYTES, max_value=2_000_000_000
|
||
),
|
||
"archive_max_total_uncompressed_bytes": _safe_int(
|
||
tabular_cfg.get("archive_max_total_uncompressed_bytes"),
|
||
MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES,
|
||
max_value=5_000_000_000,
|
||
),
|
||
}
|
||
|
||
|
||
def clear_attachment_limits_cache() -> None:
|
||
_attachments_limits.cache_clear()
|
||
|
||
|
||
def _clip_tabular_cells(df: pd.DataFrame, *, max_cell_chars: int) -> tuple[pd.DataFrame, bool]:
|
||
clipped_any = False
|
||
|
||
def _clip_cell(value: Any) -> Any:
|
||
nonlocal clipped_any
|
||
if value is None:
|
||
return value
|
||
s = str(value)
|
||
if len(s) <= max_cell_chars:
|
||
return value
|
||
clipped_any = True
|
||
return s[:max_cell_chars] + "...[cell-truncated]"
|
||
|
||
out = df.copy()
|
||
for col in out.columns:
|
||
out[col] = out[col].map(_clip_cell)
|
||
return out, clipped_any
|
||
|
||
|
||
def _dataframe_summary_text(
|
||
df: pd.DataFrame,
|
||
*,
|
||
sampled: bool = False,
|
||
clipped_columns: bool = False,
|
||
clipped_cells: bool = False,
|
||
preview_rows: int = EXCEL_PREVIEW_ROWS,
|
||
) -> str:
|
||
rows = int(len(df.index))
|
||
cols = int(len(df.columns))
|
||
header = [str(x) for x in list(df.columns)]
|
||
dtypes = [f"{str(k)}:{str(v)}" for k, v in df.dtypes.items()]
|
||
show_rows = max(1, int(preview_rows or EXCEL_PREVIEW_ROWS))
|
||
sample = df.head(show_rows).to_string(index=False)
|
||
rows_scope_line = (
|
||
f"## Preview (first {min(rows, show_rows)} rows)"
|
||
if rows > show_rows
|
||
else f"## Full table included ({rows} rows)"
|
||
)
|
||
body = (
|
||
f"# Table Summary\n"
|
||
f"- rows: {rows}\n"
|
||
f"- cols: {cols}\n"
|
||
f"- sampled: {'yes' if sampled else 'no'}\n"
|
||
f"- clipped_columns: {'yes' if clipped_columns else 'no'}\n"
|
||
f"- clipped_cells: {'yes' if clipped_cells else 'no'}\n"
|
||
f"- columns: {', '.join(header)}\n"
|
||
f"- dtypes: {', '.join(dtypes)}\n\n"
|
||
f"{rows_scope_line}\n"
|
||
f"{sample}"
|
||
)
|
||
return _truncate_text(body)
|
||
|
||
|
||
def _pdf_summary_text(reader: PdfReader) -> str:
|
||
pages = list(getattr(reader, "pages", []) or [])
|
||
chunks: list[str] = [f"# PDF Summary\n- pages: {len(pages)}\n"]
|
||
for idx, page in enumerate(pages, start=1):
|
||
txt = ""
|
||
try:
|
||
txt = str(page.extract_text() or "").strip()
|
||
except Exception:
|
||
txt = ""
|
||
if not txt:
|
||
txt = "(empty page text)"
|
||
chunks.append(f"## Page {idx}\n{txt}")
|
||
return _truncate_text("\n\n".join(chunks))
|
||
|
||
|
||
def _is_safe_zip_member_name(name: str) -> bool:
|
||
return is_safe_archive_member_name(name)
|
||
|
||
|
||
def _validate_excel_zip_payload(data: bytes) -> str | None:
|
||
try:
|
||
with zipfile.ZipFile(io.BytesIO(data)) as zf:
|
||
infos = [x for x in zf.infolist() if not x.is_dir()]
|
||
if len(infos) > MAX_EXCEL_ZIP_FILE_COUNT:
|
||
return f"excel zip has too many entries ({len(infos)} > {MAX_EXCEL_ZIP_FILE_COUNT})"
|
||
total_uncompressed = 0
|
||
for info in infos:
|
||
if not _is_safe_zip_member_name(info.filename):
|
||
return f"excel zip contains unsafe path: {info.filename}"
|
||
fsize = int(info.file_size or 0)
|
||
total_uncompressed += fsize
|
||
if fsize > MAX_EXCEL_ZIP_ENTRY_BYTES:
|
||
return f"excel zip entry too large: {info.filename} ({fsize} > {MAX_EXCEL_ZIP_ENTRY_BYTES})"
|
||
if total_uncompressed > MAX_EXCEL_ZIP_TOTAL_UNCOMPRESSED_BYTES:
|
||
return (
|
||
"excel zip total uncompressed size too large "
|
||
f"({total_uncompressed} > {MAX_EXCEL_ZIP_TOTAL_UNCOMPRESSED_BYTES})"
|
||
)
|
||
return None
|
||
except Exception:
|
||
return "invalid excel zip payload"
|
||
|
||
|
||
def process_zip(zip_data: bytes, *, _depth: int = 0) -> list[dict[str, Any]]:
|
||
limits = _attachments_limits()
|
||
return process_archive(
|
||
archive_name="bundle.zip",
|
||
archive_data=zip_data,
|
||
process_member=lambda n, b, d: process_file_data(n, b, _zip_depth=d),
|
||
depth=_depth,
|
||
limits={
|
||
"max_depth": int(limits.get("archive_max_depth") or MAX_ARCHIVE_DEPTH),
|
||
"max_file_count": int(limits.get("archive_max_file_count") or MAX_ARCHIVE_FILE_COUNT),
|
||
"max_entry_bytes": int(limits.get("archive_max_entry_bytes") or MAX_ARCHIVE_ENTRY_BYTES),
|
||
"max_total_uncompressed_bytes": int(
|
||
limits.get("archive_max_total_uncompressed_bytes") or MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES
|
||
),
|
||
},
|
||
)
|
||
|
||
|
||
def process_file_data(name: str, data: bytes, *, _zip_depth: int = 0) -> list[dict[str, Any]]:
|
||
ext = name.split(".")[-1].lower() if "." in name else ""
|
||
attachments: list[dict[str, Any]] = []
|
||
|
||
if ext in ("png", "jpg", "jpeg", "webp", "gif", "jfif", "pjpeg", "bmp", "tif", "tiff"):
|
||
if ext == "png":
|
||
mime = "image/png"
|
||
elif ext in ("jpg", "jpeg", "jfif", "pjpeg"):
|
||
mime = "image/jpeg"
|
||
elif ext == "webp":
|
||
mime = "image/webp"
|
||
elif ext == "gif":
|
||
mime = "image/gif"
|
||
elif ext == "bmp":
|
||
mime = "image/bmp"
|
||
elif ext in ("tif", "tiff"):
|
||
mime = "image/tiff"
|
||
width = None
|
||
height = None
|
||
try:
|
||
with Image.open(io.BytesIO(data)) as im:
|
||
width, height = im.size
|
||
except Exception:
|
||
pass
|
||
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=mime, width=width, height=height)
|
||
attachments.append(
|
||
{
|
||
"type": "image_ref",
|
||
"name": meta.name,
|
||
"mime": meta.mime,
|
||
"attachment_id": meta.attachment_id,
|
||
"bytes": meta.bytes,
|
||
"width": meta.width,
|
||
"height": meta.height,
|
||
}
|
||
)
|
||
return attachments
|
||
|
||
sniff_mime = _sniff_image_mime(data)
|
||
if sniff_mime:
|
||
width = None
|
||
height = None
|
||
try:
|
||
with Image.open(io.BytesIO(data)) as im:
|
||
width, height = im.size
|
||
except Exception:
|
||
pass
|
||
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=sniff_mime, width=width, height=height)
|
||
attachments.append(
|
||
{
|
||
"type": "image_ref",
|
||
"name": meta.name,
|
||
"mime": meta.mime,
|
||
"attachment_id": meta.attachment_id,
|
||
"bytes": meta.bytes,
|
||
"width": meta.width,
|
||
"height": meta.height,
|
||
}
|
||
)
|
||
return attachments
|
||
|
||
if ext in ("mp4", "mov", "mkv", "webm", "avi", "m4v"):
|
||
mime = "video/mp4"
|
||
if ext == "webm":
|
||
mime = "video/webm"
|
||
elif ext == "mov":
|
||
mime = "video/quicktime"
|
||
elif ext == "mkv":
|
||
mime = "video/x-matroska"
|
||
elif ext == "avi":
|
||
mime = "video/x-msvideo"
|
||
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=mime)
|
||
probe_meta: dict[str, Any] = {}
|
||
try:
|
||
p = AttachmentAssetStore().get_local_path(str(meta.attachment_id))
|
||
if p is not None:
|
||
probe_meta = _ffprobe_video_meta(p)
|
||
except Exception:
|
||
probe_meta = {}
|
||
attachments.append(
|
||
{
|
||
"type": "video_ref",
|
||
"name": meta.name,
|
||
"mime": meta.mime,
|
||
"attachment_id": meta.attachment_id,
|
||
"bytes": meta.bytes,
|
||
**probe_meta,
|
||
}
|
||
)
|
||
attachments.append(
|
||
{
|
||
"type": "text",
|
||
"name": name,
|
||
"content": _truncate_text(
|
||
"".join(
|
||
[
|
||
"# Video Attachment\n",
|
||
f"- name: {meta.name}\n",
|
||
f"- mime: {meta.mime}\n",
|
||
f"- bytes: {meta.bytes}\n",
|
||
(
|
||
f"- duration_sec: {probe_meta.get('duration_sec')}\n"
|
||
if probe_meta.get("duration_sec") is not None
|
||
else ""
|
||
),
|
||
(
|
||
f"- size: {probe_meta.get('width')}x{probe_meta.get('height')}\n"
|
||
if probe_meta.get("width") and probe_meta.get("height")
|
||
else ""
|
||
),
|
||
(f"- fps: {probe_meta.get('fps')}\n" if probe_meta.get("fps") is not None else ""),
|
||
"- mode: video_ref\n",
|
||
"- query_tool: query_video_attachment\n",
|
||
"- note: use `attachment_id` from accompanying `video_ref` attachment.\n",
|
||
]
|
||
),
|
||
limit=MAX_TEXT_CHARS,
|
||
),
|
||
}
|
||
)
|
||
return attachments
|
||
|
||
if ext in ("xlsx", "xls", "csv"):
|
||
try:
|
||
if ext == "xlsx":
|
||
zip_err = _validate_excel_zip_payload(data)
|
||
if zip_err:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing excel/csv: {zip_err}")})
|
||
return attachments
|
||
limits = _attachments_limits()
|
||
rows_read = int(limits.get("rows_read") or DEFAULT_TABULAR_ROWS_READ)
|
||
max_columns = int(limits.get("columns") or DEFAULT_TABULAR_COLUMNS)
|
||
max_cell_chars = int(limits.get("cell_chars") or DEFAULT_TABULAR_CELL_CHARS)
|
||
max_excel_sheets = int(limits.get("max_excel_sheets") or DEFAULT_MAX_EXCEL_SHEETS)
|
||
large_table_preview_rows = int(
|
||
limits.get("large_table_preview_rows") or DEFAULT_LARGE_TABLE_PREVIEW_ROWS
|
||
)
|
||
tool_mode_enabled = bool(limits.get("tool_mode_enabled", DEFAULT_TABULAR_TOOL_MODE_ENABLED))
|
||
tool_mode_min_rows = int(limits.get("tool_mode_min_rows") or DEFAULT_TABULAR_TOOL_MODE_MIN_ROWS)
|
||
tool_mode_max_bytes = int(limits.get("tool_mode_max_bytes") or DEFAULT_TABULAR_TOOL_MODE_MAX_BYTES)
|
||
if ext == "csv":
|
||
df = pd.read_csv(io.BytesIO(data), nrows=rows_read)
|
||
else:
|
||
df = pd.read_excel(io.BytesIO(data), nrows=rows_read)
|
||
sampled = int(len(df.index)) >= rows_read
|
||
estimated_rows = 0
|
||
if ext == "csv":
|
||
estimated_rows = max(0, int(data.count(b"\n")) - 1)
|
||
else:
|
||
estimated_rows = int(len(df.index))
|
||
clipped_columns = int(len(df.columns)) > max_columns
|
||
if clipped_columns:
|
||
df = df.iloc[:, :max_columns]
|
||
df, clipped_cells = _clip_tabular_cells(df, max_cell_chars=max_cell_chars)
|
||
tool_mode = bool(
|
||
tool_mode_enabled
|
||
and int(len(data or b"")) <= tool_mode_max_bytes
|
||
and (estimated_rows >= tool_mode_min_rows or (ext in ("xlsx", "xls") and sampled))
|
||
)
|
||
attachments.append(
|
||
{
|
||
"type": "text",
|
||
"name": name,
|
||
"content": _dataframe_summary_text(
|
||
df,
|
||
sampled=sampled,
|
||
clipped_columns=clipped_columns,
|
||
clipped_cells=clipped_cells,
|
||
preview_rows=(large_table_preview_rows if tool_mode else EXCEL_PREVIEW_ROWS),
|
||
)
|
||
+ (
|
||
(
|
||
f"\n\n# Large Table Mode\n- estimated_rows: {estimated_rows}\n- query_tool: query_tabular_attachment\n"
|
||
"- note: use the `table_id` from the accompanying `tabular_ref` attachment to query full data."
|
||
)
|
||
if tool_mode
|
||
else ""
|
||
),
|
||
}
|
||
)
|
||
if tool_mode:
|
||
workbook_sheets: dict[str, pd.DataFrame]
|
||
clipped_sheets = False
|
||
if ext == "csv":
|
||
full_df = pd.read_csv(io.BytesIO(data), dtype=str)
|
||
if int(len(full_df.columns)) > max_columns:
|
||
full_df = full_df.iloc[:, :max_columns]
|
||
full_df, _ = _clip_tabular_cells(full_df, max_cell_chars=max_cell_chars)
|
||
workbook_sheets = {"Sheet1": full_df}
|
||
else:
|
||
excel_obj = pd.read_excel(io.BytesIO(data), dtype=str, sheet_name=None)
|
||
workbook_sheets = {}
|
||
sheet_items = list((excel_obj or {}).items())
|
||
if len(sheet_items) > max_excel_sheets:
|
||
clipped_sheets = True
|
||
sheet_items = sheet_items[:max_excel_sheets]
|
||
for sheet_name, sheet_df in sheet_items:
|
||
sdf = sheet_df
|
||
if int(len(sdf.columns)) > max_columns:
|
||
sdf = sdf.iloc[:, :max_columns]
|
||
sdf, _ = _clip_tabular_cells(sdf, max_cell_chars=max_cell_chars)
|
||
workbook_sheets[str(sheet_name or f"Sheet{len(workbook_sheets)+1}")] = sdf
|
||
mime = "text/csv" if ext == "csv" else "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"
|
||
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime=mime)
|
||
table_meta = (
|
||
save_dataframe(attachment_id=str(meta.attachment_id), name=name, df=workbook_sheets.get("Sheet1", pd.DataFrame()))
|
||
if ext == "csv"
|
||
else save_workbook(attachment_id=str(meta.attachment_id), name=name, sheets=workbook_sheets)
|
||
)
|
||
attachments.append(
|
||
{
|
||
"type": "tabular_ref",
|
||
"name": str(name),
|
||
"attachment_id": str(meta.attachment_id),
|
||
"table_id": str(table_meta.get("table_id") or ""),
|
||
"rows": int(table_meta.get("rows") or 0),
|
||
"cols": int(table_meta.get("cols") or 0),
|
||
"columns": list(table_meta.get("columns") or []),
|
||
"sheets": list(table_meta.get("sheets") or []),
|
||
}
|
||
)
|
||
if clipped_sheets:
|
||
attachments.append(
|
||
{
|
||
"type": "text",
|
||
"name": f"{name}.sheet-limit",
|
||
"content": (
|
||
f"Excel workbook contains too many sheets; only the first {max_excel_sheets} sheets were stored "
|
||
"for tool-mode querying. Adjust max_excel_sheets if full coverage is required."
|
||
),
|
||
}
|
||
)
|
||
except Exception as e:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing excel/csv: {e}")})
|
||
|
||
elif ext == "docx":
|
||
try:
|
||
doc = Document(io.BytesIO(data))
|
||
text = "\n".join([p.text for p in doc.paragraphs])
|
||
limits = _attachments_limits()
|
||
inline_cap = int(limits.get("text_inline_max_chars") or DEFAULT_TEXT_INLINE_MAX_CHARS)
|
||
if len(str(text or "")) <= inline_cap:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(text, limit=inline_cap)})
|
||
else:
|
||
meta = AttachmentAssetStore().save_bytes(
|
||
data,
|
||
filename=name,
|
||
mime="application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||
)
|
||
text_meta = save_text_document(
|
||
attachment_id=str(meta.attachment_id),
|
||
name=name,
|
||
text=str(text or ""),
|
||
source_kind="docx",
|
||
chunk_size=int(limits.get("text_chunk_size") or DEFAULT_TEXT_CHUNK_SIZE),
|
||
chunk_overlap=int(limits.get("text_chunk_overlap") or DEFAULT_TEXT_CHUNK_OVERLAP),
|
||
)
|
||
top_k = int(limits.get("text_query_top_k") or DEFAULT_TEXT_QUERY_TOP_K)
|
||
attachments.append(
|
||
{
|
||
"type": "text",
|
||
"name": name,
|
||
"content": _text_ref_summary(
|
||
name=name,
|
||
chars=int(text_meta.get("chars") or 0),
|
||
chunks=int(text_meta.get("chunks") or 0),
|
||
source_kind="docx",
|
||
preview=str(text or ""),
|
||
top_k=top_k,
|
||
),
|
||
}
|
||
)
|
||
attachments.append(
|
||
{
|
||
"type": "text_ref",
|
||
"name": str(name),
|
||
"attachment_id": str(meta.attachment_id),
|
||
"text_id": str(text_meta.get("text_id") or ""),
|
||
"chars": int(text_meta.get("chars") or 0),
|
||
"chunks": int(text_meta.get("chunks") or 0),
|
||
"source_kind": "docx",
|
||
}
|
||
)
|
||
except Exception as e:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing docx: {e}")})
|
||
|
||
elif ext == "pdf":
|
||
try:
|
||
reader = PdfReader(io.BytesIO(data))
|
||
text = _pdf_summary_text(reader)
|
||
limits = _attachments_limits()
|
||
inline_cap = int(limits.get("text_inline_max_chars") or DEFAULT_TEXT_INLINE_MAX_CHARS)
|
||
if len(str(text or "")) <= inline_cap:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(text, limit=inline_cap)})
|
||
else:
|
||
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime="application/pdf")
|
||
text_meta = save_text_document(
|
||
attachment_id=str(meta.attachment_id),
|
||
name=name,
|
||
text=str(text or ""),
|
||
source_kind="pdf",
|
||
chunk_size=int(limits.get("text_chunk_size") or DEFAULT_TEXT_CHUNK_SIZE),
|
||
chunk_overlap=int(limits.get("text_chunk_overlap") or DEFAULT_TEXT_CHUNK_OVERLAP),
|
||
)
|
||
top_k = int(limits.get("text_query_top_k") or DEFAULT_TEXT_QUERY_TOP_K)
|
||
attachments.append(
|
||
{
|
||
"type": "text",
|
||
"name": name,
|
||
"content": _text_ref_summary(
|
||
name=name,
|
||
chars=int(text_meta.get("chars") or 0),
|
||
chunks=int(text_meta.get("chunks") or 0),
|
||
source_kind="pdf",
|
||
preview=str(text or ""),
|
||
top_k=top_k,
|
||
),
|
||
}
|
||
)
|
||
attachments.append(
|
||
{
|
||
"type": "text_ref",
|
||
"name": str(name),
|
||
"attachment_id": str(meta.attachment_id),
|
||
"text_id": str(text_meta.get("text_id") or ""),
|
||
"chars": int(text_meta.get("chars") or 0),
|
||
"chunks": int(text_meta.get("chunks") or 0),
|
||
"source_kind": "pdf",
|
||
}
|
||
)
|
||
except Exception as e:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(f"Error parsing pdf: {e}")})
|
||
|
||
elif ext in ("txt", "py", "js", "html", "css", "md", "json", "yaml", "yml", "sh", "log"):
|
||
text = _decode_text_bytes(data)
|
||
if ext == "html":
|
||
text = _clean_html_to_text(text)
|
||
limits = _attachments_limits()
|
||
inline_cap = int(limits.get("text_inline_max_chars") or DEFAULT_TEXT_INLINE_MAX_CHARS)
|
||
if len(str(text or "")) <= inline_cap:
|
||
attachments.append({"type": "text", "name": name, "content": _truncate_text(text, limit=inline_cap)})
|
||
else:
|
||
meta = AttachmentAssetStore().save_bytes(data, filename=name, mime="text/plain")
|
||
text_meta = save_text_document(
|
||
attachment_id=str(meta.attachment_id),
|
||
name=name,
|
||
text=str(text or ""),
|
||
source_kind=ext or "text",
|
||
chunk_size=int(limits.get("text_chunk_size") or DEFAULT_TEXT_CHUNK_SIZE),
|
||
chunk_overlap=int(limits.get("text_chunk_overlap") or DEFAULT_TEXT_CHUNK_OVERLAP),
|
||
)
|
||
top_k = int(limits.get("text_query_top_k") or DEFAULT_TEXT_QUERY_TOP_K)
|
||
attachments.append(
|
||
{
|
||
"type": "text",
|
||
"name": name,
|
||
"content": _text_ref_summary(
|
||
name=name,
|
||
chars=int(text_meta.get("chars") or 0),
|
||
chunks=int(text_meta.get("chunks") or 0),
|
||
source_kind=str(text_meta.get("source_kind") or ext or "text"),
|
||
preview=str(text or ""),
|
||
top_k=top_k,
|
||
),
|
||
}
|
||
)
|
||
attachments.append(
|
||
{
|
||
"type": "text_ref",
|
||
"name": str(name),
|
||
"attachment_id": str(meta.attachment_id),
|
||
"text_id": str(text_meta.get("text_id") or ""),
|
||
"chars": int(text_meta.get("chars") or 0),
|
||
"chunks": int(text_meta.get("chunks") or 0),
|
||
"source_kind": str(text_meta.get("source_kind") or ext or "text"),
|
||
}
|
||
)
|
||
|
||
elif detect_archive_kind(name):
|
||
limits = _attachments_limits()
|
||
attachments.extend(
|
||
process_archive(
|
||
archive_name=name,
|
||
archive_data=data,
|
||
process_member=lambda n, b, d: process_file_data(n, b, _zip_depth=d),
|
||
depth=_zip_depth,
|
||
limits={
|
||
"max_depth": int(limits.get("archive_max_depth") or MAX_ARCHIVE_DEPTH),
|
||
"max_file_count": int(limits.get("archive_max_file_count") or MAX_ARCHIVE_FILE_COUNT),
|
||
"max_entry_bytes": int(limits.get("archive_max_entry_bytes") or MAX_ARCHIVE_ENTRY_BYTES),
|
||
"max_total_uncompressed_bytes": int(
|
||
limits.get("archive_max_total_uncompressed_bytes") or MAX_ARCHIVE_TOTAL_UNCOMPRESSED_BYTES
|
||
),
|
||
},
|
||
)
|
||
)
|
||
|
||
return attachments
|
||
|
||
|
||
__all__ = ["process_zip", "process_file_data", "clear_attachment_limits_cache"]
|