diff --git a/docs/presentations/oclaw-netx-ops-expert.md b/docs/presentations/oclaw-netx-ops-expert.md index 63b4f423..cae4e763 100644 --- a/docs/presentations/oclaw-netx-ops-expert.md +++ b/docs/presentations/oclaw-netx-ops-expert.md @@ -561,7 +561,7 @@ Recipe 中 `goal` 示例: ```mermaid flowchart LR - A[write_file / run_command\n生成 report.xlsx] --> B[save_deliverable_attachment\npath=...] + A[write_xlsx\n返回 attachment_id] --> B[save_deliverable_attachment\nattachment_id=...] B --> C[Agent 文字回复摘要] C --> D[出站携带附件] D --> E[WhatsApp Sidecar 发送文件] diff --git a/runtime/tools/public/save_deliverable_attachment_tool.py b/runtime/tools/public/save_deliverable_attachment_tool.py index 902d9798..da9bda87 100644 --- a/runtime/tools/public/save_deliverable_attachment_tool.py +++ b/runtime/tools/public/save_deliverable_attachment_tool.py @@ -69,8 +69,8 @@ def save_deliverable_attachment_tool() -> ToolSpec: description=( "Mark an attachment for outbound channel delivery (WhatsApp/WeChat). " "Required before the user receives any generated file, image, or video in a messaging channel. " - "Use path for workspace files (after write_file or run_command), or attachment_id for assets " - "already in the attachment store (e.g. after cloudflare_image_generate). " + "Prefer attachment_id for assets already in the store (write_xlsx, cloudflare_image_generate, etc.). " + "Use path for workspace files (after write_file or run_command). " "Generating content alone does not send attachments to the channel." ), parameters={ @@ -82,7 +82,10 @@ def save_deliverable_attachment_tool() -> ToolSpec: }, "attachment_id": { "type": "string", - "description": "Existing attachment_id from image/video generation tools.", + "description": ( + "Existing attachment_id from generation tools " + "(write_xlsx, cloudflare_image_generate, image_edit, etc.)." + ), }, "name": { "type": "string", diff --git a/runtime/tools/public/write_xlsx_tool.py b/runtime/tools/public/write_xlsx_tool.py new file mode 100644 index 00000000..f4e3790f --- /dev/null +++ b/runtime/tools/public/write_xlsx_tool.py @@ -0,0 +1,252 @@ +from __future__ import annotations + +import io +import re +from typing import Any + +from openpyxl import Workbook +from openpyxl.utils import get_column_letter + +from runtime.tools.base import ToolSpec +from runtime.tools.path_guard import resolve_workspace_path +from svc.files.attachment_assets import AttachmentAssetStore + +XLSX_MIME = "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" +_MAX_SHEETS = 50 +_MAX_COLS = 200 +_MAX_ROWS = 100_000 +_MAX_CELL_CHARS = 32767 +_MAX_AUTO_WIDTH = 48 +_SHEET_NAME_RE = re.compile(r"[\[\]\*\?/\\:]") + + +def _safe_sheet_name(raw: str, *, index: int, used: set[str]) -> str: + name = _SHEET_NAME_RE.sub("_", str(raw or "").strip()) or f"Sheet{index}" + name = name[:31] + base = name + n = 2 + while name.lower() in used: + suffix = f"_{n}" + name = (base[: max(1, 31 - len(suffix))] + suffix)[:31] + n += 1 + used.add(name.lower()) + return name + + +def _cell_value(value: Any) -> Any: + if value is None: + return None + if isinstance(value, bool): + return value + if isinstance(value, (int, float)): + return value + text = str(value) + if len(text) > _MAX_CELL_CHARS: + return text[:_MAX_CELL_CHARS] + return text + + +def _normalize_headers(raw: Any, *, col_count: int) -> list[str]: + if isinstance(raw, list) and raw: + headers = [str(h) if h is not None else "" for h in raw[:_MAX_COLS]] + while len(headers) < col_count: + headers.append(f"col_{len(headers) + 1}") + return headers[: max(col_count, len(headers))][:_MAX_COLS] + return [f"col_{i + 1}" for i in range(max(1, col_count))] + + +def _normalize_rows(raw: Any) -> tuple[list[list[Any]], int]: + if not isinstance(raw, list): + return [], 0 + rows: list[list[Any]] = [] + max_cols = 0 + for item in raw[:_MAX_ROWS]: + if isinstance(item, dict): + # Stable insertion order for dict rows without headers mapping. + values = list(item.values()) + elif isinstance(item, (list, tuple)): + values = list(item) + else: + values = [item] + values = [_cell_value(v) for v in values[:_MAX_COLS]] + max_cols = max(max_cols, len(values)) + rows.append(values) + return rows, max_cols + + +def _apply_auto_width(ws: Any, *, col_count: int, sample_rows: list[list[Any]], headers: list[str]) -> None: + widths: list[int] = [] + for idx in range(col_count): + best = len(str(headers[idx])) if idx < len(headers) else 0 + for row in sample_rows[:200]: + if idx < len(row) and row[idx] is not None: + best = max(best, min(_MAX_AUTO_WIDTH, len(str(row[idx])))) + widths.append(min(_MAX_AUTO_WIDTH, max(8, best + 2))) + for idx, width in enumerate(widths, start=1): + ws.column_dimensions[get_column_letter(idx)].width = width + + +def _build_workbook(sheets: list[dict[str, Any]], *, freeze_header: bool, auto_width: bool) -> tuple[bytes, list[dict[str, Any]]]: + wb = Workbook() + # Remove the default sheet; recreate from input for predictable naming. + default = wb.active + wb.remove(default) + + used_names: set[str] = set() + summary: list[dict[str, Any]] = [] + for i, sheet in enumerate(sheets, start=1): + if not isinstance(sheet, dict): + continue + rows, inferred_cols = _normalize_rows(sheet.get("rows")) + headers_raw = sheet.get("headers") + if isinstance(headers_raw, list) and headers_raw: + col_count = max(len(headers_raw), inferred_cols, 1) + else: + col_count = max(inferred_cols, 1) + headers = _normalize_headers(headers_raw, col_count=col_count) + col_count = min(_MAX_COLS, max(len(headers), col_count)) + headers = headers[:col_count] + + title = _safe_sheet_name(str(sheet.get("name") or ""), index=i, used=used_names) + ws = wb.create_sheet(title=title) + ws.append(headers) + for row in rows: + padded = list(row[:col_count]) + [None] * max(0, col_count - len(row)) + ws.append(padded) + if freeze_header: + ws.freeze_panes = "A2" + if auto_width: + _apply_auto_width(ws, col_count=col_count, sample_rows=rows, headers=headers) + summary.append({"name": title, "columns": col_count, "rows": len(rows)}) + + if not summary: + ws = wb.create_sheet(title="Sheet1") + ws.append(["col_1"]) + summary.append({"name": "Sheet1", "columns": 1, "rows": 0}) + + buf = io.BytesIO() + wb.save(buf) + return buf.getvalue(), summary + + +def write_xlsx_tool() -> ToolSpec: + def _handler(args: dict[str, Any]) -> dict[str, Any]: + sheets_raw = args.get("sheets") + if not isinstance(sheets_raw, list) or not sheets_raw: + return {"ok": False, "error": "sheets_required"} + if len(sheets_raw) > _MAX_SHEETS: + return {"ok": False, "error": "too_many_sheets", "max_sheets": _MAX_SHEETS} + + filename = str(args.get("name") or "").strip() or "report.xlsx" + if not filename.lower().endswith(".xlsx"): + filename = f"{filename}.xlsx" + freeze_header = args.get("freeze_header") is not False + auto_width = args.get("auto_width") is not False + raw_path = str(args.get("path") or "").strip().strip('"').strip("'") + + try: + blob, summary = _build_workbook( + [s for s in sheets_raw if isinstance(s, dict)], + freeze_header=freeze_header, + auto_width=auto_width, + ) + except Exception as exc: + return {"ok": False, "error": "xlsx_build_failed", "detail": str(exc)} + + path_written = "" + if raw_path: + try: + p = resolve_workspace_path(raw_path) + except ValueError as exc: + return {"ok": False, "error": str(exc)} + if not str(p).lower().endswith(".xlsx"): + p = p.with_suffix(".xlsx") + try: + p.parent.mkdir(parents=True, exist_ok=True) + p.write_bytes(blob) + path_written = str(p) + except Exception as exc: + return {"ok": False, "error": "path_write_failed", "detail": str(exc)} + + meta = AttachmentAssetStore().save_bytes(blob, filename=filename, mime=XLSX_MIME) + out: dict[str, Any] = { + "ok": True, + "attachment_id": meta.attachment_id, + "name": meta.name, + "mime": meta.mime, + "bytes": meta.bytes, + "sheet_count": len(summary), + "sheets": summary, + "hint": ( + "Excel saved to attachment store. Not sent to channel yet. " + "If the user asked to receive the file, call save_deliverable_attachment " + "with this attachment_id." + ), + } + if path_written: + out["path"] = path_written + return out + + return ToolSpec( + name="write_xlsx", + description=( + "Build a real .xlsx workbook from structured sheet data (headers + rows) and save it " + "to the attachment store. Returns attachment_id/name/mime/bytes — does NOT mark " + "deliverable and does NOT send to WhatsApp/WeChat. To deliver, call " + "save_deliverable_attachment(attachment_id=...). Prefer this over run_command/openpyxl." + ), + parameters={ + "type": "object", + "properties": { + "sheets": { + "type": "array", + "description": "One or more sheets. Each item: {name, headers[], rows[][]}.", + "items": { + "type": "object", + "properties": { + "name": {"type": "string", "description": "Sheet tab name (max 31 chars)."}, + "headers": { + "type": "array", + "items": {"type": "string"}, + "description": "Column headers (first row).", + }, + "rows": { + "type": "array", + "description": "Data rows: each row is an array of cell values (string/number/bool/null).", + "items": {"type": "array"}, + }, + }, + "required": ["rows"], + "additionalProperties": False, + }, + }, + "name": { + "type": "string", + "description": "Download filename, e.g. alarm_summary.xlsx", + }, + "path": { + "type": "string", + "description": "Optional workspace path to also write the .xlsx file (mirror).", + }, + "freeze_header": { + "type": "boolean", + "default": True, + "description": "Freeze the header row (default true).", + }, + "auto_width": { + "type": "boolean", + "default": True, + "description": "Best-effort column width from sample cells (default true).", + }, + }, + "required": ["sheets"], + "additionalProperties": False, + }, + handler=_handler, + tags=frozenset({"public", "workspace", "attachment", "xlsx"}), + read_only=False, + risk_level="low", + ) + + +__all__ = ["write_xlsx_tool"] diff --git a/skills/_workspace/public/channel-file-delivery/SKILL.md b/skills/_workspace/public/channel-file-delivery/SKILL.md index f6f6ee07..b1c2988e 100644 --- a/skills/_workspace/public/channel-file-delivery/SKILL.md +++ b/skills/_workspace/public/channel-file-delivery/SKILL.md @@ -12,7 +12,7 @@ description: "在 WhatsApp/微信等渠道会话中,把生成的附件发回 ## 关键规则(统一) 1. **生成工具不会自动发送附件** - `write_file`、`run_command`、`cloudflare_image_generate` 等只产生内容,渠道出站看不到。 + `write_xlsx`、`write_file`、`run_command`、`cloudflare_image_generate` 等只产生内容,渠道出站看不到。 2. **必须调用 `save_deliverable_attachment`** 生成完成后用该工具标记 `deliverable`,系统才会随回复发送。 @@ -22,24 +22,32 @@ description: "在 WhatsApp/微信等渠道会话中,把生成的附件发回 | 生成方式 | `save_deliverable_attachment` 参数 | |----------|--------------------------------------| -| workspace 文件(csv/txt/xlsx) | `path="data/workspace/..."` | -| 生图/生视频(已有 attachment_id) | `attachment_id="..."` | +| `write_xlsx` / 生图等(已有 attachment_id) | `attachment_id="..."` | +| workspace 文件(csv/txt 等) | `path="data/workspace/..."` | **用户上传的文件不要回传**;那是分析输入,不是生成输出。 -## 工具选择(文本 vs 二进制) +## 工具选择(文本 vs Excel vs 二进制) | 目标格式 | 用哪个工具 | 说明 | |----------|------------|------| +| `.xlsx` Excel | **`write_xlsx`** | 传 sheets/headers/rows;返回 `attachment_id`;**禁止**再用 `run_command`+openpyxl | | `.csv`、`.txt`、`.md`、`.json` 等纯文本 | `write_file` | 只能写文本内容 | -| `.xlsx`、`.xls` 等 Excel | `run_command` | 用 Python(`openpyxl` 已安装) | | 图片 | `cloudflare_image_generate` 等 | 生成后用 `attachment_id` 标记 | | 视频 | 对应生成工具 | 生成后用 `attachment_id` 标记 | -**不要声称没有 `run_command`。** 用户要真 `.xlsx` 时优先用 `run_command`,不要只写 CSV 代替。 +**不要声称没有 `write_xlsx`。** 用户要真 `.xlsx` 时必须用 `write_xlsx`,不要只写 CSV 代替。 ## 推荐流程 +**Excel(xlsx)— 优先:** + +```text +1. write_xlsx(sheets=[...], name="report.xlsx") → 得到 attachment_id +2. 若用户要求发文件:save_deliverable_attachment(attachment_id="...") +3. 文字回复摘要 +``` + **文本文件(CSV/TXT):** ```text @@ -48,14 +56,6 @@ description: "在 WhatsApp/微信等渠道会话中,把生成的附件发回 3. 文字回复 ``` -**Excel(xlsx):** - -```text -1. run_command → python 写入 data/workspace/tmp/report.xlsx -2. save_deliverable_attachment(path="data/workspace/tmp/report.xlsx", name="report.xlsx") -3. 文字回复 -``` - **图片:** ```text @@ -70,12 +70,19 @@ description: "在 WhatsApp/微信等渠道会话中,把生成的附件发回 用户:帮我把统计结果导出 Excel 发我 步骤: -1. run_command(示例): - python -c "from openpyxl import Workbook; wb=Workbook(); ws=wb.active; ws.append(['姓名','数量']); ws.append(['A',10]); wb.save('data/workspace/tmp/summary.xlsx')" -2. save_deliverable_attachment(path="data/workspace/tmp/summary.xlsx", name="summary.xlsx") +1. write_xlsx: + name="summary.xlsx" + sheets=[{ + "name": "汇总", + "headers": ["姓名", "数量"], + "rows": [["A", 10], ["B", 3]] + }] +2. save_deliverable_attachment(attachment_id=<上一步返回的 attachment_id>) 3. 回复摘要 ``` +若只需在 workspace 留一份副本(不投递),可额外传 `path="data/workspace/tmp/summary.xlsx"`;**是否发给渠道仍由你决定是否调用 save_deliverable_attachment**。 + ## 限制 - 每轮回复通常只发送**第一个** deliverable 附件(约 8MB 上限)。 diff --git a/svc/files/attachment_assets.py b/svc/files/attachment_assets.py index a0264ff2..c9b3bf68 100644 --- a/svc/files/attachment_assets.py +++ b/svc/files/attachment_assets.py @@ -23,6 +23,11 @@ _ATTACHMENT_BLOB_EXTENSIONS: Final[tuple[str, ...]] = ( ".mp4", ".webm", ".mov", + ".xlsx", + ".xls", + ".csv", + ".txt", + ".pdf", "", # extensionless fallback ) @@ -53,6 +58,16 @@ def _ext_from_mime(mime: str | None) -> str: return ".webm" if m in ("video/quicktime", "video/mov"): return ".mov" + if m == "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet": + return ".xlsx" + if m == "application/vnd.ms-excel": + return ".xls" + if m == "text/csv": + return ".csv" + if m in ("text/plain", "text/markdown"): + return ".txt" + if m == "application/pdf": + return ".pdf" return "" @@ -72,6 +87,16 @@ def _guess_mime_from_ext(ext: str) -> str: return "video/webm" if e == "mov": return "video/quicktime" + if e == "xlsx": + return "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" + if e == "xls": + return "application/vnd.ms-excel" + if e == "csv": + return "text/csv" + if e in ("txt", "md"): + return "text/plain" + if e == "pdf": + return "application/pdf" return "application/octet-stream" diff --git a/tests/test_write_xlsx_tool.py b/tests/test_write_xlsx_tool.py new file mode 100644 index 00000000..d28e06de --- /dev/null +++ b/tests/test_write_xlsx_tool.py @@ -0,0 +1,140 @@ +from __future__ import annotations + +import io + +from openpyxl import load_workbook + +from runtime.chat.tool_runtime import _attachments_from_tool_result, _ref_type_for_mime +from runtime.tools.public.write_xlsx_tool import XLSX_MIME, write_xlsx_tool +from svc.files.attachment_assets import AttachmentAssetStore + + +def test_attachment_store_keeps_xlsx_extension(tmp_path) -> None: + store = AttachmentAssetStore(root_dir=tmp_path / "att") + meta = store.save_bytes(b"PK\x03\x04fake", filename="report.xlsx", mime=XLSX_MIME) + assert meta.mime == XLSX_MIME + path = store.get_local_path(meta.attachment_id) + assert path is not None + assert path.suffix == ".xlsx" + blob, loaded = store.load_bytes(meta.attachment_id) + assert blob.startswith(b"PK") + assert loaded is not None + assert loaded.name == "report.xlsx" + + +def test_write_xlsx_returns_attachment_id_without_deliverable(tmp_path, monkeypatch) -> None: + store = AttachmentAssetStore(root_dir=tmp_path / "att") + monkeypatch.setattr( + "runtime.tools.public.write_xlsx_tool.AttachmentAssetStore", + lambda root_dir=None: store if root_dir is None else AttachmentAssetStore(root_dir=root_dir), + ) + + spec = write_xlsx_tool() + out = spec.handler( + { + "name": "alarm_summary.xlsx", + "sheets": [ + { + "name": "按网元", + "headers": ["host_name", "severity", "count"], + "rows": [["NE-A", "critical", 12], ["NE-B", "major", 5]], + }, + { + "name": "汇总", + "headers": ["metric", "value"], + "rows": [["total", 17]], + }, + ], + } + ) + assert out.get("ok") is True + assert out.get("attachment_id") + assert out.get("name") == "alarm_summary.xlsx" + assert out.get("mime") == XLSX_MIME + assert out.get("deliverable") is not True + assert out.get("sheet_count") == 2 + assert "hint" in out + + blob, meta = store.load_bytes(str(out["attachment_id"])) + assert meta is not None + assert blob[:2] == b"PK" + wb = load_workbook(io.BytesIO(blob)) + assert wb.sheetnames == ["按网元", "汇总"] + ws = wb["按网元"] + assert [c.value for c in ws[1]] == ["host_name", "severity", "count"] + assert ws["A2"].value == "NE-A" + assert ws["C2"].value == 12 + assert ws.freeze_panes == "A2" + + +def test_write_xlsx_optional_workspace_path(tmp_path, monkeypatch) -> None: + store = AttachmentAssetStore(root_dir=tmp_path / "att") + monkeypatch.setattr( + "runtime.tools.public.write_xlsx_tool.AttachmentAssetStore", + lambda root_dir=None: store if root_dir is None else AttachmentAssetStore(root_dir=root_dir), + ) + target = tmp_path / "workspace" / "tmp" / "out.xlsx" + monkeypatch.setattr( + "runtime.tools.public.write_xlsx_tool.resolve_workspace_path", + lambda raw: target, + ) + + spec = write_xlsx_tool() + out = spec.handler( + { + "path": "data/workspace/tmp/out.xlsx", + "sheets": [{"name": "S1", "headers": ["a"], "rows": [[1]]}], + } + ) + assert out.get("ok") is True + assert out.get("attachment_id") + assert out.get("path") == str(target) + assert target.exists() + assert target.read_bytes()[:2] == b"PK" + + +def test_write_xlsx_requires_sheets() -> None: + spec = write_xlsx_tool() + out = spec.handler({}) + assert out.get("ok") is False + assert out.get("error") == "sheets_required" + + +def test_write_xlsx_tool_result_maps_to_binary_ref() -> None: + assert _ref_type_for_mime(XLSX_MIME) == "binary_ref" + refs = _attachments_from_tool_result( + { + "ok": True, + "attachment_id": "a" * 64, + "name": "r.xlsx", + "mime": XLSX_MIME, + "bytes": 12, + } + ) + assert len(refs) == 1 + assert refs[0]["type"] == "binary_ref" + assert refs[0].get("deliverable") is not True + + +def test_save_deliverable_marks_write_xlsx_attachment(tmp_path, monkeypatch) -> None: + from runtime.tools.public.save_deliverable_attachment_tool import save_deliverable_attachment_tool + + store = AttachmentAssetStore(root_dir=tmp_path / "att") + monkeypatch.setattr( + "runtime.tools.public.write_xlsx_tool.AttachmentAssetStore", + lambda root_dir=None: store if root_dir is None else AttachmentAssetStore(root_dir=root_dir), + ) + monkeypatch.setattr( + "runtime.tools.public.save_deliverable_attachment_tool.AttachmentAssetStore", + lambda root_dir=None: store if root_dir is None else AttachmentAssetStore(root_dir=root_dir), + ) + + gen = write_xlsx_tool().handler( + {"sheets": [{"headers": ["x"], "rows": [["你好"]]}], "name": "cn.xlsx"} + ) + aid = str(gen["attachment_id"]) + marked = save_deliverable_attachment_tool().handler({"attachment_id": aid}) + assert marked.get("ok") is True + assert marked.get("deliverable") is True + assert marked.get("attachment_id") == aid + assert marked.get("mime") == XLSX_MIME