diff --git a/docs/ENVIRONMENT_VARIABLES.md b/docs/ENVIRONMENT_VARIABLES.md index 16030e22..4b42a84d 100644 --- a/docs/ENVIRONMENT_VARIABLES.md +++ b/docs/ENVIRONMENT_VARIABLES.md @@ -185,15 +185,21 @@ - `AIA_TOOL_LLM_MESSAGE_MAX_CHARS` - 默认:`0`(不限制) - - 作用:工具结果写回 LLM 的消息长度上限 - - 说明:`0` 表示不做限制(不推荐,可能触发部分网关的单条消息上限 400) + - 作用:工具结果写回 **模型上下文** 的消息长度上限(当前轮分析用) + - 说明:`0` 表示不做限制;持久化截断见 `AIA_TOOL_PERSIST_MAX_CHARS` - 生效:`oclaw/runtime/chat/tool_runtime.py` - 观测:管理端聊天流 `tool_use_result` 事件会携带 `llm_wire.{truncated_for_llm,max_chars,result_bytes,result_for_llm_bytes,truncate_ms}` +- `AIA_TOOL_PERSIST_MAX_CHARS` + - 默认:`24000` + - 作用:一轮结束后压缩 `chat_message`(role=tool)入库内容,减缓跨天会话撑爆 SQLite + - 说明:`0` 关闭回合后压缩;未设置时也可回退读 `AIA_TOOL_LLM_MESSAGE_MAX_CHARS` + - 生效:`compact_turn_tool_messages_for_storage`(`runtime/chat/tool_runtime.py`) + - `AIA_TOOL_LOG_MAX_CHARS` - - 默认:`200000` + - 默认:`64000` - 作用:`tool_log` 中 args/result 截断上限 - - 生效:`oclaw/platform/persistence/sqlite_store.py` + - 生效:`svc/persistence/sqlite_store.py` - `AIA_MAX_ATTACHMENT_BYTES` - 默认:`26214400`(25MB) diff --git a/interfaces/admin/routes.py b/interfaces/admin/routes.py index 300819e4..9675261e 100644 --- a/interfaces/admin/routes.py +++ b/interfaces/admin/routes.py @@ -2778,7 +2778,7 @@ def build_admin_router() -> APIRouter: sse_raw = str(store.get_setting("AIA_SSE_QUEUE_MAXSIZE") or "").strip() sse_queue_maxsize = max(200, min(int(sse_raw), 50_000)) if sse_raw.isdigit() else 2000 tl_raw = str(store.get_setting("AIA_TOOL_LOG_MAX_CHARS") or "").strip() - tool_log_max_chars = max(20_000, min(int(tl_raw), 2_000_000)) if tl_raw.isdigit() else 200_000 + tool_log_max_chars = max(20_000, min(int(tl_raw), 2_000_000)) if tl_raw.isdigit() else 64_000 emcp_raw = str(store.get_setting("AIA_ENABLE_MCP_TOOLS") or "").strip().lower() enable_mcp_tools = emcp_raw not in ("0", "false", "no", "off") epl_raw = str(store.get_setting("AIA_ENABLE_PLUGIN_TOOLS") or "").strip().lower() @@ -2882,7 +2882,7 @@ def build_admin_router() -> APIRouter: tmc_n = 80 md = "" sse_q = payload.get("sse_queue_maxsize", 2000) - tl_cap = payload.get("tool_log_max_chars", 200000) + tl_cap = payload.get("tool_log_max_chars", 64000) try: sse_q_n = max(200, min(int(sse_q), 50_000)) except Exception: @@ -2890,7 +2890,7 @@ def build_admin_router() -> APIRouter: try: tl_cap_n = max(20_000, min(int(tl_cap), 2_000_000)) except Exception: - tl_cap_n = 200_000 + tl_cap_n = 64_000 emcp = bool(payload.get("enable_mcp_tools", True)) epl = bool(payload.get("enable_plugin_tools", False)) erc = bool(payload.get("enable_run_command", True)) diff --git a/runtime/chat/tool_runtime.py b/runtime/chat/tool_runtime.py index 7a803b74..9faf47f8 100644 --- a/runtime/chat/tool_runtime.py +++ b/runtime/chat/tool_runtime.py @@ -440,6 +440,11 @@ def normalize_tool_result(result: Any) -> dict[str, Any]: def tool_llm_message_max_chars() -> int: + """Cap for tool payloads fed back into the *model* mid-context. + + Default 0 = unlimited for the active turn (large UME dumps stay usable + until the turn ends). Persistence uses :func:`tool_persist_max_chars`. + """ raw = str(os.getenv("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip() if raw.isdigit(): n = int(raw) @@ -449,6 +454,37 @@ def tool_llm_message_max_chars() -> int: return 0 +def tool_persist_max_chars(store: Any | None = None) -> int: + """Cap for rewriting ``role=tool`` chat_message rows after a turn finishes. + + Multi-day field sessions rarely need multi-MB tool JSON in history; default + 24_000 keeps SQLite growth down. Set ``AIA_TOOL_PERSIST_MAX_CHARS=0`` (or + the same DB setting) to disable post-turn compaction. + """ + raw = "" + if store is not None: + try: + raw = str(store.get_setting("AIA_TOOL_PERSIST_MAX_CHARS") or "").strip() + except Exception: + raw = "" + if not raw: + raw = str(os.getenv("AIA_TOOL_PERSIST_MAX_CHARS") or "").strip() + # Fall back to the older LLM-message setting when operators already tuned it. + if not raw and store is not None: + try: + raw = str(store.get_setting("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip() + except Exception: + raw = "" + if not raw: + raw = str(os.getenv("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip() + if raw.isdigit(): + n = int(raw) + if n == 0: + return 0 + return max(4096, min(n, 500_000)) + return 24_000 + + def tool_history_summary_after_calls() -> int: raw = str(os.getenv("AIA_TOOL_HISTORY_SUMMARY_AFTER_CALLS") or "").strip() if raw.isdigit(): @@ -1423,6 +1459,9 @@ def compact_turn_tool_messages_for_storage( tid = str(turn_uuid or "").strip() if not tid: return {"scanned": 0, "updated": 0} + persist_cap = tool_persist_max_chars(store) + if persist_cap <= 0: + return {"scanned": 0, "updated": 0, "skipped": 1} try: rows = store.get_messages(session_id=session_id, limit=800) except Exception: @@ -1444,7 +1483,7 @@ def compact_turn_tool_messages_for_storage( continue if not isinstance(obj, dict): continue - compacted = truncate_tool_result_for_llm_messages(obj) + compacted = truncate_tool_result_for_llm_messages(obj, max_chars=persist_cap) if compacted == obj: continue try: @@ -1457,7 +1496,7 @@ def compact_turn_tool_messages_for_storage( updated += 1 except Exception: continue - return {"scanned": int(scanned), "updated": int(updated)} + return {"scanned": int(scanned), "updated": int(updated), "persist_cap": int(persist_cap)} __all__ = [ "ToolExecutionConfig", @@ -1466,6 +1505,7 @@ __all__ = [ "normalize_tool_result", "partition_tool_use_batches", "tool_llm_message_max_chars", + "tool_persist_max_chars", "truncate_tool_result_for_llm_messages", "compact_turn_tool_messages_for_storage", ] diff --git a/runtime/scheduler/recipe.py b/runtime/scheduler/recipe.py index baf6c744..6203f599 100644 --- a/runtime/scheduler/recipe.py +++ b/runtime/scheduler/recipe.py @@ -242,20 +242,23 @@ def synthesize_recipe_from_prompt(prompt_text: str, *, session_id: str = "") -> ] if need_attachments: success.append("Generated files are saved via save_deliverable_attachment.") - recipe = normalize_recipe( - { - "version": 1, - "goal": goal, - "steps": steps, - "constraints": constraints, - "success_criteria": success, - "output": {"need_attachments": need_attachments, "style": "channel_update"}, - "source": { - "session_id": str(session_id or "").strip(), - "compiled_at": "", - "compiled_from": "prompt_text", - }, - } + recipe = ensure_batch_cli_constraint( + normalize_recipe( + { + "version": 1, + "goal": goal, + "steps": steps, + "constraints": constraints, + "success_criteria": success, + "output": {"need_attachments": need_attachments, "style": "channel_update"}, + "source": { + "session_id": str(session_id or "").strip(), + "compiled_at": "", + "compiled_from": "prompt_text", + }, + } + ), + lang="en", ) # Preserve compiled_from beyond normalize (normalize only keeps session_id/compiled_at). recipe.setdefault("source", {})["compiled_from"] = "prompt_text" @@ -275,19 +278,22 @@ def synthesize_recipe_from_prompt(prompt_text: str, *, session_id: str = "") -> ] if text != goal: steps.insert(1, f"Full prompt/algorithm to follow:\n{text}") - recipe = normalize_recipe( - { - "version": 1, - "goal": goal, - "steps": steps, - "constraints": [], - "success_criteria": [ - "Workflow completed with tools as required.", - "Channel update delivered.", - ], - "output": {"need_attachments": need_attachments, "style": "channel_update"}, - "source": {"session_id": str(session_id or "").strip(), "compiled_at": ""}, - } + recipe = ensure_batch_cli_constraint( + normalize_recipe( + { + "version": 1, + "goal": goal, + "steps": steps, + "constraints": [], + "success_criteria": [ + "Workflow completed with tools as required.", + "Channel update delivered.", + ], + "output": {"need_attachments": need_attachments, "style": "channel_update"}, + "source": {"session_id": str(session_id or "").strip(), "compiled_at": ""}, + } + ), + lang="en", ) recipe.setdefault("source", {})["compiled_from"] = "prompt_text" return recipe @@ -398,9 +404,66 @@ def preview_markdown( return "\n".join(lines).strip() -def compile_playbook_instruction(*, recipe: dict[str, Any], lang: str = "zh") -> str: - norm = normalize_recipe(recipe) +_BATCH_CLI_MARKERS = ( + "execmanagedne", + "listclitargets", + "exec-batch", + "ume_ne_ids", + "ne_ids[", + "show ", + "display ", + "optical", + "cli valid", + "cli confirm", + "登设备", + "只读 cli", +) + + +def _recipe_mentions_cli(recipe: dict[str, Any]) -> bool: + """True when playbook steps/goal look like device CLI work.""" + parts: list[str] = [] + for key in ("goal",): + parts.append(str(recipe.get(key) or "")) + for key in ("steps", "constraints", "success_criteria"): + for item in list(recipe.get(key) or []): + parts.append(str(item or "")) + blob = " ".join(parts).lower() + return any(m in blob for m in _BATCH_CLI_MARKERS) + + +def _batch_cli_constraint(*, lang: str) -> str: is_en = str(lang or "").lower().startswith("en") + if is_en: + return ( + "Multi-NE CLI default: one execManagedNe with ne_ids[] or ume_ne_ids[] " + "(or targets[]) + shared commands (server concurrent batch). " + "Do not loop one-NE execManagedNe for the same show commands. Cap to top ~5–20." + ) + return ( + "多台 CLI 默认:一次 execManagedNe 传 ne_ids[] / ume_ne_ids[](或 targets[])+ 共享 commands" + "(服务端并发 batch)。禁止对同一 show 命令逐台循环。建议最多 top 5–20 台。" + ) + + +def ensure_batch_cli_constraint(recipe: dict[str, Any] | None, *, lang: str = "en") -> dict[str, Any]: + """Inject batch-CLI constraint when recipe mentions device CLI.""" + norm = normalize_recipe(recipe or {}) + if not _recipe_mentions_cli(norm): + return norm + constraint = _batch_cli_constraint(lang=lang) + existing = list(norm.get("constraints") or []) + low = " ".join(str(c).lower() for c in existing) + if "ne_ids" in low or "ume_ne_ids" in low or "exec-batch" in low or "并发" in low: + return norm + existing.append(constraint) + norm["constraints"] = existing + return norm + + +def compile_playbook_instruction(*, recipe: dict[str, Any], lang: str = "zh") -> str: + is_en = str(lang or "").lower().startswith("en") + norm = ensure_batch_cli_constraint(recipe, lang="en" if is_en else "zh") steps = list(norm.get("steps") or []) constraints = list(norm.get("constraints") or []) criteria = list(norm.get("success_criteria") or []) @@ -534,12 +597,13 @@ OPS_RECIPE_TEMPLATES: dict[str, dict[str, Any]] = { "version": 1, "goal": "Weekly NE license/capacity check summary for ops WhatsApp", "steps": [ - "Resolve target NEs via listManagedNe or known constants (avoid repeated listCliTargets)", - "Run execManagedNe license/capacity show commands with read_timeout_sec>=60", + "Resolve target NEs once via listManagedNe or known constants (listCliTargets at most once)", + "One execManagedNe(ne_ids=[…] or ume_ne_ids=[…], commands=[license/capacity show…], read_timeout_sec>=60) — concurrent batch, never one-NE loops", "Summarize near-limit or failed NEs in English; attach xlsx only if many rows", ], "constraints": [ "Prefer English for WhatsApp field ops", + "Multi-NE CLI: ne_ids/ume_ne_ids batch only; no per-NE execManagedNe loops", "On timeout/unreachable, classify failure and do not blind-retry identical args", "Keep the group update short and actionable", ], @@ -554,11 +618,13 @@ OPS_RECIPE_TEMPLATES: dict[str, dict[str, Any]] = { "steps": [ "Call aggregateUmeAlarms or queryUmeAlarmsRaw with bandwidth/congestion/utilization keywords", "Optionally ume_alarm_xlsx_report(mode=list) if the user wants a file (deliverable=true)", + "If CLI validation is needed: take top 3–5 host ume_ne_ids and one execManagedNe(ume_ne_ids=[…], commands=[…]) batch — never loop one-NE calls", "Summarize top congested hosts/ports in concise English — avoid sqlQueryUme unless scoped", ], "constraints": [ "Prefer English for WhatsApp field ops", "Do not spam CLI or identical alarm re-queries", + "CLI confirmations use ume_ne_ids/ne_ids batch (cap top 5)", "If insufficient_scope on SQL, switch to aggregate/report tools immediately", ], "success_criteria": [ @@ -607,6 +673,7 @@ __all__ = [ "COMPLEX_PROMPT_HINTS", "OPS_RECIPE_TEMPLATES", "compile_playbook_instruction", + "ensure_batch_cli_constraint", "list_ops_recipe_templates", "load_recipe_from_job", "looks_like_complex_schedule_prompt", diff --git a/runtime/scheduler/turn_text.py b/runtime/scheduler/turn_text.py index 83e74580..78332ded 100644 --- a/runtime/scheduler/turn_text.py +++ b/runtime/scheduler/turn_text.py @@ -185,12 +185,14 @@ def scheduled_turn_system_suffix(*, lang: str, playbook: bool = False) -> str: "(including save_deliverable_attachment for generated files). " "Lead the final reply with a short English summary (3–8 lines: what ran, key counts, " "ok/failed highlights), then optional detail. " + "For multi-NE CLI, prefer one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch. " "Do not pretend the user just messaged you." ) return ( "\n\n【定时工作流模式】你正在执行周期性工作流。" "按 playbook 步骤完成任务,按需调用工具;若生成文件须 save_deliverable_attachment。" "最终回复先给 3–8 行摘要(做了什么、关键计数、成败),再写细节。" + "多台 CLI 优先一次 execManagedNe(ne_ids|ume_ne_ids=..., commands=...) 批量并发。" "不要假装用户刚刚发了消息,不要只回一句空提醒。" ) if is_en: diff --git a/runtime/tools/mcp/adapter.py b/runtime/tools/mcp/adapter.py index c1d1a961..6c3bf226 100644 --- a/runtime/tools/mcp/adapter.py +++ b/runtime/tools/mcp/adapter.py @@ -184,6 +184,10 @@ class _McpBoundTool: from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error res = enrich_exec_managed_ne_error(res) + if tool_name == "getManagedNe" and res.get("ok") is False: + from runtime.tools.tool_error_hints import enrich_get_managed_ne_error + + res = enrich_get_managed_ne_error(res) return res return ToolSpec( diff --git a/runtime/tools/tool_error_hints.py b/runtime/tools/tool_error_hints.py index fee6de56..4ed4bbb3 100644 --- a/runtime/tools/tool_error_hints.py +++ b/runtime/tools/tool_error_hints.py @@ -169,6 +169,84 @@ def _unwrap_nested_error_blob(raw: Any) -> tuple[str, str]: return str(text or "").strip(), str(code or "").strip() +def enrich_get_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]: + """Steer agents away from empty getManagedNe loops (common WA failure mode).""" + if not isinstance(result, dict) or result.get("ok") is not False: + return result + out = dict(result) + raw_err = out.get("error") + raw_code = str(out.get("error_code") or "") + unwrapped, nested_code = _unwrap_nested_error_blob(raw_err) + if unwrapped and unwrapped != str(raw_err or "").strip(): + out["error_detail"] = unwrapped + code = (nested_code or raw_code or "").strip() + detail = str(out.get("detail") or "") + blob = f"{unwrapped} {code} {raw_err} {detail}".lower() + + error_class = "get_managed_ne_failed" + hint = ( + "getManagedNe failed. Use listManagedNe(keyword=...) or listCliTargets(source=managed) " + "to resolve a *managed* ne_id. For UME inventory UUIDs use getUmeNe or " + "execManagedNe(ume_ne_id=...) — do not retry the same getManagedNe args." + ) + example: dict[str, Any] = {"ne_id": ""} + + if code in {"ne_id_required", "managed_ne_id_required"} or "ne_id_required" in blob: + error_class = "ne_id_required" + hint = ( + "ne_id is required and must be a managed NE id from listManagedNe / " + "listCliTargets(source=managed). UME UUIDs belong in getUmeNe / execManagedNe(ume_ne_id=...)." + ) + elif any( + x in blob + for x in ( + "404", + "not_found", + "not found", + "netx_http_404", + "no such", + "unknown ne", + "ne not found", + ) + ): + error_class = "not_found" + hint = ( + "Managed NE not found for this ne_id (often a UME UUID was passed). " + "Next: listManagedNe(keyword=host_name) or listCliTargets(source=managed); " + "if the id is from alarms/UME inventory, call getUmeNe / execManagedNe(ume_ne_id=...) instead. " + "Do not blind-retry getManagedNe with the same id." + ) + example = { + "next": [ + {"tool": "listManagedNe", "args": {"keyword": ""}}, + {"tool": "execManagedNe", "args": {"ume_ne_id": "", "commands": ["show version"]}}, + ] + } + elif "timeout" in blob or code in {"tool_timeout_or_failed", "read_timeout", "deadline_exceeded"}: + error_class = "timeout" + hint = ( + "getManagedNe timed out. Prefer listManagedNe for discovery; only call getManagedNe " + "when you need connect_detail — do not spam retries." + ) + + out["error_class"] = error_class + if code and not out.get("error_code"): + out["error_code"] = code + # Prefer our steer when prior hint is empty or too vague. + prior = str(out.get("hint") or "").strip() + if (not prior) or ("listManagedNe" not in prior and "ume_ne_id" not in prior.lower()): + out["hint"] = hint + if not out.get("example"): + out["example"] = example + out["next_tools"] = [ + "mcp__netx__listManagedNe", + "mcp__netx__listCliTargets", + "mcp__netx__getUmeNe", + "mcp__netx__execManagedNe", + ] + return out + + def enrich_exec_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]: """Classify execManagedNe failures so agents stop blind-retrying.""" if not isinstance(result, dict) or result.get("ok") is not False: @@ -185,7 +263,7 @@ def enrich_exec_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]: error_class = "exec_failed" hint = ( "CLI failed. Check ne_id/ume_ne_id, avoid identical blind retries, " - "and prefer batching show commands in one execManagedNe call." + "and for many NEs prefer one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch." ) if "timeout" in blob or code in {"tool_timeout_or_failed", "read_timeout", "deadline_exceeded"}: error_class = "timeout" @@ -264,6 +342,12 @@ def classify_tool_failure(result: dict[str, Any]) -> str: return "retry_guard" if code in {"tool_not_registered"}: return "not_registered" + if code in {"ne_id_required"}: + return "ne_id_required" + if any(x in blob for x in ("not found", "netx_http_404")) or ( + "404" in blob and ("managed" in blob or "ne_id" in blob) + ): + return "not_found" return "runtime" @@ -330,6 +414,7 @@ __all__ = [ "build_finalize_system_suffix", "classify_tool_failure", "enrich_exec_managed_ne_error", + "enrich_get_managed_ne_error", "enrich_mcp_scope_error", "format_unregistered_tool_error", "stamp_tool_failure_class", diff --git a/runtime/workspaces/ops/ROLE_SYSTEM.en.md b/runtime/workspaces/ops/ROLE_SYSTEM.en.md index bb2b4a32..52ff1e70 100644 --- a/runtime/workspaces/ops/ROLE_SYSTEM.en.md +++ b/runtime/workspaces/ops/ROLE_SYSTEM.en.md @@ -64,6 +64,7 @@ Hard preferences: - **Field default is English**: WhatsApp channel dispatch defaults to `lang=en`; user-visible replies must contain **zero CJK**. Translate Chinese tool fields before display. - Group chats default to **per-speaker session isolation** (members do not share dialogue memory within the same group). - Call `listCliTargets` at most once per session and reuse ids; for many NEs with the same show commands use one `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` (server concurrency) — do not loop one-NE calls; default `read_timeout_sec=60` — on timeout raise it, no blind retries. +- `getManagedNe` needs a *managed* `ne_id` only; on failure (often a UME UUID was passed) switch to `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)` — no blind retries. - Replies like `YES` / `confirm` / `继续` / `please continue`: continue the previous unfinished task — do **not** re-ask for confirmation or restart the query. - On `tool_invalid_arguments`, fix args using the returned `example`; on timeout hints, raise `read_timeout_sec` or shrink commands. diff --git a/runtime/workspaces/ops/ROLE_SYSTEM.md b/runtime/workspaces/ops/ROLE_SYSTEM.md index 0ae093ef..6e487a6a 100644 --- a/runtime/workspaces/ops/ROLE_SYSTEM.md +++ b/runtime/workspaces/ops/ROLE_SYSTEM.md @@ -55,6 +55,7 @@ - **现场默认英文**:WhatsApp 渠道默认 `lang=en`;英文会话回复不得含汉字;工具中文字段先翻译再展示。 - 群聊默认按**发言人隔离会话**(同群不同人互不串上下文);勿假设「群共享一个对话记忆」。 - `listCliTargets` 每会话最多查一次并复用 id;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` 一批并发,勿逐台循环;超时调 `read_timeout_sec`(默认 60),禁止盲重试。 +- `getManagedNe` 仅用纳管 `ne_id`;失败(常见:把 UME UUID 当 ne_id)→ `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)`,勿盲重试。 - 用户回复 `YES` / `confirm` / `确认` / `可以` / `继续` / `please continue`:直接承接上一未完成任务继续执行,**不要**再问一遍确认或重开查询。 - 工具返回 `tool_invalid_arguments` 时按返回的 `example` 修正参数;返回超时 hint 时提高 `read_timeout_sec` 或减命令,禁止相同参数重试。 diff --git a/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md b/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md index 114f6c51..7023abda 100644 --- a/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md +++ b/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md @@ -16,8 +16,9 @@ description: 面向 ops 专家的 netx 纳管网元(网元管理)作业手 优先 **MCP**(`mcp__netx__*`)。legacy:`netx_list_managed_ne` 等(`OCLAW_NETX_BUILTIN_TOOLS=1`)。 1. **定位设备** - - `mcp__netx__listManagedNe`:`keyword`、`connect_status=pass` - - `mcp__netx__getManagedNe`:单条详情、`connect_detail` + - `mcp__netx__listManagedNe`:`keyword`、`connect_status=pass`(**首选定位**) + - `mcp__netx__getManagedNe`:仅当需要单条 `connect_detail` 时调用;**入参必须是纳管 ne_id**(来自 listManagedNe / listCliTargets `source=managed`) + - **禁止**把告警/UME 的 UUID 当 `getManagedNe` 的 `ne_id`;失败时跟返回 `hint`:改 `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)`,禁止相同参数盲重试 - **UME 清单(无需逐台纳管)**:`mcp__netx__listCliTargets`(`source=ume`)或 `queryUmeNeInventory` 取 `ne_id`,再用 `ume_ne_id` 执行 CLI(需先在 netx **UME → CLI 连接** 配置统一凭据/跳板) 2. **登录查信息** - `mcp__netx__execManagedNe`:`ne_id` **或** `ume_ne_id` + `commands`(默认最多 5 条,可由 `NETX_NE_EXEC_MAX_COMMANDS` 调高,硬上限 50) diff --git a/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md b/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md index 893c4601..7694b03f 100644 --- a/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md +++ b/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md @@ -38,7 +38,7 @@ description: 面向 ops 专家的 netx UME 运维作业手册。覆盖告警查 4. 自定义聚合:`aggregateUmeAlarmsRaw`(`group_by=alarm_host_name` 等)。 5. SQL:`sqlQueryUme`(仅 SELECT;设 `statement_timeout_ms`)。 6. **告警关联拓扑**:两台相关网元取 `ne_id` → `findTopologyPaths`(最短路径优先)。 -7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`(`listCliTargets` / `execManagedNe`)。 +7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=…)` 一批,勿逐台循环。 ## 快速决策树 @@ -64,7 +64,7 @@ Prefer these fixed paths for short group/DM asks (EN first; ZH aliases still wor | how many alarms / tally | ① `runUmeDiagnostics` or `aggregateUmeAlarms`; ② report by_severity + freshness | | export Excel / send spreadsheet | `ume_alarm_xlsx_report` **or** `write_xlsx(..., deliverable=true)`; never split into 3 steps | | CRC in area PAD / ACH / … | `queryUmeAlarmsRaw(keyword=CRC)` then keep rows whose `alarm_host_name` / `ne_host_name` starts with area prefix (`PAD-`, `ACH-`, …). Optional xlsx via `write_xlsx(deliverable=true)` | -| bandwidth / congestion / usage rate (+ area) | keyword=`bandwidth` (do **not** require event_type unless user asks); filter hostname prefix for area; CLI validate only top 3–5 if user asks to confirm false positives | +| bandwidth / congestion / usage rate (+ area) | keyword=`bandwidth` (do **not** require event_type unless user asks); filter hostname prefix for area; if CLI confirm false positives: top 3–5 `ume_ne_ids` in **one** `execManagedNe(ume_ne_ids=[…], commands=[…])` batch — never one-NE loops | | BN EMS / dying gasp / unmanaged (+ area) | keyword or native cause match (`BN EMS` / `dying gasp`); filter area prefix; short EN summary + optional xlsx | | power / temperature / fan alarms (+ area/NE) | keyword=`power` / `temperature` / `fan`; scope to host or area prefix | | alarm on **one hostname** (e.g. `MDN-PLSP`, `MKS-SWBP-EN1`) | `queryUmeAlarms` / `queryUmeAlarmsRaw` with `host_name` / keyword=hostname. **Never** start a scheduled License/daily playbook | @@ -119,7 +119,7 @@ Use this shell for ops/alarm/NE/CLI/schedule asks (preferred default; keep it sh - SQL:建议 `statement_timeout_ms=8000`;非 `count(*)` 应带过滤;时间窗相对 **数据新鲜度**,不是盲目 `now()`。 - 若返回 `insufficient_scope:sql:query`:改用 `aggregateUmeAlarms` / `queryUmeAlarmsRaw` / `ume_alarm_xlsx_report`,勿盲重试 SQL。 - `WITH` CTE 可用;**禁止** `WITH RECURSIVE`。 -- `getManagedNe` 只要 **纳管 ne_id**(来自 listManagedNe);UME UUID 用 `getUmeNe` / `execManagedNe(ume_ne_id=...)`。 +- `getManagedNe` 只要 **纳管 ne_id**(来自 listManagedNe);UME UUID 用 `getUmeNe` / `execManagedNe(ume_ne_id=...)`。失败时跟 `hint` 换工具,禁止相同 id 盲重试。 ## 输出约定 diff --git a/skills/_workspace/ops/ops-netx-ume-playbook/reference.md b/skills/_workspace/ops/ops-netx-ume-playbook/reference.md index de7e06dc..be44450b 100644 --- a/skills/_workspace/ops/ops-netx-ume-playbook/reference.md +++ b/skills/_workspace/ops/ops-netx-ume-playbook/reference.md @@ -88,4 +88,4 @@ limit 50 ## 6) 登设备 - 见 `ops-netx-managed-ne-playbook` -- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI) +- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI);多台同 show → `execManagedNe(ume_ne_ids=[…], commands=[…])` 一批 diff --git a/svc/persistence/sqlite_store.py b/svc/persistence/sqlite_store.py index 8a430173..9960f2a3 100644 --- a/svc/persistence/sqlite_store.py +++ b/svc/persistence/sqlite_store.py @@ -2296,7 +2296,8 @@ class SqliteStore(ScheduledJobStoreMixin): raw_cap = str(self.get_setting("AIA_TOOL_LOG_MAX_CHARS") or "").strip() if not raw_cap: raw_cap = str(os.getenv("AIA_TOOL_LOG_MAX_CHARS") or "").strip() - cap = 200_000 + # Default 64k: field ops rarely need multi-100k tool dumps in tool_log. + cap = 64_000 if raw_cap.isdigit(): cap = max(20_000, min(int(raw_cap), 2_000_000)) args_capped = self._cap_json_for_log(args, max_chars=cap, keep_keys=()) diff --git a/tests/test_ops_short_intent_and_exec_hints.py b/tests/test_ops_short_intent_and_exec_hints.py index 856dd48c..5460b5a1 100644 --- a/tests/test_ops_short_intent_and_exec_hints.py +++ b/tests/test_ops_short_intent_and_exec_hints.py @@ -10,7 +10,7 @@ from runtime.application.gateway.ops_short_intent import ( should_send_group_mention_nudge, ) from runtime.tools.base import ToolSpec -from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error +from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error, enrich_get_managed_ne_error def test_detect_ops_short_intent_english_field() -> None: @@ -105,3 +105,22 @@ def test_enrich_exec_unreachable_nested_json() -> None: ) assert out["error_class"] == "unreachable" assert "unreachable" in out["hint"].lower() + + +def test_enrich_get_managed_ne_not_found() -> None: + out = enrich_get_managed_ne_error( + {"ok": False, "error": "netx_http_404", "error_code": "netx_http_404", "detail": "Not Found"} + ) + assert out["error_class"] == "not_found" + assert "listManagedNe" in out["hint"] + assert "ume_ne_id" in out["hint"] + assert "listManagedNe" in " ".join(out.get("next_tools") or []) + + +def test_enrich_get_managed_ne_id_required() -> None: + out = enrich_get_managed_ne_error( + {"ok": False, "error": "ne_id_required", "error_code": "ne_id_required"} + ) + assert out["error_class"] == "ne_id_required" + assert "listManagedNe" in out["hint"] + assert out.get("example") diff --git a/tests/test_schedule_recipe.py b/tests/test_schedule_recipe.py index 62d61a83..52e0f5d3 100644 --- a/tests/test_schedule_recipe.py +++ b/tests/test_schedule_recipe.py @@ -83,8 +83,38 @@ class RecipeHelpersTests(unittest.TestCase): cong = resolve_ops_recipe_template("congestion") assert cong is not None self.assertEqual((cong.get("source") or {}).get("template_id"), "bandwidth_congestion_daily") + cong_blob = " ".join(str(s) for s in (cong.get("steps") or []) + (cong.get("constraints") or [])) + self.assertIn("ume_ne_ids", cong_blob) + self.assertIn("batch", cong_blob.lower()) + license_tmpl = resolve_ops_recipe_template("license_check") + assert license_tmpl is not None + lic_blob = " ".join(str(s) for s in (license_tmpl.get("steps") or [])) + self.assertIn("ne_ids", lic_blob) + self.assertIn("never one-NE", lic_blob) self.assertIsNone(resolve_ops_recipe_template("nope")) + def test_compile_injects_batch_cli_constraint(self) -> None: + recipe = { + "goal": "CLI check top hosts", + "steps": [ + "Pull congestion alarms", + "Run execManagedNe show interface on top hosts", + ], + "success_criteria": ["Group gets summary"], + } + instr = compile_playbook_instruction(recipe=recipe, lang="en") + self.assertIn("ne_ids", instr) + self.assertIn("ume_ne_ids", instr) + self.assertIn("batch", instr.lower()) + # Alarm-only playbook should not get CLI batch constraint. + alarm_only = { + "goal": "Alarm tally", + "steps": ["aggregateUmeAlarms", "Summarize by_severity"], + "success_criteria": ["Done"], + } + alarm_instr = compile_playbook_instruction(recipe=alarm_only, lang="en") + self.assertNotIn("Multi-NE CLI default", alarm_instr) + def test_turn_instruction_modes(self) -> None: reminder = build_scheduled_turn_instruction(prompt_text="喝水", mode="scheduled", lang="zh") self.assertIn("提醒意图", reminder) diff --git a/tests/test_tool_llm_truncation.py b/tests/test_tool_llm_truncation.py index 370770a5..6271fc65 100644 --- a/tests/test_tool_llm_truncation.py +++ b/tests/test_tool_llm_truncation.py @@ -1,11 +1,16 @@ from __future__ import annotations import json +import os +import tempfile import unittest +import uuid +from unittest import mock from runtime.chat.tool_runtime import ( compact_turn_tool_messages_for_storage, tool_llm_message_max_chars, + tool_persist_max_chars, truncate_tool_result_for_llm_messages, ) from svc.persistence.sqlite_store import SqliteStore @@ -27,16 +32,25 @@ class ToolLlmTruncationTests(unittest.TestCase): self.assertGreater(out["files_total"], len(out.get("files") or [])) def test_tool_llm_max_chars_env(self) -> None: - import os - from unittest import mock - with mock.patch.dict(os.environ, {"AIA_TOOL_LLM_MESSAGE_MAX_CHARS": "9000"}, clear=False): self.assertEqual(tool_llm_message_max_chars(), 9000) - def test_compact_turn_tool_messages_for_storage(self) -> None: - import tempfile - import uuid + def test_tool_persist_max_chars_default(self) -> None: + env = { + k: v + for k, v in os.environ.items() + if k not in {"AIA_TOOL_PERSIST_MAX_CHARS", "AIA_TOOL_LLM_MESSAGE_MAX_CHARS"} + } + with mock.patch.dict(os.environ, env, clear=True): + self.assertEqual(tool_persist_max_chars(), 24_000) + self.assertEqual(tool_llm_message_max_chars(), 0) + def test_tool_persist_max_chars_disable(self) -> None: + with mock.patch.dict(os.environ, {"AIA_TOOL_PERSIST_MAX_CHARS": "0"}, clear=False): + self.assertEqual(tool_persist_max_chars(), 0) + + def test_compact_turn_defaults_to_persist_cap(self) -> None: + """Post-turn compact runs with default 24k even when LLM wire cap is unlimited.""" db = f"{tempfile.gettempdir()}/oclaw-test-{uuid.uuid4().hex}.sqlite" store = SqliteStore(db) sess = store.create_session("t") @@ -56,22 +70,63 @@ class ToolLlmTruncationTests(unittest.TestCase): tool_calls={"tool_call_id": "c1", "name": "echo", "assistant_message_id": 1}, turn_uuid=turn_uuid, ) - before = store.get_messages(session_id=sess.id, limit=20) - before_tool = [m for m in before if m.id == row.id][0] - self.assertNotIn("_truncated_for_llm", str(before_tool.content or "")) - from unittest import mock - - with mock.patch.dict("os.environ", {"AIA_TOOL_LLM_MESSAGE_MAX_CHARS": "8000"}, clear=False): + env = { + k: v + for k, v in os.environ.items() + if k not in {"AIA_TOOL_PERSIST_MAX_CHARS", "AIA_TOOL_LLM_MESSAGE_MAX_CHARS"} + } + with mock.patch.dict(os.environ, env, clear=True): stats = compact_turn_tool_messages_for_storage( store=store, session_id=sess.id, turn_uuid=turn_uuid, ) + self.assertEqual(int(stats.get("persist_cap") or 0), 24_000) self.assertGreaterEqual(int(stats.get("scanned") or 0), 1) self.assertGreaterEqual(int(stats.get("updated") or 0), 1) after = store.get_messages(session_id=sess.id, limit=20) after_tool = [m for m in after if m.id == row.id][0] self.assertIn("_truncated_for_llm", str(after_tool.content or "")) + self.assertLess(len(str(after_tool.content or "")), 40_000) + + def test_compact_turn_can_be_disabled(self) -> None: + db = f"{tempfile.gettempdir()}/oclaw-test-{uuid.uuid4().hex}.sqlite" + store = SqliteStore(db) + sess = store.create_session("t") + turn_uuid = "turn-off" + store.add_message( + session_id=sess.id, + role="tool", + content=json.dumps({"ok": True, "blob": "x" * 50_000}, ensure_ascii=False), + tool_calls={"tool_call_id": "c1", "name": "echo"}, + turn_uuid=turn_uuid, + ) + with mock.patch.dict(os.environ, {"AIA_TOOL_PERSIST_MAX_CHARS": "0"}, clear=False): + stats = compact_turn_tool_messages_for_storage( + store=store, + session_id=sess.id, + turn_uuid=turn_uuid, + ) + self.assertEqual(int(stats.get("skipped") or 0), 1) + self.assertEqual(int(stats.get("updated") or 0), 0) + + def test_tool_log_default_cap(self) -> None: + db = f"{tempfile.gettempdir()}/oclaw-test-{uuid.uuid4().hex}.sqlite" + store = SqliteStore(db) + sess = store.create_session("t") + env = {k: v for k, v in os.environ.items() if k != "AIA_TOOL_LOG_MAX_CHARS"} + with mock.patch.dict(os.environ, env, clear=True): + store.add_tool_log( + session_id=sess.id, + tool_name="echo", + args={}, + result={"ok": True, "blob": "y" * 200_000}, + ) + logs = store.get_tool_logs(sess.id, limit=5) + self.assertEqual(len(logs), 1) + blob = json.dumps(logs[0]["result"], ensure_ascii=False) + self.assertLessEqual(len(blob), 70_000) + self.assertTrue(logs[0]["result"].get("ok") is True) if __name__ == "__main__":