Prefer batch CLI in playbooks, cap tool persistence, and steer getManagedNe failures.

Default scheduled CLI steps to ne_ids/ume_ne_ids batch, compact oversized tool chat_message/tool_log after turns, and on getManagedNe miss guide agents to listManagedNe/UME paths instead of blind retries.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-11 00:46:46 +08:00
parent ae70d4b89a
commit e2fc72607e
16 changed files with 373 additions and 61 deletions

View file

@ -185,15 +185,21 @@
- `AIA_TOOL_LLM_MESSAGE_MAX_CHARS`
- 默认:`0`(不限制)
- 作用:工具结果写回 LLM 的消息长度上限
- 说明:`0` 表示不做限制(不推荐,可能触发部分网关的单条消息上限 400)
- 作用:工具结果写回 **模型上下文** 的消息长度上限(当前轮分析用)
- 说明:`0` 表示不做限制;持久化截断见 `AIA_TOOL_PERSIST_MAX_CHARS`
- 生效:`oclaw/runtime/chat/tool_runtime.py`
- 观测:管理端聊天流 `tool_use_result` 事件会携带 `llm_wire.{truncated_for_llm,max_chars,result_bytes,result_for_llm_bytes,truncate_ms}`
- `AIA_TOOL_PERSIST_MAX_CHARS`
- 默认:`24000`
- 作用:一轮结束后压缩 `chat_message`(role=tool)入库内容,减缓跨天会话撑爆 SQLite
- 说明:`0` 关闭回合后压缩;未设置时也可回退读 `AIA_TOOL_LLM_MESSAGE_MAX_CHARS`
- 生效:`compact_turn_tool_messages_for_storage`(`runtime/chat/tool_runtime.py`)
- `AIA_TOOL_LOG_MAX_CHARS`
- 默认:`200000`
- 默认:`64000`
- 作用:`tool_log` 中 args/result 截断上限
- 生效:`oclaw/platform/persistence/sqlite_store.py`
- 生效:`svc/persistence/sqlite_store.py`
- `AIA_MAX_ATTACHMENT_BYTES`
- 默认:`26214400`(25MB)

View file

@ -2778,7 +2778,7 @@ def build_admin_router() -> APIRouter:
sse_raw = str(store.get_setting("AIA_SSE_QUEUE_MAXSIZE") or "").strip()
sse_queue_maxsize = max(200, min(int(sse_raw), 50_000)) if sse_raw.isdigit() else 2000
tl_raw = str(store.get_setting("AIA_TOOL_LOG_MAX_CHARS") or "").strip()
tool_log_max_chars = max(20_000, min(int(tl_raw), 2_000_000)) if tl_raw.isdigit() else 200_000
tool_log_max_chars = max(20_000, min(int(tl_raw), 2_000_000)) if tl_raw.isdigit() else 64_000
emcp_raw = str(store.get_setting("AIA_ENABLE_MCP_TOOLS") or "").strip().lower()
enable_mcp_tools = emcp_raw not in ("0", "false", "no", "off")
epl_raw = str(store.get_setting("AIA_ENABLE_PLUGIN_TOOLS") or "").strip().lower()
@ -2882,7 +2882,7 @@ def build_admin_router() -> APIRouter:
tmc_n = 80
md = ""
sse_q = payload.get("sse_queue_maxsize", 2000)
tl_cap = payload.get("tool_log_max_chars", 200000)
tl_cap = payload.get("tool_log_max_chars", 64000)
try:
sse_q_n = max(200, min(int(sse_q), 50_000))
except Exception:
@ -2890,7 +2890,7 @@ def build_admin_router() -> APIRouter:
try:
tl_cap_n = max(20_000, min(int(tl_cap), 2_000_000))
except Exception:
tl_cap_n = 200_000
tl_cap_n = 64_000
emcp = bool(payload.get("enable_mcp_tools", True))
epl = bool(payload.get("enable_plugin_tools", False))
erc = bool(payload.get("enable_run_command", True))

View file

@ -440,6 +440,11 @@ def normalize_tool_result(result: Any) -> dict[str, Any]:
def tool_llm_message_max_chars() -> int:
"""Cap for tool payloads fed back into the *model* mid-context.
Default 0 = unlimited for the active turn (large UME dumps stay usable
until the turn ends). Persistence uses :func:`tool_persist_max_chars`.
"""
raw = str(os.getenv("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip()
if raw.isdigit():
n = int(raw)
@ -449,6 +454,37 @@ def tool_llm_message_max_chars() -> int:
return 0
def tool_persist_max_chars(store: Any | None = None) -> int:
"""Cap for rewriting ``role=tool`` chat_message rows after a turn finishes.
Multi-day field sessions rarely need multi-MB tool JSON in history; default
24_000 keeps SQLite growth down. Set ``AIA_TOOL_PERSIST_MAX_CHARS=0`` (or
the same DB setting) to disable post-turn compaction.
"""
raw = ""
if store is not None:
try:
raw = str(store.get_setting("AIA_TOOL_PERSIST_MAX_CHARS") or "").strip()
except Exception:
raw = ""
if not raw:
raw = str(os.getenv("AIA_TOOL_PERSIST_MAX_CHARS") or "").strip()
# Fall back to the older LLM-message setting when operators already tuned it.
if not raw and store is not None:
try:
raw = str(store.get_setting("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip()
except Exception:
raw = ""
if not raw:
raw = str(os.getenv("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip()
if raw.isdigit():
n = int(raw)
if n == 0:
return 0
return max(4096, min(n, 500_000))
return 24_000
def tool_history_summary_after_calls() -> int:
raw = str(os.getenv("AIA_TOOL_HISTORY_SUMMARY_AFTER_CALLS") or "").strip()
if raw.isdigit():
@ -1423,6 +1459,9 @@ def compact_turn_tool_messages_for_storage(
tid = str(turn_uuid or "").strip()
if not tid:
return {"scanned": 0, "updated": 0}
persist_cap = tool_persist_max_chars(store)
if persist_cap <= 0:
return {"scanned": 0, "updated": 0, "skipped": 1}
try:
rows = store.get_messages(session_id=session_id, limit=800)
except Exception:
@ -1444,7 +1483,7 @@ def compact_turn_tool_messages_for_storage(
continue
if not isinstance(obj, dict):
continue
compacted = truncate_tool_result_for_llm_messages(obj)
compacted = truncate_tool_result_for_llm_messages(obj, max_chars=persist_cap)
if compacted == obj:
continue
try:
@ -1457,7 +1496,7 @@ def compact_turn_tool_messages_for_storage(
updated += 1
except Exception:
continue
return {"scanned": int(scanned), "updated": int(updated)}
return {"scanned": int(scanned), "updated": int(updated), "persist_cap": int(persist_cap)}
__all__ = [
"ToolExecutionConfig",
@ -1466,6 +1505,7 @@ __all__ = [
"normalize_tool_result",
"partition_tool_use_batches",
"tool_llm_message_max_chars",
"tool_persist_max_chars",
"truncate_tool_result_for_llm_messages",
"compact_turn_tool_messages_for_storage",
]

View file

@ -242,20 +242,23 @@ def synthesize_recipe_from_prompt(prompt_text: str, *, session_id: str = "") ->
]
if need_attachments:
success.append("Generated files are saved via save_deliverable_attachment.")
recipe = normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": constraints,
"success_criteria": success,
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {
"session_id": str(session_id or "").strip(),
"compiled_at": "",
"compiled_from": "prompt_text",
},
}
recipe = ensure_batch_cli_constraint(
normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": constraints,
"success_criteria": success,
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {
"session_id": str(session_id or "").strip(),
"compiled_at": "",
"compiled_from": "prompt_text",
},
}
),
lang="en",
)
# Preserve compiled_from beyond normalize (normalize only keeps session_id/compiled_at).
recipe.setdefault("source", {})["compiled_from"] = "prompt_text"
@ -275,19 +278,22 @@ def synthesize_recipe_from_prompt(prompt_text: str, *, session_id: str = "") ->
]
if text != goal:
steps.insert(1, f"Full prompt/algorithm to follow:\n{text}")
recipe = normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": [],
"success_criteria": [
"Workflow completed with tools as required.",
"Channel update delivered.",
],
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {"session_id": str(session_id or "").strip(), "compiled_at": ""},
}
recipe = ensure_batch_cli_constraint(
normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": [],
"success_criteria": [
"Workflow completed with tools as required.",
"Channel update delivered.",
],
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {"session_id": str(session_id or "").strip(), "compiled_at": ""},
}
),
lang="en",
)
recipe.setdefault("source", {})["compiled_from"] = "prompt_text"
return recipe
@ -398,9 +404,66 @@ def preview_markdown(
return "\n".join(lines).strip()
def compile_playbook_instruction(*, recipe: dict[str, Any], lang: str = "zh") -> str:
norm = normalize_recipe(recipe)
_BATCH_CLI_MARKERS = (
"execmanagedne",
"listclitargets",
"exec-batch",
"ume_ne_ids",
"ne_ids[",
"show ",
"display ",
"optical",
"cli valid",
"cli confirm",
"登设备",
"只读 cli",
)
def _recipe_mentions_cli(recipe: dict[str, Any]) -> bool:
"""True when playbook steps/goal look like device CLI work."""
parts: list[str] = []
for key in ("goal",):
parts.append(str(recipe.get(key) or ""))
for key in ("steps", "constraints", "success_criteria"):
for item in list(recipe.get(key) or []):
parts.append(str(item or ""))
blob = " ".join(parts).lower()
return any(m in blob for m in _BATCH_CLI_MARKERS)
def _batch_cli_constraint(*, lang: str) -> str:
is_en = str(lang or "").lower().startswith("en")
if is_en:
return (
"Multi-NE CLI default: one execManagedNe with ne_ids[] or ume_ne_ids[] "
"(or targets[]) + shared commands (server concurrent batch). "
"Do not loop one-NE execManagedNe for the same show commands. Cap to top ~5–20."
)
return (
"多台 CLI 默认:一次 execManagedNe 传 ne_ids[] / ume_ne_ids[](或 targets[])+ 共享 commands"
"(服务端并发 batch)。禁止对同一 show 命令逐台循环。建议最多 top 5–20 台。"
)
def ensure_batch_cli_constraint(recipe: dict[str, Any] | None, *, lang: str = "en") -> dict[str, Any]:
"""Inject batch-CLI constraint when recipe mentions device CLI."""
norm = normalize_recipe(recipe or {})
if not _recipe_mentions_cli(norm):
return norm
constraint = _batch_cli_constraint(lang=lang)
existing = list(norm.get("constraints") or [])
low = " ".join(str(c).lower() for c in existing)
if "ne_ids" in low or "ume_ne_ids" in low or "exec-batch" in low or "并发" in low:
return norm
existing.append(constraint)
norm["constraints"] = existing
return norm
def compile_playbook_instruction(*, recipe: dict[str, Any], lang: str = "zh") -> str:
is_en = str(lang or "").lower().startswith("en")
norm = ensure_batch_cli_constraint(recipe, lang="en" if is_en else "zh")
steps = list(norm.get("steps") or [])
constraints = list(norm.get("constraints") or [])
criteria = list(norm.get("success_criteria") or [])
@ -534,12 +597,13 @@ OPS_RECIPE_TEMPLATES: dict[str, dict[str, Any]] = {
"version": 1,
"goal": "Weekly NE license/capacity check summary for ops WhatsApp",
"steps": [
"Resolve target NEs via listManagedNe or known constants (avoid repeated listCliTargets)",
"Run execManagedNe license/capacity show commands with read_timeout_sec>=60",
"Resolve target NEs once via listManagedNe or known constants (listCliTargets at most once)",
"One execManagedNe(ne_ids=[…] or ume_ne_ids=[…], commands=[license/capacity show…], read_timeout_sec>=60) — concurrent batch, never one-NE loops",
"Summarize near-limit or failed NEs in English; attach xlsx only if many rows",
],
"constraints": [
"Prefer English for WhatsApp field ops",
"Multi-NE CLI: ne_ids/ume_ne_ids batch only; no per-NE execManagedNe loops",
"On timeout/unreachable, classify failure and do not blind-retry identical args",
"Keep the group update short and actionable",
],
@ -554,11 +618,13 @@ OPS_RECIPE_TEMPLATES: dict[str, dict[str, Any]] = {
"steps": [
"Call aggregateUmeAlarms or queryUmeAlarmsRaw with bandwidth/congestion/utilization keywords",
"Optionally ume_alarm_xlsx_report(mode=list) if the user wants a file (deliverable=true)",
"If CLI validation is needed: take top 3–5 host ume_ne_ids and one execManagedNe(ume_ne_ids=[…], commands=[…]) batch — never loop one-NE calls",
"Summarize top congested hosts/ports in concise English — avoid sqlQueryUme unless scoped",
],
"constraints": [
"Prefer English for WhatsApp field ops",
"Do not spam CLI or identical alarm re-queries",
"CLI confirmations use ume_ne_ids/ne_ids batch (cap top 5)",
"If insufficient_scope on SQL, switch to aggregate/report tools immediately",
],
"success_criteria": [
@ -607,6 +673,7 @@ __all__ = [
"COMPLEX_PROMPT_HINTS",
"OPS_RECIPE_TEMPLATES",
"compile_playbook_instruction",
"ensure_batch_cli_constraint",
"list_ops_recipe_templates",
"load_recipe_from_job",
"looks_like_complex_schedule_prompt",

View file

@ -185,12 +185,14 @@ def scheduled_turn_system_suffix(*, lang: str, playbook: bool = False) -> str:
"(including save_deliverable_attachment for generated files). "
"Lead the final reply with a short English summary (3–8 lines: what ran, key counts, "
"ok/failed highlights), then optional detail. "
"For multi-NE CLI, prefer one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch. "
"Do not pretend the user just messaged you."
)
return (
"\n\n【定时工作流模式】你正在执行周期性工作流。"
"按 playbook 步骤完成任务,按需调用工具;若生成文件须 save_deliverable_attachment。"
"最终回复先给 3–8 行摘要(做了什么、关键计数、成败),再写细节。"
"多台 CLI 优先一次 execManagedNe(ne_ids|ume_ne_ids=..., commands=...) 批量并发。"
"不要假装用户刚刚发了消息,不要只回一句空提醒。"
)
if is_en:

View file

@ -184,6 +184,10 @@ class _McpBoundTool:
from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error
res = enrich_exec_managed_ne_error(res)
if tool_name == "getManagedNe" and res.get("ok") is False:
from runtime.tools.tool_error_hints import enrich_get_managed_ne_error
res = enrich_get_managed_ne_error(res)
return res
return ToolSpec(

View file

@ -169,6 +169,84 @@ def _unwrap_nested_error_blob(raw: Any) -> tuple[str, str]:
return str(text or "").strip(), str(code or "").strip()
def enrich_get_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]:
"""Steer agents away from empty getManagedNe loops (common WA failure mode)."""
if not isinstance(result, dict) or result.get("ok") is not False:
return result
out = dict(result)
raw_err = out.get("error")
raw_code = str(out.get("error_code") or "")
unwrapped, nested_code = _unwrap_nested_error_blob(raw_err)
if unwrapped and unwrapped != str(raw_err or "").strip():
out["error_detail"] = unwrapped
code = (nested_code or raw_code or "").strip()
detail = str(out.get("detail") or "")
blob = f"{unwrapped} {code} {raw_err} {detail}".lower()
error_class = "get_managed_ne_failed"
hint = (
"getManagedNe failed. Use listManagedNe(keyword=...) or listCliTargets(source=managed) "
"to resolve a *managed* ne_id. For UME inventory UUIDs use getUmeNe or "
"execManagedNe(ume_ne_id=...) — do not retry the same getManagedNe args."
)
example: dict[str, Any] = {"ne_id": "<managed-ne-id-from-listManagedNe>"}
if code in {"ne_id_required", "managed_ne_id_required"} or "ne_id_required" in blob:
error_class = "ne_id_required"
hint = (
"ne_id is required and must be a managed NE id from listManagedNe / "
"listCliTargets(source=managed). UME UUIDs belong in getUmeNe / execManagedNe(ume_ne_id=...)."
)
elif any(
x in blob
for x in (
"404",
"not_found",
"not found",
"netx_http_404",
"no such",
"unknown ne",
"ne not found",
)
):
error_class = "not_found"
hint = (
"Managed NE not found for this ne_id (often a UME UUID was passed). "
"Next: listManagedNe(keyword=host_name) or listCliTargets(source=managed); "
"if the id is from alarms/UME inventory, call getUmeNe / execManagedNe(ume_ne_id=...) instead. "
"Do not blind-retry getManagedNe with the same id."
)
example = {
"next": [
{"tool": "listManagedNe", "args": {"keyword": "<host_name>"}},
{"tool": "execManagedNe", "args": {"ume_ne_id": "<ume-uuid>", "commands": ["show version"]}},
]
}
elif "timeout" in blob or code in {"tool_timeout_or_failed", "read_timeout", "deadline_exceeded"}:
error_class = "timeout"
hint = (
"getManagedNe timed out. Prefer listManagedNe for discovery; only call getManagedNe "
"when you need connect_detail — do not spam retries."
)
out["error_class"] = error_class
if code and not out.get("error_code"):
out["error_code"] = code
# Prefer our steer when prior hint is empty or too vague.
prior = str(out.get("hint") or "").strip()
if (not prior) or ("listManagedNe" not in prior and "ume_ne_id" not in prior.lower()):
out["hint"] = hint
if not out.get("example"):
out["example"] = example
out["next_tools"] = [
"mcp__netx__listManagedNe",
"mcp__netx__listCliTargets",
"mcp__netx__getUmeNe",
"mcp__netx__execManagedNe",
]
return out
def enrich_exec_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]:
"""Classify execManagedNe failures so agents stop blind-retrying."""
if not isinstance(result, dict) or result.get("ok") is not False:
@ -185,7 +263,7 @@ def enrich_exec_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]:
error_class = "exec_failed"
hint = (
"CLI failed. Check ne_id/ume_ne_id, avoid identical blind retries, "
"and prefer batching show commands in one execManagedNe call."
"and for many NEs prefer one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch."
)
if "timeout" in blob or code in {"tool_timeout_or_failed", "read_timeout", "deadline_exceeded"}:
error_class = "timeout"
@ -264,6 +342,12 @@ def classify_tool_failure(result: dict[str, Any]) -> str:
return "retry_guard"
if code in {"tool_not_registered"}:
return "not_registered"
if code in {"ne_id_required"}:
return "ne_id_required"
if any(x in blob for x in ("not found", "netx_http_404")) or (
"404" in blob and ("managed" in blob or "ne_id" in blob)
):
return "not_found"
return "runtime"
@ -330,6 +414,7 @@ __all__ = [
"build_finalize_system_suffix",
"classify_tool_failure",
"enrich_exec_managed_ne_error",
"enrich_get_managed_ne_error",
"enrich_mcp_scope_error",
"format_unregistered_tool_error",
"stamp_tool_failure_class",

View file

@ -64,6 +64,7 @@ Hard preferences:
- **Field default is English**: WhatsApp channel dispatch defaults to `lang=en`; user-visible replies must contain **zero CJK**. Translate Chinese tool fields before display.
- Group chats default to **per-speaker session isolation** (members do not share dialogue memory within the same group).
- Call `listCliTargets` at most once per session and reuse ids; for many NEs with the same show commands use one `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` (server concurrency) — do not loop one-NE calls; default `read_timeout_sec=60` — on timeout raise it, no blind retries.
- `getManagedNe` needs a *managed* `ne_id` only; on failure (often a UME UUID was passed) switch to `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)` — no blind retries.
- Replies like `YES` / `confirm` / `继续` / `please continue`: continue the previous unfinished task — do **not** re-ask for confirmation or restart the query.
- On `tool_invalid_arguments`, fix args using the returned `example`; on timeout hints, raise `read_timeout_sec` or shrink commands.

View file

@ -55,6 +55,7 @@
- **现场默认英文**:WhatsApp 渠道默认 `lang=en`;英文会话回复不得含汉字;工具中文字段先翻译再展示。
- 群聊默认按**发言人隔离会话**(同群不同人互不串上下文);勿假设「群共享一个对话记忆」。
- `listCliTargets` 每会话最多查一次并复用 id;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` 一批并发,勿逐台循环;超时调 `read_timeout_sec`(默认 60),禁止盲重试。
- `getManagedNe` 仅用纳管 `ne_id`;失败(常见:把 UME UUID 当 ne_id)→ `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)`,勿盲重试。
- 用户回复 `YES` / `confirm` / `确认` / `可以` / `继续` / `please continue`:直接承接上一未完成任务继续执行,**不要**再问一遍确认或重开查询。
- 工具返回 `tool_invalid_arguments` 时按返回的 `example` 修正参数;返回超时 hint 时提高 `read_timeout_sec` 或减命令,禁止相同参数重试。

View file

@ -16,8 +16,9 @@ description: 面向 ops 专家的 netx 纳管网元(网元管理)作业手
优先 **MCP**(`mcp__netx__*`)。legacy:`netx_list_managed_ne` 等(`OCLAW_NETX_BUILTIN_TOOLS=1`)。
1. **定位设备**
- `mcp__netx__listManagedNe`:`keyword`、`connect_status=pass`
- `mcp__netx__getManagedNe`:单条详情、`connect_detail`
- `mcp__netx__listManagedNe`:`keyword`、`connect_status=pass`(**首选定位**)
- `mcp__netx__getManagedNe`:仅当需要单条 `connect_detail` 时调用;**入参必须是纳管 ne_id**(来自 listManagedNe / listCliTargets `source=managed`)
- **禁止**把告警/UME 的 UUID 当 `getManagedNe` 的 `ne_id`;失败时跟返回 `hint`:改 `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)`,禁止相同参数盲重试
- **UME 清单(无需逐台纳管)**:`mcp__netx__listCliTargets`(`source=ume`)或 `queryUmeNeInventory` 取 `ne_id`,再用 `ume_ne_id` 执行 CLI(需先在 netx **UME → CLI 连接** 配置统一凭据/跳板)
2. **登录查信息**
- `mcp__netx__execManagedNe`:`ne_id` **或** `ume_ne_id` + `commands`(默认最多 5 条,可由 `NETX_NE_EXEC_MAX_COMMANDS` 调高,硬上限 50)

View file

@ -38,7 +38,7 @@ description: 面向 ops 专家的 netx UME 运维作业手册。覆盖告警查
4. 自定义聚合:`aggregateUmeAlarmsRaw`(`group_by=alarm_host_name` 等)。
5. SQL:`sqlQueryUme`(仅 SELECT;设 `statement_timeout_ms`)。
6. **告警关联拓扑**:两台相关网元取 `ne_id` → `findTopologyPaths`(最短路径优先)。
7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`(`listCliTargets` / `execManagedNe`)。
7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=…)` 一批,勿逐台循环。
## 快速决策树
@ -64,7 +64,7 @@ Prefer these fixed paths for short group/DM asks (EN first; ZH aliases still wor
| how many alarms / tally | ① `runUmeDiagnostics` or `aggregateUmeAlarms`; ② report by_severity + freshness |
| export Excel / send spreadsheet | `ume_alarm_xlsx_report` **or** `write_xlsx(..., deliverable=true)`; never split into 3 steps |
| CRC in area PAD / ACH / … | `queryUmeAlarmsRaw(keyword=CRC)` then keep rows whose `alarm_host_name` / `ne_host_name` starts with area prefix (`PAD-`, `ACH-`, …). Optional xlsx via `write_xlsx(deliverable=true)` |
| bandwidth / congestion / usage rate (+ area) | keyword=`bandwidth` (do **not** require event_type unless user asks); filter hostname prefix for area; CLI validate only top 3–5 if user asks to confirm false positives |
| bandwidth / congestion / usage rate (+ area) | keyword=`bandwidth` (do **not** require event_type unless user asks); filter hostname prefix for area; if CLI confirm false positives: top 3–5 `ume_ne_ids` in **one** `execManagedNe(ume_ne_ids=[…], commands=[…])` batch — never one-NE loops |
| BN EMS / dying gasp / unmanaged (+ area) | keyword or native cause match (`BN EMS` / `dying gasp`); filter area prefix; short EN summary + optional xlsx |
| power / temperature / fan alarms (+ area/NE) | keyword=`power` / `temperature` / `fan`; scope to host or area prefix |
| alarm on **one hostname** (e.g. `MDN-PLSP`, `MKS-SWBP-EN1`) | `queryUmeAlarms` / `queryUmeAlarmsRaw` with `host_name` / keyword=hostname. **Never** start a scheduled License/daily playbook |
@ -119,7 +119,7 @@ Use this shell for ops/alarm/NE/CLI/schedule asks (preferred default; keep it sh
- SQL:建议 `statement_timeout_ms=8000`;非 `count(*)` 应带过滤;时间窗相对 **数据新鲜度**,不是盲目 `now()`。
- 若返回 `insufficient_scope:sql:query`:改用 `aggregateUmeAlarms` / `queryUmeAlarmsRaw` / `ume_alarm_xlsx_report`,勿盲重试 SQL。
- `WITH` CTE 可用;**禁止** `WITH RECURSIVE`。
- `getManagedNe` 只要 **纳管 ne_id**(来自 listManagedNe);UME UUID 用 `getUmeNe` / `execManagedNe(ume_ne_id=...)`。
- `getManagedNe` 只要 **纳管 ne_id**(来自 listManagedNe);UME UUID 用 `getUmeNe` / `execManagedNe(ume_ne_id=...)`。失败时跟 `hint` 换工具,禁止相同 id 盲重试。
## 输出约定

View file

@ -88,4 +88,4 @@ limit 50
## 6) 登设备
- 见 `ops-netx-managed-ne-playbook`
- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI)
- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI);多台同 show → `execManagedNe(ume_ne_ids=[…], commands=[…])` 一批

View file

@ -2296,7 +2296,8 @@ class SqliteStore(ScheduledJobStoreMixin):
raw_cap = str(self.get_setting("AIA_TOOL_LOG_MAX_CHARS") or "").strip()
if not raw_cap:
raw_cap = str(os.getenv("AIA_TOOL_LOG_MAX_CHARS") or "").strip()
cap = 200_000
# Default 64k: field ops rarely need multi-100k tool dumps in tool_log.
cap = 64_000
if raw_cap.isdigit():
cap = max(20_000, min(int(raw_cap), 2_000_000))
args_capped = self._cap_json_for_log(args, max_chars=cap, keep_keys=())

View file

@ -10,7 +10,7 @@ from runtime.application.gateway.ops_short_intent import (
should_send_group_mention_nudge,
)
from runtime.tools.base import ToolSpec
from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error
from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error, enrich_get_managed_ne_error
def test_detect_ops_short_intent_english_field() -> None:
@ -105,3 +105,22 @@ def test_enrich_exec_unreachable_nested_json() -> None:
)
assert out["error_class"] == "unreachable"
assert "unreachable" in out["hint"].lower()
def test_enrich_get_managed_ne_not_found() -> None:
out = enrich_get_managed_ne_error(
{"ok": False, "error": "netx_http_404", "error_code": "netx_http_404", "detail": "Not Found"}
)
assert out["error_class"] == "not_found"
assert "listManagedNe" in out["hint"]
assert "ume_ne_id" in out["hint"]
assert "listManagedNe" in " ".join(out.get("next_tools") or [])
def test_enrich_get_managed_ne_id_required() -> None:
out = enrich_get_managed_ne_error(
{"ok": False, "error": "ne_id_required", "error_code": "ne_id_required"}
)
assert out["error_class"] == "ne_id_required"
assert "listManagedNe" in out["hint"]
assert out.get("example")

View file

@ -83,8 +83,38 @@ class RecipeHelpersTests(unittest.TestCase):
cong = resolve_ops_recipe_template("congestion")
assert cong is not None
self.assertEqual((cong.get("source") or {}).get("template_id"), "bandwidth_congestion_daily")
cong_blob = " ".join(str(s) for s in (cong.get("steps") or []) + (cong.get("constraints") or []))
self.assertIn("ume_ne_ids", cong_blob)
self.assertIn("batch", cong_blob.lower())
license_tmpl = resolve_ops_recipe_template("license_check")
assert license_tmpl is not None
lic_blob = " ".join(str(s) for s in (license_tmpl.get("steps") or []))
self.assertIn("ne_ids", lic_blob)
self.assertIn("never one-NE", lic_blob)
self.assertIsNone(resolve_ops_recipe_template("nope"))
def test_compile_injects_batch_cli_constraint(self) -> None:
recipe = {
"goal": "CLI check top hosts",
"steps": [
"Pull congestion alarms",
"Run execManagedNe show interface on top hosts",
],
"success_criteria": ["Group gets summary"],
}
instr = compile_playbook_instruction(recipe=recipe, lang="en")
self.assertIn("ne_ids", instr)
self.assertIn("ume_ne_ids", instr)
self.assertIn("batch", instr.lower())
# Alarm-only playbook should not get CLI batch constraint.
alarm_only = {
"goal": "Alarm tally",
"steps": ["aggregateUmeAlarms", "Summarize by_severity"],
"success_criteria": ["Done"],
}
alarm_instr = compile_playbook_instruction(recipe=alarm_only, lang="en")
self.assertNotIn("Multi-NE CLI default", alarm_instr)
def test_turn_instruction_modes(self) -> None:
reminder = build_scheduled_turn_instruction(prompt_text="喝水", mode="scheduled", lang="zh")
self.assertIn("提醒意图", reminder)

View file

@ -1,11 +1,16 @@
from __future__ import annotations
import json
import os
import tempfile
import unittest
import uuid
from unittest import mock
from runtime.chat.tool_runtime import (
compact_turn_tool_messages_for_storage,
tool_llm_message_max_chars,
tool_persist_max_chars,
truncate_tool_result_for_llm_messages,
)
from svc.persistence.sqlite_store import SqliteStore
@ -27,16 +32,25 @@ class ToolLlmTruncationTests(unittest.TestCase):
self.assertGreater(out["files_total"], len(out.get("files") or []))
def test_tool_llm_max_chars_env(self) -> None:
import os
from unittest import mock
with mock.patch.dict(os.environ, {"AIA_TOOL_LLM_MESSAGE_MAX_CHARS": "9000"}, clear=False):
self.assertEqual(tool_llm_message_max_chars(), 9000)
def test_compact_turn_tool_messages_for_storage(self) -> None:
import tempfile
import uuid
def test_tool_persist_max_chars_default(self) -> None:
env = {
k: v
for k, v in os.environ.items()
if k not in {"AIA_TOOL_PERSIST_MAX_CHARS", "AIA_TOOL_LLM_MESSAGE_MAX_CHARS"}
}
with mock.patch.dict(os.environ, env, clear=True):
self.assertEqual(tool_persist_max_chars(), 24_000)
self.assertEqual(tool_llm_message_max_chars(), 0)
def test_tool_persist_max_chars_disable(self) -> None:
with mock.patch.dict(os.environ, {"AIA_TOOL_PERSIST_MAX_CHARS": "0"}, clear=False):
self.assertEqual(tool_persist_max_chars(), 0)
def test_compact_turn_defaults_to_persist_cap(self) -> None:
"""Post-turn compact runs with default 24k even when LLM wire cap is unlimited."""
db = f"{tempfile.gettempdir()}/oclaw-test-{uuid.uuid4().hex}.sqlite"
store = SqliteStore(db)
sess = store.create_session("t")
@ -56,22 +70,63 @@ class ToolLlmTruncationTests(unittest.TestCase):
tool_calls={"tool_call_id": "c1", "name": "echo", "assistant_message_id": 1},
turn_uuid=turn_uuid,
)
before = store.get_messages(session_id=sess.id, limit=20)
before_tool = [m for m in before if m.id == row.id][0]
self.assertNotIn("_truncated_for_llm", str(before_tool.content or ""))
from unittest import mock
with mock.patch.dict("os.environ", {"AIA_TOOL_LLM_MESSAGE_MAX_CHARS": "8000"}, clear=False):
env = {
k: v
for k, v in os.environ.items()
if k not in {"AIA_TOOL_PERSIST_MAX_CHARS", "AIA_TOOL_LLM_MESSAGE_MAX_CHARS"}
}
with mock.patch.dict(os.environ, env, clear=True):
stats = compact_turn_tool_messages_for_storage(
store=store,
session_id=sess.id,
turn_uuid=turn_uuid,
)
self.assertEqual(int(stats.get("persist_cap") or 0), 24_000)
self.assertGreaterEqual(int(stats.get("scanned") or 0), 1)
self.assertGreaterEqual(int(stats.get("updated") or 0), 1)
after = store.get_messages(session_id=sess.id, limit=20)
after_tool = [m for m in after if m.id == row.id][0]
self.assertIn("_truncated_for_llm", str(after_tool.content or ""))
self.assertLess(len(str(after_tool.content or "")), 40_000)
def test_compact_turn_can_be_disabled(self) -> None:
db = f"{tempfile.gettempdir()}/oclaw-test-{uuid.uuid4().hex}.sqlite"
store = SqliteStore(db)
sess = store.create_session("t")
turn_uuid = "turn-off"
store.add_message(
session_id=sess.id,
role="tool",
content=json.dumps({"ok": True, "blob": "x" * 50_000}, ensure_ascii=False),
tool_calls={"tool_call_id": "c1", "name": "echo"},
turn_uuid=turn_uuid,
)
with mock.patch.dict(os.environ, {"AIA_TOOL_PERSIST_MAX_CHARS": "0"}, clear=False):
stats = compact_turn_tool_messages_for_storage(
store=store,
session_id=sess.id,
turn_uuid=turn_uuid,
)
self.assertEqual(int(stats.get("skipped") or 0), 1)
self.assertEqual(int(stats.get("updated") or 0), 0)
def test_tool_log_default_cap(self) -> None:
db = f"{tempfile.gettempdir()}/oclaw-test-{uuid.uuid4().hex}.sqlite"
store = SqliteStore(db)
sess = store.create_session("t")
env = {k: v for k, v in os.environ.items() if k != "AIA_TOOL_LOG_MAX_CHARS"}
with mock.patch.dict(os.environ, env, clear=True):
store.add_tool_log(
session_id=sess.id,
tool_name="echo",
args={},
result={"ok": True, "blob": "y" * 200_000},
)
logs = store.get_tool_logs(sess.id, limit=5)
self.assertEqual(len(logs), 1)
blob = json.dumps(logs[0]["result"], ensure_ascii=False)
self.assertLessEqual(len(blob), 70_000)
self.assertTrue(logs[0]["result"].get("ok") is True)
if __name__ == "__main__":