Prefer batch CLI in playbooks, cap tool persistence, and steer getManagedNe failures.

Default scheduled CLI steps to ne_ids/ume_ne_ids batch, compact oversized tool chat_message/tool_log after turns, and on getManagedNe miss guide agents to listManagedNe/UME paths instead of blind retries.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-11 00:46:46 +08:00
parent ae70d4b89a
commit e2fc72607e
16 changed files with 373 additions and 61 deletions

View file

@ -440,6 +440,11 @@ def normalize_tool_result(result: Any) -> dict[str, Any]:
def tool_llm_message_max_chars() -> int:
"""Cap for tool payloads fed back into the *model* mid-context.
Default 0 = unlimited for the active turn (large UME dumps stay usable
until the turn ends). Persistence uses :func:`tool_persist_max_chars`.
"""
raw = str(os.getenv("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip()
if raw.isdigit():
n = int(raw)
@ -449,6 +454,37 @@ def tool_llm_message_max_chars() -> int:
return 0
def tool_persist_max_chars(store: Any | None = None) -> int:
"""Cap for rewriting ``role=tool`` chat_message rows after a turn finishes.
Multi-day field sessions rarely need multi-MB tool JSON in history; default
24_000 keeps SQLite growth down. Set ``AIA_TOOL_PERSIST_MAX_CHARS=0`` (or
the same DB setting) to disable post-turn compaction.
"""
raw = ""
if store is not None:
try:
raw = str(store.get_setting("AIA_TOOL_PERSIST_MAX_CHARS") or "").strip()
except Exception:
raw = ""
if not raw:
raw = str(os.getenv("AIA_TOOL_PERSIST_MAX_CHARS") or "").strip()
# Fall back to the older LLM-message setting when operators already tuned it.
if not raw and store is not None:
try:
raw = str(store.get_setting("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip()
except Exception:
raw = ""
if not raw:
raw = str(os.getenv("AIA_TOOL_LLM_MESSAGE_MAX_CHARS") or "").strip()
if raw.isdigit():
n = int(raw)
if n == 0:
return 0
return max(4096, min(n, 500_000))
return 24_000
def tool_history_summary_after_calls() -> int:
raw = str(os.getenv("AIA_TOOL_HISTORY_SUMMARY_AFTER_CALLS") or "").strip()
if raw.isdigit():
@ -1423,6 +1459,9 @@ def compact_turn_tool_messages_for_storage(
tid = str(turn_uuid or "").strip()
if not tid:
return {"scanned": 0, "updated": 0}
persist_cap = tool_persist_max_chars(store)
if persist_cap <= 0:
return {"scanned": 0, "updated": 0, "skipped": 1}
try:
rows = store.get_messages(session_id=session_id, limit=800)
except Exception:
@ -1444,7 +1483,7 @@ def compact_turn_tool_messages_for_storage(
continue
if not isinstance(obj, dict):
continue
compacted = truncate_tool_result_for_llm_messages(obj)
compacted = truncate_tool_result_for_llm_messages(obj, max_chars=persist_cap)
if compacted == obj:
continue
try:
@ -1457,7 +1496,7 @@ def compact_turn_tool_messages_for_storage(
updated += 1
except Exception:
continue
return {"scanned": int(scanned), "updated": int(updated)}
return {"scanned": int(scanned), "updated": int(updated), "persist_cap": int(persist_cap)}
__all__ = [
"ToolExecutionConfig",
@ -1466,6 +1505,7 @@ __all__ = [
"normalize_tool_result",
"partition_tool_use_batches",
"tool_llm_message_max_chars",
"tool_persist_max_chars",
"truncate_tool_result_for_llm_messages",
"compact_turn_tool_messages_for_storage",
]

View file

@ -242,20 +242,23 @@ def synthesize_recipe_from_prompt(prompt_text: str, *, session_id: str = "") ->
]
if need_attachments:
success.append("Generated files are saved via save_deliverable_attachment.")
recipe = normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": constraints,
"success_criteria": success,
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {
"session_id": str(session_id or "").strip(),
"compiled_at": "",
"compiled_from": "prompt_text",
},
}
recipe = ensure_batch_cli_constraint(
normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": constraints,
"success_criteria": success,
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {
"session_id": str(session_id or "").strip(),
"compiled_at": "",
"compiled_from": "prompt_text",
},
}
),
lang="en",
)
# Preserve compiled_from beyond normalize (normalize only keeps session_id/compiled_at).
recipe.setdefault("source", {})["compiled_from"] = "prompt_text"
@ -275,19 +278,22 @@ def synthesize_recipe_from_prompt(prompt_text: str, *, session_id: str = "") ->
]
if text != goal:
steps.insert(1, f"Full prompt/algorithm to follow:\n{text}")
recipe = normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": [],
"success_criteria": [
"Workflow completed with tools as required.",
"Channel update delivered.",
],
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {"session_id": str(session_id or "").strip(), "compiled_at": ""},
}
recipe = ensure_batch_cli_constraint(
normalize_recipe(
{
"version": 1,
"goal": goal,
"steps": steps,
"constraints": [],
"success_criteria": [
"Workflow completed with tools as required.",
"Channel update delivered.",
],
"output": {"need_attachments": need_attachments, "style": "channel_update"},
"source": {"session_id": str(session_id or "").strip(), "compiled_at": ""},
}
),
lang="en",
)
recipe.setdefault("source", {})["compiled_from"] = "prompt_text"
return recipe
@ -398,9 +404,66 @@ def preview_markdown(
return "\n".join(lines).strip()
def compile_playbook_instruction(*, recipe: dict[str, Any], lang: str = "zh") -> str:
norm = normalize_recipe(recipe)
_BATCH_CLI_MARKERS = (
"execmanagedne",
"listclitargets",
"exec-batch",
"ume_ne_ids",
"ne_ids[",
"show ",
"display ",
"optical",
"cli valid",
"cli confirm",
"登设备",
"只读 cli",
)
def _recipe_mentions_cli(recipe: dict[str, Any]) -> bool:
"""True when playbook steps/goal look like device CLI work."""
parts: list[str] = []
for key in ("goal",):
parts.append(str(recipe.get(key) or ""))
for key in ("steps", "constraints", "success_criteria"):
for item in list(recipe.get(key) or []):
parts.append(str(item or ""))
blob = " ".join(parts).lower()
return any(m in blob for m in _BATCH_CLI_MARKERS)
def _batch_cli_constraint(*, lang: str) -> str:
is_en = str(lang or "").lower().startswith("en")
if is_en:
return (
"Multi-NE CLI default: one execManagedNe with ne_ids[] or ume_ne_ids[] "
"(or targets[]) + shared commands (server concurrent batch). "
"Do not loop one-NE execManagedNe for the same show commands. Cap to top ~5–20."
)
return (
"多台 CLI 默认:一次 execManagedNe 传 ne_ids[] / ume_ne_ids[](或 targets[])+ 共享 commands"
"(服务端并发 batch)。禁止对同一 show 命令逐台循环。建议最多 top 5–20 台。"
)
def ensure_batch_cli_constraint(recipe: dict[str, Any] | None, *, lang: str = "en") -> dict[str, Any]:
"""Inject batch-CLI constraint when recipe mentions device CLI."""
norm = normalize_recipe(recipe or {})
if not _recipe_mentions_cli(norm):
return norm
constraint = _batch_cli_constraint(lang=lang)
existing = list(norm.get("constraints") or [])
low = " ".join(str(c).lower() for c in existing)
if "ne_ids" in low or "ume_ne_ids" in low or "exec-batch" in low or "并发" in low:
return norm
existing.append(constraint)
norm["constraints"] = existing
return norm
def compile_playbook_instruction(*, recipe: dict[str, Any], lang: str = "zh") -> str:
is_en = str(lang or "").lower().startswith("en")
norm = ensure_batch_cli_constraint(recipe, lang="en" if is_en else "zh")
steps = list(norm.get("steps") or [])
constraints = list(norm.get("constraints") or [])
criteria = list(norm.get("success_criteria") or [])
@ -534,12 +597,13 @@ OPS_RECIPE_TEMPLATES: dict[str, dict[str, Any]] = {
"version": 1,
"goal": "Weekly NE license/capacity check summary for ops WhatsApp",
"steps": [
"Resolve target NEs via listManagedNe or known constants (avoid repeated listCliTargets)",
"Run execManagedNe license/capacity show commands with read_timeout_sec>=60",
"Resolve target NEs once via listManagedNe or known constants (listCliTargets at most once)",
"One execManagedNe(ne_ids=[…] or ume_ne_ids=[…], commands=[license/capacity show…], read_timeout_sec>=60) — concurrent batch, never one-NE loops",
"Summarize near-limit or failed NEs in English; attach xlsx only if many rows",
],
"constraints": [
"Prefer English for WhatsApp field ops",
"Multi-NE CLI: ne_ids/ume_ne_ids batch only; no per-NE execManagedNe loops",
"On timeout/unreachable, classify failure and do not blind-retry identical args",
"Keep the group update short and actionable",
],
@ -554,11 +618,13 @@ OPS_RECIPE_TEMPLATES: dict[str, dict[str, Any]] = {
"steps": [
"Call aggregateUmeAlarms or queryUmeAlarmsRaw with bandwidth/congestion/utilization keywords",
"Optionally ume_alarm_xlsx_report(mode=list) if the user wants a file (deliverable=true)",
"If CLI validation is needed: take top 3–5 host ume_ne_ids and one execManagedNe(ume_ne_ids=[…], commands=[…]) batch — never loop one-NE calls",
"Summarize top congested hosts/ports in concise English — avoid sqlQueryUme unless scoped",
],
"constraints": [
"Prefer English for WhatsApp field ops",
"Do not spam CLI or identical alarm re-queries",
"CLI confirmations use ume_ne_ids/ne_ids batch (cap top 5)",
"If insufficient_scope on SQL, switch to aggregate/report tools immediately",
],
"success_criteria": [
@ -607,6 +673,7 @@ __all__ = [
"COMPLEX_PROMPT_HINTS",
"OPS_RECIPE_TEMPLATES",
"compile_playbook_instruction",
"ensure_batch_cli_constraint",
"list_ops_recipe_templates",
"load_recipe_from_job",
"looks_like_complex_schedule_prompt",

View file

@ -185,12 +185,14 @@ def scheduled_turn_system_suffix(*, lang: str, playbook: bool = False) -> str:
"(including save_deliverable_attachment for generated files). "
"Lead the final reply with a short English summary (3–8 lines: what ran, key counts, "
"ok/failed highlights), then optional detail. "
"For multi-NE CLI, prefer one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch. "
"Do not pretend the user just messaged you."
)
return (
"\n\n【定时工作流模式】你正在执行周期性工作流。"
"按 playbook 步骤完成任务,按需调用工具;若生成文件须 save_deliverable_attachment。"
"最终回复先给 3–8 行摘要(做了什么、关键计数、成败),再写细节。"
"多台 CLI 优先一次 execManagedNe(ne_ids|ume_ne_ids=..., commands=...) 批量并发。"
"不要假装用户刚刚发了消息,不要只回一句空提醒。"
)
if is_en:

View file

@ -184,6 +184,10 @@ class _McpBoundTool:
from runtime.tools.tool_error_hints import enrich_exec_managed_ne_error
res = enrich_exec_managed_ne_error(res)
if tool_name == "getManagedNe" and res.get("ok") is False:
from runtime.tools.tool_error_hints import enrich_get_managed_ne_error
res = enrich_get_managed_ne_error(res)
return res
return ToolSpec(

View file

@ -169,6 +169,84 @@ def _unwrap_nested_error_blob(raw: Any) -> tuple[str, str]:
return str(text or "").strip(), str(code or "").strip()
def enrich_get_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]:
"""Steer agents away from empty getManagedNe loops (common WA failure mode)."""
if not isinstance(result, dict) or result.get("ok") is not False:
return result
out = dict(result)
raw_err = out.get("error")
raw_code = str(out.get("error_code") or "")
unwrapped, nested_code = _unwrap_nested_error_blob(raw_err)
if unwrapped and unwrapped != str(raw_err or "").strip():
out["error_detail"] = unwrapped
code = (nested_code or raw_code or "").strip()
detail = str(out.get("detail") or "")
blob = f"{unwrapped} {code} {raw_err} {detail}".lower()
error_class = "get_managed_ne_failed"
hint = (
"getManagedNe failed. Use listManagedNe(keyword=...) or listCliTargets(source=managed) "
"to resolve a *managed* ne_id. For UME inventory UUIDs use getUmeNe or "
"execManagedNe(ume_ne_id=...) — do not retry the same getManagedNe args."
)
example: dict[str, Any] = {"ne_id": "<managed-ne-id-from-listManagedNe>"}
if code in {"ne_id_required", "managed_ne_id_required"} or "ne_id_required" in blob:
error_class = "ne_id_required"
hint = (
"ne_id is required and must be a managed NE id from listManagedNe / "
"listCliTargets(source=managed). UME UUIDs belong in getUmeNe / execManagedNe(ume_ne_id=...)."
)
elif any(
x in blob
for x in (
"404",
"not_found",
"not found",
"netx_http_404",
"no such",
"unknown ne",
"ne not found",
)
):
error_class = "not_found"
hint = (
"Managed NE not found for this ne_id (often a UME UUID was passed). "
"Next: listManagedNe(keyword=host_name) or listCliTargets(source=managed); "
"if the id is from alarms/UME inventory, call getUmeNe / execManagedNe(ume_ne_id=...) instead. "
"Do not blind-retry getManagedNe with the same id."
)
example = {
"next": [
{"tool": "listManagedNe", "args": {"keyword": "<host_name>"}},
{"tool": "execManagedNe", "args": {"ume_ne_id": "<ume-uuid>", "commands": ["show version"]}},
]
}
elif "timeout" in blob or code in {"tool_timeout_or_failed", "read_timeout", "deadline_exceeded"}:
error_class = "timeout"
hint = (
"getManagedNe timed out. Prefer listManagedNe for discovery; only call getManagedNe "
"when you need connect_detail — do not spam retries."
)
out["error_class"] = error_class
if code and not out.get("error_code"):
out["error_code"] = code
# Prefer our steer when prior hint is empty or too vague.
prior = str(out.get("hint") or "").strip()
if (not prior) or ("listManagedNe" not in prior and "ume_ne_id" not in prior.lower()):
out["hint"] = hint
if not out.get("example"):
out["example"] = example
out["next_tools"] = [
"mcp__netx__listManagedNe",
"mcp__netx__listCliTargets",
"mcp__netx__getUmeNe",
"mcp__netx__execManagedNe",
]
return out
def enrich_exec_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]:
"""Classify execManagedNe failures so agents stop blind-retrying."""
if not isinstance(result, dict) or result.get("ok") is not False:
@ -185,7 +263,7 @@ def enrich_exec_managed_ne_error(result: dict[str, Any]) -> dict[str, Any]:
error_class = "exec_failed"
hint = (
"CLI failed. Check ne_id/ume_ne_id, avoid identical blind retries, "
"and prefer batching show commands in one execManagedNe call."
"and for many NEs prefer one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch."
)
if "timeout" in blob or code in {"tool_timeout_or_failed", "read_timeout", "deadline_exceeded"}:
error_class = "timeout"
@ -264,6 +342,12 @@ def classify_tool_failure(result: dict[str, Any]) -> str:
return "retry_guard"
if code in {"tool_not_registered"}:
return "not_registered"
if code in {"ne_id_required"}:
return "ne_id_required"
if any(x in blob for x in ("not found", "netx_http_404")) or (
"404" in blob and ("managed" in blob or "ne_id" in blob)
):
return "not_found"
return "runtime"
@ -330,6 +414,7 @@ __all__ = [
"build_finalize_system_suffix",
"classify_tool_failure",
"enrich_exec_managed_ne_error",
"enrich_get_managed_ne_error",
"enrich_mcp_scope_error",
"format_unregistered_tool_error",
"stamp_tool_failure_class",

View file

@ -64,6 +64,7 @@ Hard preferences:
- **Field default is English**: WhatsApp channel dispatch defaults to `lang=en`; user-visible replies must contain **zero CJK**. Translate Chinese tool fields before display.
- Group chats default to **per-speaker session isolation** (members do not share dialogue memory within the same group).
- Call `listCliTargets` at most once per session and reuse ids; for many NEs with the same show commands use one `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` (server concurrency) — do not loop one-NE calls; default `read_timeout_sec=60` — on timeout raise it, no blind retries.
- `getManagedNe` needs a *managed* `ne_id` only; on failure (often a UME UUID was passed) switch to `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)` — no blind retries.
- Replies like `YES` / `confirm` / `继续` / `please continue`: continue the previous unfinished task — do **not** re-ask for confirmation or restart the query.
- On `tool_invalid_arguments`, fix args using the returned `example`; on timeout hints, raise `read_timeout_sec` or shrink commands.

View file

@ -55,6 +55,7 @@
- **现场默认英文**:WhatsApp 渠道默认 `lang=en`;英文会话回复不得含汉字;工具中文字段先翻译再展示。
- 群聊默认按**发言人隔离会话**(同群不同人互不串上下文);勿假设「群共享一个对话记忆」。
- `listCliTargets` 每会话最多查一次并复用 id;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` 一批并发,勿逐台循环;超时调 `read_timeout_sec`(默认 60),禁止盲重试。
- `getManagedNe` 仅用纳管 `ne_id`;失败(常见:把 UME UUID 当 ne_id)→ `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)`,勿盲重试。
- 用户回复 `YES` / `confirm` / `确认` / `可以` / `继续` / `please continue`:直接承接上一未完成任务继续执行,**不要**再问一遍确认或重开查询。
- 工具返回 `tool_invalid_arguments` 时按返回的 `example` 修正参数;返回超时 hint 时提高 `read_timeout_sec` 或减命令,禁止相同参数重试。