From 25d09d78ce601c63a647c234539c188ebea46b41 Mon Sep 17 00:00:00 2001 From: oliver Date: Wed, 12 Aug 2026 21:27:30 +0800 Subject: [PATCH] Prefer hetero targets batch for multi-NE CLI instead of one-NE loops. Co-authored-by: Cursor --- runtime/chat/exec_managed_ne_guard.py | 25 ++++++++++++++----- runtime/scheduler/recipe.py | 11 ++++---- runtime/workspaces/ops/ROLE_SYSTEM.en.md | 2 +- runtime/workspaces/ops/ROLE_SYSTEM.md | 2 +- .../ops/ops-netx-managed-ne-playbook/SKILL.md | 15 ++++++++--- .../ops/ops-netx-ume-playbook/SKILL.md | 4 +-- .../ops/ops-netx-ume-playbook/reference.md | 2 +- tests/test_exec_managed_ne_guard.py | 22 ++++++++++++++++ 8 files changed, 63 insertions(+), 20 deletions(-) diff --git a/runtime/chat/exec_managed_ne_guard.py b/runtime/chat/exec_managed_ne_guard.py index a0573dc6..5b8c4258 100644 --- a/runtime/chat/exec_managed_ne_guard.py +++ b/runtime/chat/exec_managed_ne_guard.py @@ -134,22 +134,27 @@ def budget_block_payload( code = "cli_fail_budget_exceeded" hint = ( f"execManagedNe already failed {fail_used}/{fail_budget} times this turn. " - "Stop one-NE loops; use one execManagedNe(ne_ids|ume_ne_ids=..., commands=...) batch " - "or summarize reachable failures — do not keep probing." + "Stop one-NE loops; use ONE execManagedNe batch: " + "ne_ids|ume_ne_ids + shared commands, or targets=[{ume_ne_id, commands},…] when commands differ — " + "or summarize reachable failures; do not keep probing." if en else f"本轮 execManagedNe 已失败 {fail_used}/{fail_budget} 次。" - "停止单台循环;改用一次 ne_ids/ume_ne_ids 批量,或汇总可达性失败,勿继续盲探。" + "停止单台循环;改用一次 batch:" + "同命令用 ne_ids/ume_ne_ids,每台命令不同用 targets=[{ume_ne_id, commands},…];" + "或汇总可达性失败,勿继续盲探。" ) err = "cli_fail_budget_exceeded" else: code = "cli_call_budget_exceeded" hint = ( f"Single-NE execManagedNe budget exhausted ({single_used}/{single_budget} this turn). " - "For more NEs call ONE execManagedNe with ne_ids[] or ume_ne_ids[] (shared commands). " - "Do not loop one-NE execManagedNe." + "For more NEs call ONE execManagedNe batch: " + "ne_ids[]/ume_ne_ids[] + shared commands, OR targets=[{ume_ne_id|ne_id, commands:[…]}, …] " + "when each NE needs different CLI. Do not loop one-NE execManagedNe." if en else f"单台 execManagedNe 预算已用尽(本轮 {single_used}/{single_budget})。" - "更多网元请一次传入 ne_ids[] / ume_ne_ids[] 批量执行,禁止逐台循环。" + "更多网元请一次 batch:同命令用 ne_ids[]/ume_ne_ids[];" + "每台命令不同用 targets=[{ume_ne_id|ne_id, commands:[…]}, …]。禁止逐台循环。" ) err = "cli_call_budget_exceeded" return { @@ -164,6 +169,14 @@ def budget_block_payload( "read_timeout_sec": 90, "concurrency": 4, }, + "example_hetero_targets": { + "targets": [ + {"ume_ne_id": "", "commands": ["display optical-module brief"]}, + {"ume_ne_id": "", "commands": ["show interface transceiver"]}, + ], + "read_timeout_sec": 90, + "concurrency": 4, + }, "single_used": int(single_used), "single_budget": int(single_budget), "fail_used": int(fail_used), diff --git a/runtime/scheduler/recipe.py b/runtime/scheduler/recipe.py index 6203f599..6724c30b 100644 --- a/runtime/scheduler/recipe.py +++ b/runtime/scheduler/recipe.py @@ -436,13 +436,14 @@ def _batch_cli_constraint(*, lang: str) -> str: is_en = str(lang or "").lower().startswith("en") if is_en: return ( - "Multi-NE CLI default: one execManagedNe with ne_ids[] or ume_ne_ids[] " - "(or targets[]) + shared commands (server concurrent batch). " - "Do not loop one-NE execManagedNe for the same show commands. Cap to top ~5–20." + "Multi-NE CLI batch-first: one execManagedNe — same show → ne_ids[]/ume_ne_ids[] + shared commands; " + "different CLI per NE → targets=[{ume_ne_id|ne_id, commands:[…]}, …] (server concurrent). " + "Do not loop one-NE execManagedNe. Cap to top ~5–20." ) return ( - "多台 CLI 默认:一次 execManagedNe 传 ne_ids[] / ume_ne_ids[](或 targets[])+ 共享 commands" - "(服务端并发 batch)。禁止对同一 show 命令逐台循环。建议最多 top 5–20 台。" + "多台 CLI 必须 batch-first、一次调用:同命令用 ne_ids[]/ume_ne_ids[] + 共享 commands;" + "每台命令不同用 targets=[{ume_ne_id|ne_id, commands:[…]}, …](服务端并发)。" + "禁止逐台循环。建议最多 top 5–20 台。" ) diff --git a/runtime/workspaces/ops/ROLE_SYSTEM.en.md b/runtime/workspaces/ops/ROLE_SYSTEM.en.md index 0ecc3328..a7c1733b 100644 --- a/runtime/workspaces/ops/ROLE_SYSTEM.en.md +++ b/runtime/workspaces/ops/ROLE_SYSTEM.en.md @@ -63,7 +63,7 @@ Hard preferences: - Spreadsheet delivery: `ume_alarm_xlsx_report` or `write_xlsx(deliverable=true)` — never claim a file was sent without deliverable marking. - **Field default is English**: WhatsApp channel dispatch defaults to `lang=en`; user-visible replies must contain **zero CJK**. Translate Chinese tool fields before display. - Group chats default to **per-speaker session isolation** (members do not share dialogue memory within the same group). -- Call `listCliTargets` at most once per session and reuse ids; for many NEs with the same show commands use one `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` (server concurrency) — do not loop one-NE calls; default `read_timeout_sec=60` — on timeout raise it, no blind retries. +- Call `listCliTargets` at most once per session and reuse ids. Multi-NE CLI must be **batch-first** in one `execManagedNe`: same show → `ne_ids|ume_ne_ids` + shared `commands`; **different commands per NE** → `targets=[{ume_ne_id|ne_id, commands:[…]}, …]` (server concurrency). Do not loop one-NE calls. Default `read_timeout_sec=60` — on timeout raise it, no blind retries. - `getManagedNe` needs a *managed* `ne_id` only; on failure (often a UME UUID was passed) switch to `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)` — no blind retries. - Replies like `YES` / `confirm` / `继续` / `please continue`: continue the previous unfinished task — do **not** re-ask for confirmation or restart the query. - On `tool_invalid_arguments`, fix args using the returned `example`; on timeout hints, raise `read_timeout_sec` or shrink commands. diff --git a/runtime/workspaces/ops/ROLE_SYSTEM.md b/runtime/workspaces/ops/ROLE_SYSTEM.md index 6c08ed12..1e8cd3a0 100644 --- a/runtime/workspaces/ops/ROLE_SYSTEM.md +++ b/runtime/workspaces/ops/ROLE_SYSTEM.md @@ -54,7 +54,7 @@ - 用户要表格/Excel:`ume_alarm_xlsx_report` 或 `write_xlsx(deliverable=true)`;禁止只写文件不投递。 - **现场默认英文**:WhatsApp 渠道默认 `lang=en`;英文会话回复不得含汉字;工具中文字段先翻译再展示。 - 群聊默认按**发言人隔离会话**(同群不同人互不串上下文);勿假设「群共享一个对话记忆」。 -- `listCliTargets` 每会话最多查一次并复用 id;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` 一批并发,勿逐台循环;超时调 `read_timeout_sec`(默认 60),禁止盲重试。 +- `listCliTargets` 每会话最多查一次并复用 id。多台 CLI 必须 **batch-first、一次调用**:同命令用 `ne_ids|ume_ne_ids` + 共享 `commands`;**每台命令不同**用 `targets=[{ume_ne_id|ne_id, commands:[…]}, …]`(服务端并发)。禁止逐台循环。超时调 `read_timeout_sec`(默认 60),禁止盲重试。 - `getManagedNe` 仅用纳管 `ne_id`;失败(常见:把 UME UUID 当 ne_id)→ `listManagedNe` / `getUmeNe` / `execManagedNe(ume_ne_id=...)`,勿盲重试。 - 用户回复 `YES` / `confirm` / `确认` / `可以` / `继续` / `please continue`:直接承接上一未完成任务继续执行,**不要**再问一遍确认或重开查询。 - 工具返回 `tool_invalid_arguments` 时按返回的 `example` 修正参数;返回超时 hint 时提高 `read_timeout_sec` 或减命令,禁止相同参数重试。 diff --git a/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md b/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md index bd33a21a..e2b0a918 100644 --- a/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md +++ b/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md @@ -22,8 +22,12 @@ description: 面向 ops 专家的 netx 纳管网元(网元管理)作业手 - **UME 清单(无需逐台纳管)**:`mcp__netx__listCliTargets`(`source=ume`)或 `queryUmeNeInventory` 取 `ne_id`,再用 `ume_ne_id` 执行 CLI(需先在 netx **UME → CLI 连接** 配置统一凭据/跳板) 2. **登录查信息** - `mcp__netx__execManagedNe`:`ne_id` **或** `ume_ne_id` + `commands`(默认最多 5 条,可由 `NETX_NE_EXEC_MAX_COMMANDS` 调高,硬上限 50) - - **多台同命令(推荐)**:一次调用传 `ne_ids` / `ume_ne_ids`(或 `targets`)+ 共享 `commands`,服务端并发执行(默认 concurrency=4,最多 20 台)。禁止对同一 show 命令逐台循环 `execManagedNe` - - **一次会话内**:`listCliTargets` 最多调用一次,缓存返回的 id;多条 show 合并进同一次 `commands`,禁止「list→exec→list→exec」循环 + - **多台必须 batch-first(一次调用,服务端并发登录,默认 concurrency=4,最多 20 台)**: + - **同命令**:`ne_ids` / `ume_ne_ids` + 共享 `commands` + - **每台命令不同(厂商/角色不同)**:用 `targets=[{ume_ne_id|ne_id, commands:[…]}, …]` 一次提交;**不要**因为命令不同就退化成逐台 `execManagedNe` + - 可按厂商拆成 1~2 次 batch(华为一批、Cisco 一批),仍远好于 N 次单台 + - **禁止**对多台排查逐台循环 `execManagedNe`(stdio 串行 + 重复登录) + - **一次会话内**:`listCliTargets` 最多调用一次,缓存返回的 id;同台多条 show 合并进该台的 `commands[]`,禁止「list→exec→list→exec」循环 - 超时:提高 `read_timeout_sec`(默认 60,慢命令 90–120)或减少命令条数,禁止对同一命令盲重试 ## Field link recipes (WhatsApp EN) @@ -49,9 +53,12 @@ Try in order; **one failure → switch command, do not retry the same spelling** Cisco/Huawei: use `show interface transceiver` / `display optical-module` style allowlisted commands as applicable. -### Multi-NE same show +### Multi-NE CLI (batch-first) -Prefer `execManagedNe(ume_ne_ids=[…], commands=[…])` once. Cap to the NEs on the asked path (usually 2). +- **Same show on many NEs**: `execManagedNe(ume_ne_ids=[…], commands=[…])` once. +- **Different commands per NE** (vendor / role): one call with + `targets=[{ume_ne_id, commands:[…]}, {ume_ne_id, commands:[…]}, …]` — still concurrent on the server. +- Cap to NEs on the asked path (usually 2–5). Never one-NE `execManagedNe` loops. ## CLI 约束(服务端强制) diff --git a/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md b/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md index b9f5b7d9..1a906116 100644 --- a/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md +++ b/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md @@ -38,7 +38,7 @@ description: 面向 ops 专家的 netx UME 运维作业手册。覆盖告警查 4. 自定义聚合:`aggregateUmeAlarmsRaw`(`group_by=alarm_host_name` 等)。 5. SQL:`sqlQueryUme`(仅 SELECT;设 `statement_timeout_ms`)。 6. **告警关联拓扑**:两台相关网元取 `ne_id` → `findTopologyPaths`(最短路径优先)。 -7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=…)` 一批,勿逐台循环。 +7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`;多台用一次 `execManagedNe` batch(同命令用 `ne_ids|ume_ne_ids`;每台命令不同用 `targets=[{ume_ne_id, commands},…]`),勿逐台循环。 ## 快速决策树 @@ -64,7 +64,7 @@ Prefer these fixed paths for short group/DM asks (EN first; ZH aliases still wor | how many alarms / tally | ① `runUmeDiagnostics` or `aggregateUmeAlarms`; ② report by_severity + freshness | | export Excel / send spreadsheet | `ume_alarm_xlsx_report` **or** `write_xlsx(..., deliverable=true)`; never split into 3 steps | | CRC in area PAD / ACH / … | `queryUmeAlarmsRaw(keyword=CRC)` then keep rows whose `alarm_host_name` / `ne_host_name` starts with area prefix (`PAD-`, `ACH-`, …). Optional xlsx via `write_xlsx(deliverable=true)` | -| bandwidth / congestion / usage rate (+ area) | keyword=`bandwidth` (do **not** require event_type unless user asks); filter hostname prefix for area; if CLI confirm false positives: top 3–5 `ume_ne_ids` in **one** `execManagedNe(ume_ne_ids=[…], commands=[…])` batch — never one-NE loops | +| bandwidth / congestion / usage rate (+ area) | keyword=`bandwidth` (do **not** require event_type unless user asks); filter hostname prefix for area; if CLI confirm false positives: top 3–5 NEs in **one** batch — same show → `ume_ne_ids=[…]`+`commands`; mixed vendors → `targets=[{ume_ne_id, commands},…]` — never one-NE loops | | BN EMS / dying gasp / unmanaged (+ area) | keyword or native cause match (`BN EMS` / `dying gasp`); filter area prefix; short EN summary + optional xlsx | | power / temperature / fan alarms (+ area/NE) | keyword=`power` / `temperature` / `fan`; scope to host or area prefix | | alarm on **one hostname** (e.g. `MDN-PLSP`, `MKS-SWBP-EN1`) | `queryUmeAlarms` / `queryUmeAlarmsRaw` with `host_name` / keyword=hostname. **Never** start a scheduled License/daily playbook | diff --git a/skills/_workspace/ops/ops-netx-ume-playbook/reference.md b/skills/_workspace/ops/ops-netx-ume-playbook/reference.md index be44450b..a1e1a6cd 100644 --- a/skills/_workspace/ops/ops-netx-ume-playbook/reference.md +++ b/skills/_workspace/ops/ops-netx-ume-playbook/reference.md @@ -88,4 +88,4 @@ limit 50 ## 6) 登设备 - 见 `ops-netx-managed-ne-playbook` -- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI);多台同 show → `execManagedNe(ume_ne_ids=[…], commands=[…])` 一批 +- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI);多台同 show → `execManagedNe(ume_ne_ids=[…], commands=[…])` 一批;每台命令不同 → `execManagedNe(targets=[{ume_ne_id, commands},…])` 一批,勿逐台循环 diff --git a/tests/test_exec_managed_ne_guard.py b/tests/test_exec_managed_ne_guard.py index 4dcc6ef5..7e90043d 100644 --- a/tests/test_exec_managed_ne_guard.py +++ b/tests/test_exec_managed_ne_guard.py @@ -19,10 +19,32 @@ def test_is_batch_exec_args() -> None: assert is_batch_exec_args({"ne_ids": ["a", "b"], "commands": ["show version"]}) assert is_batch_exec_args({"ume_ne_ids": ["u1"]}) assert is_batch_exec_args({"targets": [{"ne_id": "x"}]}) + assert is_batch_exec_args( + { + "targets": [ + {"ume_ne_id": "hw1", "commands": ["display optical-module brief"]}, + {"ume_ne_id": "cs1", "commands": ["show interface transceiver"]}, + ] + } + ) assert not is_batch_exec_args({"ne_id": "x", "commands": ["show version"]}) assert not is_batch_exec_args({"ne_ids": []}) +def test_budget_block_payload_mentions_hetero_targets() -> None: + from runtime.chat.exec_managed_ne_guard import budget_block_payload + + payload = budget_block_payload( + reason="call_budget", + lang="en", + single_used=6, + single_budget=6, + ) + assert "targets=" in str(payload.get("hint") or "") + assert isinstance(payload.get("example_hetero_targets"), dict) + assert payload["example_hetero_targets"]["targets"] + + def test_normalize_exec_managed_ne_args_defaults_and_clamps() -> None: assert normalize_exec_managed_ne_args({})["read_timeout_sec"] == 60 assert normalize_exec_managed_ne_args({"read_timeout_sec": 5})["read_timeout_sec"] == 10