From 4660cc7507e7963ac41e512d55e46175d8e59d12 Mon Sep 17 00:00:00 2001 From: oliver Date: Wed, 12 Aug 2026 22:28:24 +0800 Subject: [PATCH] Treat missing MCP binding keys as empty and stress batch-only multi-NE CLI in ops skills. Co-authored-by: Cursor --- docs/MCP_LOCAL_SERVER.md | 1 + runtime/tools/mcp/adapter.py | 8 +-- .../ops/ops-netx-managed-ne-playbook/SKILL.md | 52 ++++++++++++++++--- .../ops/ops-netx-ume-playbook/SKILL.md | 3 +- .../ops/ops-netx-ume-playbook/reference.md | 18 ++++++- tests/test_mcp_adapter.py | 23 ++++++++ 6 files changed, 93 insertions(+), 12 deletions(-) diff --git a/docs/MCP_LOCAL_SERVER.md b/docs/MCP_LOCAL_SERVER.md index 94d89097..fcfcd030 100644 --- a/docs/MCP_LOCAL_SERVER.md +++ b/docs/MCP_LOCAL_SERVER.md @@ -181,6 +181,7 @@ for raw in sys.stdin: - 管理台 Plugins:**【6】专家 MCP 绑定看板**、**【7】MCP 专家绑定(编辑)** - 持久化键:`mcp_specialist_server_binding`(及粗粒度兜底 `mcp_allowed_specialists` / `AIA_MCP_SPECIALISTS`) +- 语义:绑定 JSON **已有条目**时,某专家**缺键 / null / `[]`** → 该专家 **不挂任何 MCP**;仅当绑定未配置或为 `{}` 时才回退粗粒度 allowlist(可见全部已启用 MCP) - 运行时:`materialize_mcp_tools_for_specialist`;上送前仅做 schema complete 与可选 JSON 体积压缩(`prepare_openai_tools_for_llm_api`) --- diff --git a/runtime/tools/mcp/adapter.py b/runtime/tools/mcp/adapter.py index 6fd96de8..7461f0a8 100644 --- a/runtime/tools/mcp/adapter.py +++ b/runtime/tools/mcp/adapter.py @@ -240,21 +240,21 @@ def materialize_mcp_tools_for_specialist( if sp in {"manager", "manager_self", "main"}: sp = "generalist" # Preferred mapping: specialist -> server_ids + # - missing/null key while a binding map exists → treat as [] (no MCP) + # - empty binding setting or {} → fall back to coarse allowlist (all enabled MCP) binding_server_ids: set[str] | None = None try: if store is not None and sp: raw_binding = str(store.get_setting("mcp_specialist_server_binding") or "").strip() if raw_binding: obj = json.loads(raw_binding) - if isinstance(obj, dict): + if isinstance(obj, dict) and obj: rows = obj.get(sp) # Legacy: if this specialist has no key, try manager bindings for generalist. if rows is None and sp == "generalist" and "manager" in obj: rows = obj.get("manager") - # 缺键或 null:视为未配置该专家的绑定 → 走下方「仅 coarse allowlist」逻辑(可见全部已启用 MCP)。 - # 仅当键存在且为 JSON 数组时,才按白名单过滤(含空数组 = 刻意不给该专家任何 MCP)。 if rows is None: - binding_server_ids = None + binding_server_ids = set() elif isinstance(rows, list): binding_server_ids = {str(x).strip() for x in rows if str(x).strip()} else: diff --git a/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md b/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md index e2b0a918..4f474a74 100644 --- a/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md +++ b/skills/_workspace/ops/ops-netx-managed-ne-playbook/SKILL.md @@ -22,11 +22,12 @@ description: 面向 ops 专家的 netx 纳管网元(网元管理)作业手 - **UME 清单(无需逐台纳管)**:`mcp__netx__listCliTargets`(`source=ume`)或 `queryUmeNeInventory` 取 `ne_id`,再用 `ume_ne_id` 执行 CLI(需先在 netx **UME → CLI 连接** 配置统一凭据/跳板) 2. **登录查信息** - `mcp__netx__execManagedNe`:`ne_id` **或** `ume_ne_id` + `commands`(默认最多 5 条,可由 `NETX_NE_EXEC_MAX_COMMANDS` 调高,硬上限 50) - - **多台必须 batch-first(一次调用,服务端并发登录,默认 concurrency=4,最多 20 台)**: + - **多台必须 batch-first(一次工具调用,netx 侧并发登录,默认 concurrency=4,最多 20 台)**: - **同命令**:`ne_ids` / `ume_ne_ids` + 共享 `commands` - **每台命令不同(厂商/角色不同)**:用 `targets=[{ume_ne_id|ne_id, commands:[…]}, …]` 一次提交;**不要**因为命令不同就退化成逐台 `execManagedNe` - 可按厂商拆成 1~2 次 batch(华为一批、Cisco 一批),仍远好于 N 次单台 - - **禁止**对多台排查逐台循环 `execManagedNe`(stdio 串行 + 重复登录) + - **关键:一轮里连发多次 `execManagedNe` ≠ 并行**。该工具不走只读并行调度,且同一 MCP stdio 串行排队——5 次单台调用就是串行 5 次。多台排查只允许 **一次** batch(`ne_ids`/`ume_ne_ids`/`targets`),不要「并行」下 N 个单台 tool call。 + - **禁止**对多台排查逐台循环 / 同轮 fan-out 多个单台 `execManagedNe`(stdio 串行 + 重复登录 + 易撞预算) - **一次会话内**:`listCliTargets` 最多调用一次,缓存返回的 id;同台多条 show 合并进该台的 `commands[]`,禁止「list→exec→list→exec」循环 - 超时:提高 `read_timeout_sec`(默认 60,慢命令 90–120)或减少命令条数,禁止对同一命令盲重试 @@ -55,10 +56,49 @@ Cisco/Huawei: use `show interface transceiver` / `display optical-module` style ### Multi-NE CLI (batch-first) -- **Same show on many NEs**: `execManagedNe(ume_ne_ids=[…], commands=[…])` once. -- **Different commands per NE** (vendor / role): one call with - `targets=[{ume_ne_id, commands:[…]}, {ume_ne_id, commands:[…]}, …]` — still concurrent on the server. -- Cap to NEs on the asked path (usually 2–5). Never one-NE `execManagedNe` loops. +- **Same show on many NEs**: `execManagedNe(ume_ne_ids=[…], commands=[…])` **once**. +- **Different commands per NE** (vendor / role): **one** call with + `targets=[{ume_ne_id, commands:[…]}, {ume_ne_id, commands:[…]}, …]` — concurrency is inside that batch on netx. +- **Wrong**: emit N× `execManagedNe(ume_ne_id=…)` in the same turn hoping they run in parallel — they run **serial** (no read-only parallel batch; MCP stdio lock). +- Cap to NEs on the asked path (usually 2–5). Never one-NE loops / fan-out. + +#### Examples(对 / 错) + +**✓ 同命令多台(一次调用,服务端并发)** + +```json +{ + "ume_ne_ids": ["uuid-a", "uuid-b", "uuid-c"], + "commands": ["show version"], + "read_timeout_sec": 60, + "concurrency": 4 +} +``` + +纳管 id 同理:`ne_ids` + `commands`。 + +**✓ 每台命令不同(仍一次调用)** + +```json +{ + "targets": [ + {"ume_ne_id": "uuid-zte", "commands": ["show opticalinfo brief"]}, + {"ume_ne_id": "uuid-hw", "commands": ["display optical-module brief"]}, + {"ume_ne_id": "uuid-cisco", "commands": ["show interface transceiver"]} + ], + "read_timeout_sec": 90 +} +``` + +**✗ 错误:同轮 fan-out 三次单台(会串行,且易撞预算)** + +```text +call1: execManagedNe({ "ume_ne_id": "uuid-a", "commands": ["show version"] }) +call2: execManagedNe({ "ume_ne_id": "uuid-b", "commands": ["show version"] }) +call3: execManagedNe({ "ume_ne_id": "uuid-c", "commands": ["show version"] }) +``` + +应合并成上面的 `ume_ne_ids` 一次调用。 ## CLI 约束(服务端强制) diff --git a/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md b/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md index 1a906116..77a99e4e 100644 --- a/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md +++ b/skills/_workspace/ops/ops-netx-ume-playbook/SKILL.md @@ -38,7 +38,8 @@ description: 面向 ops 专家的 netx UME 运维作业手册。覆盖告警查 4. 自定义聚合:`aggregateUmeAlarmsRaw`(`group_by=alarm_host_name` 等)。 5. SQL:`sqlQueryUme`(仅 SELECT;设 `statement_timeout_ms`)。 6. **告警关联拓扑**:两台相关网元取 `ne_id` → `findTopologyPaths`(最短路径优先)。 -7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`;多台用一次 `execManagedNe` batch(同命令用 `ne_ids|ume_ne_ids`;每台命令不同用 `targets=[{ume_ne_id, commands},…]`),勿逐台循环。 +7. **登设备查 CLI**:见 `ops-netx-managed-ne-playbook`(含对/错 JSON 示例);多台必须 **一次** `execManagedNe` batch(同命令用 `ne_ids|ume_ne_ids`;每台命令不同用 `targets=[{ume_ne_id, commands},…]`)。同轮连发多次单台 `execManagedNe` **不会并行**(stdio 串行),禁止。 + - 例:`{"ume_ne_ids":["uuid-a","uuid-b"],"commands":["show version"]}`;混厂商用 `targets=[…]` 仍一次调用。 ## 快速决策树 diff --git a/skills/_workspace/ops/ops-netx-ume-playbook/reference.md b/skills/_workspace/ops/ops-netx-ume-playbook/reference.md index a1e1a6cd..433f7d21 100644 --- a/skills/_workspace/ops/ops-netx-ume-playbook/reference.md +++ b/skills/_workspace/ops/ops-netx-ume-playbook/reference.md @@ -88,4 +88,20 @@ limit 50 ## 6) 登设备 - 见 `ops-netx-managed-ne-playbook` -- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI);多台同 show → `execManagedNe(ume_ne_ids=[…], commands=[…])` 一批;每台命令不同 → `execManagedNe(targets=[{ume_ne_id, commands},…])` 一批,勿逐台循环 +- UME `ne_id` → `listCliTargets` / `execManagedNe(ume_ne_id=…)`(需已配 UME→CLI);多台同 show → `execManagedNe(ume_ne_ids=[…], commands=[…])` **一次**;每台命令不同 → `execManagedNe(targets=[{ume_ne_id, commands},…])` **一次**。同轮 N 次单台调用会串行,不算并行;勿逐台循环 / fan-out + +示例: + +```json +// ✓ 同命令 +{"ume_ne_ids": ["uuid-1", "uuid-2"], "commands": ["show version"], "read_timeout_sec": 60} + +// ✓ 混厂商 +{"targets": [ + {"ume_ne_id": "uuid-zte", "commands": ["show opticalinfo brief"]}, + {"ume_ne_id": "uuid-hw", "commands": ["display optical-module brief"]} +]} + +// ✗ 同轮三次单台(串行)— 禁止 +// execManagedNe(ume_ne_id=uuid-1, …); execManagedNe(ume_ne_id=uuid-2, …); … +``` diff --git a/tests/test_mcp_adapter.py b/tests/test_mcp_adapter.py index 35da69b9..7ab754c4 100644 --- a/tests/test_mcp_adapter.py +++ b/tests/test_mcp_adapter.py @@ -154,6 +154,29 @@ class McpAdapterTests(unittest.TestCase): self.assertIn("mcp__echo-a__ping", names) self.assertIn("mcp__echo-b__ping", names) + def test_binding_missing_key_denies_mcp_when_map_exists(self) -> None: + """Sibling keys present but this specialist missing → treat as [] (no MCP).""" + with tempfile.TemporaryDirectory(ignore_cleanup_errors=True) as td: + store = SqliteStore(str(Path(td) / "ops.sqlite")) + store.upsert_mcp_server( + server_id="echo-a", + source_type="github", + source_ref="https://github.com/acme/a", + entry_command="python", + entry_args=["-m", "a"], + enabled=True, + ) + store.replace_mcp_server_tools( + server_id="echo-a", + tools=[{"tool_name": "ping", "description": "P", "parameters": {"type": "object", "properties": {}}}], + ) + store.set_setting("mcp_specialist_server_binding", '{"ops":["echo-a"]}') + g_specs = materialize_mcp_tools_for_specialist(store, specialist="generalist") + o_specs = materialize_mcp_tools_for_specialist(store, specialist="ops") + self.assertEqual(len(g_specs), 0) + self.assertEqual(len(o_specs), 1) + self.assertEqual(o_specs[0].name, "mcp__echo-a__ping") + def test_mcp_local_env_file_path_prefers_src_local(self) -> None: from runtime.operations import mcp_env