mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 00:40:45 +08:00
Prefer multi-NE execManagedNe batch and parallelize read MCP tools.
Guide ops to use ne_ids/ume_ne_ids for sweeps, raise exec wall-clock for batches, and mark inventory/list MCP tools read_only so consecutive queries can run in parallel. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
021ed0cc4f
commit
f31ebeb30e
5 changed files with 91 additions and 3 deletions
|
|
@ -21,8 +21,9 @@ def _mcp_row_env_config(row: dict[str, Any]) -> tuple[list[str], dict[str, str]]
|
|||
|
||||
# Long-running netx tools exceed the generic MCP row timeout (often 30s).
|
||||
# Production WA ops showed execManagedNe p90/p95 glued to ~30000ms timeouts.
|
||||
# Batch multi-NE exec posts once to /exec-batch (server concurrency); allow longer wall clock.
|
||||
_MCP_TOOL_TIMEOUT_OVERRIDES_S: dict[str, float] = {
|
||||
"execManagedNe": 320.0,
|
||||
"execManagedNe": 620.0,
|
||||
"sqlQueryUme": 90.0,
|
||||
"findTopologyPaths": 60.0,
|
||||
"aggregateUmeAlarmsRaw": 60.0,
|
||||
|
|
@ -44,6 +45,26 @@ _MCP_LIST_CACHE_TTL_S: dict[str, float] = {
|
|||
"findTopologyPaths": 60.0,
|
||||
}
|
||||
|
||||
# Safe to run in parallel with other consecutive read-only tools (separate MCP stdio processes).
|
||||
# Do NOT include execManagedNe: same-tool fans share one stdio lock — use ne_ids/ume_ne_ids batch instead.
|
||||
_MCP_READ_ONLY_TOOLS: frozenset[str] = frozenset(
|
||||
{
|
||||
"listCliTargets",
|
||||
"listManagedNe",
|
||||
"getManagedNe",
|
||||
"getUmeNe",
|
||||
"queryUmeNeInventory",
|
||||
"queryUmeAlarms",
|
||||
"queryUmeAlarmsRaw",
|
||||
"aggregateUmeAlarms",
|
||||
"aggregateUmeAlarmsRaw",
|
||||
"listUmeAlarmFields",
|
||||
"runUmeDiagnostics",
|
||||
"findTopologyPaths",
|
||||
"sqlQueryUme",
|
||||
}
|
||||
)
|
||||
|
||||
_MCP_LIST_CACHE_LOCK = threading.Lock()
|
||||
_MCP_LIST_CACHE: dict[str, tuple[float, float, dict[str, Any]]] = {}
|
||||
|
||||
|
|
@ -176,6 +197,7 @@ class _McpBoundTool:
|
|||
timeout_s=self.timeout_s,
|
||||
required_permissions=self.required_permissions,
|
||||
execution_mode="subprocess",
|
||||
read_only=self.tool_name in _MCP_READ_ONLY_TOOLS,
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -38,7 +38,7 @@ You are the ops specialist (network operations expert).
|
|||
- Spreadsheet delivery: `ume_alarm_xlsx_report` or `write_xlsx(deliverable=true)` — never claim a file was sent without deliverable marking.
|
||||
- **Field default is English**: WhatsApp channel dispatch defaults to `lang=en`; user-visible replies must contain **zero CJK**. Translate Chinese tool fields before display.
|
||||
- Group chats default to **per-speaker session isolation** (members do not share dialogue memory within the same group).
|
||||
- Call `listCliTargets` at most once per session and reuse ids; batch `execManagedNe` commands; default `read_timeout_sec=60` — on timeout raise it, no blind retries.
|
||||
- Call `listCliTargets` at most once per session and reuse ids; for many NEs with the same show commands use one `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` (server concurrency) — do not loop one-NE calls; default `read_timeout_sec=60` — on timeout raise it, no blind retries.
|
||||
- Replies like `YES` / `confirm` / `继续` / `please continue`: continue the previous unfinished task — do **not** re-ask for confirmation or restart the query.
|
||||
- On `tool_invalid_arguments`, fix args using the returned `example`; on timeout hints, raise `read_timeout_sec` or shrink commands.
|
||||
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@
|
|||
- 用户要表格/Excel:`ume_alarm_xlsx_report` 或 `write_xlsx(deliverable=true)`;禁止只写文件不投递。
|
||||
- **现场默认英文**:WhatsApp 渠道默认 `lang=en`;英文会话回复不得含汉字;工具中文字段先翻译再展示。
|
||||
- 群聊默认按**发言人隔离会话**(同群不同人互不串上下文);勿假设「群共享一个对话记忆」。
|
||||
- `listCliTargets` 每会话最多查一次并复用 id;`execManagedNe` 合并 commands,超时调 `read_timeout_sec`(默认 60),禁止盲重试。
|
||||
- `listCliTargets` 每会话最多查一次并复用 id;多台同命令用 `execManagedNe(ne_ids|ume_ne_ids=..., commands=...)` 一批并发,勿逐台循环;超时调 `read_timeout_sec`(默认 60),禁止盲重试。
|
||||
- 用户回复 `YES` / `confirm` / `确认` / `可以` / `继续` / `please continue`:直接承接上一未完成任务继续执行,**不要**再问一遍确认或重开查询。
|
||||
- 工具返回 `tool_invalid_arguments` 时按返回的 `example` 修正参数;返回超时 hint 时提高 `read_timeout_sec` 或减命令,禁止相同参数重试。
|
||||
|
||||
|
|
|
|||
|
|
@ -21,6 +21,7 @@ description: 面向 ops 专家的 netx 纳管网元(网元管理)作业手
|
|||
- **UME 清单(无需逐台纳管)**:`mcp__netx__listCliTargets`(`source=ume`)或 `queryUmeNeInventory` 取 `ne_id`,再用 `ume_ne_id` 执行 CLI(需先在 netx **UME → CLI 连接** 配置统一凭据/跳板)
|
||||
2. **登录查信息**
|
||||
- `mcp__netx__execManagedNe`:`ne_id` **或** `ume_ne_id` + `commands`(默认最多 5 条,可由 `NETX_NE_EXEC_MAX_COMMANDS` 调高,硬上限 50)
|
||||
- **多台同命令(推荐)**:一次调用传 `ne_ids` / `ume_ne_ids`(或 `targets`)+ 共享 `commands`,服务端并发执行(默认 concurrency=4,最多 20 台)。禁止对同一 show 命令逐台循环 `execManagedNe`
|
||||
- **一次会话内**:`listCliTargets` 最多调用一次,缓存返回的 id;多条 show 合并进同一次 `commands`,禁止「list→exec→list→exec」循环
|
||||
- 超时:提高 `read_timeout_sec`(默认 60,慢命令 90–120)或减少命令条数,禁止对同一命令盲重试
|
||||
|
||||
|
|
|
|||
65
tests/test_mcp_read_only_parallel.py
Normal file
65
tests/test_mcp_read_only_parallel.py
Normal file
|
|
@ -0,0 +1,65 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from runtime.tools.base import ToolRegistry, ToolSpec
|
||||
from runtime.tools.mcp.adapter import _MCP_READ_ONLY_TOOLS, _McpBoundTool, mcp_timeout_for_tool
|
||||
|
||||
|
||||
def test_list_query_mcp_tools_marked_read_only() -> None:
|
||||
for name in ("listManagedNe", "queryUmeAlarmsRaw", "listCliTargets", "findTopologyPaths"):
|
||||
assert name in _MCP_READ_ONLY_TOOLS
|
||||
spec = _McpBoundTool(
|
||||
server_id="netx",
|
||||
tool_name=name,
|
||||
description="t",
|
||||
parameters={"type": "object", "properties": {}},
|
||||
command=["echo"],
|
||||
).to_spec()
|
||||
assert spec.read_only is True
|
||||
assert spec.name == f"mcp__netx__{name}"
|
||||
|
||||
|
||||
def test_exec_managed_ne_not_read_only_use_batch_instead() -> None:
|
||||
assert "execManagedNe" not in _MCP_READ_ONLY_TOOLS
|
||||
spec = _McpBoundTool(
|
||||
server_id="netx",
|
||||
tool_name="execManagedNe",
|
||||
description="t",
|
||||
parameters={"type": "object", "properties": {}},
|
||||
command=["echo"],
|
||||
).to_spec()
|
||||
assert spec.read_only is False
|
||||
|
||||
|
||||
def test_exec_managed_ne_timeout_allows_batch_wall_clock() -> None:
|
||||
assert mcp_timeout_for_tool("execManagedNe", 30.0) >= 600.0
|
||||
|
||||
|
||||
def test_partition_would_parallelize_consecutive_list_tools() -> None:
|
||||
from runtime.chat.tool_runtime import partition_tool_use_batches
|
||||
from svc.llm.chat_models import LLMToolCall
|
||||
|
||||
reg = ToolRegistry(
|
||||
[
|
||||
ToolSpec(
|
||||
name="mcp__netx__listManagedNe",
|
||||
description="a",
|
||||
parameters={},
|
||||
handler=lambda _a: {"ok": True},
|
||||
read_only=True,
|
||||
),
|
||||
ToolSpec(
|
||||
name="mcp__netx__queryUmeAlarmsRaw",
|
||||
description="b",
|
||||
parameters={},
|
||||
handler=lambda _a: {"ok": True},
|
||||
read_only=True,
|
||||
),
|
||||
]
|
||||
)
|
||||
calls = [
|
||||
LLMToolCall(id="1", name="mcp__netx__listManagedNe", arguments={}),
|
||||
LLMToolCall(id="2", name="mcp__netx__queryUmeAlarmsRaw", arguments={}),
|
||||
]
|
||||
batches = partition_tool_use_batches(calls, reg)
|
||||
assert len(batches) == 1
|
||||
assert len(batches[0]) == 2
|
||||
Loading…
Add table
Add a link
Reference in a new issue