Improve UME ops triage APIs and slim fabric path payloads.

Add severity/time filters, exclude missing hosts from aggregates, and default topology paths to compact summary fields for MCP-friendly alarm triage.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-10 22:17:45 +08:00
parent 753740d64e
commit 78d174232d
6 changed files with 505 additions and 50 deletions

View file

@ -468,11 +468,15 @@ def find_fabric_paths(
max_paths: int = 3,
max_hops: int = 6,
layer: str = "physical",
detail: str = "summary",
) -> dict[str, Any]:
"""Find up to max_paths simple paths between two fabric nodes.
Accepts ume_ne_id (from UME alarms) or managed_ne_id (from managed NE) — resolved
to fabric_node_id internally so agents can use alarm ne_id directly.
detail=summary (default): compact node/edge fields for ops/MCP.
detail=full: full FabricNodeOut/FabricEdgeOut payloads.
"""
from_uid = str(from_ume_ne_id or "").strip()
from_mid = str(from_managed_ne_id or "").strip()
@ -502,6 +506,9 @@ def find_fabric_paths(
max_paths = max(1, min(10, int(max_paths or 3)))
max_hops = max(1, min(12, int(max_hops or 6)))
layer_v = str(layer or "physical").strip() or "physical"
detail_v = str(detail or "summary").strip().lower() or "summary"
if detail_v not in {"summary", "full"}:
raise HTTPException(400, detail="invalid_detail_use_summary_or_full")
# Lazy adjacency: only fetch edges for nodes the BFS actually expands
# (avoids loading the entire fabric layer on large graphs).
@ -565,7 +572,29 @@ def find_fabric_paths(
node_ids.add(e.b_node_id)
node_map = _nodes_by_ids(db, node_ids)
def _path_nodes(edge_ids: list[str]) -> list[dict]:
def _node_summary(n: TopoFabricNode) -> dict[str, Any]:
return {
"name": n.name or "",
"ip": n.ip or "",
"ume_ne_id": n.ume_ne_id or "",
"managed_ne_id": n.managed_ne_id or "",
"vendor": n.vendor or "",
}
def _edge_summary(e: TopoFabricEdge) -> dict[str, Any]:
a_node = node_map.get(e.a_node_id)
b_node = node_map.get(e.b_node_id)
return {
"a_name": (a_node.name if a_node else "") or "",
"b_name": (b_node.name if b_node else "") or "",
"a_ip": (a_node.ip if a_node else "") or "",
"b_ip": (b_node.ip if b_node else "") or "",
"a_port": e.a_port or "",
"b_port": e.b_port or "",
"status": _normalize_edge_status(e.status or "active"),
}
def _path_node_ids(edge_ids: list[str]) -> list[str]:
ids = [from_id]
cur = from_id
for eid in edge_ids:
@ -575,18 +604,53 @@ def find_fabric_paths(
nxt = e.b_node_id if e.a_node_id == cur else e.a_node_id
ids.append(nxt)
cur = nxt
return [_node_out(node_map[nid]).model_dump() for nid in ids if nid in node_map]
return ids
def _path_nodes(edge_ids: list[str]) -> list[dict]:
ids = _path_node_ids(edge_ids)
if detail_v == "full":
return [_node_out(node_map[nid]).model_dump() for nid in ids if nid in node_map]
return [_node_summary(node_map[nid]) for nid in ids if nid in node_map]
def _path_edges(edge_ids: list[str]) -> list[dict]:
if detail_v == "full":
return [
_edge_out(edge_map[eid], nodes_by_id=node_map).model_dump()
for eid in edge_ids
if eid in edge_map
]
return [_edge_summary(edge_map[eid]) for eid in edge_ids if eid in edge_map]
def _path_label(edge_ids: list[str]) -> str:
ids = _path_node_ids(edge_ids)
names = [(node_map[nid].name if nid in node_map else nid) or nid for nid in ids]
if not edge_ids:
return " -> ".join(names)
parts: list[str] = [names[0]]
for i, eid in enumerate(edge_ids):
e = edge_map.get(eid)
nxt = names[i + 1] if i + 1 < len(names) else "?"
if not e:
parts.append(f"-> {nxt}")
continue
# Port facing the next hop from current node orientation.
cur_id = ids[i]
port = e.a_port if e.a_node_id == cur_id else e.b_port
parts.append(f"-[{port or '?'}]-> {nxt}")
return " ".join(parts)
return {
"from_node_id": from_id,
"to_node_id": to_id,
"layer": layer_v,
"detail": detail_v,
"path_count": len(found),
"paths": [
{
"hops": len(p),
"label": _path_label(p),
"nodes": _path_nodes(p),
"edges": [_edge_out(edge_map[eid], nodes_by_id=node_map).model_dump() for eid in p if eid in edge_map],
"edges": _path_edges(p),
}
for p in found
],

View file

@ -160,6 +160,7 @@ def api_fabric_paths(body: dict[str, Any] = Body(...), db: Session = Depends(get
max_paths=int(body.get("max_paths") or 3),
max_hops=int(body.get("max_hops") or 6),
layer=str(body.get("layer") or "physical"),
detail=str(body.get("detail") or "summary"),
)

View file

@ -6,7 +6,7 @@ from datetime import datetime, timezone
from typing import Any
from fastapi import APIRouter, Depends, HTTPException, Query
from sqlalchemy import or_
from sqlalchemy import func, or_
from sqlalchemy.orm import Session
from .config import settings
@ -78,6 +78,8 @@ def ume_list_alarms(
ne_id: str | None = Query(default=None),
host_name: str | None = Query(default=None),
keyword: str | None = Query(default=None),
time_from: str | None = Query(default=None, description="Filter by last_seen_at >="),
time_to: str | None = Query(default=None, description="Filter by last_seen_at <="),
page: int = Query(default=1, ge=1),
page_size: int = Query(default=50, ge=1, le=500),
db: Session = Depends(get_db),
@ -109,6 +111,12 @@ def ume_list_alarms(
| UmeInventoryNE.ip_address.contains(kw)
| UmeInventoryNE.host_name.contains(kw)
)
dt_from = _parse_time(time_from)
dt_to = _parse_time(time_to)
if dt_from:
stmt = stmt.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None))
if dt_to:
stmt = stmt.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None))
total = int(stmt.count())
rows = (
stmt.order_by(
@ -139,7 +147,13 @@ def ume_list_alarms(
}
for alarm, ne in rows
]
return {"total": total, "page": page, "page_size": page_size, "items": items}
return {
"total": total,
"page": page,
"page_size": page_size,
"meta": {"time_filter_field": "last_seen_at"},
"items": items,
}
@router.get("/v1/ume/alarms/fields")
@ -309,6 +323,29 @@ def ume_alarms_raw(
}
_HOST_MISSING_LABEL = "(host_name missing)"
_HOST_GROUP_FIELDS = frozenset({"alarm_host_name", "ne_host_name", "ne_user_label"})
def _normalize_ne_bucket_key(raw: Any) -> str:
s = str(raw or "").strip()
if not s or s.lower() in {"unknown", "none", "null"}:
return _HOST_MISSING_LABEL
# Bare UUID or ME{uuid}
if len(s) >= 32 and s.count("-") >= 4 and " " not in s:
return _HOST_MISSING_LABEL
if s.startswith("ME{") and s.endswith("}"):
return _HOST_MISSING_LABEL
return s
def _raw_group_bucket_key(field: str, value: str) -> str:
if field in _HOST_GROUP_FIELDS:
return _normalize_ne_bucket_key(value)
s = str(value or "").strip()
return s if s else "(empty)"
@router.get("/v1/ume/alarms/aggregate/raw")
def ume_alarms_aggregate_raw(
group_by: str = Query(default="alarm_perceived_severity"),
@ -320,6 +357,10 @@ def ume_alarms_aggregate_raw(
keyword: str | None = Query(default=None),
time_from: str | None = Query(default=None),
time_to: str | None = Query(default=None),
exclude_missing_host: bool = Query(
default=True,
description="When grouping by host/user_label fields, omit (host_name missing) from buckets.",
),
limit: int = Query(default=200, ge=1, le=2000),
db: Session = Depends(get_db),
) -> dict[str, Any]:
@ -362,17 +403,30 @@ def ume_alarms_aggregate_raw(
stmt = stmt.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None))
rows = stmt.order_by(UmeAlarmCurrent.last_seen_at.desc()).all()
counts: dict[tuple[str, str], int] = {}
merged: dict[tuple[str, str], int] = {}
by_ne_missing = 0
for alarm, ne in rows:
k1 = _extract_ume_raw_group_field(alarm, ne, g1)
k2 = _extract_ume_raw_group_field(alarm, ne, g2) if g2 else ""
kk = (k1, k2)
counts[kk] = int(counts.get(kk, 0)) + 1
buckets = sorted(counts.items(), key=lambda x: x[1], reverse=True)[: int(limit)]
nk1 = _raw_group_bucket_key(g1, _extract_ume_raw_group_field(alarm, ne, g1))
nk2 = _raw_group_bucket_key(g2, _extract_ume_raw_group_field(alarm, ne, g2)) if g2 else ""
missing_hit = False
if g1 in _HOST_GROUP_FIELDS and nk1 == _HOST_MISSING_LABEL:
missing_hit = True
if g2 and g2 in _HOST_GROUP_FIELDS and nk2 == _HOST_MISSING_LABEL:
missing_hit = True
if missing_hit:
by_ne_missing += 1
if exclude_missing_host:
continue
kk = (nk1, nk2)
merged[kk] = int(merged.get(kk, 0)) + 1
buckets = sorted(merged.items(), key=lambda x: x[1], reverse=True)[: int(limit)]
return {
"total": len(rows),
"group_by": g1,
"group_by2": g2 or None,
"by_ne_missing": by_ne_missing,
"exclude_missing_host": bool(exclude_missing_host),
"meta": {
"available_fields": sorted(selectable_fields),
"group_by_allowed": sorted(selectable_fields),
@ -387,60 +441,274 @@ def ume_alarms_aggregate_raw(
},
"time_filter_field": "last_seen_at",
"limit": int(limit),
"host_missing_label": _HOST_MISSING_LABEL,
},
"buckets": [
{"key": k1, "key2": (k2 if g2 else None), "count": int(v)}
{
"key": k1,
"key2": (k2 if g2 else None),
"count": int(v),
}
for (k1, k2), v in buckets
],
}
@router.get("/v1/ume/alarms/aggregate")
def ume_alarms_aggregate(db: Session = Depends(get_db)) -> dict[str, Any]:
rows = db.query(UmeAlarmCurrent, UmeInventoryNE).outerjoin(
UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id
).all()
by_severity = _aggregate_rows(rows, lambda x: x[0].perceived_severity)
by_ne = _aggregate_rows(rows, lambda x: _ume_alarm_ne_group_key(x[0], x[1]))
return {"total": len(rows), "by_severity": by_severity, "by_ne": by_ne}
def ume_alarms_aggregate(
top_ne: int = Query(
default=50,
ge=0,
le=5000,
description="Max NE buckets to return (0 = all). Severity buckets are always complete.",
),
exclude_missing_host: bool = Query(
default=True,
description="When true, omit (host_name missing) from by_ne ranking (count still in by_ne_missing).",
),
severity: str | None = Query(
default=None,
description="Optional perceived_severity filter (e.g. critical) for top-NE ranking.",
),
time_from: str | None = Query(default=None, description="Filter by last_seen_at >="),
time_to: str | None = Query(default=None, description="Filter by last_seen_at <="),
db: Session = Depends(get_db),
) -> dict[str, Any]:
"""Aggregate current alarms by severity and NE (SQL group-by; top_ne capped)."""
dt_from = _parse_time(time_from)
dt_to = _parse_time(time_to)
sev = str(severity or "").strip() or None
base = db.query(UmeAlarmCurrent)
if sev:
base = base.filter(UmeAlarmCurrent.perceived_severity == sev)
if dt_from:
base = base.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None))
if dt_to:
base = base.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None))
# Prefer .count() — with_entities(func.count()) without select_from can return 1.
total = int(base.count())
sev_q = db.query(UmeAlarmCurrent.perceived_severity, func.count())
if sev:
sev_q = sev_q.filter(UmeAlarmCurrent.perceived_severity == sev)
if dt_from:
sev_q = sev_q.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None))
if dt_to:
sev_q = sev_q.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None))
sev_rows = (
sev_q.group_by(UmeAlarmCurrent.perceived_severity).order_by(func.count().desc()).all()
)
by_severity = [
{"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), "count": int(v)}
for k, v in sev_rows
]
# Display key: host_name / inventory host / user_label only (never bare UUID).
ne_key = func.coalesce(
func.nullif(func.trim(UmeAlarmCurrent.host_name), ""),
func.nullif(func.trim(UmeInventoryNE.host_name), ""),
func.nullif(func.trim(UmeInventoryNE.user_label), ""),
_HOST_MISSING_LABEL,
)
ne_q = (
db.query(ne_key.label("ne_key"), func.count().label("cnt"))
.select_from(UmeAlarmCurrent)
.outerjoin(UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id)
)
if sev:
ne_q = ne_q.filter(UmeAlarmCurrent.perceived_severity == sev)
if dt_from:
ne_q = ne_q.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None))
if dt_to:
ne_q = ne_q.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None))
ne_q = ne_q.group_by(ne_key).order_by(func.count().desc())
all_ne_rows = ne_q.all()
by_ne_missing = 0
named_counts: dict[str, int] = {}
for k, v in all_ne_rows:
label = _normalize_ne_bucket_key(k)
cnt = int(v)
if label == _HOST_MISSING_LABEL:
by_ne_missing += cnt
else:
named_counts[label] = int(named_counts.get(label, 0)) + cnt
named_rows = sorted(named_counts.items(), key=lambda kv: kv[1], reverse=True)
by_ne_total = len(named_rows) + (1 if by_ne_missing else 0)
ranked = list(named_rows)
if not exclude_missing_host and by_ne_missing:
ranked.append((_HOST_MISSING_LABEL, by_ne_missing))
ranked.sort(key=lambda kv: int(kv[1]), reverse=True)
if top_ne > 0:
ranked = ranked[: int(top_ne)]
by_ne = [{"key": str(k), "count": int(v)} for k, v in ranked]
seen_bounds = db.query(
func.min(UmeAlarmCurrent.last_seen_at),
func.max(UmeAlarmCurrent.last_seen_at),
)
if sev:
seen_bounds = seen_bounds.filter(UmeAlarmCurrent.perceived_severity == sev)
if dt_from:
seen_bounds = seen_bounds.filter(
UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None)
)
if dt_to:
seen_bounds = seen_bounds.filter(
UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None)
)
min_seen, max_seen = seen_bounds.one()
return {
"total": total,
"by_severity": by_severity,
"by_ne": by_ne,
"by_ne_total": by_ne_total,
"by_ne_missing": by_ne_missing,
"top_ne": int(top_ne),
"exclude_missing_host": bool(exclude_missing_host),
"severity": sev,
"meta": {
"time_filter_field": "last_seen_at",
"last_seen_min": (_ensure_utc(min_seen).isoformat() if min_seen else None),
"last_seen_max": (_ensure_utc(max_seen).isoformat() if max_seen else None),
},
}
@router.get("/v1/ume/diagnostics")
def ume_diagnostics(
lang: str | None = Query(default=None),
top_n: int = Query(default=10, ge=1, le=50),
db: Session = Depends(get_db),
) -> dict[str, Any]:
rows = db.query(UmeAlarmCurrent, UmeInventoryNE).outerjoin(
UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id
).all()
by_severity = _aggregate_rows(rows, lambda x: x[0].perceived_severity)
by_alarm_code = _aggregate_rows(rows, lambda x: x[0].event_type)[:10]
by_ne = _aggregate_rows(rows, lambda x: _ume_alarm_ne_group_key(x[0], x[1]))[:10]
"""Fast SQL diagnostics; top_ne excludes missing host_name; includes data freshness."""
total = int(db.query(func.count()).select_from(UmeAlarmCurrent).scalar() or 0)
min_seen, max_seen = db.query(
func.min(UmeAlarmCurrent.last_seen_at),
func.max(UmeAlarmCurrent.last_seen_at),
).one()
sev_rows = (
db.query(UmeAlarmCurrent.perceived_severity, func.count())
.group_by(UmeAlarmCurrent.perceived_severity)
.order_by(func.count().desc())
.all()
)
by_severity = [
{"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), "count": int(v)}
for k, v in sev_rows
]
event_rows = (
db.query(UmeAlarmCurrent.event_type, func.count())
.group_by(UmeAlarmCurrent.event_type)
.order_by(func.count().desc())
.limit(top_n)
.all()
)
top_event_types = [
{"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), "count": int(v)}
for k, v in event_rows
]
# Prefer real UME alarmCode from raw_json when present (Postgres JSON text).
top_alarm_codes: list[dict[str, Any]] = []
try:
from sqlalchemy import cast
from sqlalchemy.dialects.postgresql import JSON
code_expr = func.coalesce(
func.nullif(
func.json_extract_path_text(cast(UmeAlarmCurrent.raw_json, JSON), "alarmCode"),
"",
),
"(none)",
)
code_rows = (
db.query(code_expr.label("code"), func.count())
.group_by(code_expr)
.order_by(func.count().desc())
.limit(top_n)
.all()
)
top_alarm_codes = [{"key": str(k), "count": int(v)} for k, v in code_rows]
except Exception:
# SQLite / non-JSON: fall back to native_probable_cause tops.
cause_rows = (
db.query(UmeAlarmCurrent.native_probable_cause, func.count())
.group_by(UmeAlarmCurrent.native_probable_cause)
.order_by(func.count().desc())
.limit(top_n)
.all()
)
top_alarm_codes = [
{
"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"),
"count": int(v),
}
for k, v in cause_rows
]
ne_key = func.coalesce(
func.nullif(func.trim(UmeAlarmCurrent.host_name), ""),
func.nullif(func.trim(UmeInventoryNE.host_name), ""),
func.nullif(func.trim(UmeInventoryNE.user_label), ""),
_HOST_MISSING_LABEL,
)
ne_rows = (
db.query(ne_key.label("ne_key"), func.count().label("cnt"))
.select_from(UmeAlarmCurrent)
.outerjoin(UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id)
.group_by(ne_key)
.order_by(func.count().desc())
.all()
)
by_ne_missing = 0
named_counts: dict[str, int] = {}
for k, v in ne_rows:
label = _normalize_ne_bucket_key(k)
cnt = int(v)
if label == _HOST_MISSING_LABEL:
by_ne_missing += cnt
continue
named_counts[label] = int(named_counts.get(label, 0)) + cnt
top_ne = [
{"key": label, "count": cnt}
for label, cnt in sorted(named_counts.items(), key=lambda kv: kv[1], reverse=True)[:top_n]
]
# Protocol buckets still need text classify; stream rows lightly (cause+event only).
lang_norm = _normalize_netx_lang(lang)
proto_counts: dict[str, int] = {}
for alarm, ne in rows:
blob = " | ".join(
[
str(alarm.event_type or ""),
str(alarm.native_probable_cause or ""),
str(alarm.object_name or ""),
str(ne.ne_name if ne else ""),
str(ne.user_label if ne else ""),
str(ne.ip_address if ne else ""),
]
)
light = db.query(
UmeAlarmCurrent.event_type,
UmeAlarmCurrent.native_probable_cause,
UmeAlarmCurrent.object_name,
).yield_per(2000)
for event_type, cause, obj in light:
blob = " | ".join([str(event_type or ""), str(cause or ""), str(obj or "")])
bucket = _protocol_bucket_label(blob, lang=lang_norm)
proto_counts[bucket] = int(proto_counts.get(bucket, 0)) + 1
protocol_summary = sorted(proto_counts.items(), key=lambda x: x[1], reverse=True)[:10]
protocol_summary = sorted(proto_counts.items(), key=lambda x: x[1], reverse=True)[:top_n]
return {
"source": "ume_alarms_current",
"total_alarms": len(rows),
"total_alarms": total,
"severity_summary": by_severity,
"top_alarm_codes": by_alarm_code,
"top_ne": by_ne,
"top_event_types": top_event_types,
# Backward-compatible alias — historically event_type, now real alarmCode when possible.
"top_alarm_codes": top_alarm_codes,
"top_ne": top_ne,
"by_ne_missing": by_ne_missing,
"protocol_summary": [{"key": k, "count": v} for k, v in protocol_summary],
"meta": {
"last_seen_min": (_ensure_utc(min_seen).isoformat() if min_seen else None),
"last_seen_max": (_ensure_utc(max_seen).isoformat() if max_seen else None),
"time_filter_field": "last_seen_at",
"host_missing_label": _HOST_MISSING_LABEL,
},
}

View file

@ -123,12 +123,24 @@ def _query_ume_alarms(args: dict[str, Any]) -> dict[str, Any]:
params["keyword"] = ne_name
if str(args.get("ne_id") or "").strip():
params["ne_id"] = str(args.get("ne_id")).strip()
if str(args.get("host_name") or "").strip():
params["host_name"] = str(args.get("host_name")).strip()
for k in ("time_from", "time_to"):
if str(args.get(k) or "").strip():
params[k] = str(args.get(k)).strip()
return http_json("GET", "/v1/ume/alarms", params=params)
def _aggregate_ume_alarms(args: dict[str, Any]) -> dict[str, Any]:
_ = args
return http_json("GET", "/v1/ume/alarms/aggregate", params=None)
# Default top 50 named NEs — missing host buckets reported separately.
top_ne = max(0, min(500, int(args.get("top_ne") if args.get("top_ne") is not None else 50)))
params: dict[str, Any] = {"top_ne": top_ne}
if "exclude_missing_host" in args:
params["exclude_missing_host"] = bool(args.get("exclude_missing_host"))
for k in ("severity", "time_from", "time_to"):
if str(args.get(k) or "").strip():
params[k] = str(args.get(k)).strip()
return http_json("GET", "/v1/ume/alarms/aggregate", params=params)
def _run_ume_diagnostics(args: dict[str, Any]) -> dict[str, Any]:
@ -192,6 +204,8 @@ def _aggregate_ume_alarms_raw(args: dict[str, Any]) -> dict[str, Any]:
sv = str(v).strip()
if sv:
params[k] = sv
if "exclude_missing_host" in args:
params["exclude_missing_host"] = bool(args.get("exclude_missing_host"))
return http_json("GET", "/v1/ume/alarms/aggregate/raw", params=params)
@ -294,10 +308,14 @@ def _find_topology_paths(args: dict[str, Any]) -> dict[str, Any]:
return {"ok": False, "error": "exactly_one_of_from_ume_ne_id_or_from_managed_ne_id_required"}
if bool(to_uid) == bool(to_mid):
return {"ok": False, "error": "exactly_one_of_to_ume_ne_id_or_to_managed_ne_id_required"}
detail = str(args.get("detail") or "summary").strip().lower() or "summary"
if detail not in {"summary", "full"}:
detail = "summary"
body: dict[str, Any] = {
"max_paths": max(1, min(10, int(args.get("max_paths") or 3))),
"max_hops": max(1, min(12, int(args.get("max_hops") or 6))),
"layer": str(args.get("layer") or "physical").strip() or "physical",
"detail": detail,
}
if from_uid:
body["from_ume_ne_id"] = from_uid
@ -313,14 +331,21 @@ def _find_topology_paths(args: dict[str, Any]) -> dict[str, Any]:
HTTP_MCP_TOOLS: list[dict[str, Any]] = [
{
"name": "queryUmeAlarms",
"description": "Query UME current alarms (each row includes host_name). Supports severity/ne_id/keyword and pagination.",
"description": (
"Query UME current alarms (each row includes host_name). "
"Supports severity/ne_id/host_name/keyword, last_seen time_from/time_to, pagination. "
"Prefer host_name for display; ne_id is for filters only."
),
"inputSchema": {
"type": "object",
"properties": {
"severity": {"type": "string"},
"ne_id": {"type": "string"},
"ne_id": {"type": "string", "description": "Filter only; do not show UUID to users"},
"host_name": {"type": "string", "description": "Filter by NE host_name"},
"ne_name": {"type": "string", "description": "Legacy alias mapped to keyword"},
"keyword": {"type": "string"},
"time_from": {"type": "string", "description": "ISO time; filters last_seen_at >="},
"time_to": {"type": "string", "description": "ISO time; filters last_seen_at <="},
"page": {"type": "integer", "minimum": 1, "default": 1},
"page_size": {"type": "integer", "minimum": 1, "maximum": 500, "default": 50},
},
@ -330,12 +355,45 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [
},
{
"name": "aggregateUmeAlarms",
"description": "Aggregate UME current alarms (by_severity/by_ne).",
"inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False},
"description": (
"Aggregate UME current alarms (by_severity + top by_ne). "
"Optional severity filter (e.g. critical) for risk Top-N. "
"by_ne is capped by top_ne (default 50) and excludes missing host_name by default "
"(see by_ne_missing). meta.last_seen_min/max show data freshness. "
"For custom grouping use aggregateUmeAlarmsRaw."
),
"inputSchema": {
"type": "object",
"properties": {
"severity": {
"type": "string",
"description": "Optional perceived_severity filter (critical/major/minor/warning).",
},
"top_ne": {
"type": "integer",
"minimum": 0,
"maximum": 500,
"default": 50,
"description": "Max NE buckets to return (0 = all, capped at API 5000).",
},
"exclude_missing_host": {
"type": "boolean",
"default": True,
"description": "Omit (host_name missing) from by_ne ranking.",
},
"time_from": {"type": "string", "description": "ISO time; filters last_seen_at >="},
"time_to": {"type": "string", "description": "ISO time; filters last_seen_at <="},
},
"required": [],
"additionalProperties": False,
},
},
{
"name": "runUmeDiagnostics",
"description": "UME alarm diagnostics summary (severity distribution, top codes/NEs, protocol buckets).",
"description": (
"UME alarm diagnostics: severity, top_event_types, top_alarm_codes (UME alarmCode), "
"top_ne (excludes missing host), protocol buckets, and meta.last_seen_min/max freshness."
),
"inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False},
},
{
@ -391,7 +449,11 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [
},
{
"name": "aggregateUmeAlarmsRaw",
"description": "Dynamic aggregation on UME raw fields (group_by/group_by2); prefer alarm_host_name for NE grouping.",
"description": (
"Dynamic aggregation on UME raw fields (group_by/group_by2); prefer alarm_host_name. "
"When grouping by host fields, (host_name missing) is omitted by default "
"(see by_ne_missing)."
),
"inputSchema": {
"type": "object",
"properties": {
@ -404,6 +466,11 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [
"keyword": {"type": "string"},
"time_from": {"type": "string"},
"time_to": {"type": "string"},
"exclude_missing_host": {
"type": "boolean",
"default": True,
"description": "Omit missing host buckets when grouping by host/user_label fields.",
},
"limit": {"type": "integer", "minimum": 1, "maximum": 2000, "default": 200},
},
"required": ["group_by"],
@ -498,9 +565,10 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [
"name": "findTopologyPaths",
"description": (
"Find up to max_paths simple paths between two fabric nodes for troubleshooting. "
"For each endpoint provide exactly one of ume_ne_id (from UME alarms) or "
"For each endpoint provide exactly one of ume_ne_id (from UME alarm ne_id) or "
"managed_ne_id — resolved to fabric node internally. Returns shortest paths "
"first with node sequence + edge status (up/down)."
"first with compact label + node/edge summary (detail=summary default). "
"Use after critical alarms to correlate neighboring NEs before CLI login."
),
"inputSchema": {
"type": "object",
@ -512,6 +580,12 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [
"max_paths": {"type": "integer", "minimum": 1, "maximum": 10, "default": 3},
"max_hops": {"type": "integer", "minimum": 1, "maximum": 12, "default": 6},
"layer": {"type": "string", "default": "physical"},
"detail": {
"type": "string",
"enum": ["summary", "full"],
"default": "summary",
"description": "summary=compact ops fields; full=include attrs/coords.",
},
},
"required": [],
"additionalProperties": False,

View file

@ -40,6 +40,39 @@ def test_call_query_ume_alarms_forwards_http() -> None:
assert payload["ok"] is True
def test_call_aggregate_ume_alarms_forwards_top_ne() -> None:
with patch("netx_mcp.http_tools.http_json") as mock_http:
mock_http.return_value = {
"ok": True,
"data": {"total": 10, "by_severity": [], "by_ne": [], "by_ne_total": 3, "top_ne": 20},
}
out = call_http_tool("aggregateUmeAlarms", {"top_ne": 20, "severity": "critical"})
mock_http.assert_called_once_with(
"GET",
"/v1/ume/alarms/aggregate",
params={"top_ne": 20, "severity": "critical"},
)
payload = json.loads(out["content"][0]["text"])
assert payload["ok"] is True
assert payload["data"]["top_ne"] == 20
def test_call_find_topology_paths_defaults_summary_detail() -> None:
with patch("netx_mcp.http_tools.http_post_json") as mock_post:
mock_post.return_value = {"ok": True, "data": {"path_count": 1, "detail": "summary", "paths": []}}
out = call_http_tool(
"findTopologyPaths",
{"from_ume_ne_id": "a", "to_ume_ne_id": "b"},
)
mock_post.assert_called_once()
body = mock_post.call_args[0][1]
assert body["detail"] == "summary"
assert body["from_ume_ne_id"] == "a"
assert body["to_ume_ne_id"] == "b"
payload = json.loads(out["content"][0]["text"])
assert payload["ok"] is True
def test_call_get_ume_ne_requires_id() -> None:
out = call_http_tool("getUmeNe", {})
assert out.get("isError") is True

View file

@ -2547,9 +2547,24 @@ Management Addresses:
max_hops=6,
)
self.assertEqual(out["path_count"], 2)
self.assertEqual(out["detail"], "summary")
self.assertEqual(out["paths"][0]["hops"], 1)
self.assertEqual(out["paths"][1]["hops"], 2)
self.assertEqual(len(out["paths"][0]["nodes"]), 2)
self.assertNotIn("attrs", out["paths"][0]["nodes"][0])
self.assertIn("label", out["paths"][0])
self.assertIn("gi0/0", out["paths"][0]["label"].lower())
full = svc.find_fabric_paths(
self.db,
from_managed_ne_id=nes[0].id,
to_managed_ne_id=nes[1].id,
max_paths=1,
max_hops=6,
detail="full",
)
self.assertEqual(full["detail"], "full")
self.assertIn("attrs", full["paths"][0]["nodes"][0])
if __name__ == "__main__":