From 78d174232d7285f11103193c9a3e10023950b67e Mon Sep 17 00:00:00 2001 From: oliver Date: Mon, 10 Aug 2026 22:17:45 +0800 Subject: [PATCH] Improve UME ops triage APIs and slim fabric path payloads. Add severity/time filters, exclude missing hosts from aggregates, and default topology paths to compact summary fields for MCP-friendly alarm triage. Co-authored-by: Cursor --- netx_api/topology_fabric_nodes.py | 70 +++- netx_api/topology_router.py | 1 + netx_api/ume_alarms_router.py | 342 +++++++++++++++++-- packages/netx-mcp/src/netx_mcp/http_tools.py | 94 ++++- packages/netx-mcp/tests/test_mcp_http.py | 33 ++ tests/test_topology.py | 15 + 6 files changed, 505 insertions(+), 50 deletions(-) diff --git a/netx_api/topology_fabric_nodes.py b/netx_api/topology_fabric_nodes.py index c4cd238..0ce979e 100644 --- a/netx_api/topology_fabric_nodes.py +++ b/netx_api/topology_fabric_nodes.py @@ -468,11 +468,15 @@ def find_fabric_paths( max_paths: int = 3, max_hops: int = 6, layer: str = "physical", + detail: str = "summary", ) -> dict[str, Any]: """Find up to max_paths simple paths between two fabric nodes. Accepts ume_ne_id (from UME alarms) or managed_ne_id (from managed NE) — resolved to fabric_node_id internally so agents can use alarm ne_id directly. + + detail=summary (default): compact node/edge fields for ops/MCP. + detail=full: full FabricNodeOut/FabricEdgeOut payloads. """ from_uid = str(from_ume_ne_id or "").strip() from_mid = str(from_managed_ne_id or "").strip() @@ -502,6 +506,9 @@ def find_fabric_paths( max_paths = max(1, min(10, int(max_paths or 3))) max_hops = max(1, min(12, int(max_hops or 6))) layer_v = str(layer or "physical").strip() or "physical" + detail_v = str(detail or "summary").strip().lower() or "summary" + if detail_v not in {"summary", "full"}: + raise HTTPException(400, detail="invalid_detail_use_summary_or_full") # Lazy adjacency: only fetch edges for nodes the BFS actually expands # (avoids loading the entire fabric layer on large graphs). @@ -565,7 +572,29 @@ def find_fabric_paths( node_ids.add(e.b_node_id) node_map = _nodes_by_ids(db, node_ids) - def _path_nodes(edge_ids: list[str]) -> list[dict]: + def _node_summary(n: TopoFabricNode) -> dict[str, Any]: + return { + "name": n.name or "", + "ip": n.ip or "", + "ume_ne_id": n.ume_ne_id or "", + "managed_ne_id": n.managed_ne_id or "", + "vendor": n.vendor or "", + } + + def _edge_summary(e: TopoFabricEdge) -> dict[str, Any]: + a_node = node_map.get(e.a_node_id) + b_node = node_map.get(e.b_node_id) + return { + "a_name": (a_node.name if a_node else "") or "", + "b_name": (b_node.name if b_node else "") or "", + "a_ip": (a_node.ip if a_node else "") or "", + "b_ip": (b_node.ip if b_node else "") or "", + "a_port": e.a_port or "", + "b_port": e.b_port or "", + "status": _normalize_edge_status(e.status or "active"), + } + + def _path_node_ids(edge_ids: list[str]) -> list[str]: ids = [from_id] cur = from_id for eid in edge_ids: @@ -575,18 +604,53 @@ def find_fabric_paths( nxt = e.b_node_id if e.a_node_id == cur else e.a_node_id ids.append(nxt) cur = nxt - return [_node_out(node_map[nid]).model_dump() for nid in ids if nid in node_map] + return ids + + def _path_nodes(edge_ids: list[str]) -> list[dict]: + ids = _path_node_ids(edge_ids) + if detail_v == "full": + return [_node_out(node_map[nid]).model_dump() for nid in ids if nid in node_map] + return [_node_summary(node_map[nid]) for nid in ids if nid in node_map] + + def _path_edges(edge_ids: list[str]) -> list[dict]: + if detail_v == "full": + return [ + _edge_out(edge_map[eid], nodes_by_id=node_map).model_dump() + for eid in edge_ids + if eid in edge_map + ] + return [_edge_summary(edge_map[eid]) for eid in edge_ids if eid in edge_map] + + def _path_label(edge_ids: list[str]) -> str: + ids = _path_node_ids(edge_ids) + names = [(node_map[nid].name if nid in node_map else nid) or nid for nid in ids] + if not edge_ids: + return " -> ".join(names) + parts: list[str] = [names[0]] + for i, eid in enumerate(edge_ids): + e = edge_map.get(eid) + nxt = names[i + 1] if i + 1 < len(names) else "?" + if not e: + parts.append(f"-> {nxt}") + continue + # Port facing the next hop from current node orientation. + cur_id = ids[i] + port = e.a_port if e.a_node_id == cur_id else e.b_port + parts.append(f"-[{port or '?'}]-> {nxt}") + return " ".join(parts) return { "from_node_id": from_id, "to_node_id": to_id, "layer": layer_v, + "detail": detail_v, "path_count": len(found), "paths": [ { "hops": len(p), + "label": _path_label(p), "nodes": _path_nodes(p), - "edges": [_edge_out(edge_map[eid], nodes_by_id=node_map).model_dump() for eid in p if eid in edge_map], + "edges": _path_edges(p), } for p in found ], diff --git a/netx_api/topology_router.py b/netx_api/topology_router.py index d996f83..168cc86 100644 --- a/netx_api/topology_router.py +++ b/netx_api/topology_router.py @@ -160,6 +160,7 @@ def api_fabric_paths(body: dict[str, Any] = Body(...), db: Session = Depends(get max_paths=int(body.get("max_paths") or 3), max_hops=int(body.get("max_hops") or 6), layer=str(body.get("layer") or "physical"), + detail=str(body.get("detail") or "summary"), ) diff --git a/netx_api/ume_alarms_router.py b/netx_api/ume_alarms_router.py index 702fa07..e4757b7 100644 --- a/netx_api/ume_alarms_router.py +++ b/netx_api/ume_alarms_router.py @@ -6,7 +6,7 @@ from datetime import datetime, timezone from typing import Any from fastapi import APIRouter, Depends, HTTPException, Query -from sqlalchemy import or_ +from sqlalchemy import func, or_ from sqlalchemy.orm import Session from .config import settings @@ -78,6 +78,8 @@ def ume_list_alarms( ne_id: str | None = Query(default=None), host_name: str | None = Query(default=None), keyword: str | None = Query(default=None), + time_from: str | None = Query(default=None, description="Filter by last_seen_at >="), + time_to: str | None = Query(default=None, description="Filter by last_seen_at <="), page: int = Query(default=1, ge=1), page_size: int = Query(default=50, ge=1, le=500), db: Session = Depends(get_db), @@ -109,6 +111,12 @@ def ume_list_alarms( | UmeInventoryNE.ip_address.contains(kw) | UmeInventoryNE.host_name.contains(kw) ) + dt_from = _parse_time(time_from) + dt_to = _parse_time(time_to) + if dt_from: + stmt = stmt.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None)) + if dt_to: + stmt = stmt.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None)) total = int(stmt.count()) rows = ( stmt.order_by( @@ -139,7 +147,13 @@ def ume_list_alarms( } for alarm, ne in rows ] - return {"total": total, "page": page, "page_size": page_size, "items": items} + return { + "total": total, + "page": page, + "page_size": page_size, + "meta": {"time_filter_field": "last_seen_at"}, + "items": items, + } @router.get("/v1/ume/alarms/fields") @@ -309,6 +323,29 @@ def ume_alarms_raw( } +_HOST_MISSING_LABEL = "(host_name missing)" +_HOST_GROUP_FIELDS = frozenset({"alarm_host_name", "ne_host_name", "ne_user_label"}) + + +def _normalize_ne_bucket_key(raw: Any) -> str: + s = str(raw or "").strip() + if not s or s.lower() in {"unknown", "none", "null"}: + return _HOST_MISSING_LABEL + # Bare UUID or ME{uuid} + if len(s) >= 32 and s.count("-") >= 4 and " " not in s: + return _HOST_MISSING_LABEL + if s.startswith("ME{") and s.endswith("}"): + return _HOST_MISSING_LABEL + return s + + +def _raw_group_bucket_key(field: str, value: str) -> str: + if field in _HOST_GROUP_FIELDS: + return _normalize_ne_bucket_key(value) + s = str(value or "").strip() + return s if s else "(empty)" + + @router.get("/v1/ume/alarms/aggregate/raw") def ume_alarms_aggregate_raw( group_by: str = Query(default="alarm_perceived_severity"), @@ -320,6 +357,10 @@ def ume_alarms_aggregate_raw( keyword: str | None = Query(default=None), time_from: str | None = Query(default=None), time_to: str | None = Query(default=None), + exclude_missing_host: bool = Query( + default=True, + description="When grouping by host/user_label fields, omit (host_name missing) from buckets.", + ), limit: int = Query(default=200, ge=1, le=2000), db: Session = Depends(get_db), ) -> dict[str, Any]: @@ -362,17 +403,30 @@ def ume_alarms_aggregate_raw( stmt = stmt.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None)) rows = stmt.order_by(UmeAlarmCurrent.last_seen_at.desc()).all() - counts: dict[tuple[str, str], int] = {} + merged: dict[tuple[str, str], int] = {} + by_ne_missing = 0 for alarm, ne in rows: - k1 = _extract_ume_raw_group_field(alarm, ne, g1) - k2 = _extract_ume_raw_group_field(alarm, ne, g2) if g2 else "" - kk = (k1, k2) - counts[kk] = int(counts.get(kk, 0)) + 1 - buckets = sorted(counts.items(), key=lambda x: x[1], reverse=True)[: int(limit)] + nk1 = _raw_group_bucket_key(g1, _extract_ume_raw_group_field(alarm, ne, g1)) + nk2 = _raw_group_bucket_key(g2, _extract_ume_raw_group_field(alarm, ne, g2)) if g2 else "" + missing_hit = False + if g1 in _HOST_GROUP_FIELDS and nk1 == _HOST_MISSING_LABEL: + missing_hit = True + if g2 and g2 in _HOST_GROUP_FIELDS and nk2 == _HOST_MISSING_LABEL: + missing_hit = True + if missing_hit: + by_ne_missing += 1 + if exclude_missing_host: + continue + kk = (nk1, nk2) + merged[kk] = int(merged.get(kk, 0)) + 1 + buckets = sorted(merged.items(), key=lambda x: x[1], reverse=True)[: int(limit)] + return { "total": len(rows), "group_by": g1, "group_by2": g2 or None, + "by_ne_missing": by_ne_missing, + "exclude_missing_host": bool(exclude_missing_host), "meta": { "available_fields": sorted(selectable_fields), "group_by_allowed": sorted(selectable_fields), @@ -387,60 +441,274 @@ def ume_alarms_aggregate_raw( }, "time_filter_field": "last_seen_at", "limit": int(limit), + "host_missing_label": _HOST_MISSING_LABEL, }, "buckets": [ - {"key": k1, "key2": (k2 if g2 else None), "count": int(v)} + { + "key": k1, + "key2": (k2 if g2 else None), + "count": int(v), + } for (k1, k2), v in buckets ], } @router.get("/v1/ume/alarms/aggregate") -def ume_alarms_aggregate(db: Session = Depends(get_db)) -> dict[str, Any]: - rows = db.query(UmeAlarmCurrent, UmeInventoryNE).outerjoin( - UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id - ).all() - by_severity = _aggregate_rows(rows, lambda x: x[0].perceived_severity) - by_ne = _aggregate_rows(rows, lambda x: _ume_alarm_ne_group_key(x[0], x[1])) - return {"total": len(rows), "by_severity": by_severity, "by_ne": by_ne} +def ume_alarms_aggregate( + top_ne: int = Query( + default=50, + ge=0, + le=5000, + description="Max NE buckets to return (0 = all). Severity buckets are always complete.", + ), + exclude_missing_host: bool = Query( + default=True, + description="When true, omit (host_name missing) from by_ne ranking (count still in by_ne_missing).", + ), + severity: str | None = Query( + default=None, + description="Optional perceived_severity filter (e.g. critical) for top-NE ranking.", + ), + time_from: str | None = Query(default=None, description="Filter by last_seen_at >="), + time_to: str | None = Query(default=None, description="Filter by last_seen_at <="), + db: Session = Depends(get_db), +) -> dict[str, Any]: + """Aggregate current alarms by severity and NE (SQL group-by; top_ne capped).""" + dt_from = _parse_time(time_from) + dt_to = _parse_time(time_to) + sev = str(severity or "").strip() or None + + base = db.query(UmeAlarmCurrent) + if sev: + base = base.filter(UmeAlarmCurrent.perceived_severity == sev) + if dt_from: + base = base.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None)) + if dt_to: + base = base.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None)) + # Prefer .count() — with_entities(func.count()) without select_from can return 1. + total = int(base.count()) + + sev_q = db.query(UmeAlarmCurrent.perceived_severity, func.count()) + if sev: + sev_q = sev_q.filter(UmeAlarmCurrent.perceived_severity == sev) + if dt_from: + sev_q = sev_q.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None)) + if dt_to: + sev_q = sev_q.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None)) + sev_rows = ( + sev_q.group_by(UmeAlarmCurrent.perceived_severity).order_by(func.count().desc()).all() + ) + by_severity = [ + {"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), "count": int(v)} + for k, v in sev_rows + ] + + # Display key: host_name / inventory host / user_label only (never bare UUID). + ne_key = func.coalesce( + func.nullif(func.trim(UmeAlarmCurrent.host_name), ""), + func.nullif(func.trim(UmeInventoryNE.host_name), ""), + func.nullif(func.trim(UmeInventoryNE.user_label), ""), + _HOST_MISSING_LABEL, + ) + ne_q = ( + db.query(ne_key.label("ne_key"), func.count().label("cnt")) + .select_from(UmeAlarmCurrent) + .outerjoin(UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id) + ) + if sev: + ne_q = ne_q.filter(UmeAlarmCurrent.perceived_severity == sev) + if dt_from: + ne_q = ne_q.filter(UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None)) + if dt_to: + ne_q = ne_q.filter(UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None)) + ne_q = ne_q.group_by(ne_key).order_by(func.count().desc()) + + all_ne_rows = ne_q.all() + by_ne_missing = 0 + named_counts: dict[str, int] = {} + for k, v in all_ne_rows: + label = _normalize_ne_bucket_key(k) + cnt = int(v) + if label == _HOST_MISSING_LABEL: + by_ne_missing += cnt + else: + named_counts[label] = int(named_counts.get(label, 0)) + cnt + named_rows = sorted(named_counts.items(), key=lambda kv: kv[1], reverse=True) + by_ne_total = len(named_rows) + (1 if by_ne_missing else 0) + ranked = list(named_rows) + if not exclude_missing_host and by_ne_missing: + ranked.append((_HOST_MISSING_LABEL, by_ne_missing)) + ranked.sort(key=lambda kv: int(kv[1]), reverse=True) + if top_ne > 0: + ranked = ranked[: int(top_ne)] + by_ne = [{"key": str(k), "count": int(v)} for k, v in ranked] + + seen_bounds = db.query( + func.min(UmeAlarmCurrent.last_seen_at), + func.max(UmeAlarmCurrent.last_seen_at), + ) + if sev: + seen_bounds = seen_bounds.filter(UmeAlarmCurrent.perceived_severity == sev) + if dt_from: + seen_bounds = seen_bounds.filter( + UmeAlarmCurrent.last_seen_at >= dt_from.replace(tzinfo=None) + ) + if dt_to: + seen_bounds = seen_bounds.filter( + UmeAlarmCurrent.last_seen_at <= dt_to.replace(tzinfo=None) + ) + min_seen, max_seen = seen_bounds.one() + + return { + "total": total, + "by_severity": by_severity, + "by_ne": by_ne, + "by_ne_total": by_ne_total, + "by_ne_missing": by_ne_missing, + "top_ne": int(top_ne), + "exclude_missing_host": bool(exclude_missing_host), + "severity": sev, + "meta": { + "time_filter_field": "last_seen_at", + "last_seen_min": (_ensure_utc(min_seen).isoformat() if min_seen else None), + "last_seen_max": (_ensure_utc(max_seen).isoformat() if max_seen else None), + }, + } @router.get("/v1/ume/diagnostics") def ume_diagnostics( lang: str | None = Query(default=None), + top_n: int = Query(default=10, ge=1, le=50), db: Session = Depends(get_db), ) -> dict[str, Any]: - rows = db.query(UmeAlarmCurrent, UmeInventoryNE).outerjoin( - UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id - ).all() - by_severity = _aggregate_rows(rows, lambda x: x[0].perceived_severity) - by_alarm_code = _aggregate_rows(rows, lambda x: x[0].event_type)[:10] - by_ne = _aggregate_rows(rows, lambda x: _ume_alarm_ne_group_key(x[0], x[1]))[:10] + """Fast SQL diagnostics; top_ne excludes missing host_name; includes data freshness.""" + total = int(db.query(func.count()).select_from(UmeAlarmCurrent).scalar() or 0) + min_seen, max_seen = db.query( + func.min(UmeAlarmCurrent.last_seen_at), + func.max(UmeAlarmCurrent.last_seen_at), + ).one() + sev_rows = ( + db.query(UmeAlarmCurrent.perceived_severity, func.count()) + .group_by(UmeAlarmCurrent.perceived_severity) + .order_by(func.count().desc()) + .all() + ) + by_severity = [ + {"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), "count": int(v)} + for k, v in sev_rows + ] + + event_rows = ( + db.query(UmeAlarmCurrent.event_type, func.count()) + .group_by(UmeAlarmCurrent.event_type) + .order_by(func.count().desc()) + .limit(top_n) + .all() + ) + top_event_types = [ + {"key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), "count": int(v)} + for k, v in event_rows + ] + + # Prefer real UME alarmCode from raw_json when present (Postgres JSON text). + top_alarm_codes: list[dict[str, Any]] = [] + try: + from sqlalchemy import cast + from sqlalchemy.dialects.postgresql import JSON + + code_expr = func.coalesce( + func.nullif( + func.json_extract_path_text(cast(UmeAlarmCurrent.raw_json, JSON), "alarmCode"), + "", + ), + "(none)", + ) + code_rows = ( + db.query(code_expr.label("code"), func.count()) + .group_by(code_expr) + .order_by(func.count().desc()) + .limit(top_n) + .all() + ) + top_alarm_codes = [{"key": str(k), "count": int(v)} for k, v in code_rows] + except Exception: + # SQLite / non-JSON: fall back to native_probable_cause tops. + cause_rows = ( + db.query(UmeAlarmCurrent.native_probable_cause, func.count()) + .group_by(UmeAlarmCurrent.native_probable_cause) + .order_by(func.count().desc()) + .limit(top_n) + .all() + ) + top_alarm_codes = [ + { + "key": (str(k).strip() if k is not None and str(k).strip() else "unknown"), + "count": int(v), + } + for k, v in cause_rows + ] + + ne_key = func.coalesce( + func.nullif(func.trim(UmeAlarmCurrent.host_name), ""), + func.nullif(func.trim(UmeInventoryNE.host_name), ""), + func.nullif(func.trim(UmeInventoryNE.user_label), ""), + _HOST_MISSING_LABEL, + ) + ne_rows = ( + db.query(ne_key.label("ne_key"), func.count().label("cnt")) + .select_from(UmeAlarmCurrent) + .outerjoin(UmeInventoryNE, UmeAlarmCurrent.ne_id == UmeInventoryNE.ne_id) + .group_by(ne_key) + .order_by(func.count().desc()) + .all() + ) + by_ne_missing = 0 + named_counts: dict[str, int] = {} + for k, v in ne_rows: + label = _normalize_ne_bucket_key(k) + cnt = int(v) + if label == _HOST_MISSING_LABEL: + by_ne_missing += cnt + continue + named_counts[label] = int(named_counts.get(label, 0)) + cnt + top_ne = [ + {"key": label, "count": cnt} + for label, cnt in sorted(named_counts.items(), key=lambda kv: kv[1], reverse=True)[:top_n] + ] + + # Protocol buckets still need text classify; stream rows lightly (cause+event only). lang_norm = _normalize_netx_lang(lang) proto_counts: dict[str, int] = {} - for alarm, ne in rows: - blob = " | ".join( - [ - str(alarm.event_type or ""), - str(alarm.native_probable_cause or ""), - str(alarm.object_name or ""), - str(ne.ne_name if ne else ""), - str(ne.user_label if ne else ""), - str(ne.ip_address if ne else ""), - ] - ) + light = db.query( + UmeAlarmCurrent.event_type, + UmeAlarmCurrent.native_probable_cause, + UmeAlarmCurrent.object_name, + ).yield_per(2000) + for event_type, cause, obj in light: + blob = " | ".join([str(event_type or ""), str(cause or ""), str(obj or "")]) bucket = _protocol_bucket_label(blob, lang=lang_norm) proto_counts[bucket] = int(proto_counts.get(bucket, 0)) + 1 - protocol_summary = sorted(proto_counts.items(), key=lambda x: x[1], reverse=True)[:10] + protocol_summary = sorted(proto_counts.items(), key=lambda x: x[1], reverse=True)[:top_n] return { "source": "ume_alarms_current", - "total_alarms": len(rows), + "total_alarms": total, "severity_summary": by_severity, - "top_alarm_codes": by_alarm_code, - "top_ne": by_ne, + "top_event_types": top_event_types, + # Backward-compatible alias — historically event_type, now real alarmCode when possible. + "top_alarm_codes": top_alarm_codes, + "top_ne": top_ne, + "by_ne_missing": by_ne_missing, "protocol_summary": [{"key": k, "count": v} for k, v in protocol_summary], + "meta": { + "last_seen_min": (_ensure_utc(min_seen).isoformat() if min_seen else None), + "last_seen_max": (_ensure_utc(max_seen).isoformat() if max_seen else None), + "time_filter_field": "last_seen_at", + "host_missing_label": _HOST_MISSING_LABEL, + }, } diff --git a/packages/netx-mcp/src/netx_mcp/http_tools.py b/packages/netx-mcp/src/netx_mcp/http_tools.py index 8954a61..33edb6c 100644 --- a/packages/netx-mcp/src/netx_mcp/http_tools.py +++ b/packages/netx-mcp/src/netx_mcp/http_tools.py @@ -123,12 +123,24 @@ def _query_ume_alarms(args: dict[str, Any]) -> dict[str, Any]: params["keyword"] = ne_name if str(args.get("ne_id") or "").strip(): params["ne_id"] = str(args.get("ne_id")).strip() + if str(args.get("host_name") or "").strip(): + params["host_name"] = str(args.get("host_name")).strip() + for k in ("time_from", "time_to"): + if str(args.get(k) or "").strip(): + params[k] = str(args.get(k)).strip() return http_json("GET", "/v1/ume/alarms", params=params) def _aggregate_ume_alarms(args: dict[str, Any]) -> dict[str, Any]: - _ = args - return http_json("GET", "/v1/ume/alarms/aggregate", params=None) + # Default top 50 named NEs — missing host buckets reported separately. + top_ne = max(0, min(500, int(args.get("top_ne") if args.get("top_ne") is not None else 50))) + params: dict[str, Any] = {"top_ne": top_ne} + if "exclude_missing_host" in args: + params["exclude_missing_host"] = bool(args.get("exclude_missing_host")) + for k in ("severity", "time_from", "time_to"): + if str(args.get(k) or "").strip(): + params[k] = str(args.get(k)).strip() + return http_json("GET", "/v1/ume/alarms/aggregate", params=params) def _run_ume_diagnostics(args: dict[str, Any]) -> dict[str, Any]: @@ -192,6 +204,8 @@ def _aggregate_ume_alarms_raw(args: dict[str, Any]) -> dict[str, Any]: sv = str(v).strip() if sv: params[k] = sv + if "exclude_missing_host" in args: + params["exclude_missing_host"] = bool(args.get("exclude_missing_host")) return http_json("GET", "/v1/ume/alarms/aggregate/raw", params=params) @@ -294,10 +308,14 @@ def _find_topology_paths(args: dict[str, Any]) -> dict[str, Any]: return {"ok": False, "error": "exactly_one_of_from_ume_ne_id_or_from_managed_ne_id_required"} if bool(to_uid) == bool(to_mid): return {"ok": False, "error": "exactly_one_of_to_ume_ne_id_or_to_managed_ne_id_required"} + detail = str(args.get("detail") or "summary").strip().lower() or "summary" + if detail not in {"summary", "full"}: + detail = "summary" body: dict[str, Any] = { "max_paths": max(1, min(10, int(args.get("max_paths") or 3))), "max_hops": max(1, min(12, int(args.get("max_hops") or 6))), "layer": str(args.get("layer") or "physical").strip() or "physical", + "detail": detail, } if from_uid: body["from_ume_ne_id"] = from_uid @@ -313,14 +331,21 @@ def _find_topology_paths(args: dict[str, Any]) -> dict[str, Any]: HTTP_MCP_TOOLS: list[dict[str, Any]] = [ { "name": "queryUmeAlarms", - "description": "Query UME current alarms (each row includes host_name). Supports severity/ne_id/keyword and pagination.", + "description": ( + "Query UME current alarms (each row includes host_name). " + "Supports severity/ne_id/host_name/keyword, last_seen time_from/time_to, pagination. " + "Prefer host_name for display; ne_id is for filters only." + ), "inputSchema": { "type": "object", "properties": { "severity": {"type": "string"}, - "ne_id": {"type": "string"}, + "ne_id": {"type": "string", "description": "Filter only; do not show UUID to users"}, + "host_name": {"type": "string", "description": "Filter by NE host_name"}, "ne_name": {"type": "string", "description": "Legacy alias mapped to keyword"}, "keyword": {"type": "string"}, + "time_from": {"type": "string", "description": "ISO time; filters last_seen_at >="}, + "time_to": {"type": "string", "description": "ISO time; filters last_seen_at <="}, "page": {"type": "integer", "minimum": 1, "default": 1}, "page_size": {"type": "integer", "minimum": 1, "maximum": 500, "default": 50}, }, @@ -330,12 +355,45 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [ }, { "name": "aggregateUmeAlarms", - "description": "Aggregate UME current alarms (by_severity/by_ne).", - "inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False}, + "description": ( + "Aggregate UME current alarms (by_severity + top by_ne). " + "Optional severity filter (e.g. critical) for risk Top-N. " + "by_ne is capped by top_ne (default 50) and excludes missing host_name by default " + "(see by_ne_missing). meta.last_seen_min/max show data freshness. " + "For custom grouping use aggregateUmeAlarmsRaw." + ), + "inputSchema": { + "type": "object", + "properties": { + "severity": { + "type": "string", + "description": "Optional perceived_severity filter (critical/major/minor/warning).", + }, + "top_ne": { + "type": "integer", + "minimum": 0, + "maximum": 500, + "default": 50, + "description": "Max NE buckets to return (0 = all, capped at API 5000).", + }, + "exclude_missing_host": { + "type": "boolean", + "default": True, + "description": "Omit (host_name missing) from by_ne ranking.", + }, + "time_from": {"type": "string", "description": "ISO time; filters last_seen_at >="}, + "time_to": {"type": "string", "description": "ISO time; filters last_seen_at <="}, + }, + "required": [], + "additionalProperties": False, + }, }, { "name": "runUmeDiagnostics", - "description": "UME alarm diagnostics summary (severity distribution, top codes/NEs, protocol buckets).", + "description": ( + "UME alarm diagnostics: severity, top_event_types, top_alarm_codes (UME alarmCode), " + "top_ne (excludes missing host), protocol buckets, and meta.last_seen_min/max freshness." + ), "inputSchema": {"type": "object", "properties": {}, "required": [], "additionalProperties": False}, }, { @@ -391,7 +449,11 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [ }, { "name": "aggregateUmeAlarmsRaw", - "description": "Dynamic aggregation on UME raw fields (group_by/group_by2); prefer alarm_host_name for NE grouping.", + "description": ( + "Dynamic aggregation on UME raw fields (group_by/group_by2); prefer alarm_host_name. " + "When grouping by host fields, (host_name missing) is omitted by default " + "(see by_ne_missing)." + ), "inputSchema": { "type": "object", "properties": { @@ -404,6 +466,11 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [ "keyword": {"type": "string"}, "time_from": {"type": "string"}, "time_to": {"type": "string"}, + "exclude_missing_host": { + "type": "boolean", + "default": True, + "description": "Omit missing host buckets when grouping by host/user_label fields.", + }, "limit": {"type": "integer", "minimum": 1, "maximum": 2000, "default": 200}, }, "required": ["group_by"], @@ -498,9 +565,10 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [ "name": "findTopologyPaths", "description": ( "Find up to max_paths simple paths between two fabric nodes for troubleshooting. " - "For each endpoint provide exactly one of ume_ne_id (from UME alarms) or " + "For each endpoint provide exactly one of ume_ne_id (from UME alarm ne_id) or " "managed_ne_id — resolved to fabric node internally. Returns shortest paths " - "first with node sequence + edge status (up/down)." + "first with compact label + node/edge summary (detail=summary default). " + "Use after critical alarms to correlate neighboring NEs before CLI login." ), "inputSchema": { "type": "object", @@ -512,6 +580,12 @@ HTTP_MCP_TOOLS: list[dict[str, Any]] = [ "max_paths": {"type": "integer", "minimum": 1, "maximum": 10, "default": 3}, "max_hops": {"type": "integer", "minimum": 1, "maximum": 12, "default": 6}, "layer": {"type": "string", "default": "physical"}, + "detail": { + "type": "string", + "enum": ["summary", "full"], + "default": "summary", + "description": "summary=compact ops fields; full=include attrs/coords.", + }, }, "required": [], "additionalProperties": False, diff --git a/packages/netx-mcp/tests/test_mcp_http.py b/packages/netx-mcp/tests/test_mcp_http.py index 6a881f3..c70b7b0 100644 --- a/packages/netx-mcp/tests/test_mcp_http.py +++ b/packages/netx-mcp/tests/test_mcp_http.py @@ -40,6 +40,39 @@ def test_call_query_ume_alarms_forwards_http() -> None: assert payload["ok"] is True +def test_call_aggregate_ume_alarms_forwards_top_ne() -> None: + with patch("netx_mcp.http_tools.http_json") as mock_http: + mock_http.return_value = { + "ok": True, + "data": {"total": 10, "by_severity": [], "by_ne": [], "by_ne_total": 3, "top_ne": 20}, + } + out = call_http_tool("aggregateUmeAlarms", {"top_ne": 20, "severity": "critical"}) + mock_http.assert_called_once_with( + "GET", + "/v1/ume/alarms/aggregate", + params={"top_ne": 20, "severity": "critical"}, + ) + payload = json.loads(out["content"][0]["text"]) + assert payload["ok"] is True + assert payload["data"]["top_ne"] == 20 + + +def test_call_find_topology_paths_defaults_summary_detail() -> None: + with patch("netx_mcp.http_tools.http_post_json") as mock_post: + mock_post.return_value = {"ok": True, "data": {"path_count": 1, "detail": "summary", "paths": []}} + out = call_http_tool( + "findTopologyPaths", + {"from_ume_ne_id": "a", "to_ume_ne_id": "b"}, + ) + mock_post.assert_called_once() + body = mock_post.call_args[0][1] + assert body["detail"] == "summary" + assert body["from_ume_ne_id"] == "a" + assert body["to_ume_ne_id"] == "b" + payload = json.loads(out["content"][0]["text"]) + assert payload["ok"] is True + + def test_call_get_ume_ne_requires_id() -> None: out = call_http_tool("getUmeNe", {}) assert out.get("isError") is True diff --git a/tests/test_topology.py b/tests/test_topology.py index 8557cc7..ecd3e41 100644 --- a/tests/test_topology.py +++ b/tests/test_topology.py @@ -2547,9 +2547,24 @@ Management Addresses: max_hops=6, ) self.assertEqual(out["path_count"], 2) + self.assertEqual(out["detail"], "summary") self.assertEqual(out["paths"][0]["hops"], 1) self.assertEqual(out["paths"][1]["hops"], 2) self.assertEqual(len(out["paths"][0]["nodes"]), 2) + self.assertNotIn("attrs", out["paths"][0]["nodes"][0]) + self.assertIn("label", out["paths"][0]) + self.assertIn("gi0/0", out["paths"][0]["label"].lower()) + + full = svc.find_fabric_paths( + self.db, + from_managed_ne_id=nes[0].id, + to_managed_ne_id=nes[1].id, + max_paths=1, + max_hops=6, + detail="full", + ) + self.assertEqual(full["detail"], "full") + self.assertIn("attrs", full["paths"][0]["nodes"][0]) if __name__ == "__main__":