Harden LLDP collect ops: edge list, job reclaim, and schedule hygiene.

Add fabric link browsing, conservative job retention, hour-based intervals, multi-worker start locking, and trim canvas discover UI in favor of the LLDP page.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-02 03:58:03 +08:00
parent d05ba0f1f7
commit e7f97b7cf6
18 changed files with 1126 additions and 479 deletions

View file

@ -3,26 +3,36 @@
from __future__ import annotations
import unittest
from datetime import datetime, timedelta
from uuid import uuid4
from netx_api.db import Base, SessionLocal, engine
from netx_api.lldp_collect_schemas import LldpCollectPolicyUpdate
from netx_api.lldp_collect_service import (
build_discover_request,
ensure_policy,
get_dashboard,
has_running_job,
next_due_at,
update_policy,
)
from netx_api.models import LldpCollectPolicy
from netx_api.models import LldpCollectPolicy, ManagedNE, TopoDiscoverJob, TopoDiscoverJobItem
from netx_api.topology_service import prune_discover_jobs, reclaim_stale_discover_jobs
class LldpCollectTests(unittest.TestCase):
@classmethod
def setUpClass(cls) -> None:
Base.metadata.create_all(bind=engine)
from netx_api.topology_migrate import ensure_topology_schema
with engine.begin() as conn:
ensure_topology_schema(conn)
def setUp(self) -> None:
self.db = SessionLocal()
self.db.query(TopoDiscoverJobItem).delete()
self.db.query(TopoDiscoverJob).delete()
self.db.query(LldpCollectPolicy).delete()
self.db.commit()
@ -41,6 +51,7 @@ class LldpCollectTests(unittest.TestCase):
dash = get_dashboard(self.db)
self.assertFalse(dash.policy.enabled)
self.assertIsNone(dash.next_due_at)
self.assertEqual(dash.policy.history_keep, 30)
def test_policy_enable_updates(self) -> None:
ensure_policy(self.db)
@ -48,18 +59,189 @@ class LldpCollectTests(unittest.TestCase):
self.db,
LldpCollectPolicyUpdate(
enabled=True,
interval_days=2,
interval_hours=48,
concurrency=6,
scope_mode="all",
auto_add_unmatched=True,
history_keep=5,
),
)
self.assertTrue(out.enabled)
self.assertEqual(out.interval_hours, 48)
self.assertEqual(out.interval_days, 2)
self.assertEqual(out.concurrency, 6)
self.assertEqual(out.history_keep, 5)
due = next_due_at(self.db, ensure_policy(self.db))
self.assertIsNotNone(due)
def test_next_due_uses_hours_and_ignores_manual(self) -> None:
policy = ensure_policy(self.db)
policy.enabled = True
policy.interval_hours = 6
policy.interval_days = 1
self.db.commit()
now = datetime.utcnow()
sched = TopoDiscoverJob(
id=uuid4().hex,
scope="all_inventory",
trigger_mode="schedule",
status="done",
ended_at=now - timedelta(hours=1),
created_at=now - timedelta(hours=2),
updated_at=now - timedelta(hours=1),
)
manual = TopoDiscoverJob(
id=uuid4().hex,
scope="all_inventory",
trigger_mode="manual",
status="done",
ended_at=now,
created_at=now,
updated_at=now,
)
self.db.add(sched)
self.db.add(manual)
self.db.commit()
due = next_due_at(self.db, ensure_policy(self.db))
self.assertIsNotNone(due)
assert due is not None
self.assertEqual(due, sched.ended_at + timedelta(hours=6))
def test_start_discover_rejects_second_while_running(self) -> None:
from netx_api.topology_schemas import FabricDiscoverRequest
from netx_api.topology_service import start_discover_job
now = datetime.utcnow()
running = TopoDiscoverJob(
id=uuid4().hex,
scope="all_inventory",
trigger_mode="manual",
status="running",
created_at=now,
updated_at=now,
started_at=now,
)
self.db.add(running)
ensure_policy(self.db)
self.db.commit()
with self.assertRaises(Exception) as ctx:
start_discover_job(self.db, FabricDiscoverRequest(scope="ne_ids", ne_ids=["x"]))
detail = getattr(ctx.exception, "detail", str(ctx.exception))
self.assertEqual(detail, "lldp_collect_already_running")
def test_build_request_respects_source(self) -> None:
suffix = uuid4().hex[:8]
ne = ManagedNE(
id=f"m-{suffix}",
name=f"M-{suffix}",
ip_address=f"203.0.113.{(int(suffix[:2], 16) % 200) + 1}",
vendor="Cisco",
device_type="cisco_ios",
)
self.db.add(ne)
policy = ensure_policy(self.db)
policy.scope_mode = "selected"
# Same id string marked as ume should NOT resolve via managed path when building lists.
policy.selected_targets = [
{"source": "managed", "id": ne.id},
{"source": "ume", "id": f"ume-{suffix}"},
]
self.db.commit()
req = build_discover_request(ensure_policy(self.db))
self.assertEqual(req.managed_ne_ids, [ne.id])
self.assertEqual(req.ume_ne_ids, [f"ume-{suffix}"])
self.assertEqual(req.ne_ids, [])
self.db.delete(ne)
self.db.commit()
def test_prune_discover_jobs_keeps_newest(self) -> None:
now = datetime.utcnow()
ids: list[str] = []
for i in range(5):
jid = uuid4().hex
ids.append(jid)
self.db.add(
TopoDiscoverJob(
id=jid,
scope="all_inventory",
trigger_mode="manual",
status="done",
created_at=now - timedelta(minutes=5 - i),
updated_at=now - timedelta(minutes=5 - i),
ended_at=now - timedelta(minutes=5 - i),
)
)
self.db.add(
TopoDiscoverJobItem(
id=uuid4().hex,
job_id=jid,
ne_name=f"n{i}",
raw_preview="x" * 20,
created_at=now,
)
)
# Keep one open job — must survive prune.
open_id = uuid4().hex
self.db.add(
TopoDiscoverJob(
id=open_id,
scope="all_inventory",
trigger_mode="manual",
status="running",
created_at=now,
updated_at=now,
)
)
self.db.commit()
dropped = prune_discover_jobs(self.db, keep=2)
self.assertEqual(dropped, 3)
left = {r.id for r in self.db.query(TopoDiscoverJob).all()}
self.assertIn(open_id, left)
self.assertEqual(len(left), 3) # 2 finished + 1 running
def test_reclaim_force_all_open_on_startup(self) -> None:
now = datetime.utcnow()
job = TopoDiscoverJob(
id=uuid4().hex,
scope="all_inventory",
trigger_mode="schedule",
status="running",
total=10,
done=1,
created_at=now,
updated_at=now,
started_at=now,
)
self.db.add(job)
self.db.commit()
closed = reclaim_stale_discover_jobs(self.db, force_all_open=True)
self.assertEqual(closed, 1)
self.db.refresh(job)
self.assertEqual(job.status, "failed")
self.assertIn("stale_running_reset_on_startup", job.error or "")
self.assertIsNone(has_running_job(self.db))
def test_reclaim_running_by_stale_updated_at(self) -> None:
old = datetime.utcnow() - timedelta(hours=5)
job = TopoDiscoverJob(
id=uuid4().hex,
scope="ne_ids",
trigger_mode="manual",
status="running",
total=3,
done=0,
created_at=old,
updated_at=old,
started_at=old,
)
self.db.add(job)
self.db.commit()
closed = reclaim_stale_discover_jobs(self.db)
self.assertEqual(closed, 1)
self.db.refresh(job)
self.assertEqual(job.status, "failed")
self.assertIn("running_stale_timeout", job.error or "")
if __name__ == "__main__":
unittest.main()

View file

@ -87,6 +87,10 @@ class FabricTopologyTests(unittest.TestCase):
@classmethod
def setUpClass(cls) -> None:
Base.metadata.create_all(bind=engine)
from netx_api.topology_migrate import ensure_topology_schema
with engine.begin() as conn:
ensure_topology_schema(conn)
def setUp(self) -> None:
self.db = SessionLocal()
@ -482,6 +486,112 @@ Management Addresses:
self.db.commit()
return fa, fb, ne_a, ne_b
def test_merge_lldp_placeholder_into_real_inventory(self) -> None:
suffix = uuid4().hex[:8]
# Real inventory NE + fabric node.
real = ManagedNE(
id=f"real-{suffix}",
name=f"R1-{suffix}",
ip_address=f"198.51.100.{(int(suffix[:2], 16) % 80) + 10}",
vendor="Cisco",
device_type="cisco_ios",
)
self.db.add(real)
self.db.commit()
fr = svc.ensure_fabric_node_for_managed(self.db, real)
# LLDP placeholder with same hostname key + seen mgmt IP.
ph = svc.ensure_lldp_discovered_managed_ne(
self.db,
remote_name=f"R1-{suffix}",
remote_ip=real.ip_address,
)
fp = svc.ensure_fabric_node_for_managed(self.db, ph)
self.db.commit()
# Edge hanging off placeholder should retarget to real.
edge, _ = svc.upsert_fabric_edge(
self.db,
a_node_id=fr.id,
b_node_id=fp.id,
a_port="Gi0/0",
b_port="Gi0/1",
source="lldp",
)
# Need a third node so edge isn't self-loop after merge… actually A=real B=placeholder
# after absorb B→A becomes self-loop and edge is deleted. Use external peer.
peer_ne = ManagedNE(
id=f"peer-{suffix}",
name=f"P-{suffix}",
ip_address=f"198.51.100.{(int(suffix[2:4], 16) % 80) + 100}",
vendor="Cisco",
device_type="cisco_ios",
)
self.db.add(peer_ne)
self.db.commit()
fpeer = svc.ensure_fabric_node_for_managed(self.db, peer_ne)
edge2, _ = svc.upsert_fabric_edge(
self.db,
a_node_id=fp.id,
b_node_id=fpeer.id,
a_port="Gi1/0",
b_port="Gi1/1",
source="lldp",
)
self.db.commit()
edge2_id = edge2.id
out = svc.merge_duplicate_fabric_nodes(self.db)
self.assertGreaterEqual(out["merged"], 1)
self.assertGreaterEqual(out.get("placeholders_removed", 0), 1)
self.db.expire_all()
self.assertIsNone(self.db.get(TopoFabricNode, fp.id))
self.assertIsNone(self.db.get(ManagedNE, ph.id))
# Edge from placeholder→peer should now be real→peer.
moved = self.db.get(TopoFabricEdge, edge2_id)
self.assertIsNotNone(moved)
assert moved is not None
ends = {moved.a_node_id, moved.b_node_id}
self.assertEqual(ends, {fr.id, fpeer.id})
self.db.delete(real)
self.db.delete(peer_ne)
self.db.commit()
def test_list_fabric_edges_missing_filter_and_names(self) -> None:
suffix = uuid4().hex[:8]
fa, fb, ne_a, ne_b = self._pair_nodes(suffix)
edge, _ = svc.upsert_fabric_edge(
self.db,
a_node_id=fa.id,
b_node_id=fb.id,
a_port="Gi0/0",
b_port="Gi0/1",
source="lldp",
)
self.db.commit()
svc._apply_missing_and_purge(
self.db, scanned_ok={fa.id}, touched_edge_ids=set()
)
self.db.commit()
missing = svc.list_fabric_edges(self.db, status="missing", page=1, page_size=50)
self.assertGreaterEqual(missing["total"], 1)
hit = next(i for i in missing["items"] if i["id"] == edge.id)
self.assertEqual(hit["status"], "missing")
self.assertTrue(hit["a_name"] or hit["b_name"])
self.assertGreaterEqual(int((hit.get("attrs") or {}).get("miss_count") or 0), 1)
by_kw = svc.list_fabric_edges(
self.db, keyword=ne_a.name[:6], status="missing", page=1, page_size=50
)
self.assertTrue(any(i["id"] == edge.id for i in by_kw["items"]))
active = svc.list_fabric_edges(self.db, status="active", page=1, page_size=50)
self.assertFalse(any(i["id"] == edge.id for i in active["items"]))
self.db.delete(ne_a)
self.db.delete(ne_b)
self.db.commit()
def test_edge_missing_after_one_absent_cycle(self) -> None:
suffix = uuid4().hex[:8]
fa, fb, ne_a, ne_b = self._pair_nodes(suffix)