Wait indefinitely for UME topology dumps while the connection stays up.

Default topology read timeout to 0 (no deadline) and extend stale-running reap to 24h so long TopoNodes/TopologicalLinks pulls are not killed mid-stream.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-06 19:04:13 +08:00
parent 6622d4ce7a
commit cd7d1fc6e7
4 changed files with 32 additions and 13 deletions

View file

@ -32,8 +32,10 @@ NETX_UME_TOPO_NODES_PATH=/restconf/data/zte-resources-module:TopoNodes
NETX_UME_TOPOLOGICAL_LINKS_PATH=/restconf/data/zte-resources-module:TopologicalLinks
NETX_UME_SYNC_TOPOLOGY_AUTO_ENABLED=true
NETX_UME_SYNC_TOPOLOGY_EVERY_HOURS=24
# NETX_UME_TOPOLOGY_TIMEOUT_S=600
# NETX_UME_SYNC_TOPOLOGY_STALE_RUNNING_SEC=1800
# Topology dump: 0 = no read timeout (wait while TCP stays up). Set e.g. 3600 to cap.
# NETX_UME_TOPOLOGY_TIMEOUT_S=0
# Reap stuck topology jobs only after this many seconds (default 24h).
# NETX_UME_SYNC_TOPOLOGY_STALE_RUNNING_SEC=86400
NETX_UME_SYNC_ALARMS_CURRENT_ENABLED=true
NETX_UME_SYNC_ALARMS_CURRENT_INTERVAL_S=18000
NETX_UME_SYNC_ALARMS_CURRENT_SKIP_WHEN_WS=true

View file

@ -58,8 +58,10 @@ class Settings(BaseSettings):
ume_sync_inventory_every_hours: int = 48
ume_sync_topology_auto_enabled: bool = True
ume_sync_topology_every_hours: int = 24
ume_topology_timeout_s: float = 600.0
ume_sync_topology_stale_running_sec: int = 1800
# Topology REST dump: 0 = no read timeout (wait while the connection stays up).
ume_topology_timeout_s: float = 0.0
# Do not reap a still-running topology pull until this age (long UME dumps).
ume_sync_topology_stale_running_sec: int = 86400
ume_token_path: str = "/restconf/operations/zte-security:oauth_token"
ume_token_handshake_path: str = "/restconf/operations/zte-security:oauth_handshake"
ume_token_logout_path: str = "/restconf/operations/zte-security:oauth_token"

View file

@ -237,16 +237,30 @@ class UMEClient:
headers[self.auth_header] = token
return headers
def _client(self, *, timeout_s: float | None = None) -> httpx.Client:
def _client(
self,
*,
timeout_s: float | None = None,
timeout: httpx.Timeout | None = None,
) -> httpx.Client:
# Use explicit HTTPTransport to keep behavior consistent with onsite validation.
# In this mode, requests run over HTTP/1.1 and avoid HTTP/2 negotiation issues.
transport = httpx.HTTPTransport(verify=self.verify_tls, http2=False)
if timeout_s is None:
to: float | httpx.Timeout = self.timeout_s
if timeout is not None:
to: float | httpx.Timeout = timeout
elif timeout_s is None:
to = self.timeout_s
elif float(timeout_s) <= 0:
# No read/write deadline — keep waiting while the peer still streams.
to = httpx.Timeout(connect=30.0, read=None, write=None, pool=30.0)
else:
# Large topology dumps need a long read window; connect stays short.
read_s = max(3.0, float(timeout_s))
to = httpx.Timeout(connect=min(30.0, read_s), read=read_s, write=min(60.0, read_s), pool=30.0)
to = httpx.Timeout(
connect=min(30.0, read_s),
read=read_s,
write=min(60.0, read_s),
pool=30.0,
)
return httpx.Client(transport=transport, timeout=to)
@ -618,10 +632,11 @@ class UMEClient:
return rows, diag
def _topology_timeout_s(self) -> float:
"""Read timeout seconds for topology dumps; ``0`` = wait while connected."""
base = float(getattr(settings, "ume_topology_timeout_s", 0) or 0)
if base > 0:
return max(30.0, base)
return max(self.timeout_s, 600.0)
if base <= 0:
return 0.0
return max(30.0, base)
@staticmethod
def _extract_restconf_list(payload: dict[str, Any], *, containers: list[str], items: list[str]) -> list[dict[str, Any]]:

View file

@ -109,7 +109,7 @@ def _ume_ne_id_for_topo_node(*, node_type: str, name: str) -> str:
def _stale_running_sec() -> int:
return max(300, int(getattr(settings, "ume_sync_topology_stale_running_sec", 1800) or 1800))
return max(3600, int(getattr(settings, "ume_sync_topology_stale_running_sec", 86400) or 86400))
def fail_stale_topology_running_jobs(db: Session | None = None) -> int: