diff --git a/.env.example b/.env.example index 0a7b75e..41ee727 100644 --- a/.env.example +++ b/.env.example @@ -32,8 +32,10 @@ NETX_UME_TOPO_NODES_PATH=/restconf/data/zte-resources-module:TopoNodes NETX_UME_TOPOLOGICAL_LINKS_PATH=/restconf/data/zte-resources-module:TopologicalLinks NETX_UME_SYNC_TOPOLOGY_AUTO_ENABLED=true NETX_UME_SYNC_TOPOLOGY_EVERY_HOURS=24 -# NETX_UME_TOPOLOGY_TIMEOUT_S=600 -# NETX_UME_SYNC_TOPOLOGY_STALE_RUNNING_SEC=1800 +# Topology dump: 0 = no read timeout (wait while TCP stays up). Set e.g. 3600 to cap. +# NETX_UME_TOPOLOGY_TIMEOUT_S=0 +# Reap stuck topology jobs only after this many seconds (default 24h). +# NETX_UME_SYNC_TOPOLOGY_STALE_RUNNING_SEC=86400 NETX_UME_SYNC_ALARMS_CURRENT_ENABLED=true NETX_UME_SYNC_ALARMS_CURRENT_INTERVAL_S=18000 NETX_UME_SYNC_ALARMS_CURRENT_SKIP_WHEN_WS=true diff --git a/netx_api/config.py b/netx_api/config.py index d814a50..83c95b6 100644 --- a/netx_api/config.py +++ b/netx_api/config.py @@ -58,8 +58,10 @@ class Settings(BaseSettings): ume_sync_inventory_every_hours: int = 48 ume_sync_topology_auto_enabled: bool = True ume_sync_topology_every_hours: int = 24 - ume_topology_timeout_s: float = 600.0 - ume_sync_topology_stale_running_sec: int = 1800 + # Topology REST dump: 0 = no read timeout (wait while the connection stays up). + ume_topology_timeout_s: float = 0.0 + # Do not reap a still-running topology pull until this age (long UME dumps). + ume_sync_topology_stale_running_sec: int = 86400 ume_token_path: str = "/restconf/operations/zte-security:oauth_token" ume_token_handshake_path: str = "/restconf/operations/zte-security:oauth_handshake" ume_token_logout_path: str = "/restconf/operations/zte-security:oauth_token" diff --git a/netx_api/ume_client.py b/netx_api/ume_client.py index 799fc4c..36bc7cb 100644 --- a/netx_api/ume_client.py +++ b/netx_api/ume_client.py @@ -237,16 +237,30 @@ class UMEClient: headers[self.auth_header] = token return headers - def _client(self, *, timeout_s: float | None = None) -> httpx.Client: + def _client( + self, + *, + timeout_s: float | None = None, + timeout: httpx.Timeout | None = None, + ) -> httpx.Client: # Use explicit HTTPTransport to keep behavior consistent with onsite validation. # In this mode, requests run over HTTP/1.1 and avoid HTTP/2 negotiation issues. transport = httpx.HTTPTransport(verify=self.verify_tls, http2=False) - if timeout_s is None: - to: float | httpx.Timeout = self.timeout_s + if timeout is not None: + to: float | httpx.Timeout = timeout + elif timeout_s is None: + to = self.timeout_s + elif float(timeout_s) <= 0: + # No read/write deadline — keep waiting while the peer still streams. + to = httpx.Timeout(connect=30.0, read=None, write=None, pool=30.0) else: - # Large topology dumps need a long read window; connect stays short. read_s = max(3.0, float(timeout_s)) - to = httpx.Timeout(connect=min(30.0, read_s), read=read_s, write=min(60.0, read_s), pool=30.0) + to = httpx.Timeout( + connect=min(30.0, read_s), + read=read_s, + write=min(60.0, read_s), + pool=30.0, + ) return httpx.Client(transport=transport, timeout=to) @@ -618,10 +632,11 @@ class UMEClient: return rows, diag def _topology_timeout_s(self) -> float: + """Read timeout seconds for topology dumps; ``0`` = wait while connected.""" base = float(getattr(settings, "ume_topology_timeout_s", 0) or 0) - if base > 0: - return max(30.0, base) - return max(self.timeout_s, 600.0) + if base <= 0: + return 0.0 + return max(30.0, base) @staticmethod def _extract_restconf_list(payload: dict[str, Any], *, containers: list[str], items: list[str]) -> list[dict[str, Any]]: diff --git a/netx_api/ume_sync_topology.py b/netx_api/ume_sync_topology.py index 9fa57db..ae62f6a 100644 --- a/netx_api/ume_sync_topology.py +++ b/netx_api/ume_sync_topology.py @@ -109,7 +109,7 @@ def _ume_ne_id_for_topo_node(*, node_type: str, name: str) -> str: def _stale_running_sec() -> int: - return max(300, int(getattr(settings, "ume_sync_topology_stale_running_sec", 1800) or 1800)) + return max(3600, int(getattr(settings, "ume_sync_topology_stale_running_sec", 86400) or 86400)) def fail_stale_topology_running_jobs(db: Session | None = None) -> int: