Fix UME topology sync stuck running and unique-key flush failures.

Use savepoint upserts with payload dedupe, finalize jobs on a fresh sibling session, serialize concurrent topology syncs, and reap stale running rows so the scheduler does not keep spawning orphans.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-06 18:42:02 +08:00
parent fe483f51cc
commit 4fcec36514
9 changed files with 565 additions and 169 deletions

View file

@ -25,6 +25,7 @@ from .ume_alarm_ws import (
start_ume_alarm_ws_consumer,
)
from .ume_sync_service import sync_alarms_current, sync_inventory_full, sync_topology_full
from .ume_sync_topology import fail_stale_topology_running_jobs
from .runtime_task_messages import (
RT_ALARMS_SYNC_IN_PROGRESS_SKIP,
RT_KEEPALIVE_FAILED,
@ -323,6 +324,7 @@ def start_api_sideband_threads() -> None:
)
db = SessionLocal()
try:
fail_stale_topology_running_jobs(db)
client = ume_support._ume_client()
sync_topology_full(db, client, trigger_mode="schedule")
_schedule_log.info("topology_auto_sync: sync finished ok")
@ -334,6 +336,17 @@ def start_api_sideband_threads() -> None:
)
finally:
db.close()
except RuntimeError as exc:
if str(exc).startswith("topology_sync_busy"):
_schedule_log.info("topology_auto_sync: skipped busy (%s)", exc)
ume_support._refresh_runtime_task_idle(
"topology_auto_sync",
"topology",
last_error="rt:topology_sync_busy",
)
time.sleep(30)
else:
raise
except Exception as exc:
_schedule_log.exception("topology_auto_sync: sync failed: %s", exc)
ume_support._set_runtime_task(