mirror of
https://github.com/hansjone/netx.git
synced 2026-10-10 09:50:44 +08:00
Harden config sync defaults, crash resume, and network nav collapse.
Disable auto-sync by default, enforce single-flight cycles with crash requeue, delay new scheduled runs after restart, and make the network sidebar collapsible. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
0eead0e662
commit
f5b4a14c0f
14 changed files with 387 additions and 99 deletions
|
|
@ -75,6 +75,8 @@ class Settings(BaseSettings):
|
|||
# Config sync (periodic running-config backup into DB)
|
||||
config_sync_scheduler_enabled: bool = True
|
||||
config_sync_scheduler_tick_sec: int = 60
|
||||
# After process start / unexpected restart, wait before any scheduled sync.
|
||||
config_sync_startup_grace_sec: int = 3600
|
||||
# Managed NE exec: max CLI commands per request (lab can raise; hard-capped in ne_exec).
|
||||
ne_exec_max_commands: int = 5
|
||||
# WebCRT interactive terminal sessions
|
||||
|
|
|
|||
|
|
@ -3,66 +3,112 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
from datetime import datetime
|
||||
|
||||
from sqlalchemy.orm import Session
|
||||
|
||||
from .config_sync_runner import dispatch_cycle
|
||||
from .config_sync_service import finalize_cycle, sync_cycle_progress
|
||||
from .models import ConfigSyncCycle, ConfigSyncTask
|
||||
from datetime import datetime
|
||||
|
||||
_log = logging.getLogger("netx.config_sync.recovery")
|
||||
|
||||
_ACTIVE = ("running", "paused", "pending")
|
||||
|
||||
|
||||
def _close_cycle(db: Session, cycle: ConfigSyncCycle, *, reason: str) -> None:
|
||||
"""Fail in-flight work and mark cycle terminal so it cannot block forever."""
|
||||
cycle_id = str(cycle.id)
|
||||
now = datetime.utcnow()
|
||||
for task in (
|
||||
db.query(ConfigSyncTask)
|
||||
.filter(
|
||||
ConfigSyncTask.cycle_id == cycle_id,
|
||||
ConfigSyncTask.status.in_(("running", "pending")),
|
||||
)
|
||||
.all()
|
||||
):
|
||||
was_running = str(task.status) == "running"
|
||||
task.status = "fail" if was_running else "cancelled"
|
||||
task.message = reason
|
||||
task.ended_at = now
|
||||
db.commit()
|
||||
sync_cycle_progress(db, cycle_id)
|
||||
db.refresh(cycle)
|
||||
cycle.status = "fail"
|
||||
cycle.error_message = reason
|
||||
cycle.ended_at = now
|
||||
db.commit()
|
||||
|
||||
|
||||
def recover_config_sync_on_startup(db: Session) -> int:
|
||||
"""
|
||||
Mark orphaned running tasks as fail(orphan_recovered), then resume pending
|
||||
tasks for cycles still marked running/paused.
|
||||
Resume at most one interrupted cycle after process restart.
|
||||
|
||||
Rules:
|
||||
- Only one active cycle (running/pending/paused) may exist; older actives are closed.
|
||||
- Orphan ``running`` tasks are re-queued to ``pending`` and continued (crash 续跑).
|
||||
- ``paused`` cycles stay paused (no auto dispatch) but still occupy the single-flight slot.
|
||||
- New scheduled cycles remain blocked while this active cycle exists.
|
||||
"""
|
||||
cycles = (
|
||||
db.query(ConfigSyncCycle)
|
||||
.filter(ConfigSyncCycle.status.in_(("running", "paused", "pending")))
|
||||
.filter(ConfigSyncCycle.status.in_(_ACTIVE))
|
||||
.all()
|
||||
)
|
||||
resumed = 0
|
||||
for cycle in cycles:
|
||||
cycle_id = str(cycle.id)
|
||||
orphans = (
|
||||
db.query(ConfigSyncTask)
|
||||
.filter(ConfigSyncTask.cycle_id == cycle_id, ConfigSyncTask.status == "running")
|
||||
.all()
|
||||
cycles = sorted(cycles, key=lambda c: c.created_at or datetime.min)
|
||||
if not cycles:
|
||||
return 0
|
||||
|
||||
# Single-flight hygiene: keep newest, close older interrupted cycles.
|
||||
primary = cycles[-1]
|
||||
for stale in cycles[:-1]:
|
||||
_log.warning(
|
||||
"config_sync recovery closing older active cycle=%s (keep=%s)",
|
||||
stale.id,
|
||||
primary.id,
|
||||
)
|
||||
for task in orphans:
|
||||
task.status = "fail"
|
||||
task.message = "orphan_recovered"
|
||||
task.ended_at = datetime.utcnow()
|
||||
if orphans:
|
||||
db.commit()
|
||||
_log.info("config_sync recovery cycle=%s orphaned_tasks=%s", cycle_id, len(orphans))
|
||||
_close_cycle(db, stale, reason="superseded_active_cycle")
|
||||
|
||||
sync_cycle_progress(db, cycle_id)
|
||||
db.refresh(cycle)
|
||||
cycle_id = str(primary.id)
|
||||
orphans = (
|
||||
db.query(ConfigSyncTask)
|
||||
.filter(ConfigSyncTask.cycle_id == cycle_id, ConfigSyncTask.status == "running")
|
||||
.all()
|
||||
)
|
||||
for task in orphans:
|
||||
task.status = "pending"
|
||||
task.message = "requeued_after_restart"
|
||||
task.started_at = None
|
||||
task.ended_at = None
|
||||
if orphans:
|
||||
db.commit()
|
||||
_log.info("config_sync recovery cycle=%s requeued_orphans=%s", cycle_id, len(orphans))
|
||||
|
||||
if str(cycle.status) == "paused":
|
||||
continue
|
||||
sync_cycle_progress(db, cycle_id)
|
||||
db.refresh(primary)
|
||||
|
||||
pending = (
|
||||
db.query(ConfigSyncTask)
|
||||
.filter(ConfigSyncTask.cycle_id == cycle_id, ConfigSyncTask.status == "pending")
|
||||
.count()
|
||||
)
|
||||
if pending <= 0:
|
||||
if str(cycle.status) in ("running", "pending"):
|
||||
finalize_cycle(db, cycle_id)
|
||||
continue
|
||||
if str(primary.status) == "paused":
|
||||
_log.info("config_sync recovery cycle=%s stays paused (blocks new cycles)", cycle_id)
|
||||
return 0
|
||||
|
||||
if str(cycle.status) == "pending":
|
||||
cycle.status = "running"
|
||||
if not cycle.started_at:
|
||||
cycle.started_at = datetime.utcnow()
|
||||
db.commit()
|
||||
pending = (
|
||||
db.query(ConfigSyncTask)
|
||||
.filter(ConfigSyncTask.cycle_id == cycle_id, ConfigSyncTask.status == "pending")
|
||||
.all()
|
||||
)
|
||||
if not pending:
|
||||
finalize_cycle(db, cycle_id)
|
||||
_log.info("config_sync recovery cycle=%s finalized (no pending)", cycle_id)
|
||||
return 0
|
||||
|
||||
n = dispatch_cycle(cycle_id)
|
||||
resumed += n
|
||||
_log.info("config_sync recovery resumed cycle=%s pending=%s", cycle_id, n)
|
||||
return resumed
|
||||
if str(primary.status) == "pending":
|
||||
primary.status = "running"
|
||||
if not primary.started_at:
|
||||
primary.started_at = datetime.utcnow()
|
||||
db.commit()
|
||||
|
||||
task_ids = [str(t.id) for t in pending]
|
||||
n = dispatch_cycle(cycle_id)
|
||||
_log.info("config_sync recovery resumed cycle=%s pending=%s dispatched=%s", cycle_id, len(task_ids), n)
|
||||
return n
|
||||
|
|
|
|||
|
|
@ -70,7 +70,7 @@ def _update_task(task_id: str, **fields: Any) -> None:
|
|||
db.close()
|
||||
|
||||
|
||||
def _update_task(task_id: str, **fields: Any) -> None:
|
||||
def _claim_task(cycle_id: str, task_id: str) -> bool:
|
||||
for attempt in range(10):
|
||||
db = SessionLocal()
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ from __future__ import annotations
|
|||
import logging
|
||||
import threading
|
||||
import time
|
||||
from datetime import datetime
|
||||
from datetime import datetime, timedelta
|
||||
from uuid import uuid4
|
||||
|
||||
from .config import settings
|
||||
|
|
@ -22,20 +22,50 @@ from .models import ConfigSyncCycle, ConfigSyncTask
|
|||
_log = logging.getLogger("netx.config_sync.scheduler")
|
||||
_stop = threading.Event()
|
||||
_thread: threading.Thread | None = None
|
||||
_BOOT_MONO = time.monotonic()
|
||||
|
||||
|
||||
def _utcnow() -> datetime:
|
||||
return datetime.utcnow()
|
||||
|
||||
|
||||
def startup_grace_remaining_sec() -> float:
|
||||
grace = max(0, int(settings.config_sync_startup_grace_sec or 0))
|
||||
elapsed = time.monotonic() - _BOOT_MONO
|
||||
return max(0.0, float(grace) - elapsed)
|
||||
|
||||
|
||||
def in_startup_grace() -> bool:
|
||||
return startup_grace_remaining_sec() > 0
|
||||
|
||||
|
||||
def startup_grace_until() -> datetime | None:
|
||||
rem = startup_grace_remaining_sec()
|
||||
if rem <= 0:
|
||||
return None
|
||||
return _utcnow() + timedelta(seconds=rem)
|
||||
|
||||
|
||||
def try_start_scheduled_cycle() -> str | None:
|
||||
"""Create and dispatch a scheduled cycle if policy is due. Returns cycle id or None."""
|
||||
"""Create and dispatch a scheduled cycle if policy is due. Returns cycle id or None.
|
||||
|
||||
Never starts while another cycle is active (running/pending/paused), including
|
||||
a cycle being resumed after crash. Startup grace only delays *new* scheduled runs.
|
||||
"""
|
||||
if in_startup_grace():
|
||||
return None
|
||||
db = SessionLocal()
|
||||
try:
|
||||
policy = ensure_policy(db)
|
||||
if not policy.enabled:
|
||||
return None
|
||||
if has_running_cycle(db):
|
||||
active = has_running_cycle(db)
|
||||
if active:
|
||||
_log.debug(
|
||||
"config_sync schedule skip: active cycle=%s status=%s",
|
||||
active.id,
|
||||
active.status,
|
||||
)
|
||||
return None
|
||||
due = next_due_at(db, policy)
|
||||
if due is not None and due > _utcnow():
|
||||
|
|
@ -85,7 +115,8 @@ def try_start_scheduled_cycle() -> str | None:
|
|||
|
||||
def _loop() -> None:
|
||||
tick = max(15, int(settings.config_sync_scheduler_tick_sec or 60))
|
||||
_log.info("config_sync scheduler started tick=%ss", tick)
|
||||
grace = max(0, int(settings.config_sync_startup_grace_sec or 0))
|
||||
_log.info("config_sync scheduler started tick=%ss startup_grace=%ss", tick, grace)
|
||||
while not _stop.is_set():
|
||||
try:
|
||||
if bool(settings.config_sync_scheduler_enabled):
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ def _utcnow() -> datetime:
|
|||
def ensure_policy(db: Session) -> ConfigSyncPolicy:
|
||||
row = db.get(ConfigSyncPolicy, POLICY_ID)
|
||||
if row is None:
|
||||
row = ConfigSyncPolicy(id=POLICY_ID)
|
||||
row = ConfigSyncPolicy(id=POLICY_ID, enabled=False)
|
||||
db.add(row)
|
||||
db.commit()
|
||||
db.refresh(row)
|
||||
|
|
@ -202,15 +202,21 @@ def expand_targets(db: Session, policy: ConfigSyncPolicy) -> list[dict[str, str]
|
|||
return out
|
||||
|
||||
|
||||
def has_running_cycle(db: Session) -> ConfigSyncCycle | None:
|
||||
def has_active_cycle(db: Session) -> ConfigSyncCycle | None:
|
||||
"""Any non-terminal cycle occupies the single-flight slot (incl. paused)."""
|
||||
return (
|
||||
db.query(ConfigSyncCycle)
|
||||
.filter(ConfigSyncCycle.status.in_(("running", "pending")))
|
||||
.filter(ConfigSyncCycle.status.in_(("running", "pending", "paused")))
|
||||
.order_by(ConfigSyncCycle.created_at.desc())
|
||||
.first()
|
||||
)
|
||||
|
||||
|
||||
def has_running_cycle(db: Session) -> ConfigSyncCycle | None:
|
||||
"""Backward-compatible alias: treat paused as active so a new cycle cannot start."""
|
||||
return has_active_cycle(db)
|
||||
|
||||
|
||||
def last_finished_cycle(db: Session) -> ConfigSyncCycle | None:
|
||||
return (
|
||||
db.query(ConfigSyncCycle)
|
||||
|
|
@ -224,6 +230,8 @@ def next_due_at(db: Session, policy: ConfigSyncPolicy | None = None) -> datetime
|
|||
pol = policy or ensure_policy(db)
|
||||
if not pol.enabled:
|
||||
return None
|
||||
from .config_sync_scheduler import startup_grace_until
|
||||
|
||||
last = (
|
||||
db.query(ConfigSyncCycle)
|
||||
.filter(ConfigSyncCycle.status == "success", ConfigSyncCycle.ended_at.isnot(None))
|
||||
|
|
@ -232,8 +240,14 @@ def next_due_at(db: Session, policy: ConfigSyncPolicy | None = None) -> datetime
|
|||
)
|
||||
days = max(1, int(pol.interval_days or 3))
|
||||
if last and last.ended_at:
|
||||
return last.ended_at + timedelta(days=days)
|
||||
return _utcnow()
|
||||
due = last.ended_at + timedelta(days=days)
|
||||
else:
|
||||
# Never synced successfully: do not fire immediately on enable / first boot.
|
||||
due = _utcnow() + timedelta(days=days)
|
||||
grace_until = startup_grace_until()
|
||||
if grace_until is not None and due < grace_until:
|
||||
return grace_until
|
||||
return due
|
||||
|
||||
|
||||
def create_cycle(db: Session, body: ConfigSyncCycleCreate) -> ConfigSyncCycleOut:
|
||||
|
|
|
|||
|
|
@ -840,7 +840,7 @@ def on_startup() -> None:
|
|||
ensure_policy(db)
|
||||
cfg_resumed = recover_config_sync_on_startup(db)
|
||||
if cfg_resumed:
|
||||
_schedule_log.info("startup: resumed %s pending config_sync tasks", cfg_resumed)
|
||||
_schedule_log.info("startup: resumed %s config_sync task(s) from interrupted cycle", cfg_resumed)
|
||||
except Exception:
|
||||
_schedule_log.exception("startup: ne collection / config_sync recovery failed")
|
||||
finally:
|
||||
|
|
|
|||
|
|
@ -492,7 +492,7 @@ class ConfigSyncPolicy(Base):
|
|||
__tablename__ = "config_sync_policy"
|
||||
|
||||
id: Mapped[int] = mapped_column(Integer, primary_key=True, default=1)
|
||||
enabled: Mapped[bool] = mapped_column(Boolean, default=True)
|
||||
enabled: Mapped[bool] = mapped_column(Boolean, default=False)
|
||||
interval_days: Mapped[int] = mapped_column(Integer, default=3)
|
||||
concurrency: Mapped[int] = mapped_column(Integer, default=5)
|
||||
scope_mode: Mapped[str] = mapped_column(String(32), default="all") # all | selected
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue