mirror of
https://github.com/hansjone/netx.git
synced 2026-10-11 09:10:50 +08:00
Scale biz_state collects with dedicated workers and non-blocking UI poll.
Add PG claim/NE mutex, persist pool, and biz_state_worker replicas; fix double-SSH and row-count bugs; stop 32m collectNow while-loop from freezing page switches. Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
parent
0eb65a839e
commit
3a0de0d7fb
22 changed files with 1210 additions and 165 deletions
64
netx_api/biz_state_worker.py
Normal file
64
netx_api/biz_state_worker.py
Normal file
|
|
@ -0,0 +1,64 @@
|
|||
"""Dedicated biz_state collect worker process.
|
||||
|
||||
When ``NETX_BIZ_STATE_DEDICATED_WORKERS=true`` (default) and API uses external
|
||||
schedulers (``NETX_RUN_INLINE_SCHEDULERS=false``), run one or more replicas:
|
||||
|
||||
python -m netx_api.biz_state_worker
|
||||
|
||||
Each process enqueues due tasks, claims queued batches (SKIP LOCKED + NE mutex),
|
||||
runs SSH collect → spool, and drains the persist pool. Scale by starting N
|
||||
processes on the same host (start_netx.ps1 / .sh do this via replicas).
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import logging
|
||||
import signal
|
||||
import time
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s %(levelname)s %(name)s: %(message)s",
|
||||
)
|
||||
_log = logging.getLogger("netx.biz_state.worker")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
from .biz_state_scheduler import start_biz_state_scheduler, stop_biz_state_scheduler
|
||||
from .config import settings
|
||||
from .scheduler_heartbeat import start_scheduler_heartbeat_publisher
|
||||
|
||||
stop = False
|
||||
|
||||
def _handle(_sig: int, _frame: object) -> None:
|
||||
nonlocal stop
|
||||
_log.info("shutdown signal received")
|
||||
stop = True
|
||||
|
||||
for sig in (signal.SIGINT, signal.SIGTERM):
|
||||
try:
|
||||
signal.signal(sig, _handle)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
start_biz_state_scheduler()
|
||||
try:
|
||||
start_scheduler_heartbeat_publisher(role="biz_state_worker")
|
||||
except Exception:
|
||||
_log.exception("biz_state worker heartbeat failed to start")
|
||||
|
||||
_log.info(
|
||||
"biz_state worker started dedicated=%s max_concurrent=%s",
|
||||
bool(getattr(settings, "biz_state_dedicated_workers", True)),
|
||||
int(getattr(settings, "biz_state_max_concurrent_tasks", 16) or 16),
|
||||
)
|
||||
|
||||
while not stop:
|
||||
time.sleep(1.0)
|
||||
|
||||
stop_biz_state_scheduler()
|
||||
_log.info("biz_state worker exiting")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue