mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 04:00:44 +08:00
重构仓库目录为统一的 runtime 分层并清理历史 openclaw 残留。
本次迁移将网关/通道/工具/技能/脚本与协议资源集中到新结构,统一路径常量与脚本转发机制,减少顶层噪音并保证运行与测试行为一致。 Made-with: Cursor
This commit is contained in:
parent
ba3836f00f
commit
4a23b715a2
498 changed files with 2760 additions and 2200 deletions
62
runtime/orchestration/evaluation.py
Normal file
62
runtime/orchestration/evaluation.py
Normal file
|
|
@ -0,0 +1,62 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from collections import defaultdict
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
|
||||
|
||||
def log_eval_event(
|
||||
store: SqliteStore,
|
||||
*,
|
||||
session_id: str,
|
||||
specialist: str,
|
||||
task_kind: str,
|
||||
success: bool,
|
||||
latency_ms: int,
|
||||
cost_hint: float = 0.0,
|
||||
notes: str = "",
|
||||
) -> None:
|
||||
store.add_agent_eval_log(
|
||||
session_id=session_id,
|
||||
specialist=specialist,
|
||||
task_kind=task_kind,
|
||||
success=success,
|
||||
latency_ms=latency_ms,
|
||||
cost_hint=cost_hint,
|
||||
notes=notes,
|
||||
)
|
||||
|
||||
|
||||
def eval_summary(store: SqliteStore, *, limit: int = 200) -> dict[str, Any]:
|
||||
rows = store.list_agent_eval_logs(limit=limit)
|
||||
if not rows:
|
||||
return {"total": 0, "success_rate": 0.0, "p95_latency_ms": 0}
|
||||
success_cnt = sum(1 for r in rows if bool(r.get("success")))
|
||||
lats = sorted(int(r.get("latency_ms") or 0) for r in rows)
|
||||
idx = max(0, int(len(lats) * 0.95) - 1)
|
||||
by_specialist: dict[str, dict[str, Any]] = defaultdict(lambda: {"total": 0, "ok": 0, "lat": []})
|
||||
plan_rows = 0
|
||||
for r in rows:
|
||||
sp = str(r.get("specialist") or "unknown")
|
||||
by_specialist[sp]["total"] += 1
|
||||
by_specialist[sp]["ok"] += 1 if bool(r.get("success")) else 0
|
||||
by_specialist[sp]["lat"].append(int(r.get("latency_ms") or 0))
|
||||
if "manager_plan_generated" in str(r.get("notes") or ""):
|
||||
plan_rows += 1
|
||||
specialist_metrics: dict[str, dict[str, Any]] = {}
|
||||
for sp, m in by_specialist.items():
|
||||
l = sorted(m["lat"])
|
||||
p95_idx = max(0, int(len(l) * 0.95) - 1)
|
||||
specialist_metrics[sp] = {
|
||||
"total": m["total"],
|
||||
"success_rate": round((m["ok"] / m["total"]) if m["total"] else 0.0, 4),
|
||||
"p95_latency_ms": l[p95_idx] if l else 0,
|
||||
}
|
||||
return {
|
||||
"total": len(rows),
|
||||
"success_rate": round(success_cnt / len(rows), 4),
|
||||
"p95_latency_ms": lats[idx],
|
||||
"plan_events": plan_rows,
|
||||
"by_specialist": specialist_metrics,
|
||||
}
|
||||
Loading…
Add table
Add a link
Reference in a new issue