mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 05:10:45 +08:00
重构仓库目录为统一的 runtime 分层并清理历史 openclaw 残留。
本次迁移将网关/通道/工具/技能/脚本与协议资源集中到新结构,统一路径常量与脚本转发机制,减少顶层噪音并保证运行与测试行为一致。 Made-with: Cursor
This commit is contained in:
parent
ba3836f00f
commit
4a23b715a2
498 changed files with 2760 additions and 2200 deletions
0
runtime/tools/evals/__init__.py
Normal file
0
runtime/tools/evals/__init__.py
Normal file
105
runtime/tools/evals/assistant_runner.py
Normal file
105
runtime/tools/evals/assistant_runner.py
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from oclaw.runtime.application.gateway import process_inbound_payload_usecase
|
||||
from oclaw.platform.config.paths import db_path
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class Case:
|
||||
case_id: str
|
||||
kind: str
|
||||
payload: dict[str, Any]
|
||||
assert_contains: list[str]
|
||||
assert_not_contains: list[str]
|
||||
|
||||
|
||||
def _load_cases(path: str) -> list[Case]:
|
||||
p = Path(path)
|
||||
if not p.exists():
|
||||
raise FileNotFoundError(path)
|
||||
out: list[Case] = []
|
||||
for idx, line in enumerate(p.read_text(encoding="utf-8").splitlines(), start=1):
|
||||
raw = line.strip()
|
||||
if not raw:
|
||||
continue
|
||||
row = json.loads(raw)
|
||||
cid = str(row.get("id") or f"line-{idx}")
|
||||
kind = str(row.get("kind") or "gateway")
|
||||
payload = row.get("payload") if isinstance(row.get("payload"), dict) else {}
|
||||
ac = row.get("assert_contains") or []
|
||||
anc = row.get("assert_not_contains") or []
|
||||
out.append(
|
||||
Case(
|
||||
case_id=cid,
|
||||
kind=kind,
|
||||
payload=payload,
|
||||
assert_contains=[str(x) for x in ac if str(x).strip()],
|
||||
assert_not_contains=[str(x) for x in anc if str(x).strip()],
|
||||
)
|
||||
)
|
||||
return out
|
||||
|
||||
|
||||
def _extract_reply_text(resp: dict[str, Any]) -> str:
|
||||
try:
|
||||
replies = resp.get("replies")
|
||||
if isinstance(replies, list) and replies:
|
||||
first = replies[0]
|
||||
if isinstance(first, dict):
|
||||
return str(first.get("text") or "")
|
||||
except Exception:
|
||||
pass
|
||||
return ""
|
||||
|
||||
|
||||
def run_gateway_eval(dataset_path: str) -> dict[str, Any]:
|
||||
store = SqliteStore(db_path())
|
||||
# Seed a tenant + bind code for tests
|
||||
tenants = store.list_tenants(limit=1)
|
||||
if tenants:
|
||||
tenant_id = tenants[0]["id"]
|
||||
else:
|
||||
tenant_id = store.create_tenant("Eval")["id"]
|
||||
code = "EVALCODE"
|
||||
try:
|
||||
store.create_bind_code(tenant_id=tenant_id, role="member", code=code)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Binding creates the user; we will use external ids in payloads.
|
||||
cases = _load_cases(dataset_path)
|
||||
results = []
|
||||
passed = 0
|
||||
for c in cases:
|
||||
payload = dict(c.payload)
|
||||
# inject tenant/code shortcuts
|
||||
payload.setdefault("channel", "wecom")
|
||||
payload.setdefault("chat_id", "room_eval")
|
||||
payload.setdefault("user_id", "wxid_eval_u1")
|
||||
payload.setdefault("is_group", True)
|
||||
payload["text"] = str(payload.get("text") or "").replace("EVALCODE", code)
|
||||
resp = process_inbound_payload_usecase(payload)
|
||||
text = _extract_reply_text(resp)
|
||||
failures = []
|
||||
for must in c.assert_contains:
|
||||
if must not in text:
|
||||
failures.append(f"missing:{must}")
|
||||
for bad in c.assert_not_contains:
|
||||
if bad in text:
|
||||
failures.append(f"unexpected:{bad}")
|
||||
ok = not failures
|
||||
passed += 1 if ok else 0
|
||||
results.append({"id": c.case_id, "ok": ok, "text": text, "failures": failures})
|
||||
return {"total": len(results), "passed": passed, "pass_rate": (passed / len(results)) if results else 0.0, "results": results}
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
rep = run_gateway_eval("data/eval/assistant_gateway.jsonl")
|
||||
print(json.dumps({k: v for k, v in rep.items() if k != "results"}, ensure_ascii=False, indent=2))
|
||||
|
||||
125
runtime/tools/evals/runner.py
Normal file
125
runtime/tools/evals/runner.py
Normal file
|
|
@ -0,0 +1,125 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import time
|
||||
from pathlib import Path
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
from oclaw.runtime.agents.factory import build_gateway_executor
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
from oclaw.runtime.orchestration.evaluation import eval_summary
|
||||
from oclaw.platform.config.paths import db_path
|
||||
from oclaw.runtime.gateway import OclawGateway
|
||||
from oclaw.runtime.types import StandardMessage
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EvalCase:
|
||||
case_id: str
|
||||
input_text: str
|
||||
assert_contains: list[str]
|
||||
assert_not_contains: list[str]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class EvalCaseResult:
|
||||
case_id: str
|
||||
ok: bool
|
||||
latency_ms: int
|
||||
failures: list[str]
|
||||
|
||||
|
||||
def _load_dataset(dataset_path: str) -> list[EvalCase]:
|
||||
ds = Path(dataset_path)
|
||||
if not ds.exists():
|
||||
raise FileNotFoundError(dataset_path)
|
||||
cases: list[EvalCase] = []
|
||||
with ds.open("r", encoding="utf-8") as f:
|
||||
for idx, line in enumerate(f, start=1):
|
||||
raw = line.strip()
|
||||
if not raw:
|
||||
continue
|
||||
row = json.loads(raw)
|
||||
input_text = str(row.get("input") or "").strip()
|
||||
if not input_text:
|
||||
continue
|
||||
case_id = str(row.get("id") or row.get("case_id") or f"line-{idx}").strip()
|
||||
ac = row.get("assert_contains") or []
|
||||
anc = row.get("assert_not_contains") or []
|
||||
assert_contains = [str(x) for x in ac if str(x).strip()]
|
||||
assert_not_contains = [str(x) for x in anc if str(x).strip()]
|
||||
cases.append(
|
||||
EvalCase(
|
||||
case_id=case_id,
|
||||
input_text=input_text,
|
||||
assert_contains=assert_contains,
|
||||
assert_not_contains=assert_not_contains,
|
||||
)
|
||||
)
|
||||
return cases
|
||||
|
||||
|
||||
def run_eval(
|
||||
dataset_path: str,
|
||||
*,
|
||||
report_path: str | None = None,
|
||||
limit: int | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Run a simple offline regression eval.
|
||||
|
||||
Dataset format: JSONL, each line:
|
||||
{"id": "...", "input": "...", "assert_contains": ["..."], "assert_not_contains": ["..."]}
|
||||
"""
|
||||
store = SqliteStore(db_path())
|
||||
agent = build_gateway_executor(store)
|
||||
session = store.create_session("offline-eval")
|
||||
gw = OclawGateway(store=store)
|
||||
cases = _load_dataset(dataset_path)
|
||||
if limit is not None:
|
||||
cases = cases[: max(0, int(limit))]
|
||||
|
||||
results: list[EvalCaseResult] = []
|
||||
for c in cases:
|
||||
t0 = time.perf_counter()
|
||||
msg = StandardMessage(
|
||||
session_id=str(session.id),
|
||||
tenant_id="",
|
||||
user_id="",
|
||||
role="owner",
|
||||
channel="eval",
|
||||
text=str(c.input_text or ""),
|
||||
attachments=[],
|
||||
metadata={"channel": "eval"},
|
||||
)
|
||||
out = str(gw.handle_turn(msg=msg, lang="zh", executor=agent).reply_text or "")
|
||||
latency_ms = int((time.perf_counter() - t0) * 1000)
|
||||
failures: list[str] = []
|
||||
for must in c.assert_contains:
|
||||
if must not in out:
|
||||
failures.append(f"missing_substring:{must}")
|
||||
for bad in c.assert_not_contains:
|
||||
if bad in out:
|
||||
failures.append(f"unexpected_substring:{bad}")
|
||||
results.append(EvalCaseResult(case_id=c.case_id, ok=not failures, latency_ms=latency_ms, failures=failures))
|
||||
|
||||
passed = sum(1 for r in results if r.ok)
|
||||
report = {
|
||||
"dataset": str(dataset_path),
|
||||
"total": len(results),
|
||||
"passed": passed,
|
||||
"pass_rate": round((passed / len(results)) if results else 0.0, 4),
|
||||
"results": [
|
||||
{"id": r.case_id, "ok": r.ok, "latency_ms": r.latency_ms, "failures": r.failures} for r in results
|
||||
],
|
||||
"agent_metrics": eval_summary(store, limit=5000),
|
||||
}
|
||||
if report_path:
|
||||
Path(report_path).parent.mkdir(parents=True, exist_ok=True)
|
||||
Path(report_path).write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
|
||||
return report
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
result = run_eval("data/eval/mvp_tasks.jsonl", report_path="data/eval/report.json")
|
||||
print(json.dumps({k: v for k, v in result.items() if k != "results"}, ensure_ascii=False, indent=2))
|
||||
Loading…
Add table
Add a link
Reference in a new issue