Cut ops turn idle loops and WhatsApp progress spam.

Align tool validation with playbook recipes, add turn checklist/idle guard, and cap interim WA status ticks so multi-hop CLI turns stop flooding the group.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
oliver 2026-08-11 11:13:41 +08:00
parent c3f3992a51
commit 76f62d8fac
12 changed files with 944 additions and 43 deletions

View file

@ -162,7 +162,16 @@ def build_ops_short_intent_hint(*, intent: str, lang: str = "en") -> str:
if not pair:
return ""
en, zh = pair
return zh if str(lang or "").strip().lower().startswith("zh") else en
base = zh if str(lang or "").strip().lower().startswith("zh") else en
try:
from runtime.tools.playbook_contracts import build_turn_checklist
checklist = build_turn_checklist(intent=key, lang=lang)
if checklist:
return f"{base}\n{checklist}".strip()
except Exception:
pass
return base
def maybe_ops_short_intent_system_hint(*, text: str, lang: str = "en") -> str:

View file

@ -1,4 +1,8 @@
"""Rate-limited WhatsApp interim progress during long channel turns."""
"""Rate-limited WhatsApp interim progress during long channel turns.
Field groups hate spam: keep at most a few ticks, never alternate
"Running CLI" ↔ "composing", and announce each long tool at most once.
"""
from __future__ import annotations
@ -52,11 +56,21 @@ def whatsapp_turn_progress_enabled() -> bool:
def progress_min_interval_sec() -> float:
"""Default 45s — field turns often run multi-hop CLI for minutes."""
raw = str(os.environ.get("OCLAW_WHATSAPP_PROGRESS_MIN_INTERVAL_SEC") or "").strip()
try:
return max(3.0, min(float(raw), 120.0))
return max(5.0, min(float(raw), 300.0))
except Exception:
return 12.0
return 45.0
def progress_max_per_turn() -> int:
"""Hard cap on interim WA ticks per inbound turn (final reply is separate)."""
raw = str(os.environ.get("OCLAW_WHATSAPP_PROGRESS_MAX_PER_TURN") or "").strip()
try:
return max(0, min(int(raw), 20))
except Exception:
return 2
def normalize_tool_key(name: str) -> str:
@ -84,7 +98,7 @@ def humanize_long_tool(*, tool_name: str, lang: str = "en") -> str | None:
def should_forward_progress_text(text: str) -> bool:
"""Filter noisy think/retry ticks; keep meaningful wait signals."""
"""Filter noisy think/retry/composing ticks; keep rare meaningful waits only."""
t = str(text or "").strip()
if not t:
return False
@ -95,13 +109,12 @@ def should_forward_progress_text(text: str) -> bool:
return False
if "retry-empty" in low or "retry-native-tool-calls" in low:
return False
m = _TOOLS_DONE_RE.search(t)
if m:
try:
return int(m.group(1)) >= 8000
except Exception:
return False
# Specialist / other explicit progress lines
if "idle-guard" in low:
return False
# Mid-turn "tools done / composing" is the main WA spam pattern when CLI loops.
if _TOOLS_DONE_RE.search(t) or "composing" in low or "整理回复" in t:
return False
# Specialist / other explicit progress lines (rare)
if low.startswith("oclaw:"):
return True
return len(t) >= 8
@ -110,11 +123,11 @@ def should_forward_progress_text(text: str) -> bool:
def humanize_progress_text(*, text: str, lang: str = "en") -> str:
t = str(text or "").strip()
is_zh = str(lang or "").strip().lower().startswith("zh")
m = _TOOLS_DONE_RE.search(t)
if m:
# tools-done is filtered by should_forward; keep a soft fallback if callers bypass.
if _TOOLS_DONE_RE.search(t):
if is_zh:
return "工具已完成,正在整理回复…"
return "Tools finished; composing the reply…"
return "仍在处理,请稍候…"
return "Still working, please wait…"
if t.lower().startswith("oclaw:"):
body = t.split(":", 1)[-1].strip()
if is_zh:
@ -150,6 +163,7 @@ class WhatsappTurnProgressPublisher:
is_group: bool = False,
inbound: Any = None,
min_interval_sec: float | None = None,
max_per_turn: int | None = None,
enabled: bool | None = None,
clock: Callable[[], float] | None = None,
) -> None:
@ -158,12 +172,14 @@ class WhatsappTurnProgressPublisher:
self._is_group = bool(is_group)
self._inbound = inbound
self._min_interval = float(min_interval_sec if min_interval_sec is not None else progress_min_interval_sec())
self._max_per_turn = int(max_per_turn if max_per_turn is not None else progress_max_per_turn())
self._enabled = bool(whatsapp_turn_progress_enabled() if enabled is None else enabled)
self._clock = clock or time.monotonic
self._lock = threading.Lock()
self._last_sent_at = 0.0
self._last_text = ""
self._sent_count = 0
self._announced_tools: set[str] = set()
@property
def sent_count(self) -> int:
@ -183,10 +199,18 @@ class WhatsappTurnProgressPublisher:
if str(event or "").strip() != "tool_use_call":
return
pl = payload if isinstance(payload, dict) else {}
label = humanize_long_tool(tool_name=str(pl.get("tool_name") or ""), lang=self._lang)
tool_name = str(pl.get("tool_name") or "")
key = normalize_tool_key(tool_name)
label = humanize_long_tool(tool_name=tool_name, lang=self._lang)
if not label:
return
self._maybe_send(label)
with self._lock:
# Same long tool (e.g. repeated execManagedNe hops) → one WA tick only.
if key and key in self._announced_tools:
return
if self._maybe_send(label) and key:
with self._lock:
self._announced_tools.add(key)
def _reply_metadata(self) -> dict[str, Any] | None:
if not self._is_group or self._inbound is None:
@ -196,23 +220,26 @@ class WhatsappTurnProgressPublisher:
except Exception:
return None
def _maybe_send(self, text: str) -> None:
def _maybe_send(self, text: str) -> bool:
msg = str(text or "").strip()
if not msg:
return
return False
with self._lock:
if self._max_per_turn >= 0 and self._sent_count >= self._max_per_turn:
return False
now = float(self._clock())
if msg == self._last_text and self._sent_count > 0:
return
return False
if self._sent_count > 0 and (now - self._last_sent_at) < self._min_interval:
return
return False
try:
self._enqueue(msg, self._reply_metadata())
except Exception:
return
return False
self._last_sent_at = now
self._last_text = msg
self._sent_count += 1
return True
__all__ = [
@ -221,6 +248,7 @@ __all__ = [
"humanize_long_tool",
"humanize_progress_text",
"normalize_tool_key",
"progress_max_per_turn",
"progress_min_interval_sec",
"should_forward_progress_text",
"whatsapp_turn_progress_enabled",

View file

@ -829,10 +829,18 @@ class ToolExecutor:
ok, v_err = validate_tool_arguments(tool.parameters, tool_args)
if not ok:
intent = None
try:
md = ctx.inbound_metadata if isinstance(ctx.inbound_metadata, dict) else {}
intent = str(md.get("ops_short_intent") or "").strip() or None
except Exception:
intent = None
return format_invalid_arguments_error(
tool.parameters or {},
str(v_err or "invalid"),
lang=str(ctx.lang or "zh"),
tool_name=str(tc.name or tool.name or ""),
intent=intent,
), int((time.perf_counter() - t0) * 1000)
try:

View file

@ -0,0 +1,209 @@
"""Turn-local idle / checklist guard to cut narration-only and no-progress loops."""
from __future__ import annotations
import os
from dataclasses import dataclass, field
from typing import Any, Literal
IdleAction = Literal["continue", "nudge", "early_finalize"]
def _env_int(name: str, default: int, *, min_v: int = 1, max_v: int = 20) -> int:
raw = str(os.getenv(name) or "").strip()
if not raw:
return default
try:
n = int(raw)
except Exception:
return default
return max(min_v, min(int(n), max_v))
def idle_guard_enabled() -> bool:
raw = str(os.getenv("AIA_TURN_IDLE_GUARD") or "1").strip().lower()
return raw not in ("0", "false", "no", "off")
@dataclass
class RoundStats:
round_idx: int
had_tool_calls: bool
ok_count: int = 0
fail_count: int = 0
schema_fail_count: int = 0
retry_guard_count: int = 0
tool_names: list[str] = field(default_factory=list)
@dataclass
class IdleGuardDecision:
action: IdleAction
reason: str
nudge_text: str = ""
@dataclass
class TurnIdleTracker:
"""Tracks per-turn progress and decides nudge / early finalize."""
lang: str = "en"
short_intent: str | None = None
max_idle_rounds: int = field(default_factory=lambda: _env_int("AIA_TURN_IDLE_MAX_ROUNDS", 2))
max_schema_fails: int = field(default_factory=lambda: _env_int("AIA_TURN_IDLE_MAX_SCHEMA_FAILS", 3))
rounds: list[RoundStats] = field(default_factory=list)
nudged: bool = False
total_ok: int = 0
total_schema_fails: int = 0
def record_from_traces(
self,
*,
round_idx: int,
had_tool_calls: bool,
round_traces: list[dict[str, Any]],
results_by_id: dict[str, tuple[dict[str, Any], int]] | None = None,
) -> RoundStats:
ok = 0
fail = 0
schema = 0
guard = 0
names: list[str] = []
# Prefer live results when available (richer failure_class).
payloads: list[dict[str, Any]] = []
if results_by_id:
for _cid, (payload, _dur) in results_by_id.items():
if isinstance(payload, dict):
payloads.append(payload)
for tr in round_traces or []:
name = str((tr or {}).get("name") or "").strip()
if name:
names.append(name)
if not payloads:
if (tr or {}).get("ok") is True:
ok += 1
elif (tr or {}).get("ok") is False:
fail += 1
for payload in payloads:
if payload.get("ok") is True:
ok += 1
continue
fail += 1
klass = str(payload.get("failure_class") or payload.get("error_class") or "").lower()
code = str(payload.get("error_code") or "").lower()
if klass == "schema_validation" or code in {"tool_invalid_arguments", "invalid_arguments"}:
schema += 1
if klass == "retry_guard" or code in {
"identical_retry_blocked",
"retry_forbidden_blocked",
"tool_loop_guard",
}:
guard += 1
stats = RoundStats(
round_idx=int(round_idx),
had_tool_calls=bool(had_tool_calls),
ok_count=int(ok),
fail_count=int(fail),
schema_fail_count=int(schema),
retry_guard_count=int(guard),
tool_names=names,
)
self.rounds.append(stats)
self.total_ok += int(ok)
self.total_schema_fails += int(schema)
return stats
def decide_after_assistant_no_tools(self) -> IdleGuardDecision:
"""Model returned text without tool_calls."""
if not idle_guard_enabled():
return IdleGuardDecision(action="continue", reason="disabled")
# Short-intent recipes require an evidence tool before answering.
intent = str(self.short_intent or "").strip()
if intent and intent != "continue" and self.total_ok == 0 and not self.nudged:
self.nudged = True
return IdleGuardDecision(
action="nudge",
reason="short_intent_narration_only",
nudge_text=self._nudge_text(reason="narration_only"),
)
return IdleGuardDecision(action="continue", reason="allow_text_reply")
def decide_after_tools(self, stats: RoundStats) -> IdleGuardDecision:
if not idle_guard_enabled():
return IdleGuardDecision(action="continue", reason="disabled")
if self.total_schema_fails >= int(self.max_schema_fails) and self.total_ok == 0:
return IdleGuardDecision(
action="early_finalize",
reason="schema_fail_budget",
nudge_text=self._nudge_text(reason="schema_fail_budget"),
)
# Count trailing idle rounds: tools ran but zero successes.
idle_streak = 0
for r in reversed(self.rounds):
if r.had_tool_calls and r.ok_count == 0:
idle_streak += 1
continue
break
if idle_streak >= int(self.max_idle_rounds) and self.total_ok == 0:
return IdleGuardDecision(
action="early_finalize",
reason="idle_no_progress",
nudge_text=self._nudge_text(reason="idle_no_progress"),
)
# One soft nudge when a round is all retry_guard / schema with no success yet.
if (
stats.had_tool_calls
and stats.ok_count == 0
and (stats.schema_fail_count + stats.retry_guard_count) > 0
and self.total_ok == 0
and not self.nudged
):
self.nudged = True
return IdleGuardDecision(
action="nudge",
reason="round_no_progress",
nudge_text=self._nudge_text(reason="round_no_progress"),
)
return IdleGuardDecision(action="continue", reason="progress_or_recoverable")
def _nudge_text(self, *, reason: str) -> str:
is_zh = str(self.lang or "").strip().lower().startswith("zh")
intent = str(self.short_intent or "").strip()
from runtime.tools.playbook_contracts import short_intent_first_step
step = short_intent_first_step(intent)
if is_zh:
lines = ["[Idle guard] 本轮未取得有效工具证据,禁止空转。"]
if step:
tool, example = step
lines.append(f"立即调用:{tool} 参数示例 {example}")
elif reason == "schema_fail_budget":
lines.append("参数多次不合法:按 tool 返回的 example/playbook 修正后只重试一次,然后直接作答。")
else:
lines.append("改参数或换 fallback 工具;若仍无证据则基于已知信息直接答复用户。")
return "\n".join(lines)
lines = ["[Idle guard] No usable tool evidence yet — stop spinning."]
if step:
tool, example = step
lines.append(f"Call now: {tool} with example args {example}")
elif reason == "schema_fail_budget":
lines.append(
"Repeated invalid arguments: fix once using the tool example/playbook, then answer."
)
else:
lines.append("Change args or switch fallback tools; if still blocked, answer from what you have.")
return "\n".join(lines)
__all__ = [
"IdleGuardDecision",
"RoundStats",
"TurnIdleTracker",
"idle_guard_enabled",
]

View file

@ -1255,8 +1255,48 @@ def run_oclaw_direct_loop(
user_facing_hints: list[str] = []
final_text = ""
hit_tool_round_limit = False
early_finalize_reason = ""
workspace_lane_role = str(skill_binding_role or wire_policy_role or "generalist").strip().lower() or "generalist"
# Turn checklist / idle guard (P0: cut narration-only + no-progress loops).
short_intent = ""
try:
md0 = inbound_metadata if isinstance(inbound_metadata, dict) else {}
short_intent = str(md0.get("ops_short_intent") or "").strip()
except Exception:
short_intent = ""
if not short_intent:
try:
from runtime.application.gateway.ops_short_intent import detect_ops_short_intent
short_intent = str(detect_ops_short_intent(str(user_text or "")) or "").strip()
except Exception:
short_intent = ""
turn_system_suffix = ""
if short_intent:
try:
from runtime.tools.playbook_contracts import build_turn_checklist
turn_system_suffix = build_turn_checklist(
intent=short_intent,
lang=lang,
goal=str(user_text or "")[:160],
)
except Exception:
turn_system_suffix = ""
from runtime.chat.turn_idle_guard import TurnIdleTracker
idle_tracker = TurnIdleTracker(lang=str(lang or "en"), short_intent=short_intent or None)
def _effective_system_prompt() -> str:
base = str(system_prompt or "")
extra = str(turn_system_suffix or "").strip()
if not extra:
return base
if extra in base:
return base
return f"{base}\n\n{extra}".strip()
base_url = str(getattr(model, "base_url", "") or "")
model_id = str(getattr(model, "model", "") or "")
allow_dsml_text_tools = dsml_text_tools_enabled(base_url=base_url, model_id=model_id)
@ -1275,7 +1315,7 @@ def run_oclaw_direct_loop(
store=store,
session_id=session_id,
max_messages=max_messages,
system_prompt=system_prompt,
system_prompt=_effective_system_prompt(),
model=model,
lang=lang,
memory_context=memory_context,
@ -1373,6 +1413,36 @@ def run_oclaw_direct_loop(
)
final_text = step.assistant_text
if not step.llm_tool_calls:
idle_decision = idle_tracker.decide_after_assistant_no_tools()
if idle_decision.action == "nudge" and idle_decision.nudge_text:
turn_system_suffix = (
f"{turn_system_suffix}\n\n{idle_decision.nudge_text}".strip()
if turn_system_suffix
else idle_decision.nudge_text
)
try:
if trace_id:
_emit_direct_loop_trace(
store=store,
session_id=session_id,
trace_id=trace_id,
parent_span_id=parent_span_id,
event_type="turn_idle_guard",
payload={
"action": idle_decision.action,
"reason": idle_decision.reason,
"round": int(round_idx + 1),
"short_intent": short_intent or "",
},
run_id=run_id,
attempt_no=attempt_no,
lang=lang,
)
except Exception:
pass
if on_progress:
on_progress("oclaw: idle-guard nudge…")
continue
break
if round_idx == (max_rounds - 1):
# Reached tool-round cap with pending tool calls. Execute this batch, then
@ -1420,17 +1490,18 @@ def run_oclaw_direct_loop(
inbound_metadata=inbound_metadata,
)
round_traces: list[dict[str, Any]] = []
for tc in step.llm_tool_calls:
result, dur = results_by_id.get(str(getattr(tc, "id", "") or ""), ({}, 0))
tool_traces.append(
{
"name": str(getattr(tc, "name", "") or ""),
"tool_call_id": str(getattr(tc, "id", "") or ""),
"ok": bool((result or {}).get("ok")) if isinstance(result, dict) else None,
"duration_ms": int(dur),
"round": int(round_idx + 1),
}
)
tr = {
"name": str(getattr(tc, "name", "") or ""),
"tool_call_id": str(getattr(tc, "id", "") or ""),
"ok": bool((result or {}).get("ok")) if isinstance(result, dict) else None,
"duration_ms": int(dur),
"round": int(round_idx + 1),
}
round_traces.append(tr)
tool_traces.append(tr)
if isinstance(result, dict):
uh = str(result.get("user_facing_hint") or "").strip()
if uh:
@ -1439,7 +1510,80 @@ def run_oclaw_direct_loop(
if on_progress:
on_progress(f"oclaw: tools done ({elapsed_ms}ms)")
need_finalize = hit_tool_round_limit or (bool(tool_traces) and not str(final_text or "").strip())
idle_stats = idle_tracker.record_from_traces(
round_idx=int(round_idx + 1),
had_tool_calls=True,
round_traces=round_traces,
results_by_id=results_by_id,
)
idle_decision = idle_tracker.decide_after_tools(idle_stats)
if idle_decision.action == "nudge" and idle_decision.nudge_text:
turn_system_suffix = (
f"{turn_system_suffix}\n\n{idle_decision.nudge_text}".strip()
if turn_system_suffix
else idle_decision.nudge_text
)
try:
if trace_id:
_emit_direct_loop_trace(
store=store,
session_id=session_id,
trace_id=trace_id,
parent_span_id=parent_span_id,
event_type="turn_idle_guard",
payload={
"action": idle_decision.action,
"reason": idle_decision.reason,
"round": int(round_idx + 1),
"short_intent": short_intent or "",
"ok_count": int(idle_stats.ok_count),
"schema_fail_count": int(idle_stats.schema_fail_count),
},
run_id=run_id,
attempt_no=attempt_no,
lang=lang,
)
except Exception:
pass
if on_progress:
on_progress("oclaw: idle-guard nudge…")
elif idle_decision.action == "early_finalize":
early_finalize_reason = str(idle_decision.reason or "idle")
if idle_decision.nudge_text:
turn_system_suffix = (
f"{turn_system_suffix}\n\n{idle_decision.nudge_text}".strip()
if turn_system_suffix
else idle_decision.nudge_text
)
try:
if trace_id:
_emit_direct_loop_trace(
store=store,
session_id=session_id,
trace_id=trace_id,
parent_span_id=parent_span_id,
event_type="turn_idle_guard",
payload={
"action": idle_decision.action,
"reason": idle_decision.reason,
"round": int(round_idx + 1),
"short_intent": short_intent or "",
},
run_id=run_id,
attempt_no=attempt_no,
lang=lang,
)
except Exception:
pass
if on_progress:
on_progress("oclaw: idle-guard finalize…")
break
need_finalize = (
hit_tool_round_limit
or bool(early_finalize_reason)
or (bool(tool_traces) and not str(final_text or "").strip())
)
if need_finalize:
_check_stop(should_stop)
if on_progress:
@ -1448,10 +1592,10 @@ def run_oclaw_direct_loop(
finalize_suffix = build_finalize_system_suffix(
lang=lang,
hit_tool_round_limit=hit_tool_round_limit,
hit_tool_round_limit=hit_tool_round_limit or bool(early_finalize_reason),
user_facing_hints=user_facing_hints,
)
finalize_system = str(system_prompt or "")
finalize_system = _effective_system_prompt()
if finalize_suffix:
finalize_system = f"{finalize_system}\n\n{finalize_suffix}".strip()
msgs = _build_model_context(

View file

@ -870,6 +870,14 @@ class OclawGateway:
short_intent = detect_ops_short_intent(
str(msg.text or md_intent.get("raw_inbound_text") or "")
)
if short_intent:
# Stamp for tool validation playbook examples + turn idle guard.
try:
if not isinstance(msg.metadata, dict):
msg.metadata = {}
msg.metadata["ops_short_intent"] = str(short_intent)
except Exception:
pass
if ops_short_intent_should_filter_tools(short_intent):
before_n = len(tools.list())
filtered_specs = filter_tool_specs_for_ops_short_intent(

View file

@ -0,0 +1,229 @@
"""Canonical ops playbook tool contracts.
Keeps skill recipes (ops-netx-*-playbook) and runtime JSON schemas aligned by
providing executable examples used in:
- invalid-argument error payloads (self-correct without blind retry)
- short-intent turn checklists
- schema↔playbook regression tests
"""
from __future__ import annotations
from typing import Any
# Bare MCP tool names (without mcp__netx__) and always-on expert tools.
_PLAYBOOK_EXAMPLES: dict[str, dict[str, dict[str, Any]]] = {
"ume_alarm_xlsx_report": {
"fiber_cut": {"mode": "fiber_cut", "deliverable": True},
"offline": {"mode": "offline", "deliverable": True},
"alarm_tally": {"mode": "aggregate_by_host", "severity": "critical", "deliverable": True},
"excel_export": {"mode": "list", "deliverable": True},
"license": {"mode": "list", "keyword": "license", "deliverable": True},
"congestion": {"mode": "list", "keyword": "bandwidth", "deliverable": True},
"default": {"mode": "list", "deliverable": True},
},
"write_xlsx": {
"excel_export": {
"sheets": [{"name": "Sheet1", "headers": ["host_name", "count"], "rows": [["NE-1", 1]]}],
"deliverable": True,
"name": "alarms.xlsx",
},
"default": {
"sheets": [{"name": "Sheet1", "headers": ["col"], "rows": [["val"]]}],
"deliverable": True,
"name": "report.xlsx",
},
},
"aggregateumealarms": {
"alarm_tally": {"severity": "critical", "top_ne": 20},
"default": {"severity": "critical", "top_ne": 20},
},
"queryumealarms": {
"default": {"host_name": "<host_name>", "page_size": 50},
},
"queryumealarmsraw": {
"congestion": {"keyword": "bandwidth", "field_preset": "evidence", "page_size": 50},
"license": {"keyword": "license", "field_preset": "evidence", "page_size": 50},
"default": {"keyword": "<cause>", "field_preset": "evidence", "page_size": 50},
},
"runumediagnostics": {
"default": {},
},
"listmanagedne": {
"default": {"keyword": "<host_or_area>", "connect_status": "pass"},
},
"getmanagedne": {
"default": {"ne_id": "<managed-ne-id-from-listManagedNe>"},
},
"execmanagedne": {
"default": {
"ume_ne_ids": ["<ume-uuid-1>", "<ume-uuid-2>"],
"commands": ["show version"],
"read_timeout_sec": 60,
},
},
"listclitargets": {
"default": {"source": "ume", "keyword": "<host_or_area>"},
},
"findtopologypaths": {
"default": {
"from_ume_ne_id": "<ume-uuid-a>",
"to_ume_ne_id": "<ume-uuid-b>",
"detail": "summary",
},
},
}
# Short-intent → preferred first tool + example key (playbook recipe).
_SHORT_INTENT_FIRST_STEP: dict[str, tuple[str, str]] = {
"fiber_cut": ("ume_alarm_xlsx_report", "fiber_cut"),
"offline": ("ume_alarm_xlsx_report", "offline"),
"alarm_tally": ("ume_alarm_xlsx_report", "alarm_tally"),
"excel_export": ("ume_alarm_xlsx_report", "excel_export"),
"license": ("ume_alarm_xlsx_report", "license"),
"congestion": ("ume_alarm_xlsx_report", "congestion"),
}
def canonical_tool_key(tool_name: str) -> str:
raw = str(tool_name or "").strip()
if not raw:
return ""
if "__" in raw:
raw = raw.rsplit("__", 1)[-1]
# Legacy snake_case netx_* → last segment style already handled by rsplit.
if raw.lower().startswith("netx_"):
raw = raw[5:]
return raw.strip().lower()
def playbook_examples_for_tool(tool_name: str) -> dict[str, dict[str, Any]]:
return dict(_PLAYBOOK_EXAMPLES.get(canonical_tool_key(tool_name)) or {})
def playbook_example_for_tool(
tool_name: str,
*,
intent: str | None = None,
) -> dict[str, Any] | None:
examples = playbook_examples_for_tool(tool_name)
if not examples:
return None
key = str(intent or "").strip()
if key and key in examples:
return dict(examples[key])
if "default" in examples:
return dict(examples["default"])
# First recipe as fallback.
first = next(iter(examples.values()), None)
return dict(first) if isinstance(first, dict) else None
def short_intent_first_step(intent: str | None) -> tuple[str, dict[str, Any]] | None:
key = str(intent or "").strip()
if not key:
return None
pair = _SHORT_INTENT_FIRST_STEP.get(key)
if not pair:
return None
tool, example_key = pair
example = playbook_example_for_tool(tool, intent=example_key) or {}
return tool, example
def build_turn_checklist(
*,
intent: str | None = None,
lang: str = "en",
goal: str | None = None,
) -> str:
"""Compact in-turn checklist injected into system prompt (ops short intents)."""
step = short_intent_first_step(intent)
is_zh = str(lang or "").strip().lower().startswith("zh")
lines: list[str] = []
if is_zh:
lines.append("[本轮 checklist — 先工具后结论,禁止只叙述不调用]")
else:
lines.append("[Turn checklist — call tools first; do not narrate-only]")
goal_s = str(goal or "").strip()
if goal_s:
lines.append(f"- goal: {goal_s[:160]}")
if step:
tool, example = step
lines.append(f"- step1: {tool}({_fmt_args(example)})")
if is_zh:
lines.append("- 完成后用 Result/Evidence 短答;勿翻页或开无关 playbook")
else:
lines.append("- then reply with Result/Evidence; no pagination / unrelated playbooks")
elif is_zh:
lines.append("- 需要证据时立刻调用工具;失败时改参数或换 fallback,禁止相同参数盲重试")
else:
lines.append("- If evidence is required, call a tool now; on failure change args or switch tools")
return "\n".join(lines)
def enrich_invalid_arguments_with_playbook(
payload: dict[str, Any],
*,
tool_name: str,
intent: str | None = None,
) -> dict[str, Any]:
"""Attach playbook-aligned example when schema validation fails."""
if not isinstance(payload, dict):
return payload
out = dict(payload)
example = playbook_example_for_tool(tool_name, intent=intent)
if example:
# Prefer playbook recipe over generic schema-derived example.
out["example"] = example
out["playbook_example"] = True
prior = str(out.get("hint") or "").strip()
tip = f"Playbook recipe: {_fmt_args(example)}"
if tip not in prior:
out["hint"] = f"{prior} {tip}".strip() if prior else tip
out["tool"] = str(tool_name or "").strip() or out.get("tool")
return out
def schema_playbook_mismatches(
tool_name: str,
parameters: dict[str, Any] | None,
) -> list[str]:
"""Return human-readable mismatches between playbook examples and JSON schema."""
from runtime.tools.tool_validation import validate_tool_arguments
examples = playbook_examples_for_tool(tool_name)
if not examples or not isinstance(parameters, dict) or not parameters:
return []
issues: list[str] = []
for label, args in examples.items():
ok, err = validate_tool_arguments(parameters, dict(args))
if not ok:
issues.append(f"{canonical_tool_key(tool_name)}/{label}: {err}")
return issues
def _fmt_args(args: dict[str, Any]) -> str:
parts: list[str] = []
for k, v in (args or {}).items():
if isinstance(v, bool):
parts.append(f"{k}={'true' if v else 'false'}")
elif isinstance(v, (int, float)):
parts.append(f"{k}={v}")
elif isinstance(v, str):
parts.append(f'{k}="{v}"')
else:
parts.append(f"{k}=…")
return ", ".join(parts)
__all__ = [
"build_turn_checklist",
"canonical_tool_key",
"enrich_invalid_arguments_with_playbook",
"playbook_example_for_tool",
"playbook_examples_for_tool",
"schema_playbook_mismatches",
"short_intent_first_step",
]

View file

@ -58,6 +58,8 @@ def format_invalid_arguments_error(
message: str,
*,
lang: str = "en",
tool_name: str | None = None,
intent: str | None = None,
) -> dict[str, Any]:
"""Rich invalid-arg payload so the model can self-correct without blind retries."""
props = parameters.get("properties") if isinstance(parameters.get("properties"), dict) else {}
@ -69,7 +71,7 @@ def format_invalid_arguments_error(
else:
err = f"Invalid arguments: {message}"
hint = "Fix arguments to match the schema example; do not retry with the same payload."
return {
out: dict[str, Any] = {
"ok": False,
"error_code": "tool_invalid_arguments",
"error": err,
@ -78,7 +80,20 @@ def format_invalid_arguments_error(
"properties": sorted(str(k) for k in props.keys()),
"example": example,
"hint": hint,
"failure_class": "schema_validation",
}
if tool_name:
try:
from runtime.tools.playbook_contracts import enrich_invalid_arguments_with_playbook
out = enrich_invalid_arguments_with_playbook(
out,
tool_name=str(tool_name),
intent=intent,
)
except Exception:
out["tool"] = str(tool_name)
return out
def validate_tool_arguments(parameters: dict[str, Any], arguments: dict[str, Any]) -> tuple[bool, str | None]: