mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 03:23:23 +08:00
完善 MCP/会话上下文治理并补齐搜索与诊断能力。
补充历史压缩与会话诊断链路,强化 tool pairing 与空响应兜底观测;完善 Admin MCP 导入/绑定与相关前端展示;新增 web_search_fast/web_fetch_clean 及多项测试与文档更新,并同步技能安装与角色绑定策略改进。 Made-with: Cursor
This commit is contained in:
parent
df91434936
commit
ae44cbcad5
43 changed files with 3549 additions and 57 deletions
|
|
@ -23,6 +23,10 @@ from oclaw.runtime.relay_pointer import parse_pointer_uri
|
|||
logger = logging.getLogger(__name__)
|
||||
_THINK_BLOCK_RE = re.compile(r"<think>\s*(.*?)\s*</think>\s*", flags=re.IGNORECASE | re.DOTALL)
|
||||
_TOOL_CONTEXT_RESULT_MAX_CHARS = 50
|
||||
_LAST_BUILD_LLM_MESSAGES_STATS: dict[str, int] = {
|
||||
"dropped_unpaired_tool_rows": 0,
|
||||
"dropped_no_id_tool_rows": 0,
|
||||
}
|
||||
|
||||
|
||||
def _replay_recent_tool_rounds() -> int:
|
||||
|
|
@ -210,6 +214,8 @@ def build_llm_messages(
|
|||
older user attachments are replayed as text metadata only.
|
||||
"""
|
||||
out: list[dict[str, Any]] = [{"role": "system", "content": (system_prompt or "").strip()}]
|
||||
dropped_unpaired_tool_rows = 0
|
||||
dropped_no_id_tool_rows = 0
|
||||
thinking_mode_enabled = bool(getattr(model, "thinking_mode_enabled", False))
|
||||
allow_signature_replay = _allow_reasoning_signature_replay(model)
|
||||
reasoning_by_turn: dict[str, list[tuple[int, str]]] = {}
|
||||
|
|
@ -250,6 +256,7 @@ def build_llm_messages(
|
|||
seen_tool_ids.add(tcid)
|
||||
tool_ids_after.append(set(seen_tool_ids))
|
||||
tool_ids_after.reverse()
|
||||
pending_tool_ids_for_next_tool_rows: set[str] = set()
|
||||
|
||||
last_user_msg_idx = -1
|
||||
for _ui, _um in enumerate(store_messages or []):
|
||||
|
|
@ -550,6 +557,11 @@ def build_llm_messages(
|
|||
entry["extra_content"] = {"google": {"thought_signature": raw_sig}}
|
||||
api_tool_calls.append(entry)
|
||||
if api_tool_calls:
|
||||
pending_tool_ids_for_next_tool_rows = {
|
||||
str(tc.get("id") or "").strip()
|
||||
for tc in api_tool_calls
|
||||
if str(tc.get("id") or "").strip()
|
||||
}
|
||||
out.append(
|
||||
_attach_reasoning_content(
|
||||
{
|
||||
|
|
@ -561,6 +573,7 @@ def build_llm_messages(
|
|||
)
|
||||
)
|
||||
else:
|
||||
pending_tool_ids_for_next_tool_rows = set()
|
||||
out.append(
|
||||
_attach_reasoning_content(
|
||||
{"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")},
|
||||
|
|
@ -568,6 +581,7 @@ def build_llm_messages(
|
|||
)
|
||||
)
|
||||
else:
|
||||
pending_tool_ids_for_next_tool_rows = set()
|
||||
out.append(
|
||||
_attach_reasoning_content(
|
||||
{"role": "assistant", "content": _strip_reasoning_blocks(getattr(m, "content", "") or "")},
|
||||
|
|
@ -593,26 +607,15 @@ def build_llm_messages(
|
|||
tool_call_id = ""
|
||||
if tool_call_id:
|
||||
# Guard against dangling tool_call_id (assistant tool_calls missing from this trimmed context window).
|
||||
if str(tool_call_id) not in valid_tool_call_ids:
|
||||
# Preserve tool evidence, but downgrade to plain assistant text when pairing is broken.
|
||||
# Some OpenAI-compatible gateways reject a role=tool message if tool_call_id cannot be paired
|
||||
# to an assistant.tool_calls.id within the same request context.
|
||||
r0 = getattr(m, "content", "") or ""
|
||||
cap0 = tool_llm_message_max_chars()
|
||||
pretty = _summarize_unpaired_tool_content(r0, cap=cap0)
|
||||
if tool_context_truncate_enabled:
|
||||
pretty = _truncate_tool_context(pretty, lang=lang)
|
||||
out.append(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": render_prompt(
|
||||
"tools/tool_result_unpaired.md",
|
||||
variables={"tag": "tool_use_result:unpaired", "payload": pretty},
|
||||
strict=True,
|
||||
),
|
||||
}
|
||||
)
|
||||
# Also require strict immediate-turn pairing: a tool row must follow the assistant
|
||||
# tool_calls message that introduced this id (no unrelated message in-between).
|
||||
# Some OpenAI-compatible gateways enforce this strictly.
|
||||
if str(tool_call_id) not in valid_tool_call_ids or str(tool_call_id) not in pending_tool_ids_for_next_tool_rows:
|
||||
# Strict pairing mode: drop unpaired tool rows entirely.
|
||||
# This avoids provider-side 400 errors caused by orphan tool_result blocks.
|
||||
dropped_unpaired_tool_rows += 1
|
||||
continue
|
||||
pending_tool_ids_for_next_tool_rows.discard(str(tool_call_id))
|
||||
raw_tc_content = getattr(m, "content", "") or ""
|
||||
_tun = str(getattr(m, "turn_uuid", "") or "").strip()
|
||||
_aus = str(active_turn_uuid or "").strip()
|
||||
|
|
@ -661,23 +664,21 @@ def build_llm_messages(
|
|||
tool_row["name"] = str(meta2["name"])
|
||||
out.append(tool_row)
|
||||
else:
|
||||
r = getattr(m, "content", "") or ""
|
||||
cap2 = tool_llm_message_max_chars()
|
||||
pretty2 = _summarize_unpaired_tool_content(r, cap=cap2)
|
||||
# Unpaired tool rows are already summarized; avoid extra 50-char clipping.
|
||||
out.append(
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": render_prompt(
|
||||
"tools/tool_result_unpaired.md",
|
||||
variables={"tag": "tool_use_result:no_id", "payload": pretty2},
|
||||
strict=True,
|
||||
),
|
||||
}
|
||||
)
|
||||
pending_tool_ids_for_next_tool_rows = set()
|
||||
# Strict pairing mode: drop no-id tool rows entirely.
|
||||
dropped_no_id_tool_rows += 1
|
||||
continue
|
||||
|
||||
global _LAST_BUILD_LLM_MESSAGES_STATS
|
||||
_LAST_BUILD_LLM_MESSAGES_STATS = {
|
||||
"dropped_unpaired_tool_rows": int(dropped_unpaired_tool_rows),
|
||||
"dropped_no_id_tool_rows": int(dropped_no_id_tool_rows),
|
||||
}
|
||||
return out
|
||||
|
||||
|
||||
__all__ = ["build_llm_messages"]
|
||||
def get_last_build_llm_messages_stats() -> dict[str, int]:
|
||||
return dict(_LAST_BUILD_LLM_MESSAGES_STATS)
|
||||
|
||||
|
||||
__all__ = ["build_llm_messages", "get_last_build_llm_messages_stats"]
|
||||
|
|
|
|||
244
runtime/chat/history_tool_result_compact.py
Normal file
244
runtime/chat/history_tool_result_compact.py
Normal file
|
|
@ -0,0 +1,244 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
from oclaw.runtime.chat.media_redact import redact_embedded_image_blobs
|
||||
from oclaw.runtime.direct_loop import (
|
||||
_OCLAW_TOOL_RESULT_HARD_CAP_CHARS,
|
||||
_image_tool_result_replay_cap_chars,
|
||||
_video_tool_result_replay_cap_chars,
|
||||
)
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class HistoryCompactionResult:
|
||||
ok: bool
|
||||
session_id: str
|
||||
scanned_tool_messages: int = 0
|
||||
compacted_tool_messages: int = 0
|
||||
rewritten_all_tool_messages: int = 0
|
||||
skipped_already_guarded: int = 0
|
||||
max_original_chars_seen: int = 0
|
||||
cap_chars: int = 0
|
||||
detail: str = ""
|
||||
|
||||
|
||||
def _json_dumps_safe(obj: Any) -> str:
|
||||
try:
|
||||
return json.dumps(obj, ensure_ascii=False, default=str)
|
||||
except Exception:
|
||||
return json.dumps({"ok": False, "error": "not_json_serializable"}, ensure_ascii=False)
|
||||
|
||||
|
||||
def _guard_tool_result_text_for_history(
|
||||
*,
|
||||
store: Any,
|
||||
raw: str,
|
||||
cap_chars: int,
|
||||
image_cap_chars: int,
|
||||
video_cap_chars: int,
|
||||
) -> tuple[str, bool]:
|
||||
"""Return (new_raw, changed) following the same strategy as context replay guard."""
|
||||
text = str(raw or "")
|
||||
if not text:
|
||||
return text, False
|
||||
# Redact embedded blobs first (same as context guard).
|
||||
try:
|
||||
p0 = json.loads(text)
|
||||
p1 = redact_embedded_image_blobs(p0)
|
||||
text = _json_dumps_safe(p1)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# If already guarded, do not rewrite again.
|
||||
try:
|
||||
obj0 = json.loads(text)
|
||||
if isinstance(obj0, dict) and (obj0.get("_tool_result_guarded") or obj0.get("_image_tool_result_guarded") or obj0.get("_video_tool_result_guarded")):
|
||||
return text, False
|
||||
except Exception:
|
||||
obj0 = None
|
||||
|
||||
ok = None
|
||||
error_code = ""
|
||||
error = ""
|
||||
obj: dict[str, Any] | None = None
|
||||
try:
|
||||
parsed = json.loads(text)
|
||||
if isinstance(parsed, dict):
|
||||
obj = parsed
|
||||
ok = obj.get("ok")
|
||||
error_code = str(obj.get("error_code") or "").strip()
|
||||
error = str(obj.get("error") or "").strip()
|
||||
except Exception:
|
||||
obj = None
|
||||
|
||||
if isinstance(obj, dict):
|
||||
task = str(obj.get("task") or "").strip().lower()
|
||||
t = str(obj.get("text") or "")
|
||||
has_attachment_id = bool(str(obj.get("attachment_id") or "").strip())
|
||||
if task in {"describe", "ocr"} and has_attachment_id and len(t) > int(image_cap_chars):
|
||||
preview = t[: int(image_cap_chars)] + "\n...<image_tool_result_truncated_for_context_replay>"
|
||||
guarded_obj = dict(obj)
|
||||
guarded_obj["text"] = preview
|
||||
guarded_obj["_image_tool_result_guarded"] = True
|
||||
guarded_obj["image_result_original_chars"] = len(t)
|
||||
guarded_obj["image_result_replay_cap_chars"] = int(image_cap_chars)
|
||||
guarded_obj["image_result_hint"] = (
|
||||
"Image analysis result was truncated for context replay. "
|
||||
"Refine query_image_attachment(question=...) for narrower evidence. / "
|
||||
"图片分析结果在上下文回放中已截断,请缩小 query_image_attachment 的问题范围。"
|
||||
)
|
||||
return _json_dumps_safe(guarded_obj), True
|
||||
if task == "transcript" and has_attachment_id and len(t) > int(video_cap_chars):
|
||||
preview = t[: int(video_cap_chars)] + "\n...<video_tool_result_truncated_for_context_replay>"
|
||||
guarded_obj = dict(obj)
|
||||
guarded_obj["text"] = preview
|
||||
guarded_obj["_video_tool_result_guarded"] = True
|
||||
guarded_obj["video_result_original_chars"] = len(t)
|
||||
guarded_obj["video_result_replay_cap_chars"] = int(video_cap_chars)
|
||||
return _json_dumps_safe(guarded_obj), True
|
||||
|
||||
if len(text) <= int(cap_chars):
|
||||
return text, False
|
||||
|
||||
preview = text[: max(1, min(4000, int(cap_chars) - 400))] + "\n...<tool_result_guard_truncated>"
|
||||
guarded_obj = {
|
||||
"ok": bool(ok) if ok is not None else None,
|
||||
"error_code": error_code,
|
||||
"error": error,
|
||||
"_tool_result_guarded": True,
|
||||
"original_chars": len(text),
|
||||
"guard_cap_chars": int(cap_chars),
|
||||
"preview": preview,
|
||||
"hint": (
|
||||
"Tool output was too large for safe context replay; it was truncated for history storage. "
|
||||
"Use narrower queries (e.g., smaller glob/max_results) or adjust AIA_TOOL_LLM_MESSAGE_MAX_CHARS. / "
|
||||
"工具输出过大,已压缩写回历史;请缩小范围或配置 AIA_TOOL_LLM_MESSAGE_MAX_CHARS。"
|
||||
),
|
||||
}
|
||||
return _json_dumps_safe(guarded_obj), True
|
||||
|
||||
|
||||
def compact_tool_results_in_session_history(
|
||||
*,
|
||||
store: Any,
|
||||
session_id: str,
|
||||
cap_chars: int | None = None,
|
||||
limit_messages: int = 5000,
|
||||
rewrite_all: bool = True,
|
||||
) -> HistoryCompactionResult:
|
||||
"""Rewrite overlarge `role=tool` chat_message.content in DB using the replay-guard strategy."""
|
||||
sid = str(session_id or "").strip()
|
||||
if not sid:
|
||||
return HistoryCompactionResult(ok=False, session_id="", detail="session_id_required")
|
||||
cap = max(4096, min(int(cap_chars or _OCLAW_TOOL_RESULT_HARD_CAP_CHARS), 500_000))
|
||||
image_cap = int(_image_tool_result_replay_cap_chars(store))
|
||||
video_cap = int(_video_tool_result_replay_cap_chars(store))
|
||||
scanned = 0
|
||||
compacted = 0
|
||||
rewritten_all = 0
|
||||
skipped = 0
|
||||
max_seen = 0
|
||||
|
||||
# We only need to scan tool messages; fetching ids+content is enough.
|
||||
with store._connect() as conn: # noqa: SLF001
|
||||
cur = conn.execute(
|
||||
"select id, content from chat_message where session_id=? and role='tool' "
|
||||
"order by id asc limit ?",
|
||||
(sid, max(1, min(int(limit_messages or 5000), 200_000))),
|
||||
)
|
||||
rows = cur.fetchall() or []
|
||||
for mid, raw in rows:
|
||||
scanned += 1
|
||||
txt = str(raw or "")
|
||||
max_seen = max(max_seen, len(txt))
|
||||
new_txt, changed = _guard_tool_result_text_for_history(
|
||||
store=store,
|
||||
raw=txt,
|
||||
cap_chars=cap,
|
||||
image_cap_chars=image_cap,
|
||||
video_cap_chars=video_cap,
|
||||
)
|
||||
# Defensive fallback: if content is still over cap but guard didn't report change,
|
||||
# force a minimal guard so polluted history can always be compacted.
|
||||
if (not changed) and len(txt) > int(cap):
|
||||
preview = txt[: max(1, min(4000, int(cap) - 400))] + "\n...<tool_result_guard_truncated>"
|
||||
new_txt = _json_dumps_safe(
|
||||
{
|
||||
"ok": None,
|
||||
"error_code": "",
|
||||
"error": "",
|
||||
"_tool_result_guarded": True,
|
||||
"original_chars": len(txt),
|
||||
"guard_cap_chars": int(cap),
|
||||
"preview": preview,
|
||||
"hint": (
|
||||
"Tool output was too large for safe context replay; it was truncated for history storage. / "
|
||||
"工具输出过大,已压缩写回历史。"
|
||||
),
|
||||
}
|
||||
)
|
||||
changed = True
|
||||
# Full rewrite mode: compact every tool_result row into guarded envelope,
|
||||
# even when current content is under cap. This keeps history bounded and
|
||||
# prevents heterogeneous huge payload persistence.
|
||||
if (not changed) and bool(rewrite_all):
|
||||
preview_cap = max(200, min(1200, int(cap) - 200))
|
||||
preview = txt[:preview_cap]
|
||||
if len(txt) > preview_cap:
|
||||
preview += "\n...<tool_result_guard_truncated>"
|
||||
ok_val = None
|
||||
try:
|
||||
_obj = json.loads(txt)
|
||||
if isinstance(_obj, dict):
|
||||
_ok = _obj.get("ok")
|
||||
ok_val = bool(_ok) if _ok is not None else None
|
||||
except Exception:
|
||||
ok_val = None
|
||||
new_txt = _json_dumps_safe(
|
||||
{
|
||||
"ok": ok_val,
|
||||
"error_code": "",
|
||||
"error": "",
|
||||
"_tool_result_guarded": True,
|
||||
"_history_full_rewrite": True,
|
||||
"original_chars": len(txt),
|
||||
"guard_cap_chars": int(cap),
|
||||
"preview": preview,
|
||||
"hint": (
|
||||
"Tool result history was compacted by operator action. / "
|
||||
"该工具结果历史已按运维操作统一压缩。"
|
||||
),
|
||||
}
|
||||
)
|
||||
changed = (new_txt != txt)
|
||||
if changed:
|
||||
rewritten_all += 1
|
||||
if not changed:
|
||||
# could be already guarded, or under cap
|
||||
if txt and ("_tool_result_guarded" in txt or "_image_tool_result_guarded" in txt or "_video_tool_result_guarded" in txt):
|
||||
skipped += 1
|
||||
continue
|
||||
conn.execute("update chat_message set content=? where id=? and session_id=?", (new_txt, int(mid), sid))
|
||||
compacted += 1
|
||||
|
||||
return HistoryCompactionResult(
|
||||
ok=True,
|
||||
session_id=sid,
|
||||
scanned_tool_messages=int(scanned),
|
||||
compacted_tool_messages=int(compacted),
|
||||
rewritten_all_tool_messages=int(rewritten_all),
|
||||
skipped_already_guarded=int(skipped),
|
||||
max_original_chars_seen=int(max_seen),
|
||||
cap_chars=int(cap),
|
||||
detail="ok",
|
||||
)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"HistoryCompactionResult",
|
||||
"compact_tool_results_in_session_history",
|
||||
]
|
||||
|
||||
105
runtime/chat/session_context_diagnostics.py
Normal file
105
runtime/chat/session_context_diagnostics.py
Normal file
|
|
@ -0,0 +1,105 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class SessionContextStats:
|
||||
session_id: str
|
||||
total_messages: int
|
||||
sampled_messages: int
|
||||
last_n: int
|
||||
|
||||
last_n_total_chars: int
|
||||
last_n_tool_total_chars: int
|
||||
last_n_max_msg_chars: int
|
||||
last_n_max_tool_chars: int
|
||||
last_n_max_user_chars: int
|
||||
|
||||
empty_assistant_text_in_sampled: int
|
||||
empty_assistant_text_ids: tuple[int, ...]
|
||||
|
||||
max_content_chars_in_sampled: int
|
||||
max_tool_chars_in_sampled: int
|
||||
|
||||
|
||||
def compute_session_context_stats(
|
||||
*,
|
||||
store: Any,
|
||||
session_id: str,
|
||||
sample_n: int = 120,
|
||||
last_n: int = 80,
|
||||
) -> SessionContextStats:
|
||||
"""Compute lightweight DB-backed stats for diagnosing context overflow vs empty responses.
|
||||
|
||||
This intentionally only inspects DB text lengths (not token counts).
|
||||
"""
|
||||
sid = str(session_id or "").strip()
|
||||
if not sid:
|
||||
raise ValueError("session_id required")
|
||||
sample_n = max(1, min(int(sample_n or 120), 2000))
|
||||
last_n = max(1, min(int(last_n or 80), 2000))
|
||||
|
||||
total = 0
|
||||
sampled_rows: list[tuple[int, str, str, int]] = []
|
||||
last_rows: list[tuple[int, str, str, int]] = []
|
||||
try:
|
||||
# Prefer raw SQL to avoid store-level object hydration overhead.
|
||||
with store._connect() as conn: # noqa: SLF001
|
||||
cur = conn.execute("select count(1) from chat_message where session_id=?", (sid,))
|
||||
total = int((cur.fetchone() or [0])[0] or 0)
|
||||
cur = conn.execute(
|
||||
"select id, role, event_type, coalesce(length(content),0) as n "
|
||||
"from chat_message where session_id=? order by id desc limit ?",
|
||||
(sid, int(sample_n)),
|
||||
)
|
||||
sampled_rows = [(int(r[0]), str(r[1] or ""), str(r[2] or ""), int(r[3] or 0)) for r in (cur.fetchall() or [])]
|
||||
cur = conn.execute(
|
||||
"select id, role, event_type, coalesce(length(content),0) as n "
|
||||
"from chat_message where session_id=? order by id desc limit ?",
|
||||
(sid, int(last_n)),
|
||||
)
|
||||
last_rows = [(int(r[0]), str(r[1] or ""), str(r[2] or ""), int(r[3] or 0)) for r in (cur.fetchall() or [])]
|
||||
except Exception:
|
||||
# Fallback path via store API if direct SQL fails for any reason.
|
||||
msgs = list(store.get_messages(session_id=sid, limit=int(sample_n)))
|
||||
total = len(msgs)
|
||||
sampled_rows = [
|
||||
(int(getattr(m, "id", 0) or 0), str(getattr(m, "role", "") or ""), str(getattr(m, "event_type", "") or ""), len(str(getattr(m, "content", "") or "")))
|
||||
for m in msgs
|
||||
]
|
||||
last_rows = sampled_rows[: int(last_n)]
|
||||
|
||||
empty_ids = tuple(sorted([mid for mid, role, ev, n in sampled_rows if role == "assistant" and ev == "assistant_text" and int(n) == 0]))
|
||||
max_content = max([n for *_rest, n in sampled_rows] or [0])
|
||||
max_tool = max([n for _mid, role, _ev, n in sampled_rows if role == "tool"] or [0])
|
||||
|
||||
last_total = sum(int(n) for *_rest, n in last_rows)
|
||||
last_tool_total = sum(int(n) for _mid, role, _ev, n in last_rows if role == "tool")
|
||||
last_max = max([int(n) for *_rest, n in last_rows] or [0])
|
||||
last_tool_max = max([int(n) for _mid, role, _ev, n in last_rows if role == "tool"] or [0])
|
||||
last_user_max = max([int(n) for _mid, role, _ev, n in last_rows if role == "user"] or [0])
|
||||
|
||||
return SessionContextStats(
|
||||
session_id=sid,
|
||||
total_messages=int(total),
|
||||
sampled_messages=len(sampled_rows),
|
||||
last_n=int(last_n),
|
||||
last_n_total_chars=int(last_total),
|
||||
last_n_tool_total_chars=int(last_tool_total),
|
||||
last_n_max_msg_chars=int(last_max),
|
||||
last_n_max_tool_chars=int(last_tool_max),
|
||||
last_n_max_user_chars=int(last_user_max),
|
||||
empty_assistant_text_in_sampled=len(empty_ids),
|
||||
empty_assistant_text_ids=empty_ids,
|
||||
max_content_chars_in_sampled=int(max_content),
|
||||
max_tool_chars_in_sampled=int(max_tool),
|
||||
)
|
||||
|
||||
|
||||
__all__ = [
|
||||
"SessionContextStats",
|
||||
"compute_session_context_stats",
|
||||
]
|
||||
|
||||
45
runtime/chat/tool_invocation_context.py
Normal file
45
runtime/chat/tool_invocation_context.py
Normal file
|
|
@ -0,0 +1,45 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from contextlib import contextmanager
|
||||
from contextvars import ContextVar
|
||||
from typing import Iterator
|
||||
|
||||
_tool_lane_owner: ContextVar[str | None] = ContextVar("tool_lane_owner", default=None)
|
||||
_tool_lane_session: ContextVar[str | None] = ContextVar("tool_lane_session", default=None)
|
||||
_tool_workspace_lane_role: ContextVar[str | None] = ContextVar("tool_workspace_lane_role", default=None)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def tool_workspace_lane_scope(
|
||||
*,
|
||||
workspace_owner_session_id: str | None,
|
||||
session_id: str | None,
|
||||
workspace_lane_role: str | None = None,
|
||||
) -> Iterator[None]:
|
||||
o = str(workspace_owner_session_id or "").strip() or None
|
||||
s = str(session_id or "").strip() or None
|
||||
r = str(workspace_lane_role or "").strip().lower() or None
|
||||
t_o = _tool_lane_owner.set(o)
|
||||
t_s = _tool_lane_session.set(s)
|
||||
t_r = _tool_workspace_lane_role.set(r)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
_tool_lane_owner.reset(t_o)
|
||||
_tool_lane_session.reset(t_s)
|
||||
_tool_workspace_lane_role.reset(t_r)
|
||||
|
||||
|
||||
def current_tool_lane_sessions() -> tuple[str | None, str | None]:
|
||||
return _tool_lane_owner.get(), _tool_lane_session.get()
|
||||
|
||||
|
||||
def current_tool_workspace_lane_role() -> str | None:
|
||||
return _tool_workspace_lane_role.get()
|
||||
|
||||
|
||||
__all__ = [
|
||||
"current_tool_lane_sessions",
|
||||
"current_tool_workspace_lane_role",
|
||||
"tool_workspace_lane_scope",
|
||||
]
|
||||
|
|
@ -26,6 +26,7 @@ from oclaw.runtime.tools.path_guard import (
|
|||
workspace_path_access_scope,
|
||||
workspace_write_namespace_scope,
|
||||
)
|
||||
from oclaw.runtime.chat.tool_invocation_context import tool_workspace_lane_scope
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
_tool_exec_log = logging.getLogger("oclaw.tool_exec")
|
||||
|
|
@ -474,6 +475,8 @@ class ToolExecutionContext:
|
|||
path_policy_user_id: str | None = None
|
||||
workspace_dir: str | None = None
|
||||
turn_uuid: str | None = None
|
||||
#: Binding role for private ``skill_auto_install`` paths (``_workspace/<role>/``, sibling of ``public/``).
|
||||
workspace_lane_role: str | None = None
|
||||
|
||||
|
||||
class ToolExecutor:
|
||||
|
|
@ -516,7 +519,11 @@ class ToolExecutor:
|
|||
owner_fallback_session_id=ctx.workspace_owner_session_id,
|
||||
allowlist_tenant_id=ctx.path_policy_tenant_id,
|
||||
allowlist_user_id=ctx.path_policy_user_id,
|
||||
), workspace_write_namespace_scope(ws_ns):
|
||||
), workspace_write_namespace_scope(ws_ns), tool_workspace_lane_scope(
|
||||
workspace_owner_session_id=ctx.workspace_owner_session_id,
|
||||
session_id=ctx.session_id,
|
||||
workspace_lane_role=ctx.workspace_lane_role,
|
||||
):
|
||||
return tool.handler(tc.arguments)
|
||||
|
||||
if isinstance(timeout_s, (int, float)) and float(timeout_s) > 0:
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ from pathlib import Path
|
|||
from types import SimpleNamespace
|
||||
from typing import Any, Callable, Optional
|
||||
|
||||
from oclaw.runtime.chat.agent_messages import build_llm_messages
|
||||
from oclaw.runtime.chat.agent_messages import build_llm_messages, get_last_build_llm_messages_stats
|
||||
from oclaw.runtime.chat.media_redact import redact_embedded_image_blobs
|
||||
from oclaw.runtime.chat.tool_runtime import ToolExecutionConfig
|
||||
from oclaw.runtime.chat.turn_types import TurnRunOutcome
|
||||
|
|
@ -32,6 +32,7 @@ _OCLAW_IMAGE_TOOL_RESULT_REPLAY_CAP_CHARS = 4_000
|
|||
_DIRECT_LOOP_OC_STAGE: dict[str, str] = {
|
||||
"tool_wire_filter": "wire_filter",
|
||||
"tool_result_context_guard": "tool_context_guard",
|
||||
"tool_pairing_guard": "tool_pairing_guard",
|
||||
}
|
||||
_THINK_BLOCK_RE = re.compile(r"<(think|redacted_thinking)>\s*(.*?)\s*</\1>\s*", flags=re.IGNORECASE | re.DOTALL)
|
||||
_TOOL_WIRE_CACHE_LOCK = threading.Lock()
|
||||
|
|
@ -559,6 +560,7 @@ def _build_model_context(
|
|||
attempt_no: int | None = None,
|
||||
workspace_dir: str | None = None,
|
||||
skill_binding_role: str | None = None,
|
||||
workspace_owner_session_id: str | None = None,
|
||||
user_text: str = "",
|
||||
prompt_build_context: dict[str, Any] | None = None,
|
||||
active_turn_uuid: str | None = None,
|
||||
|
|
@ -590,6 +592,8 @@ def _build_model_context(
|
|||
lang=lang,
|
||||
workspace_dir=workspace_dir,
|
||||
skill_binding_role=skill_binding_role,
|
||||
workspace_owner_session_id=workspace_owner_session_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
# Hook integration: wiki-auto-inject can prepend retrieval snippets
|
||||
# before prompt build when query/topic hints indicate supplemental lookup.
|
||||
|
|
@ -617,7 +621,7 @@ def _build_model_context(
|
|||
pass
|
||||
trunc_raw = str(store.get_setting("AIA_TOOL_CONTEXT_TRUNCATE_ENABLED") or "").strip().lower()
|
||||
tool_context_truncate_enabled = trunc_raw not in ("0", "false", "no", "off")
|
||||
return build_llm_messages(
|
||||
llm_messages = build_llm_messages(
|
||||
store_messages=rows,
|
||||
system_prompt=final_system,
|
||||
model=model,
|
||||
|
|
@ -625,6 +629,30 @@ def _build_model_context(
|
|||
tool_context_truncate_enabled=tool_context_truncate_enabled,
|
||||
active_turn_uuid=active_turn_uuid,
|
||||
)
|
||||
try:
|
||||
stats = get_last_build_llm_messages_stats()
|
||||
dropped_unpaired = int(stats.get("dropped_unpaired_tool_rows") or 0)
|
||||
dropped_no_id = int(stats.get("dropped_no_id_tool_rows") or 0)
|
||||
dropped_total = dropped_unpaired + dropped_no_id
|
||||
if dropped_total > 0 and trace_id:
|
||||
_emit_direct_loop_trace(
|
||||
store=store,
|
||||
session_id=session_id,
|
||||
trace_id=trace_id,
|
||||
parent_span_id=parent_span_id,
|
||||
event_type="tool_pairing_guard",
|
||||
payload={
|
||||
"dropped_total": int(dropped_total),
|
||||
"dropped_unpaired_tool_rows": int(dropped_unpaired),
|
||||
"dropped_no_id_tool_rows": int(dropped_no_id),
|
||||
},
|
||||
run_id=run_id,
|
||||
attempt_no=attempt_no,
|
||||
lang=lang,
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
return llm_messages
|
||||
|
||||
|
||||
def _prepare_llm_tools(
|
||||
|
|
@ -815,6 +843,10 @@ def _persist_assistant_step(
|
|||
|
||||
reasoning_chunks, assistant_body = _split_reasoning_and_body(assistant_text, explicit_reasoning=reasoning_text)
|
||||
reasoning_full = "\n".join([str(x or "").strip() for x in reasoning_chunks if str(x or "").strip()]).strip()
|
||||
if not str(assistant_body or "").strip() and not stored_tool_calls:
|
||||
# Provider/model can occasionally return an empty body; persist a visible stub
|
||||
# so UI doesn't look "stuck" and operators can diagnose from history.
|
||||
assistant_body = "(空响应)模型返回了空内容,请重试一次;若持续出现,请检查模型网关/上游返回。"
|
||||
if not thinking_mode_enabled:
|
||||
for idx, chunk in enumerate(reasoning_chunks):
|
||||
store.add_message(
|
||||
|
|
@ -860,6 +892,7 @@ def _execute_tool_step(
|
|||
on_tool_ui: Optional[Callable[[str, dict[str, Any]], None]],
|
||||
should_stop: Optional[Callable[[], bool]],
|
||||
signature_budget: int,
|
||||
workspace_lane_role: str | None = None,
|
||||
run_id: str | None = None,
|
||||
attempt_no: int | None = None,
|
||||
turn_uuid: str | None = None,
|
||||
|
|
@ -882,6 +915,7 @@ def _execute_tool_step(
|
|||
run_id=run_id,
|
||||
attempt_no=attempt_no,
|
||||
turn_uuid=turn_uuid,
|
||||
workspace_lane_role=workspace_lane_role,
|
||||
),
|
||||
assistant_msg_id=assistant_msg_id,
|
||||
skill_uses=llm_tool_calls,
|
||||
|
|
@ -945,6 +979,7 @@ def run_oclaw_direct_loop(
|
|||
tool_traces: list[dict[str, Any]] = []
|
||||
final_text = ""
|
||||
hit_tool_round_limit = False
|
||||
workspace_lane_role = str(skill_binding_role or wire_policy_role or "generalist").strip().lower() or "generalist"
|
||||
|
||||
base_url = str(getattr(model, "base_url", "") or "")
|
||||
|
||||
|
|
@ -970,6 +1005,7 @@ def run_oclaw_direct_loop(
|
|||
attempt_no=attempt_no,
|
||||
workspace_dir=workspace_dir,
|
||||
skill_binding_role=skill_binding_role,
|
||||
workspace_owner_session_id=workspace_owner_session_id,
|
||||
user_text=str(user_text or ""),
|
||||
prompt_build_context=prompt_build_context,
|
||||
active_turn_uuid=turn_uuid,
|
||||
|
|
@ -1026,6 +1062,7 @@ def run_oclaw_direct_loop(
|
|||
on_tool_ui=on_tool_ui,
|
||||
should_stop=should_stop,
|
||||
signature_budget=tool_signature_budget,
|
||||
workspace_lane_role=workspace_lane_role,
|
||||
run_id=run_id,
|
||||
attempt_no=attempt_no,
|
||||
turn_uuid=turn_uuid,
|
||||
|
|
@ -1066,6 +1103,7 @@ def run_oclaw_direct_loop(
|
|||
attempt_no=attempt_no,
|
||||
workspace_dir=workspace_dir,
|
||||
skill_binding_role=skill_binding_role,
|
||||
workspace_owner_session_id=workspace_owner_session_id,
|
||||
user_text=str(user_text or ""),
|
||||
prompt_build_context=prompt_build_context,
|
||||
active_turn_uuid=turn_uuid,
|
||||
|
|
|
|||
|
|
@ -9,7 +9,7 @@ from oclaw.platform.config.paths import PROJECT_ROOT, db_path
|
|||
|
||||
_DEFAULT_ALLOWLIST = (
|
||||
"BRAVE_API_KEY,GOOGLE_OAUTH_CREDENTIALS,GOOGLE_CALENDAR_MCP_TOKEN_PATH,"
|
||||
"GITHUB_PERSONAL_ACCESS_TOKEN,CONTEXT7_API_KEY"
|
||||
"GITHUB_PERSONAL_ACCESS_TOKEN,CONTEXT7_API_KEY,DASHSCOPE_API_KEY"
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -33,6 +33,7 @@ class SkillExecutionContext:
|
|||
attempt_no: int | None = None
|
||||
turn_uuid: str | None = None
|
||||
hook_eligibility: HookEligibilityContext | None = None
|
||||
workspace_lane_role: str | None = None
|
||||
|
||||
|
||||
class SkillExecutor:
|
||||
|
|
@ -169,6 +170,7 @@ class SkillExecutor:
|
|||
path_policy_user_id=ctx.path_policy_user_id,
|
||||
workspace_dir=ctx.workspace_dir,
|
||||
turn_uuid=ctx.turn_uuid,
|
||||
workspace_lane_role=ctx.workspace_lane_role,
|
||||
),
|
||||
assistant_msg_id=assistant_msg_id,
|
||||
tool_uses=skill_uses,
|
||||
|
|
|
|||
|
|
@ -29,6 +29,7 @@ from oclaw.runtime.skills import (
|
|||
discover_workspace_skill_manifests,
|
||||
load_skill_manifest,
|
||||
)
|
||||
from oclaw.runtime.skills_workspace_lane import skills_home_containing_workspace_lane
|
||||
|
||||
_DISABLED_SKILLS_KEY = "AIA_SKILL_DISABLED_NAMES"
|
||||
_AUTO_INSTALL_KEY = "AIA_SKILL_AUTO_INSTALL_ENABLED"
|
||||
|
|
@ -513,7 +514,7 @@ def install_skill_from_local_dir(
|
|||
auto_enabled = False
|
||||
binding_roles: tuple[str, ...] = ()
|
||||
if auto_bind:
|
||||
binding_root = root.parent if str(root.name).strip().lower() == "_workspace" else root
|
||||
binding_root = skills_home_containing_workspace_lane(root)
|
||||
auto_enabled, binding_roles = _apply_auto_enable_binding(store=store, skill_name=manifest.name, skills_root=binding_root)
|
||||
deps_ok, deps_detail = _auto_install_skill_dependencies(store=store, skill_dir=target)
|
||||
if not deps_ok:
|
||||
|
|
@ -748,6 +749,8 @@ def auto_install_skill_from_payload(
|
|||
store: Any,
|
||||
payload: dict[str, Any],
|
||||
skills_root: str | Path | None = None,
|
||||
workspace_install_parent: str | Path | None = None,
|
||||
auto_bind: bool | None = None,
|
||||
) -> SkillInstallResult:
|
||||
if not skill_auto_install_enabled(store):
|
||||
ec, rt = _classify_install_detail("auto_install_disabled")
|
||||
|
|
@ -757,9 +760,13 @@ def auto_install_skill_from_payload(
|
|||
body = str(payload.get("body_markdown") or "").strip()
|
||||
md = payload.get("metadata_oclaw")
|
||||
md = dict(md) if isinstance(md, dict) else {}
|
||||
root = Path(skills_root).resolve() if skills_root else default_skills_root()
|
||||
skills_home = Path(skills_root).resolve() if skills_root else default_skills_root()
|
||||
if workspace_install_parent is not None:
|
||||
workspace_root = Path(workspace_install_parent).resolve()
|
||||
else:
|
||||
workspace_root = skills_home / "_workspace"
|
||||
bind = True if auto_bind is None else bool(auto_bind)
|
||||
# Keep agent-authored/auto-installed skills isolated from primary managed skills.
|
||||
workspace_root = root / "_workspace"
|
||||
target = workspace_root / name
|
||||
before_disabled = _get_disabled_names(store)
|
||||
try:
|
||||
|
|
@ -776,7 +783,12 @@ def auto_install_skill_from_payload(
|
|||
raise RuntimeError("forced_error_for_test")
|
||||
if not out.ok:
|
||||
return out
|
||||
auto_enabled, roles = _apply_auto_enable_binding(store=store, skill_name=name, skills_root=root)
|
||||
if bind:
|
||||
auto_enabled, roles = _apply_auto_enable_binding(
|
||||
store=store, skill_name=name, skills_root=skills_home_containing_workspace_lane(workspace_root)
|
||||
)
|
||||
else:
|
||||
auto_enabled, roles = False, ()
|
||||
return SkillInstallResult(
|
||||
ok=out.ok,
|
||||
name=out.name,
|
||||
|
|
|
|||
|
|
@ -13,10 +13,11 @@
|
|||
## 目录分层(重要)
|
||||
- **主目录**:`oclaw/runtime/skills/<skill_name>/`
|
||||
用于官方/手工管理的稳定技能(安装、维护、评审都在这层)。
|
||||
- **自写目录**:`oclaw/runtime/skills/_workspace/<skill_name>/`
|
||||
用于 agent 自写/自动安装技能,和主目录隔离,避免混放。
|
||||
- **自写目录(扁平)**:`oclaw/runtime/skills/_workspace/<skill_name>/`
|
||||
兼容旧行为;新装技能优先按角色分桶(见下)。
|
||||
- **公共目录**:`oclaw/runtime/skills/_workspace/public/<skill_name>/`
|
||||
放在这里的 skill 默认对所有人可用,不需要做角色绑定(适合通用能力)。
|
||||
与 **按专家/角色目录** `oclaw/runtime/skills/_workspace/<role>/<skill_name>/`(如 `generalist`、`ops`)**平级**;`public` 下技能默认全员可用,不需角色绑定。
|
||||
- **遗留会话桶**:`oclaw/runtime/skills/_workspace/_agent/<segment>/` 仅用于无绑定角色时的回退(如旧会话 id),新逻辑不应依赖该层。
|
||||
|
||||
说明:
|
||||
- `auto_install_skill_from_payload` 产物默认落在 `_workspace` 下。
|
||||
|
|
|
|||
|
|
@ -0,0 +1,267 @@
|
|||
---
|
||||
name: automation-workflows
|
||||
description: Design and implement automation workflows to save time and scale operations as a solopreneur. Use when identifying repetitive tasks to automate, building workflows across tools, setting up triggers and actions, or optimizing existing automations. Covers automation opportunity identification, workflow design, tool selection (Zapier, Make, n8n), testing, and maintenance. Trigger on "automate", "automation", "workflow automation", "save time", "reduce manual work", "automate my business", "no-code automation".
|
||||
---
|
||||
|
||||
# Automation Workflows
|
||||
|
||||
## Overview
|
||||
As a solopreneur, your time is your most valuable asset. Automation lets you scale without hiring. The goal is simple: automate anything you do more than twice a week that doesn't require creative thinking. This playbook shows you how to identify automation opportunities, design workflows, and implement them without writing code.
|
||||
|
||||
---
|
||||
|
||||
## Step 1: Identify What to Automate
|
||||
|
||||
Not every task should be automated. Start by finding the highest-value opportunities.
|
||||
|
||||
**Automation audit (spend 1 hour on this):**
|
||||
|
||||
1. Track every task you do for a week (use a notebook or simple spreadsheet)
|
||||
2. For each task, note:
|
||||
- How long it takes
|
||||
- How often you do it (daily, weekly, monthly)
|
||||
- Whether it's repetitive or requires judgment
|
||||
|
||||
3. Calculate time cost per task:
|
||||
```
|
||||
Time Cost = (Minutes per task × Frequency per month) / 60
|
||||
```
|
||||
Example: 15 min task done 20x/month = 5 hours/month
|
||||
|
||||
4. Sort by time cost (highest to lowest)
|
||||
|
||||
**Good candidates for automation:**
|
||||
- Repetitive (same steps every time)
|
||||
- Rule-based (no complex judgment calls)
|
||||
- High-frequency (daily or weekly)
|
||||
- Time-consuming (takes 10+ minutes)
|
||||
|
||||
**Examples:**
|
||||
- ✅ Sending weekly reports to clients (same format, same schedule)
|
||||
- ✅ Creating invoices after payment
|
||||
- ✅ Adding new leads to CRM from form submissions
|
||||
- ✅ Posting social media content on a schedule
|
||||
- ❌ Conducting customer discovery interviews (requires nuance)
|
||||
- ❌ Writing custom proposals for clients (requires creativity)
|
||||
|
||||
**Low-hanging fruit checklist (start here):**
|
||||
- [ ] Email notifications for form submissions
|
||||
- [ ] Auto-save form responses to spreadsheet
|
||||
- [ ] Schedule social posts in advance
|
||||
- [ ] Auto-create invoices from payment confirmations
|
||||
- [ ] Sync data between tools (CRM ↔ email tool ↔ spreadsheet)
|
||||
|
||||
---
|
||||
|
||||
## Step 2: Choose Your Automation Tool
|
||||
|
||||
Three main options for no-code automation. Pick based on complexity and budget.
|
||||
|
||||
**Tool comparison:**
|
||||
|
||||
| Tool | Best For | Pricing | Learning Curve | Power Level |
|
||||
|---|---|---|---|---|
|
||||
| **Zapier** | Simple, 2-3 step workflows | $20-50/month | Easy | Low-Medium |
|
||||
| **Make (Integromat)** | Visual, multi-step workflows | $9-30/month | Medium | Medium-High |
|
||||
| **n8n** | Complex, developer-friendly, self-hosted | Free (self-hosted) or $20/month | Medium-Hard | High |
|
||||
|
||||
**Selection guide:**
|
||||
- Budget < $20/month → Try Zapier free tier or n8n self-hosted
|
||||
- Need visual workflow builder → Make
|
||||
- Simple 2-step workflows → Zapier
|
||||
- Complex workflows with branching logic → Make or n8n
|
||||
- Want full control and customization → n8n
|
||||
|
||||
**Recommendation for solopreneurs:** Start with Zapier (easiest to learn). Graduate to Make or n8n when you hit Zapier's limits.
|
||||
|
||||
---
|
||||
|
||||
## Step 3: Design Your Workflow
|
||||
|
||||
Before building, map out the workflow on paper or a whiteboard.
|
||||
|
||||
**Workflow design template:**
|
||||
|
||||
```
|
||||
TRIGGER: What event starts the workflow?
|
||||
Example: "New row added to Google Sheet"
|
||||
|
||||
CONDITIONS (optional): Should this workflow run every time, or only when certain conditions are met?
|
||||
Example: "Only if Status column = 'Approved'"
|
||||
|
||||
ACTIONS: What should happen as a result?
|
||||
Step 1: [action]
|
||||
Step 2: [action]
|
||||
Step 3: [action]
|
||||
|
||||
ERROR HANDLING: What happens if something fails?
|
||||
Example: "Send me a Slack message if action fails"
|
||||
```
|
||||
|
||||
**Example workflow (lead capture → CRM → email):**
|
||||
```
|
||||
TRIGGER: New form submission on website
|
||||
|
||||
CONDITIONS: Email field is not empty
|
||||
|
||||
ACTIONS:
|
||||
Step 1: Add lead to CRM (e.g., Airtable or HubSpot)
|
||||
Step 2: Send welcome email via email tool (e.g., ConvertKit)
|
||||
Step 3: Create task in project management tool (e.g., Notion) to follow up in 3 days
|
||||
Step 4: Send me a Slack notification: "New lead: [Name]"
|
||||
|
||||
ERROR HANDLING: If Step 1 fails, send email alert to me
|
||||
```
|
||||
|
||||
**Design principles:**
|
||||
- Keep it simple — start with 2-3 steps, add complexity later
|
||||
- Test each step individually before chaining them together
|
||||
- Add delays between actions if needed (some APIs are slow)
|
||||
- Always include error notifications so you know when things break
|
||||
|
||||
---
|
||||
|
||||
## Step 4: Build and Test Your Workflow
|
||||
|
||||
Now implement it in your chosen tool.
|
||||
|
||||
**Build workflow (Zapier example):**
|
||||
1. **Choose trigger app** (e.g., Google Forms, Typeform, website form)
|
||||
2. **Connect your account** (authenticate via OAuth)
|
||||
3. **Test trigger** (submit a test form to make sure data comes through)
|
||||
4. **Add action** (e.g., "Add row to Google Sheets")
|
||||
5. **Map fields** (match form fields to spreadsheet columns)
|
||||
6. **Test action** (run test to verify row is added correctly)
|
||||
7. **Repeat for additional actions**
|
||||
8. **Turn on workflow** (Zapier calls this "turn on Zap")
|
||||
|
||||
**Testing checklist:**
|
||||
- [ ] Submit test data through the trigger
|
||||
- [ ] Verify each action executes correctly
|
||||
- [ ] Check that data maps to the right fields
|
||||
- [ ] Test with edge cases (empty fields, special characters, long text)
|
||||
- [ ] Test error handling (intentionally cause a failure to see if alerts work)
|
||||
|
||||
**Common issues and fixes:**
|
||||
|
||||
| Issue | Cause | Fix |
|
||||
|---|---|---|
|
||||
| Workflow doesn't trigger | Trigger conditions too narrow | Check filter settings, broaden criteria |
|
||||
| Action fails | API rate limit or permissions | Add delay between actions, re-authenticate |
|
||||
| Data missing or incorrect | Field mapping wrong | Double-check which fields are mapped |
|
||||
| Workflow runs multiple times | Duplicate triggers | De-duplicate based on unique ID |
|
||||
|
||||
**Rule:** Test with real data before relying on an automation. Don't discover bugs when a real customer is involved.
|
||||
|
||||
---
|
||||
|
||||
## Step 5: Monitor and Maintain Automations
|
||||
|
||||
Automations aren't set-it-and-forget-it. They break. Tools change. APIs update. You need a maintenance plan.
|
||||
|
||||
**Weekly check (5 min):**
|
||||
- Scan workflow logs for errors (most tools show a log of runs + failures)
|
||||
- Address any failures immediately
|
||||
|
||||
**Monthly audit (15 min):**
|
||||
- Review all active workflows
|
||||
- Check: Is this still being used? Is it still saving time?
|
||||
- Disable or delete unused workflows (they clutter your dashboard and can cause confusion)
|
||||
- Update any workflows that depend on tools you've switched away from
|
||||
|
||||
**Where to store workflow documentation:**
|
||||
- Create a simple doc (Notion, Google Doc) for each workflow
|
||||
- Include: What it does, when it runs, what apps it connects, how to troubleshoot
|
||||
- If you have 10+ workflows, this doc will save you hours when something breaks
|
||||
|
||||
**Error handling setup:**
|
||||
- Route all error notifications to one place (Slack channel, email inbox, or task manager)
|
||||
- Set up: "If any workflow fails, send a message to [your error channel]"
|
||||
- Review errors weekly and fix root causes
|
||||
|
||||
---
|
||||
|
||||
## Step 6: Advanced Automation Ideas
|
||||
|
||||
Once you've automated the basics, consider these higher-leverage workflows:
|
||||
|
||||
### Client onboarding automation
|
||||
```
|
||||
TRIGGER: New client signs contract (via DocuSign, HelloSign)
|
||||
ACTIONS:
|
||||
1. Create project in project management tool
|
||||
2. Add client to CRM with "Active" status
|
||||
3. Send onboarding email sequence
|
||||
4. Create invoice in accounting software
|
||||
5. Schedule kickoff call on calendar
|
||||
6. Add client to Slack workspace (if applicable)
|
||||
```
|
||||
|
||||
### Content distribution automation
|
||||
```
|
||||
TRIGGER: New blog post published on website (via RSS or webhook)
|
||||
ACTIONS:
|
||||
1. Post link to LinkedIn with auto-generated caption
|
||||
2. Post link to Twitter as a thread
|
||||
3. Add post to email newsletter draft (in email tool)
|
||||
4. Add to content calendar (Notion or Airtable)
|
||||
5. Send notification to team (Slack) that post is live
|
||||
```
|
||||
|
||||
### Customer health monitoring
|
||||
```
|
||||
TRIGGER: Every Monday at 9am (scheduled trigger)
|
||||
ACTIONS:
|
||||
1. Pull usage data for all customers from database (via API)
|
||||
2. Flag customers with <50% of average usage
|
||||
3. Add flagged customers to "At Risk" segment in CRM
|
||||
4. Send re-engagement email campaign to at-risk customers
|
||||
5. Create task for me to personally reach out to top 10 at-risk customers
|
||||
```
|
||||
|
||||
### Invoice and payment tracking
|
||||
```
|
||||
TRIGGER: Payment received (Stripe webhook)
|
||||
ACTIONS:
|
||||
1. Mark invoice as paid in accounting software
|
||||
2. Send receipt email to customer
|
||||
3. Update CRM: customer status = "Paid"
|
||||
4. Add revenue to monthly dashboard (Google Sheets or Airtable)
|
||||
5. Send me a Slack notification: "Payment received: $X from [Customer]"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Step 7: Calculate Automation ROI
|
||||
|
||||
Not every automation is worth the time investment. Calculate ROI to prioritize.
|
||||
|
||||
**ROI formula:**
|
||||
```
|
||||
Time Saved per Month (hours) = (Minutes per task / 60) × Frequency per month
|
||||
Cost = (Setup time in hours × $50/hour) + Tool cost per month
|
||||
Payback Period (months) = Setup cost / Monthly time saved value
|
||||
|
||||
If payback period < 3 months → Worth it
|
||||
If payback period > 6 months → Probably not worth it (unless it unlocks other value)
|
||||
```
|
||||
|
||||
**Example:**
|
||||
```
|
||||
Task: Manually copying form submissions to CRM (15 min, 20x/month = 5 hours/month saved)
|
||||
Setup time: 1 hour
|
||||
Tool cost: $20/month (Zapier)
|
||||
Payback: ($50 setup cost) / ($250/month value saved) = 0.2 months → Absolutely worth it
|
||||
```
|
||||
|
||||
**Rule:** Focus on automations with payback < 3 months. Those are your highest-leverage investments.
|
||||
|
||||
---
|
||||
|
||||
## Automation Mistakes to Avoid
|
||||
- **Automating before optimizing.** Don't automate a bad process. Fix the process first, then automate it.
|
||||
- **Over-automating.** Not everything needs to be automated. If a task is rare or requires judgment, do it manually.
|
||||
- **No error handling.** If an automation breaks and you don't know, it causes silent failures. Always set up error alerts.
|
||||
- **Not testing thoroughly.** A broken automation is worse than no automation — it creates incorrect data or missed tasks.
|
||||
- **Building too complex too fast.** Start with simple 2-3 step workflows. Add complexity only when the simple version works perfectly.
|
||||
- **Not documenting workflows.** Future you will forget how this works. Write it down.
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn732qfbv22he1jqm63xbwq6e980kn8s",
|
||||
"slug": "automation-workflows",
|
||||
"version": "0.1.0",
|
||||
"publishedAt": 1770341582349
|
||||
}
|
||||
|
|
@ -0,0 +1,18 @@
|
|||
# Changelog
|
||||
|
||||
## v2.1.0 (2026-04-11)
|
||||
- Version bump to 2.1.0
|
||||
|
||||
## v2.0.1 (2026-02-06)
|
||||
- Simplified documentation
|
||||
- Removed gov-related content
|
||||
- Optimized for ClawHub publishing
|
||||
|
||||
## v2.0.0 (2026-02-06)
|
||||
- Added 9 international search engines
|
||||
- Enhanced advanced search capabilities
|
||||
- Added DuckDuckGo Bangs support
|
||||
- Added WolframAlpha knowledge queries
|
||||
|
||||
## v1.0.0 (2026-02-04)
|
||||
- Initial release with 8 domestic search engines
|
||||
|
|
@ -0,0 +1,48 @@
|
|||
# Multi Search Engine
|
||||
|
||||
## 基本信息
|
||||
|
||||
- **名称**: multi-search-engine
|
||||
- **版本**: v2.0.1
|
||||
- **描述**: 集成16个搜索引擎(7国内+9国际),支持高级搜索语法
|
||||
- **发布时间**: 2026-02-06
|
||||
|
||||
## 搜索引擎
|
||||
|
||||
**国内(7个)**: 百度、必应CN、必应INT、360、搜狗、微信、神马
|
||||
**国际(9个)**: Google、Google HK、DuckDuckGo、Yahoo、Startpage、Brave、Ecosia、Qwant、WolframAlpha
|
||||
|
||||
## 核心功能
|
||||
|
||||
- 高级搜索操作符(site:, filetype:, intitle:等)
|
||||
- DuckDuckGo Bangs快捷命令
|
||||
- 时间筛选(小时/天/周/月/年)
|
||||
- 隐私保护搜索
|
||||
- WolframAlpha知识计算
|
||||
|
||||
## 更新记录
|
||||
|
||||
### v2.0.1 (2026-02-06)
|
||||
- 精简文档,优化发布
|
||||
|
||||
### v2.0.0 (2026-02-06)
|
||||
- 新增9个国际搜索引擎
|
||||
- 强化深度搜索能力
|
||||
|
||||
### v1.0.0 (2026-02-04)
|
||||
- 初始版本:8个国内搜索引擎
|
||||
|
||||
## 使用示例
|
||||
|
||||
```javascript
|
||||
// Google搜索
|
||||
web_fetch({"url": "https://www.google.com/search?q=python"})
|
||||
|
||||
// 隐私搜索
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=privacy"})
|
||||
|
||||
// 站内搜索
|
||||
web_fetch({"url": "https://www.google.com/search?q=site:github.com+python"})
|
||||
```
|
||||
|
||||
MIT License
|
||||
|
|
@ -0,0 +1,215 @@
|
|||
---
|
||||
name: "multi-search-engine"
|
||||
description: "Multi search engine integration with 16 engines (7 CN + 9 Global). Supports advanced search operators, time filters, site search, privacy engines, and WolframAlpha knowledge queries. No API keys required."
|
||||
---
|
||||
|
||||
# Multi Search Engine
|
||||
|
||||
Integration of 16 search engines for web crawling without API keys.
|
||||
|
||||
## Workflow
|
||||
|
||||
1. **Preparation**: AI Agent initializes an empty in-memory cookie store. Cookies are only acquired dynamically during search operations when access is denied
|
||||
|
||||
2. **Language Evaluation**: Detect the language attribute of the search query. If the query is in Chinese, use Domestic search engines (Baidu, Bing CN, Bing INT, 360, Sogou, WeChat, Shenma). If the query is non-Chinese, use International search engines (Google, Google HK, DuckDuckGo, Yahoo, Startpage, Brave, Ecosia, Qwant, WolframAlpha). Select engines based on query relevance and availability.
|
||||
|
||||
3. **Controlled Search (Reachability-first)**: Use `web_search_fast` to search and `web_fetch_clean` to read result pages:
|
||||
- Prefer `provider=auto` first (current fallback order: `official_bing/official_google(if configured) -> ddg_html -> bing_html -> ddg_api`)
|
||||
- Prefer `engine` + `site` for deterministic routing only when that engine is known reachable in current environment
|
||||
- Read details only for top 1-3 URLs using `web_fetch_clean`
|
||||
- If result pages fail with anti-bot/SSRF, switch to another source URL instead of blind retries
|
||||
|
||||
4. **Cookie Management**:
|
||||
- Cookies are stored ONLY in memory during runtime
|
||||
- Cookies are acquired on-demand when search requests fail
|
||||
- No cookies are read from or written to config.json or any file
|
||||
- Cookies are cleared after search session completes
|
||||
- Only session cookies from search engine domains are captured
|
||||
|
||||
5. **Retry Mechanism**: If a search fails due to cookie/session issues, retry once with freshly acquired cookies after a 2-second delay
|
||||
|
||||
6. **Result Aggregation**: Consolidate successful results from search engines, organize and summarize them to output a core search report
|
||||
|
||||
## Search Engines
|
||||
|
||||
### Domestic (7)
|
||||
- **Baidu**: `https://www.baidu.com/s?wd={keyword}`
|
||||
- **Bing CN**: `https://cn.bing.com/search?q={keyword}&ensearch=0`
|
||||
- **Bing INT**: `https://cn.bing.com/search?q={keyword}&ensearch=1`
|
||||
- **360**: `https://www.so.com/s?q={keyword}`
|
||||
- **Sogou**: `https://sogou.com/web?query={keyword}`
|
||||
- **WeChat**: `https://wx.sogou.com/weixin?type=2&query={keyword}`
|
||||
- **Shenma**: `https://m.sm.cn/s?q={keyword}`
|
||||
|
||||
### International (9)
|
||||
- **Google**: `https://www.google.com/search?q={keyword}`
|
||||
- **Google HK**: `https://www.google.com.hk/search?q={keyword}`
|
||||
- **DuckDuckGo**: `https://duckduckgo.com/html/?q={keyword}`
|
||||
- **Yahoo**: `https://search.yahoo.com/search?p={keyword}`
|
||||
- **Startpage**: `https://www.startpage.com/sp/search?query={keyword}`
|
||||
- **Brave**: `https://search.brave.com/search?q={keyword}`
|
||||
- **Ecosia**: `https://www.ecosia.org/search?q={keyword}`
|
||||
- **Qwant**: `https://www.qwant.com/?q={keyword}`
|
||||
- **WolframAlpha**: `https://www.wolframalpha.com/input?i={keyword}`
|
||||
|
||||
## Quick Examples (Recommended)
|
||||
|
||||
```javascript
|
||||
// Basic search (auto fallback chain)
|
||||
web_search_fast({"query":"python tutorial","provider":"auto","max_results":8})
|
||||
|
||||
// Official Bing API
|
||||
web_search_fast({"query":"latest ai policy","provider":"official_bing","max_results":8})
|
||||
|
||||
// Official Google Programmable Search API
|
||||
web_search_fast({"query":"latest ai policy","provider":"official_google","max_results":8})
|
||||
|
||||
// Auto + prefer Google first among official providers
|
||||
web_search_fast({"query":"latest ai policy","provider":"auto","official_provider":"google","max_results":8})
|
||||
|
||||
// Site-specific with deterministic engine
|
||||
web_search_fast({"query":"react hooks best practices","engine":"bing","site":"github.com"})
|
||||
|
||||
// Domestic CN engine route
|
||||
web_search_fast({"query":"人工智能 最新 进展","engine":"bing_cn","max_results":10})
|
||||
|
||||
// Full custom engine URL template from config
|
||||
web_search_fast({"query":"privacy tools","engine_url":"https://duckduckgo.com/html/?q={keyword}"})
|
||||
|
||||
// Fetch details for selected result URL
|
||||
web_fetch_clean({"url":"https://www.bing.com/search?q=Iran+news","max_chars":18000})
|
||||
|
||||
// Optional: URL-level fallback when search engines are unstable
|
||||
web_fetch_clean({"url":"https://www.bing.com/search?q=site:reuters.com+iran+news"})
|
||||
```
|
||||
|
||||
## Tool Parameters (web_search_fast)
|
||||
|
||||
- `query`: required search text
|
||||
- `engine`: optional named engine (`bing`, `bing_cn`, `bing_int`, `ddg`, `google`, `google_hk`, `baidu`, `sogou`)
|
||||
- `site`: optional domain constraint; auto-converted to `site:domain query`
|
||||
- `engine_url`: optional URL template with `{keyword}`; overrides `engine`
|
||||
- `provider`: fallback strategy (`auto`, `official_api`, `official_bing`, `official_google`, `ddg_api`, `ddg_html`, `bing_html`) where `auto` routes by availability first
|
||||
- `official_provider`: optional preference hint for official providers (`auto`, `bing`, `google`)
|
||||
- `max_results`: 1..20
|
||||
- Official Bing env: `OCLAW_WEB_SEARCH_BING_API_KEY` + optional `OCLAW_WEB_SEARCH_BING_API_ENDPOINT`
|
||||
- Official Google env: `OCLAW_WEB_SEARCH_GOOGLE_API_KEY` + `OCLAW_WEB_SEARCH_GOOGLE_CSE_ID` + optional `OCLAW_WEB_SEARCH_GOOGLE_API_ENDPOINT`
|
||||
- Backward compatibility: legacy `OCLAW_WEB_SEARCH_OFFICIAL_API_KEY` and `OCLAW_WEB_SEARCH_OFFICIAL_API_ENDPOINT` still map to Bing official provider
|
||||
|
||||
## Official API Setup (Optional, can defer)
|
||||
|
||||
If you do not have official API keys yet, use:
|
||||
|
||||
```javascript
|
||||
web_search_fast({"query":"latest ai policy","provider":"auto","max_results":8})
|
||||
```
|
||||
|
||||
This will skip official providers when keys are missing and continue with HTML/API fallback.
|
||||
|
||||
### Bing official API (optional)
|
||||
|
||||
1. Create a Bing Search resource in Azure portal.
|
||||
2. Open resource page and copy Key + Endpoint.
|
||||
3. Set environment variables:
|
||||
|
||||
```powershell
|
||||
$env:OCLAW_WEB_SEARCH_BING_API_KEY="your_bing_key"
|
||||
$env:OCLAW_WEB_SEARCH_BING_API_ENDPOINT="https://api.bing.microsoft.com/v7.0/search"
|
||||
```
|
||||
|
||||
### Google official API (optional)
|
||||
|
||||
1. Enable Google Custom Search JSON API in Google Cloud.
|
||||
2. Create Programmable Search Engine and get `cx`.
|
||||
3. Create API key and set environment variables:
|
||||
|
||||
```powershell
|
||||
$env:OCLAW_WEB_SEARCH_GOOGLE_API_KEY="your_google_key"
|
||||
$env:OCLAW_WEB_SEARCH_GOOGLE_CSE_ID="your_cse_id"
|
||||
$env:OCLAW_WEB_SEARCH_GOOGLE_API_ENDPOINT="https://customsearch.googleapis.com/customsearch/v1"
|
||||
```
|
||||
|
||||
### Notes
|
||||
|
||||
- Env vars in PowerShell apply to current shell session only.
|
||||
- After changing system-level env vars, restart the runtime process.
|
||||
- If official API is not configured now, keep using `provider=auto` and revisit later.
|
||||
|
||||
### Diagnostics in return payload
|
||||
- `provider_attempts`: each attempted backend with `ok`, `count`, `elapsed_ms`, and error fields
|
||||
- `error_category`: normalized failure category (`ssrf_blocked`, `anti_bot_401_403`, `timeout`, `dns_error`, `network_error`, `upstream_4xx`, `upstream_5xx`, `no_results`, `unknown_error`)
|
||||
|
||||
## Advanced Operators
|
||||
|
||||
| Operator | Example | Description |
|
||||
|----------|---------|-------------|
|
||||
| `site:` | `site:github.com python` | Search within site |
|
||||
| `filetype:` | `filetype:pdf report` | Specific file type |
|
||||
| `""` | `"machine learning"` | Exact match |
|
||||
| `-` | `python -snake` | Exclude term |
|
||||
| `OR` | `cat OR dog` | Either term |
|
||||
|
||||
## Time Filters
|
||||
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `tbs=qdr:h` | Past hour |
|
||||
| `tbs=qdr:d` | Past day |
|
||||
| `tbs=qdr:w` | Past week |
|
||||
| `tbs=qdr:m` | Past month |
|
||||
| `tbs=qdr:y` | Past year |
|
||||
|
||||
## Privacy Engines
|
||||
|
||||
- **DuckDuckGo**: No tracking
|
||||
- **Startpage**: Google results + privacy
|
||||
- **Brave**: Independent index
|
||||
- **Qwant**: EU GDPR compliant
|
||||
|
||||
## Bangs Shortcuts (DuckDuckGo)
|
||||
|
||||
| Bang | Destination |
|
||||
|------|-------------|
|
||||
| `!g` | Google |
|
||||
| `!gh` | GitHub |
|
||||
| `!so` | Stack Overflow |
|
||||
| `!w` | Wikipedia |
|
||||
| `!yt` | YouTube |
|
||||
|
||||
## WolframAlpha Queries (Caveat)
|
||||
|
||||
- WolframAlpha HTML pages are often not suitable as structured answers in blocked environments.
|
||||
- For deterministic math/finance/units, prefer APIs/tools that provide structured outputs.
|
||||
- Use WolframAlpha URL search as a hint source, not as guaranteed machine-readable result.
|
||||
|
||||
## Documentation
|
||||
|
||||
- `references/advanced-search.md` - Domestic search guide
|
||||
- `references/international-search.md` - International search guide
|
||||
- `CHANGELOG.md` - Version history
|
||||
|
||||
## License
|
||||
|
||||
MIT
|
||||
|
||||
## Security & Privacy Notice
|
||||
|
||||
### Cookie Handling
|
||||
- **Purpose**: Cookies are used ONLY to maintain search session state when access is denied (403/429 errors)
|
||||
- **Storage**: Cookies are kept STRICTLY in memory during runtime - NEVER persisted to disk or config files
|
||||
- **Acquisition**: Cookies are acquired on-demand from search engine homepages only when search requests fail
|
||||
- **Scope**: Only session cookies from the specific search engine domain are captured
|
||||
- **Lifecycle**: Cookies are cleared immediately after the search session completes
|
||||
- **No Pre-configuration**: No cookies are loaded from config.json or any external file at startup
|
||||
- **No API Keys**: This tool uses standard web search URLs, no authentication required
|
||||
|
||||
### Crawling Ethics
|
||||
- **Rate Limiting**: Implement reasonable delays between requests (recommend 1-2 seconds)
|
||||
- **Respect robots.txt**: Honor search engine crawling policies
|
||||
- **Terms of Service**: Users are responsible for complying with search engine ToS
|
||||
- **Purpose**: Designed for legitimate search aggregation, not mass data scraping
|
||||
|
||||
### Data Handling
|
||||
- **No Personal Data**: Tool does not collect or transmit user personal information
|
||||
- **Local Execution**: All operations run locally, no external data transmission
|
||||
- **Session Isolation**: Cookies are session-specific and cleared after use
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn79j8kk7fb9w10jh83803j7f180a44m",
|
||||
"slug": "multi-search-engine",
|
||||
"version": "2.1.3",
|
||||
"publishedAt": 1775879953831
|
||||
}
|
||||
|
|
@ -0,0 +1,85 @@
|
|||
{
|
||||
"name": "multi-search-engine",
|
||||
"engines": [
|
||||
{
|
||||
"name": "Baidu",
|
||||
"url": "https://www.baidu.com/s?wd={keyword}",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "Bing CN",
|
||||
"url": "https://cn.bing.com/search?q={keyword}&ensearch=0",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "Bing INT",
|
||||
"url": "https://cn.bing.com/search?q={keyword}&ensearch=1",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "360",
|
||||
"url": "https://www.so.com/s?q={keyword}",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "Sogou",
|
||||
"url": "https://sogou.com/web?query={keyword}",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "WeChat",
|
||||
"url": "https://wx.sogou.com/weixin?type=2&query={keyword}",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "Shenma",
|
||||
"url": "https://m.sm.cn/s?q={keyword}",
|
||||
"region": "cn"
|
||||
},
|
||||
{
|
||||
"name": "Google",
|
||||
"url": "https://www.google.com/search?q={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "Google HK",
|
||||
"url": "https://www.google.com.hk/search?q={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "DuckDuckGo",
|
||||
"url": "https://duckduckgo.com/html/?q={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "Yahoo",
|
||||
"url": "https://search.yahoo.com/search?p={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "Startpage",
|
||||
"url": "https://www.startpage.com/sp/search?query={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "Brave",
|
||||
"url": "https://search.brave.com/search?q={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "Ecosia",
|
||||
"url": "https://www.ecosia.org/search?q={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "Qwant",
|
||||
"url": "https://www.qwant.com/?q={keyword}",
|
||||
"region": "global"
|
||||
},
|
||||
{
|
||||
"name": "WolframAlpha",
|
||||
"url": "https://www.wolframalpha.com/input?i={keyword}",
|
||||
"region": "global"
|
||||
}
|
||||
]
|
||||
}
|
||||
|
|
@ -0,0 +1,7 @@
|
|||
{
|
||||
"name": "multi-search-engine",
|
||||
"version": "2.1.0",
|
||||
"description": "Multi search engine with 16 engines (7 CN + 9 Global). Supports advanced operators, time filters, privacy engines.",
|
||||
"engines": 16,
|
||||
"requires_api_key": false
|
||||
}
|
||||
|
|
@ -0,0 +1,146 @@
|
|||
# 国内搜索引擎深度搜索指南
|
||||
|
||||
## 🔍 百度 (Baidu)
|
||||
|
||||
### 特色功能
|
||||
|
||||
| 功能 | 说明 | URL |
|
||||
|------|------|-----|
|
||||
| **中文优化** | 中文内容索引最全 | `https://www.baidu.com/s?wd={keyword}` |
|
||||
| **百度学术** | 学术资源搜索 | `https://xueshu.baidu.com/s?wd={keyword}` |
|
||||
| **百度新闻** | 新闻聚合 | `https://news.baidu.com/` |
|
||||
|
||||
### 搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 基础搜索
|
||||
web_fetch({"url": "https://www.baidu.com/s?wd=Python教程"})
|
||||
|
||||
// 2. 站内搜索
|
||||
web_fetch({"url": "https://www.baidu.com/s?wd=site:github.com+python"})
|
||||
|
||||
// 3. 文件类型搜索
|
||||
web_fetch({"url": "https://www.baidu.com/s?wd=机器学习+filetype:pdf"})
|
||||
|
||||
// 4. 学术搜索
|
||||
web_fetch({"url": "https://xueshu.baidu.com/s?wd=深度学习+图像识别"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔎 必应中国版 (Bing CN/INT)
|
||||
|
||||
### 特色功能
|
||||
|
||||
| 功能 | 说明 | URL |
|
||||
|------|------|-----|
|
||||
| **中文优化** | `ensearch=0` 中文结果 | `https://cn.bing.com/search?q={keyword}&ensearch=0` |
|
||||
| **国际版** | `ensearch=1` 英文结果 | `https://cn.bing.com/search?q={keyword}&ensearch=1` |
|
||||
| **学术搜索** | 学术资源 | `https://cn.bing.com/academic/search?q={keyword}` |
|
||||
|
||||
### 搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 中文搜索结果
|
||||
web_fetch({"url": "https://cn.bing.com/search?q=人工智能技术&ensearch=0"})
|
||||
|
||||
// 2. 英文搜索结果(使用中国服务器)
|
||||
web_fetch({"url": "https://cn.bing.com/search?q=artificial+intelligence&ensearch=1"})
|
||||
|
||||
// 3. 学术搜索
|
||||
web_fetch({"url": "https://cn.bing.com/academic/search?q=机器学习算法"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔍 360搜索
|
||||
|
||||
### 特色功能
|
||||
|
||||
| 功能 | 说明 | URL |
|
||||
|------|------|-----|
|
||||
| **安全搜索** | 内置安全防护 | 默认开启 |
|
||||
| **基础搜索** | 网页搜索 | `https://www.so.com/s?q={keyword}` |
|
||||
|
||||
### 搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 基础搜索
|
||||
web_fetch({"url": "https://www.so.com/s?q=网络安全"})
|
||||
|
||||
// 2. 站内搜索
|
||||
web_fetch({"url": "https://www.so.com/s?q=site:zhihu.com+python"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔍 搜狗 (Sogou) + 微信搜索
|
||||
|
||||
### 特色功能
|
||||
|
||||
| 功能 | 说明 | URL |
|
||||
|------|------|-----|
|
||||
| **网页搜索** | 通用搜索 | `https://sogou.com/web?query={keyword}` |
|
||||
| **微信公众号** | 搜公众号文章(唯一渠道) | `https://wx.sogou.com/weixin?type=2&query={keyword}` |
|
||||
| **知乎优化** | 知乎内容索引好 | `site:zhihu.com` 配合使用 |
|
||||
|
||||
### 搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 网页搜索
|
||||
web_fetch({"url": "https://sogou.com/web?query=python教程"})
|
||||
|
||||
// 2. 微信公众号文章搜索
|
||||
web_fetch({"url": "https://wx.sogou.com/weixin?type=2&query=Python编程"})
|
||||
|
||||
// 3. 搜索特定公众号
|
||||
web_fetch({"url": "https://wx.sogou.com/weixin?type=2&query=公众号:机器之心"})
|
||||
|
||||
// 4. 知乎内容搜索
|
||||
web_fetch({"url": "https://www.sogou.com/web?query=site:zhihu.com+机器学习"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 📱 神马搜索 (Shenma)
|
||||
|
||||
### 特色功能
|
||||
|
||||
| 功能 | 说明 | URL |
|
||||
|------|------|-----|
|
||||
| **移动优化** | 专注移动端搜索 | `https://m.sm.cn/s?q={keyword}` |
|
||||
| **阿里生态** | 整合阿里系内容 | UC浏览器默认搜索 |
|
||||
|
||||
### 搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 移动端搜索
|
||||
web_fetch({"url": "https://m.sm.cn/s?q=python入门教程"})
|
||||
|
||||
// 2. 移动网站点搜索
|
||||
web_fetch({"url": "https://m.sm.cn/s?q=site:zhuanlan.zhihu.com+AI"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🌍 国内搜索策略
|
||||
|
||||
### 按搜索目标选择引擎
|
||||
|
||||
| 搜索目标 | 首选引擎 | 原因 |
|
||||
|---------|---------|------|
|
||||
| **综合中文内容** | 百度 | 中文索引最全 |
|
||||
| **微信公众号** | 搜狗微信 | 唯一支持公众号搜索 |
|
||||
| **知乎内容** | 搜狗 | 知乎优化好 |
|
||||
| **移动端内容** | 神马 | 移动端优化 |
|
||||
| **学术资源** | 必应学术 | 学术索引 |
|
||||
| **中英文双语** | 必应中国/国际版 | enswitch切换 |
|
||||
| **新闻资讯** | 百度新闻 | 新闻聚合 |
|
||||
|
||||
---
|
||||
|
||||
## 📚 参考资料
|
||||
|
||||
- [百度搜索高级语法](https://baike.baidu.com/item/搜索语法)
|
||||
- [必应搜索技巧](https://cn.bing.com/tips)
|
||||
- [搜狗搜索帮助](https://help.sogou.com/)
|
||||
|
|
@ -0,0 +1,398 @@
|
|||
# 国际搜索引擎深度搜索指南
|
||||
|
||||
## 🔍 Google 深度搜索
|
||||
|
||||
### 1.1 基础高级搜索操作符
|
||||
|
||||
| 操作符 | 功能 | 示例 | URL |
|
||||
|--------|------|------|-----|
|
||||
| `""` | 精确匹配 | `"machine learning"` | `https://www.google.com/search?q=%22machine+learning%22` |
|
||||
| `-` | 排除关键词 | `python -snake` | `https://www.google.com/search?q=python+-snake` |
|
||||
| `OR` | 或运算 | `machine learning OR deep learning` | `https://www.google.com/search?q=machine+learning+OR+deep+learning` |
|
||||
| `*` | 通配符 | `machine * algorithms` | `https://www.google.com/search?q=machine+*+algorithms` |
|
||||
| `()` | 分组 | `(apple OR microsoft) phones` | `https://www.google.com/search?q=(apple+OR+microsoft)+phones` |
|
||||
| `..` | 数字范围 | `laptop $500..$1000` | `https://www.google.com/search?q=laptop+%24500..%241000` |
|
||||
|
||||
### 1.2 站点与文件搜索
|
||||
|
||||
| 操作符 | 功能 | 示例 |
|
||||
|--------|------|------|
|
||||
| `site:` | 站内搜索 | `site:github.com python projects` |
|
||||
| `filetype:` | 文件类型 | `filetype:pdf annual report` |
|
||||
| `inurl:` | URL包含 | `inurl:login admin` |
|
||||
| `intitle:` | 标题包含 | `intitle:"index of" mp3` |
|
||||
| `intext:` | 正文包含 | `intext:password filetype:txt` |
|
||||
| `cache:` | 查看缓存 | `cache:example.com` |
|
||||
| `related:` | 相关网站 | `related:github.com` |
|
||||
| `info:` | 网站信息 | `info:example.com` |
|
||||
|
||||
### 1.3 时间筛选参数
|
||||
|
||||
| 参数 | 含义 | URL示例 |
|
||||
|------|------|---------|
|
||||
| `tbs=qdr:h` | 过去1小时 | `https://www.google.com/search?q=news&tbs=qdr:h` |
|
||||
| `tbs=qdr:d` | 过去24小时 | `https://www.google.com/search?q=news&tbs=qdr:d` |
|
||||
| `tbs=qdr:w` | 过去1周 | `https://www.google.com/search?q=news&tbs=qdr:w` |
|
||||
| `tbs=qdr:m` | 过去1月 | `https://www.google.com/search?q=news&tbs=qdr:m` |
|
||||
| `tbs=qdr:y` | 过去1年 | `https://www.google.com/search?q=news&tbs=qdr:y` |
|
||||
| `tbs=cdr:1,cd_min:1/1/2024,cd_max:12/31/2024` | 自定义日期范围 | 2024年全年 |
|
||||
|
||||
### 1.4 语言和地区筛选
|
||||
|
||||
| 参数 | 功能 | 示例 |
|
||||
|------|------|------|
|
||||
| `hl=en` | 界面语言 | `https://www.google.com/search?q=test&hl=en` |
|
||||
| `lr=lang_zh-CN` | 搜索结果语言 | `https://www.google.com/search?q=test&lr=lang_zh-CN` |
|
||||
| `cr=countryCN` | 国家/地区 | `https://www.google.com/search?q=test&cr=countryCN` |
|
||||
| `gl=us` | 地理位置 | `https://www.google.com/search?q=test&gl=us` |
|
||||
|
||||
### 1.5 特殊搜索类型
|
||||
|
||||
| 类型 | URL | 说明 |
|
||||
|------|-----|------|
|
||||
| 图片搜索 | `https://www.google.com/search?q={keyword}&tbm=isch` | `tbm=isch` 表示图片 |
|
||||
| 新闻搜索 | `https://www.google.com/search?q={keyword}&tbm=nws` | `tbm=nws` 表示新闻 |
|
||||
| 视频搜索 | `https://www.google.com/search?q={keyword}&tbm=vid` | `tbm=vid` 表示视频 |
|
||||
| 地图搜索 | `https://www.google.com/search?q={keyword}&tbm=map` | `tbm=map` 表示地图 |
|
||||
| 购物搜索 | `https://www.google.com/search?q={keyword}&tbm=shop` | `tbm=shop` 表示购物 |
|
||||
| 图书搜索 | `https://www.google.com/search?q={keyword}&tbm=bks` | `tbm=bks` 表示图书 |
|
||||
| 学术搜索 | `https://scholar.google.com/scholar?q={keyword}` | Google Scholar |
|
||||
|
||||
### 1.6 Google 深度搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 搜索GitHub上的Python机器学习项目
|
||||
web_fetch({"url": "https://www.google.com/search?q=site:github.com+python+machine+learning"})
|
||||
|
||||
// 2. 搜索2024年的PDF格式机器学习教程
|
||||
web_fetch({"url": "https://www.google.com/search?q=machine+learning+tutorial+filetype:pdf&tbs=cdr:1,cd_min:1/1/2024"})
|
||||
|
||||
// 3. 搜索标题包含"tutorial"的Python相关页面
|
||||
web_fetch({"url": "https://www.google.com/search?q=intitle:tutorial+python"})
|
||||
|
||||
// 4. 搜索过去一周的新闻
|
||||
web_fetch({"url": "https://www.google.com/search?q=AI+breakthrough&tbs=qdr:w&tbm=nws"})
|
||||
|
||||
// 5. 搜索中文内容(界面英文,结果中文)
|
||||
web_fetch({"url": "https://www.google.com/search?q=人工智能&lr=lang_zh-CN&hl=en"})
|
||||
|
||||
// 6. 搜索特定价格范围的笔记本电脑
|
||||
web_fetch({"url": "https://www.google.com/search?q=laptop+%241000..%242000+best+rating"})
|
||||
|
||||
// 7. 搜索排除Wikipedia的结果
|
||||
web_fetch({"url": "https://www.google.com/search?q=python+programming+-wikipedia"})
|
||||
|
||||
// 8. 搜索学术文献
|
||||
web_fetch({"url": "https://scholar.google.com/scholar?q=deep+learning+optimization"})
|
||||
|
||||
// 9. 搜索缓存页面(查看已删除内容)
|
||||
web_fetch({"url": "https://webcache.googleusercontent.com/search?q=cache:example.com"})
|
||||
|
||||
// 10. 搜索相关网站
|
||||
web_fetch({"url": "https://www.google.com/search?q=related:stackoverflow.com"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🦆 DuckDuckGo 深度搜索
|
||||
|
||||
### 2.1 DuckDuckGo 特色功能
|
||||
|
||||
| 功能 | 语法 | 示例 |
|
||||
|------|------|------|
|
||||
| **Bangs 快捷** | `!缩写` | `!g python` → Google搜索 |
|
||||
| **密码生成** | `password` | `https://duckduckgo.com/?q=password+20` |
|
||||
| **颜色转换** | `color` | `https://duckduckgo.com/?q=+%23FF5733` |
|
||||
| **短链接** | `shorten` | `https://duckduckgo.com/?q=shorten+example.com` |
|
||||
| **二维码生成** | `qr` | `https://duckduckgo.com/?q=qr+hello+world` |
|
||||
| **生成UUID** | `uuid` | `https://duckduckgo.com/?q=uuid` |
|
||||
| **Base64编解码** | `base64` | `https://duckduckgo.com/?q=base64+hello` |
|
||||
|
||||
### 2.2 DuckDuckGo Bangs 完整列表
|
||||
|
||||
#### 搜索引擎
|
||||
|
||||
| Bang | 跳转目标 | 示例 |
|
||||
|------|---------|------|
|
||||
| `!g` | Google | `!g python tutorial` |
|
||||
| `!b` | Bing | `!b weather` |
|
||||
| `!y` | Yahoo | `!y finance` |
|
||||
| `!sp` | Startpage | `!sp privacy` |
|
||||
| `!brave` | Brave Search | `!brave tech` |
|
||||
|
||||
#### 编程开发
|
||||
|
||||
| Bang | 跳转目标 | 示例 |
|
||||
|------|---------|------|
|
||||
| `!gh` | GitHub | `!gh tensorflow` |
|
||||
| `!so` | Stack Overflow | `!so javascript error` |
|
||||
| `!npm` | npmjs.com | `!npm express` |
|
||||
| `!pypi` | PyPI | `!pypi requests` |
|
||||
| `!mdn` | MDN Web Docs | `!mdn fetch api` |
|
||||
| `!docs` | DevDocs | `!docs python` |
|
||||
| `!docker` | Docker Hub | `!docker nginx` |
|
||||
|
||||
#### 知识百科
|
||||
|
||||
| Bang | 跳转目标 | 示例 |
|
||||
|------|---------|------|
|
||||
| `!w` | Wikipedia | `!w machine learning` |
|
||||
| `!wen` | Wikipedia英文 | `!wen artificial intelligence` |
|
||||
| `!wt` | Wiktionary | `!wt serendipity` |
|
||||
| `!imdb` | IMDb | `!imdb inception` |
|
||||
|
||||
#### 购物价格
|
||||
|
||||
| Bang | 跳转目标 | 示例 |
|
||||
|------|---------|------|
|
||||
| `!a` | Amazon | `!a wireless headphones` |
|
||||
| `!e` | eBay | `!e vintage watch` |
|
||||
| `!ali` | AliExpress | `!ali phone case` |
|
||||
|
||||
#### 地图位置
|
||||
|
||||
| Bang | 跳转目标 | 示例 |
|
||||
|------|---------|------|
|
||||
| `!m` | Google Maps | `!m Beijing` |
|
||||
| `!maps` | OpenStreetMap | `!maps Paris` |
|
||||
|
||||
### 2.3 DuckDuckGo 搜索参数
|
||||
|
||||
| 参数 | 功能 | 示例 |
|
||||
|------|------|------|
|
||||
| `kp=1` | 严格安全搜索 | `https://duckduckgo.com/html/?q=test&kp=1` |
|
||||
| `kp=-1` | 关闭安全搜索 | `https://duckduckgo.com/html/?q=test&kp=-1` |
|
||||
| `kl=cn` | 中国区域 | `https://duckduckgo.com/html/?q=news&kl=cn` |
|
||||
| `kl=us-en` | 美国英文 | `https://duckduckgo.com/html/?q=news&kl=us-en` |
|
||||
| `ia=web` | 网页结果 | `https://duckduckgo.com/?q=test&ia=web` |
|
||||
| `ia=images` | 图片结果 | `https://duckduckgo.com/?q=test&ia=images` |
|
||||
| `ia=news` | 新闻结果 | `https://duckduckgo.com/?q=test&ia=news` |
|
||||
| `ia=videos` | 视频结果 | `https://duckduckgo.com/?q=test&ia=videos` |
|
||||
|
||||
### 2.4 DuckDuckGo 深度搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 使用Bang跳转到Google搜索
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=!g+machine+learning"})
|
||||
|
||||
// 2. 直接搜索GitHub上的项目
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=!gh+react"})
|
||||
|
||||
// 3. 查找Stack Overflow答案
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=!so+python+list+comprehension"})
|
||||
|
||||
// 4. 生成密码
|
||||
web_fetch({"url": "https://duckduckgo.com/?q=password+16"})
|
||||
|
||||
// 5. Base64编码
|
||||
web_fetch({"url": "https://duckduckgo.com/?q=base64+hello+world"})
|
||||
|
||||
// 6. 颜色代码转换
|
||||
web_fetch({"url": "https://duckduckgo.com/?q=%23FF5733"})
|
||||
|
||||
// 7. 搜索YouTube视频
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=!yt+python+tutorial"})
|
||||
|
||||
// 8. 查看Wikipedia
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=!w+artificial+intelligence"})
|
||||
|
||||
// 9. 亚马逊商品搜索
|
||||
web_fetch({"url": "https://duckduckgo.com/html/?q=!a+laptop"})
|
||||
|
||||
// 10. 生成二维码
|
||||
web_fetch({"url": "https://duckduckgo.com/?q=qr+https://github.com"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔎 Brave Search 深度搜索
|
||||
|
||||
### 3.1 Brave Search 特色功能
|
||||
|
||||
| 功能 | 参数 | 示例 |
|
||||
|------|------|------|
|
||||
| **独立索引** | 无依赖Google/Bing | 自有爬虫索引 |
|
||||
| **Goggles** | 自定义搜索规则 | 创建个性化过滤器 |
|
||||
| **Discussions** | 论坛讨论搜索 | 聚合Reddit等论坛 |
|
||||
| **News** | 新闻聚合 | 独立新闻索引 |
|
||||
|
||||
### 3.2 Brave Search 参数
|
||||
|
||||
| 参数 | 功能 | 示例 |
|
||||
|------|------|------|
|
||||
| `tf=pw` | 本周 | `https://search.brave.com/search?q=news&tf=pw` |
|
||||
| `tf=pm` | 本月 | `https://search.brave.com/search?q=tech&tf=pm` |
|
||||
| `tf=py` | 本年 | `https://search.brave.com/search?q=AI&tf=py` |
|
||||
| `safesearch=strict` | 严格安全 | `https://search.brave.com/search?q=test&safesearch=strict` |
|
||||
| `source=web` | 网页搜索 | 默认 |
|
||||
| `source=news` | 新闻搜索 | `https://search.brave.com/search?q=tech&source=news` |
|
||||
| `source=images` | 图片搜索 | `https://search.brave.com/search?q=cat&source=images` |
|
||||
| `source=videos` | 视频搜索 | `https://search.brave.com/search?q=music&source=videos` |
|
||||
|
||||
### 3.3 Brave Search Goggles(自定义过滤器)
|
||||
|
||||
Goggles 允许创建自定义搜索规则:
|
||||
|
||||
```
|
||||
$discard // 丢弃所有
|
||||
$boost,site=stackoverflow.com // 提升Stack Overflow
|
||||
$boost,site=github.com // 提升GitHub
|
||||
$boost,site=docs.python.org // 提升Python文档
|
||||
```
|
||||
|
||||
### 3.4 Brave Search 深度搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 本周科技新闻
|
||||
web_fetch({"url": "https://search.brave.com/search?q=technology&tf=pw&source=news"})
|
||||
|
||||
// 2. 本月AI发展
|
||||
web_fetch({"url": "https://search.brave.com/search?q=artificial+intelligence&tf=pm"})
|
||||
|
||||
// 3. 图片搜索
|
||||
web_fetch({"url": "https://search.brave.com/search?q=machine+learning&source=images"})
|
||||
|
||||
// 4. 视频教程
|
||||
web_fetch({"url": "https://search.brave.com/search?q=python+tutorial&source=videos"})
|
||||
|
||||
// 5. 使用独立索引搜索
|
||||
web_fetch({"url": "https://search.brave.com/search?q=privacy+tools"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 📊 WolframAlpha 知识计算搜索
|
||||
|
||||
### 4.1 WolframAlpha 数据类型
|
||||
|
||||
| 类型 | 查询示例 | URL |
|
||||
|------|---------|-----|
|
||||
| **数学计算** | `integrate x^2 dx` | `https://www.wolframalpha.com/input?i=integrate+x%5E2+dx` |
|
||||
| **单位换算** | `100 miles to km` | `https://www.wolframalpha.com/input?i=100+miles+to+km` |
|
||||
| **货币转换** | `100 USD to CNY` | `https://www.wolframalpha.com/input?i=100+USD+to+CNY` |
|
||||
| **股票数据** | `AAPL stock` | `https://www.wolframalpha.com/input?i=AAPL+stock` |
|
||||
| **天气查询** | `weather in Beijing` | `https://www.wolframalpha.com/input?i=weather+in+Beijing` |
|
||||
| **人口数据** | `population of China` | `https://www.wolframalpha.com/input?i=population+of+China` |
|
||||
| **化学元素** | `properties of gold` | `https://www.wolframalpha.com/input?i=properties+of+gold` |
|
||||
| **营养成分** | `nutrition of apple` | `https://www.wolframalpha.com/input?i=nutrition+of+apple` |
|
||||
| **日期计算** | `days between Jan 1 2020 and Dec 31 2024` | 日期间隔计算 |
|
||||
| **时区转换** | `10am Beijing to New York` | 时区转换 |
|
||||
| **IP地址** | `8.8.8.8` | IP信息查询 |
|
||||
| **条形码** | `scan barcode 123456789` | 条码信息 |
|
||||
| **飞机航班** | `flight AA123` | 航班信息 |
|
||||
|
||||
### 4.2 WolframAlpha 深度搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 计算积分
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=integrate+sin%28x%29+from+0+to+pi"})
|
||||
|
||||
// 2. 解方程
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=solve+x%5E2-5x%2B6%3D0"})
|
||||
|
||||
// 3. 货币实时汇率
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=100+USD+to+CNY"})
|
||||
|
||||
// 4. 股票实时数据
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=Apple+stock+price"})
|
||||
|
||||
// 5. 城市天气
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=weather+in+Shanghai+tomorrow"})
|
||||
|
||||
// 6. 国家统计信息
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=GDP+of+China+vs+USA"})
|
||||
|
||||
// 7. 化学计算
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=molar+mass+of+H2SO4"})
|
||||
|
||||
// 8. 物理常数
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=speed+of+light"})
|
||||
|
||||
// 9. 营养信息
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=calories+in+banana"})
|
||||
|
||||
// 10. 历史日期
|
||||
web_fetch({"url": "https://www.wolframalpha.com/input?i=events+on+July+20+1969"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🔧 Startpage 隐私搜索
|
||||
|
||||
### 5.1 Startpage 特色功能
|
||||
|
||||
| 功能 | 说明 | URL |
|
||||
|------|------|-----|
|
||||
| **代理浏览** | 匿名访问搜索结果 | 点击"匿名查看" |
|
||||
| **无追踪** | 不记录搜索历史 | 默认开启 |
|
||||
| **EU服务器** | 受欧盟隐私法保护 | 数据在欧洲 |
|
||||
| **代理图片** | 图片代理加载 | 隐藏IP |
|
||||
|
||||
### 5.2 Startpage 参数
|
||||
|
||||
| 参数 | 功能 | 示例 |
|
||||
|------|------|------|
|
||||
| `cat=web` | 网页搜索 | 默认 |
|
||||
| `cat=images` | 图片搜索 | `...&cat=images` |
|
||||
| `cat=video` | 视频搜索 | `...&cat=video` |
|
||||
| `cat=news` | 新闻搜索 | `...&cat=news` |
|
||||
| `language=english` | 英文结果 | `...&language=english` |
|
||||
| `time=day` | 过去24小时 | `...&time=day` |
|
||||
| `time=week` | 过去一周 | `...&time=week` |
|
||||
| `time=month` | 过去一月 | `...&time=month` |
|
||||
| `time=year` | 过去一年 | `...&time=year` |
|
||||
| `nj=0` | 关闭 family filter | `...&nj=0` |
|
||||
|
||||
### 5.3 Startpage 深度搜索示例
|
||||
|
||||
```javascript
|
||||
// 1. 隐私搜索
|
||||
web_fetch({"url": "https://www.startpage.com/sp/search?query=privacy+tools"})
|
||||
|
||||
// 2. 图片隐私搜索
|
||||
web_fetch({"url": "https://www.startpage.com/sp/search?query=nature&cat=images"})
|
||||
|
||||
// 3. 本周新闻(隐私模式)
|
||||
web_fetch({"url": "https://www.startpage.com/sp/search?query=tech+news&time=week&cat=news"})
|
||||
|
||||
// 4. 英文结果搜索
|
||||
web_fetch({"url": "https://www.startpage.com/sp/search?query=machine+learning&language=english"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🌐 其他国际搜索引擎
|
||||
|
||||
### Yahoo
|
||||
|
||||
```javascript
|
||||
web_fetch({"url": "https://search.yahoo.com/search?p={keyword}"})
|
||||
```
|
||||
|
||||
### Ecosia(环保搜索)
|
||||
|
||||
```javascript
|
||||
web_fetch({"url": "https://www.ecosia.org/search?q={keyword}"})
|
||||
```
|
||||
|
||||
### Qwant(欧盟隐私搜索)
|
||||
|
||||
```javascript
|
||||
web_fetch({"url": "https://www.qwant.com/?q={keyword}"})
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 🌍 国际搜索策略
|
||||
|
||||
### 按搜索目标选择引擎
|
||||
|
||||
| 搜索目标 | 首选引擎 | 原因 |
|
||||
|---------|---------|------|
|
||||
| **学术研究** | Google Scholar | 学术资源索引最全 |
|
||||
| **编程开发** | Google + DuckDuckGo Bangs | 技术文档全面 |
|
||||
| **隐私敏感** | DuckDuckGo / Brave | 不追踪用户 |
|
||||
| **实时新闻** | Brave News | 独立新闻索引 |
|
||||
| **知识计算** | WolframAlpha | 结构化数据计算 |
|
||||
| **隐私+Google结果** | Startpage | Google结果+隐私保护 |
|
||||
|
||||
138
runtime/skills/_workspace/generalist/skill-vetter/SKILL.md
Normal file
138
runtime/skills/_workspace/generalist/skill-vetter/SKILL.md
Normal file
|
|
@ -0,0 +1,138 @@
|
|||
---
|
||||
name: skill-vetter
|
||||
version: 1.0.0
|
||||
description: Security-first skill vetting for AI agents. Use before installing any skill from ClawdHub, GitHub, or other sources. Checks for red flags, permission scope, and suspicious patterns.
|
||||
---
|
||||
|
||||
# Skill Vetter 🔒
|
||||
|
||||
Security-first vetting protocol for AI agent skills. **Never install a skill without vetting it first.**
|
||||
|
||||
## When to Use
|
||||
|
||||
- Before installing any skill from ClawdHub
|
||||
- Before running skills from GitHub repos
|
||||
- When evaluating skills shared by other agents
|
||||
- Anytime you're asked to install unknown code
|
||||
|
||||
## Vetting Protocol
|
||||
|
||||
### Step 1: Source Check
|
||||
|
||||
```
|
||||
Questions to answer:
|
||||
- [ ] Where did this skill come from?
|
||||
- [ ] Is the author known/reputable?
|
||||
- [ ] How many downloads/stars does it have?
|
||||
- [ ] When was it last updated?
|
||||
- [ ] Are there reviews from other agents?
|
||||
```
|
||||
|
||||
### Step 2: Code Review (MANDATORY)
|
||||
|
||||
Read ALL files in the skill. Check for these **RED FLAGS**:
|
||||
|
||||
```
|
||||
🚨 REJECT IMMEDIATELY IF YOU SEE:
|
||||
─────────────────────────────────────────
|
||||
• curl/wget to unknown URLs
|
||||
• Sends data to external servers
|
||||
• Requests credentials/tokens/API keys
|
||||
• Reads ~/.ssh, ~/.aws, ~/.config without clear reason
|
||||
• Accesses MEMORY.md, USER.md, SOUL.md, IDENTITY.md
|
||||
• Uses base64 decode on anything
|
||||
• Uses eval() or exec() with external input
|
||||
• Modifies system files outside workspace
|
||||
• Installs packages without listing them
|
||||
• Network calls to IPs instead of domains
|
||||
• Obfuscated code (compressed, encoded, minified)
|
||||
• Requests elevated/sudo permissions
|
||||
• Accesses browser cookies/sessions
|
||||
• Touches credential files
|
||||
─────────────────────────────────────────
|
||||
```
|
||||
|
||||
### Step 3: Permission Scope
|
||||
|
||||
```
|
||||
Evaluate:
|
||||
- [ ] What files does it need to read?
|
||||
- [ ] What files does it need to write?
|
||||
- [ ] What commands does it run?
|
||||
- [ ] Does it need network access? To where?
|
||||
- [ ] Is the scope minimal for its stated purpose?
|
||||
```
|
||||
|
||||
### Step 4: Risk Classification
|
||||
|
||||
| Risk Level | Examples | Action |
|
||||
|------------|----------|--------|
|
||||
| 🟢 LOW | Notes, weather, formatting | Basic review, install OK |
|
||||
| 🟡 MEDIUM | File ops, browser, APIs | Full code review required |
|
||||
| 🔴 HIGH | Credentials, trading, system | Human approval required |
|
||||
| ⛔ EXTREME | Security configs, root access | Do NOT install |
|
||||
|
||||
## Output Format
|
||||
|
||||
After vetting, produce this report:
|
||||
|
||||
```
|
||||
SKILL VETTING REPORT
|
||||
═══════════════════════════════════════
|
||||
Skill: [name]
|
||||
Source: [ClawdHub / GitHub / other]
|
||||
Author: [username]
|
||||
Version: [version]
|
||||
───────────────────────────────────────
|
||||
METRICS:
|
||||
• Downloads/Stars: [count]
|
||||
• Last Updated: [date]
|
||||
• Files Reviewed: [count]
|
||||
───────────────────────────────────────
|
||||
RED FLAGS: [None / List them]
|
||||
|
||||
PERMISSIONS NEEDED:
|
||||
• Files: [list or "None"]
|
||||
• Network: [list or "None"]
|
||||
• Commands: [list or "None"]
|
||||
───────────────────────────────────────
|
||||
RISK LEVEL: [🟢 LOW / 🟡 MEDIUM / 🔴 HIGH / ⛔ EXTREME]
|
||||
|
||||
VERDICT: [✅ SAFE TO INSTALL / ⚠️ INSTALL WITH CAUTION / ❌ DO NOT INSTALL]
|
||||
|
||||
NOTES: [Any observations]
|
||||
═══════════════════════════════════════
|
||||
```
|
||||
|
||||
## Quick Vet Commands
|
||||
|
||||
For GitHub-hosted skills:
|
||||
```bash
|
||||
# Check repo stats
|
||||
curl -s "https://api.github.com/repos/OWNER/REPO" | jq '{stars: .stargazers_count, forks: .forks_count, updated: .updated_at}'
|
||||
|
||||
# List skill files
|
||||
curl -s "https://api.github.com/repos/OWNER/REPO/contents/skills/SKILL_NAME" | jq '.[].name'
|
||||
|
||||
# Fetch and review SKILL.md
|
||||
curl -s "https://raw.githubusercontent.com/OWNER/REPO/main/skills/SKILL_NAME/SKILL.md"
|
||||
```
|
||||
|
||||
## Trust Hierarchy
|
||||
|
||||
1. **Official OpenClaw skills** → Lower scrutiny (still review)
|
||||
2. **High-star repos (1000+)** → Moderate scrutiny
|
||||
3. **Known authors** → Moderate scrutiny
|
||||
4. **New/unknown sources** → Maximum scrutiny
|
||||
5. **Skills requesting credentials** → Human approval always
|
||||
|
||||
## Remember
|
||||
|
||||
- No skill is worth compromising security
|
||||
- When in doubt, don't install
|
||||
- Ask your human for high-risk decisions
|
||||
- Document what you vet for future reference
|
||||
|
||||
---
|
||||
|
||||
*Paranoia is a feature.* 🔒🦀
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn71j6xbmpwfvx4c6y1ez8cd718081mg",
|
||||
"slug": "skill-vetter",
|
||||
"version": "1.0.0",
|
||||
"publishedAt": 1769863429632
|
||||
}
|
||||
|
|
@ -9,6 +9,7 @@ from oclaw.runtime.skill_role_binding import (
|
|||
should_apply_workspace_role_filter,
|
||||
)
|
||||
from oclaw.runtime.skills import discover_workspace_skill_manifests
|
||||
from oclaw.runtime.skills_workspace_lane import skill_dir_private_lane_segment
|
||||
from oclaw.runtime.tools.base import ToolRegistry
|
||||
|
||||
|
||||
|
|
@ -71,6 +72,8 @@ def collect_skill_catalog_entries(
|
|||
registry: ToolRegistry,
|
||||
base_url: str,
|
||||
skill_binding_role: str | None = None,
|
||||
exclude_foreign_private_workspace_skills: bool = False,
|
||||
private_workspace_lane_segment: str | None = None,
|
||||
) -> list[tuple[str, str, str]]:
|
||||
"""Return (name, description, location) for model-visible prompt skills."""
|
||||
_ = registry
|
||||
|
|
@ -86,8 +89,20 @@ def collect_skill_catalog_entries(
|
|||
for m in sorted(discover_workspace_skill_manifests(), key=lambda x: x.name.lower()):
|
||||
if m.disable_model_invocation or m.name in disabled:
|
||||
continue
|
||||
priv_lane = skill_dir_private_lane_segment(str(m.skill_dir or ""))
|
||||
want_lane = str(private_workspace_lane_segment or "").strip()
|
||||
if exclude_foreign_private_workspace_skills:
|
||||
if priv_lane is not None:
|
||||
if not want_lane or priv_lane != want_lane:
|
||||
continue
|
||||
own_private_lane = bool(
|
||||
exclude_foreign_private_workspace_skills
|
||||
and priv_lane is not None
|
||||
and want_lane
|
||||
and priv_lane == want_lane
|
||||
)
|
||||
if role_filter:
|
||||
if m.name not in role_allow:
|
||||
if m.name not in role_allow and not own_private_lane:
|
||||
continue
|
||||
out.append((m.name, (m.description or "").strip() or m.name, m.skill_file))
|
||||
|
||||
|
|
@ -125,6 +140,8 @@ def build_skills_catalog_block(
|
|||
registry: ToolRegistry,
|
||||
base_url: str,
|
||||
skill_binding_role: str | None = None,
|
||||
exclude_foreign_private_workspace_skills: bool = False,
|
||||
private_workspace_lane_segment: str | None = None,
|
||||
) -> str:
|
||||
if not _skill_runtime_enabled(store) or not _skills_prompt_in_system_enabled(store):
|
||||
return ""
|
||||
|
|
@ -133,6 +150,8 @@ def build_skills_catalog_block(
|
|||
registry=registry,
|
||||
base_url=base_url,
|
||||
skill_binding_role=skill_binding_role,
|
||||
exclude_foreign_private_workspace_skills=exclude_foreign_private_workspace_skills,
|
||||
private_workspace_lane_segment=private_workspace_lane_segment,
|
||||
)
|
||||
return format_skills_for_prompt(entries, max_chars=_max_skills_prompt_chars())
|
||||
|
||||
|
|
|
|||
96
runtime/skills_workspace_lane.py
Normal file
96
runtime/skills_workspace_lane.py
Normal file
|
|
@ -0,0 +1,96 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def fs_safe_workspace_lane_segment(raw: str | None) -> str:
|
||||
s = "".join(c if (c.isalnum() or c in "._-") else "_" for c in str(raw or "").strip())
|
||||
s = s.strip("._")
|
||||
return (s or "lane")[:120]
|
||||
|
||||
|
||||
def workspace_lane_segment(*, workspace_owner_session_id: str | None, session_id: str | None) -> str:
|
||||
owner = str(workspace_owner_session_id or "").strip()
|
||||
sid = str(session_id or "").strip()
|
||||
return fs_safe_workspace_lane_segment(owner or sid)
|
||||
|
||||
|
||||
def skill_dir_private_lane_segment(skill_dir: str | Path, *, skills_home: Path | None = None) -> str | None:
|
||||
"""Return the private lane key for catalog filtering.
|
||||
|
||||
Layouts:
|
||||
|
||||
- ``_workspace/<role>/<skill>/`` (role sibling of ``public``) → ``<role>``
|
||||
- ``_workspace/_agent/<lane>/<skill>/`` (legacy session or older role bucket) → ``<lane>``
|
||||
|
||||
Flat ``_workspace/<skill>/`` (single segment) → ``None``.
|
||||
"""
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
|
||||
home = Path(skills_home).resolve() if skills_home else default_skills_root().resolve()
|
||||
ws = (home / "_workspace").resolve()
|
||||
p = Path(skill_dir).resolve()
|
||||
try:
|
||||
rel = p.relative_to(ws)
|
||||
except ValueError:
|
||||
return None
|
||||
parts = rel.parts
|
||||
if len(parts) < 2:
|
||||
return None
|
||||
head = str(parts[0]).lower()
|
||||
if head == "public":
|
||||
return None
|
||||
if head == "_agent" and len(parts) >= 2:
|
||||
return str(parts[1])
|
||||
return str(parts[0])
|
||||
|
||||
|
||||
def resolve_auto_install_parent(
|
||||
*,
|
||||
public: bool,
|
||||
workspace_lane_role: str | None = None,
|
||||
workspace_owner_session_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
skills_home: Path | None = None,
|
||||
) -> Path:
|
||||
"""Directory that directly contains the skill folder (sibling of ``<name>/SKILL.md``).
|
||||
|
||||
- ``public``: ``_workspace/public/``
|
||||
- role installs: ``_workspace/<role>/`` (same level as ``public``, e.g. ``generalist``, ``ops``)
|
||||
- legacy session bucket: ``_workspace/_agent/<segment>/`` when no binding role is available
|
||||
"""
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
|
||||
base = Path(skills_home).resolve() if skills_home else default_skills_root().resolve()
|
||||
ws = base / "_workspace"
|
||||
if public:
|
||||
return ws / "public"
|
||||
role = str(workspace_lane_role or "").strip().lower()
|
||||
if role:
|
||||
return ws / fs_safe_workspace_lane_segment(role)
|
||||
owner = str(workspace_owner_session_id or "").strip()
|
||||
sid = str(session_id or "").strip()
|
||||
if not owner and not sid:
|
||||
return ws
|
||||
seg = workspace_lane_segment(workspace_owner_session_id=owner or None, session_id=sid or None)
|
||||
return ws / "_agent" / seg
|
||||
|
||||
|
||||
def skills_home_containing_workspace_lane(install_skills_parent: Path) -> Path:
|
||||
"""Resolve the skills tree root (parent of ``_workspace``) from a workspace lane install parent."""
|
||||
cur = Path(install_skills_parent).resolve()
|
||||
if cur.name.lower() == "_workspace":
|
||||
return cur.parent
|
||||
for anc in cur.parents:
|
||||
if anc.name.lower() == "_workspace":
|
||||
return anc.parent
|
||||
return cur
|
||||
|
||||
|
||||
__all__ = [
|
||||
"fs_safe_workspace_lane_segment",
|
||||
"resolve_auto_install_parent",
|
||||
"skill_dir_private_lane_segment",
|
||||
"skills_home_containing_workspace_lane",
|
||||
"workspace_lane_segment",
|
||||
]
|
||||
|
|
@ -6,6 +6,10 @@ from typing import Any
|
|||
from oclaw.runtime.memory_stage import render_memory_context_block
|
||||
from oclaw.runtime.project_context_prompt import build_project_context_block
|
||||
from oclaw.runtime.skills_prompt import build_skills_catalog_block
|
||||
from oclaw.runtime.skills_workspace_lane import (
|
||||
fs_safe_workspace_lane_segment,
|
||||
workspace_lane_segment,
|
||||
)
|
||||
from oclaw.runtime.types import OclawMemoryContext
|
||||
from oclaw.runtime.workspaces.experts import expert_workspace_signature_token
|
||||
from oclaw.prompts.loader import render_runtime_prompt
|
||||
|
|
@ -57,6 +61,23 @@ def _unified_skill_policy_guidance() -> str:
|
|||
)
|
||||
|
||||
|
||||
def _skill_catalog_lane_flags(
|
||||
*,
|
||||
skill_binding_role: str | None,
|
||||
workspace_owner_session_id: str | None,
|
||||
session_id: str | None,
|
||||
) -> tuple[bool, str | None]:
|
||||
role = str(skill_binding_role or "").strip().lower()
|
||||
if role:
|
||||
return True, fs_safe_workspace_lane_segment(role)
|
||||
o = str(workspace_owner_session_id or "").strip()
|
||||
s = str(session_id or "").strip()
|
||||
if not o and not s:
|
||||
return False, None
|
||||
seg = workspace_lane_segment(workspace_owner_session_id=o or None, session_id=s or None)
|
||||
return True, seg
|
||||
|
||||
|
||||
def build_executor_system_prompt(
|
||||
*,
|
||||
store: Any,
|
||||
|
|
@ -67,6 +88,8 @@ def build_executor_system_prompt(
|
|||
lang: str,
|
||||
workspace_dir: str | None = None,
|
||||
skill_binding_role: str | None = None,
|
||||
workspace_owner_session_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> str:
|
||||
"""Build the final system string for the oclaw executor (memory block + skills catalog).
|
||||
|
||||
|
|
@ -80,6 +103,8 @@ def build_executor_system_prompt(
|
|||
base_system=base_system,
|
||||
workspace_dir=workspace_dir,
|
||||
skill_binding_role=skill_binding_role,
|
||||
workspace_owner_session_id=workspace_owner_session_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
mem_block = render_memory_context_block(memory_context or OclawMemoryContext())
|
||||
if not mem_block:
|
||||
|
|
@ -99,7 +124,14 @@ def get_executor_prompt_static(
|
|||
base_system: str,
|
||||
workspace_dir: str | None = None,
|
||||
skill_binding_role: str | None = None,
|
||||
workspace_owner_session_id: str | None = None,
|
||||
session_id: str | None = None,
|
||||
) -> str:
|
||||
excl, lane_seg = _skill_catalog_lane_flags(
|
||||
skill_binding_role=skill_binding_role,
|
||||
workspace_owner_session_id=workspace_owner_session_id,
|
||||
session_id=session_id,
|
||||
)
|
||||
cache_key = (
|
||||
str(base_url or "").strip(),
|
||||
str(base_system or "").strip(),
|
||||
|
|
@ -108,6 +140,8 @@ def get_executor_prompt_static(
|
|||
expert_workspace_signature_token(),
|
||||
_executor_prompt_settings_signature(store),
|
||||
bool(tools is not None),
|
||||
int(bool(excl)),
|
||||
str(lane_seg or ""),
|
||||
)
|
||||
with _EXECUTOR_STATIC_PROMPT_CACHE_LOCK:
|
||||
cached = _EXECUTOR_STATIC_PROMPT_CACHE.get(cache_key)
|
||||
|
|
@ -126,6 +160,8 @@ def get_executor_prompt_static(
|
|||
registry=tools,
|
||||
base_url=str(base_url or ""),
|
||||
skill_binding_role=skill_binding_role,
|
||||
exclude_foreign_private_workspace_skills=excl,
|
||||
private_workspace_lane_segment=lane_seg,
|
||||
)
|
||||
if cat.strip():
|
||||
final_system = render_runtime_prompt(
|
||||
|
|
@ -157,6 +193,8 @@ def warm_executor_prompt_cache(
|
|||
base_system=str(base_system or ""),
|
||||
workspace_dir=workspace_dir,
|
||||
skill_binding_role=str(role or "").strip().lower() or None,
|
||||
workspace_owner_session_id=None,
|
||||
session_id=None,
|
||||
)
|
||||
warmed += 1
|
||||
return {"roles_warmed": int(warmed)}
|
||||
|
|
|
|||
|
|
@ -78,6 +78,13 @@ class LocalAdapter:
|
|||
enabled: bool | None = None
|
||||
# Prefer DB setting when available.
|
||||
dbp = str(os.getenv("OPS_ASSISTANT_DB_PATH") or "").strip()
|
||||
try:
|
||||
if not dbp:
|
||||
from oclaw.platform.config.paths import db_path
|
||||
|
||||
dbp = str(db_path() or "").strip()
|
||||
except Exception:
|
||||
dbp = ""
|
||||
if dbp:
|
||||
try:
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
|
|
|
|||
|
|
@ -5,9 +5,13 @@ from typing import Any
|
|||
|
||||
from oclaw.platform.config.paths import db_path
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
from oclaw.runtime.chat.tool_invocation_context import (
|
||||
current_tool_lane_sessions,
|
||||
current_tool_workspace_lane_role,
|
||||
)
|
||||
from oclaw.runtime.skill_installer import auto_install_skill_from_payload, install_skill_from_registry_archive
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting
|
||||
from oclaw.runtime.skills_workspace_lane import resolve_auto_install_parent
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
|
|
@ -15,14 +19,32 @@ def _store() -> SqliteStore:
|
|||
return SqliteStore(db_path())
|
||||
|
||||
|
||||
def _agent_workspace_skills_root() -> Path:
|
||||
return default_skills_root() / "_workspace"
|
||||
def _coerce_public_flag(raw: Any) -> bool:
|
||||
if isinstance(raw, bool):
|
||||
return raw
|
||||
return str(raw or "").strip().lower() in {"1", "true", "yes", "on", "public"}
|
||||
|
||||
|
||||
def _agent_skill_install_parent(*, public: bool) -> Path:
|
||||
role = current_tool_workspace_lane_role()
|
||||
owner, sid = current_tool_lane_sessions()
|
||||
return resolve_auto_install_parent(
|
||||
public=public,
|
||||
workspace_lane_role=role,
|
||||
workspace_owner_session_id=owner,
|
||||
session_id=sid,
|
||||
skills_home=None,
|
||||
)
|
||||
|
||||
|
||||
def skill_auto_install_tool() -> ToolSpec:
|
||||
def _handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = args if isinstance(args, dict) else {}
|
||||
store = _store()
|
||||
public = _coerce_public_flag(payload.get("public"))
|
||||
install_parent = _agent_skill_install_parent(public=public)
|
||||
lane_role = current_tool_workspace_lane_role()
|
||||
auto_bind = bool(public)
|
||||
archive_url = str(payload.get("archive_url") or "").strip()
|
||||
slug = str(payload.get("slug") or "").strip()
|
||||
version = str(payload.get("version") or "").strip() or None
|
||||
|
|
@ -41,8 +63,8 @@ def skill_auto_install_tool() -> ToolSpec:
|
|||
store=store,
|
||||
archive_url=archive_url,
|
||||
overwrite=overwrite,
|
||||
skills_root=_agent_workspace_skills_root(),
|
||||
auto_bind=True,
|
||||
skills_root=install_parent,
|
||||
auto_bind=auto_bind,
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
|
|
@ -54,6 +76,9 @@ def skill_auto_install_tool() -> ToolSpec:
|
|||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
"provider": provider,
|
||||
"public": public,
|
||||
"install_lane": str(install_parent),
|
||||
"workspace_lane_role": lane_role,
|
||||
}
|
||||
|
||||
skill_payload = {
|
||||
|
|
@ -68,7 +93,12 @@ def skill_auto_install_tool() -> ToolSpec:
|
|||
return {"ok": False, "error_code": "name_required", "error": "name_required"}
|
||||
if not skill_payload["description"]:
|
||||
skill_payload["description"] = f"{skill_payload['name']} skill"
|
||||
out = auto_install_skill_from_payload(store=store, payload=skill_payload)
|
||||
out = auto_install_skill_from_payload(
|
||||
store=store,
|
||||
payload=skill_payload,
|
||||
workspace_install_parent=install_parent,
|
||||
auto_bind=auto_bind,
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
|
|
@ -78,11 +108,18 @@ def skill_auto_install_tool() -> ToolSpec:
|
|||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
"public": public,
|
||||
"install_lane": str(install_parent),
|
||||
"workspace_lane_role": lane_role,
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="skill_auto_install",
|
||||
description="Auto install a skill into _workspace lane (payload or market/archive).",
|
||||
description=(
|
||||
"Auto install a skill under the agent workspace lane. "
|
||||
"Use public=true for shared _workspace/public; omit or false for a role folder "
|
||||
"sibling to public: _workspace/<skill_binding_role>/ (e.g. generalist, ops)."
|
||||
),
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
|
|
@ -95,6 +132,10 @@ def skill_auto_install_tool() -> ToolSpec:
|
|||
"version": {"type": "string"},
|
||||
"archive_url": {"type": "string"},
|
||||
"overwrite": {"type": "boolean"},
|
||||
"public": {
|
||||
"type": "boolean",
|
||||
"description": "If true, install to shared _workspace/public with role auto-bind; if false, install under this tool run's session lane only.",
|
||||
},
|
||||
},
|
||||
"required": [],
|
||||
"additionalProperties": False,
|
||||
|
|
|
|||
158
runtime/tools/public/web_fetch_clean_tool.py
Normal file
158
runtime/tools/public/web_fetch_clean_tool.py
Normal file
|
|
@ -0,0 +1,158 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from html import unescape
|
||||
from typing import Any
|
||||
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36"
|
||||
)
|
||||
|
||||
|
||||
def _http_get(url: str, *, timeout_s: int) -> tuple[int, dict[str, str], bytes]:
|
||||
req = urllib.request.Request(
|
||||
str(url),
|
||||
headers={
|
||||
"User-Agent": _UA,
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
},
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=max(3, min(int(timeout_s), 45))) as resp:
|
||||
code = int(getattr(resp, "status", 200) or 200)
|
||||
headers = {str(k).lower(): str(v) for k, v in (getattr(resp, "headers", {}) or {}).items()}
|
||||
body = resp.read()
|
||||
return code, headers, body
|
||||
except urllib.error.HTTPError as e: # type: ignore[attr-defined]
|
||||
try:
|
||||
body = e.read() or b""
|
||||
except Exception:
|
||||
body = b""
|
||||
headers = {str(k).lower(): str(v) for k, v in (getattr(e, "headers", {}) or {}).items()}
|
||||
return int(getattr(e, "code", 0) or 0), headers, body
|
||||
|
||||
|
||||
def _strip_html_keep_lines(html: str) -> str:
|
||||
s = str(html or "")
|
||||
# Drop scripts/styles/noscript
|
||||
s = re.sub(r"(?is)<(script|style|noscript)[^>]*>.*?</\1>", " ", s)
|
||||
# Add line breaks for common block tags before stripping
|
||||
s = re.sub(r"(?i)</(p|div|br|li|h1|h2|h3|h4|h5|h6|tr|section|article)>", "\n", s)
|
||||
s = re.sub(r"(?i)<br\\s*/?>", "\n", s)
|
||||
# Strip tags
|
||||
s = re.sub(r"(?is)<[^>]+>", " ", s)
|
||||
# Decode entities and normalize whitespace
|
||||
s = unescape(s)
|
||||
s = re.sub(r"[\t\r\f\v]+", " ", s)
|
||||
s = re.sub(r" *\\n *", "\n", s)
|
||||
s = re.sub(r"\\n{3,}", "\n\n", s)
|
||||
s = re.sub(r" {2,}", " ", s)
|
||||
return s.strip()
|
||||
|
||||
|
||||
def _extract_title(html: str) -> str:
|
||||
m = re.search(r"(?is)<title[^>]*>(.*?)</title>", str(html or ""))
|
||||
if not m:
|
||||
return ""
|
||||
t = _strip_html_keep_lines(m.group(1))
|
||||
return t.splitlines()[0].strip() if t else ""
|
||||
|
||||
|
||||
def web_fetch_clean_tool() -> ToolSpec:
|
||||
def _handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = args if isinstance(args, dict) else {}
|
||||
url = str(payload.get("url") or "").strip()
|
||||
if not url:
|
||||
return {"ok": False, "error_code": "url_required", "error": "url_required"}
|
||||
timeout_s = max(3, min(int(payload.get("timeout") or 20), 45))
|
||||
max_chars = max(500, min(int(payload.get("max_chars") or 18000), 200000))
|
||||
|
||||
try:
|
||||
parsed = urllib.parse.urlparse(url)
|
||||
except Exception:
|
||||
return {"ok": False, "error_code": "invalid_url", "error": "invalid_url"}
|
||||
if parsed.scheme not in {"http", "https"}:
|
||||
return {"ok": False, "error_code": "unsupported_scheme", "error": "unsupported_scheme"}
|
||||
|
||||
try:
|
||||
code, headers, body = _http_get(url, timeout_s=timeout_s)
|
||||
except Exception as exc:
|
||||
return {"ok": False, "error_code": "fetch_failed", "error": f"fetch_failed:{type(exc).__name__}"}
|
||||
|
||||
if int(code) < 200 or int(code) >= 300:
|
||||
return {
|
||||
"ok": False,
|
||||
"error_code": "http_error",
|
||||
"error": "http_error",
|
||||
"status_code": int(code),
|
||||
"content_type": str(headers.get("content-type") or ""),
|
||||
}
|
||||
|
||||
ctype = str(headers.get("content-type") or "").lower()
|
||||
if ctype and ("text/html" not in ctype and "application/xhtml" not in ctype):
|
||||
# Still try to decode as text, but label as non_html for caller decisions.
|
||||
try:
|
||||
text = body.decode("utf-8", errors="replace")
|
||||
except Exception:
|
||||
text = ""
|
||||
text = text.strip()
|
||||
if len(text) > max_chars:
|
||||
text = text[:max_chars]
|
||||
return {
|
||||
"ok": True,
|
||||
"url": url,
|
||||
"status_code": int(code),
|
||||
"content_type": ctype,
|
||||
"title": "",
|
||||
"text": text,
|
||||
"truncated": len(body) > max_chars,
|
||||
"mode": "raw_non_html",
|
||||
}
|
||||
|
||||
html = body.decode("utf-8", errors="replace")
|
||||
title = _extract_title(html)
|
||||
text = _strip_html_keep_lines(html)
|
||||
truncated = False
|
||||
if len(text) > max_chars:
|
||||
text = text[:max_chars]
|
||||
truncated = True
|
||||
return {
|
||||
"ok": True,
|
||||
"url": url,
|
||||
"status_code": int(code),
|
||||
"content_type": ctype or "text/html",
|
||||
"title": title,
|
||||
"text": text,
|
||||
"truncated": truncated,
|
||||
"mode": "html_strip",
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="web_fetch_clean",
|
||||
description="Fetch a web page and return cleaned main text (simple HTML stripping).",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"url": {"type": "string", "description": "HTTP/HTTPS URL to fetch."},
|
||||
"timeout": {"type": "integer", "default": 20, "description": "Timeout seconds [3..45]."},
|
||||
"max_chars": {"type": "integer", "default": 18000, "description": "Max returned text chars [500..200000]."},
|
||||
},
|
||||
"required": ["url"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=_handler,
|
||||
tags=frozenset({"public", "web", "fetch", "read"}),
|
||||
risk_level="low",
|
||||
read_only=True,
|
||||
timeout_s=55.0,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["web_fetch_clean_tool"]
|
||||
|
||||
653
runtime/tools/public/web_search_fast_tool.py
Normal file
653
runtime/tools/public/web_search_fast_tool.py
Normal file
|
|
@ -0,0 +1,653 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from html import unescape
|
||||
from typing import Any
|
||||
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
_UA = (
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
|
||||
"AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36"
|
||||
)
|
||||
|
||||
_ENGINE_URLS: dict[str, str] = {
|
||||
"bing": "https://www.bing.com/search?q={keyword}&ensearch=1",
|
||||
"bing_cn": "https://cn.bing.com/search?q={keyword}&ensearch=0",
|
||||
"bing_int": "https://cn.bing.com/search?q={keyword}&ensearch=1",
|
||||
"ddg": "https://duckduckgo.com/html/?q={keyword}",
|
||||
"google": "https://www.google.com/search?q={keyword}",
|
||||
"google_hk": "https://www.google.com.hk/search?q={keyword}",
|
||||
"baidu": "https://www.baidu.com/s?wd={keyword}",
|
||||
"sogou": "https://www.sogou.com/web?query={keyword}",
|
||||
}
|
||||
|
||||
_ENGINE_ALIASES: dict[str, str] = {
|
||||
# config.json style names
|
||||
"baidu": "baidu",
|
||||
"bing cn": "bing_cn",
|
||||
"bing int": "bing_int",
|
||||
"bing": "bing",
|
||||
"360": "bing_cn", # fallback route
|
||||
"sogou": "sogou",
|
||||
"wechat": "sogou", # best-effort fallback
|
||||
"shenma": "bing_cn", # best-effort fallback
|
||||
"google": "google",
|
||||
"google hk": "google_hk",
|
||||
"duckduckgo": "ddg",
|
||||
"yahoo": "bing", # best-effort fallback
|
||||
"startpage": "ddg",
|
||||
"brave": "bing",
|
||||
"ecosia": "bing",
|
||||
"qwant": "bing",
|
||||
"wolframalpha": "bing",
|
||||
}
|
||||
|
||||
|
||||
def _normalize_engine_alias_key(raw: str) -> str:
|
||||
return re.sub(r"\s+", " ", str(raw or "").strip().lower().replace("_", " "))
|
||||
|
||||
|
||||
_ENGINE_ALIASES_NORM: dict[str, str] = {_normalize_engine_alias_key(k): v for k, v in _ENGINE_ALIASES.items()}
|
||||
_AUTO_PROVIDER_ORDER_DEFAULT = ("ddg_html", "bing_html", "ddg_api")
|
||||
_OFFICIAL_API_KEY_ENV = "OCLAW_WEB_SEARCH_OFFICIAL_API_KEY"
|
||||
_OFFICIAL_API_URL_ENV = "OCLAW_WEB_SEARCH_OFFICIAL_API_ENDPOINT"
|
||||
_OFFICIAL_API_DEFAULT_URL = "https://api.bing.microsoft.com/v7.0/search"
|
||||
_OFFICIAL_BING_KEY_ENV = "OCLAW_WEB_SEARCH_BING_API_KEY"
|
||||
_OFFICIAL_BING_URL_ENV = "OCLAW_WEB_SEARCH_BING_API_ENDPOINT"
|
||||
_OFFICIAL_BING_DEFAULT_URL = "https://api.bing.microsoft.com/v7.0/search"
|
||||
_OFFICIAL_GOOGLE_KEY_ENV = "OCLAW_WEB_SEARCH_GOOGLE_API_KEY"
|
||||
_OFFICIAL_GOOGLE_CSE_ENV = "OCLAW_WEB_SEARCH_GOOGLE_CSE_ID"
|
||||
_OFFICIAL_GOOGLE_URL_ENV = "OCLAW_WEB_SEARCH_GOOGLE_API_ENDPOINT"
|
||||
_OFFICIAL_GOOGLE_DEFAULT_URL = "https://customsearch.googleapis.com/customsearch/v1"
|
||||
|
||||
|
||||
def _http_get_text(url: str, *, timeout_s: int, extra_headers: dict[str, str] | None = None) -> str:
|
||||
headers = {
|
||||
"User-Agent": _UA,
|
||||
"Accept": "text/html,application/json;q=0.9,*/*;q=0.8",
|
||||
"Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
|
||||
}
|
||||
if isinstance(extra_headers, dict):
|
||||
for k, v in extra_headers.items():
|
||||
if str(k or "").strip():
|
||||
headers[str(k).strip()] = str(v or "")
|
||||
req = urllib.request.Request(
|
||||
str(url),
|
||||
headers=headers,
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=max(3, min(int(timeout_s), 30))) as resp:
|
||||
raw = resp.read()
|
||||
return raw.decode("utf-8", errors="replace")
|
||||
|
||||
|
||||
def _err_tag(prefix: str, exc: Exception) -> str:
|
||||
msg = str(exc or "").strip().replace("\n", " ")
|
||||
if len(msg) > 180:
|
||||
msg = msg[:180] + "..."
|
||||
return f"{prefix}:{type(exc).__name__}:{msg or 'failed'}"
|
||||
|
||||
|
||||
def _is_chinese_query(query: str) -> bool:
|
||||
return bool(re.search(r"[\u4e00-\u9fff]", str(query or "")))
|
||||
|
||||
|
||||
def _classify_error(exc: Exception, provider_name: str = "") -> str:
|
||||
p = str(provider_name or "")
|
||||
if isinstance(exc, urllib.error.HTTPError):
|
||||
code = int(getattr(exc, "code", 0) or 0)
|
||||
if p.startswith("official_"):
|
||||
if code in {401, 403}:
|
||||
return "auth_failed"
|
||||
if code == 429:
|
||||
return "rate_limited"
|
||||
if code == 402:
|
||||
return "quota_exceeded"
|
||||
if code in {401, 403, 429}:
|
||||
return "anti_bot_401_403"
|
||||
if code >= 500:
|
||||
return "upstream_5xx"
|
||||
if 400 <= code < 500:
|
||||
return "upstream_4xx"
|
||||
msg = str(exc or "").lower()
|
||||
if p.startswith("official_"):
|
||||
if "invalid api key" in msg or "access denied" in msg or "unauthorized" in msg:
|
||||
return "auth_failed"
|
||||
if "keyinvalid" in msg or "apikeyinvalid" in msg:
|
||||
return "auth_failed"
|
||||
if "quota" in msg:
|
||||
return "quota_exceeded"
|
||||
if "rate limit" in msg or "too many requests" in msg:
|
||||
return "rate_limited"
|
||||
if "ssrf" in msg or "forbidden host" in msg or "disallow" in msg:
|
||||
return "ssrf_blocked"
|
||||
if "timed out" in msg or "timeout" in msg:
|
||||
return "timeout"
|
||||
if "name or service not known" in msg or "nodename nor servname" in msg:
|
||||
return "dns_error"
|
||||
if "connection reset" in msg or "connection aborted" in msg or "refused" in msg:
|
||||
return "network_error"
|
||||
return "unknown_error"
|
||||
|
||||
|
||||
def _classify_error_from_tags(tags: list[str]) -> str:
|
||||
joined = " ".join(str(x or "").lower() for x in tags)
|
||||
if "ssrf" in joined or "forbidden host" in joined:
|
||||
return "ssrf_blocked"
|
||||
if "timed out" in joined or "timeout" in joined:
|
||||
return "timeout"
|
||||
if "httperror: http error 401" in joined or "httperror: http error 403" in joined or "httperror: http error 429" in joined:
|
||||
return "anti_bot_401_403"
|
||||
if "httperror: http error 5" in joined:
|
||||
return "upstream_5xx"
|
||||
if "httperror: http error 4" in joined:
|
||||
return "upstream_4xx"
|
||||
if "name or service not known" in joined or "nodename nor servname" in joined:
|
||||
return "dns_error"
|
||||
if "connection reset" in joined or "connection aborted" in joined or "refused" in joined:
|
||||
return "network_error"
|
||||
return "unknown_error"
|
||||
|
||||
|
||||
def _classify_error_from_attempts(attempts: list[dict[str, Any]]) -> str:
|
||||
for a in attempts:
|
||||
if not bool(a.get("ok")) and str(a.get("error_category") or "").strip():
|
||||
return str(a.get("error_category"))
|
||||
return "unknown_error"
|
||||
|
||||
|
||||
def _search_ddg_api(query: str, *, max_results: int, timeout_s: int) -> list[dict[str, str]]:
|
||||
q = urllib.parse.quote_plus(query)
|
||||
url = f"https://api.duckduckgo.com/?q={q}&format=json&no_html=1&skip_disambig=0"
|
||||
txt = _http_get_text(url, timeout_s=timeout_s)
|
||||
obj = json.loads(txt)
|
||||
out: list[dict[str, str]] = []
|
||||
|
||||
def _push(title: str, link: str, snippet: str) -> None:
|
||||
if not title or not link:
|
||||
return
|
||||
out.append({"title": title.strip(), "url": link.strip(), "snippet": snippet.strip()})
|
||||
|
||||
abs_text = str(obj.get("AbstractText") or "").strip()
|
||||
abs_url = str(obj.get("AbstractURL") or "").strip()
|
||||
heading = str(obj.get("Heading") or "").strip()
|
||||
if abs_text and abs_url:
|
||||
_push(heading or abs_url, abs_url, abs_text)
|
||||
|
||||
def _walk_related(items: list[Any]) -> None:
|
||||
for it in items:
|
||||
if len(out) >= max_results:
|
||||
return
|
||||
if isinstance(it, dict) and isinstance(it.get("Topics"), list):
|
||||
_walk_related(list(it.get("Topics") or []))
|
||||
continue
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
text = str(it.get("Text") or "").strip()
|
||||
link = str(it.get("FirstURL") or "").strip()
|
||||
if text and link:
|
||||
title = text.split(" - ", 1)[0].strip() or link
|
||||
_push(title, link, text)
|
||||
|
||||
_walk_related(list(obj.get("RelatedTopics") or []))
|
||||
return out[:max_results]
|
||||
|
||||
|
||||
def _strip_html(s: str) -> str:
|
||||
no_tags = re.sub(r"<[^>]+>", " ", str(s or ""))
|
||||
compact = re.sub(r"\s+", " ", no_tags).strip()
|
||||
return unescape(compact)
|
||||
|
||||
|
||||
def _search_ddg_html(query: str, *, max_results: int, timeout_s: int) -> list[dict[str, str]]:
|
||||
q = urllib.parse.quote_plus(query)
|
||||
url = f"https://duckduckgo.com/html/?q={q}"
|
||||
html = _http_get_text(url, timeout_s=timeout_s)
|
||||
patt = re.compile(
|
||||
r'<a[^>]+class="[^"]*result__a[^"]*"[^>]+href="(?P<href>[^"]+)"[^>]*>(?P<title>.*?)</a>',
|
||||
re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
snippets = re.findall(
|
||||
r'<a[^>]+class="[^"]*result__snippet[^"]*"[^>]*>(.*?)</a>',
|
||||
html,
|
||||
flags=re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
out: list[dict[str, str]] = []
|
||||
for idx, m in enumerate(patt.finditer(html)):
|
||||
if len(out) >= max_results:
|
||||
break
|
||||
href = unescape(_strip_html(m.group("href")))
|
||||
title = _strip_html(m.group("title"))
|
||||
if not href or not title:
|
||||
continue
|
||||
if href.startswith("/"):
|
||||
href = urllib.parse.urljoin("https://duckduckgo.com", href)
|
||||
snippet = _strip_html(snippets[idx]) if idx < len(snippets) else ""
|
||||
out.append({"title": title, "url": href, "snippet": snippet})
|
||||
return out[:max_results]
|
||||
|
||||
|
||||
def _search_bing_html(query: str, *, max_results: int, timeout_s: int, search_url: str | None = None) -> list[dict[str, str]]:
|
||||
q = urllib.parse.quote_plus(query)
|
||||
url = str(search_url or f"https://www.bing.com/search?q={q}&ensearch=1")
|
||||
html = _http_get_text(url, timeout_s=timeout_s)
|
||||
patt = re.compile(
|
||||
r'<li[^>]+class="[^"]*b_algo[^"]*"[^>]*>.*?<h2>\s*<a[^>]+href="(?P<href>[^"]+)"[^>]*>(?P<title>.*?)</a>\s*</h2>(?P<body>.*?)</li>',
|
||||
re.IGNORECASE | re.DOTALL,
|
||||
)
|
||||
out: list[dict[str, str]] = []
|
||||
for m in patt.finditer(html):
|
||||
if len(out) >= max_results:
|
||||
break
|
||||
href = unescape(_strip_html(m.group("href")))
|
||||
title = _strip_html(m.group("title"))
|
||||
body = str(m.group("body") or "")
|
||||
sm = re.search(r"<p[^>]*>(.*?)</p>", body, flags=re.IGNORECASE | re.DOTALL)
|
||||
snippet = _strip_html(sm.group(1)) if sm else _strip_html(body)[:220]
|
||||
if not href or not title:
|
||||
continue
|
||||
out.append({"title": title, "url": href, "snippet": snippet})
|
||||
return out[:max_results]
|
||||
|
||||
|
||||
def _search_generic_html(url: str, *, max_results: int, timeout_s: int) -> list[dict[str, str]]:
|
||||
html = _http_get_text(url, timeout_s=timeout_s)
|
||||
# Generic fallback parser for engines not explicitly adapted:
|
||||
# pick visible anchors with http(s) href and non-trivial title text.
|
||||
patt = re.compile(r'<a[^>]+href="(?P<href>https?://[^"]+)"[^>]*>(?P<title>.*?)</a>', re.IGNORECASE | re.DOTALL)
|
||||
out: list[dict[str, str]] = []
|
||||
seen: set[str] = set()
|
||||
for m in patt.finditer(html):
|
||||
if len(out) >= max_results:
|
||||
break
|
||||
href = unescape(_strip_html(m.group("href")))
|
||||
title = _strip_html(m.group("title"))
|
||||
if not href or not title or len(title) < 3:
|
||||
continue
|
||||
# Skip obvious nav links
|
||||
low = title.lower()
|
||||
if low in {"next", "prev", "previous", "about", "help", "privacy", "login", "sign in"}:
|
||||
continue
|
||||
if href in seen:
|
||||
continue
|
||||
seen.add(href)
|
||||
out.append({"title": title, "url": href, "snippet": ""})
|
||||
return out[:max_results]
|
||||
|
||||
|
||||
def _normalize_engine_name(raw: str) -> str:
|
||||
src = str(raw or "").strip().lower()
|
||||
if not src:
|
||||
return ""
|
||||
key_space = _normalize_engine_alias_key(src)
|
||||
alias = _ENGINE_ALIASES_NORM.get(key_space)
|
||||
if alias:
|
||||
return alias
|
||||
return src.replace(" ", "_")
|
||||
|
||||
|
||||
def _build_query_with_site(*, query: str, site: str) -> str:
|
||||
q = str(query or "").strip()
|
||||
s = str(site or "").strip()
|
||||
if not s:
|
||||
return q
|
||||
if s.startswith("site:"):
|
||||
s = s[5:].strip()
|
||||
s = s.strip("/")
|
||||
if not s:
|
||||
return q
|
||||
return f"site:{s} {q}".strip()
|
||||
|
||||
|
||||
def _search_by_engine_url(
|
||||
*,
|
||||
engine_url_template: str,
|
||||
query: str,
|
||||
max_results: int,
|
||||
timeout_s: int,
|
||||
) -> list[dict[str, str]]:
|
||||
q = urllib.parse.quote_plus(query)
|
||||
if "{keyword}" in str(engine_url_template):
|
||||
url = str(engine_url_template).replace("{keyword}", q)
|
||||
else:
|
||||
sep = "&" if ("?" in str(engine_url_template)) else "?"
|
||||
url = f"{engine_url_template}{sep}q={q}"
|
||||
low = url.lower()
|
||||
if "bing.com/search" in low:
|
||||
return _search_bing_html(query, max_results=max_results, timeout_s=timeout_s, search_url=url)
|
||||
if "duckduckgo.com" in low:
|
||||
return _search_ddg_html(query, max_results=max_results, timeout_s=timeout_s)
|
||||
return _search_generic_html(url, max_results=max_results, timeout_s=timeout_s)
|
||||
|
||||
|
||||
def _search_official_api(
|
||||
query: str,
|
||||
*,
|
||||
max_results: int,
|
||||
timeout_s: int,
|
||||
api_key: str,
|
||||
endpoint: str,
|
||||
) -> list[dict[str, str]]:
|
||||
base = str(endpoint or "").strip() or _OFFICIAL_API_DEFAULT_URL
|
||||
sep = "&" if ("?" in base) else "?"
|
||||
url = f"{base}{sep}q={urllib.parse.quote_plus(query)}&count={int(max_results)}"
|
||||
txt = _http_get_text(
|
||||
url,
|
||||
timeout_s=timeout_s,
|
||||
extra_headers={"Ocp-Apim-Subscription-Key": str(api_key or "").strip(), "Accept": "application/json"},
|
||||
)
|
||||
obj = json.loads(txt)
|
||||
out: list[dict[str, str]] = []
|
||||
web_pages = obj.get("webPages") if isinstance(obj, dict) else None
|
||||
values = web_pages.get("value") if isinstance(web_pages, dict) else None
|
||||
if isinstance(values, list):
|
||||
for it in values:
|
||||
if len(out) >= max_results:
|
||||
break
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
title = str(it.get("name") or "").strip()
|
||||
href = str(it.get("url") or "").strip()
|
||||
snippet = str(it.get("snippet") or "").strip()
|
||||
if title and href:
|
||||
out.append({"title": title, "url": href, "snippet": snippet})
|
||||
return out[:max_results]
|
||||
|
||||
|
||||
def _search_official_google_api(
|
||||
query: str,
|
||||
*,
|
||||
max_results: int,
|
||||
timeout_s: int,
|
||||
api_key: str,
|
||||
cse_id: str,
|
||||
endpoint: str,
|
||||
) -> list[dict[str, str]]:
|
||||
base = str(endpoint or "").strip() or _OFFICIAL_GOOGLE_DEFAULT_URL
|
||||
sep = "&" if ("?" in base) else "?"
|
||||
url = (
|
||||
f"{base}{sep}q={urllib.parse.quote_plus(query)}"
|
||||
f"&num={int(max_results)}&key={urllib.parse.quote_plus(str(api_key or '').strip())}"
|
||||
f"&cx={urllib.parse.quote_plus(str(cse_id or '').strip())}"
|
||||
)
|
||||
txt = _http_get_text(url, timeout_s=timeout_s, extra_headers={"Accept": "application/json"})
|
||||
obj = json.loads(txt)
|
||||
out: list[dict[str, str]] = []
|
||||
items = obj.get("items") if isinstance(obj, dict) else None
|
||||
if isinstance(items, list):
|
||||
for it in items:
|
||||
if len(out) >= max_results:
|
||||
break
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
title = str(it.get("title") or "").strip()
|
||||
href = str(it.get("link") or "").strip()
|
||||
snippet = str(it.get("snippet") or "").strip()
|
||||
if title and href:
|
||||
out.append({"title": title, "url": href, "snippet": snippet})
|
||||
return out[:max_results]
|
||||
|
||||
|
||||
def web_search_fast_tool() -> ToolSpec:
|
||||
def _handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = args if isinstance(args, dict) else {}
|
||||
query = str(payload.get("query") or "").strip()
|
||||
if not query:
|
||||
return {"ok": False, "error_code": "query_required", "error": "query_required"}
|
||||
site = str(payload.get("site") or "").strip()
|
||||
query = _build_query_with_site(query=query, site=site)
|
||||
max_results = max(1, min(int(payload.get("max_results") or 8), 20))
|
||||
timeout_s = max(3, min(int(payload.get("timeout") or 12), 30))
|
||||
provider = str(payload.get("provider") or "auto").strip().lower()
|
||||
official_provider = str(payload.get("official_provider") or "auto").strip().lower()
|
||||
engine = _normalize_engine_name(str(payload.get("engine") or ""))
|
||||
engine_url = str(payload.get("engine_url") or "").strip()
|
||||
official_bing_key = str(os.getenv(_OFFICIAL_BING_KEY_ENV) or os.getenv(_OFFICIAL_API_KEY_ENV) or "").strip()
|
||||
official_bing_endpoint = str(os.getenv(_OFFICIAL_BING_URL_ENV) or os.getenv(_OFFICIAL_API_URL_ENV) or _OFFICIAL_BING_DEFAULT_URL).strip()
|
||||
official_google_key = str(os.getenv(_OFFICIAL_GOOGLE_KEY_ENV) or "").strip()
|
||||
official_google_cse_id = str(os.getenv(_OFFICIAL_GOOGLE_CSE_ENV) or "").strip()
|
||||
official_google_endpoint = str(os.getenv(_OFFICIAL_GOOGLE_URL_ENV) or _OFFICIAL_GOOGLE_DEFAULT_URL).strip()
|
||||
used = "auto"
|
||||
errors: list[str] = []
|
||||
results: list[dict[str, str]] = []
|
||||
provider_attempts: list[dict[str, Any]] = []
|
||||
|
||||
def _attempt(provider_name: str, fn: Any) -> list[dict[str, str]]:
|
||||
started = time.perf_counter()
|
||||
try:
|
||||
r = list(fn() or [])
|
||||
provider_attempts.append(
|
||||
{
|
||||
"provider": provider_name,
|
||||
"ok": True,
|
||||
"count": len(r),
|
||||
"elapsed_ms": int((time.perf_counter() - started) * 1000),
|
||||
}
|
||||
)
|
||||
return r
|
||||
except Exception as exc:
|
||||
provider_attempts.append(
|
||||
{
|
||||
"provider": provider_name,
|
||||
"ok": False,
|
||||
"count": 0,
|
||||
"error_category": _classify_error(exc, provider_name),
|
||||
"error": _err_tag(provider_name, exc),
|
||||
"elapsed_ms": int((time.perf_counter() - started) * 1000),
|
||||
}
|
||||
)
|
||||
raise
|
||||
|
||||
# Skill-controlled path: explicit engine URL template or named engine.
|
||||
if engine_url:
|
||||
try:
|
||||
results = _attempt(
|
||||
"engine_url",
|
||||
lambda: _search_by_engine_url(
|
||||
engine_url_template=engine_url,
|
||||
query=query,
|
||||
max_results=max_results,
|
||||
timeout_s=timeout_s,
|
||||
),
|
||||
)
|
||||
used = "engine_url"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag("engine_url", exc))
|
||||
elif engine:
|
||||
template = _ENGINE_URLS.get(engine, "")
|
||||
if not template:
|
||||
errors.append(f"engine:unsupported:{engine}")
|
||||
else:
|
||||
try:
|
||||
results = _attempt(
|
||||
f"engine:{engine}",
|
||||
lambda: _search_by_engine_url(
|
||||
engine_url_template=template,
|
||||
query=query,
|
||||
max_results=max_results,
|
||||
timeout_s=timeout_s,
|
||||
),
|
||||
)
|
||||
used = f"engine:{engine}"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag(f"engine:{engine}", exc))
|
||||
|
||||
official_order: list[str] = []
|
||||
if official_provider in {"auto", "bing"} and official_bing_key:
|
||||
official_order.append("official_bing")
|
||||
if official_provider in {"auto", "google"} and official_google_key and official_google_cse_id:
|
||||
official_order.append("official_google")
|
||||
if official_provider == "google" and not official_order and official_bing_key:
|
||||
official_order.append("official_bing")
|
||||
if official_provider == "bing" and not official_order and official_google_key and official_google_cse_id:
|
||||
official_order.append("official_google")
|
||||
auto_order = [*official_order, *_AUTO_PROVIDER_ORDER_DEFAULT]
|
||||
run_order = auto_order if provider == "auto" else [provider]
|
||||
if provider == "official_api":
|
||||
run_order = official_order or ["official_bing", "official_google"]
|
||||
|
||||
for p in run_order:
|
||||
if results:
|
||||
break
|
||||
if p == "ddg_api":
|
||||
try:
|
||||
results = _attempt("ddg_api", lambda: _search_ddg_api(query, max_results=max_results, timeout_s=timeout_s))
|
||||
used = "ddg_api"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag("ddg_api", exc))
|
||||
elif p == "ddg_html":
|
||||
try:
|
||||
results = _attempt("ddg_html", lambda: _search_ddg_html(query, max_results=max_results, timeout_s=timeout_s))
|
||||
used = "ddg_html"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag("ddg_html", exc))
|
||||
elif p == "bing_html":
|
||||
try:
|
||||
results = _attempt("bing_html", lambda: _search_bing_html(query, max_results=max_results, timeout_s=timeout_s))
|
||||
used = "bing_html"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag("bing_html", exc))
|
||||
elif p == "official_api":
|
||||
continue
|
||||
elif p == "official_bing":
|
||||
if not official_bing_key:
|
||||
msg = f"official_bing:missing_key_env:{_OFFICIAL_BING_KEY_ENV}"
|
||||
errors.append(msg)
|
||||
provider_attempts.append(
|
||||
{
|
||||
"provider": "official_bing",
|
||||
"ok": False,
|
||||
"count": 0,
|
||||
"error_category": "auth_failed",
|
||||
"error": msg,
|
||||
"elapsed_ms": 0,
|
||||
}
|
||||
)
|
||||
continue
|
||||
try:
|
||||
results = _attempt(
|
||||
"official_bing",
|
||||
lambda: _search_official_api(
|
||||
query,
|
||||
max_results=max_results,
|
||||
timeout_s=timeout_s,
|
||||
api_key=official_bing_key,
|
||||
endpoint=official_bing_endpoint,
|
||||
),
|
||||
)
|
||||
used = "official_bing"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag("official_bing", exc))
|
||||
elif p == "official_google":
|
||||
if not official_google_key or not official_google_cse_id:
|
||||
miss = _OFFICIAL_GOOGLE_KEY_ENV if not official_google_key else _OFFICIAL_GOOGLE_CSE_ENV
|
||||
msg = f"official_google:missing_key_env:{miss}"
|
||||
errors.append(msg)
|
||||
provider_attempts.append(
|
||||
{
|
||||
"provider": "official_google",
|
||||
"ok": False,
|
||||
"count": 0,
|
||||
"error_category": "auth_failed",
|
||||
"error": msg,
|
||||
"elapsed_ms": 0,
|
||||
}
|
||||
)
|
||||
continue
|
||||
try:
|
||||
results = _attempt(
|
||||
"official_google",
|
||||
lambda: _search_official_google_api(
|
||||
query,
|
||||
max_results=max_results,
|
||||
timeout_s=timeout_s,
|
||||
api_key=official_google_key,
|
||||
cse_id=official_google_cse_id,
|
||||
endpoint=official_google_endpoint,
|
||||
),
|
||||
)
|
||||
used = "official_google"
|
||||
except Exception as exc:
|
||||
errors.append(_err_tag("official_google", exc))
|
||||
|
||||
if not results:
|
||||
error_category = "no_results"
|
||||
if errors:
|
||||
error_category = _classify_error_from_attempts(provider_attempts)
|
||||
if error_category == "unknown_error":
|
||||
error_category = _classify_error_from_tags(errors)
|
||||
return {
|
||||
"ok": False,
|
||||
"error_code": "search_failed",
|
||||
"error": "search_failed",
|
||||
"error_category": error_category,
|
||||
"provider": used,
|
||||
"query": query,
|
||||
"site": site,
|
||||
"engine": engine or "",
|
||||
"errors": errors,
|
||||
"tried": run_order,
|
||||
"provider_attempts": provider_attempts,
|
||||
}
|
||||
return {
|
||||
"ok": True,
|
||||
"query": query,
|
||||
"site": site,
|
||||
"engine": engine or "",
|
||||
"provider": used,
|
||||
"count": len(results),
|
||||
"results": results,
|
||||
"errors": errors,
|
||||
"provider_attempts": provider_attempts,
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="web_search_fast",
|
||||
description="Fast public web search with API-first and HTML fallback.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"query": {"type": "string", "description": "Search query."},
|
||||
"site": {"type": "string", "description": "Optional site/domain constraint; converted to `site:...` query qualifier."},
|
||||
"engine": {
|
||||
"type": "string",
|
||||
"description": "Optional named engine (e.g. bing, bing_cn, ddg, google, baidu, sogou).",
|
||||
},
|
||||
"engine_url": {
|
||||
"type": "string",
|
||||
"description": "Optional engine URL template containing `{keyword}`; overrides `engine` and provider fallback.",
|
||||
},
|
||||
"max_results": {"type": "integer", "default": 8, "description": "Max results [1..20]."},
|
||||
"provider": {
|
||||
"type": "string",
|
||||
"enum": ["auto", "official_api", "official_bing", "official_google", "ddg_api", "ddg_html", "bing_html"],
|
||||
"default": "auto",
|
||||
"description": "Search backend; auto tries configured official providers then ddg_html -> bing_html -> ddg_api.",
|
||||
},
|
||||
"official_provider": {
|
||||
"type": "string",
|
||||
"enum": ["auto", "bing", "google"],
|
||||
"default": "auto",
|
||||
"description": "Preference/order hint for official providers when provider=auto or official_api.",
|
||||
},
|
||||
"timeout": {"type": "integer", "default": 12, "description": "HTTP timeout seconds [3..30]."},
|
||||
},
|
||||
"required": ["query"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=_handler,
|
||||
tags=frozenset({"public", "search", "web", "read"}),
|
||||
risk_level="low",
|
||||
read_only=True,
|
||||
timeout_s=35.0,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["web_search_fast_tool"]
|
||||
|
||||
Loading…
Add table
Add a link
Reference in a new issue