diff --git a/interfaces/admin/skills_api.py b/interfaces/admin/skills_api.py index f6bdc11a..2231d1bf 100644 --- a/interfaces/admin/skills_api.py +++ b/interfaces/admin/skills_api.py @@ -15,6 +15,7 @@ from oclaw.runtime.skill_installer import ( install_skill_from_local_dir, install_skill_from_registry_archive, list_skills_with_status, + repair_skill_dependencies, set_skill_enabled, uninstall_skill, ) @@ -29,11 +30,13 @@ from oclaw.runtime.skill_role_binding import ( ) from oclaw.runtime.skills_prompt import collect_skill_catalog_entries from oclaw.runtime.skills import _allowed_tool_names_after_wire_policy, discover_workspace_skill_manifests -from oclaw.runtime.skills_market import get_market_adapter +from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting from oclaw.runtime.tools.skills_runtime.subprocess_exec import run_skill_runtime_entry from oclaw.platform.config.paths import db_path from oclaw.platform.persistence.sqlite_store import SqliteStore +_SKILL_MARKET_PROVIDER_KEY = "AIA_SKILL_MARKET_PROVIDER" + def include_skill_routes( router: APIRouter, @@ -95,7 +98,13 @@ def include_skill_routes( raw_toolcall = str(store.get_setting("AIA_SKILL_TOOLCALL_ENABLED") or "").strip().lower() prompt_in_system = raw_prompt not in {"0", "false", "no", "off"} toolcall_enabled = raw_toolcall in {"1", "true", "yes", "on"} - return {"ok": True, "prompt_in_system": bool(prompt_in_system), "toolcall_enabled": bool(toolcall_enabled)} + market_provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")) + return { + "ok": True, + "prompt_in_system": bool(prompt_in_system), + "toolcall_enabled": bool(toolcall_enabled), + "market_provider": market_provider, + } @sk.post("/mode") def api_skills_mode_save( @@ -110,19 +119,31 @@ def include_skill_routes( store.set_setting("AIA_SKILLS_PROMPT_IN_SYSTEM", "1" if bool(payload.get("prompt_in_system")) else "0") if "toolcall_enabled" in payload: store.set_setting("AIA_SKILL_TOOLCALL_ENABLED", "1" if bool(payload.get("toolcall_enabled")) else "0") + if "market_provider" in payload: + store.set_setting(_SKILL_MARKET_PROVIDER_KEY, normalize_skill_market_provider_setting(str(payload.get("market_provider") or ""))) raw_prompt = str(store.get_setting("AIA_SKILLS_PROMPT_IN_SYSTEM") or "").strip().lower() raw_toolcall = str(store.get_setting("AIA_SKILL_TOOLCALL_ENABLED") or "").strip().lower() prompt_in_system = raw_prompt not in {"0", "false", "no", "off"} toolcall_enabled = raw_toolcall in {"1", "true", "yes", "on"} + market_provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")) _audit( store, ctx, action="skill_mode_update", target_id="skill_mode", status="ok", - detail={"prompt_in_system": bool(prompt_in_system), "toolcall_enabled": bool(toolcall_enabled)}, + detail={ + "prompt_in_system": bool(prompt_in_system), + "toolcall_enabled": bool(toolcall_enabled), + "market_provider": market_provider, + }, ) - return {"ok": True, "prompt_in_system": bool(prompt_in_system), "toolcall_enabled": bool(toolcall_enabled)} + return { + "ok": True, + "prompt_in_system": bool(prompt_in_system), + "toolcall_enabled": bool(toolcall_enabled), + "market_provider": market_provider, + } @sk.post("/install") def api_skills_install( @@ -204,7 +225,7 @@ def include_skill_routes( query = str(q or "").strip() lim = int(limit) if isinstance(limit, int) and limit > 0 else 20 lim = max(1, min(lim, 200)) - provider = str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub").strip() + provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")) items = get_market_adapter(provider).search(query, limit=lim) return {"ok": True, "items": items} @@ -219,7 +240,7 @@ def include_skill_routes( s = str(slug or "").strip() if not s: raise HTTPException(status_code=400, detail="slug_required") - provider = str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub").strip() + provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")) detail = get_market_adapter(provider).detail(s) return {"ok": True, "detail": detail} @@ -238,7 +259,7 @@ def include_skill_routes( requested_version = str(payload.get("version") or "").strip() overwrite = bool(payload.get("overwrite")) - provider = str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub").strip() + provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")) adapter = get_market_adapter(provider) archive_url, chosen_version = adapter.resolve_archive_url(slug=s, version=requested_version or None) @@ -673,6 +694,98 @@ def include_skill_routes( }, } + @sk.post("/repair-deps") + def api_skills_repair_deps( + payload: dict[str, Any] | None = Body(default=None), + authorization: str | None = Header(default=None), + ) -> dict[str, Any]: + payload = payload or {} + store = SqliteStore(db_path()) + ctx = resolve_auth(store, authorization) + _require_admin(ctx) + name = str(payload.get("name") or "").strip() + if not name: + raise HTTPException(status_code=400, detail="name_required") + _audit( + store, + ctx, + action="skill_repair_deps_started", + target_id=name, + status="start", + ) + out = repair_skill_dependencies(store=store, skill_name=name) + _audit( + store, + ctx, + action="skill_repair_deps_finished" if bool(out.get("ok")) else "skill_repair_deps_failed", + target_id=name, + status="ok" if bool(out.get("ok")) else "fail", + detail=out, + ) + return {"ok": bool(out.get("ok")), "result": out} + + @sk.post("/repair-deps-all") + def api_skills_repair_deps_all( + payload: dict[str, Any] | None = Body(default=None), + authorization: str | None = Header(default=None), + ) -> dict[str, Any]: + _payload = payload or {} + store = SqliteStore(db_path()) + ctx = resolve_auth(store, authorization) + _require_admin(ctx) + items = list_skills_with_status(store=store) + results: list[dict[str, Any]] = [] + ok_count = 0 + warn_count = 0 + fail_count = 0 + for it in items: + name = str((it or {}).get("name") or "").strip() + if not name: + continue + _audit( + store, + ctx, + action="skill_repair_deps_started", + target_id=name, + status="start", + detail={"batch": True}, + ) + out = repair_skill_dependencies(store=store, skill_name=name) + warnings = list(out.get("warnings") or []) if isinstance(out, dict) else [] + if bool(out.get("ok")): + if warnings: + warn_count += 1 + else: + ok_count += 1 + else: + fail_count += 1 + _audit( + store, + ctx, + action="skill_repair_deps_finished" if bool(out.get("ok")) else "skill_repair_deps_failed", + target_id=name, + status="ok" if bool(out.get("ok")) else "fail", + detail={"batch": True, **(out if isinstance(out, dict) else {})}, + ) + results.append( + { + "name": name, + "ok": bool(out.get("ok")) if isinstance(out, dict) else False, + "warnings": warnings, + "detail": str((out or {}).get("detail") or "") if isinstance(out, dict) else "", + } + ) + return { + "ok": True, + "summary": { + "total": len(results), + "ok_count": ok_count, + "warn_count": warn_count, + "fail_count": fail_count, + }, + "items": results, + } + @sk.post("/test-run") def api_skills_test_run( payload: dict[str, Any] | None = Body(default=None), @@ -778,7 +891,7 @@ def include_skill_routes( "execution_checked_total": len(execution_checks), "execution_checks": execution_checks, "classification_counts": classification_counts, - "market_provider": str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub"), + "market_provider": normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")), } router.include_router(sk) diff --git a/interfaces/admin/static/app.js b/interfaces/admin/static/app.js index fa94591f..7553dc04 100644 --- a/interfaces/admin/static/app.js +++ b/interfaces/admin/static/app.js @@ -1087,6 +1087,16 @@ async function apiPost(path, body) { return data ?? {}; } +/** Skills endpoints often return HTTP 200 with `{ ok: false, result: {...} }` on failure — treat as error for UX. */ +function assertSkillMutationOk(r, fallbackMessage) { + if (!r || r.ok !== false) return; + const res = r.result && typeof r.result === "object" ? r.result : {}; + const code = res.error_code != null ? String(res.error_code).trim() : ""; + const detail = res.detail != null ? String(res.detail).trim() : ""; + const msg = [code, detail].filter(Boolean).join(": ") || String(fallbackMessage || "skill operation failed"); + throw new Error(msg); +} + async function apiRequest(method, path, body) { const url = resolveAdminApiUrl(path); const token = getStoredAuthToken(); @@ -7363,7 +7373,11 @@ async function renderSkills() { const skillModeStatus = el("div", { class: "muted", text: "" }); const skillPromptModeCb = el("input", { type: "checkbox" }); const skillToolcallModeCb = el("input", { type: "checkbox" }); - const marketQ = el("input", { class: "input", placeholder: "search ClawHub skills" }); + const skillMarketProviderSelect = el("select", { class: "input", style: "min-width:160px;" }, [ + el("option", { value: "clawhub", text: "clawhub (ClawHub)" }), + el("option", { value: "cocoloop", text: "cocoloop (CocoLoop)" }), + ]); + const marketQ = el("input", { class: "input", placeholder: "search skills (keyword)" }); const marketLimitInp = el("input", { class: "input", placeholder: "limit", value: "40", style: "max-width:120px;" }); const marketTbody = el("tbody"); const marketDetailPre = el("pre", { class: "muted pre", text: "" }); @@ -7374,6 +7388,8 @@ async function renderSkills() { const r = await apiGet("/admin/api/skills/mode"); skillPromptModeCb.checked = !!r.prompt_in_system; skillToolcallModeCb.checked = !!r.toolcall_enabled; + const mp = String(r.market_provider || "clawhub").trim().toLowerCase(); + skillMarketProviderSelect.value = mp === "cocoloop" ? "cocoloop" : "clawhub"; skillModeStatus.textContent = ""; } catch (e) { skillModeStatus.textContent = `mode: ${String(e && e.message ? e.message : e)}`; @@ -7385,10 +7401,13 @@ async function renderSkills() { const r = await apiPost("/admin/api/skills/mode", { prompt_in_system: !!skillPromptModeCb.checked, toolcall_enabled: !!skillToolcallModeCb.checked, + market_provider: String(skillMarketProviderSelect.value || "clawhub").trim(), }); skillPromptModeCb.checked = !!r.prompt_in_system; skillToolcallModeCb.checked = !!r.toolcall_enabled; - skillModeStatus.textContent = `saved: prompt=${String(!!r.prompt_in_system)} toolcall=${String(!!r.toolcall_enabled)}`; + const mp = String(r.market_provider || "clawhub").trim().toLowerCase(); + skillMarketProviderSelect.value = mp === "cocoloop" ? "cocoloop" : "clawhub"; + skillModeStatus.textContent = `saved: prompt=${String(!!r.prompt_in_system)} toolcall=${String(!!r.toolcall_enabled)} market=${String(skillMarketProviderSelect.value)}`; } catch (e) { skillModeStatus.textContent = `mode: ${String(e && e.message ? e.message : e)}`; } @@ -7429,7 +7448,8 @@ async function renderSkills() { openSkillInstallModal(`Installing ${s}...`); try { const r = await apiPost("/admin/api/skills/market/install", { slug: s, version: version ? String(version) : undefined, overwrite: false }); - status.textContent = `install-clawhub success: ${JSON.stringify(r.result || {})}`; + assertSkillMutationOk(r, "Market install failed"); + status.textContent = `install-market success: ${JSON.stringify(r.result || {})}`; marketStatus.textContent = `installed: ${s}`; await refreshSkillsState(); finishSkillInstallModal(true, `${s} installed successfully.`); @@ -7446,14 +7466,13 @@ async function renderSkills() { const slug = String(x.slug || ""); const ver = String(x.version || ""); const btnDetail = el("button", { class: "btn btn--small", text: "Detail", onclick: async () => await loadMarketDetail(slug) }); - const btnInstall = el("button", { class: "btn btn--small btn--primary", text: "Install", onclick: async () => await installFromMarket(slug, ver || undefined) }); marketTbody.appendChild( el("tr", {}, [ el("td", { text: slug }), el("td", { text: String(x.name || "") }), el("td", { text: ver }), el("td", { text: shortText(String(x.description || ""), 80) }), - el("td", {}, [btnDetail, el("span", { style: "display:inline-block;width:6px" }), btnInstall]), + el("td", {}, [btnDetail]), ]), ); }); @@ -7462,7 +7481,7 @@ async function renderSkills() { const btnMarketSearch = el("button", { class: "btn", text: "Search", onclick: async () => await loadMarket(marketQ.value) }); const btnMarketLatest = el("button", { class: "btn", text: "Latest", onclick: async () => await loadMarket("") }); const marketBox = el("details", { style: "margin:10px 0 14px 0;" }, [ - el("summary", { text: "ClawHub Market", style: "cursor:pointer;user-select:none;" }), + el("summary", { text: "Skill market (ClawHub / CocoLoop)", style: "cursor:pointer;user-select:none;" }), el("div", { style: "height:8px" }), el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [marketQ, marketLimitInp, btnMarketSearch, btnMarketLatest]), marketStatus, @@ -8400,6 +8419,24 @@ async function renderSkills() { openSkillTestRunModal(name); }, }); + const repairDepsBtn = el("button", { + class: "chat-sess-menu-item", + text: "Repair deps", + onclick: async () => { + closeSkillActionMenu(); + try { + status.textContent = `repair deps: ${name}...`; + const r = await apiPost("/admin/api/skills/repair-deps", { name }); + assertSkillMutationOk(r, "Repair deps failed"); + status.textContent = `repair deps: ${JSON.stringify((r && r.result) || {}, null, 0)}`; + await loadRows(); + await loadAudits(); + repaint(); + } catch (e) { + status.textContent = `repair deps failed: ${String(e && e.message ? e.message : e)}`; + } + }, + }); const uninstallBtn = el("button", { class: "chat-sess-menu-item", text: "Uninstall", @@ -8409,6 +8446,7 @@ async function renderSkills() { try { status.textContent = `uninstalling: ${name}...`; const r = await apiPost("/admin/api/skills/uninstall", { name }); + assertSkillMutationOk(r, "Uninstall failed"); status.textContent = `uninstall success: ${JSON.stringify((r && r.result) || {}, null, 0)}`; await loadRows(); await loadAudits(); @@ -8428,6 +8466,7 @@ async function renderSkills() { const menu = el("div", { class: "chat-sess-menu-pop", style: "position:fixed;z-index:250;" }, [ toggleBtn, testRunBtn, + repairDepsBtn, uninstallBtn, ]); const rect = ev.currentTarget.getBoundingClientRect(); @@ -8483,6 +8522,7 @@ async function renderSkills() { return; } const r = await apiPost("/admin/api/skills/retry-install", { source: src, target }); + assertSkillMutationOk(r, "Retry install failed"); status.textContent = `retry: ${JSON.stringify(r.result || {})}`; await loadRows(); await loadAudits(); @@ -8544,6 +8584,7 @@ async function renderSkills() { description: String(descInp.value || "").trim(), body_markdown: String(bodyInp.value || ""), }); + assertSkillMutationOk(r, "Create skill failed"); status.textContent = `create: ${JSON.stringify(r.result || {})}`; await loadRows(); await loadAudits(); @@ -8566,6 +8607,7 @@ async function renderSkills() { const r = await apiPost("/admin/api/skills/install-registry", { archive_url: String(regInp.value || "").trim(), }); + assertSkillMutationOk(r, "Registry install failed"); status.textContent = `install-registry success: ${JSON.stringify(r.result || {})}`; await refreshSkillsState(); finishSkillInstallModal(true, "Registry skill installed successfully."); @@ -8591,6 +8633,7 @@ async function renderSkills() { const r = await apiPost("/admin/api/skills/install", { source_dir: String(localDirInp.value || "").trim(), }); + assertSkillMutationOk(r, "Local install failed"); status.textContent = `install-local success: ${JSON.stringify(r.result || {})}`; await refreshSkillsState(); finishSkillInstallModal(true, "Local skill installed successfully."); @@ -8615,6 +8658,26 @@ async function renderSkills() { } }, }); + const btnRepairDepsAll = el("button", { + class: "btn", + text: "Repair all deps", + onclick: async () => { + const prev = btnRepairDepsAll.textContent; + btnRepairDepsAll.disabled = true; + btnRepairDepsAll.textContent = "Repairing..."; + try { + const r = await apiPost("/admin/api/skills/repair-deps-all", {}); + const s = r && typeof r.summary === "object" ? r.summary : {}; + status.textContent = `repair all deps: total=${Number(s.total || 0)} ok=${Number(s.ok_count || 0)} warn=${Number(s.warn_count || 0)} fail=${Number(s.fail_count || 0)}`; + await refreshSkillsState(); + } catch (e) { + status.textContent = `repair all deps failed: ${String(e && e.message ? e.message : e)}`; + } finally { + btnRepairDepsAll.disabled = false; + btnRepairDepsAll.textContent = prev; + } + }, + }); retryableOnlyCb.addEventListener("change", () => { localStorage.setItem(SKILL_AUDIT_RETRYABLE_ONLY_KEY, retryableOnlyCb.checked ? "1" : "0"); repaint(); @@ -8650,6 +8713,10 @@ async function renderSkills() { el("div", { class: "row", style: "gap:8px;align-items:center;flex-wrap:wrap;margin-bottom:8px;" }, [ el("label", { class: "row", style: "gap:6px;align-items:center;" }, [skillPromptModeCb, el("span", { text: "Prompt mode (inject SKILL.md)" })]), el("label", { class: "row", style: "gap:6px;align-items:center;" }, [skillToolcallModeCb, el("span", { text: "Toolcall mode (runtime as tools)" })]), + el("label", { class: "row", style: "gap:6px;align-items:center;flex-wrap:wrap;" }, [ + el("span", { text: "Market (AIA_SKILL_MARKET_PROVIDER)" }), + skillMarketProviderSelect, + ]), el("button", { class: "btn", text: "Save skill mode", onclick: saveSkillMode }), skillModeStatus, ]), @@ -8663,8 +8730,7 @@ async function renderSkills() { el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [nameInp, descInp]), el("div", { style: "margin-bottom:8px;" }, [bodyInp]), el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [btnCreate]), - el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [regInp, btnInstallRegistry, btnRefresh]), - el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [localDirInp, btnInstallLocal]), + el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [btnRefresh, btnRepairDepsAll]), el("div", { class: "table-wrap" }, [ el("table", { class: "table table--compact" }, [ el("thead", {}, [el("tr", {}, [ diff --git a/interfaces/admin/static/theme-deepseek.css b/interfaces/admin/static/theme-deepseek.css index 5ae19bc9..b8ab7cb0 100644 --- a/interfaces/admin/static/theme-deepseek.css +++ b/interfaces/admin/static/theme-deepseek.css @@ -475,6 +475,7 @@ body.theme-ds-body .card { flex: 1 1 auto; width: auto; max-width: min(max(0px, calc(50% + 2cm - 58px)), 100%); + align-items: flex-start; } .chat-msg-col--user { @@ -573,7 +574,8 @@ body.theme-ds-body .card { .chat-msg--assistant { align-self: flex-start; - width: 100%; + width: auto; + max-width: 100%; background: rgba(255, 255, 255, 0.04); border: 1px solid var(--ds-border, rgba(255, 255, 255, 0.08)); } diff --git a/runtime/chat/tool_runtime.py b/runtime/chat/tool_runtime.py index c3b7d748..1dd442b4 100644 --- a/runtime/chat/tool_runtime.py +++ b/runtime/chat/tool_runtime.py @@ -696,6 +696,8 @@ class ToolExecutor: results_by_id: dict[str, tuple[dict[str, Any], int]] = {} runnable_tool_uses: list[LLMToolCall] = [] + dedupe_alias_to_source: dict[str, str] = {} + first_tool_call_id_by_signature: dict[str, str] = {} sig_seen: dict[str, int] = {} budget = max(1, min(int(signature_budget or 2), 8)) for tc in tool_uses: @@ -796,6 +798,19 @@ class ToolExecutor: ) continue sig_seen[sig] = count + 1 + source_tool_call_id = str(first_tool_call_id_by_signature.get(sig) or "").strip() + if source_tool_call_id: + dedupe_alias_to_source[str(tc.id or "")] = source_tool_call_id + _trace( + "tool_cache_hit_same_round", + { + "tool_name": tc.name, + "tool_call_id": str(tc.id or ""), + "source_tool_call_id": source_tool_call_id, + }, + ) + continue + first_tool_call_id_by_signature[sig] = str(tc.id or "") runnable_tool_uses.append(tc) for batch in partition_tool_use_batches(runnable_tool_uses, ctx.tools): @@ -825,6 +840,10 @@ class ToolExecutor: }, ) + for tool_call_id, source_tool_call_id in dedupe_alias_to_source.items(): + if source_tool_call_id in results_by_id: + results_by_id[tool_call_id] = results_by_id[source_tool_call_id] + tool_messages: list[dict[str, Any]] = [] for tc in tool_uses: _check_stop() diff --git a/runtime/skill_installer.py b/runtime/skill_installer.py index a7abd3fd..8a1fd16e 100644 --- a/runtime/skill_installer.py +++ b/runtime/skill_installer.py @@ -1,10 +1,16 @@ from __future__ import annotations +import ast +import importlib.util import json import shutil +import subprocess +import sys import tarfile import tempfile +import time import urllib.parse +import urllib.error import urllib.request import zipfile from dataclasses import dataclass @@ -27,6 +33,33 @@ from oclaw.runtime.skills import ( _DISABLED_SKILLS_KEY = "AIA_SKILL_DISABLED_NAMES" _AUTO_INSTALL_KEY = "AIA_SKILL_AUTO_INSTALL_ENABLED" _AUTO_ENABLE_TRUSTED_KEY = "AIA_SKILL_AUTO_ENABLE_TRUSTED" +_AUTO_INSTALL_DEPS_KEY = "AIA_SKILL_AUTO_INSTALL_DEPS_ENABLED" +_ARCHIVE_DOWNLOAD_TIMEOUT_S = 45 +_ARCHIVE_DOWNLOAD_MAX_RETRIES = 3 + + +def _download_archive_bytes(url: str) -> bytes: + req = urllib.request.Request( + url, + headers={ + "User-Agent": "Oclaw-SkillInstaller/1.0 (+https://clawhub.ai)", + "Accept": "*/*", + }, + ) + last_exc: Exception | None = None + for attempt in range(1, _ARCHIVE_DOWNLOAD_MAX_RETRIES + 1): + try: + with urllib.request.urlopen(req, timeout=_ARCHIVE_DOWNLOAD_TIMEOUT_S) as resp: + return resp.read() + except Exception as exc: + last_exc = exc + code = getattr(exc, "code", None) + retryable_http = isinstance(code, int) and code in {408, 429, 500, 502, 503, 504} + retryable_net = isinstance(exc, urllib.error.URLError) or isinstance(exc, TimeoutError) + if attempt >= _ARCHIVE_DOWNLOAD_MAX_RETRIES or not (retryable_http or retryable_net): + raise + time.sleep(0.8 * attempt) + raise last_exc or RuntimeError("download_failed_unknown") @dataclass(frozen=True) @@ -85,6 +118,144 @@ def skill_auto_enable_trusted_enabled(store: Any) -> bool: return _truthy(raw) +def skill_auto_install_deps_enabled(store: Any) -> bool: + try: + raw = str(store.get_setting(_AUTO_INSTALL_DEPS_KEY) or "").strip() + except Exception: + raw = "" + # Default ON to reduce first-run dependency issues for newly installed skills. + if not raw: + return True + return _truthy(raw) + + +def _scan_dependency_manifests(skill_dir: Path) -> dict[str, list[Path]]: + req_files: list[Path] = [] + pkg_files: list[Path] = [] + try: + for p in skill_dir.rglob("requirements.txt"): + if p.is_file(): + req_files.append(p) + for p in skill_dir.rglob("package.json"): + if p.is_file(): + pkg_files.append(p) + except Exception: + return {"requirements": [], "package_json": []} + req_files = sorted(req_files) + pkg_files = sorted(pkg_files) + return {"requirements": req_files, "package_json": pkg_files} + + +def _run_dep_install(command: list[str], *, cwd: Path) -> tuple[bool, str]: + try: + cp = subprocess.run( + command, + cwd=str(cwd), + capture_output=True, + text=True, + timeout=600, + check=False, + ) + except Exception as exc: + return False, f"{type(exc).__name__}" + if int(cp.returncode) == 0: + return True, "" + err = str(cp.stderr or cp.stdout or "").strip() + if err: + err = err[:160].replace("\n", " ") + return False, err or f"exit_{cp.returncode}" + + +def _auto_install_skill_dependencies(*, store: Any, skill_dir: Path) -> tuple[bool, str]: + if not skill_auto_install_deps_enabled(store): + return True, "" + manifests = _scan_dependency_manifests(skill_dir) + req_files = manifests.get("requirements") or [] + pkg_files = manifests.get("package_json") or [] + if not req_files and not pkg_files: + return True, "" + + failures: list[str] = [] + # Python dependencies + for req in req_files: + ok, detail = _run_dep_install([sys.executable, "-m", "pip", "install", "-r", str(req)], cwd=req.parent) + if not ok: + failures.append(f"pip:{req.name}:{detail}") + + # Node dependencies (only when package.json declares deps) + for pkg in pkg_files: + try: + obj = json.loads(pkg.read_text(encoding="utf-8")) + except Exception: + obj = {} + deps = obj.get("dependencies") if isinstance(obj, dict) else None + if not isinstance(deps, dict) or not deps: + continue + ok, detail = _run_dep_install(["npm", "install", "--omit=dev"], cwd=pkg.parent) + if not ok: + failures.append(f"npm:{pkg.name}:{detail}") + if failures: + return False, "; ".join(failures[:4]) + return True, "" + + +def _collect_python_import_roots(skill_dir: Path) -> set[str]: + names: set[str] = set() + for py in skill_dir.rglob("*.py"): + if not py.is_file(): + continue + try: + tree = ast.parse(py.read_text(encoding="utf-8"), filename=str(py)) + except Exception: + continue + for node in ast.walk(tree): + if isinstance(node, ast.Import): + for alias in node.names: + root = str(alias.name or "").split(".", 1)[0].strip() + if root: + names.add(root) + elif isinstance(node, ast.ImportFrom): + if int(node.level or 0) > 0: + continue + root = str(node.module or "").split(".", 1)[0].strip() + if root: + names.add(root) + return names + + +def _local_python_module_roots(skill_dir: Path) -> set[str]: + roots: set[str] = set() + for py in skill_dir.rglob("*.py"): + if py.is_file(): + roots.add(py.stem) + for p in skill_dir.rglob("*"): + if p.is_dir() and (p / "__init__.py").exists(): + roots.add(p.name) + return {x for x in roots if x and x != "__init__"} + + +def _auto_probe_and_install_python_imports(*, store: Any, skill_dir: Path) -> tuple[bool, str]: + if not skill_auto_install_deps_enabled(store): + return True, "" + imports = _collect_python_import_roots(skill_dir) + if not imports: + return True, "" + local_roots = _local_python_module_roots(skill_dir) + stdlib = set(getattr(sys, "stdlib_module_names", set()) or set()) + missing: list[str] = [] + for name in sorted(imports): + if name in local_roots or name in stdlib: + continue + if importlib.util.find_spec(name) is None: + missing.append(name) + if not missing: + return True, "" + ok, detail = _run_dep_install([sys.executable, "-m", "pip", "install", *missing], cwd=skill_dir) + if not ok: + return False, f"probe_pip:{detail}" + return True, "" + + def _get_disabled_names(store: Any) -> set[str]: try: raw = str(store.get_setting(_DISABLED_SKILLS_KEY) or "").strip() @@ -296,6 +467,7 @@ def install_skill_from_local_dir( source_dir: str | Path, overwrite: bool = False, skills_root: str | Path | None = None, + auto_bind: bool = False, ) -> SkillInstallResult: src = Path(source_dir).resolve() ok, detail = scan_skill_source_dir(src) @@ -317,8 +489,49 @@ def install_skill_from_local_dir( shutil.rmtree(target) shutil.copytree(src, target) set_skill_enabled(store=store, skill_name=manifest.name, enabled=True) + auto_enabled = False + binding_roles: tuple[str, ...] = () + if auto_bind: + binding_root = root.parent if str(root.name).strip().lower() == "_workspace" else root + auto_enabled, binding_roles = _apply_auto_enable_binding(store=store, skill_name=manifest.name, skills_root=binding_root) + deps_ok, deps_detail = _auto_install_skill_dependencies(store=store, skill_dir=target) + if not deps_ok: + # Keep install successful; surface dependency warning for operator follow-up. + ec, rt = _classify_install_detail("installed") + return SkillInstallResult( + ok=True, + name=manifest.name, + target_dir=str(target), + detail=f"installed_with_dependency_warnings:{deps_detail}", + error_code=ec, + retryable=rt, + auto_enabled=auto_enabled, + binding_applied_roles=binding_roles, + ) + probe_ok, probe_detail = _auto_probe_and_install_python_imports(store=store, skill_dir=target) + if not probe_ok: + ec, rt = _classify_install_detail("installed") + return SkillInstallResult( + ok=True, + name=manifest.name, + target_dir=str(target), + detail=f"installed_with_dependency_warnings:{probe_detail}", + error_code=ec, + retryable=rt, + auto_enabled=auto_enabled, + binding_applied_roles=binding_roles, + ) ec, rt = _classify_install_detail("installed") - return SkillInstallResult(ok=True, name=manifest.name, target_dir=str(target), detail="installed", error_code=ec, retryable=rt) + return SkillInstallResult( + ok=True, + name=manifest.name, + target_dir=str(target), + detail="installed", + error_code=ec, + retryable=rt, + auto_enabled=auto_enabled, + binding_applied_roles=binding_roles, + ) def install_skill_from_registry_archive( @@ -327,6 +540,7 @@ def install_skill_from_registry_archive( archive_url: str, overwrite: bool = False, skills_root: str | Path | None = None, + auto_bind: bool = False, ) -> SkillInstallResult: url = str(archive_url or "").strip() if not url: @@ -339,15 +553,7 @@ def install_skill_from_registry_archive( return SkillInstallResult(ok=False, name="", target_dir="", detail="unsupported_url_scheme", error_code=ec, retryable=rt) tmp_file = Path(tempfile.mkstemp(prefix="skill_pkg_", suffix=".bin")[1]) try: - req = urllib.request.Request( - url, - headers={ - "User-Agent": "Oclaw-SkillInstaller/1.0 (+https://clawhub.ai)", - "Accept": "*/*", - }, - ) - with urllib.request.urlopen(req, timeout=20) as resp: - data = resp.read() + data = _download_archive_bytes(url) if not data: ec, rt = _classify_install_detail("empty_archive") return SkillInstallResult(ok=False, name="", target_dir="", detail="empty_archive", error_code=ec, retryable=rt) @@ -369,7 +575,13 @@ def install_skill_from_registry_archive( ec, rt = _classify_install_detail("skill_md_missing") return SkillInstallResult(ok=False, name="", target_dir="", detail="skill_md_missing", error_code=ec, retryable=rt) chosen = sorted(candidates, key=lambda x: len(x.parts))[0] - return install_skill_from_local_dir(store=store, source_dir=chosen, overwrite=overwrite, skills_root=skills_root) + return install_skill_from_local_dir( + store=store, + source_dir=chosen, + overwrite=overwrite, + skills_root=skills_root, + auto_bind=auto_bind, + ) except Exception as exc: detail = f"download_failed:{type(exc).__name__}" code = getattr(exc, "code", None) @@ -569,6 +781,36 @@ def auto_install_skill_from_payload( return SkillInstallResult(ok=False, name=name, target_dir=str(target), detail=detail, error_code=ec, retryable=rt) +def repair_skill_dependencies( + *, + store: Any, + skill_name: str, + skills_root: str | Path | None = None, +) -> dict[str, Any]: + nm = str(skill_name or "").strip() + if not nm: + return {"ok": False, "error_code": "name_required", "error": "name_required"} + manifests = list(discover_workspace_skill_manifests(skills_root)) + mf = next((m for m in manifests if str(m.name or "").strip() == nm), None) + if mf is None: + return {"ok": False, "error_code": "skill_not_found", "error": "skill_not_found", "name": nm} + skill_dir = Path(str(mf.skill_dir or "")).resolve() + deps_ok, deps_detail = _auto_install_skill_dependencies(store=store, skill_dir=skill_dir) + probe_ok, probe_detail = _auto_probe_and_install_python_imports(store=store, skill_dir=skill_dir) + warnings: list[str] = [] + if not deps_ok and deps_detail: + warnings.append(f"manifest:{deps_detail}") + if not probe_ok and probe_detail: + warnings.append(f"probe:{probe_detail}") + return { + "ok": True, + "name": nm, + "skill_dir": str(skill_dir), + "warnings": warnings, + "detail": "ok" if not warnings else f"warnings:{'; '.join(warnings[:4])}", + } + + __all__ = [ "SkillInstallResult", "auto_install_skill_from_payload", @@ -577,6 +819,7 @@ __all__ = [ "install_skill_from_local_dir", "install_skill_from_registry_archive", "list_skills_with_status", + "repair_skill_dependencies", "set_skill_enabled", "skill_auto_install_enabled", "uninstall_skill", diff --git a/runtime/skills/README.md b/runtime/skills/README.md index f4ae595e..39817c6e 100644 --- a/runtime/skills/README.md +++ b/runtime/skills/README.md @@ -25,6 +25,12 @@ - 若设置了环境变量 `AIA_SKILLS_ROOT`,以该变量为准。 - 为兼容旧工程,仍可回退读取旧路径 `oclaw/runtime/skills/`(如存在)。 +## 技能市场(安装来源) + +- 租户设置 **`AIA_SKILL_MARKET_PROVIDER`**:`clawhub`(默认)或 **`cocoloop`**,由 `runtime/skills_market.get_market_adapter` 选择适配器;Admin「市场搜索 / 按 slug 安装」共用同一套路由。 +- ClawHub:见 `runtime/tools/skills/clawhub_client.py`(`AIA_CLAWHUB_*` / `CLAWHUB_*`)。公开 API 说明可参考 [openclaw/clawhub CLI 文档](https://github.com/openclaw/clawhub/blob/main/docs/cli.md)。 +- CocoLoop:`runtime/tools/skills/cocoloop_client.py`,默认 API 基址 `https://api.cocoloop.com`,可用 **`AIA_COCOLOOP_API_BASE`** 覆盖。 + ## 推荐实用 Skills(workspace) 以下为当前已落地并可直接在 Admin `Test run` 使用的实用技能: diff --git a/runtime/skills/_workspace/SkillScan/SKILL.md b/runtime/skills/_workspace/SkillScan/SKILL.md new file mode 100644 index 00000000..9ab299ba --- /dev/null +++ b/runtime/skills/_workspace/SkillScan/SKILL.md @@ -0,0 +1,97 @@ +--- +name: SkillScan +metadata: + version: "1.1.6" +description: > + Security gate for skills. Every new skill MUST pass SkillScan before use. + Activate on any install, load, add, evaluate, or safety question about a skill. + On first load, run first-run to scan all existing skills. + Blocks HIGH/CRITICAL skills. No exceptions. +--- + +# SkillScan v1.1.6 + +Security check for skill packages. Every new skill must pass this scan before use. + +Implemented in `scripts/scanner.py`. Supports Windows / macOS / Linux. + +Do NOT generate report files inside the skill directory. + +--- + +## When to Activate + +Run SkillScan whenever a skill is installed, loaded, added, evaluated, or questioned for safety — regardless of method or phrasing. + +Trigger examples: install, load, add, use, set up, "is this safe", upload .zip/.skill, paste SKILL.md, mention unknown skill name, install from hub/store. + +--- + +## First Load + +Check `scripts/.first_run_done`: + +- **Not found** → **Tell user** SkillScan is installed, ask to scan all existing skills: + ```bash + python /scripts/scanner.py first-run + ``` + `.first_run_done` is created automatically after completion. + +- **Found** → Skip, operate normally. + +--- + +## Commands + +| Command | Usage | +|---------|-------| +| `scanner.py scan ` | Scan a single skill (.zip or directory) | +| `scanner.py scan-all` | Scan all installed skills | +| `scanner.py first-run` | First-time full scan | +| `scanner.py upgrade` | Manual upgrade | + +--- + +## Scan Rules + +- **.zip files** → Scan BEFORE installation. Block if fails. +- **Directory installs** (cp, mv, git clone, ln -s, any method) → Scan AFTER files land on disk. +- **Remote installs** (clawhub, skillhub, npx skills add, etc.) → Scan immediately after install. +- **Unknown skills** → If user mentions a skill you haven't seen, scan it. + +--- + +## Exit Codes + +| Code | Verdict | Action | +|------|---------|--------| +| `0` | UNKNOWN / SAFE | Proceed | +| `1` | LOW / MEDIUM | Warn user, ask to confirm | +| `2` | HIGH / CRITICAL | Block, show details | +| `3` | Scan failed | Explain, offer retry | + +--- + +## Auto Update + +Checks for updates every day automatically. Silent, no user action needed. Manual: `scanner.py upgrade`. + +--- + +## Environment Variables + +| Variable | Description | +|----------|-------------| +| `SKILL_SCANNER_UPDATE_URL` | Custom update source (optional) | + +--- + +## API + +Base URL: `https://skillscan.tokauth.com` + +| Step | Method | Path | +|------|--------|------| +| ① Cache lookup | GET | `/oapi/v1/skill-scan/search?dir_sha256=` | +| ② Upload | POST | `/oapi/v1/skill-scan/upload` | +| ③ Poll result | GET | `/oapi/v1/skill-scan/result?task_no=` (poll every 20s, max 180s) | diff --git a/runtime/skills/_workspace/SkillScan/_meta.json b/runtime/skills/_workspace/SkillScan/_meta.json new file mode 100644 index 00000000..e9cce2eb --- /dev/null +++ b/runtime/skills/_workspace/SkillScan/_meta.json @@ -0,0 +1,6 @@ +{ + "ownerId": "kn791cyx98pcsezkh5088g8jxn84c7mm", + "slug": "skillscan", + "version": "1.1.6", + "publishedAt": 1776650587310 +} \ No newline at end of file diff --git a/runtime/skills/_workspace/SkillScan/scripts/scanner.py b/runtime/skills/_workspace/SkillScan/scripts/scanner.py new file mode 100644 index 00000000..5579be50 --- /dev/null +++ b/runtime/skills/_workspace/SkillScan/scripts/scanner.py @@ -0,0 +1,959 @@ +#!/usr/bin/env python3 +""" +SkillScan v1.1.5 — OpenClaw Skill security scanner. +Supports Windows / macOS / Linux. All temp files use the standard tempfile module. + +Usage (invoked by the agent via bash): + python scanner.py first-run # First install: list installed skills and ask to scan + python scanner.py scan # Scan a single skill (.zip or directory) + python scanner.py scan-all # Scan all installed skills + python scanner.py upgrade # Auto-upgrade +""" + +import sys, os, json, time, zipfile, hashlib, shutil, tempfile, uuid, platform, base64 +import urllib.request, urllib.error, urllib.parse +from pathlib import Path +from datetime import datetime, timezone + +# ───────────────────────────────────────────────────────────────────────────── +# Configuration +# ───────────────────────────────────────────────────────────────────────────── + +SCANNER_VERSION = "1.1.5" + +BASE_URL = "https://skillscan.tokauth.com" +API_SEARCH = f"{BASE_URL}/oapi/v1/skill-scan/search" +API_UPLOAD = f"{BASE_URL}/oapi/v1/skill-scan/upload" +API_RESULT = f"{BASE_URL}/oapi/v1/skill-scan/result" +UPDATE_URL = os.environ.get("SKILL_SCANNER_UPDATE_URL", + f"{BASE_URL}/downloads/SkillScan/manifest") + +POLL_INTERVAL = 20 # Poll interval (seconds) +POLL_TIMEOUT = 180 # Max wait time (seconds) + +# First-run marker file (in the same directory as scanner.py) +STATE_FILE = Path(__file__).parent / ".first_run_done" + +# Auto-update check marker file and interval (7 days) +LAST_UPDATE_CHECK_FILE = Path(__file__).parent / ".last_update_check" +AUTO_UPDATE_INTERVAL = 1 * 24 * 3600 # 1 day (seconds) + +# Client info file (generated on first run, reused afterwards) +CLIENT_INFO_FILE = Path(__file__).parent / ".client_info" + +# Files and directories to skip during scanning, hashing, and packing +SKIP_FILES = {".first_run_done", ".last_update_check", ".client_info", "cloud_report.json", ".DS_Store"} +SKIP_DIRS = {".git", "__pycache__", ".venv", "node_modules", ".idea", ".vscode", ".clawhub"} + +# Resolve the root directory of SkillScan itself (parent of scripts/) +SELF_ROOT = Path(__file__).parent.parent.resolve() + + +# Skill installation paths (cross-platform) +def skill_install_paths(): + # type: () -> list + """Auto-enumerate OpenClaw and local skill paths across platforms.""" + home = Path.home() + oc_dir = home / ".openclaw" + candidates = [ + # OpenClaw standard paths + oc_dir / "skills", + oc_dir / "workspace/skills", + # Shared agent skill paths + home / ".agents/skills", + home / ".config/agents/skills", + # Agent-specific global paths + home / ".gemini/antigravity/skills", + home / ".gemini/skills", + home / ".augment/skills", + home / ".claude/skills", + home / ".codex/skills", + home / ".commandcode/skills", + home / ".continue/skills", + home / ".snowflake/cortex/skills", + home / ".config/crush/skills", + home / ".cursor/skills", + home / ".deepagents/agent/skills", + home / ".factory/skills", + home / ".firebender/skills", + home / ".copilot/skills", + home / ".config/goose/skills", + home / ".junie/skills", + home / ".iflow/skills", + home / ".kilocode/skills", + home / ".kiro/skills", + home / ".kode/skills", + home / ".mcpjam/skills", + home / ".vibe/skills", + home / ".mux/skills", + home / ".config/opencode/skills", + home / ".openhands/skills", + home / ".pi/agent/skills", + home / ".qoder/skills", + home / ".qwen/skills", + home / ".roo/skills", + home / ".trae/skills", + home / ".trae-cn/skills", + home / ".codeium/windsurf/skills", + home / ".zencoder/skills", + home / ".neovate/skills", + home / ".pochi/skills", + home / ".adal/skills", + home / ".npm-global/lib/node_modules/openclaw/skills", + # Container default paths + Path("/mnt/skills/public"), + Path("/mnt/skills/private"), + Path("/mnt/skills/user"), + # User dev/download paths + home / "Downloads/skills", + ] + + # Windows-specific paths + if os.name == "nt": + appdata = os.environ.get("APPDATA") + if appdata: + candidates.append(Path(appdata) / "OpenClaw/skills") + candidates.append(Path(appdata) / "Programs/LobsterAI/resources/SKILLs") + + # Dynamically scan extensions: .openclaw/extensions/{xxxx}/skills + if oc_dir.exists(): + ext_root = oc_dir / "extensions" + if ext_root.exists(): + for sub in ext_root.iterdir(): + if sub.is_dir(): + s_dir = sub / "skills" + if s_dir.exists(): + candidates.append(s_dir) + + # Include script run path and workspace + candidates.append(Path.cwd() / "skills") + candidates.append(Path(__file__).parent.parent / "skills") + + # Deduplicate and filter non-existent paths + seen = set() + result = [] + for p in candidates: + try: + abs_p = p.resolve() + if abs_p.exists() and abs_p not in seen: + result.append(p) + seen.add(abs_p) + except Exception: + continue + return result + +RISK_EMOJI = {"SAFE":"✅","LOW":"⚠️ ","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️ "} + + +# ───────────────────────────────────────────────────────────────────────────── +# Client Info (X-Client-Info) +# ───────────────────────────────────────────────────────────────────────────── + +def _get_mac_address(): + """Try to get the MAC address; return empty string on failure.""" + try: + import uuid as _uuid + mac_int = _uuid.getnode() + # getnode() returns a random value (bit 8 set) when it can't get the real MAC + if (mac_int >> 40) & 1: + return "" + mac_str = ":".join(("%012X" % mac_int)[i:i+2] for i in range(0, 12, 2)) + return mac_str + except Exception: + return "" + + +def _build_client_info(): + """Build client info dict and persist to file; reuse on subsequent runs.""" + # If a record file already exists, read it + if CLIENT_INFO_FILE.exists(): + try: + data = json.loads(CLIENT_INFO_FILE.read_text(encoding="utf-8")) + if data.get("client_id"): + return data + except Exception: + pass + + # First run: generate new client info + info = { + "client_id": str(uuid.uuid4()), + "os": platform.system() or "", + "platform": platform.machine() or "", + "os_version": platform.release() or "", + "client": "SkillScanner/%s" % SCANNER_VERSION, + } + + mac = _get_mac_address() + if mac: + info["mac"] = mac + + # Python version as extra + info["extra"] = { + "python": platform.python_version(), + } + + # Persist + try: + CLIENT_INFO_FILE.write_text( + json.dumps(info, ensure_ascii=False, indent=2), + encoding="utf-8" + ) + except Exception: + pass + + return info + + +def _get_client_info_header(): + """Return Base64-encoded X-Client-Info header value; empty string on failure.""" + try: + info = _build_client_info() + json_str = json.dumps(info, ensure_ascii=False) + encoded = base64.b64encode(json_str.encode("utf-8")).decode("ascii") + return encoded + except Exception: + return "" + +# ───────────────────────────────────────────────────────────────────────────── +# Output Helpers +# ───────────────────────────────────────────────────────────────────────────── + +def banner(title: str): + w = 58 + print(f"\n{'═'*w}") + print(f" {title}") + print(f"{'═'*w}") + +def divider(title: str = ""): + if title: + print(f"\n ── {title} {'─'*(48-len(title))}") + else: + print(f" {'─'*52}") + +def log(msg: str): + print(f" {msg}", flush=True) + +def ask(prompt: str) -> str: + """Read user input (compatible with non-interactive environments).""" + try: + return input(f"\n {prompt} ").strip() + except (EOFError, KeyboardInterrupt): + return "" + +# ───────────────────────────────────────────────────────────────────────────── +# HTTP Helpers +# ───────────────────────────────────────────────────────────────────────────── + +def http_get(url: str) -> dict: + req = urllib.request.Request(url) + with urllib.request.urlopen(req, timeout=30) as r: + return json.loads(r.read().decode("utf-8", errors="replace")) + +def http_post(url: str, payload: dict) -> dict: + headers = {"Content-Type": "application/json"} + data = json.dumps(payload, ensure_ascii=False).encode("utf-8") + req = urllib.request.Request(url, data=data, headers=headers, method="POST") + with urllib.request.urlopen(req, timeout=60) as r: + return json.loads(r.read().decode("utf-8", errors="replace")) + + +# ───────────────────────────────────────────────────────────────────────────── +# Skill Utilities +# ───────────────────────────────────────────────────────────────────────────── + +def skill_name_from_dir(skill_dir: Path) -> str: + md = skill_dir / "SKILL.md" + if md.exists(): + for line in md.read_text(encoding="utf-8", errors="replace").splitlines(): + s = line.strip() + if s.startswith("name:"): + return s.split(":", 1)[1].strip().strip("\"'") + return skill_dir.name + +def sha256_of(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + +def calculate_dir_sha256(directory: Path) -> str: + """Calculate SHA256 hash of a skill directory (based on all file contents + relative paths). + Excludes _meta.json and files/dirs in SKIP_FILES/SKIP_DIRS.""" + file_hashes = [] + for file_path in sorted(directory.rglob('*')): + if not file_path.is_file(): + continue + if file_path.name == '_meta.json': + continue + rel = file_path.relative_to(directory) + if any(part in SKIP_DIRS for part in rel.parts): + continue + if file_path.name in SKIP_FILES: + continue + rel_path = str(rel) + file_hash = hashlib.sha256() + file_hash.update(rel_path.encode('utf-8')) + file_hash.update(b'\x00') + with open(file_path, 'rb') as f: + for chunk in iter(lambda: f.read(8192), b''): + file_hash.update(chunk) + file_hashes.append(file_hash.hexdigest()) + file_hashes.sort() + final_hash = hashlib.sha256() + for h in file_hashes: + final_hash.update(h.encode('utf-8')) + final_hash.update(b'\x00') + return final_hash.hexdigest() + +def collect_files(skill_dir: Path) -> dict: + """Collect files for scanning, skipping redundant or sensitive directories.""" + exts = {".md",".py",".js",".ts",".sh",".yaml",".yml",".json",".txt"} + out = {} + for p in sorted(skill_dir.rglob("*")): + if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts): + continue + if p.is_file() and p.name not in SKIP_FILES: + if p.suffix.lower() in exts or p.name == "SKILL.md": + try: + out[str(p.relative_to(skill_dir))] = \ + p.read_text(encoding="utf-8", errors="replace") + except Exception: + pass + return out + +def pack_zip(skill_dir: Path) -> bytes: + """Pack a skill directory into a zip byte stream, excluding redundant directories.""" + import io + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf: + for p in sorted(skill_dir.rglob("*")): + if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts): + continue + if p.is_file() and p.name not in SKIP_FILES: + zf.write(p, p.relative_to(skill_dir)) + return buf.getvalue() + +def unpack_zip(zip_path: Path) -> Path: + """Extract a .zip to a system temp directory. Returns the extraction path. Prevents zip-slip.""" + tmp = Path(tempfile.mkdtemp(prefix="skillscan-")) + log(f"📦 Extracting {zip_path.name} → {tmp}") + with zipfile.ZipFile(zip_path, "r") as zf: + for member in zf.namelist(): + dest = (tmp / member).resolve() + if not str(dest).startswith(str(tmp.resolve())): + raise ValueError(f"zip-slip path rejected: {member}") + zf.extractall(tmp) + return tmp + +def find_installed_skills(): + # type: () -> list + """Find all installed skill directories (first-level subdirectories containing SKILL.md). + Excludes SkillScan itself.""" + found = set() + for base in skill_install_paths(): + if not base.exists(): + continue + for md in base.rglob("SKILL.md"): + skill_path = md.parent + try: + rel = skill_path.relative_to(base) + if len(rel.parts) == 1: + resolved = skill_path.resolve() + # Skip self + if resolved == SELF_ROOT: + continue + found.add(resolved) + except ValueError: + pass + return sorted(found) + + +# ───────────────────────────────────────────────────────────────────────────── +# Scan Core (3 steps) +# ───────────────────────────────────────────────────────────────────────────── + +def _extract_result(resp, sha256): + """Internal: extract core data from API response, handling SHA256 wrapping/nested result.""" + # 1. Handle API response keyed by SHA256 (e.g. { "sha256": { "status": "success", "data": {...} } }) + if sha256 and sha256 in resp: + resp = resp[sha256] + + # 2. Extract data body (data or result) + data = resp.get("data") or resp.get("result") or resp + + # 3. Handle nested result inside data + if isinstance(data, dict) and "result" in data: + inner = data["result"] + if isinstance(inner, dict): + # Merge sibling metadata (analysis_level/reason etc.) into result + for k, v in data.items(): + if k != "result" and k not in inner: + inner[k] = v + return inner + + return data if isinstance(data, dict) and (data.get("verdict") or data.get("is_safe") is not None or data.get("analysis_level")) else None + + +def cloud_search(dir_sha256): + """Step 1: Query scan cache by dir_sha256. Returns result dict or None.""" + extra_headers = {} + ci = _get_client_info_header() + if ci: + extra_headers["X-Client-Info"] = ci + + url = "%s?%s" % (API_SEARCH, urllib.parse.urlencode({"dir_sha256": dir_sha256})) + try: + headers = {} + headers.update(extra_headers) + req = urllib.request.Request(url, headers=headers) + with urllib.request.urlopen(req, timeout=30) as r: + resp = json.loads(r.read().decode("utf-8", errors="replace")) + res = _extract_result(resp, dir_sha256) + if res: + log(" ✅ Cache hit (dir_sha256 %s…)" % dir_sha256[:16]) + return res + except urllib.error.HTTPError as e: + if e.code == 404: + return None + raise RuntimeError("Search API error HTTP %d" % e.code) + except urllib.error.URLError as e: + raise RuntimeError("Cannot connect to server: %s" % e) + return None + + +def cloud_upload(skill_dir, name, dir_hash): + """Step 2: Upload skill (multipart/form-data), returns task_no.""" + # Pack the entire directory for full code context + zip_data = pack_zip(skill_dir) + filename = "%s.zip" % name + + # Build multipart/form-data boundary + boundary = "----WebKitFormBoundary%s" % uuid.uuid4().hex + + # Manually construct multipart byte stream (no requests library needed) + parts = [] + parts.append(("--%s" % boundary).encode()) + parts.append(('Content-Disposition: form-data; name="file"; filename="%s"' % filename).encode()) + parts.append(b"Content-Type: application/zip") + parts.append(b"") + parts.append(zip_data) + parts.append(("--%s--" % boundary).encode()) + parts.append(b"") # trailing newline + + body = b"\r\n".join(parts) + + headers = { + "Content-Type": "multipart/form-data; boundary=%s" % boundary, + "Content-Length": str(len(body)), + "Accept": "application/json" + } + + # Add X-Client-Info header + ci = _get_client_info_header() + if ci: + headers["X-Client-Info"] = ci + + log(" 📤 Uploading: %s (%.1f KB)..." % (filename, len(zip_data) / 1024.0)) + req = urllib.request.Request(API_UPLOAD, data=body, headers=headers, method="POST") + try: + with urllib.request.urlopen(req, timeout=60) as r: + resp = json.loads(r.read().decode("utf-8", errors="replace")) + except urllib.error.HTTPError as e: + err_body = e.read().decode(errors="replace") + raise RuntimeError("Upload failed HTTP %d: %s" % (e.code, err_body)) + + task_no = (resp.get("data") or {}).get("task_no") or resp.get("task_no") or resp.get("taskNo") or resp.get("task_id") or "" + if not task_no: + raise RuntimeError("Upload succeeded but no valid task_no in response: %s" % resp) + + log(" ✅ Upload complete, task_no: %s" % task_no) + return str(task_no) + + +def cloud_poll(task_no: str) -> dict: + """Step 3: Poll until complete or timeout. Queries every 20s. + status: 0=pending, 1=scanning, 2=completed, 3=failed, 4=cancelled + """ + url = f"{API_RESULT}?{urllib.parse.urlencode({'task_no': task_no})}" + deadline = time.time() + POLL_TIMEOUT + attempt = 0 + while time.time() < deadline: + attempt += 1 + elapsed = int(time.time() - (deadline - POLL_TIMEOUT)) + try: + resp = http_get(url) + data = resp.get("data") or resp + status = data.get("status") + + if status == 2: # completed + print() + log(f" ✅ Scan complete (attempt {attempt}, {elapsed}s elapsed)") + return _extract_result(resp, "") or resp + elif status == 3: # failed + print() + err_msg = data.get("error_message") or resp.get("message", "unknown error") + raise RuntimeError(f"Analysis failed: {err_msg}") + elif status == 4: # cancelled + print() + raise RuntimeError("Scan task was cancelled") + else: + # 0=pending, 1=scanning -> keep waiting + status_text = data.get("status_text", "processing") + print(f" ⏳ [{status_text}] attempt {attempt}, {elapsed}s / {POLL_TIMEOUT}s elapsed", + end="\r", flush=True) + time.sleep(POLL_INTERVAL) + except (RuntimeError, ValueError): + raise + except Exception as e: + raise RuntimeError(f"Poll error: {e}") + print() + raise RuntimeError(f"Timeout ({POLL_TIMEOUT}s), task_no={task_no}, please retry later") + + +def cloud_check(skill_dir: Path) -> dict: + """Run full security scan on a skill directory, return normalized result.""" + md = skill_dir / "SKILL.md" + if not md.exists(): + raise FileNotFoundError(f"SKILL.md not found: {skill_dir}") + + name = skill_name_from_dir(skill_dir) + dir_hash = calculate_dir_sha256(skill_dir) + log(f"🔍 Scanning: {name}") + log(f" dir_sha256: {dir_hash}") + + log(f"🔎 [1/3] Checking scan cache...") + raw = cloud_search(dir_hash) + + if raw is None: + log(f" ℹ️ No cache record, submitting new scan task") + log(f"📤 [2/3] Uploading skill for analysis...") + task_no = cloud_upload(skill_dir, name, dir_hash) + log(f"⏳ [3/3] Waiting for analysis (polling every {POLL_INTERVAL}s, max {POLL_TIMEOUT}s)...") + raw = cloud_poll(task_no) + else: + log(f" ⏭️ Skipping upload, using cached result") + + return _normalize(raw, name, dir_hash) + + +def _normalize(raw: dict, name: str, dir_hash: str) -> dict: + """Normalize scan result: + 1. Extract is_safe (bool) and max_severity (str). + 2. Map API-specific fields (analysis_reason, analysis_suggestion) to standard fields. + """ + is_safe = raw.get("is_safe") + + # Severity field priority: max_severity > analysis_level > verdict > level + v_raw = (raw.get("max_severity") or raw.get("analysis_level") or + raw.get("verdict") or raw.get("risk_level") or + raw.get("level") or "UNKNOWN").upper() + + # Combined verdict logic + if is_safe is True and v_raw in ("UNKNOWN", "SAFE"): + verdict = "SAFE" + elif is_safe is False and v_raw in ("UNKNOWN", "SAFE"): + verdict = "CRITICAL" # Explicitly marked unsafe -> critical + else: + verdict = v_raw + + return { + "skill_name": name, + "dir_sha256": dir_hash, + "verdict": verdict, + "confidence": raw.get("confidence") or raw.get("score"), + "threat_labels": raw.get("threat_labels") or raw.get("tags") or [], + "summary": raw.get("analysis_reason") or raw.get("summary") or raw.get("description") or "", + "findings": raw.get("findings") or raw.get("issues") or [], + "recommendation":raw.get("analysis_suggestion") or raw.get("recommendation") or raw.get("action") or "", + } + + +# ───────────────────────────────────────────────────────────────────────────── +# Result Display +# ───────────────────────────────────────────────────────────────────────────── + +def print_result(r: dict): + verdict = r.get("verdict","UNKNOWN") + emoji = RISK_EMOJI.get(verdict,"❓") + conf = r.get("confidence") + labels = r.get("threat_labels",[]) + summary = r.get("summary","") + findings = r.get("findings",[]) + rec = r.get("recommendation","") + conf_str = f" confidence {float(conf):.0%}" if conf is not None else "" + + divider() + log(f"{emoji} Result: {verdict}{conf_str}") + if summary: + log(f"📋 {summary}") + if labels: + log(f"🏷️ Threat labels: {', '.join(labels)}") + if findings: + SEV = {"LOW":"🔵","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️"} + log(f"🔍 Findings ({len(findings)} items):") + for f in findings: + sev = str(f.get("severity","")).upper() + desc = f.get("description") or f.get("detail") or str(f) + rid = f.get("id") or "" + tag = f"[{rid}] " if rid else "" + log(f" {SEV.get(sev,'⚪')} {tag}{desc}") + if rec: + log(f"💡 Recommendation: {rec}") + divider() + +# ───────────────────────────────────────────────────────────────────────────── +# Prompt: malicious detected -> ask whether to delete +# ───────────────────────────────────────────────────────────────────────────── + +def prompt_delete(skill_path: Path, result: dict) -> bool: + """When result is HIGH/CRITICAL, ask user whether to delete the skill. + skill_path is the original install path (not temp dir). + Returns True if deleted. + """ + verdict = result.get("verdict","") + if verdict not in ("HIGH","CRITICAL"): + return False + + if not skill_path or not skill_path.exists(): + return False + + emoji = RISK_EMOJI.get(verdict,"🔴") + log(f"\n{emoji} This skill is marked as [{verdict}] high risk by security scan.") + log(f" Path: {skill_path}") + + answer = ask("Delete this skill now? [y/n]") + if answer in ("y","Y","yes","Yes"): + try: + if skill_path.is_dir(): + shutil.rmtree(skill_path) + else: + skill_path.unlink() + log(f"✅ Deleted: {skill_path}") + return True + except Exception as e: + log(f"❌ Delete failed: {e} (please delete manually)") + return False + else: + log(f"⚠️ Skipped deletion. Use this skill with caution.") + return False + + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: first-run (first install) +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_first_run(): + """First install: list installed skills, ask user to scan, show results.""" + if STATE_FILE.exists(): + log("ℹ️ First-run scan already completed. Use scan-all to rescan.") + return + + banner("🛡️ SkillScan First-Run Check") + log("Welcome to SkillScan!") + log("Searching for installed skills...\n") + + skills = find_installed_skills() + if not skills: + log("✅ No installed skills found, nothing to scan.") + STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8") + return + + # Print installed skill list + log(f"Found {len(skills)} installed skill(s):\n") + for i, s in enumerate(skills, 1): + log(f" {i:2d}. {s.name}") + + answer = ask("Run security scan on all listed skills? [y/n]") + if answer not in ("y","Y","yes","Yes"): + log("Skipped. You can run scan-all anytime to rescan.") + STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8") + return + + # Scan one by one + results = [] + for idx, skill_path in enumerate(skills, 1): + divider(f"[{idx}/{len(skills)}] {skill_path.name}") + tmp = None + try: + # Copy to temp dir (source may be read-only) + tmp = Path(tempfile.mkdtemp(prefix="skillscan-")) + scan_dir = tmp / skill_path.name + shutil.copytree(skill_path, scan_dir) + + r = cloud_check(scan_dir) + print_result(r) + + # High risk -> ask to delete (targeting original install path) + prompt_delete(skill_path, r) + results.append(r) + + except RuntimeError as e: + log(f"❌ Scan failed: {e}") + results.append({"skill_name": skill_path.name, + "verdict": "ERROR", "threat_labels": [], + "summary": str(e)[:100]}) + finally: + if tmp: + shutil.rmtree(tmp, ignore_errors=True) + + _print_summary(results) + STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8") + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: scan (single skill) +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_scan(path_str: str): + skill_path = Path(path_str) + if not skill_path.exists(): + log(f"❌ Path not found: {skill_path}") + sys.exit(1) + + banner(f"Skill Security Scan v{SCANNER_VERSION}") + + tmp = None + original_path = skill_path if skill_path.is_dir() else None + try: + if skill_path.is_file(): + if skill_path.suffix.lower() not in (".zip",): + log(f"❌ Unsupported format: {skill_path.suffix} (use .zip)") + sys.exit(1) + tmp = unpack_zip(skill_path) + scan_dir = tmp + else: + scan_dir = skill_path + + result = cloud_check(scan_dir) + print_result(result) + + # High risk -> ask to delete + if original_path: + prompt_delete(original_path, result) + elif skill_path.is_file() and result.get("verdict") in ("HIGH","CRITICAL"): + # Zip file: ask to delete source file + prompt_delete(skill_path, result) + + v = result.get("verdict","UNKNOWN") + sys.exit(0 if v in ("SAFE","LOW") else 1 if v=="MEDIUM" else 2) + + except RuntimeError as e: + log(f"\n❌ Scan failed: {e}") + sys.exit(3) + finally: + if tmp: + shutil.rmtree(tmp, ignore_errors=True) + + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: scan-all +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_scan_all(): + banner(f"Full Skill Security Scan v{SCANNER_VERSION}") + + skills = find_installed_skills() + if not skills: + log("ℹ️ No installed skills detected.") + return + + log(f"Found {len(skills)} installed skill(s):\n") + for i, s in enumerate(skills, 1): + log(f" {i:2d}. {s.name:<30} {s}") + + answer = ask("Start security scan? [y/n]") + if answer not in ("y","Y","yes","Yes"): + log("Cancelled.") + return + + results = [] + for idx, skill_path in enumerate(skills, 1): + divider(f"[{idx}/{len(skills)}] {skill_path.name}") + tmp = None + try: + tmp = Path(tempfile.mkdtemp(prefix="skillscan-")) + scan_dir = tmp / skill_path.name + shutil.copytree(skill_path, scan_dir) + + r = cloud_check(scan_dir) + v = r.get("verdict","UNKNOWN") + log(f"{RISK_EMOJI.get(v,'❓')} Scan complete: {v}") + if r.get("threat_labels"): + log(f" Threat labels: {', '.join(r['threat_labels'])}") + + # High risk: ask to delete + prompt_delete(skill_path, r) + results.append(r) + + except RuntimeError as e: + log(f"❌ Scan failed: {e}") + results.append({"skill_name": skill_path.name, "verdict":"ERROR", + "threat_labels":[], "summary":str(e)[:100]}) + finally: + if tmp: + shutil.rmtree(tmp, ignore_errors=True) + + _print_summary(results) + +# ───────────────────────────────────────────────────────────────────────────── +# Summary Table +# ───────────────────────────────────────────────────────────────────────────── + +def _print_summary(results): + banner("📊 Scan Summary") + print(f" {'Skill Name':<28} {'Result':<12} {'Threat Labels'}") + divider() + for r in results: + v = r.get("verdict","?") + name = r.get("skill_name","?")[:27] + labels = ", ".join(r.get("threat_labels",[]))[:20] or "-" + print(f" {name:<28} {RISK_EMOJI.get(v,'❓')}{v:<10} {labels}") + + safes = [r for r in results if r["verdict"] in {"SAFE","LOW"}] + mediums = [r for r in results if r["verdict"] == "MEDIUM"] + highs = [r for r in results if r["verdict"] in {"HIGH","CRITICAL"}] + errors = [r for r in results if r["verdict"] in {"ERROR","UNKNOWN"}] + + print() + log(f"Total {len(results)} | ✅ Safe {len(safes)} " + f"🟡 Suspicious {len(mediums)} 🔴 Dangerous {len(highs)} ❓ Error {len(errors)}") + if highs: + log(f"\n⚠️ High-risk skills: {', '.join(r['skill_name'] for r in highs)}") + elif not mediums and not errors: + log("\n🎉 All skills passed security scan.") + + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: upgrade +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_upgrade(): + banner("SkillScan Auto-Upgrade") + log(f"Current version: {SCANNER_VERSION}") + log(f"Update source: {UPDATE_URL}") + try: + manifest = http_get(UPDATE_URL) + except Exception as e: + log(f"❌ Failed to fetch update manifest: {e}") + return + + latest = manifest.get("version", SCANNER_VERSION) + if (tuple(int(x) for x in latest.split(".")) <= + tuple(int(x) for x in SCANNER_VERSION.split("."))): + log(f"✅ Already up to date ({SCANNER_VERSION})") + return + + log(f"New version found: {SCANNER_VERSION} → {latest}") + log(f"Changelog: {manifest.get('changelog','(none)')}") + + download_url = manifest.get("download_url", "") + if not download_url: + log("⚠️ No download URL in manifest, skipping upgrade") + return + + # Download new version zip + log(f"📥 Downloading: {download_url}") + try: + req = urllib.request.Request(download_url) + with urllib.request.urlopen(req, timeout=60) as r: + zip_data = r.read() + except Exception as e: + log(f"❌ Download failed: {e}") + return + + # SHA256 verification + expected_sha = manifest.get("sha256", "") + if expected_sha: + actual_sha = hashlib.sha256(zip_data).hexdigest() + if actual_sha != expected_sha: + log(f"❌ SHA256 mismatch, upgrade aborted (expected {expected_sha[:16]}…, got {actual_sha[:16]}…)") + return + log(f" ✅ SHA256 verified") + + # Backup current skill directory + skill_root = Path(__file__).parent.parent + backup_dir = skill_root.parent / f"SkillScan-backup-{SCANNER_VERSION}" + if backup_dir.exists(): + shutil.rmtree(backup_dir) + shutil.copytree(skill_root, backup_dir) + log(f"📦 Backed up to: {backup_dir}") + + # Extract and replace files + tmp = Path(tempfile.mkdtemp(prefix="skillupgrade-")) + try: + zip_path = tmp / "update.zip" + zip_path.write_bytes(zip_data) + with zipfile.ZipFile(zip_path, "r") as zf: + # Security check: prevent zip-slip + for member in zf.namelist(): + dest = (tmp / "extracted" / member).resolve() + if not str(dest).startswith(str((tmp / "extracted").resolve())): + raise ValueError(f"zip-slip path rejected: {member}") + zf.extractall(tmp / "extracted") + + # Overwrite skill directory with new files + extracted = tmp / "extracted" + for item in extracted.rglob("*"): + if not item.is_file(): + continue + rel = item.relative_to(extracted) + target = skill_root / rel + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(item, target) + log(f" ✅ Updated: {rel}") + + log(f"🎉 Upgraded to v{latest}") + except Exception as e: + log(f"❌ Upgrade failed: {e}") + log(f" You can restore from backup: {backup_dir}") + finally: + shutil.rmtree(tmp, ignore_errors=True) + +# ───────────────────────────────────────────────────────────────────────────── +# Entry Point +# ───────────────────────────────────────────────────────────────────────────── + +def auto_upgrade_if_needed(): + """Auto-check for updates every 7 days, runs silently.""" + try: + if LAST_UPDATE_CHECK_FILE.exists(): + last_check = float(LAST_UPDATE_CHECK_FILE.read_text(encoding="utf-8").strip()) + if time.time() - last_check < AUTO_UPDATE_INTERVAL: + return # Not time to check yet + log("🔄 Checking for updates...") + manifest = http_get(UPDATE_URL) + latest = manifest.get("version", SCANNER_VERSION) + if (tuple(int(x) for x in latest.split(".")) <= + tuple(int(x) for x in SCANNER_VERSION.split("."))): + log(f" ✅ Already up to date ({SCANNER_VERSION})") + else: + log(f" New version found: {SCANNER_VERSION} → {latest}, auto-updating...") + cmd_upgrade() + LAST_UPDATE_CHECK_FILE.write_text(str(time.time()), encoding="utf-8") + except Exception as e: + log(f" ⚠️ Auto-update check failed: {e} (normal operation unaffected)") + + +def main(): + if len(sys.argv) < 2: + print(__doc__) + sys.exit(0) + + # Check for auto-update on every run (once every 7 days) + auto_upgrade_if_needed() + + cmd = sys.argv[1] + if cmd == "first-run": + cmd_first_run() + elif cmd == "scan": + if len(sys.argv) < 3: + log("Usage: scanner.py scan ") + sys.exit(1) + cmd_scan(sys.argv[2]) + elif cmd == "scan-all": + cmd_scan_all() + elif cmd == "upgrade": + cmd_upgrade() + else: + log(f"Unknown command: {cmd}") + log("Available commands: first-run / scan / scan-all / upgrade") + sys.exit(1) + +if __name__ == "__main__": + main() diff --git a/runtime/skills/_workspace/word-reader/DEVELOPMENT.md b/runtime/skills/_workspace/word-reader/DEVELOPMENT.md new file mode 100644 index 00000000..796643f0 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/DEVELOPMENT.md @@ -0,0 +1,208 @@ +# Word Reader 技能开发完成 + +## 🎯 技能概述 + +成功创建了一个功能完整的 Word 文档读取技能,支持读取 .docx 和 .doc 格式的 Word 文档,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。 + +## 📁 技能结构 + +``` +word-reader/ +├── SKILL.md # 技能定义文件 +├── README.md # 使用说明 +├── skill.json # 技能配置 +├── demo.sh # 演示脚本 +├── install.sh # 安装脚本 +├── test.md # 测试文档 +└── scripts/ + └── read_word.py # 核心脚本 +``` + +## ✨ 主要功能 + +### 1. 文档解析能力 +- ✅ **文本提取** - 提取文档中的所有段落文本 +- ✅ **表格解析** - 解析表格数据并转换为结构化格式 +- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等) +- ✅ **图片信息** - 获取文档中图片的基本信息 + +### 2. 格式支持 +- ✅ **.docx** - Office 2007+ 格式(主要支持) +- ✅ **.doc** - 旧版 Word 格式(需要 antiword) + +### 3. 输出格式 +- ✅ **JSON** - 结构化数据,适合程序处理 +- ✅ **Text** - 纯文本格式,简单易读 +- ✅ **Markdown** - 格式化输出,保留文档结构 + +### 4. 高级功能 +- ✅ **批量处理** - 支持处理整个目录的文档 +- ✅ **选择性提取** - 可只提取特定内容类型 +- ✅ **文件输出** - 支持保存结果到文件 +- ✅ **编码支持** - 支持多种文本编码 + +## 🚀 使用示例 + +### 基本用法 +```bash +# 读取文档 +python3 scripts/read_word.py 文档.docx + +# JSON 格式输出 +python3 scripts/read_word.py 文档.docx --format json + +# Markdown 格式输出 +python3 scripts/read_word.py 文档.docx --format markdown + +# 只提取文本 +python3 scripts/read_word.py 文档.docx --extract text +``` + +### 批量处理 +```bash +# 批量处理目录下所有文档 +python3 scripts/read_word.py ./文档目录 --batch + +# 批量处理并保存结果 +python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json +``` + +## 🔧 安装和配置 + +### 自动安装 +```bash +cd word-reader/ +./install.sh +``` + +### 手动安装 +```bash +# 安装 Python 依赖 +pip3 install python-docx + +# 安装系统依赖(可选) +sudo apt-get install antiword # Ubuntu/Debian +brew install antiword # macOS + +# 设置执行权限 +chmod +x scripts/read_word.py +``` + +## 📊 输出示例 + +### JSON 格式 +```json +{ + "metadata": { + "filename": "文档.docx", + "title": "文档标题", + "author": "作者", + "created": "2024-01-01T10:00:00", + "modified": "2024-01-01T12:00:00" + }, + "format": "docx", + "text": "文档内容...", + "tables": [...], + "images": [...] +} +``` + +### Markdown 格式 +```markdown +# 文档.docx + +**标题**:文档标题 +**作者**:作者 +**创建时间**:2024-01-01T10:00:00 + +## 正文内容 + +文档内容... + +## 表格内容 + +| 表头1 | 表头2 | +|-------|-------| +| 数据1 | 数据2 | +``` + +## 🎨 技能特点 + +### 1. 智能错误处理 +- 友好的错误提示 +- 自动检测文档格式 +- 优雅的异常处理 + +### 2. 性能优化 +- 流式处理大文件 +- 内存使用优化 +- 进度显示(批量模式) + +### 3. 用户友好 +- 详细的帮助信息 +- 多种使用方式 +- 完整的文档说明 + +### 4. 可扩展性 +- 模块化设计 +- 易于添加新功能 +- 支持自定义输出格式 + +## 🎯 应用场景 + +### 1. 文档内容分析 +- 快速查看 Word 文档内容 +- 提取特定信息 +- 文档摘要生成 + +### 2. 批量处理 +- 处理大量文档 +- 文档格式转换 +- 内容索引创建 + +### 3. 自动化工作流 +- 集到文档处理系统 +- 自动化文档分析 +- 内容管理系统集成 + +## 📝 开发总结 + +### 实现的功能 +- 完整的 Word 文档解析框架 +- 支持多种输出格式 +- 批量处理能力 +- 错误处理和用户友好性 + +### 技术亮点 +- 模块化设计,易于维护 +- 优雅的错误处理机制 +- 支持多种文件格式 +- 灵活的输出选项 + +### 改进空间 +- 可以添加 PDF 支持 +- 可以增加图片提取功能 +- 可以优化大文件处理性能 +- 可以添加更多文档元素支持 + +## 🚀 发布到 ClawHub + +要发布此技能到 ClawHub,可以运行: + +```bash +# 安装 ClawHub CLI +npm i -g clawhub + +# 登录 +clawhub login + +# 发布技能 +clawhub publish ./word-reader \ + --slug word-reader \ + --name "Word Reader" \ + --version 1.0.0 \ + --changelog "Initial release with .docx and .doc support" \ + --tags document,word,office,text-extraction +``` + +这个技能现在已经准备好使用了!它可以帮助用户轻松读取和处理 Word 文档,支持多种格式和输出选项。 \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/PUBLISHING.md b/runtime/skills/_workspace/word-reader/PUBLISHING.md new file mode 100644 index 00000000..35b32821 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/PUBLISHING.md @@ -0,0 +1,177 @@ +# Word Reader 技能发布指南 + +## 🚀 发布到 ClawHub + +### 1. 准备工作 + +#### 确保技能完整 +- [ ] SKILL.md 文件完整且格式正确 +- [ ] 脚本功能正常 +- [ ] 安装脚本工作正常 +- [ ] README.md 说明清晰 +- [ ] 所有依赖已在 SKILL.md 中声明 + +#### 环境准备 +```bash +# 安装 ClawHub CLI +npm install -g clawhub +# 或 +pnpm add -g clawhub +``` + +#### 登录 ClawHub +```bash +# 登录(会打开浏览器进行 OAuth 认证) +clawhub login + +# 验证登录状态 +clawhub whoami +``` + +> **注意**:GitHub 账号需要注册满一周才能发布技能 + +### 2. 发布流程 + +#### 检查技能 +```bash +# 验证技能结构 +clawhub validate ./word-reader +``` + +#### 发布技能 +```bash +clawhub publish ./word-reader \ + --slug word-reader \ + --name "Word Reader" \ + --version 1.0.0 \ + --changelog "支持 .docx 和 .doc 格式的 Word 文档读取,提取文本、表格、元数据等" \ + --tags document,word,office,text-extraction,reader,parsing \ + --license MIT \ + --visibility public +``` + +#### 参数说明 +- `--slug`: URL 友好的唯一标识符 +- `--name`: 技能显示名称 +- `--version`: 遵循语义化版本控制 +- `--changelog`: 版本变更说明 +- `--tags`: 搜索标签(逗号分隔) +- `--license`: 许可证类型 +- `--visibility`: public/private + +### 3. 发布后操作 + +#### 验证发布 +```bash +# 查看已发布的技能 +clawhub search word-reader + +# 安装测试 +clawhub install word-reader-test +``` + +#### 分享技能 +- 技能将在 `https://clawhub.com/skills/word-reader` 可见 +- 其他用户可通过 `clawhub install word-reader` 安装 + +### 4. 版本管理 + +#### 更新技能 +```bash +# 修改技能后更新版本号 +clawhub publish ./word-reader --version 1.0.1 --changelog "修复了某些文档格式的解析问题" +``` + +#### 批量操作 +```bash +# 同步所有技能 +clawhub sync --all + +# 发布并标记 +clawhub publish ./word-reader --tags latest,stable +``` + +### 5. 自动化发布 + +#### GitHub Actions 示例 +```yaml +name: Publish Skill +on: + push: + tags: + - 'v*' + +jobs: + publish: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v3 + + - name: Setup Node.js + uses: actions/setup-node@v3 + with: + node-version: '18' + + - name: Install ClawHub CLI + run: npm install -g clawhub + + - name: Login to ClawHub + run: echo "${{ secrets.CLAWHUB_TOKEN }}" | clawhub login --token + + - name: Publish Skill + run: | + clawhub publish ./skills/word-reader \ + --slug word-reader \ + --version ${{ github.ref_name }} \ + --changelog "Published from GitHub Actions" +``` + +### 6. 发布注意事项 + +#### 必须遵守的规则 +- [ ] 技能名称不能与其他技能冲突 +- [ ] 版本号遵循 SemVer 规范 +- [ ] changelog 清晰描述变更 +- [ ] 代码无安全漏洞 +- [ ] 许可证声明清晰 + +#### 最佳实践 +- [ ] 发布前充分测试 +- [ ] 提供清晰的使用示例 +- [ ] 维护更新日志 +- [ ] 及时修复问题 +- [ ] 关注用户反馈 + +### 7. 故障排除 + +#### 常见问题 +```bash +# 验证发布权限 +clawhub whoami + +# 检查技能格式 +clawhub validate ./word-reader + +# 查看详细错误信息 +clawhub publish ./word-reader --verbose +``` + +#### 重新发布 +如果发布失败,可以: +1. 修正问题 +2. 增加版本号 +3. 重新发布 + +### 8. 维护指南 + +#### 监控使用情况 +- 定期查看下载统计 +- 关注用户反馈 +- 及时修复问题 + +#### 更新策略 +- 重要修复:紧急发布补丁版本 +- 新功能:发布次版本号 +- 重大变更:发布主版本号 + +现在你的 Word Reader 技能已经准备好发布到 ClawHub 了! \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/README.md b/runtime/skills/_workspace/word-reader/README.md new file mode 100644 index 00000000..3e881839 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/README.md @@ -0,0 +1,171 @@ +# Word Reader 技能 + +## 📋 概述 + +Word Reader 是一个强大的 Word 文档读取工具,支持 .docx 和 .doc 格式,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。 + +## ✨ 功能特性 + +- ✅ **文本提取** - 提取文档中的所有段落文本 +- ✅ **表格解析** - 解析表格数据并转换为结构化格式 +- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等) +- ✅ **图片信息** - 获取文档中图片的基本信息 +- ✅ **多格式支持** - 支持 .docx 和 .doc 格式 +- ✅ **多种输出** - JSON、Text、Markdown 格式 +- ✅ **批量处理** - 支持处理整个目录的文档 +- ✅ **自动安装** - 一键安装所有依赖 + +## 🚀 安装 + +### 自动安装(推荐) +```bash +cd word-reader/ +./install.sh +``` + +### 手动安装 +```bash +# 安装 Python 依赖 +pip3 install python-docx --break-system-packages + +# 安装系统依赖(可选,用于 .doc 格式支持) +# Ubuntu/Debian +sudo apt-get install antiword + +# macOS +brew install antiword + +# 设置执行权限 +chmod +x scripts/read_word.py +``` + +## 📖 使用方法 + +### 基本用法 +```bash +# 读取文档并输出为文本格式 +python3 scripts/read_word.py 文档.docx + +# 输出为 JSON 格式 +python3 scripts/read_word.py 文档.docx --format json + +# 输出为 Markdown 格式 +python3 scripts/read_word.py 文档.docx --format markdown + +# 只提取文本内容 +python3 scripts/read_word.py 文档.docx --extract text +``` + +### 批量处理 +```bash +# 批量处理目录下所有 Word 文档 +python3 scripts/read_word.py ./文档目录 --batch + +# 批量处理并保存为 JSON 文件 +python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json +``` + +### 高级用法 +```bash +# 将结果保存到文件 +python3 scripts/read_word.py 文档.docx --format markdown --output output.md + +# 提取表格数据 +python3 scripts/read_word.py 文档.docx --extract tables + +# 获取文档元数据 +python3 scripts/read_word.py 文档.docx --extract metadata +``` + +## 📊 输出示例 + +### JSON 格式输出 +```json +{ + "metadata": { + "filename": "测试文档.docx", + "size": "2048 bytes", + "created": "2024-01-01T10:00:00", + "modified": "2024-01-01T12:00:00", + "title": "测试文档", + "author": "测试用户" + }, + "format": "docx", + "text": "这是文档的正文内容...", + "tables": [ + { + "id": 1, + "rows": 3, + "columns": 3, + "data": [ + ["表头1", "表头2", "表头3"], + ["数据1", "数据2", "数据3"], + ["数据4", "数据5", "数据6"] + ] + } + ], + "images": [ + { + "id": "rId1", + "filename": "image1.png", + "size": "1024 bytes" + } + ] +} +``` + +### Markdown 格式输出 +```markdown +# 测试文档.docx + +**标题**:测试文档 +**作者**:测试用户 +**文件大小**:2048 bytes +**创建时间**:2024-01-01T10:00:00 +**修改时间**:2024-01-01T12:00:00 + +## 正文内容 + +这是文档的正文内容... + +## 表格内容 + +### 表格 1 (3行 x 3列) + +| 表头1 | 表头2 | 表头3 | +|-------|-------|-------| +| 数据1 | 数据2 | 数据3 | +| 数据4 | 数据5 | 数据6 | +``` + +## 🎯 应用场景 + +- **文档内容分析** - 快速查看 Word 文档内容 +- **批量处理** - 处理大量文档 +- **内容提取** - 提取特定信息 +- **格式转换** - 转换为其他格式 +- **自动化工作流** - 集成到文档处理系统 + +## 📤 发布到 ClawHub + +要将此技能发布到 ClawHub,请参考 `PUBLISHING.md` 文件。 + +## 🔧 故障排除 + +### 常见问题 +1. **ModuleNotFoundError**: 确保已安装 python-docx +2. **PermissionError**: 检查文件读取权限 +3. **FileNotFoundError**: 确认文件路径正确 +4. **编码问题**: 尝试使用 `--encoding gb2312` 参数 + +### 性能优化 +- 大文档处理时建议使用 `--format json` 以获得更好的性能 +- 批量模式下建议使用 `--output` 参数将结果保存到文件 + +## 🤝 贡献 + +欢迎提交 Issue 和 Pull Request 来改进这个技能! + +## 📄 许可证 + +MIT License \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/SKILL.md b/runtime/skills/_workspace/word-reader/SKILL.md new file mode 100644 index 00000000..8e5bd6b2 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/SKILL.md @@ -0,0 +1,225 @@ +--- +name: word-reader +description: | + 读取 Word 文档(.docx 和 .doc 格式)并提取文本内容。支持文档解析、表格提取、图片处理等功能。使用当用户需要分析 Word 文档内容、提取文本信息或批量处理文档时。 +homepage: https://python-docx.readthedocs.io/ +metadata: + { + "openclaw": + { + "emoji": "📄", + "requires": { "bins": ["python3"], "env": ["PYTHONPATH"] }, + "install": + [ + { + "id": "pip", + "kind": "pip", + "package": "python-docx", + "bins": ["python3"], + "label": "Install python-docx (pip)", + }, + { + "id": "system", + "kind": "system", + "command": "sudo apt-get install antiword -y", + "label": "Install antiword for .doc support (optional)", + "platform": "linux-debian" + } + ], + }, + } +--- + +# Word 文档读取器 + +使用 Python 解析 Word 文档,提取文本内容和结构化信息。 + +## 支持的功能 + +- **文档文本提取** - 提取段落、标题、页眉页脚内容 +- **表格解析** - 读取表格数据并转换为结构化格式 +- **图片处理** - 提取文档中的图片信息 +- **元数据获取** - 读取文档属性(作者、标题、创建时间等) +- **批量处理** - 支持处理多个文档 + +## 用法 + +### 基本文本提取 + +```bash +python3 {baseDir}/scripts/read_word.py <文件路径> +``` + +### 指定输出格式 + +```bash +# JSON 输出 +python3 {baseDir}/scripts/read_word.py <文件路径> --format json + +# 纯文本输出 +python3 {baseDir}/scripts/read_word.py <文件路径> --format text + +# Markdown 格式 +python3 {baseDir}/scripts/read_word.py <文件路径> --format markdown +``` + +### 提取特定内容 + +```bash +# 只提取文本 +python3 {baseDir}/scripts/read_word.py <文件路径> --extract text + +# 提取表格数据 +python3 {baseDir}/scripts/read_word.py <文件路径> --extract tables + +# 获取文档元数据 +python3 {baseDir}/scripts/read_word.py <文件路径> --extract metadata +``` + +### 批量处理 + +```bash +# 处理目录下所有 .docx 文件 +python3 {baseDir}/scripts/read_word.py <目录路径> --batch +``` + +## 参数说明 + +| 参数 | 说明 | 默认值 | +|------|------|--------| +| `--format` | 输出格式(json/text/markdown) | text | +| `--extract` | 提取内容类型(text/tables/images/metadata/all) | all | +| `--batch` | 批量处理模式 | false | +| `--output` | 输出文件路径 | stdout | +| `--encoding` | 文本编码(utf-8/gb2312) | utf-8 | + +## 输出格式 + +### JSON 格式 + +```json +{ + "metadata": { + "title": "文档标题", + "author": "作者姓名", + "created": "2024-01-01T10:00:00", + "modified": "2024-01-01T12:00:00" + }, + "text": "文档全文内容...", + "tables": [ + [ + ["表头1", "表头2"], + ["行1列1", "行1列2"], + ["行2列1", "行2列2"] + ] + ], + "images": [ + { + "filename": "image1.png", + "description": "图片描述", + "size": "1024x768" + } + ] +} +``` + +### Markdown 格式 + +```markdown +# 文档标题 + +**作者**:作者姓名 +**创建时间**:2024-01-01 10:00:00 + +## 正文内容 + +这是文档的正文内容... + +### 表格示例 + +| 表头1 | 表头2 | +|-------|-------| +| 行1列1 | 行1列2 | +| 行2列1 | 行2列2 | + +![图片描述](image1.png) + +## 图片列表 + +1. **image1.png** (1024x768) - 图片描述 +``` + +## 错误处理 + +- 文件不存在:显示错误信息并退出 +- 格式不支持:提示支持的文件类型 +- 权限问题:提示文件访问权限 +- 编码问题:尝试自动检测编码 + +## 示例场景 + +### 1. 查看项目文档 + +```bash +python3 {baseDir}/scripts/read_word.py 项目需求.docx --format markdown +``` + +### 2. 提取会议记录 + +```bash +python3 {baseDir}/scripts/read_word.py 会议记录.docx --extract text +``` + +### 3. 批量处理文档 + +```bash +python3 {baseDir}/scripts/read_word.py ./文档目录 --batch --format json --output results.json +``` + +## 注意事项 + +- 支持 .docx 格式(Office 2007+) +- .doc 格式需要额外依赖(如 antiword) +- 大文档处理可能需要较长时间 +- 图片提取仅获取元数据,不包含实际图片数据 +- 表格格式可能需要手动调整 + +## 故障排除 + +### 常见问题 + +1. **ModuleNotFoundError**: 确保已安装 python-docx +2. **PermissionError**: 检查文件读取权限 +3. **UnicodeDecodeError**: 尝试不同的编码格式 + +### 安装依赖 + +```bash +pip3 install python-docx +``` + +对于 .doc 格式支持: +```bash +# Ubuntu/Debian +sudo apt-get install antiword + +# macOS +brew install antiword +``` + +## 高级功能 + +### 自定义样式处理 + +脚本会自动处理以下文档元素: +- 标题级别(H1-H6) +- 段落样式 +- 列表项目 +- 页眉页脚 +- 文档属性 + +### 性能优化 + +- 大文件流式处理 +- 内存使用优化 +- 进度显示(批量模式) \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/_meta.json b/runtime/skills/_workspace/word-reader/_meta.json new file mode 100644 index 00000000..3f2ee4ad --- /dev/null +++ b/runtime/skills/_workspace/word-reader/_meta.json @@ -0,0 +1,11 @@ +{ + "owner": "xtfnhcyjpgf", + "slug": "word-reader", + "displayName": "Word Reader", + "latest": { + "version": "1.0.0", + "publishedAt": 1770700102926, + "commit": "https://github.com/openclaw/skills/commit/91b71e101c57b69a4d4eb2678e1b79992eb7032f" + }, + "history": [] +} diff --git a/runtime/skills/_workspace/word-reader/demo.sh b/runtime/skills/_workspace/word-reader/demo.sh new file mode 100644 index 00000000..8d6f2630 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/demo.sh @@ -0,0 +1,89 @@ +#!/bin/bash + +# Word Reader 技能演示脚本 +# 此脚本展示如何使用 word-reader 技能 + +echo "=== Word Reader 技能演示 ===" +echo "" + +# 检查脚本是否存在 +SCRIPT_PATH="/root/.openclaw/workspace/skills/word-reader/scripts/read_word.py" +if [ ! -f "$SCRIPT_PATH" ]; then + echo "❌ 错误:脚本不存在" + echo "请确保技能已正确安装" + exit 1 +fi + +# 检查脚本是否有执行权限 +if [ ! -x "$SCRIPT_PATH" ]; then + echo "❌ 错误:脚本没有执行权限" + echo "正在添加执行权限..." + chmod +x "$SCRIPT_PATH" +fi + +echo "✅ 脚本已就绪" +echo "" + +# 显示技能信息 +echo "📋 技能信息:" +echo " 名称:word-reader" +echo " 功能:读取 Word 文档(.docx 和 .doc 格式)" +echo " 位置:$SCRIPT_PATH" +echo "" + +# 显示使用示例 +echo "📖 使用示例:" +echo "" + +echo "1. 显示帮助信息:" +echo " python3 $SCRIPT_PATH --help" +echo "" + +echo "2. 读取文档(文本格式):" +echo " python3 $SCRIPT_PATH 文档路径.docx" +echo "" + +echo "3. 读取文档(JSON 格式):" +echo " python3 $SCRIPT_PATH 文档路径.docx --format json" +echo "" + +echo "4. 读取文档(Markdown 格式):" +echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown" +echo "" + +echo "5. 只提取文本内容:" +echo " python3 $SCRIPT_PATH 文档路径.docx --extract text" +echo "" + +echo "6. 批量处理目录:" +echo " python3 $SCRIPT_PATH ./文档目录 --batch" +echo "" + +echo "7. 保存结果到文件:" +echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown --output output.md" +echo "" + +echo "🔧 安装依赖:" +echo " pip3 install python-docx" +echo " # 对于 .doc 格式支持:" +echo " # Ubuntu: sudo apt-get install antiword" +echo " # macOS: brew install antiword" +echo "" + +echo "📊 支持的功能:" +echo " ✅ 文本提取" +echo " ✅ 表格解析" +echo " ✅ 元数据获取" +echo " ✅ 图片信息" +echo " ✅ 多格式支持" +echo " ✅ 批量处理" +echo "" + +echo "💡 提示:" +echo " - 支持 .docx 和 .doc 格式" +echo " - 输出格式:JSON、Text、Markdown" +echo " - 如遇错误,请检查依赖是否安装" +echo "" + +echo "演示完成!" +echo "如需使用,请替换 '文档路径.docx' 为实际的文档路径" \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/install.sh b/runtime/skills/_workspace/word-reader/install.sh new file mode 100644 index 00000000..4947e28e --- /dev/null +++ b/runtime/skills/_workspace/word-reader/install.sh @@ -0,0 +1,101 @@ +#!/bin/bash + +# Word Reader 技能安装脚本 +# 此脚本会自动安装依赖并设置技能 + +set -e + +echo "=== Word Reader 技能安装 ===" +echo "" + +# 检查 Python 版本 +echo "🔍 检查 Python 版本..." +python_version=$(python3 --version 2>&1) +echo " Python 版本: $python_version" + +if ! python3 -c "import sys; assert sys.version_info >= (3, 6)"; then + echo "❌ 错误:需要 Python 3.6 或更高版本" + exit 1 +fi + +echo "✅ Python 版本检查通过" +echo "" + +# 检查并安装依赖 +echo "📦 检查依赖..." + +# 检查 pip +if ! command -v pip3 &> /dev/null; then + echo " 🔧 安装 pip..." + python3 -m ensurepip --upgrade 2>/dev/null || { + echo " ❌ 无法安装 pip,尝试使用系统包管理器" + if command -v apt &> /dev/null; then + sudo apt update + sudo apt install -y python3-pip + elif command -v yum &> /dev/null; then + sudo yum install -y python3-pip + elif command -v brew &> /dev/null; then + brew install python3 + else + echo " ❌ 无法自动安装 pip,请手动安装" + exit 1 + fi + } +fi + +# 检查 python-docx +if ! python3 -c "import docx" 2>/dev/null; then + echo " 🔧 安装 python-docx..." + if python3 -m pip install python-docx --break-system-packages 2>/dev/null; then + echo " ✅ python-docx 安装完成" + elif python3 -m pip install python-docx 2>/dev/null; then + echo " ✅ python-docx 安装完成" + else + echo "❌ 无法安装 python-docx" + exit 1 + fi +else + echo " ✅ python-docx 已安装" +fi + +# 检查 antiword(可选) +if command -v antiword >/dev/null 2>&1; then + echo " ✅ antiword 已安装" +else + echo " ⚠️ antiword 未安装(可选,用于 .doc 格式支持)" + echo " 推荐安装命令:" + echo " Ubuntu/Debian: sudo apt-get install antiword" + echo " macOS: brew install antiword" +fi + +echo "" + +# 设置执行权限 +echo "🔐 设置执行权限..." +chmod +x scripts/read_word.py +echo "✅ 执行权限已设置" +echo "" + +# 验证安装 +echo "🧪 验证安装..." +python3 scripts/read_word.py --help >/dev/null 2>&1 +if [ $? -eq 0 ]; then + echo "✅ 安装验证成功" +else + echo "❌ 安装验证失败" + exit 1 +fi + +echo "" +echo "🎉 Word Reader 技能安装完成!" +echo "" +echo "📖 使用方法:" +echo " python3 scripts/read_word.py 文档.docx" +echo " python3 scripts/read_word.py 文档.docx --format json" +echo " python3 scripts/read_word.py 文档.docx --format markdown" +echo "" +echo "📖 更多帮助:" +echo " python3 scripts/read_word.py --help" +echo "" +echo "📖 运行演示:" +echo " ./demo.sh" \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/scripts/read_word.py b/runtime/skills/_workspace/word-reader/scripts/read_word.py new file mode 100644 index 00000000..27448eec --- /dev/null +++ b/runtime/skills/_workspace/word-reader/scripts/read_word.py @@ -0,0 +1,396 @@ +#!/usr/bin/env python3 +""" +Word 文档读取器 +支持 .docx 和 .doc 格式的 Word 文档解析 +""" + +import argparse +import json +import os +import sys +import re +import traceback +from datetime import datetime +from pathlib import Path + +try: + from docx import Document + from docx.opc.constants import RELATIONSHIP_TYPE as RT + from docx.oxml.table import CT_Tbl + from docx.oxml.text.paragraph import CT_P + from docx.table import Table + from docx.text.paragraph import Paragraph + DOCX_AVAILABLE = True +except ImportError: + DOCX_AVAILABLE = False + +try: + import subprocess + SUBPROCESS_AVAILABLE = True +except ImportError: + SUBPROCESS_AVAILABLE = False + +class WordReader: + """Word 文档读取器""" + + def __init__(self, file_path): + self.file_path = Path(file_path) + self.document = None + self.format_type = None + self.encoding = 'utf-8' + + # 检查文件是否存在 + if not self.file_path.exists(): + raise FileNotFoundError(f"文件不存在: {file_path}") + + # 检查文件扩展名 + if self.file_path.suffix.lower() not in ['.docx', '.doc']: + raise ValueError(f"不支持的文件格式: {self.file_path.suffix}") + + def read_docx(self): + """读取 .docx 格式文档""" + if not DOCX_AVAILABLE: + raise Exception("缺少 python-docx 库。请安装:pip3 install python-docx") + + try: + self.document = Document(str(self.file_path)) + self.format_type = 'docx' + return True + except Exception as e: + raise Exception(f"读取 .docx 文件失败: {str(e)}") + + def read_doc(self): + """读取 .doc 格式文档(使用 antiword)""" + if not SUBPROCESS_AVAILABLE: + raise Exception("缺少 subprocess 模块") + + try: + # 检查 antiword 是否可用 + result = subprocess.run(['which', 'antiword'], + capture_output=True, text=True) + if result.returncode != 0: + raise Exception("antiword 未安装。请安装 antiword: Ubuntu/Debian: sudo apt-get install antiword; macOS: brew install antiword") + + # 使用 antiword 转换 + result = subprocess.run(['antiword', str(self.file_path)], + capture_output=True, text=True, encoding='utf-8') + + if result.returncode != 0: + raise Exception(f"antiword 转换失败: {result.stderr}") + + # 创建临时文档对象 + class TempDocument: + def __init__(self, text): + self.text = text + self.paragraphs = [TempParagraph(p) for p in text.split('\n') if p.strip()] + + class TempParagraph: + def __init__(self, text): + self.text = text + + self.document = TempDocument(result.stdout) + self.format_type = 'doc' + return True + except Exception as e: + raise Exception(f"读取 .doc 文件失败: {str(e)}") + + def read_metadata(self): + """读取文档元数据""" + metadata = { + 'filename': self.file_path.name, + 'size': f"{self.file_path.stat().st_size} bytes", + 'created': datetime.fromtimestamp(self.file_path.stat().st_ctime).isoformat(), + 'modified': datetime.fromtimestamp(self.file_path.stat().st_mtime).isoformat() + } + + if self.format_type == 'docx' and hasattr(self.document, 'core_properties'): + props = self.document.core_properties + metadata.update({ + 'title': getattr(props, 'title', ''), + 'author': getattr(props, 'author', ''), + 'subject': getattr(props, 'subject', ''), + 'keywords': getattr(props, 'keywords', ''), + 'comments': getattr(props, 'comments', ''), + 'application': getattr(props, 'application', ''), + 'category': getattr(props, 'category', '') + }) + + return metadata + + def extract_text(self): + """提取文档文本""" + text_content = [] + + if self.format_type == 'docx': + # 提取段落文本 + for para in self.document.paragraphs: + if para.text.strip(): + text_content.append(para.text) + + # 提取表格文本 + for table in self.document.tables: + table_text = [] + for row in table.rows: + row_text = [] + for cell in row.cells: + row_text.append(cell.text.strip()) + table_text.append(' | '.join(row_text)) + text_content.append('\n'.join(table_text)) + + else: # doc 格式 + text_content = [para.text for para in self.document.paragraphs if para.text.strip()] + + return '\n\n'.join(text_content) + + def extract_tables(self): + """提取表格数据""" + tables = [] + + if self.format_type == 'docx': + for i, table in enumerate(self.document.tables): + table_data = [] + for row in table.rows: + row_data = [] + for cell in row.cells: + row_data.append(cell.text.strip()) + table_data.append(row_data) + tables.append({ + 'id': i + 1, + 'rows': len(table.rows), + 'columns': len(table.columns) if table.rows else 0, + 'data': table_data + }) + + return tables + + def extract_images(self): + """提取图片信息""" + images = [] + + if self.format_type == 'docx': + try: + # 获取文档中的关系 + part = self.document.part + image_parts = part.related_parts + + for rel in part.relationships: + if rel.reltype == RT.IMAGE: + image_data = image_parts[rel.rId]._blob + image_info = { + 'id': rel.rId, + 'filename': f"image_{rel.rId}.{rel.target_ref.split('.')[-1]}", + 'size': f"{len(image_data)} bytes" + } + images.append(image_info) + except: + # 图片提取可能失败,忽略错误 + pass + + return images + + def extract_all(self): + """提取所有内容""" + result = { + 'metadata': self.read_metadata(), + 'format': self.format_type, + 'text': self.extract_text(), + 'tables': self.extract_tables(), + 'images': self.extract_images() + } + return result + + def to_markdown(self, extract_type='all'): + """转换为 Markdown 格式""" + if extract_type == 'text': + return self.extract_text() + + result = self.extract_all() + md_content = [] + + # 标题 + md_content.append(f"# {result['metadata']['filename']}") + md_content.append("") + + # 元数据 + metadata = result['metadata'] + if metadata.get('title'): + md_content.append(f"**标题**:{metadata['title']}") + if metadata.get('author'): + md_content.append(f"**作者**:{metadata['author']}") + md_content.append(f"**文件大小**:{metadata['size']}") + md_content.append(f"**创建时间**:{metadata['created']}") + md_content.append(f"**修改时间**:{metadata['modified']}") + md_content.append("") + + # 文本内容 + if result['text']: + md_content.append("## 正文内容") + md_content.append("") + md_content.append(result['text']) + md_content.append("") + + # 表格 + if result['tables']: + md_content.append("## 表格内容") + md_content.append("") + for table in result['tables']: + md_content.append(f"### 表格 {table['id']} ({table['rows']}行 x {table['columns']}列)") + md_content.append("") + # 转换为 Markdown 表格 + for row in table['data']: + md_row = " | ".join([str(cell) for cell in row]) + md_content.append(f"| {md_row} |") + md_content.append("") + + # 图片 + if result['images']: + md_content.append("## 图片列表") + md_content.append("") + for img in result['images']: + md_content.append(f"- **{img['filename']}** ({img['size']})") + md_content.append("") + + return '\n'.join(md_content) + + def to_text(self, extract_type='all'): + """转换为纯文本格式""" + if extract_type == 'text': + return self.extract_text() + + result = self.extract_all() + text_content = [] + + # 标题和元数据 + text_content.append(f"文件:{result['metadata']['filename']}") + text_content.append("=" * 50) + text_content.append("") + + for key, value in result['metadata'].items(): + if value and key not in ['filename', 'size', 'created', 'modified']: + text_content.append(f"{key}:{value}") + + text_content.append("") + + # 文本内容 + if result['text']: + text_content.append("正文内容:") + text_content.append("-" * 20) + text_content.append(result['text']) + text_content.append("") + + # 表格 + if result['tables']: + text_content.append("表格内容:") + text_content.append("-" * 20) + for table in result['tables']: + text_content.append(f"表格 {table['id']}:") + for row in table['data']: + text_content.append(" " + " | ".join([str(cell) for cell in row])) + text_content.append("") + + return '\n'.join(text_content) + +def main(): + parser = argparse.ArgumentParser(description='读取 Word 文档') + parser.add_argument('path', help='文档路径或目录路径(批量模式)') + parser.add_argument('--format', choices=['json', 'text', 'markdown'], + default='text', help='输出格式') + parser.add_argument('--extract', choices=['text', 'tables', 'images', 'metadata', 'all'], + default='all', help='提取内容类型') + parser.add_argument('--batch', action='store_true', help='批量处理模式') + parser.add_argument('--output', help='输出文件路径') + parser.add_argument('--encoding', default='utf-8', help='文本编码') + + args = parser.parse_args() + + try: + if args.batch: + # 批量处理模式 + path = Path(args.path) + if not path.is_dir(): + print("错误:批量模式需要指定目录路径") + sys.exit(1) + + # 查找所有 Word 文档 + word_files = [] + for ext in ['.docx', '.doc']: + word_files.extend(path.glob(f"**/*{ext}")) + + if not word_files: + print("未找到 Word 文档") + sys.exit(0) + + print(f"找到 {len(word_files)} 个 Word 文档") + + results = {} + for file_path in word_files: + print(f"正在处理: {file_path}") + try: + reader = WordReader(file_path) + if file_path.suffix.lower() == '.docx': + reader.read_docx() + else: + reader.read_doc() + + if args.format == 'json': + content = reader.extract_all() + elif args.format == 'markdown': + content = reader.to_markdown(args.extract) + else: + content = reader.to_text(args.extract) + + results[str(file_path)] = { + 'filename': file_path.name, + 'content': content, + 'status': 'success' + } + + except Exception as e: + results[str(file_path)] = { + 'filename': file_path.name, + 'error': str(e), + 'status': 'failed' + } + + # 保存结果 + if args.output: + with open(args.output, 'w', encoding='utf-8') as f: + json.dump(results, f, ensure_ascii=False, indent=2) + print(f"结果已保存到: {args.output}") + else: + print(json.dumps(results, ensure_ascii=False, indent=2)) + + else: + # 单文件处理模式 + reader = WordReader(args.path) + + # 根据文件类型读取 + if args.path.lower().endswith('.docx'): + reader.read_docx() + else: + reader.read_doc() + + # 根据格式输出 + if args.format == 'json': + content = reader.extract_all() + elif args.format == 'markdown': + content = reader.to_markdown(args.extract) + else: + content = reader.to_text(args.extract) + + # 输出结果 + if args.output: + with open(args.output, 'w', encoding=args.encoding) as f: + f.write(content) + print(f"结果已保存到: {args.output}") + else: + print(content) + + except Exception as e: + print(f"错误: {str(e)}", file=sys.stderr) + if '--debug' in sys.argv or '-d' in sys.argv: + traceback.print_exc() + sys.exit(1) + +if __name__ == '__main__': + main() \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/skill.json b/runtime/skills/_workspace/word-reader/skill.json new file mode 100644 index 00000000..435f3ea9 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/skill.json @@ -0,0 +1,47 @@ +{ + "name": "word-reader", + "version": "1.0.0", + "description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容", + "author": "OpenClaw User", + "tags": ["document", "word", "office", "text-extraction"], + "dependencies": { + "python": ">=3.6", + "packages": ["python-docx"], + "system": ["antiword (optional for .doc support)"] + }, + "features": { + "text_extraction": true, + "table_parsing": true, + "metadata_extraction": true, + "image_info": true, + "batch_processing": true, + "multiple_formats": ["json", "text", "markdown"] + }, + "installation": { + "steps": [ + "pip3 install python-docx", + "sudo apt-get install antiword # 可选,支持 .doc 格式", + "chmod +x scripts/read_word.py" + ] + }, + "usage_examples": [ + { + "description": "读取文档文本", + "command": "python3 scripts/read_word.py document.docx" + }, + { + "description": "转换为 Markdown", + "command": "python3 scripts/read_word.py document.docx --format markdown" + }, + { + "description": "批量处理", + "command": "python3 scripts/read_word.py ./docs --batch --format json" + } + ], + "supported_file_types": [".docx", ".doc"], + "notes": [ + ".doc 格式需要安装 antiword", + "大文档处理可能需要较长时间", + "图片提取仅获取元数据,不包含实际图片数据" + ] +} \ No newline at end of file diff --git a/runtime/skills/_workspace/word-reader/test.md b/runtime/skills/_workspace/word-reader/test.md new file mode 100644 index 00000000..8029a815 --- /dev/null +++ b/runtime/skills/_workspace/word-reader/test.md @@ -0,0 +1,42 @@ +# Word Reader 技能测试 + +这是一个简单的测试文档,用于验证 Word Reader 技能的功能。 + +## 测试内容 + +### 1. 基本文本 +这是一段测试文本,用于验证文本提取功能是否正常工作。 + +### 2. 表格测试 + +| 功能 | 状态 | 描述 | +|------|------|------| +| 文本提取 | ✅ | 能够提取文档中的所有文本内容 | +| 表格解析 | ✅ | 能够正确解析表格数据 | +| 元数据获取 | ✅ | 能够获取文档属性信息 | +| 多格式支持 | ✅ | 支持 .docx 和 .doc 格式 | +| 输出格式 | ✅ | 支持 JSON、Text、Markdown 格式 | + +### 3. 列表测试 + +- 第一项:文本提取功能 +- 第二项:表格解析功能 +- 第三项:图片信息获取 +- 第四项:文档元数据读取 + +### 4. 代码块示例 + +```python +def read_word_document(file_path): + """读取 Word 文档""" + reader = WordReader(file_path) + if file_path.endswith('.docx'): + reader.read_docx() + else: + reader.read_doc() + return reader.extract_all() +``` + +## 测试完成 + +如果这个技能能够正确读取并解析上述内容,说明功能正常。 \ No newline at end of file diff --git a/runtime/skills/cocoloop/README.md b/runtime/skills/cocoloop/README.md deleted file mode 100644 index a85e77cd..00000000 --- a/runtime/skills/cocoloop/README.md +++ /dev/null @@ -1,140 +0,0 @@ -# Cocoloop - -一个更快速、更安全的 Skill 管理器,用于安装、管理、更新和卸载 Skills。 - -[![License](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE) - -## 简介 - -Cocoloop 是一个安全优先的 Skill 管理器,提供比 clawhub 更智能的安装体验和集成 BSS 安全认证。 - -## 功能特性 - -- **单个 Skill 安装** - 支持 URL、名称搜索、GitHub 等多种来源 -- **批量 Skills 安装** - 依次安装多个 skills -- **Skill 更新** - 检查并更新到最新版本 -- **Skill 卸载** - 安全卸载已安装的 skills -- **安全检查** - 集成 BSS 安全认证系统 - -## 安装 - -```bash -# 克隆仓库 -git clone https://github.com/CatREFuse/cocoloop.git -cd cocoloop -``` - -## 使用方法 - -### 安装单个 Skill - -```bash -# 通过名称安装 -cocoloop install pdf-processor - -# 通过 URL 安装 -cocoloop install https://example.com/skill-name.skill - -# 通过 GitHub 安装 -cocoloop install owner/repo -``` - -### 批量安装 Skills - -```bash -cocoloop install skill1 skill2 skill3 -``` - -### 更新 Skill - -```bash -cocoloop update pdf-processor -``` - -### 卸载 Skill - -```bash -cocoloop uninstall pdf-processor -``` - -### 安全检查 - -```bash -cocoloop check pdf-processor -``` - -## 安全检查系统 - -Cocoloop 集成了 BSS (Berry Skills Safe) 安全认证检查,评级标准: - -- **S+** - 最高安全等级 -- **S** - 优秀 -- **A** - 良好 -- **B** - 一般(需谨慎) -- **C** - 风险较高 -- **D** - 不建议使用 - -### 动态代码加载检查 - -实施最多 2 层的 URL 递归检查,识别隐藏的多层动态加载风险: - -- 无动态加载:正常评级流程 -- 仅第 1 层动态加载:根据来源分级处理 -- 存在第 2 层动态加载:最高评级为 C 级 -- 第 2 层后仍有动态加载:强制标记为 C 级 - -## 支持的平台 - -- OpenClaw -- Molili -- Claude Code - -## 文档 - -- [安装流程指南](references/install-guide.md) -- [搜索流程指南](references/search-guide.md) -- [卸载流程指南](references/uninstall-guide.md) -- [安全检查流程指南](references/safety-check-guide.md) -- [Cocoloop Safe Check 标准](references/cocoloop-safe-check.md) - -## 工作流程 - -### Skill 安装流程 - -1. **平台检测** - 确定当前运行环境和安装方式 -2. **来源识别** - 支持直接 URL、Skill 名称、GitHub 短链接 -3. **搜索与下载** - 从 Cocoloop API、clawhub 或 GitHub 获取 -4. **安全检查** - BSS 安全认证检查 -5. **安装执行** - 安装到对应平台的 skill 目录 - -### 搜索优先级 - -1. Cocoloop API 搜索 -2. Fallback 到 clawhub -3. Fallback 到 GitHub 搜索 - -## 项目结构 - -``` -cocoloop/ -├── SKILL.md # Skill 定义文件 -├── README.md # 项目说明文档 -└── references/ # 详细指南文档 - ├── install-guide.md # 安装流程指南 - ├── search-guide.md # 搜索流程指南 - ├── uninstall-guide.md # 卸载流程指南 - ├── safety-check-guide.md # 安全检查流程指南 - └── cocoloop-safe-check.md # 安全检查标准 -``` - -## 贡献 - -欢迎提交 Issue 和 Pull Request! - -## 许可证 - -[MIT](LICENSE) - ---- - -Made with ❤️ by Cocoloop Team diff --git a/runtime/skills/cocoloop/SKILL.md b/runtime/skills/cocoloop/SKILL.md deleted file mode 100644 index 949d9cfb..00000000 --- a/runtime/skills/cocoloop/SKILL.md +++ /dev/null @@ -1,257 +0,0 @@ ---- -name: cocoloop -description: 一个更快速、更安全的 Skill 管理器,用于安装、管理、更新和卸载 Skills。优先使用当用户需要安装 skill、更新 skill、卸载 skill、管理 skills 或进行 skill 安全检查时。支持通过 URL、名称搜索、GitHub 等多种方式定位并安装 skills,集成 BSS 安全认证系统。 ---- - -# Cocoloop Skill 管理器 - -Cocoloop 是一个安全优先的 Skill 管理器,提供比 clawhub 更智能的安装体验和集成 BSS 安全认证。 - -## 核心功能 - -1. **单个 Skill 安装** - 支持 URL、名称搜索、GitHub 等多种来源 -2. **批量 Skills 安装** - 依次安装多个 skills -3. **Skill 更新** - 检查并更新到最新版本 -4. **Skill 卸载** - 安全卸载已安装的 skills -5. **安全检查** - 集成 BSS 安全认证系统 - -## 工作流程概览 - -### 平台检测 - -首先检测当前运行环境,确定 skill 安装方式. - -### 1. 单个 Skill 安装流程 - -用户输入可能是以下三种情况之一: - -#### 情况 1: 直接 URL - -输入格式:`https://example.com/skill-name.skill` 或 `http://...` - -处理流程: - -1. 使用 HTTP GET 请求下载内容 -2. 处理 3xx 重定向(自动跟随跳转 URL) -3. 保存到临时路径(如 `/tmp/cocoloop-{timestamp}.skill`) -4. 调用检测到的平台安装命令 -5. 清理临时文件 -6. 返回安装结果 - -异常情况处理: - -- **URL 无法访问** → 返回错误信息,提示用户检查 URL -- **内容无法识别** → 尝试查找页面中的跳转链接或下载按钮 -- **下载成功但安装失败** → 保留临时文件,提示用户手动安装 - -#### 情况 2: Skill 名称(最常见) - -输入格式:`skill-name`(如 `pdf-processor`) - -处理流程(按优先级): - -**步骤 1: CocoLoop API 搜索(最优先使用)** - -- 调用 `https://api.cocoloop.com/api/v1/store/skills?page={page}&page_size={page_size}&keyword={keyword}&sort=downloads` -- **请优先使用 `curl` 命令工具进行请求** -- 返回格式示例: - ```json - { - "results": [ - { - "name": "pdf-processor", - "description": "PDF processing skill", - "url": "https://...", - "version": "1.0.0", - "author": "cocoloop" - } - ] - } - ``` -- 如果找到结果 → 展示列表,询问用户选择 - -**步骤 2: Fallback 到平台 Skills API 安装(API 失败时)** - -- 不要依赖 `run_command` 执行外部安装命令(如 `npx clawhub ...`)。 -- 优先调用平台内置 Skills API: - - 通过 registry 安装:`POST /admin/api/skills/install-registry` - - 通过 market 安装:`POST /admin/api/skills/market/install` - - 本地目录安装:`POST /admin/api/skills/install` -- 如果 API 安装成功 → 完成安装 -- 如果 API 安装失败 → 进入步骤 3 - -**步骤 3: Fallback 到 GitHub 搜索** - -- 调用 GitHub API: `https://api.github.com/search/repositories?q={query}+filename:SKILL.md` -- 筛选条件:仓库中包含 `SKILL.md` 文件 -- 返回结果按 stars 数排序 -- 展示搜索结果(最多 5 个): - ``` - 📋 GitHub 搜索结果: - 1. owner/skill-name (⭐ 150) - 🏢 Organization | 描述文本 - 2. user/another-skill (⭐ 45) - 👤 User | 描述文本 - ``` -- 询问用户是否安装选中的 skill - -#### 情况 3: GitHub 短链接 - -输入格式:`owner/repo`(如 `anthropic/claude-skill`) - -处理流程: - -1. 识别为 GitHub 格式 -2. 调用 GitHub API 获取仓库信息 -3. 检查是否存在 `SKILL.md` 文件 -4. 询问用户确认 -5. 下载并安装 - -### 2. 批量 Skills 安装流程 - -输入格式:`skill1 skill2 skill3 ...` - -处理流程: - -1. 解析输入为多个 skill 标识符 -2. 遍历每个 skill,依次执行「单个 Skill 安装流程」 -3. 记录每个 skill 的安装结果 -4. 汇总输出结果: - ``` - 📊 批量安装结果: - skill1: ✅ 成功 - skill2: ❌ 失败 (原因) - skill3: ✅ 成功 - ``` - -注意事项: - -- 每个 skill 独立处理,一个失败不影响其他 - -### 3. Skill 更新流程 - -处理流程: - -1. 确定当前已安装的 skill 列表(读取平台配置) -2. 对于指定 skill: - a. 查询最新版本(通过 Cocoloop API 或 GitHub) - b. 比较本地版本与远程版本 - c. 如果有更新 → 执行「单个 Skill 安装流程」(覆盖安装) - d. 备份旧版本(可选) -3. 返回更新结果 - -版本比较逻辑: - -- 使用语义化版本号比较(major.minor.patch) -- 支持 `^`、`~` 等版本范围(如果配置中有) - -### 4. Skill 卸载流程 - -详见 [references/uninstall-guide.md](references/uninstall-guide.md) - -处理概要: - -1. 检测当前平台的 skill 安装目录: - - OpenClaw: `~/.openclaw/skills/` - - Molili: `~/.molili/skills/` - - Claude Code: `~/.claude/skills/` -2. 确认 skill 存在 -3. 询问用户确认卸载 -4. 删除 skill 目录 -5. 清理相关配置 -6. 返回卸载结果 - -### 5. 安全检查流程 - -详见 [references/safety-check-guide.md](references/safety-check-guide.md) 和 [references/cocoloop-safe-check.md](references/cocoloop-safe-check.md) - -处理概要: - -1. 询问用户是否进行安全检查 -2. 对要安装的 skill 进行 Cocoloop Safe Check 安全认证检查 -3. 评级标准:S+/S/A/B/C/D -4. 如果评级 <= B,强烈建议用户查看详细报告 -5. 询问用户是否继续安装 - -**动态代码加载检查(URL 递归检查):** - -检查 skill 是否从网络动态加载可执行代码,实施最多 2 层的 URL 递归检查: - -``` -Skill 代码(第 0 层) - ↓ 发现 fetch/import/require 远程 URL -第 1 层:下载并检查该 URL 内容 - ↓ 如包含新的动态加载 -第 2 层:继续检查下一层内容 - ↓ 如第 2 层仍有动态加载 - 强制标记为 C 级(多层动态加载风险) -``` - -**递归检查规则:** - -- **无动态加载**:正常评级流程 -- **仅第 1 层动态加载**:根据来源分级处理(T1→B级, T2→C级, T3→禁止) -- **存在第 2 层动态加载**:最高评级为 C 级 -- **第 2 层后仍有动态加载**:强制标记为 C 级 - -此机制用于识别隐藏的多层动态加载风险,防止通过间接方式引入未经验证的代码。 - -## 资源引用 - -- **安装流程详细指南**: [references/install-guide.md](references/install-guide.md) -- **搜索流程详细指南**: [references/search-guide.md](references/search-guide.md) -- **卸载流程详细指南**: [references/uninstall-guide.md](references/uninstall-guide.md) -- **安全检查流程指南**: [references/safety-check-guide.md](references/safety-check-guide.md) -- **Cocoloop Safe Check 安全检查标准**: [references/cocoloop-safe-check.md](references/cocoloop-safe-check.md) - -## 使用示例 - -### 安装单个 skill - -``` -用户: 安装 pdf-processor -→ 执行单个 skill 安装流程 -→ 搜索 → 确认 → 安装 → 安全检查(可选) -``` - -### 安装多个 skills - -``` -用户: 安装 pdf-processor image-editor code-formatter -→ 批量安装流程 -→ 依次处理每个 skill -``` - -### 更新 skill - -``` -用户: 更新 pdf-processor -→ 查询最新版本 -→ 对比本地版本 -→ 执行更新 -``` - -### 卸载 skill - -``` -用户: 卸载 pdf-processor -→ 检测平台 -→ 确认卸载 -→ 删除文件 -``` - -### 安全检查 - -``` -用户: 检查 pdf-processor 安全 -→ 下载/定位 skill -→ 执行 Cocoloop Safe Check 检查 -→ 生成报告 -→ 询问保存位置 -``` - -## 注意事项 - -- 每个 skill 独立处理,一个失败不影响其他 -- 询问用户请使用当前平台下的询问命令,例如 Claude Code 下的 `AskUserQuestion` -- 在 OpenClaw 环境中,安装与卸载优先使用 `/admin/api/skills/*` 路由,不要假设可用 shell 安装命令。 diff --git a/runtime/skills/cocoloop/references/install-guide.md b/runtime/skills/cocoloop/references/install-guide.md deleted file mode 100644 index c312508b..00000000 --- a/runtime/skills/cocoloop/references/install-guide.md +++ /dev/null @@ -1,245 +0,0 @@ -# Skill 安装流程详细指南 - -本文档详细描述单个 skill 的安装流程,包括所有分支逻辑和异常处理。 - -## 流程图 - -``` -开始 - ↓ -接收用户输入 (URL / 名称 / GitHub短链) - ↓ -检测运行平台 - ↓ -判断输入类型 - ├── URL ─────────→ 下载内容 ──→ 保存临时文件 ──→ 平台安装 ──→ 清理 ──→ 完成 - │ ↑ │ - │ └──────── 失败 ──────────────┘ - │ - ├── 名称 ─────────→ Cocoloop API 搜索 - │ │ - 成功? ──是──→ 展示结果 ──→ 用户确认 ──→ 下载安装 ──→ 完成 - │ │否 - │ ↓ - │ Skills API install - │ │ - 成功? ──是──→ 完成 - │ │否 - │ ↓ - │ GitHub API 搜索 - │ │ - 成功? ──是──→ 展示结果 ──→ 用户确认 ──→ 下载安装 ──→ 完成 - │ │否 - │ ↓ - │ 返回错误 - │ - └── GitHub短链 ───→ 获取仓库信息 ──→ 确认SKILL.md存在 ──→ 下载安装 ──→ 完成 -``` - -## 详细步骤 - -### 第一步:平台检测 - -检测逻辑: -``` -IF 环境变量 OPENCLAW_HOME 存在 或 /usr/local/openclaw 存在: - 平台 = OpenClaw - 安装方式 = "POST /admin/api/skills/install 或 /admin/api/skills/install-registry" - 安装目录 = ~/.openclaw/skills/ - -ELSE IF 环境变量 MOLILI_HOME 存在 或 /usr/local/molili 存在: - 平台 = Molili - 安装方式 = "molili skills install" - 安装目录 = ~/.molili/skills/ - -ELSE IF 环境变量 CLAUDE_CODE_HOME 存在 或 /usr/local/claude-code 存在: - 平台 = Claude Code - 安装方式 = "claude skills install" - 安装目录 = ~/.claude/skills/ - -ELSE: - 平台 = 通用 (clawhub fallback) - 安装方式 = "优先平台 Skills API,必要时再提示人工执行命令" - 安装目录 = ~/.claude/skills/ (或 clawhub 默认目录) -``` - -### 第二步:URL 安装流程 - -完整流程: - -1. **发送 HTTP GET 请求** - - URL: 用户提供的地址 - - Headers: - ``` - User-Agent: Cocoloop-Skill-Manager/1.0 - ``` - -2. **处理响应** - - 状态码 200 → 获取内容,进入步骤 3 - - 状态码 3xx → 从 Location header 获取跳转 URL,递归步骤 1 - - 其他状态码 → 返回错误 - -3. **保存临时文件** - - 临时路径: `/tmp/cocoloop-{timestamp}.skill` - - 写入下载内容 - -4. **执行平台安装命令** - ```bash - OpenClaw: 调用 /admin/api/skills/install(source_dir 或 archive_url) - ``` - -5. **清理与返回** - - 安装成功 → 删除临时文件 → 返回成功 - - 安装失败 → 保留临时文件(便于调试)→ 返回错误 - -异常处理: - -| 异常情况 | 处理方式 | -|---------|---------| -| URL 无法访问 | 返回错误 "无法访问该 URL,请检查网络连接或 URL 是否正确" | -| 重定向过多 | 返回错误 "该 URL 重定向次数过多,可能存在循环跳转" | -| 下载内容为空 | 返回错误 "下载内容为空,请检查 URL 是否正确" | -| 安装命令失败 | 返回错误 "安装失败,临时文件保留在 {path},可尝试手动安装" | - -### 第三步:名称搜索安装流程 - -#### 3.1 Cocoloop API 搜索 - -请求: -``` -GET https://api.cocoloop.cn/search={encoded_query} -``` - -成功响应示例: -```json -{ - "results": [ - { - "name": "pdf-processor", - "description": "PDF processing and manipulation skill", - "url": "https://skills.cocoloop.cn/pdf-processor/v1.0.0.skill", - "version": "1.0.0", - "author": "cocoloop-team", - "downloads": 1500, - "rating": "S" - } - ], - "total": 1 -} -``` - -处理: -- 如果 results.length > 0 → 展示结果,询问用户选择 -- 如果 results.length = 0 或 API 失败 → 进入 3.2 - -#### 3.2 clawhub Fallback - -执行: -```bash -POST /admin/api/skills/market/install { "slug": "{skill_name}" } -``` - -处理: -- 成功 → 完成安装 -- 失败(退出码非0)→ 进入 3.3 - -#### 3.3 GitHub API 搜索 - -请求: -``` -GET https://api.github.com/search/repositories?q={query}+filename:SKILL.md&sort=stars&order=desc -``` - -Headers: -``` -User-Agent: Cocoloop-Skill-Manager/1.0 -``` - -成功响应处理: -```javascript -results = data.items - .filter(repo => repo.name.includes(query) || repo.description?.includes(query)) - .map(repo => ({ - name: repo.name, - fullName: repo.full_name, - description: repo.description, - url: repo.html_url, - stars: repo.stargazers_count, - owner: { - name: repo.owner.login, - type: repo.owner.type // 'User' 或 'Organization' - } - })) - .slice(0, 5) // 取前5个 -``` - -展示格式: -``` -📋 GitHub 搜索结果 (找到 {total} 个): - - 1. company/pdf-processor ⭐ 1250 - 🏢 Organization | Advanced PDF processing tools - - 2. user/simple-pdf ⭐ 45 - 👤 User | Basic PDF operations - -请选择要安装的 skill (输入序号,或输入 0 取消): -``` - -用户选择后: -1. 获取仓库详情(确认存在 SKILL.md) -2. 询问用户确认安装 -3. 下载 raw SKILL.md 和相关资源 -4. 打包为 .skill 文件(如果需要) -5. 执行平台安装 - -### 第四步:GitHub 短链安装流程 - -输入格式识别: -- 包含 `/` 但不以 `http` 开头 -- 格式:`owner/repo` 或 `owner/repo/subpath` - -处理流程: -1. 解析 owner 和 repo -2. 调用 GitHub API 获取仓库信息: - ``` - GET https://api.github.com/repos/{owner}/{repo} - ``` -3. 检查是否存在 SKILL.md: - ``` - GET https://api.github.com/repos/{owner}/{repo}/contents/SKILL.md - ``` -4. 如果存在 → 展示仓库信息,询问确认 -5. 下载并安装 - -### 第五步:安全检查(可选但推荐) - -在安装前或安装后,询问用户是否进行安全检查: - -``` -⚠️ 安全提醒: 该 skill 来源为 {source_level},建议进行安全检查。 -是否进行 BSS 安全认证检查? [Y/n] -``` - -如果用户选择是: -1. 执行 [safety-check-guide.md](safety-check-guide.md) 和 [cocoloop-safe-check.md](cocoloop-safe-check.md) 中的检查流程 -2. 生成报告 -3. 如果评级 <= B,询问用户是否继续安装 - -## 安装后处理 - -安装完成后,执行: -1. 验证安装是否成功(检查安装目录) -2. 如果是更新操作,清理旧版本备份 -3. 可选:显示 skill 使用帮助 - ``` - ✅ 安装成功! - - Skill: pdf-processor - 版本: 1.0.0 - 来源: cocoloop (S级认证) - - 使用方式: - - 转换 PDF: 使用 pdf-processor 转换 xxx.pdf 为 docx - - 合并 PDF: 使用 pdf-processor 合并 a.pdf b.pdf - ``` diff --git a/runtime/skills/cocoloop/references/search-guide.md b/runtime/skills/cocoloop/references/search-guide.md deleted file mode 100644 index b6874732..00000000 --- a/runtime/skills/cocoloop/references/search-guide.md +++ /dev/null @@ -1,254 +0,0 @@ -# Skill 搜索流程详细指南 - -本文档详细描述 Cocoloop 的多源搜索机制。 - -## 搜索源优先级 - -1. **Cocoloop API** - 官方技能仓库(优先) -2. **GitHub API** - 开源社区(fallback) -3. **本地缓存** - 已下载的 skill 信息(辅助) - -## Cocoloop API 搜索 - -### 请求格式 - -``` -GET https://api.cocoloop.cn/search={encoded_query} -``` - -### 请求头 - -``` -User-Agent: Cocoloop-Skill-Manager/1.0 -Accept: application/json -``` - -### 响应格式 - -```json -{ - "results": [ - { - "name": "skill-name", - "displayName": "Skill Display Name", - "description": "Skill description", - "url": "https://skills.cocoloop.cn/skill-name/v1.0.0.skill", - "version": "1.0.0", - "author": "author-name", - "authorUrl": "https://github.com/author", - "license": "MIT", - "downloads": 1500, - "rating": "S", - "tags": ["pdf", "document"], - "updatedAt": "2024-01-15T10:30:00Z" - } - ], - "total": 10, - "page": 1, - "perPage": 20 -} -``` - -### 处理逻辑 - -1. 发送请求 -2. 解析 JSON 响应 -3. 过滤结果(匹配度排序) -4. 返回前 10 个结果 - -## GitHub API 搜索 - -### 请求格式 - -``` -GET https://api.github.com/search/repositories?q={query}+filename:SKILL.md&sort=stars&order=desc&per_page=10 -``` - -### 搜索查询构建 - -基础查询:`{query} filename:SKILL.md` - -可选追加: -- `+language:javascript` - 限定语言 -- `+stars:>10` - 限定 stars 数 -- `+topic:claude-skill` - 限定 topic - -### 响应处理 - -原始响应字段映射: - -```javascript -{ - name: item.name, // 仓库名 - fullName: item.full_name, // 完整名 owner/repo - description: item.description, // 描述 - url: item.html_url, // GitHub 页面 - stars: item.stargazers_count, // stars 数 - forks: item.forks_count, // forks 数 - language: item.language, // 主要语言 - updatedAt: item.updated_at, // 更新时间 - owner: { - name: item.owner.login, // 所有者名 - type: item.owner.type, // 'User' 或 'Organization' - avatar: item.owner.avatar_url // 头像 URL - }, - license: item.license?.name, // 许可证 - topics: item.topics // 标签数组 -} -``` - -### 结果过滤与排序 - -过滤条件: -1. 仓库名或描述包含查询词 -2. 不是 fork 的仓库(可选) -3. 最近 2 年有更新(可选) - -排序规则: -1. 组织账号优先于个人账号 -2. stars 数高优先 -3. 最近更新优先 - -### 展示格式 - -``` -🐙 GitHub 搜索结果 (按 stars 排序): - - 1. company/skill-name ⭐ 1.2k - 🏢 Organization | MIT License - 📄 PDF processing and manipulation tools - 🏷️ pdf, document, converter - - 2. user/another-skill ⭐ 45 - 👤 User | Apache-2.0 - 📄 Simple PDF utilities - 🏷️ pdf, utils - - 3. ... -``` - -## 综合搜索流程 - -当用户搜索时,执行以下流程: - -``` -并行执行: -├── Cocoloop API 搜索 ──────→ 结果 A -└── GitHub API 搜索 ────────→ 结果 B - -合并结果: -1. 优先展示 Cocoloop 结果(官方源) -2. 然后展示 GitHub 结果(社区源) -3. 去重(相同 fullName 只保留一个) - -展示: -- 最多展示 10 个结果(可配置) -- 标注来源(🌟 Cocoloop / 🐙 GitHub) -- 显示关键信息(名称、描述、stars、来源类型) -``` - -## 获取 Skill 详情 - -当用户选择某个 skill 后,获取详细信息: - -### 对于 Cocoloop 源 - -直接读取 API 返回的完整信息。 - -### 对于 GitHub 源 - -1. **获取仓库详情** - ``` - GET https://api.github.com/repos/{owner}/{repo} - ``` - -2. **获取 SKILL.md 内容** - ``` - GET https://api.github.com/repos/{owner}/{repo}/contents/SKILL.md - ``` - 响应中的 `content` 字段是 base64 编码的,需要解码。 - -3. **解析 SKILL.md** - - 提取 frontmatter(name, description) - - 提取前 500 字作为预览 - -4. **获取最新 release(可选)** - ``` - GET https://api.github.com/repos/{owner}/{repo}/releases/latest - ``` - -### 详情展示格式 - -``` -📋 Skill 详情 - -名称: pdf-processor -版本: 1.0.0 -来源: 🐙 GitHub (Organization) -⭐ Stars: 1250 | 🍴 Forks: 45 -📄 许可证: MIT -🏷️ 标签: pdf, document, converter - -描述: -Advanced PDF processing and manipulation tools. Supports conversion, -merging, splitting, and encryption. - -SKILL.md 预览: ---- -name: pdf-processor -description: PDF processing skill... ---- -# PDF Processor -This skill provides tools for working with PDF files... - -来源可信度: T2 (可信组织) -安全评级: 待检查 - -是否安装此 skill? [Y/n] -``` - -## 本地缓存搜索 - -为了提高重复搜索的速度,维护本地缓存: - -### 缓存位置 - -`~/.cocoloop/cache/search.json` - -### 缓存格式 - -```json -{ - "query": "pdf", - "timestamp": "2024-01-15T10:30:00Z", - "results": [...], - "expires": "2024-01-16T10:30:00Z" -} -``` - -### 缓存策略 - -- 缓存有效期:24 小时 -- 命中缓存时,询问用户是否使用缓存结果 -- 提供 `--fresh` 或 `-f` 参数强制刷新 - -## 错误处理 - -| 错误场景 | 处理方式 | -|---------|---------| -| Cocoloop API 超时 | 自动 fallback 到 GitHub | -| GitHub API 限流 | 提示用户稍后重试,或使用本地缓存 | -| 网络错误 | 显示错误信息,建议使用离线模式(如果有缓存)| -| 解析错误 | 记录日志,跳过该结果,继续其他 | - -## 高级搜索语法 - -支持以下搜索修饰符: - -| 修饰符 | 含义 | 示例 | -|-------|------|------| -| `author:` | 限定作者 | `author:anthropic pdf` | -| `lang:` | 限定语言 | `lang:javascript tool` | -| `stars:>n` | stars 数大于 | `stars:>100 utility` | -| `source:cocoloop` | 仅官方源 | `source:cocoloop document` | -| `source:github` | 仅 GitHub | `source:github utility` | diff --git a/runtime/skills/cocoloop/references/uninstall-guide.md b/runtime/skills/cocoloop/references/uninstall-guide.md deleted file mode 100644 index 4cb3e5d9..00000000 --- a/runtime/skills/cocoloop/references/uninstall-guide.md +++ /dev/null @@ -1,163 +0,0 @@ -# Skill 卸载流程详细指南 - -本文档详细描述 skill 的卸载流程。 - -## 卸载前准备 - -### 1. 检测平台 - -使用与安装相同的平台检测逻辑: - -``` -IF OpenClaw: - 安装目录 = ~/.openclaw/skills/ - 配置文件 = ~/.openclaw/config.json - -ELSE IF Molili: - 安装目录 = ~/.molili/skills/ - 配置文件 = ~/.molili/config.json - -ELSE IF Claude Code: - 安装目录 = ~/.claude/skills/ - 配置文件 = ~/.claude/config.json - -ELSE: - 安装目录 = ~/.claude/skills/ (clawhub 默认) - 配置文件 = ~/.claude/config.json -``` - -### 2. 确认 Skill 存在 - -检查 skill 目录是否存在: - -``` -{安装目录}/{skill-name}/ - ├── SKILL.md - ├── scripts/ - ├── references/ - └── assets/ -``` - -如果不存在: -- 返回错误 "未找到该 skill,可能已卸载或名称错误" -- 建议用户使用 `list` 命令查看已安装 skills - -### 3. 获取 Skill 信息 - -读取 SKILL.md 获取基本信息: -- name -- description -- version(如果有) - -## 卸载流程 - -### 第一步:用户确认 - -展示将要卸载的 skill 信息,请求确认: - -``` -⚠️ 即将卸载以下 skill: - -名称: pdf-processor -描述: PDF processing and manipulation tools -安装路径: ~/.claude/skills/pdf-processor/ - -⚠️ 此操作将删除该 skill 的所有文件,不可恢复。 - -是否确认卸载? [y/N] -``` - -可选:添加 `--force` 或 `-f` 参数跳过确认。 - -### 第二步:备份(可选) - -如果用户指定 `--backup` 或 `-b` 参数: - -1. 创建备份目录:`~/.cocoloop/backups/` -2. 打包 skill 目录:`tar -czf ~/.cocoloop/backups/{skill-name}-{timestamp}.tar.gz {skill-path}/` -3. 提示备份位置 - -### 第三步:执行卸载 - -1. **删除 skill 目录** - ```bash - rm -rf {安装目录}/{skill-name}/ - ``` - -2. **更新平台配置(如果需要)** - - 某些平台维护已安装 skill 列表 - - 从列表中移除该 skill - -3. **清理相关缓存** - - 删除 Cocoloop 本地缓存中该 skill 的搜索记录 - - 删除安全检查缓存(如果有) - -### 第四步:验证卸载 - -检查 skill 目录是否还存在: -- 如果存在 → 返回错误 "卸载失败,请检查权限或手动删除" -- 如果不存在 → 卸载成功 - -## 批量卸载 - -支持一次卸载多个 skills: - -``` -卸载 skill1 skill2 skill3 -``` - -处理流程: -1. 遍历每个 skill -2. 执行单个卸载流程(不询问确认,或统一确认) -3. 汇总结果: - ``` - 📊 卸载结果: - skill1: ✅ 已卸载 - skill2: ❌ 未找到 - skill3: ✅ 已卸载 - ``` - -## 卸载后处理 - -### 依赖检查(可选) - -检查是否有其他 skill 依赖被卸载的 skill: -1. 遍历所有已安装 skills -2. 检查它们的 dependencies(如果有记录) -3. 如果有依赖关系,警告用户: - ``` - ⚠️ 警告: 以下 skill 可能依赖 pdf-processor: - - document-workflow - - 继续使用这些 skill 可能会出现问题。 - ``` - -### 清理孤立依赖(高级) - -如果 skill 安装了独立的依赖(如 node_modules),检查是否可以清理: -- 如果其他 skill 不使用 → 可以删除 -- 如果有共享依赖 → 保留 - -## 错误处理 - -| 错误场景 | 处理方式 | -|---------|---------| -| 权限不足 | 提示使用 `sudo` 或检查目录权限 | -| 文件被占用 | 提示关闭使用该 skill 的程序后重试 | -| 目录非空但无法删除 | 保留日志,提示手动删除 | -| 配置文件损坏 | 尝试修复或重建配置 | - -## 恢复卸载 - -如果用户误卸载,提供恢复选项(前提是备份存在): - -``` -恢复 pdf-processor -``` - -流程: -1. 查找备份目录:`~/.cocoloop/backups/pdf-processor-*.tar.gz` -2. 列出可用备份(按时间排序) -3. 询问用户选择恢复哪个版本 -4. 解压到安装目录 -5. 验证恢复 diff --git a/runtime/skills/data-analyst/SKILL.md b/runtime/skills/data-analyst/SKILL.md new file mode 100644 index 00000000..db85e53c --- /dev/null +++ b/runtime/skills/data-analyst/SKILL.md @@ -0,0 +1,21 @@ +--- +name: data-analyst +description: Complete the data analysis tasks delegated by the user.If the code needs to operate on files, please ensure that the file is listed in the `upload_files` parameter, and **pay special attention** that, in the code, you should directly use the filename (e.g., `open('data.csv', 'r')`) to access the uploaded files, because they will be placed under the working directory `./`. +--- + +# Data Analyst + +## Overview + +This skill provides specialized capabilities for data analyst. + +## Instructions + +Complete the data analysis tasks delegated by the user.If the code needs to operate on files, please ensure that the file is listed in the `upload_files` parameter, and **pay special attention** that, in the code, you should directly use the filename (e.g., `open('data.csv', 'r')`) to access the uploaded files, because they will be placed under the working directory `./`. + + +## Usage Notes + +- This skill is based on the data_analyst agent configuration +- Template variables (if any) like $DATE$, $SESSION_GROUP_ID$ may require runtime substitution +- Follow the instructions and guidelines provided in the content above diff --git a/runtime/skills/data-analyst/_meta.json b/runtime/skills/data-analyst/_meta.json new file mode 100644 index 00000000..bd29be5c --- /dev/null +++ b/runtime/skills/data-analyst/_meta.json @@ -0,0 +1,6 @@ +{ + "ownerId": "kn71xkrq2fawjvteej73gsx71s80p3kb", + "slug": "data-analyst-pro", + "version": "0.1.0", + "publishedAt": 1771141367019 +} \ No newline at end of file diff --git a/runtime/skills/oclaw-skill-manager/README.md b/runtime/skills/oclaw-skill-manager/README.md new file mode 100644 index 00000000..f1b5ccb7 --- /dev/null +++ b/runtime/skills/oclaw-skill-manager/README.md @@ -0,0 +1,7 @@ +# oclaw-skill-manager + +Oclaw **内置** Skill:说明如何在当前仓库中安装、更新、卸载技能,以及依赖与健康检查。 + +- 主文档:[SKILL.md](SKILL.md) +- 本包**不是**任何外部「技能市场 CLI」的封装;平台不提供官方 shell 一键安装命令。 +- 模型安装策略:仅允许 `skill_auto_install`;失败时仅报告 `error_code/detail`,禁止绕路安装。 diff --git a/runtime/skills/oclaw-skill-manager/SKILL.md b/runtime/skills/oclaw-skill-manager/SKILL.md new file mode 100644 index 00000000..bfc19726 --- /dev/null +++ b/runtime/skills/oclaw-skill-manager/SKILL.md @@ -0,0 +1,87 @@ +--- +name: oclaw-skill-manager +description: Oclaw 官方 Skill 生命周期手册:安装、更新、卸载、依赖与健康检查。仅使用本仓库 Admin API、安装器与内置工具;不提供也不推荐任何 shell「一键装技能」命令。 +--- + +# Oclaw Skill 管理器(官方手册) + +本技能是 **Oclaw 自有** 的 Skill 运维说明,用于指导模型与运维在**同一套契约**下管理 `runtime/skills` 下的技能包。 + +## 系统强约束(模型必须遵守) + +- 规范优先:安装决策只依据本文件,不依据临时推断或“经验性排障说法”。 +- 路径约束:当任务目标是“为当前会话/用户安装 skill”时,**只允许**走 `skill_auto_install`(`_workspace` lane)。 +- 禁止绕路:不得调用任何非 auto 安装路径(包括但不限于 `skill_market_install`、`skill_registry_install`、Admin 安装 API、手写下载/解压脚本落盘到 `runtime/skills`)。 +- 失败处理:`skill_auto_install` 失败后,**不得继续安装**;只报告 `error_code/detail` 与最小下一步,等待用户指示。 +- 禁止臆测前置条件:未在本规范显式声明的环境变量/端口/服务状态,不得被表述为“安装必经条件”。 + +## 平台事实(无「原生安装命令」) + +- Oclaw **不提供**任何官方 shell「一键装技能」命令(包括市场 CLI、`npx …` 拉 CLI 再 `install` 等模式)。 +- 模型安装策略:默认仅使用 **`skill_auto_install`**;其他安装能力仅供管理员/后端运维链路使用。 +- 在沙箱里执行 `run_command` 时,**外部技能 CLI 安装模式会被拦截**(见 `shell_tools`),请改用下文 API。 + +## 目录策略 + +| 场景 | 路径 | +|------|------| +| 人工 / Admin 市场或 registry 安装 | `//` | +| 智能体 payload 自动安装 | `/_workspace//` | + +`` 默认 `runtime/skills/`,可被 **`AIA_SKILLS_ROOT`** 覆盖。 + +## 技能市场提供方(ClawHub + CocoLoop) + +租户设置 **`AIA_SKILL_MARKET_PROVIDER`** 选择市场(网关 `get_market_adapter` 读取): + +| 取值 | 说明 | +|------|------| +| **`clawhub`**(默认) | [ClawHub](https://clawhub.ai) 公开技能注册表;HTTP 形态与官方 CLI 一致,见上游文档 [CLI / Registry](https://github.com/openclaw/clawhub/blob/main/docs/cli.md)(`/api/v1/search`、`/api/v1/skills/{slug}`、`/api/v1/download?slug=&version=`)。本仓库客户端:`runtime/tools/skills/clawhub_client.py`,环境变量 **`AIA_CLAWHUB_SITE` / `AIA_CLAWHUB_REGISTRY` / `AIA_CLAWHUB_TOKEN`**(或 `CLAWHUB_*`)与官方 `CLAWHUB_*` 对齐。 | +| **`cocoloop`** | [CocoLoop 技能商店](https://hub.cocoloop.cn) 开放列表接口:`GET {api}/api/v1/store/skills`(分页、`keyword`、`sort`),详情:`GET {api}/api/v1/store/skills/{id}`;列表项中的 **`download_url`** 为 zip 直链(常见域名 `dl.cocoloop.cn`)。实现:`runtime/tools/skills/cocoloop_client.py`;可选 **`AIA_COCOLOOP_API_BASE`**(默认 `https://api.cocoloop.com`)。别名:`cocoloop-cn`、`cocoloop_cn` 与 `cocoloop` 相同。 | + +安装仍统一走 **`install_skill_from_registry_archive`**:对 ClawHub 与 CocoLoop 均为 **HTTPS zip 归档 URL**,无需在服务器上安装 `clawhub` / `cocoloop` CLI。 + +## 发现与安装(模型视角) + +### 唯一安装路径(必须) + +- **`skill_auto_install`**:仅写入 `_workspace` lane(见 `skill_installer.auto_install_skill_from_payload`)。 +- **非前置条件澄清**:`AIA_INTERNAL_BASE_URL`、Admin `market/search`、本地 5173 服务都不是 `skill_auto_install` 的必需前置。 +- 若安装失败,向用户返回: + - `error_code` + - `detail` + - 建议下一步(例如补充 name/description、检查依赖、重试) +- 不允许改走其它安装入口完成同一目标。 + +## 列表、开关、卸载 + +- **`skill_list`**(若已绑定到当前专家) +- `GET /admin/api/skills` +- `POST /admin/api/skills/enable` / `disable`,`{ "name": "..." }` +- `POST /admin/api/skills/uninstall`,`{ "name": "..." }` + +## 依赖与健康 + +- 安装后:`requirements.txt`、`package.json`(`dependencies`)、Python import 探测与补装(受 `AIA_SKILL_AUTO_INSTALL_DEPS_ENABLED` 控制)。 +- Admin:**Repair deps** / **Repair all deps** +- `GET /admin/api/skills/self-check?include_execution=...` + +## 更新 + +模型侧无独立 update 安装路径;如需更新,按用户指示走新的 `skill_auto_install` 版本草案或转人工管理员操作。 + +## 可选人工安全审查 + +- [references/safety-check-guide.md](references/safety-check-guide.md) +- [references/skill-safety-rubric.md](references/skill-safety-rubric.md) + 均为**人工参考**,平台不自动执行远程认证。 + +## 子文档 + +- [references/install-guide.md](references/install-guide.md) +- [references/search-guide.md](references/search-guide.md) +- [references/uninstall-guide.md](references/uninstall-guide.md) + +--- + +维护本技能时,请只增删 **Oclaw 已实现** 的行为,勿再引入第三方「技能 CLI」作为默认路径。 diff --git a/runtime/skills/oclaw-skill-manager/references/install-guide.md b/runtime/skills/oclaw-skill-manager/references/install-guide.md new file mode 100644 index 00000000..26868d1a --- /dev/null +++ b/runtime/skills/oclaw-skill-manager/references/install-guide.md @@ -0,0 +1,53 @@ +# Oclaw — Skill 安装详细指南 + +本文档仅描述 **Oclaw** 内的安装行为,替代旧版「多平台检测 + Cocoloop API」流程。 + +## 模型执行硬规则 + +- 仅允许 `skill_auto_install`。 +- 安装失败时只允许“报告失败原因给用户”,禁止改走其它安装入口(market/install、install-registry、install、本地脚本解压落盘等)。 +- 不得把未在规范中声明的环境变量/端口/服务状态当作安装前置条件。 + +## 1. 技能根目录 + +- 默认:`runtime/skills/`(或环境变量 `AIA_SKILLS_ROOT` 指向的目录) +- **主目录**:`//` — Admin 市场 / registry / 本地目录安装默认落点 +- **智能体自写目录**:`/_workspace//` — `skill_auto_install` / `auto_install_skill_from_payload` 等 + +具体以安装接口返回的 `target_dir` 为准。 + +## 2. 安装入口对照(运维参考) + +| 场景 | 方式 | HTTP(Admin) | +|------|------|----------------| +| ClawHub slug | 解析 `archive_url` 后安装 | `POST /admin/api/skills/market/install` | +| 已知归档 URL | 直接拉取 zip/tar | `POST /admin/api/skills/install-registry` | +| 本地已展开目录 | 目录内含 `SKILL.md` | `POST /admin/api/skills/install` | +| 从模板创建 | 生成新包 | `POST /admin/api/skills/create` | +| Workspace 模板 | 带 runtime 桩 | `POST /admin/api/skills/create-workspace` | + +认证:Admin 路由需带网关要求的 `Authorization`(与现有 Admin 一致)。 + +## 3. 依赖与自检 + +安装成功后,安装器会尽量: + +1. 处理 `requirements.txt`、`package.json`(`dependencies` 非空) +2. 扫描 `.py` 的 import,对缺失的第三方模块尝试 `pip install` + +失败不一定会回滚整个目录,可能返回带 `installed_with_dependency_warnings` 的 `detail`。此时在 Admin 使用 **Repair deps** 或 **Repair all deps**。 + +## 4. 重试与覆盖 + +- 安装失败审计里若带 `retryable`,可用 `POST /admin/api/skills/retry-install`(见 Admin 实现) +- 覆盖安装:`overwrite: true` + +## 5. 禁止项 + +- **不要**使用任何「第三方技能 CLI + install」作为安装路径(`run_command` 会拦截常见模式) +- **不要**把非本仓库契约的 HTTP 商店当作主源;技能发现以 **ClawHub 市场适配器**(`AIA_SKILL_MARKET_PROVIDER`)为准 +- **模型侧不要**在 `skill_auto_install` 失败后切换到 Admin 安装 API 或脚本直装。 + +## 6. slug 与包名 + +ClawHub 返回的 **slug** 可能与解压后 `SKILL.md` frontmatter 里的 **name** 不同。卸载、启用、绑定角色时以 **`skill_list` / API 返回的 `name`** 为准。 diff --git a/runtime/skills/cocoloop/references/safety-check-guide.md b/runtime/skills/oclaw-skill-manager/references/safety-check-guide.md similarity index 95% rename from runtime/skills/cocoloop/references/safety-check-guide.md rename to runtime/skills/oclaw-skill-manager/references/safety-check-guide.md index 1a2b84c8..281bd4a4 100644 --- a/runtime/skills/cocoloop/references/safety-check-guide.md +++ b/runtime/skills/oclaw-skill-manager/references/safety-check-guide.md @@ -1,6 +1,8 @@ -# Cocoloop Safe Check 安全检查流程指南 +> **Oclaw 说明**:本文件为**可选的人工安全审查**参考流程。Oclaw **不会**自动调用 Cocoloop BSS 或远程「Safe Check」服务;安装与运行以平台自带沙箱、工具策略与租户配置为准。执行审查前请确认符合本地合规要求。 -本文档详细描述 Cocoloop 安全检查的执行流程,基于 cocoloop-safe-check 安全认证体系。 +# Skill 安全检查流程指南(参考) + +本文档描述一套**可参考**的人工安全检查执行流程,可与 [skill-safety-rubric.md](skill-safety-rubric.md) 配合阅读。Oclaw 不保证与任何第三方「安全认证产品」行为一致。 ## 检查触发时机 diff --git a/runtime/skills/oclaw-skill-manager/references/search-guide.md b/runtime/skills/oclaw-skill-manager/references/search-guide.md new file mode 100644 index 00000000..6a82cfe0 --- /dev/null +++ b/runtime/skills/oclaw-skill-manager/references/search-guide.md @@ -0,0 +1,41 @@ +# Oclaw — Skill 搜索指南 + +## 市场提供方 + +由设置 **`AIA_SKILL_MARKET_PROVIDER`** 决定:`clawhub`(默认)或 `cocoloop`(见主 `SKILL.md` 对照表)。Admin 的 `market/search` 与 `market/detail` 会调用当前提供方适配器。 + +## 主源:ClawHub + +当 `AIA_SKILL_MARKET_PROVIDER=clawhub` 时使用。 + +### Admin HTTP(ClawHub 模式下) + +- 搜索:`GET /admin/api/skills/market/search?q=<关键词>&limit=` +- 详情:`GET /admin/api/skills/market/detail?slug=` + +从详情中读取:`slug`、`version`、描述、以及安装所需的 **`archiveUrl`**(ClawHub 下载链)。 + +## 主源:CocoLoop 商店 + +当 `AIA_SKILL_MARKET_PROVIDER=cocoloop` 时,同一组 Admin 路由背后走 **`cocoloop_client`**:关键词搜索商店列表,按技能 **`name` 字段** 匹配 slug;详情中的安装 URL 来自列表 **`download_url`**(或按 `asset_name` 拼 zip 直链)。商店前端:[hub.cocoloop.cn](https://hub.cocoloop.cn)。 + +### 模型侧 + +若无 Admin 权限,请用户代为搜索/安装,或提供准确 **slug** / **archive_url**。 + +> 重要:市场搜索不是安装前置条件。 +> 模型安装策略下只允许 `skill_auto_install`。若没有可用市场结果,应向用户索取可安装内容(如技能描述、`SKILL.md` 或源文件)并走 `skill_auto_install`,而不是要求先配置 `AIA_INTERNAL_BASE_URL` 或先启动本地 5173 服务。 + +## 辅助源:GitHub(可选) + +当市场无结果或用户指定开源仓库时: + +``` +GET https://api.github.com/search/repositories?q=<关键词>+filename:SKILL.md&sort=stars&order=desc +``` + +需自备 `User-Agent`,注意 API 速率限制。找到仓库后仍需**可安装的归档 URL** 再走 `install-registry`。 + +## 合并展示建议 + +向用户展示时标注来源:`[ClawHub]` / `[GitHub]`,并给出 **slug** 或 **full_name**。 diff --git a/runtime/skills/cocoloop/references/cocoloop-safe-check.md b/runtime/skills/oclaw-skill-manager/references/skill-safety-rubric.md similarity index 92% rename from runtime/skills/cocoloop/references/cocoloop-safe-check.md rename to runtime/skills/oclaw-skill-manager/references/skill-safety-rubric.md index d14a4279..a1c7e4b8 100644 --- a/runtime/skills/cocoloop/references/cocoloop-safe-check.md +++ b/runtime/skills/oclaw-skill-manager/references/skill-safety-rubric.md @@ -1,6 +1,8 @@ -# Cocoloop Safe Check 安全检查标准 +> **Oclaw 说明**:以下为**评级与检查维度参考**,供人工审阅 skill 时使用。Oclaw **不**根据本文件自动打分或拦截安装;实际风险控制依赖权限、审计、工具白名单与运行环境隔离。 -本文件定义了 Cocoloop Skill 管理器的安全检查标准。 +# Skill 安全审查量表(参考) + +本文件提供人工审阅时可用的检查维度与分级思路,**不**作为 Oclaw 运行时强制契约。 ## 评级标准 diff --git a/runtime/skills/oclaw-skill-manager/references/uninstall-guide.md b/runtime/skills/oclaw-skill-manager/references/uninstall-guide.md new file mode 100644 index 00000000..3ad64723 --- /dev/null +++ b/runtime/skills/oclaw-skill-manager/references/uninstall-guide.md @@ -0,0 +1,32 @@ +# Oclaw — Skill 卸载指南 + +## 路径 + +卸载逻辑会依次查找(存在则删除): + +1. `//` +2. `/_workspace//` + +其中 `` 为 `AIA_SKILLS_ROOT` 或默认 `runtime/skills/`。 + +`` 为 **`SKILL.md` frontmatter 中的 `name`**(与 `skill_list` 中 `name` 字段一致),不一定等于 ClawHub **slug**。 + +## Admin + +- `POST /admin/api/skills/uninstall` + Body:`{ "name": "" }` + +卸载前应在 UI 或对话中向用户确认;删除后不可恢复(除非有外部备份)。 + +## 启用状态 + +卸载实现中会尝试将技能从禁用列表恢复为可用状态(见 `skill_installer.uninstall_skill`);若需保留禁用记录,请在产品中另行约定(当前以代码为准)。 + +## 批量卸载 + +对每个名称依次调用卸载接口,独立汇总结果。 + +## 与旧版差异 + +- ~~`rm -rf ~/.openclaw/skills/`~~ 等路径不适用于本仓库默认布局 +- 不使用任何外部「技能卸载 CLI」;一律走 Admin `uninstall` 或安装器 API diff --git a/runtime/skills/skillscan/SKILL.md b/runtime/skills/skillscan/SKILL.md new file mode 100644 index 00000000..9ab299ba --- /dev/null +++ b/runtime/skills/skillscan/SKILL.md @@ -0,0 +1,97 @@ +--- +name: SkillScan +metadata: + version: "1.1.6" +description: > + Security gate for skills. Every new skill MUST pass SkillScan before use. + Activate on any install, load, add, evaluate, or safety question about a skill. + On first load, run first-run to scan all existing skills. + Blocks HIGH/CRITICAL skills. No exceptions. +--- + +# SkillScan v1.1.6 + +Security check for skill packages. Every new skill must pass this scan before use. + +Implemented in `scripts/scanner.py`. Supports Windows / macOS / Linux. + +Do NOT generate report files inside the skill directory. + +--- + +## When to Activate + +Run SkillScan whenever a skill is installed, loaded, added, evaluated, or questioned for safety — regardless of method or phrasing. + +Trigger examples: install, load, add, use, set up, "is this safe", upload .zip/.skill, paste SKILL.md, mention unknown skill name, install from hub/store. + +--- + +## First Load + +Check `scripts/.first_run_done`: + +- **Not found** → **Tell user** SkillScan is installed, ask to scan all existing skills: + ```bash + python /scripts/scanner.py first-run + ``` + `.first_run_done` is created automatically after completion. + +- **Found** → Skip, operate normally. + +--- + +## Commands + +| Command | Usage | +|---------|-------| +| `scanner.py scan ` | Scan a single skill (.zip or directory) | +| `scanner.py scan-all` | Scan all installed skills | +| `scanner.py first-run` | First-time full scan | +| `scanner.py upgrade` | Manual upgrade | + +--- + +## Scan Rules + +- **.zip files** → Scan BEFORE installation. Block if fails. +- **Directory installs** (cp, mv, git clone, ln -s, any method) → Scan AFTER files land on disk. +- **Remote installs** (clawhub, skillhub, npx skills add, etc.) → Scan immediately after install. +- **Unknown skills** → If user mentions a skill you haven't seen, scan it. + +--- + +## Exit Codes + +| Code | Verdict | Action | +|------|---------|--------| +| `0` | UNKNOWN / SAFE | Proceed | +| `1` | LOW / MEDIUM | Warn user, ask to confirm | +| `2` | HIGH / CRITICAL | Block, show details | +| `3` | Scan failed | Explain, offer retry | + +--- + +## Auto Update + +Checks for updates every day automatically. Silent, no user action needed. Manual: `scanner.py upgrade`. + +--- + +## Environment Variables + +| Variable | Description | +|----------|-------------| +| `SKILL_SCANNER_UPDATE_URL` | Custom update source (optional) | + +--- + +## API + +Base URL: `https://skillscan.tokauth.com` + +| Step | Method | Path | +|------|--------|------| +| ① Cache lookup | GET | `/oapi/v1/skill-scan/search?dir_sha256=` | +| ② Upload | POST | `/oapi/v1/skill-scan/upload` | +| ③ Poll result | GET | `/oapi/v1/skill-scan/result?task_no=` (poll every 20s, max 180s) | diff --git a/runtime/skills/skillscan/_meta.json b/runtime/skills/skillscan/_meta.json new file mode 100644 index 00000000..e9cce2eb --- /dev/null +++ b/runtime/skills/skillscan/_meta.json @@ -0,0 +1,6 @@ +{ + "ownerId": "kn791cyx98pcsezkh5088g8jxn84c7mm", + "slug": "skillscan", + "version": "1.1.6", + "publishedAt": 1776650587310 +} \ No newline at end of file diff --git a/runtime/skills/skillscan/scripts/scanner.py b/runtime/skills/skillscan/scripts/scanner.py new file mode 100644 index 00000000..5579be50 --- /dev/null +++ b/runtime/skills/skillscan/scripts/scanner.py @@ -0,0 +1,959 @@ +#!/usr/bin/env python3 +""" +SkillScan v1.1.5 — OpenClaw Skill security scanner. +Supports Windows / macOS / Linux. All temp files use the standard tempfile module. + +Usage (invoked by the agent via bash): + python scanner.py first-run # First install: list installed skills and ask to scan + python scanner.py scan # Scan a single skill (.zip or directory) + python scanner.py scan-all # Scan all installed skills + python scanner.py upgrade # Auto-upgrade +""" + +import sys, os, json, time, zipfile, hashlib, shutil, tempfile, uuid, platform, base64 +import urllib.request, urllib.error, urllib.parse +from pathlib import Path +from datetime import datetime, timezone + +# ───────────────────────────────────────────────────────────────────────────── +# Configuration +# ───────────────────────────────────────────────────────────────────────────── + +SCANNER_VERSION = "1.1.5" + +BASE_URL = "https://skillscan.tokauth.com" +API_SEARCH = f"{BASE_URL}/oapi/v1/skill-scan/search" +API_UPLOAD = f"{BASE_URL}/oapi/v1/skill-scan/upload" +API_RESULT = f"{BASE_URL}/oapi/v1/skill-scan/result" +UPDATE_URL = os.environ.get("SKILL_SCANNER_UPDATE_URL", + f"{BASE_URL}/downloads/SkillScan/manifest") + +POLL_INTERVAL = 20 # Poll interval (seconds) +POLL_TIMEOUT = 180 # Max wait time (seconds) + +# First-run marker file (in the same directory as scanner.py) +STATE_FILE = Path(__file__).parent / ".first_run_done" + +# Auto-update check marker file and interval (7 days) +LAST_UPDATE_CHECK_FILE = Path(__file__).parent / ".last_update_check" +AUTO_UPDATE_INTERVAL = 1 * 24 * 3600 # 1 day (seconds) + +# Client info file (generated on first run, reused afterwards) +CLIENT_INFO_FILE = Path(__file__).parent / ".client_info" + +# Files and directories to skip during scanning, hashing, and packing +SKIP_FILES = {".first_run_done", ".last_update_check", ".client_info", "cloud_report.json", ".DS_Store"} +SKIP_DIRS = {".git", "__pycache__", ".venv", "node_modules", ".idea", ".vscode", ".clawhub"} + +# Resolve the root directory of SkillScan itself (parent of scripts/) +SELF_ROOT = Path(__file__).parent.parent.resolve() + + +# Skill installation paths (cross-platform) +def skill_install_paths(): + # type: () -> list + """Auto-enumerate OpenClaw and local skill paths across platforms.""" + home = Path.home() + oc_dir = home / ".openclaw" + candidates = [ + # OpenClaw standard paths + oc_dir / "skills", + oc_dir / "workspace/skills", + # Shared agent skill paths + home / ".agents/skills", + home / ".config/agents/skills", + # Agent-specific global paths + home / ".gemini/antigravity/skills", + home / ".gemini/skills", + home / ".augment/skills", + home / ".claude/skills", + home / ".codex/skills", + home / ".commandcode/skills", + home / ".continue/skills", + home / ".snowflake/cortex/skills", + home / ".config/crush/skills", + home / ".cursor/skills", + home / ".deepagents/agent/skills", + home / ".factory/skills", + home / ".firebender/skills", + home / ".copilot/skills", + home / ".config/goose/skills", + home / ".junie/skills", + home / ".iflow/skills", + home / ".kilocode/skills", + home / ".kiro/skills", + home / ".kode/skills", + home / ".mcpjam/skills", + home / ".vibe/skills", + home / ".mux/skills", + home / ".config/opencode/skills", + home / ".openhands/skills", + home / ".pi/agent/skills", + home / ".qoder/skills", + home / ".qwen/skills", + home / ".roo/skills", + home / ".trae/skills", + home / ".trae-cn/skills", + home / ".codeium/windsurf/skills", + home / ".zencoder/skills", + home / ".neovate/skills", + home / ".pochi/skills", + home / ".adal/skills", + home / ".npm-global/lib/node_modules/openclaw/skills", + # Container default paths + Path("/mnt/skills/public"), + Path("/mnt/skills/private"), + Path("/mnt/skills/user"), + # User dev/download paths + home / "Downloads/skills", + ] + + # Windows-specific paths + if os.name == "nt": + appdata = os.environ.get("APPDATA") + if appdata: + candidates.append(Path(appdata) / "OpenClaw/skills") + candidates.append(Path(appdata) / "Programs/LobsterAI/resources/SKILLs") + + # Dynamically scan extensions: .openclaw/extensions/{xxxx}/skills + if oc_dir.exists(): + ext_root = oc_dir / "extensions" + if ext_root.exists(): + for sub in ext_root.iterdir(): + if sub.is_dir(): + s_dir = sub / "skills" + if s_dir.exists(): + candidates.append(s_dir) + + # Include script run path and workspace + candidates.append(Path.cwd() / "skills") + candidates.append(Path(__file__).parent.parent / "skills") + + # Deduplicate and filter non-existent paths + seen = set() + result = [] + for p in candidates: + try: + abs_p = p.resolve() + if abs_p.exists() and abs_p not in seen: + result.append(p) + seen.add(abs_p) + except Exception: + continue + return result + +RISK_EMOJI = {"SAFE":"✅","LOW":"⚠️ ","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️ "} + + +# ───────────────────────────────────────────────────────────────────────────── +# Client Info (X-Client-Info) +# ───────────────────────────────────────────────────────────────────────────── + +def _get_mac_address(): + """Try to get the MAC address; return empty string on failure.""" + try: + import uuid as _uuid + mac_int = _uuid.getnode() + # getnode() returns a random value (bit 8 set) when it can't get the real MAC + if (mac_int >> 40) & 1: + return "" + mac_str = ":".join(("%012X" % mac_int)[i:i+2] for i in range(0, 12, 2)) + return mac_str + except Exception: + return "" + + +def _build_client_info(): + """Build client info dict and persist to file; reuse on subsequent runs.""" + # If a record file already exists, read it + if CLIENT_INFO_FILE.exists(): + try: + data = json.loads(CLIENT_INFO_FILE.read_text(encoding="utf-8")) + if data.get("client_id"): + return data + except Exception: + pass + + # First run: generate new client info + info = { + "client_id": str(uuid.uuid4()), + "os": platform.system() or "", + "platform": platform.machine() or "", + "os_version": platform.release() or "", + "client": "SkillScanner/%s" % SCANNER_VERSION, + } + + mac = _get_mac_address() + if mac: + info["mac"] = mac + + # Python version as extra + info["extra"] = { + "python": platform.python_version(), + } + + # Persist + try: + CLIENT_INFO_FILE.write_text( + json.dumps(info, ensure_ascii=False, indent=2), + encoding="utf-8" + ) + except Exception: + pass + + return info + + +def _get_client_info_header(): + """Return Base64-encoded X-Client-Info header value; empty string on failure.""" + try: + info = _build_client_info() + json_str = json.dumps(info, ensure_ascii=False) + encoded = base64.b64encode(json_str.encode("utf-8")).decode("ascii") + return encoded + except Exception: + return "" + +# ───────────────────────────────────────────────────────────────────────────── +# Output Helpers +# ───────────────────────────────────────────────────────────────────────────── + +def banner(title: str): + w = 58 + print(f"\n{'═'*w}") + print(f" {title}") + print(f"{'═'*w}") + +def divider(title: str = ""): + if title: + print(f"\n ── {title} {'─'*(48-len(title))}") + else: + print(f" {'─'*52}") + +def log(msg: str): + print(f" {msg}", flush=True) + +def ask(prompt: str) -> str: + """Read user input (compatible with non-interactive environments).""" + try: + return input(f"\n {prompt} ").strip() + except (EOFError, KeyboardInterrupt): + return "" + +# ───────────────────────────────────────────────────────────────────────────── +# HTTP Helpers +# ───────────────────────────────────────────────────────────────────────────── + +def http_get(url: str) -> dict: + req = urllib.request.Request(url) + with urllib.request.urlopen(req, timeout=30) as r: + return json.loads(r.read().decode("utf-8", errors="replace")) + +def http_post(url: str, payload: dict) -> dict: + headers = {"Content-Type": "application/json"} + data = json.dumps(payload, ensure_ascii=False).encode("utf-8") + req = urllib.request.Request(url, data=data, headers=headers, method="POST") + with urllib.request.urlopen(req, timeout=60) as r: + return json.loads(r.read().decode("utf-8", errors="replace")) + + +# ───────────────────────────────────────────────────────────────────────────── +# Skill Utilities +# ───────────────────────────────────────────────────────────────────────────── + +def skill_name_from_dir(skill_dir: Path) -> str: + md = skill_dir / "SKILL.md" + if md.exists(): + for line in md.read_text(encoding="utf-8", errors="replace").splitlines(): + s = line.strip() + if s.startswith("name:"): + return s.split(":", 1)[1].strip().strip("\"'") + return skill_dir.name + +def sha256_of(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + +def calculate_dir_sha256(directory: Path) -> str: + """Calculate SHA256 hash of a skill directory (based on all file contents + relative paths). + Excludes _meta.json and files/dirs in SKIP_FILES/SKIP_DIRS.""" + file_hashes = [] + for file_path in sorted(directory.rglob('*')): + if not file_path.is_file(): + continue + if file_path.name == '_meta.json': + continue + rel = file_path.relative_to(directory) + if any(part in SKIP_DIRS for part in rel.parts): + continue + if file_path.name in SKIP_FILES: + continue + rel_path = str(rel) + file_hash = hashlib.sha256() + file_hash.update(rel_path.encode('utf-8')) + file_hash.update(b'\x00') + with open(file_path, 'rb') as f: + for chunk in iter(lambda: f.read(8192), b''): + file_hash.update(chunk) + file_hashes.append(file_hash.hexdigest()) + file_hashes.sort() + final_hash = hashlib.sha256() + for h in file_hashes: + final_hash.update(h.encode('utf-8')) + final_hash.update(b'\x00') + return final_hash.hexdigest() + +def collect_files(skill_dir: Path) -> dict: + """Collect files for scanning, skipping redundant or sensitive directories.""" + exts = {".md",".py",".js",".ts",".sh",".yaml",".yml",".json",".txt"} + out = {} + for p in sorted(skill_dir.rglob("*")): + if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts): + continue + if p.is_file() and p.name not in SKIP_FILES: + if p.suffix.lower() in exts or p.name == "SKILL.md": + try: + out[str(p.relative_to(skill_dir))] = \ + p.read_text(encoding="utf-8", errors="replace") + except Exception: + pass + return out + +def pack_zip(skill_dir: Path) -> bytes: + """Pack a skill directory into a zip byte stream, excluding redundant directories.""" + import io + buf = io.BytesIO() + with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf: + for p in sorted(skill_dir.rglob("*")): + if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts): + continue + if p.is_file() and p.name not in SKIP_FILES: + zf.write(p, p.relative_to(skill_dir)) + return buf.getvalue() + +def unpack_zip(zip_path: Path) -> Path: + """Extract a .zip to a system temp directory. Returns the extraction path. Prevents zip-slip.""" + tmp = Path(tempfile.mkdtemp(prefix="skillscan-")) + log(f"📦 Extracting {zip_path.name} → {tmp}") + with zipfile.ZipFile(zip_path, "r") as zf: + for member in zf.namelist(): + dest = (tmp / member).resolve() + if not str(dest).startswith(str(tmp.resolve())): + raise ValueError(f"zip-slip path rejected: {member}") + zf.extractall(tmp) + return tmp + +def find_installed_skills(): + # type: () -> list + """Find all installed skill directories (first-level subdirectories containing SKILL.md). + Excludes SkillScan itself.""" + found = set() + for base in skill_install_paths(): + if not base.exists(): + continue + for md in base.rglob("SKILL.md"): + skill_path = md.parent + try: + rel = skill_path.relative_to(base) + if len(rel.parts) == 1: + resolved = skill_path.resolve() + # Skip self + if resolved == SELF_ROOT: + continue + found.add(resolved) + except ValueError: + pass + return sorted(found) + + +# ───────────────────────────────────────────────────────────────────────────── +# Scan Core (3 steps) +# ───────────────────────────────────────────────────────────────────────────── + +def _extract_result(resp, sha256): + """Internal: extract core data from API response, handling SHA256 wrapping/nested result.""" + # 1. Handle API response keyed by SHA256 (e.g. { "sha256": { "status": "success", "data": {...} } }) + if sha256 and sha256 in resp: + resp = resp[sha256] + + # 2. Extract data body (data or result) + data = resp.get("data") or resp.get("result") or resp + + # 3. Handle nested result inside data + if isinstance(data, dict) and "result" in data: + inner = data["result"] + if isinstance(inner, dict): + # Merge sibling metadata (analysis_level/reason etc.) into result + for k, v in data.items(): + if k != "result" and k not in inner: + inner[k] = v + return inner + + return data if isinstance(data, dict) and (data.get("verdict") or data.get("is_safe") is not None or data.get("analysis_level")) else None + + +def cloud_search(dir_sha256): + """Step 1: Query scan cache by dir_sha256. Returns result dict or None.""" + extra_headers = {} + ci = _get_client_info_header() + if ci: + extra_headers["X-Client-Info"] = ci + + url = "%s?%s" % (API_SEARCH, urllib.parse.urlencode({"dir_sha256": dir_sha256})) + try: + headers = {} + headers.update(extra_headers) + req = urllib.request.Request(url, headers=headers) + with urllib.request.urlopen(req, timeout=30) as r: + resp = json.loads(r.read().decode("utf-8", errors="replace")) + res = _extract_result(resp, dir_sha256) + if res: + log(" ✅ Cache hit (dir_sha256 %s…)" % dir_sha256[:16]) + return res + except urllib.error.HTTPError as e: + if e.code == 404: + return None + raise RuntimeError("Search API error HTTP %d" % e.code) + except urllib.error.URLError as e: + raise RuntimeError("Cannot connect to server: %s" % e) + return None + + +def cloud_upload(skill_dir, name, dir_hash): + """Step 2: Upload skill (multipart/form-data), returns task_no.""" + # Pack the entire directory for full code context + zip_data = pack_zip(skill_dir) + filename = "%s.zip" % name + + # Build multipart/form-data boundary + boundary = "----WebKitFormBoundary%s" % uuid.uuid4().hex + + # Manually construct multipart byte stream (no requests library needed) + parts = [] + parts.append(("--%s" % boundary).encode()) + parts.append(('Content-Disposition: form-data; name="file"; filename="%s"' % filename).encode()) + parts.append(b"Content-Type: application/zip") + parts.append(b"") + parts.append(zip_data) + parts.append(("--%s--" % boundary).encode()) + parts.append(b"") # trailing newline + + body = b"\r\n".join(parts) + + headers = { + "Content-Type": "multipart/form-data; boundary=%s" % boundary, + "Content-Length": str(len(body)), + "Accept": "application/json" + } + + # Add X-Client-Info header + ci = _get_client_info_header() + if ci: + headers["X-Client-Info"] = ci + + log(" 📤 Uploading: %s (%.1f KB)..." % (filename, len(zip_data) / 1024.0)) + req = urllib.request.Request(API_UPLOAD, data=body, headers=headers, method="POST") + try: + with urllib.request.urlopen(req, timeout=60) as r: + resp = json.loads(r.read().decode("utf-8", errors="replace")) + except urllib.error.HTTPError as e: + err_body = e.read().decode(errors="replace") + raise RuntimeError("Upload failed HTTP %d: %s" % (e.code, err_body)) + + task_no = (resp.get("data") or {}).get("task_no") or resp.get("task_no") or resp.get("taskNo") or resp.get("task_id") or "" + if not task_no: + raise RuntimeError("Upload succeeded but no valid task_no in response: %s" % resp) + + log(" ✅ Upload complete, task_no: %s" % task_no) + return str(task_no) + + +def cloud_poll(task_no: str) -> dict: + """Step 3: Poll until complete or timeout. Queries every 20s. + status: 0=pending, 1=scanning, 2=completed, 3=failed, 4=cancelled + """ + url = f"{API_RESULT}?{urllib.parse.urlencode({'task_no': task_no})}" + deadline = time.time() + POLL_TIMEOUT + attempt = 0 + while time.time() < deadline: + attempt += 1 + elapsed = int(time.time() - (deadline - POLL_TIMEOUT)) + try: + resp = http_get(url) + data = resp.get("data") or resp + status = data.get("status") + + if status == 2: # completed + print() + log(f" ✅ Scan complete (attempt {attempt}, {elapsed}s elapsed)") + return _extract_result(resp, "") or resp + elif status == 3: # failed + print() + err_msg = data.get("error_message") or resp.get("message", "unknown error") + raise RuntimeError(f"Analysis failed: {err_msg}") + elif status == 4: # cancelled + print() + raise RuntimeError("Scan task was cancelled") + else: + # 0=pending, 1=scanning -> keep waiting + status_text = data.get("status_text", "processing") + print(f" ⏳ [{status_text}] attempt {attempt}, {elapsed}s / {POLL_TIMEOUT}s elapsed", + end="\r", flush=True) + time.sleep(POLL_INTERVAL) + except (RuntimeError, ValueError): + raise + except Exception as e: + raise RuntimeError(f"Poll error: {e}") + print() + raise RuntimeError(f"Timeout ({POLL_TIMEOUT}s), task_no={task_no}, please retry later") + + +def cloud_check(skill_dir: Path) -> dict: + """Run full security scan on a skill directory, return normalized result.""" + md = skill_dir / "SKILL.md" + if not md.exists(): + raise FileNotFoundError(f"SKILL.md not found: {skill_dir}") + + name = skill_name_from_dir(skill_dir) + dir_hash = calculate_dir_sha256(skill_dir) + log(f"🔍 Scanning: {name}") + log(f" dir_sha256: {dir_hash}") + + log(f"🔎 [1/3] Checking scan cache...") + raw = cloud_search(dir_hash) + + if raw is None: + log(f" ℹ️ No cache record, submitting new scan task") + log(f"📤 [2/3] Uploading skill for analysis...") + task_no = cloud_upload(skill_dir, name, dir_hash) + log(f"⏳ [3/3] Waiting for analysis (polling every {POLL_INTERVAL}s, max {POLL_TIMEOUT}s)...") + raw = cloud_poll(task_no) + else: + log(f" ⏭️ Skipping upload, using cached result") + + return _normalize(raw, name, dir_hash) + + +def _normalize(raw: dict, name: str, dir_hash: str) -> dict: + """Normalize scan result: + 1. Extract is_safe (bool) and max_severity (str). + 2. Map API-specific fields (analysis_reason, analysis_suggestion) to standard fields. + """ + is_safe = raw.get("is_safe") + + # Severity field priority: max_severity > analysis_level > verdict > level + v_raw = (raw.get("max_severity") or raw.get("analysis_level") or + raw.get("verdict") or raw.get("risk_level") or + raw.get("level") or "UNKNOWN").upper() + + # Combined verdict logic + if is_safe is True and v_raw in ("UNKNOWN", "SAFE"): + verdict = "SAFE" + elif is_safe is False and v_raw in ("UNKNOWN", "SAFE"): + verdict = "CRITICAL" # Explicitly marked unsafe -> critical + else: + verdict = v_raw + + return { + "skill_name": name, + "dir_sha256": dir_hash, + "verdict": verdict, + "confidence": raw.get("confidence") or raw.get("score"), + "threat_labels": raw.get("threat_labels") or raw.get("tags") or [], + "summary": raw.get("analysis_reason") or raw.get("summary") or raw.get("description") or "", + "findings": raw.get("findings") or raw.get("issues") or [], + "recommendation":raw.get("analysis_suggestion") or raw.get("recommendation") or raw.get("action") or "", + } + + +# ───────────────────────────────────────────────────────────────────────────── +# Result Display +# ───────────────────────────────────────────────────────────────────────────── + +def print_result(r: dict): + verdict = r.get("verdict","UNKNOWN") + emoji = RISK_EMOJI.get(verdict,"❓") + conf = r.get("confidence") + labels = r.get("threat_labels",[]) + summary = r.get("summary","") + findings = r.get("findings",[]) + rec = r.get("recommendation","") + conf_str = f" confidence {float(conf):.0%}" if conf is not None else "" + + divider() + log(f"{emoji} Result: {verdict}{conf_str}") + if summary: + log(f"📋 {summary}") + if labels: + log(f"🏷️ Threat labels: {', '.join(labels)}") + if findings: + SEV = {"LOW":"🔵","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️"} + log(f"🔍 Findings ({len(findings)} items):") + for f in findings: + sev = str(f.get("severity","")).upper() + desc = f.get("description") or f.get("detail") or str(f) + rid = f.get("id") or "" + tag = f"[{rid}] " if rid else "" + log(f" {SEV.get(sev,'⚪')} {tag}{desc}") + if rec: + log(f"💡 Recommendation: {rec}") + divider() + +# ───────────────────────────────────────────────────────────────────────────── +# Prompt: malicious detected -> ask whether to delete +# ───────────────────────────────────────────────────────────────────────────── + +def prompt_delete(skill_path: Path, result: dict) -> bool: + """When result is HIGH/CRITICAL, ask user whether to delete the skill. + skill_path is the original install path (not temp dir). + Returns True if deleted. + """ + verdict = result.get("verdict","") + if verdict not in ("HIGH","CRITICAL"): + return False + + if not skill_path or not skill_path.exists(): + return False + + emoji = RISK_EMOJI.get(verdict,"🔴") + log(f"\n{emoji} This skill is marked as [{verdict}] high risk by security scan.") + log(f" Path: {skill_path}") + + answer = ask("Delete this skill now? [y/n]") + if answer in ("y","Y","yes","Yes"): + try: + if skill_path.is_dir(): + shutil.rmtree(skill_path) + else: + skill_path.unlink() + log(f"✅ Deleted: {skill_path}") + return True + except Exception as e: + log(f"❌ Delete failed: {e} (please delete manually)") + return False + else: + log(f"⚠️ Skipped deletion. Use this skill with caution.") + return False + + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: first-run (first install) +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_first_run(): + """First install: list installed skills, ask user to scan, show results.""" + if STATE_FILE.exists(): + log("ℹ️ First-run scan already completed. Use scan-all to rescan.") + return + + banner("🛡️ SkillScan First-Run Check") + log("Welcome to SkillScan!") + log("Searching for installed skills...\n") + + skills = find_installed_skills() + if not skills: + log("✅ No installed skills found, nothing to scan.") + STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8") + return + + # Print installed skill list + log(f"Found {len(skills)} installed skill(s):\n") + for i, s in enumerate(skills, 1): + log(f" {i:2d}. {s.name}") + + answer = ask("Run security scan on all listed skills? [y/n]") + if answer not in ("y","Y","yes","Yes"): + log("Skipped. You can run scan-all anytime to rescan.") + STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8") + return + + # Scan one by one + results = [] + for idx, skill_path in enumerate(skills, 1): + divider(f"[{idx}/{len(skills)}] {skill_path.name}") + tmp = None + try: + # Copy to temp dir (source may be read-only) + tmp = Path(tempfile.mkdtemp(prefix="skillscan-")) + scan_dir = tmp / skill_path.name + shutil.copytree(skill_path, scan_dir) + + r = cloud_check(scan_dir) + print_result(r) + + # High risk -> ask to delete (targeting original install path) + prompt_delete(skill_path, r) + results.append(r) + + except RuntimeError as e: + log(f"❌ Scan failed: {e}") + results.append({"skill_name": skill_path.name, + "verdict": "ERROR", "threat_labels": [], + "summary": str(e)[:100]}) + finally: + if tmp: + shutil.rmtree(tmp, ignore_errors=True) + + _print_summary(results) + STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8") + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: scan (single skill) +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_scan(path_str: str): + skill_path = Path(path_str) + if not skill_path.exists(): + log(f"❌ Path not found: {skill_path}") + sys.exit(1) + + banner(f"Skill Security Scan v{SCANNER_VERSION}") + + tmp = None + original_path = skill_path if skill_path.is_dir() else None + try: + if skill_path.is_file(): + if skill_path.suffix.lower() not in (".zip",): + log(f"❌ Unsupported format: {skill_path.suffix} (use .zip)") + sys.exit(1) + tmp = unpack_zip(skill_path) + scan_dir = tmp + else: + scan_dir = skill_path + + result = cloud_check(scan_dir) + print_result(result) + + # High risk -> ask to delete + if original_path: + prompt_delete(original_path, result) + elif skill_path.is_file() and result.get("verdict") in ("HIGH","CRITICAL"): + # Zip file: ask to delete source file + prompt_delete(skill_path, result) + + v = result.get("verdict","UNKNOWN") + sys.exit(0 if v in ("SAFE","LOW") else 1 if v=="MEDIUM" else 2) + + except RuntimeError as e: + log(f"\n❌ Scan failed: {e}") + sys.exit(3) + finally: + if tmp: + shutil.rmtree(tmp, ignore_errors=True) + + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: scan-all +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_scan_all(): + banner(f"Full Skill Security Scan v{SCANNER_VERSION}") + + skills = find_installed_skills() + if not skills: + log("ℹ️ No installed skills detected.") + return + + log(f"Found {len(skills)} installed skill(s):\n") + for i, s in enumerate(skills, 1): + log(f" {i:2d}. {s.name:<30} {s}") + + answer = ask("Start security scan? [y/n]") + if answer not in ("y","Y","yes","Yes"): + log("Cancelled.") + return + + results = [] + for idx, skill_path in enumerate(skills, 1): + divider(f"[{idx}/{len(skills)}] {skill_path.name}") + tmp = None + try: + tmp = Path(tempfile.mkdtemp(prefix="skillscan-")) + scan_dir = tmp / skill_path.name + shutil.copytree(skill_path, scan_dir) + + r = cloud_check(scan_dir) + v = r.get("verdict","UNKNOWN") + log(f"{RISK_EMOJI.get(v,'❓')} Scan complete: {v}") + if r.get("threat_labels"): + log(f" Threat labels: {', '.join(r['threat_labels'])}") + + # High risk: ask to delete + prompt_delete(skill_path, r) + results.append(r) + + except RuntimeError as e: + log(f"❌ Scan failed: {e}") + results.append({"skill_name": skill_path.name, "verdict":"ERROR", + "threat_labels":[], "summary":str(e)[:100]}) + finally: + if tmp: + shutil.rmtree(tmp, ignore_errors=True) + + _print_summary(results) + +# ───────────────────────────────────────────────────────────────────────────── +# Summary Table +# ───────────────────────────────────────────────────────────────────────────── + +def _print_summary(results): + banner("📊 Scan Summary") + print(f" {'Skill Name':<28} {'Result':<12} {'Threat Labels'}") + divider() + for r in results: + v = r.get("verdict","?") + name = r.get("skill_name","?")[:27] + labels = ", ".join(r.get("threat_labels",[]))[:20] or "-" + print(f" {name:<28} {RISK_EMOJI.get(v,'❓')}{v:<10} {labels}") + + safes = [r for r in results if r["verdict"] in {"SAFE","LOW"}] + mediums = [r for r in results if r["verdict"] == "MEDIUM"] + highs = [r for r in results if r["verdict"] in {"HIGH","CRITICAL"}] + errors = [r for r in results if r["verdict"] in {"ERROR","UNKNOWN"}] + + print() + log(f"Total {len(results)} | ✅ Safe {len(safes)} " + f"🟡 Suspicious {len(mediums)} 🔴 Dangerous {len(highs)} ❓ Error {len(errors)}") + if highs: + log(f"\n⚠️ High-risk skills: {', '.join(r['skill_name'] for r in highs)}") + elif not mediums and not errors: + log("\n🎉 All skills passed security scan.") + + +# ───────────────────────────────────────────────────────────────────────────── +# Subcommand: upgrade +# ───────────────────────────────────────────────────────────────────────────── + +def cmd_upgrade(): + banner("SkillScan Auto-Upgrade") + log(f"Current version: {SCANNER_VERSION}") + log(f"Update source: {UPDATE_URL}") + try: + manifest = http_get(UPDATE_URL) + except Exception as e: + log(f"❌ Failed to fetch update manifest: {e}") + return + + latest = manifest.get("version", SCANNER_VERSION) + if (tuple(int(x) for x in latest.split(".")) <= + tuple(int(x) for x in SCANNER_VERSION.split("."))): + log(f"✅ Already up to date ({SCANNER_VERSION})") + return + + log(f"New version found: {SCANNER_VERSION} → {latest}") + log(f"Changelog: {manifest.get('changelog','(none)')}") + + download_url = manifest.get("download_url", "") + if not download_url: + log("⚠️ No download URL in manifest, skipping upgrade") + return + + # Download new version zip + log(f"📥 Downloading: {download_url}") + try: + req = urllib.request.Request(download_url) + with urllib.request.urlopen(req, timeout=60) as r: + zip_data = r.read() + except Exception as e: + log(f"❌ Download failed: {e}") + return + + # SHA256 verification + expected_sha = manifest.get("sha256", "") + if expected_sha: + actual_sha = hashlib.sha256(zip_data).hexdigest() + if actual_sha != expected_sha: + log(f"❌ SHA256 mismatch, upgrade aborted (expected {expected_sha[:16]}…, got {actual_sha[:16]}…)") + return + log(f" ✅ SHA256 verified") + + # Backup current skill directory + skill_root = Path(__file__).parent.parent + backup_dir = skill_root.parent / f"SkillScan-backup-{SCANNER_VERSION}" + if backup_dir.exists(): + shutil.rmtree(backup_dir) + shutil.copytree(skill_root, backup_dir) + log(f"📦 Backed up to: {backup_dir}") + + # Extract and replace files + tmp = Path(tempfile.mkdtemp(prefix="skillupgrade-")) + try: + zip_path = tmp / "update.zip" + zip_path.write_bytes(zip_data) + with zipfile.ZipFile(zip_path, "r") as zf: + # Security check: prevent zip-slip + for member in zf.namelist(): + dest = (tmp / "extracted" / member).resolve() + if not str(dest).startswith(str((tmp / "extracted").resolve())): + raise ValueError(f"zip-slip path rejected: {member}") + zf.extractall(tmp / "extracted") + + # Overwrite skill directory with new files + extracted = tmp / "extracted" + for item in extracted.rglob("*"): + if not item.is_file(): + continue + rel = item.relative_to(extracted) + target = skill_root / rel + target.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(item, target) + log(f" ✅ Updated: {rel}") + + log(f"🎉 Upgraded to v{latest}") + except Exception as e: + log(f"❌ Upgrade failed: {e}") + log(f" You can restore from backup: {backup_dir}") + finally: + shutil.rmtree(tmp, ignore_errors=True) + +# ───────────────────────────────────────────────────────────────────────────── +# Entry Point +# ───────────────────────────────────────────────────────────────────────────── + +def auto_upgrade_if_needed(): + """Auto-check for updates every 7 days, runs silently.""" + try: + if LAST_UPDATE_CHECK_FILE.exists(): + last_check = float(LAST_UPDATE_CHECK_FILE.read_text(encoding="utf-8").strip()) + if time.time() - last_check < AUTO_UPDATE_INTERVAL: + return # Not time to check yet + log("🔄 Checking for updates...") + manifest = http_get(UPDATE_URL) + latest = manifest.get("version", SCANNER_VERSION) + if (tuple(int(x) for x in latest.split(".")) <= + tuple(int(x) for x in SCANNER_VERSION.split("."))): + log(f" ✅ Already up to date ({SCANNER_VERSION})") + else: + log(f" New version found: {SCANNER_VERSION} → {latest}, auto-updating...") + cmd_upgrade() + LAST_UPDATE_CHECK_FILE.write_text(str(time.time()), encoding="utf-8") + except Exception as e: + log(f" ⚠️ Auto-update check failed: {e} (normal operation unaffected)") + + +def main(): + if len(sys.argv) < 2: + print(__doc__) + sys.exit(0) + + # Check for auto-update on every run (once every 7 days) + auto_upgrade_if_needed() + + cmd = sys.argv[1] + if cmd == "first-run": + cmd_first_run() + elif cmd == "scan": + if len(sys.argv) < 3: + log("Usage: scanner.py scan ") + sys.exit(1) + cmd_scan(sys.argv[2]) + elif cmd == "scan-all": + cmd_scan_all() + elif cmd == "upgrade": + cmd_upgrade() + else: + log(f"Unknown command: {cmd}") + log("Available commands: first-run / scan / scan-all / upgrade") + sys.exit(1) + +if __name__ == "__main__": + main() diff --git a/runtime/skills/wiki-first-autonomy/reference.md b/runtime/skills/wiki-first-autonomy/reference.md index ae54c5e2..f6a41871 100644 --- a/runtime/skills/wiki-first-autonomy/reference.md +++ b/runtime/skills/wiki-first-autonomy/reference.md @@ -11,7 +11,7 @@ - `tavily-search-pro`:搜索与深度研究覆盖面广,能力强。 - `weather`:基础可用,但能力边界较窄。 - `self-improvement`:框架不错,但过去依赖手动触发。 -- `cocoloop`:管理类能力实用,但使用场景相对集中。 +- `oclaw-skill-manager`:技能安装与运维说明,使用场景相对集中。 核心缺口: diff --git a/runtime/skills/word-reader/DEVELOPMENT.md b/runtime/skills/word-reader/DEVELOPMENT.md new file mode 100644 index 00000000..796643f0 --- /dev/null +++ b/runtime/skills/word-reader/DEVELOPMENT.md @@ -0,0 +1,208 @@ +# Word Reader 技能开发完成 + +## 🎯 技能概述 + +成功创建了一个功能完整的 Word 文档读取技能,支持读取 .docx 和 .doc 格式的 Word 文档,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。 + +## 📁 技能结构 + +``` +word-reader/ +├── SKILL.md # 技能定义文件 +├── README.md # 使用说明 +├── skill.json # 技能配置 +├── demo.sh # 演示脚本 +├── install.sh # 安装脚本 +├── test.md # 测试文档 +└── scripts/ + └── read_word.py # 核心脚本 +``` + +## ✨ 主要功能 + +### 1. 文档解析能力 +- ✅ **文本提取** - 提取文档中的所有段落文本 +- ✅ **表格解析** - 解析表格数据并转换为结构化格式 +- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等) +- ✅ **图片信息** - 获取文档中图片的基本信息 + +### 2. 格式支持 +- ✅ **.docx** - Office 2007+ 格式(主要支持) +- ✅ **.doc** - 旧版 Word 格式(需要 antiword) + +### 3. 输出格式 +- ✅ **JSON** - 结构化数据,适合程序处理 +- ✅ **Text** - 纯文本格式,简单易读 +- ✅ **Markdown** - 格式化输出,保留文档结构 + +### 4. 高级功能 +- ✅ **批量处理** - 支持处理整个目录的文档 +- ✅ **选择性提取** - 可只提取特定内容类型 +- ✅ **文件输出** - 支持保存结果到文件 +- ✅ **编码支持** - 支持多种文本编码 + +## 🚀 使用示例 + +### 基本用法 +```bash +# 读取文档 +python3 scripts/read_word.py 文档.docx + +# JSON 格式输出 +python3 scripts/read_word.py 文档.docx --format json + +# Markdown 格式输出 +python3 scripts/read_word.py 文档.docx --format markdown + +# 只提取文本 +python3 scripts/read_word.py 文档.docx --extract text +``` + +### 批量处理 +```bash +# 批量处理目录下所有文档 +python3 scripts/read_word.py ./文档目录 --batch + +# 批量处理并保存结果 +python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json +``` + +## 🔧 安装和配置 + +### 自动安装 +```bash +cd word-reader/ +./install.sh +``` + +### 手动安装 +```bash +# 安装 Python 依赖 +pip3 install python-docx + +# 安装系统依赖(可选) +sudo apt-get install antiword # Ubuntu/Debian +brew install antiword # macOS + +# 设置执行权限 +chmod +x scripts/read_word.py +``` + +## 📊 输出示例 + +### JSON 格式 +```json +{ + "metadata": { + "filename": "文档.docx", + "title": "文档标题", + "author": "作者", + "created": "2024-01-01T10:00:00", + "modified": "2024-01-01T12:00:00" + }, + "format": "docx", + "text": "文档内容...", + "tables": [...], + "images": [...] +} +``` + +### Markdown 格式 +```markdown +# 文档.docx + +**标题**:文档标题 +**作者**:作者 +**创建时间**:2024-01-01T10:00:00 + +## 正文内容 + +文档内容... + +## 表格内容 + +| 表头1 | 表头2 | +|-------|-------| +| 数据1 | 数据2 | +``` + +## 🎨 技能特点 + +### 1. 智能错误处理 +- 友好的错误提示 +- 自动检测文档格式 +- 优雅的异常处理 + +### 2. 性能优化 +- 流式处理大文件 +- 内存使用优化 +- 进度显示(批量模式) + +### 3. 用户友好 +- 详细的帮助信息 +- 多种使用方式 +- 完整的文档说明 + +### 4. 可扩展性 +- 模块化设计 +- 易于添加新功能 +- 支持自定义输出格式 + +## 🎯 应用场景 + +### 1. 文档内容分析 +- 快速查看 Word 文档内容 +- 提取特定信息 +- 文档摘要生成 + +### 2. 批量处理 +- 处理大量文档 +- 文档格式转换 +- 内容索引创建 + +### 3. 自动化工作流 +- 集到文档处理系统 +- 自动化文档分析 +- 内容管理系统集成 + +## 📝 开发总结 + +### 实现的功能 +- 完整的 Word 文档解析框架 +- 支持多种输出格式 +- 批量处理能力 +- 错误处理和用户友好性 + +### 技术亮点 +- 模块化设计,易于维护 +- 优雅的错误处理机制 +- 支持多种文件格式 +- 灵活的输出选项 + +### 改进空间 +- 可以添加 PDF 支持 +- 可以增加图片提取功能 +- 可以优化大文件处理性能 +- 可以添加更多文档元素支持 + +## 🚀 发布到 ClawHub + +要发布此技能到 ClawHub,可以运行: + +```bash +# 安装 ClawHub CLI +npm i -g clawhub + +# 登录 +clawhub login + +# 发布技能 +clawhub publish ./word-reader \ + --slug word-reader \ + --name "Word Reader" \ + --version 1.0.0 \ + --changelog "Initial release with .docx and .doc support" \ + --tags document,word,office,text-extraction +``` + +这个技能现在已经准备好使用了!它可以帮助用户轻松读取和处理 Word 文档,支持多种格式和输出选项。 \ No newline at end of file diff --git a/runtime/skills/word-reader/PUBLISHING.md b/runtime/skills/word-reader/PUBLISHING.md new file mode 100644 index 00000000..35b32821 --- /dev/null +++ b/runtime/skills/word-reader/PUBLISHING.md @@ -0,0 +1,177 @@ +# Word Reader 技能发布指南 + +## 🚀 发布到 ClawHub + +### 1. 准备工作 + +#### 确保技能完整 +- [ ] SKILL.md 文件完整且格式正确 +- [ ] 脚本功能正常 +- [ ] 安装脚本工作正常 +- [ ] README.md 说明清晰 +- [ ] 所有依赖已在 SKILL.md 中声明 + +#### 环境准备 +```bash +# 安装 ClawHub CLI +npm install -g clawhub +# 或 +pnpm add -g clawhub +``` + +#### 登录 ClawHub +```bash +# 登录(会打开浏览器进行 OAuth 认证) +clawhub login + +# 验证登录状态 +clawhub whoami +``` + +> **注意**:GitHub 账号需要注册满一周才能发布技能 + +### 2. 发布流程 + +#### 检查技能 +```bash +# 验证技能结构 +clawhub validate ./word-reader +``` + +#### 发布技能 +```bash +clawhub publish ./word-reader \ + --slug word-reader \ + --name "Word Reader" \ + --version 1.0.0 \ + --changelog "支持 .docx 和 .doc 格式的 Word 文档读取,提取文本、表格、元数据等" \ + --tags document,word,office,text-extraction,reader,parsing \ + --license MIT \ + --visibility public +``` + +#### 参数说明 +- `--slug`: URL 友好的唯一标识符 +- `--name`: 技能显示名称 +- `--version`: 遵循语义化版本控制 +- `--changelog`: 版本变更说明 +- `--tags`: 搜索标签(逗号分隔) +- `--license`: 许可证类型 +- `--visibility`: public/private + +### 3. 发布后操作 + +#### 验证发布 +```bash +# 查看已发布的技能 +clawhub search word-reader + +# 安装测试 +clawhub install word-reader-test +``` + +#### 分享技能 +- 技能将在 `https://clawhub.com/skills/word-reader` 可见 +- 其他用户可通过 `clawhub install word-reader` 安装 + +### 4. 版本管理 + +#### 更新技能 +```bash +# 修改技能后更新版本号 +clawhub publish ./word-reader --version 1.0.1 --changelog "修复了某些文档格式的解析问题" +``` + +#### 批量操作 +```bash +# 同步所有技能 +clawhub sync --all + +# 发布并标记 +clawhub publish ./word-reader --tags latest,stable +``` + +### 5. 自动化发布 + +#### GitHub Actions 示例 +```yaml +name: Publish Skill +on: + push: + tags: + - 'v*' + +jobs: + publish: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v3 + + - name: Setup Node.js + uses: actions/setup-node@v3 + with: + node-version: '18' + + - name: Install ClawHub CLI + run: npm install -g clawhub + + - name: Login to ClawHub + run: echo "${{ secrets.CLAWHUB_TOKEN }}" | clawhub login --token + + - name: Publish Skill + run: | + clawhub publish ./skills/word-reader \ + --slug word-reader \ + --version ${{ github.ref_name }} \ + --changelog "Published from GitHub Actions" +``` + +### 6. 发布注意事项 + +#### 必须遵守的规则 +- [ ] 技能名称不能与其他技能冲突 +- [ ] 版本号遵循 SemVer 规范 +- [ ] changelog 清晰描述变更 +- [ ] 代码无安全漏洞 +- [ ] 许可证声明清晰 + +#### 最佳实践 +- [ ] 发布前充分测试 +- [ ] 提供清晰的使用示例 +- [ ] 维护更新日志 +- [ ] 及时修复问题 +- [ ] 关注用户反馈 + +### 7. 故障排除 + +#### 常见问题 +```bash +# 验证发布权限 +clawhub whoami + +# 检查技能格式 +clawhub validate ./word-reader + +# 查看详细错误信息 +clawhub publish ./word-reader --verbose +``` + +#### 重新发布 +如果发布失败,可以: +1. 修正问题 +2. 增加版本号 +3. 重新发布 + +### 8. 维护指南 + +#### 监控使用情况 +- 定期查看下载统计 +- 关注用户反馈 +- 及时修复问题 + +#### 更新策略 +- 重要修复:紧急发布补丁版本 +- 新功能:发布次版本号 +- 重大变更:发布主版本号 + +现在你的 Word Reader 技能已经准备好发布到 ClawHub 了! \ No newline at end of file diff --git a/runtime/skills/word-reader/README.md b/runtime/skills/word-reader/README.md new file mode 100644 index 00000000..3e881839 --- /dev/null +++ b/runtime/skills/word-reader/README.md @@ -0,0 +1,171 @@ +# Word Reader 技能 + +## 📋 概述 + +Word Reader 是一个强大的 Word 文档读取工具,支持 .docx 和 .doc 格式,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。 + +## ✨ 功能特性 + +- ✅ **文本提取** - 提取文档中的所有段落文本 +- ✅ **表格解析** - 解析表格数据并转换为结构化格式 +- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等) +- ✅ **图片信息** - 获取文档中图片的基本信息 +- ✅ **多格式支持** - 支持 .docx 和 .doc 格式 +- ✅ **多种输出** - JSON、Text、Markdown 格式 +- ✅ **批量处理** - 支持处理整个目录的文档 +- ✅ **自动安装** - 一键安装所有依赖 + +## 🚀 安装 + +### 自动安装(推荐) +```bash +cd word-reader/ +./install.sh +``` + +### 手动安装 +```bash +# 安装 Python 依赖 +pip3 install python-docx --break-system-packages + +# 安装系统依赖(可选,用于 .doc 格式支持) +# Ubuntu/Debian +sudo apt-get install antiword + +# macOS +brew install antiword + +# 设置执行权限 +chmod +x scripts/read_word.py +``` + +## 📖 使用方法 + +### 基本用法 +```bash +# 读取文档并输出为文本格式 +python3 scripts/read_word.py 文档.docx + +# 输出为 JSON 格式 +python3 scripts/read_word.py 文档.docx --format json + +# 输出为 Markdown 格式 +python3 scripts/read_word.py 文档.docx --format markdown + +# 只提取文本内容 +python3 scripts/read_word.py 文档.docx --extract text +``` + +### 批量处理 +```bash +# 批量处理目录下所有 Word 文档 +python3 scripts/read_word.py ./文档目录 --batch + +# 批量处理并保存为 JSON 文件 +python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json +``` + +### 高级用法 +```bash +# 将结果保存到文件 +python3 scripts/read_word.py 文档.docx --format markdown --output output.md + +# 提取表格数据 +python3 scripts/read_word.py 文档.docx --extract tables + +# 获取文档元数据 +python3 scripts/read_word.py 文档.docx --extract metadata +``` + +## 📊 输出示例 + +### JSON 格式输出 +```json +{ + "metadata": { + "filename": "测试文档.docx", + "size": "2048 bytes", + "created": "2024-01-01T10:00:00", + "modified": "2024-01-01T12:00:00", + "title": "测试文档", + "author": "测试用户" + }, + "format": "docx", + "text": "这是文档的正文内容...", + "tables": [ + { + "id": 1, + "rows": 3, + "columns": 3, + "data": [ + ["表头1", "表头2", "表头3"], + ["数据1", "数据2", "数据3"], + ["数据4", "数据5", "数据6"] + ] + } + ], + "images": [ + { + "id": "rId1", + "filename": "image1.png", + "size": "1024 bytes" + } + ] +} +``` + +### Markdown 格式输出 +```markdown +# 测试文档.docx + +**标题**:测试文档 +**作者**:测试用户 +**文件大小**:2048 bytes +**创建时间**:2024-01-01T10:00:00 +**修改时间**:2024-01-01T12:00:00 + +## 正文内容 + +这是文档的正文内容... + +## 表格内容 + +### 表格 1 (3行 x 3列) + +| 表头1 | 表头2 | 表头3 | +|-------|-------|-------| +| 数据1 | 数据2 | 数据3 | +| 数据4 | 数据5 | 数据6 | +``` + +## 🎯 应用场景 + +- **文档内容分析** - 快速查看 Word 文档内容 +- **批量处理** - 处理大量文档 +- **内容提取** - 提取特定信息 +- **格式转换** - 转换为其他格式 +- **自动化工作流** - 集成到文档处理系统 + +## 📤 发布到 ClawHub + +要将此技能发布到 ClawHub,请参考 `PUBLISHING.md` 文件。 + +## 🔧 故障排除 + +### 常见问题 +1. **ModuleNotFoundError**: 确保已安装 python-docx +2. **PermissionError**: 检查文件读取权限 +3. **FileNotFoundError**: 确认文件路径正确 +4. **编码问题**: 尝试使用 `--encoding gb2312` 参数 + +### 性能优化 +- 大文档处理时建议使用 `--format json` 以获得更好的性能 +- 批量模式下建议使用 `--output` 参数将结果保存到文件 + +## 🤝 贡献 + +欢迎提交 Issue 和 Pull Request 来改进这个技能! + +## 📄 许可证 + +MIT License \ No newline at end of file diff --git a/runtime/skills/word-reader/SKILL.md b/runtime/skills/word-reader/SKILL.md new file mode 100644 index 00000000..8e5bd6b2 --- /dev/null +++ b/runtime/skills/word-reader/SKILL.md @@ -0,0 +1,225 @@ +--- +name: word-reader +description: | + 读取 Word 文档(.docx 和 .doc 格式)并提取文本内容。支持文档解析、表格提取、图片处理等功能。使用当用户需要分析 Word 文档内容、提取文本信息或批量处理文档时。 +homepage: https://python-docx.readthedocs.io/ +metadata: + { + "openclaw": + { + "emoji": "📄", + "requires": { "bins": ["python3"], "env": ["PYTHONPATH"] }, + "install": + [ + { + "id": "pip", + "kind": "pip", + "package": "python-docx", + "bins": ["python3"], + "label": "Install python-docx (pip)", + }, + { + "id": "system", + "kind": "system", + "command": "sudo apt-get install antiword -y", + "label": "Install antiword for .doc support (optional)", + "platform": "linux-debian" + } + ], + }, + } +--- + +# Word 文档读取器 + +使用 Python 解析 Word 文档,提取文本内容和结构化信息。 + +## 支持的功能 + +- **文档文本提取** - 提取段落、标题、页眉页脚内容 +- **表格解析** - 读取表格数据并转换为结构化格式 +- **图片处理** - 提取文档中的图片信息 +- **元数据获取** - 读取文档属性(作者、标题、创建时间等) +- **批量处理** - 支持处理多个文档 + +## 用法 + +### 基本文本提取 + +```bash +python3 {baseDir}/scripts/read_word.py <文件路径> +``` + +### 指定输出格式 + +```bash +# JSON 输出 +python3 {baseDir}/scripts/read_word.py <文件路径> --format json + +# 纯文本输出 +python3 {baseDir}/scripts/read_word.py <文件路径> --format text + +# Markdown 格式 +python3 {baseDir}/scripts/read_word.py <文件路径> --format markdown +``` + +### 提取特定内容 + +```bash +# 只提取文本 +python3 {baseDir}/scripts/read_word.py <文件路径> --extract text + +# 提取表格数据 +python3 {baseDir}/scripts/read_word.py <文件路径> --extract tables + +# 获取文档元数据 +python3 {baseDir}/scripts/read_word.py <文件路径> --extract metadata +``` + +### 批量处理 + +```bash +# 处理目录下所有 .docx 文件 +python3 {baseDir}/scripts/read_word.py <目录路径> --batch +``` + +## 参数说明 + +| 参数 | 说明 | 默认值 | +|------|------|--------| +| `--format` | 输出格式(json/text/markdown) | text | +| `--extract` | 提取内容类型(text/tables/images/metadata/all) | all | +| `--batch` | 批量处理模式 | false | +| `--output` | 输出文件路径 | stdout | +| `--encoding` | 文本编码(utf-8/gb2312) | utf-8 | + +## 输出格式 + +### JSON 格式 + +```json +{ + "metadata": { + "title": "文档标题", + "author": "作者姓名", + "created": "2024-01-01T10:00:00", + "modified": "2024-01-01T12:00:00" + }, + "text": "文档全文内容...", + "tables": [ + [ + ["表头1", "表头2"], + ["行1列1", "行1列2"], + ["行2列1", "行2列2"] + ] + ], + "images": [ + { + "filename": "image1.png", + "description": "图片描述", + "size": "1024x768" + } + ] +} +``` + +### Markdown 格式 + +```markdown +# 文档标题 + +**作者**:作者姓名 +**创建时间**:2024-01-01 10:00:00 + +## 正文内容 + +这是文档的正文内容... + +### 表格示例 + +| 表头1 | 表头2 | +|-------|-------| +| 行1列1 | 行1列2 | +| 行2列1 | 行2列2 | + +![图片描述](image1.png) + +## 图片列表 + +1. **image1.png** (1024x768) - 图片描述 +``` + +## 错误处理 + +- 文件不存在:显示错误信息并退出 +- 格式不支持:提示支持的文件类型 +- 权限问题:提示文件访问权限 +- 编码问题:尝试自动检测编码 + +## 示例场景 + +### 1. 查看项目文档 + +```bash +python3 {baseDir}/scripts/read_word.py 项目需求.docx --format markdown +``` + +### 2. 提取会议记录 + +```bash +python3 {baseDir}/scripts/read_word.py 会议记录.docx --extract text +``` + +### 3. 批量处理文档 + +```bash +python3 {baseDir}/scripts/read_word.py ./文档目录 --batch --format json --output results.json +``` + +## 注意事项 + +- 支持 .docx 格式(Office 2007+) +- .doc 格式需要额外依赖(如 antiword) +- 大文档处理可能需要较长时间 +- 图片提取仅获取元数据,不包含实际图片数据 +- 表格格式可能需要手动调整 + +## 故障排除 + +### 常见问题 + +1. **ModuleNotFoundError**: 确保已安装 python-docx +2. **PermissionError**: 检查文件读取权限 +3. **UnicodeDecodeError**: 尝试不同的编码格式 + +### 安装依赖 + +```bash +pip3 install python-docx +``` + +对于 .doc 格式支持: +```bash +# Ubuntu/Debian +sudo apt-get install antiword + +# macOS +brew install antiword +``` + +## 高级功能 + +### 自定义样式处理 + +脚本会自动处理以下文档元素: +- 标题级别(H1-H6) +- 段落样式 +- 列表项目 +- 页眉页脚 +- 文档属性 + +### 性能优化 + +- 大文件流式处理 +- 内存使用优化 +- 进度显示(批量模式) \ No newline at end of file diff --git a/runtime/skills/word-reader/_meta.json b/runtime/skills/word-reader/_meta.json new file mode 100644 index 00000000..3f2ee4ad --- /dev/null +++ b/runtime/skills/word-reader/_meta.json @@ -0,0 +1,11 @@ +{ + "owner": "xtfnhcyjpgf", + "slug": "word-reader", + "displayName": "Word Reader", + "latest": { + "version": "1.0.0", + "publishedAt": 1770700102926, + "commit": "https://github.com/openclaw/skills/commit/91b71e101c57b69a4d4eb2678e1b79992eb7032f" + }, + "history": [] +} diff --git a/runtime/skills/word-reader/demo.sh b/runtime/skills/word-reader/demo.sh new file mode 100644 index 00000000..8d6f2630 --- /dev/null +++ b/runtime/skills/word-reader/demo.sh @@ -0,0 +1,89 @@ +#!/bin/bash + +# Word Reader 技能演示脚本 +# 此脚本展示如何使用 word-reader 技能 + +echo "=== Word Reader 技能演示 ===" +echo "" + +# 检查脚本是否存在 +SCRIPT_PATH="/root/.openclaw/workspace/skills/word-reader/scripts/read_word.py" +if [ ! -f "$SCRIPT_PATH" ]; then + echo "❌ 错误:脚本不存在" + echo "请确保技能已正确安装" + exit 1 +fi + +# 检查脚本是否有执行权限 +if [ ! -x "$SCRIPT_PATH" ]; then + echo "❌ 错误:脚本没有执行权限" + echo "正在添加执行权限..." + chmod +x "$SCRIPT_PATH" +fi + +echo "✅ 脚本已就绪" +echo "" + +# 显示技能信息 +echo "📋 技能信息:" +echo " 名称:word-reader" +echo " 功能:读取 Word 文档(.docx 和 .doc 格式)" +echo " 位置:$SCRIPT_PATH" +echo "" + +# 显示使用示例 +echo "📖 使用示例:" +echo "" + +echo "1. 显示帮助信息:" +echo " python3 $SCRIPT_PATH --help" +echo "" + +echo "2. 读取文档(文本格式):" +echo " python3 $SCRIPT_PATH 文档路径.docx" +echo "" + +echo "3. 读取文档(JSON 格式):" +echo " python3 $SCRIPT_PATH 文档路径.docx --format json" +echo "" + +echo "4. 读取文档(Markdown 格式):" +echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown" +echo "" + +echo "5. 只提取文本内容:" +echo " python3 $SCRIPT_PATH 文档路径.docx --extract text" +echo "" + +echo "6. 批量处理目录:" +echo " python3 $SCRIPT_PATH ./文档目录 --batch" +echo "" + +echo "7. 保存结果到文件:" +echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown --output output.md" +echo "" + +echo "🔧 安装依赖:" +echo " pip3 install python-docx" +echo " # 对于 .doc 格式支持:" +echo " # Ubuntu: sudo apt-get install antiword" +echo " # macOS: brew install antiword" +echo "" + +echo "📊 支持的功能:" +echo " ✅ 文本提取" +echo " ✅ 表格解析" +echo " ✅ 元数据获取" +echo " ✅ 图片信息" +echo " ✅ 多格式支持" +echo " ✅ 批量处理" +echo "" + +echo "💡 提示:" +echo " - 支持 .docx 和 .doc 格式" +echo " - 输出格式:JSON、Text、Markdown" +echo " - 如遇错误,请检查依赖是否安装" +echo "" + +echo "演示完成!" +echo "如需使用,请替换 '文档路径.docx' 为实际的文档路径" \ No newline at end of file diff --git a/runtime/skills/word-reader/install.sh b/runtime/skills/word-reader/install.sh new file mode 100644 index 00000000..4947e28e --- /dev/null +++ b/runtime/skills/word-reader/install.sh @@ -0,0 +1,101 @@ +#!/bin/bash + +# Word Reader 技能安装脚本 +# 此脚本会自动安装依赖并设置技能 + +set -e + +echo "=== Word Reader 技能安装 ===" +echo "" + +# 检查 Python 版本 +echo "🔍 检查 Python 版本..." +python_version=$(python3 --version 2>&1) +echo " Python 版本: $python_version" + +if ! python3 -c "import sys; assert sys.version_info >= (3, 6)"; then + echo "❌ 错误:需要 Python 3.6 或更高版本" + exit 1 +fi + +echo "✅ Python 版本检查通过" +echo "" + +# 检查并安装依赖 +echo "📦 检查依赖..." + +# 检查 pip +if ! command -v pip3 &> /dev/null; then + echo " 🔧 安装 pip..." + python3 -m ensurepip --upgrade 2>/dev/null || { + echo " ❌ 无法安装 pip,尝试使用系统包管理器" + if command -v apt &> /dev/null; then + sudo apt update + sudo apt install -y python3-pip + elif command -v yum &> /dev/null; then + sudo yum install -y python3-pip + elif command -v brew &> /dev/null; then + brew install python3 + else + echo " ❌ 无法自动安装 pip,请手动安装" + exit 1 + fi + } +fi + +# 检查 python-docx +if ! python3 -c "import docx" 2>/dev/null; then + echo " 🔧 安装 python-docx..." + if python3 -m pip install python-docx --break-system-packages 2>/dev/null; then + echo " ✅ python-docx 安装完成" + elif python3 -m pip install python-docx 2>/dev/null; then + echo " ✅ python-docx 安装完成" + else + echo "❌ 无法安装 python-docx" + exit 1 + fi +else + echo " ✅ python-docx 已安装" +fi + +# 检查 antiword(可选) +if command -v antiword >/dev/null 2>&1; then + echo " ✅ antiword 已安装" +else + echo " ⚠️ antiword 未安装(可选,用于 .doc 格式支持)" + echo " 推荐安装命令:" + echo " Ubuntu/Debian: sudo apt-get install antiword" + echo " macOS: brew install antiword" +fi + +echo "" + +# 设置执行权限 +echo "🔐 设置执行权限..." +chmod +x scripts/read_word.py +echo "✅ 执行权限已设置" +echo "" + +# 验证安装 +echo "🧪 验证安装..." +python3 scripts/read_word.py --help >/dev/null 2>&1 +if [ $? -eq 0 ]; then + echo "✅ 安装验证成功" +else + echo "❌ 安装验证失败" + exit 1 +fi + +echo "" +echo "🎉 Word Reader 技能安装完成!" +echo "" +echo "📖 使用方法:" +echo " python3 scripts/read_word.py 文档.docx" +echo " python3 scripts/read_word.py 文档.docx --format json" +echo " python3 scripts/read_word.py 文档.docx --format markdown" +echo "" +echo "📖 更多帮助:" +echo " python3 scripts/read_word.py --help" +echo "" +echo "📖 运行演示:" +echo " ./demo.sh" \ No newline at end of file diff --git a/runtime/skills/word-reader/scripts/read_word.py b/runtime/skills/word-reader/scripts/read_word.py new file mode 100644 index 00000000..27448eec --- /dev/null +++ b/runtime/skills/word-reader/scripts/read_word.py @@ -0,0 +1,396 @@ +#!/usr/bin/env python3 +""" +Word 文档读取器 +支持 .docx 和 .doc 格式的 Word 文档解析 +""" + +import argparse +import json +import os +import sys +import re +import traceback +from datetime import datetime +from pathlib import Path + +try: + from docx import Document + from docx.opc.constants import RELATIONSHIP_TYPE as RT + from docx.oxml.table import CT_Tbl + from docx.oxml.text.paragraph import CT_P + from docx.table import Table + from docx.text.paragraph import Paragraph + DOCX_AVAILABLE = True +except ImportError: + DOCX_AVAILABLE = False + +try: + import subprocess + SUBPROCESS_AVAILABLE = True +except ImportError: + SUBPROCESS_AVAILABLE = False + +class WordReader: + """Word 文档读取器""" + + def __init__(self, file_path): + self.file_path = Path(file_path) + self.document = None + self.format_type = None + self.encoding = 'utf-8' + + # 检查文件是否存在 + if not self.file_path.exists(): + raise FileNotFoundError(f"文件不存在: {file_path}") + + # 检查文件扩展名 + if self.file_path.suffix.lower() not in ['.docx', '.doc']: + raise ValueError(f"不支持的文件格式: {self.file_path.suffix}") + + def read_docx(self): + """读取 .docx 格式文档""" + if not DOCX_AVAILABLE: + raise Exception("缺少 python-docx 库。请安装:pip3 install python-docx") + + try: + self.document = Document(str(self.file_path)) + self.format_type = 'docx' + return True + except Exception as e: + raise Exception(f"读取 .docx 文件失败: {str(e)}") + + def read_doc(self): + """读取 .doc 格式文档(使用 antiword)""" + if not SUBPROCESS_AVAILABLE: + raise Exception("缺少 subprocess 模块") + + try: + # 检查 antiword 是否可用 + result = subprocess.run(['which', 'antiword'], + capture_output=True, text=True) + if result.returncode != 0: + raise Exception("antiword 未安装。请安装 antiword: Ubuntu/Debian: sudo apt-get install antiword; macOS: brew install antiword") + + # 使用 antiword 转换 + result = subprocess.run(['antiword', str(self.file_path)], + capture_output=True, text=True, encoding='utf-8') + + if result.returncode != 0: + raise Exception(f"antiword 转换失败: {result.stderr}") + + # 创建临时文档对象 + class TempDocument: + def __init__(self, text): + self.text = text + self.paragraphs = [TempParagraph(p) for p in text.split('\n') if p.strip()] + + class TempParagraph: + def __init__(self, text): + self.text = text + + self.document = TempDocument(result.stdout) + self.format_type = 'doc' + return True + except Exception as e: + raise Exception(f"读取 .doc 文件失败: {str(e)}") + + def read_metadata(self): + """读取文档元数据""" + metadata = { + 'filename': self.file_path.name, + 'size': f"{self.file_path.stat().st_size} bytes", + 'created': datetime.fromtimestamp(self.file_path.stat().st_ctime).isoformat(), + 'modified': datetime.fromtimestamp(self.file_path.stat().st_mtime).isoformat() + } + + if self.format_type == 'docx' and hasattr(self.document, 'core_properties'): + props = self.document.core_properties + metadata.update({ + 'title': getattr(props, 'title', ''), + 'author': getattr(props, 'author', ''), + 'subject': getattr(props, 'subject', ''), + 'keywords': getattr(props, 'keywords', ''), + 'comments': getattr(props, 'comments', ''), + 'application': getattr(props, 'application', ''), + 'category': getattr(props, 'category', '') + }) + + return metadata + + def extract_text(self): + """提取文档文本""" + text_content = [] + + if self.format_type == 'docx': + # 提取段落文本 + for para in self.document.paragraphs: + if para.text.strip(): + text_content.append(para.text) + + # 提取表格文本 + for table in self.document.tables: + table_text = [] + for row in table.rows: + row_text = [] + for cell in row.cells: + row_text.append(cell.text.strip()) + table_text.append(' | '.join(row_text)) + text_content.append('\n'.join(table_text)) + + else: # doc 格式 + text_content = [para.text for para in self.document.paragraphs if para.text.strip()] + + return '\n\n'.join(text_content) + + def extract_tables(self): + """提取表格数据""" + tables = [] + + if self.format_type == 'docx': + for i, table in enumerate(self.document.tables): + table_data = [] + for row in table.rows: + row_data = [] + for cell in row.cells: + row_data.append(cell.text.strip()) + table_data.append(row_data) + tables.append({ + 'id': i + 1, + 'rows': len(table.rows), + 'columns': len(table.columns) if table.rows else 0, + 'data': table_data + }) + + return tables + + def extract_images(self): + """提取图片信息""" + images = [] + + if self.format_type == 'docx': + try: + # 获取文档中的关系 + part = self.document.part + image_parts = part.related_parts + + for rel in part.relationships: + if rel.reltype == RT.IMAGE: + image_data = image_parts[rel.rId]._blob + image_info = { + 'id': rel.rId, + 'filename': f"image_{rel.rId}.{rel.target_ref.split('.')[-1]}", + 'size': f"{len(image_data)} bytes" + } + images.append(image_info) + except: + # 图片提取可能失败,忽略错误 + pass + + return images + + def extract_all(self): + """提取所有内容""" + result = { + 'metadata': self.read_metadata(), + 'format': self.format_type, + 'text': self.extract_text(), + 'tables': self.extract_tables(), + 'images': self.extract_images() + } + return result + + def to_markdown(self, extract_type='all'): + """转换为 Markdown 格式""" + if extract_type == 'text': + return self.extract_text() + + result = self.extract_all() + md_content = [] + + # 标题 + md_content.append(f"# {result['metadata']['filename']}") + md_content.append("") + + # 元数据 + metadata = result['metadata'] + if metadata.get('title'): + md_content.append(f"**标题**:{metadata['title']}") + if metadata.get('author'): + md_content.append(f"**作者**:{metadata['author']}") + md_content.append(f"**文件大小**:{metadata['size']}") + md_content.append(f"**创建时间**:{metadata['created']}") + md_content.append(f"**修改时间**:{metadata['modified']}") + md_content.append("") + + # 文本内容 + if result['text']: + md_content.append("## 正文内容") + md_content.append("") + md_content.append(result['text']) + md_content.append("") + + # 表格 + if result['tables']: + md_content.append("## 表格内容") + md_content.append("") + for table in result['tables']: + md_content.append(f"### 表格 {table['id']} ({table['rows']}行 x {table['columns']}列)") + md_content.append("") + # 转换为 Markdown 表格 + for row in table['data']: + md_row = " | ".join([str(cell) for cell in row]) + md_content.append(f"| {md_row} |") + md_content.append("") + + # 图片 + if result['images']: + md_content.append("## 图片列表") + md_content.append("") + for img in result['images']: + md_content.append(f"- **{img['filename']}** ({img['size']})") + md_content.append("") + + return '\n'.join(md_content) + + def to_text(self, extract_type='all'): + """转换为纯文本格式""" + if extract_type == 'text': + return self.extract_text() + + result = self.extract_all() + text_content = [] + + # 标题和元数据 + text_content.append(f"文件:{result['metadata']['filename']}") + text_content.append("=" * 50) + text_content.append("") + + for key, value in result['metadata'].items(): + if value and key not in ['filename', 'size', 'created', 'modified']: + text_content.append(f"{key}:{value}") + + text_content.append("") + + # 文本内容 + if result['text']: + text_content.append("正文内容:") + text_content.append("-" * 20) + text_content.append(result['text']) + text_content.append("") + + # 表格 + if result['tables']: + text_content.append("表格内容:") + text_content.append("-" * 20) + for table in result['tables']: + text_content.append(f"表格 {table['id']}:") + for row in table['data']: + text_content.append(" " + " | ".join([str(cell) for cell in row])) + text_content.append("") + + return '\n'.join(text_content) + +def main(): + parser = argparse.ArgumentParser(description='读取 Word 文档') + parser.add_argument('path', help='文档路径或目录路径(批量模式)') + parser.add_argument('--format', choices=['json', 'text', 'markdown'], + default='text', help='输出格式') + parser.add_argument('--extract', choices=['text', 'tables', 'images', 'metadata', 'all'], + default='all', help='提取内容类型') + parser.add_argument('--batch', action='store_true', help='批量处理模式') + parser.add_argument('--output', help='输出文件路径') + parser.add_argument('--encoding', default='utf-8', help='文本编码') + + args = parser.parse_args() + + try: + if args.batch: + # 批量处理模式 + path = Path(args.path) + if not path.is_dir(): + print("错误:批量模式需要指定目录路径") + sys.exit(1) + + # 查找所有 Word 文档 + word_files = [] + for ext in ['.docx', '.doc']: + word_files.extend(path.glob(f"**/*{ext}")) + + if not word_files: + print("未找到 Word 文档") + sys.exit(0) + + print(f"找到 {len(word_files)} 个 Word 文档") + + results = {} + for file_path in word_files: + print(f"正在处理: {file_path}") + try: + reader = WordReader(file_path) + if file_path.suffix.lower() == '.docx': + reader.read_docx() + else: + reader.read_doc() + + if args.format == 'json': + content = reader.extract_all() + elif args.format == 'markdown': + content = reader.to_markdown(args.extract) + else: + content = reader.to_text(args.extract) + + results[str(file_path)] = { + 'filename': file_path.name, + 'content': content, + 'status': 'success' + } + + except Exception as e: + results[str(file_path)] = { + 'filename': file_path.name, + 'error': str(e), + 'status': 'failed' + } + + # 保存结果 + if args.output: + with open(args.output, 'w', encoding='utf-8') as f: + json.dump(results, f, ensure_ascii=False, indent=2) + print(f"结果已保存到: {args.output}") + else: + print(json.dumps(results, ensure_ascii=False, indent=2)) + + else: + # 单文件处理模式 + reader = WordReader(args.path) + + # 根据文件类型读取 + if args.path.lower().endswith('.docx'): + reader.read_docx() + else: + reader.read_doc() + + # 根据格式输出 + if args.format == 'json': + content = reader.extract_all() + elif args.format == 'markdown': + content = reader.to_markdown(args.extract) + else: + content = reader.to_text(args.extract) + + # 输出结果 + if args.output: + with open(args.output, 'w', encoding=args.encoding) as f: + f.write(content) + print(f"结果已保存到: {args.output}") + else: + print(content) + + except Exception as e: + print(f"错误: {str(e)}", file=sys.stderr) + if '--debug' in sys.argv or '-d' in sys.argv: + traceback.print_exc() + sys.exit(1) + +if __name__ == '__main__': + main() \ No newline at end of file diff --git a/runtime/skills/word-reader/skill.json b/runtime/skills/word-reader/skill.json new file mode 100644 index 00000000..435f3ea9 --- /dev/null +++ b/runtime/skills/word-reader/skill.json @@ -0,0 +1,47 @@ +{ + "name": "word-reader", + "version": "1.0.0", + "description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容", + "author": "OpenClaw User", + "tags": ["document", "word", "office", "text-extraction"], + "dependencies": { + "python": ">=3.6", + "packages": ["python-docx"], + "system": ["antiword (optional for .doc support)"] + }, + "features": { + "text_extraction": true, + "table_parsing": true, + "metadata_extraction": true, + "image_info": true, + "batch_processing": true, + "multiple_formats": ["json", "text", "markdown"] + }, + "installation": { + "steps": [ + "pip3 install python-docx", + "sudo apt-get install antiword # 可选,支持 .doc 格式", + "chmod +x scripts/read_word.py" + ] + }, + "usage_examples": [ + { + "description": "读取文档文本", + "command": "python3 scripts/read_word.py document.docx" + }, + { + "description": "转换为 Markdown", + "command": "python3 scripts/read_word.py document.docx --format markdown" + }, + { + "description": "批量处理", + "command": "python3 scripts/read_word.py ./docs --batch --format json" + } + ], + "supported_file_types": [".docx", ".doc"], + "notes": [ + ".doc 格式需要安装 antiword", + "大文档处理可能需要较长时间", + "图片提取仅获取元数据,不包含实际图片数据" + ] +} \ No newline at end of file diff --git a/runtime/skills/word-reader/test.md b/runtime/skills/word-reader/test.md new file mode 100644 index 00000000..8029a815 --- /dev/null +++ b/runtime/skills/word-reader/test.md @@ -0,0 +1,42 @@ +# Word Reader 技能测试 + +这是一个简单的测试文档,用于验证 Word Reader 技能的功能。 + +## 测试内容 + +### 1. 基本文本 +这是一段测试文本,用于验证文本提取功能是否正常工作。 + +### 2. 表格测试 + +| 功能 | 状态 | 描述 | +|------|------|------| +| 文本提取 | ✅ | 能够提取文档中的所有文本内容 | +| 表格解析 | ✅ | 能够正确解析表格数据 | +| 元数据获取 | ✅ | 能够获取文档属性信息 | +| 多格式支持 | ✅ | 支持 .docx 和 .doc 格式 | +| 输出格式 | ✅ | 支持 JSON、Text、Markdown 格式 | + +### 3. 列表测试 + +- 第一项:文本提取功能 +- 第二项:表格解析功能 +- 第三项:图片信息获取 +- 第四项:文档元数据读取 + +### 4. 代码块示例 + +```python +def read_word_document(file_path): + """读取 Word 文档""" + reader = WordReader(file_path) + if file_path.endswith('.docx'): + reader.read_docx() + else: + reader.read_doc() + return reader.extract_all() +``` + +## 测试完成 + +如果这个技能能够正确读取并解析上述内容,说明功能正常。 \ No newline at end of file diff --git a/runtime/skills/xlsx-pro/README.md b/runtime/skills/xlsx-pro/README.md new file mode 100644 index 00000000..bd60e358 --- /dev/null +++ b/runtime/skills/xlsx-pro/README.md @@ -0,0 +1,61 @@ +# XLSX Pro + +![Version](https://img.shields.io/badge/version-1.0.1-blue) ![License](https://img.shields.io/badge/license-MIT-green) + +Un skill **Clawdbot / OpenClawd** pour générer et modifier des fichiers Excel **propres** (XLSX / XLSM / CSV / TSV) avec : +- formatage “pro” +- **formules Excel** (au lieu de valeurs hardcodées) +- recalcul optionnel des formules via **LibreOffice headless** +- contrôle qualité : détection des erreurs Excel (`#REF!`, `#DIV/0!`, `#VALUE!`, `#N/A`, `#NAME?`, …) + +## Pourquoi ce skill ? + +`openpyxl` sait **écrire** des formules, mais ne sait pas **calculer** leurs résultats. En production, ça crée des fichiers où les formules ne sont pas évaluées et où les erreurs ne sont pas détectées. + +`XLSX Pro` ajoute une étape serveur fiable : **recalcul via LibreOffice** + scan d’erreurs. + +## Prérequis + +### Python +```bash +pip install openpyxl pandas xlrd xlwt +``` + +### LibreOffice (uniquement si tu veux recalculer les formules) +Ubuntu/Debian : +```bash +sudo apt-get update +sudo apt-get install -y libreoffice-calc libreoffice-common +``` + +## Quickstart + +### 1) Générer un fichier Excel (avec openpyxl) +Tu peux créer ton `.xlsx` comme d’habitude en Python, en mettant des **formules** dans les cellules. + +### 2) Recalculer + valider +```bash +python scripts/recalc.py ton_fichier.xlsx 60 +``` + +Sortie JSON : +- `status: success | errors_found` +- `total_errors` +- `error_summary` (types + emplacements) +- `total_formulas` + +## Bonnes pratiques (résumé) +- **Préférer les formules Excel** plutôt que calculer en Python puis écrire des valeurs. +- **Zéro erreur de formule** dans le livrable. +- Si tu modifies un template existant : **respecte exactement** les styles/conventions. + +## Troubleshooting +- Si `soffice` est introuvable : installe LibreOffice (voir Prérequis). +- Si le recalcul “timeout” : augmente le timeout (2e argument) et/ou teste sur un fichier plus petit. +- Si erreur du type "macro mal configurée" / "macro not configured" : supprime le fichier de macro puis relance : + - Linux : `~/.config/libreoffice/4/user/basic/Standard/Module1.xba` + - macOS : `~/Library/Application Support/LibreOffice/4/user/basic/Standard/Module1.xba` +- En conteneur (Docker) : ajoute la variable d'env `SAL_USE_VCLPLUGIN=svp` (ça évite des soucis d'UI en headless). + +## Licence +MIT (à ajuster si tu veux une autre licence). diff --git a/runtime/skills/xlsx-pro/SKILL.md b/runtime/skills/xlsx-pro/SKILL.md new file mode 100644 index 00000000..6b6626a6 --- /dev/null +++ b/runtime/skills/xlsx-pro/SKILL.md @@ -0,0 +1,232 @@ +--- +name: xlsx-pro +description: "Compétence pour manipuler les fichiers Excel (.xlsx, .xlsm, .csv, .tsv). Utiliser quand l'utilisateur veut : ouvrir, lire, éditer ou créer un fichier tableur ; ajouter des colonnes, calculer des formules, formater, créer des graphiques, nettoyer des données ; convertir entre formats tabulaires. Le livrable doit être un fichier tableur. NE PAS utiliser si le livrable est un document Word, HTML, script Python standalone, ou intégration Google Sheets." +version: "1.0.1" +author: "Eric Barotte" +--- + +# Compétence Excel pour OpenClawd + +## TL;DR +- Génère/édite des fichiers Excel avec des **formules** (pas des valeurs hardcodées). +- Optionnel: **recalcul** via LibreOffice headless + détection d’erreurs Excel. +- Livrable attendu: un fichier tableur propre (XLSX/XLSM/CSV/TSV). + + +## Prérequis + +### Dépendances Python +```bash +pip install openpyxl pandas xlrd xlwt +``` + +### LibreOffice (pour recalcul des formules) +```bash +# Ubuntu/Debian +sudo apt-get install libreoffice-calc libreoffice-common +``` + +## Règles de Qualité + +### Police Professionnelle +- Utiliser une police cohérente (Arial, Times New Roman) sauf instruction contraire + +### Zéro Erreur de Formule +- Tout fichier Excel DOIT être livré SANS erreurs (#REF!, #DIV/0!, #VALUE!, #N/A, #NAME?) + +### Préservation des Templates +- Respecter EXACTEMENT le format et style existants lors de modifications +- Les conventions du template préexistant ont TOUJOURS priorité + +## Standards pour Modèles Financiers + +### Code Couleur (Standards Industrie) +- **Texte bleu (RGB: 0,0,255)** : Inputs hardcodés, valeurs modifiables +- **Texte noir (RGB: 0,0,0)** : TOUTES les formules et calculs +- **Texte vert (RGB: 0,128,0)** : Liens vers autres feuilles du même classeur +- **Texte rouge (RGB: 255,0,0)** : Liens externes vers autres fichiers +- **Fond jaune (RGB: 255,255,0)** : Hypothèses clés ou cellules à mettre à jour + +### Formatage des Nombres +- **Années** : Format texte ("2024" pas "2,024") +- **Devises** : Format $#,##0 ; spécifier unités dans les en-têtes ("Revenue ($mm)") +- **Zéros** : Afficher comme "-" (format: "$#,##0;($#,##0);-") +- **Pourcentages** : Format 0.0% par défaut +- **Multiples** : Format 0.0x (EV/EBITDA, P/E) +- **Négatifs** : Parenthèses (123) pas moins -123 + +## CRITIQUE : Utiliser des Formules, PAS des Valeurs Hardcodées + +**TOUJOURS utiliser des formules Excel au lieu de calculer en Python et hardcoder.** + +### ❌ MAUVAIS - Hardcoding +```python +# Mauvais: Calcul Python puis hardcode +total = df['Sales'].sum() +sheet['B10'] = total # Hardcode 5000 + +# Mauvais: Taux de croissance calculé en Python +growth = (df.iloc[-1]['Revenue'] - df.iloc[0]['Revenue']) / df.iloc[0]['Revenue'] +sheet['C5'] = growth # Hardcode 0.15 +``` + +### ✅ CORRECT - Formules Excel +```python +# Bon: Laisser Excel calculer +sheet['B10'] = '=SUM(B2:B9)' + +# Bon: Taux de croissance en formule Excel +sheet['C5'] = '=(C4-C2)/C2' + +# Bon: Moyenne en fonction Excel +sheet['D20'] = '=AVERAGE(D2:D19)' +``` + +## Workflows + +### Workflow Standard +1. **Choisir l'outil** : pandas pour données, openpyxl pour formules/formatage +2. **Créer/Charger** : Nouveau classeur ou fichier existant +3. **Modifier** : Données, formules, formatage +4. **Sauvegarder** : Écrire le fichier +5. **Recalculer (OBLIGATOIRE si formules)** : `python scripts/recalc.py output.xlsx` +6. **Vérifier et corriger** les erreurs détectées + +### Lecture et Analyse avec pandas +```python +import pandas as pd + +# Lire Excel +df = pd.read_excel('file.xlsx') # Première feuille par défaut +all_sheets = pd.read_excel('file.xlsx', sheet_name=None) # Dict de toutes les feuilles + +# Analyser +df.head() # Aperçu +df.info() # Info colonnes +df.describe() # Statistiques + +# Écrire +df.to_excel('output.xlsx', index=False) +``` + +### Création de Fichiers Excel +```python +from openpyxl import Workbook +from openpyxl.styles import Font, PatternFill, Alignment + +wb = Workbook() +sheet = wb.active + +# Données +sheet['A1'] = 'Hello' +sheet['B1'] = 'World' +sheet.append(['Row', 'of', 'data']) + +# Formule +sheet['B2'] = '=SUM(A1:A10)' + +# Formatage +sheet['A1'].font = Font(bold=True, color='FF0000') +sheet['A1'].fill = PatternFill('solid', start_color='FFFF00') +sheet['A1'].alignment = Alignment(horizontal='center') + +# Largeur colonne +sheet.column_dimensions['A'].width = 20 + +wb.save('output.xlsx') +``` + +### Édition de Fichiers Existants +```python +from openpyxl import load_workbook + +# Charger fichier existant +wb = load_workbook('existing.xlsx') +sheet = wb.active # ou wb['NomFeuille'] + +# Parcourir les feuilles +for sheet_name in wb.sheetnames: + sheet = wb[sheet_name] + print(f"Feuille: {sheet_name}") + +# Modifier +sheet['A1'] = 'Nouvelle Valeur' +sheet.insert_rows(2) # Insérer ligne +sheet.delete_cols(3) # Supprimer colonne + +# Ajouter feuille +new_sheet = wb.create_sheet('NouvelleFeuille') +new_sheet['A1'] = 'Data' + +wb.save('modified.xlsx') +``` + +## Recalcul des Formules + +Les fichiers créés par openpyxl contiennent les formules comme chaînes mais pas les valeurs calculées. Utiliser le script `recalc.py` : + +```bash +python scripts/recalc.py [timeout_secondes] +``` + +Le script : +- Configure automatiquement la macro LibreOffice au premier lancement +- Recalcule toutes les formules +- Scanne TOUTES les cellules pour erreurs Excel +- Retourne JSON avec détails et emplacements des erreurs + +### Interprétation de la Sortie +```json +{ + "status": "success", // ou "errors_found" + "total_errors": 0, // Nombre total d'erreurs + "total_formulas": 42, // Nombre de formules + "error_summary": { // Présent si erreurs + "#REF!": { + "count": 2, + "locations": ["Sheet1!B5", "Sheet1!C10"] + } + } +} +``` + +## Checklist de Vérification + +### Vérifications Essentielles +- [ ] **Tester 2-3 références** : Vérifier qu'elles tirent les bonnes valeurs +- [ ] **Mapping colonnes** : Confirmer correspondance (colonne 64 = BL, pas BK) +- [ ] **Offset lignes** : Excel est 1-indexé (DataFrame row 5 = Excel row 6) + +### Pièges Courants +- [ ] **Gestion NaN** : Vérifier valeurs nulles avec `pd.notna()` +- [ ] **Colonnes éloignées** : Données FY souvent en colonnes 50+ +- [ ] **Correspondances multiples** : Chercher toutes les occurrences +- [ ] **Division par zéro** : Vérifier dénominateurs (#DIV/0!) +- [ ] **Références invalides** : Vérifier que toutes pointent vers cellules existantes (#REF!) +- [ ] **Références inter-feuilles** : Format correct (Sheet1!A1) + +## Bonnes Pratiques + +### Sélection de Bibliothèque +- **pandas** : Analyse de données, opérations en masse, export simple +- **openpyxl** : Formatage complexe, formules, fonctionnalités Excel spécifiques + +### Avec openpyxl +- Indices de cellules en base 1 (row=1, column=1 = cellule A1) +- `data_only=True` pour lire valeurs calculées +- **Attention** : Sauvegarder après `data_only=True` remplace définitivement les formules par les valeurs +- Pour gros fichiers : `read_only=True` ou `write_only=True` + +### Avec pandas +- Spécifier types de données : `pd.read_excel('file.xlsx', dtype={'id': str})` +- Pour gros fichiers, colonnes spécifiques : `usecols=['A', 'C', 'E']` +- Gestion des dates : `parse_dates=['date_column']` + +## Style de Code + +**IMPORTANT** : Code Python minimal et concis, sans commentaires superflus. + +**Pour les fichiers Excel** : +- Commenter les cellules avec formules complexes +- Documenter les sources des données hardcodées +- Inclure notes pour calculs clés diff --git a/runtime/skills/xlsx-pro/_meta.json b/runtime/skills/xlsx-pro/_meta.json new file mode 100644 index 00000000..e5e08a4a --- /dev/null +++ b/runtime/skills/xlsx-pro/_meta.json @@ -0,0 +1,6 @@ +{ + "ownerId": "kn7aga06tewbtydgnw3v1468x9804b6x", + "slug": "xlsx-pro", + "version": "1.0.1", + "publishedAt": 1770059662515 +} \ No newline at end of file diff --git a/runtime/skills/xlsx-pro/scripts/office/__init__.py b/runtime/skills/xlsx-pro/scripts/office/__init__.py new file mode 100644 index 00000000..99c43da6 --- /dev/null +++ b/runtime/skills/xlsx-pro/scripts/office/__init__.py @@ -0,0 +1,8 @@ +""" +Module office pour OpenClawd +Gestion des opérations LibreOffice +""" + +from .soffice import get_soffice_env, run_soffice + +__all__ = ['get_soffice_env', 'run_soffice'] diff --git a/runtime/skills/xlsx-pro/scripts/office/soffice.py b/runtime/skills/xlsx-pro/scripts/office/soffice.py new file mode 100644 index 00000000..7240d235 --- /dev/null +++ b/runtime/skills/xlsx-pro/scripts/office/soffice.py @@ -0,0 +1,211 @@ +""" +Helper pour exécuter LibreOffice (soffice) dans des environnements +où les sockets AF_UNIX peuvent être bloqués (VMs sandboxées). + +Usage: + from office.soffice import run_soffice, get_soffice_env + + # Option 1 – exécuter soffice directement + result = run_soffice(["--headless", "--convert-to", "pdf", "input.docx"]) + + # Option 2 – obtenir env dict pour vos propres appels subprocess + env = get_soffice_env() + subprocess.run(["soffice", ...], env=env) +""" + +import os +import socket +import subprocess +import tempfile +from pathlib import Path + + +def get_soffice_env() -> dict: + """Retourne un env dict adapté pour exécuter soffice en headless. + + Définit toujours SAL_USE_VCLPLUGIN=svp pour le rendu headless (pas de X11). + Dans les environnements sandboxés où AF_UNIX est bloqué, ajoute aussi + LD_PRELOAD (socket shim). + """ + env = os.environ.copy() + env["SAL_USE_VCLPLUGIN"] = "svp" + + if _needs_shim(): + shim = _ensure_shim() + env["LD_PRELOAD"] = str(shim) + + return env + + +def run_soffice(args: list, **kwargs) -> subprocess.CompletedProcess: + """Exécute soffice avec les arguments donnés, appliquant le socket shim + si nécessaire. Accepte les mêmes arguments que subprocess.run. + + Dans les environnements sandboxés, le shim gère l'arrêt propre en appelant + _exit(0) quand le socket listener de soffice.bin se ferme (après conversion). + """ + env = get_soffice_env() + return subprocess.run(["soffice"] + args, env=env, **kwargs) + + +# --------------------------------------------------------------------------- +# Internals +# --------------------------------------------------------------------------- + +_SHIM_SO = Path(tempfile.gettempdir()) / "lo_socket_shim.so" + + +def _needs_shim() -> bool: + """Vérifie si les sockets AF_UNIX sont bloqués.""" + try: + s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + s.close() + return False + except OSError: + return True + + +def _ensure_shim() -> Path: + """Compile le shim .so s'il n'est pas déjà en cache.""" + if _SHIM_SO.exists(): + return _SHIM_SO + + src = Path(tempfile.gettempdir()) / "lo_socket_shim.c" + src.write_text(_SHIM_SOURCE) + try: + subprocess.run( + ["gcc", "-shared", "-fPIC", "-o", str(_SHIM_SO), str(src), "-ldl"], + check=True, + capture_output=True, + ) + except (subprocess.CalledProcessError, FileNotFoundError): + # Si gcc n'est pas disponible ou échoue, on continue sans shim + pass + finally: + if src.exists(): + src.unlink() + return _SHIM_SO + + +# --------------------------------------------------------------------------- +# LD_PRELOAD shim – source C +# +# Problème +# -------- +# LibreOffice utilise des sockets AF_UNIX pour la gestion single-instance +# (OSL_PIPE). Dans les environnements sandboxés, le filtre seccomp bloque +# socket(AF_UNIX) tout en permettant socketpair(AF_UNIX). Sans ce shim, +# soffice crash ou reste bloqué après conversion. +# +# Solution +# -------- +# Intercepte les appels concernés et fournit des substituts fonctionnels. +# --------------------------------------------------------------------------- + +_SHIM_SOURCE = r""" +#define _GNU_SOURCE +#include +#include +#include +#include +#include +#include +#include + +static int (*real_socket)(int, int, int); +static int (*real_socketpair)(int, int, int, int[2]); +static int (*real_listen)(int, int); +static int (*real_accept)(int, struct sockaddr *, socklen_t *); +static int (*real_close)(int); +static int (*real_read)(int, void *, size_t); + +static int is_shimmed[1024]; +static int peer_of[1024]; +static int wake_r[1024]; +static int wake_w[1024]; +static int listener_fd = -1; + +__attribute__((constructor)) +static void init(void) { + real_socket = dlsym(RTLD_NEXT, "socket"); + real_socketpair = dlsym(RTLD_NEXT, "socketpair"); + real_listen = dlsym(RTLD_NEXT, "listen"); + real_accept = dlsym(RTLD_NEXT, "accept"); + real_close = dlsym(RTLD_NEXT, "close"); + real_read = dlsym(RTLD_NEXT, "read"); + for (int i = 0; i < 1024; i++) { + peer_of[i] = -1; + wake_r[i] = -1; + wake_w[i] = -1; + } +} + +int socket(int domain, int type, int protocol) { + if (domain == AF_UNIX) { + int fd = real_socket(domain, type, protocol); + if (fd >= 0) return fd; + int sv[2]; + if (real_socketpair(domain, type, protocol, sv) == 0) { + if (sv[0] >= 0 && sv[0] < 1024) { + is_shimmed[sv[0]] = 1; + peer_of[sv[0]] = sv[1]; + int wp[2]; + if (pipe(wp) == 0) { + wake_r[sv[0]] = wp[0]; + wake_w[sv[0]] = wp[1]; + } + } + return sv[0]; + } + errno = EPERM; + return -1; + } + return real_socket(domain, type, protocol); +} + +int listen(int sockfd, int backlog) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + listener_fd = sockfd; + return 0; + } + return real_listen(sockfd, backlog); +} + +int accept(int sockfd, struct sockaddr *addr, socklen_t *addrlen) { + if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) { + if (wake_r[sockfd] >= 0) { + char buf; + real_read(wake_r[sockfd], &buf, 1); + } + errno = ECONNABORTED; + return -1; + } + return real_accept(sockfd, addr, addrlen); +} + +int close(int fd) { + if (fd >= 0 && fd < 1024 && is_shimmed[fd]) { + int was_listener = (fd == listener_fd); + is_shimmed[fd] = 0; + + if (wake_w[fd] >= 0) { + char c = 0; + write(wake_w[fd], &c, 1); + real_close(wake_w[fd]); + wake_w[fd] = -1; + } + if (wake_r[fd] >= 0) { real_close(wake_r[fd]); wake_r[fd] = -1; } + if (peer_of[fd] >= 0) { real_close(peer_of[fd]); peer_of[fd] = -1; } + + if (was_listener) + _exit(0); + } + return real_close(fd); +} +""" + + +if __name__ == "__main__": + import sys + result = run_soffice(sys.argv[1:]) + sys.exit(result.returncode) diff --git a/runtime/skills/xlsx-pro/scripts/recalc.py b/runtime/skills/xlsx-pro/scripts/recalc.py new file mode 100644 index 00000000..d6d328b0 --- /dev/null +++ b/runtime/skills/xlsx-pro/scripts/recalc.py @@ -0,0 +1,225 @@ +#!/usr/bin/env python3 +""" +Script de recalcul des formules Excel +Recalcule toutes les formules d'un fichier Excel via LibreOffice +Adapté pour OpenClawd +""" + +import json +import os +import platform +import subprocess +import sys +from pathlib import Path + +try: + from office.soffice import get_soffice_env +except ImportError: + # Fallback si le module office n'est pas disponible + def get_soffice_env(): + env = os.environ.copy() + env["SAL_USE_VCLPLUGIN"] = "svp" + return env + +try: + from openpyxl import load_workbook +except ImportError: + print("Erreur: openpyxl non installé. Exécuter: pip install openpyxl") + sys.exit(1) + +# Répertoire macro LibreOffice selon plateforme +MACRO_DIR_MACOS = "~/Library/Application Support/LibreOffice/4/user/basic/Standard" +MACRO_DIR_LINUX = "~/.config/libreoffice/4/user/basic/Standard" +MACRO_FILENAME = "Module1.xba" + +# Macro LibreOffice Basic pour recalcul +RECALCULATE_MACRO = """ + + + Sub RecalculateAndSave() + ThisComponent.calculateAll() + ThisComponent.store() + ThisComponent.close(True) + End Sub +""" + + +def has_gtimeout(): + """Vérifie si gtimeout est disponible sur macOS""" + try: + subprocess.run( + ["gtimeout", "--version"], capture_output=True, timeout=1, check=False + ) + return True + except (FileNotFoundError, subprocess.TimeoutExpired): + return False + + +def setup_libreoffice_macro(): + """Configure la macro LibreOffice si pas déjà fait""" + macro_dir = os.path.expanduser( + MACRO_DIR_MACOS if platform.system() == "Darwin" else MACRO_DIR_LINUX + ) + macro_file = os.path.join(macro_dir, MACRO_FILENAME) + + # Vérifier si macro existe déjà + if ( + os.path.exists(macro_file) + and "RecalculateAndSave" in Path(macro_file).read_text() + ): + return True + + # Créer répertoire macro si nécessaire + if not os.path.exists(macro_dir): + try: + subprocess.run( + ["soffice", "--headless", "--terminate_after_init"], + capture_output=True, + timeout=10, + env=get_soffice_env(), + ) + except Exception: + pass + os.makedirs(macro_dir, exist_ok=True) + + # Écrire fichier macro + try: + Path(macro_file).write_text(RECALCULATE_MACRO) + return True + except Exception: + return False + + +def recalc(filename, timeout=30): + """ + Recalcule les formules d'un fichier Excel et rapporte les erreurs + + Args: + filename: Chemin vers le fichier Excel + timeout: Temps max d'attente pour le recalcul (secondes) + + Returns: + dict avec emplacements et compteurs d'erreurs + """ + if not Path(filename).exists(): + return {"error": f"Fichier {filename} inexistant"} + + abs_path = str(Path(filename).absolute()) + + if not setup_libreoffice_macro(): + return {"error": "Échec configuration macro LibreOffice"} + + cmd = [ + "soffice", + "--headless", + "--norestore", + "vnd.sun.star.script:Standard.Module1.RecalculateAndSave?language=Basic&location=application", + abs_path, + ] + + # Encapsuler avec timeout si disponible + if platform.system() == "Linux": + cmd = ["timeout", str(timeout)] + cmd + elif platform.system() == "Darwin" and has_gtimeout(): + cmd = ["gtimeout", str(timeout)] + cmd + + try: + result = subprocess.run(cmd, capture_output=True, text=True, env=get_soffice_env(), timeout=timeout+10) + except subprocess.TimeoutExpired: + return {"error": "Timeout lors du recalcul"} + + if result.returncode != 0 and result.returncode != 124: # 124 = code timeout + error_msg = result.stderr or "Erreur inconnue lors du recalcul" + if "Module1" in error_msg or "RecalculateAndSave" not in error_msg: + return {"error": "Macro LibreOffice mal configurée"} + return {"error": error_msg} + + # Vérifier erreurs Excel dans le fichier recalculé + try: + wb = load_workbook(filename, data_only=True) + + excel_errors = [ + "#VALUE!", + "#DIV/0!", + "#REF!", + "#NAME?", + "#NULL!", + "#NUM!", + "#N/A", + ] + error_details = {err: [] for err in excel_errors} + total_errors = 0 + + for sheet_name in wb.sheetnames: + ws = wb[sheet_name] + for row in ws.iter_rows(): + for cell in row: + if cell.value is not None and isinstance(cell.value, str): + for err in excel_errors: + if err in cell.value: + location = f"{sheet_name}!{cell.coordinate}" + error_details[err].append(location) + total_errors += 1 + break + + wb.close() + + # Construire résumé + result = { + "status": "success" if total_errors == 0 else "errors_found", + "total_errors": total_errors, + "error_summary": {}, + } + + # Ajouter catégories d'erreurs non vides + for err_type, locations in error_details.items(): + if locations: + result["error_summary"][err_type] = { + "count": len(locations), + "locations": locations[:20], # Max 20 emplacements affichés + } + + # Ajouter compte de formules + wb_formulas = load_workbook(filename, data_only=False) + formula_count = 0 + for sheet_name in wb_formulas.sheetnames: + ws = wb_formulas[sheet_name] + for row in ws.iter_rows(): + for cell in row: + if ( + cell.value + and isinstance(cell.value, str) + and cell.value.startswith("=") + ): + formula_count += 1 + wb_formulas.close() + + result["total_formulas"] = formula_count + + return result + + except Exception as e: + return {"error": str(e)} + + +def main(): + if len(sys.argv) < 2: + print("Usage: python recalc.py [timeout_secondes]") + print("\nRecalcule toutes les formules d'un fichier Excel via LibreOffice") + print("\nRetourne JSON avec détails d'erreurs:") + print(" - status: 'success' ou 'errors_found'") + print(" - total_errors: Nombre total d'erreurs Excel") + print(" - total_formulas: Nombre de formules dans le fichier") + print(" - error_summary: Détail par type avec emplacements") + print(" - #VALUE!, #DIV/0!, #REF!, #NAME?, #NULL!, #NUM!, #N/A") + sys.exit(1) + + filename = sys.argv[1] + timeout = int(sys.argv[2]) if len(sys.argv) > 2 else 30 + + result = recalc(filename, timeout) + print(json.dumps(result, indent=2, ensure_ascii=False)) + + +if __name__ == "__main__": + main() diff --git a/runtime/skills_market.py b/runtime/skills_market.py index 921a332e..1c10b58a 100644 --- a/runtime/skills_market.py +++ b/runtime/skills_market.py @@ -4,6 +4,16 @@ from dataclasses import dataclass from typing import Any, Protocol from oclaw.runtime.tools.skills.clawhub_client import get_skill_detail, search_skills +from oclaw.runtime.tools.skills.cocoloop_client import get_skill_detail_by_slug as cocoloop_get_skill_detail +from oclaw.runtime.tools.skills.cocoloop_client import search_store_skills as cocoloop_search_skills + + +def normalize_skill_market_provider_setting(raw: str | None) -> str: + """Tenant setting value for ``AIA_SKILL_MARKET_PROVIDER``: ``clawhub`` or ``cocoloop``.""" + p = str(raw or "").strip().lower() + if p in {"cocoloop", "cocoloop-cn", "cocoloop_cn"}: + return "cocoloop" + return "clawhub" class SkillMarketAdapter(Protocol): @@ -42,12 +52,45 @@ class ClawHubMarketAdapter: return str(detail.get("archiveUrl") or "").strip(), latest +@dataclass(frozen=True) +class CocoloopMarketAdapter: + provider: str = "cocoloop" + + def search(self, query: str, *, limit: int = 20) -> list[dict[str, Any]]: + return cocoloop_search_skills(query, limit=limit) + + def detail(self, slug: str) -> dict[str, Any]: + return cocoloop_get_skill_detail(slug) + + def resolve_archive_url(self, *, slug: str, version: str | None = None) -> tuple[str, str]: + detail = self.detail(slug) + requested = str(version or "").strip().lstrip("vV") + if requested: + for row in detail.get("versions") or []: + if not isinstance(row, dict): + continue + ver = str(row.get("version") or "").strip().lstrip("vV") + if ver != requested: + continue + return str(row.get("archiveUrl") or "").strip(), str(row.get("version") or requested) + latest = str(detail.get("latestVersion") or "").strip() + return str(detail.get("archiveUrl") or "").strip(), latest + + def get_market_adapter(provider: str | None) -> SkillMarketAdapter: - p = str(provider or "clawhub").strip().lower() + p = normalize_skill_market_provider_setting(provider) if p in {"clawhub", "openclaw"}: return ClawHubMarketAdapter(provider="clawhub") + if p in {"cocoloop", "cocoloop-cn", "cocoloop_cn"}: + return CocoloopMarketAdapter(provider="cocoloop") raise ValueError(f"unsupported_market_provider:{p}") -__all__ = ["SkillMarketAdapter", "ClawHubMarketAdapter", "get_market_adapter"] +__all__ = [ + "SkillMarketAdapter", + "ClawHubMarketAdapter", + "CocoloopMarketAdapter", + "get_market_adapter", + "normalize_skill_market_provider_setting", +] diff --git a/runtime/system_prompt.py b/runtime/system_prompt.py index 8ac33bb8..ec9718f4 100644 --- a/runtime/system_prompt.py +++ b/runtime/system_prompt.py @@ -50,6 +50,10 @@ def _unified_skill_policy_guidance() -> str: "- 如果脚本依赖相对路径(例如 `.learnings/`),请将工作目录设置为用户工作区。\n" "- 在没有显式工具调用成功结果前,不要假设脚本已经执行。\n" "- 在 Windows 上,`.sh` 可能需要 Git Bash、WSL 或等效环境。\n" + "- 当用户目标是“安装 skill/技能”时,必须遵循 `oclaw-skill-manager` 的安装策略,并以其为唯一规范来源。\n" + "- 安装路径强约束:仅允许 `skill_auto_install`(`_workspace` lane);不得改用任何非 auto 路径或脚本绕过。\n" + "- 严禁臆测前置条件:不要把未在规范中声明的环境变量、端口、服务启动状态当作必需前提。\n" + "- 若安装失败,仅输出可验证事实(至少包含 `error_code` 与 `detail`)和最小下一步,不得编造基础设施依赖。\n" ) diff --git a/runtime/tools/catalog.py b/runtime/tools/catalog.py index cb891a5d..1f4cc36c 100644 --- a/runtime/tools/catalog.py +++ b/runtime/tools/catalog.py @@ -17,6 +17,13 @@ from oclaw.runtime.tools.skills_runtime.materialize_skill_tools import materiali from oclaw.runtime.skills import SkillSpec, materialize_skills_from_tool_specs logger = logging.getLogger(__name__) +# Tools hidden from model-facing registry to enforce auto-install only policy. +_MODEL_TOOLS_DENYLIST = frozenset( + { + "skill_market_install", + "skill_registry_install", + } +) # Legacy export: some modules still import TOOL_FACTORIES. Tools are now intentionally # restricted to a single safe builtin (`system_time`), so this is left empty. @@ -114,6 +121,16 @@ def _resolve_tool_conflicts(collected: list[tuple[str, ToolSpec]]) -> list[ToolS def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]: + # Lazy imports to avoid circular dependency at module import time. + from oclaw.runtime.skill_installer import ( + auto_install_skill_from_payload, + create_skill_from_template, + install_skill_from_registry_archive, + list_skills_with_status, + ) + from oclaw.runtime.skills import default_skills_root + from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting + def _create_skill_handler(args: dict[str, Any]) -> dict[str, Any]: out = create_skill_from_template( store=store, @@ -126,14 +143,60 @@ def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]: return {"ok": bool(out.ok), "name": out.name, "target_dir": out.target_dir, "detail": out.detail} def _auto_install_skill_handler(args: dict[str, Any]) -> dict[str, Any]: + archive_url = str(args.get("archive_url") or "").strip() + slug = str(args.get("slug") or "").strip() + version = str(args.get("version") or "").strip() or None + overwrite = bool(args.get("overwrite")) + provider = normalize_skill_market_provider_setting( + str(args.get("provider") or store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "") + ) + if not archive_url and slug: + try: + adapter = get_market_adapter(provider) + archive_url, _chosen_version = adapter.resolve_archive_url(slug=slug, version=version) + except Exception: + archive_url = "" + if archive_url: + out = install_skill_from_registry_archive( + store=store, + archive_url=archive_url, + overwrite=overwrite, + skills_root=default_skills_root() / "_workspace", + auto_bind=True, + ) + return { + "ok": bool(out.ok), + "name": out.name, + "target_dir": out.target_dir, + "detail": out.detail, + "error_code": out.error_code, + "retryable": bool(out.retryable), + "auto_enabled": bool(getattr(out, "auto_enabled", False)), + "binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []), + "provider": provider, + } + payload = { "name": str(args.get("name") or "").strip(), "description": str(args.get("description") or "").strip(), "body_markdown": str(args.get("body_markdown") or "").strip(), "metadata_oclaw": dict(args.get("metadata_oclaw") or {}) if isinstance(args.get("metadata_oclaw"), dict) else {}, } + if not payload["name"]: + return {"ok": False, "error_code": "name_required", "error": "name_required"} + if not payload["description"]: + payload["description"] = f"{payload['name']} skill" out = auto_install_skill_from_payload(store=store, payload=payload) - return {"ok": bool(out.ok), "name": out.name, "target_dir": out.target_dir, "detail": out.detail} + return { + "ok": bool(out.ok), + "name": out.name, + "target_dir": out.target_dir, + "detail": out.detail, + "error_code": out.error_code, + "retryable": bool(out.retryable), + "auto_enabled": bool(getattr(out, "auto_enabled", False)), + "binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []), + } def _list_skills_handler(args: dict[str, Any]) -> dict[str, Any]: del args @@ -146,9 +209,13 @@ def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]: "description": {"type": "string"}, "body_markdown": {"type": "string"}, "metadata_oclaw": {"type": "object"}, + "slug": {"type": "string"}, + "provider": {"type": "string"}, + "version": {"type": "string"}, + "archive_url": {"type": "string"}, "overwrite": {"type": "boolean"}, }, - "required": ["name", "description"], + "required": [], } return [ ToolSpec( @@ -167,7 +234,7 @@ def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]: handler=_auto_install_skill_handler, tags=frozenset({"skill", "oclaw", "installer"}), risk_level="high", - timeout_s=20.0, + timeout_s=120.0, ), ToolSpec( name="skill_list", @@ -199,6 +266,9 @@ def materialize_tool_specs( _ = factories collected: list[tuple[str, ToolSpec]] = [] + def _hidden_from_model(name: str) -> bool: + return str(name or "").strip() in _MODEL_TOOLS_DENYLIST + def _risk_allowed(spec: ToolSpec) -> bool: # Optional safety gate for public tools. # Default: only allow low risk public tools to be visible to all roles. @@ -213,6 +283,9 @@ def materialize_tool_specs( for spec in list(materialize_public_tools()): if not isinstance(spec, ToolSpec): continue + if _hidden_from_model(str(spec.name or "")): + logger.info("public tool hidden from model registry: %s", str(spec.name or "")) + continue if not _risk_allowed(spec): logger.warning("public tool blocked by risk gate: %s", str(spec.name or "")) continue @@ -225,6 +298,9 @@ def materialize_tool_specs( for spec in materialize_tools_for_expert(str(expert or "").strip() or None): if not isinstance(spec, ToolSpec): continue + if _hidden_from_model(str(spec.name or "")): + logger.info("expert tool hidden from model registry: %s", str(spec.name or "")) + continue collected.append(("expert", spec)) except Exception as exc: logger.warning("expert tool load skipped: %s", exc) @@ -235,6 +311,9 @@ def materialize_tool_specs( for spec in materialize_executable_skill_tools(store=store): if not isinstance(spec, ToolSpec): continue + if _hidden_from_model(str(spec.name or "")): + logger.info("skill runtime tool hidden from model registry: %s", str(spec.name or "")) + continue collected.append(("skill_runtime", spec)) except Exception as exc: logger.warning("skill runtime tool load skipped: %s", exc) @@ -260,6 +339,9 @@ def materialize_tool_specs( path_policy_user_id=path_policy_user_id, ): if isinstance(spec, ToolSpec): + if _hidden_from_model(str(spec.name or "")): + logger.info("mcp tool hidden from model registry: %s", str(spec.name or "")) + continue collected.append(("mcp", spec)) except Exception as exc: logger.warning("mcp tool load skipped: %s", exc) @@ -293,6 +375,9 @@ def materialize_tool_specs( continue tags_raw = row.get("tags") tags = frozenset(str(x).strip() for x in (tags_raw or []) if str(x).strip()) + if _hidden_from_model(name): + logger.info("plugin tool hidden from model registry: %s", name) + continue collected.append(( "plugin", ToolSpec( diff --git a/runtime/tools/experts/workspace/shell_tools.py b/runtime/tools/experts/workspace/shell_tools.py index a5492a73..f5b32b68 100644 --- a/runtime/tools/experts/workspace/shell_tools.py +++ b/runtime/tools/experts/workspace/shell_tools.py @@ -172,6 +172,36 @@ def run_command_tool() -> ToolSpec: command, normalized_cd_removed = _strip_leading_cd_chain(command) command, command_rewritten = _rewrite_workspace_absolute_refs(command, workdir=workdir) command, script_path_rewritten = _rewrite_python_script_arg(command, workdir=workdir) + + def _external_skill_install_cli_blocked(raw_cmd: str) -> bool: + s = str(raw_cmd or "").strip() + if not s: + return False + low = s.lower() + if re.match(r"^\s*cocoloop(?:\.cmd|\.exe)?\s+install(?:\s|$)", s, flags=re.IGNORECASE): + return True + if re.match(r"^\s*clawhub(?:\.cmd|\.exe)?\s+install(?:\s|$)", s, flags=re.IGNORECASE): + return True + if re.search(r"\bnpx\b", low) and "clawhub" in low: + return True + if re.match(r"^\s*npm(?:\.cmd|\.exe)?\s+install\b", low) and "clawhub" in low: + return True + return False + + if _external_skill_install_cli_blocked(str(command or "")): + return { + "ok": False, + "error_code": "skill_install_cli_blocked", + "error": "skill_install_cli_blocked", + "hint": "Oclaw has no shell skill installer. Use Admin POST /admin/api/skills/market/install or install-registry, or skill_auto_install.", + "command": command, + "cwd": str(workdir), + "normalized_cd_removed": bool(normalized_cd_removed), + "cwd_redirected_to_sandbox": bool(cwd_redirected_to_sandbox), + "command_rewritten": bool(command_rewritten), + "script_path_rewritten": bool(script_path_rewritten), + "original_command": original_command, + } try: os.makedirs(workdir, exist_ok=True) except Exception: diff --git a/runtime/tools/public/skills_install_tool.py b/runtime/tools/public/skills_install_tool.py new file mode 100644 index 00000000..05aff54b --- /dev/null +++ b/runtime/tools/public/skills_install_tool.py @@ -0,0 +1,141 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Any + +from oclaw.platform.config.paths import db_path +from oclaw.platform.persistence.sqlite_store import SqliteStore +from oclaw.runtime.skill_installer import install_skill_from_registry_archive +from oclaw.runtime.skills import default_skills_root +from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting +from oclaw.runtime.tools.base import ToolSpec + + +def _store() -> SqliteStore: + return SqliteStore(db_path()) + + +def _agent_workspace_skills_root() -> Path: + # Agent-origin installs are isolated under _workspace lane. + return default_skills_root() / "_workspace" + + +def skill_market_install_tool() -> ToolSpec: + def _handler(args: dict[str, Any]) -> dict[str, Any]: + payload = args if isinstance(args, dict) else {} + slug = str(payload.get("slug") or "").strip() + if not slug: + return {"ok": False, "error_code": "slug_required", "error": "slug_required"} + version = str(payload.get("version") or "").strip() or None + overwrite = bool(payload.get("overwrite")) + store = _store() + provider_arg = str(payload.get("provider") or "").strip() + if provider_arg: + provider = normalize_skill_market_provider_setting(provider_arg) + else: + provider = normalize_skill_market_provider_setting(str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "")) + try: + adapter = get_market_adapter(provider) + archive_url, chosen_version = adapter.resolve_archive_url(slug=slug, version=version) + except Exception as exc: + return { + "ok": False, + "error_code": "market_resolve_failed", + "error": f"market_resolve_failed:{type(exc).__name__}", + "provider": provider, + "slug": slug, + } + if not str(archive_url or "").strip(): + return { + "ok": False, + "error_code": "archive_url_unavailable", + "error": "archive_url_unavailable", + "provider": provider, + "slug": slug, + } + out = install_skill_from_registry_archive( + store=store, + archive_url=str(archive_url), + overwrite=overwrite, + skills_root=_agent_workspace_skills_root(), + ) + return { + "ok": bool(out.ok), + "result": { + "name": out.name, + "target_dir": out.target_dir, + "detail": out.detail, + "error_code": out.error_code, + "retryable": bool(out.retryable), + }, + "provider": provider, + "slug": slug, + "version": str(chosen_version or version or ""), + } + + return ToolSpec( + name="skill_market_install", + description="Install a skill from configured market by slug/version.", + parameters={ + "type": "object", + "properties": { + "slug": {"type": "string"}, + "provider": {"type": "string", "description": "Optional provider override: clawhub or cocoloop."}, + "version": {"type": "string"}, + "overwrite": {"type": "boolean"}, + }, + "required": ["slug"], + "additionalProperties": False, + }, + handler=_handler, + tags=frozenset({"skill", "installer", "market"}), + risk_level="medium", + timeout_s=120.0, + ) + + +def skill_registry_install_tool() -> ToolSpec: + def _handler(args: dict[str, Any]) -> dict[str, Any]: + payload = args if isinstance(args, dict) else {} + archive_url = str(payload.get("archive_url") or "").strip() + if not archive_url: + return {"ok": False, "error_code": "archive_url_required", "error": "archive_url_required"} + overwrite = bool(payload.get("overwrite")) + out = install_skill_from_registry_archive( + store=_store(), + archive_url=archive_url, + overwrite=overwrite, + skills_root=_agent_workspace_skills_root(), + ) + return { + "ok": bool(out.ok), + "result": { + "name": out.name, + "target_dir": out.target_dir, + "detail": out.detail, + "error_code": out.error_code, + "retryable": bool(out.retryable), + }, + } + + return ToolSpec( + name="skill_registry_install", + description="Install a skill from archive URL (registry/market artifact).", + parameters={ + "type": "object", + "properties": { + "archive_url": {"type": "string"}, + "overwrite": {"type": "boolean"}, + }, + "required": ["archive_url"], + "additionalProperties": False, + }, + handler=_handler, + tags=frozenset({"skill", "installer", "registry"}), + risk_level="medium", + timeout_s=120.0, + ) + + +__all__ = ["skill_market_install_tool", "skill_registry_install_tool"] + diff --git a/runtime/tools/skills/cocoloop_client.py b/runtime/tools/skills/cocoloop_client.py new file mode 100644 index 00000000..4818d007 --- /dev/null +++ b/runtime/tools/skills/cocoloop_client.py @@ -0,0 +1,187 @@ +"""CocoLoop 技能商店 HTTP 客户端(与 ClawHub 并列,供 `skills_market` 使用)。""" + +from __future__ import annotations + +import os +from dataclasses import dataclass +from typing import Any + +import httpx + + +def _strip_trailing_slash(url: str) -> str: + return str(url or "").strip().rstrip("/") + + +def _join_url(base: str, path: str) -> str: + b = _strip_trailing_slash(base) + p = str(path or "").strip() + if not p: + return b + if not p.startswith("/"): + p = "/" + p + return b + p + + +@dataclass(frozen=True) +class CocoloopConfig: + api_base_url: str = "https://api.cocoloop.com" + + +def load_cocoloop_config() -> CocoloopConfig: + base = str(os.getenv("AIA_COCOLOOP_API_BASE") or os.getenv("COCOLOOP_API_BASE") or "https://api.cocoloop.com").strip() + return CocoloopConfig(api_base_url=_strip_trailing_slash(base)) + + +def _default_headers() -> dict[str, str]: + return { + "User-Agent": "Oclaw-SkillMarket/1.0 (+https://github.com/oclaw)", + "Accept": "application/json", + } + + +def _get_json(url: str, *, params: dict[str, Any] | None = None) -> dict[str, Any]: + try: + with httpx.Client(timeout=12.0, follow_redirects=True) as c: + r = c.get(url, params=params or {}, headers=_default_headers()) + if r.status_code != 200: + return {} + obj = r.json() + return obj if isinstance(obj, dict) else {} + except Exception: + return {} + + +def _list_items(cfg: CocoloopConfig, *, keyword: str, page: int, page_size: int) -> list[dict[str, Any]]: + url = _join_url(cfg.api_base_url, "/api/v1/store/skills") + blob = _get_json( + url, + params={ + "page": max(1, int(page)), + "page_size": max(1, min(int(page_size), 100)), + "keyword": str(keyword or "").strip(), + "sort": "downloads", + }, + ) + data = blob.get("data") if isinstance(blob.get("data"), dict) else {} + items = data.get("items") + if not isinstance(items, list): + return [] + return [x for x in items if isinstance(x, dict)] + + +def _normalize_list_row(raw: dict[str, Any]) -> dict[str, Any]: + slug = str(raw.get("name") or "").strip() + dl = str(raw.get("download_url") or "").strip() + ver = str(raw.get("version") or "").strip() or "latest" + return { + "source": "cocoloop", + "slug": slug, + "name": str(raw.get("subtitle") or raw.get("summary") or slug), + "description": str(raw.get("brief") or raw.get("summary") or raw.get("original_desc") or ""), + "version": ver, + "owner": str(raw.get("author") or ""), + "updatedAt": "", + "downloads": _parse_count(raw.get("downloads")), + "stars": _parse_count(raw.get("github_stars")), + "homepage": f"https://hub.cocoloop.cn/skills/{raw.get('id')}" if raw.get("id") else "", + "archiveUrl": dl, + "raw": raw, + } + + +def _parse_count(v: Any) -> int: + if isinstance(v, int): + return v + s = str(v or "").strip().lower().replace(",", "") + if not s: + return 0 + mult = 1 + if s.endswith("k"): + mult = 1000 + s = s[:-1] + if s.endswith("m"): + mult = 1_000_000 + s = s[:-1] + try: + return int(float(s) * mult) + except ValueError: + return 0 + + +def search_store_skills(query: str, *, limit: int = 20, cfg: CocoloopConfig | None = None) -> list[dict[str, Any]]: + cfg = cfg or load_cocoloop_config() + lim = max(1, min(int(limit or 20), 100)) + rows = _list_items(cfg, keyword=str(query or "").strip(), page=1, page_size=lim) + return [_normalize_list_row(r) for r in rows if str(r.get("name") or "").strip()] + + +def get_skill_detail_by_slug(slug: str, *, cfg: CocoloopConfig | None = None) -> dict[str, Any]: + """按商店 `name`(slug)解析技能;必要时用数字 id 直查。""" + cfg = cfg or load_cocoloop_config() + s = str(slug or "").strip() + if not s: + return {} + if s.isdigit(): + return _detail_from_id(cfg, int(s)) + rows = _list_items(cfg, keyword=s, page=1, page_size=80) + want = s.lower() + hit: dict[str, Any] | None = None + for r in rows: + if str(r.get("name") or "").strip().lower() == want: + hit = r + break + if hit is None: + for r in rows: + nm = str(r.get("name") or "").strip().lower() + if want in nm or nm in want: + hit = r + break + if hit is None: + return {"slug": s, "source": "cocoloop"} + return _detail_from_list_row(cfg, hit) + + +def _detail_from_id(cfg: CocoloopConfig, skill_id: int) -> dict[str, Any]: + url = _join_url(cfg.api_base_url, f"/api/v1/store/skills/{int(skill_id)}") + blob = _get_json(url) + data = blob.get("data") if isinstance(blob.get("data"), dict) else {} + if not data: + return {"slug": str(skill_id), "source": "cocoloop"} + return _detail_from_list_row(cfg, data) + + +def _detail_from_list_row(cfg: CocoloopConfig, row: dict[str, Any]) -> dict[str, Any]: + slug = str(row.get("name") or "").strip() + dl = str(row.get("download_url") or "").strip() + if not dl and slug: + asset = str(row.get("asset_name") or f"{slug}.zip").strip() + if not asset.endswith(".zip"): + asset = f"{asset}.zip" + dl = f"https://dl.cocoloop.cn/bss/skills/{asset.lstrip('/')}" + ver = str(row.get("version") or "").strip() or "latest" + ver_clean = ver.lstrip("vV") if ver not in {"", "latest"} else ver + versions: list[dict[str, Any]] = [{"version": ver_clean or "latest", "changelog": "", "createdAt": "", "archiveUrl": dl, "raw": row}] + return { + "source": "cocoloop", + "slug": slug, + "name": str(row.get("subtitle") or row.get("summary") or slug), + "description": str(row.get("brief") or row.get("summary") or row.get("original_desc") or ""), + "owner": str(row.get("author") or ""), + "updatedAt": "", + "homepage": f"https://hub.cocoloop.cn/skills/{row.get('id')}" if row.get("id") else "", + "latestVersion": ver_clean if ver_clean else "latest", + "archiveUrl": dl, + "downloads": _parse_count(row.get("downloads")), + "stars": _parse_count(row.get("github_stars")), + "versions": versions, + "raw": row, + } + + +__all__ = [ + "CocoloopConfig", + "load_cocoloop_config", + "search_store_skills", + "get_skill_detail_by_slug", +] diff --git a/tests/test_admin_skills_api.py b/tests/test_admin_skills_api.py index 74e15420..80578207 100644 --- a/tests/test_admin_skills_api.py +++ b/tests/test_admin_skills_api.py @@ -140,9 +140,11 @@ class AdminSkillsApiTests(unittest.TestCase): self.assertEqual(g.status_code, 200, g.text) gb = g.json() or {} self.assertTrue(gb.get("ok")) + self.assertIn("market_provider", gb) + self.assertIn(str(gb.get("market_provider") or ""), {"clawhub", "cocoloop"}) s = self.client.post( "/admin/api/skills/mode", - json={"prompt_in_system": True, "toolcall_enabled": False}, + json={"prompt_in_system": True, "toolcall_enabled": False, "market_provider": "cocoloop"}, headers=self._h(), ) self.assertEqual(s.status_code, 200, s.text) @@ -150,6 +152,9 @@ class AdminSkillsApiTests(unittest.TestCase): self.assertTrue(sb.get("ok")) self.assertTrue(bool(sb.get("prompt_in_system"))) self.assertFalse(bool(sb.get("toolcall_enabled"))) + self.assertEqual(str(sb.get("market_provider") or ""), "cocoloop") + g2 = self.client.get("/admin/api/skills/mode", headers=self._h()) + self.assertEqual((g2.json() or {}).get("market_provider"), "cocoloop") def test_skills_effective_dashboard(self) -> None: c = self.client.post( diff --git a/tests/test_skill_installer.py b/tests/test_skill_installer.py index f61479e3..2b006693 100644 --- a/tests/test_skill_installer.py +++ b/tests/test_skill_installer.py @@ -2,6 +2,7 @@ from __future__ import annotations from pathlib import Path import zipfile +import subprocess from oclaw.runtime.skill_installer import ( auto_install_skill_from_payload, @@ -9,6 +10,7 @@ from oclaw.runtime.skill_installer import ( install_skill_from_local_dir, install_skill_from_registry_archive, list_skills_with_status, + repair_skill_dependencies, set_skill_enabled, ) from oclaw.platform.persistence.sqlite_store import SqliteStore @@ -102,6 +104,33 @@ def test_install_skill_from_registry_archive_file_url(tmp_path: Path) -> None: assert out.name == "reg_demo" +def test_install_skill_from_registry_archive_workspace_auto_bind(tmp_path: Path) -> None: + db = tmp_path / "ops.sqlite" + store = SqliteStore(str(db)) + pkg_dir = tmp_path / "pkg_ws" + inner = pkg_dir / "demo" + inner.mkdir(parents=True, exist_ok=True) + (inner / "SKILL.md").write_text( + "---\nname: reg_ws_demo\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n", + encoding="utf-8", + ) + archive = tmp_path / "reg_ws.zip" + with zipfile.ZipFile(archive, "w") as zf: + zf.write(inner / "SKILL.md", arcname="demo/SKILL.md") + root = tmp_path / "skills" + out = install_skill_from_registry_archive( + store=store, + archive_url=archive.resolve().as_uri(), + skills_root=root / "_workspace", + auto_bind=True, + ) + assert out.ok + assert out.name == "reg_ws_demo" + assert out.auto_enabled is True + assert len(out.binding_applied_roles) >= 1 + assert (root / "_workspace" / "reg_ws_demo" / "SKILL.md").exists() + + def test_install_skill_from_clawhub_page_url(tmp_path: Path, monkeypatch) -> None: db = tmp_path / "ops.sqlite" store = SqliteStore(str(db)) @@ -149,3 +178,103 @@ def test_install_local_allows_sh_files(tmp_path: Path) -> None: assert out.ok assert out.name == "local_with_sh" + +def test_install_local_auto_installs_python_requirements(tmp_path: Path, monkeypatch) -> None: + db = tmp_path / "ops.sqlite" + store = SqliteStore(str(db)) + src = tmp_path / "skill_with_reqs" + src.mkdir(parents=True, exist_ok=True) + (src / "SKILL.md").write_text( + "---\nname: with_reqs\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n", + encoding="utf-8", + ) + (src / "requirements.txt").write_text("requests>=2.0.0\n", encoding="utf-8") + + calls: list[list[str]] = [] + + def _mock_run(cmd, **kwargs): # noqa: ANN001 + calls.append([str(x) for x in cmd]) + return subprocess.CompletedProcess(args=cmd, returncode=0, stdout="", stderr="") + + monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run) + out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills") + assert out.ok + assert any(("pip" in " ".join(c) and "-r" in c) for c in calls) + + +def test_install_local_dependency_install_failure_returns_warning(tmp_path: Path, monkeypatch) -> None: + db = tmp_path / "ops.sqlite" + store = SqliteStore(str(db)) + src = tmp_path / "skill_with_bad_reqs" + src.mkdir(parents=True, exist_ok=True) + (src / "SKILL.md").write_text( + "---\nname: with_bad_reqs\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n", + encoding="utf-8", + ) + (src / "requirements.txt").write_text("not_a_real_pkg_zzz\n", encoding="utf-8") + + def _mock_run(cmd, **kwargs): # noqa: ANN001,ARG001 + return subprocess.CompletedProcess(args=cmd, returncode=1, stdout="", stderr="install failed") + + monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run) + out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills") + assert out.ok + assert out.detail.startswith("installed_with_dependency_warnings:") + + +def test_install_local_probe_missing_imports_and_install(tmp_path: Path, monkeypatch) -> None: + db = tmp_path / "ops.sqlite" + store = SqliteStore(str(db)) + src = tmp_path / "skill_probe_imports" + src.mkdir(parents=True, exist_ok=True) + (src / "SKILL.md").write_text( + "---\nname: probe_imports\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n", + encoding="utf-8", + ) + (src / "main.py").write_text( + "import json\nimport office\nimport pandas\nimport totally_missing_pkg_xyz\n", + encoding="utf-8", + ) + office_dir = src / "office" + office_dir.mkdir(parents=True, exist_ok=True) + (office_dir / "__init__.py").write_text("", encoding="utf-8") + + calls: list[list[str]] = [] + + def _mock_run(cmd, **kwargs): # noqa: ANN001,ARG001 + calls.append([str(x) for x in cmd]) + return subprocess.CompletedProcess(args=cmd, returncode=0, stdout="", stderr="") + + monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run) + out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills") + assert out.ok + pip_calls = [c for c in calls if ("pip" in " ".join(c))] + assert pip_calls + assert any("totally_missing_pkg_xyz" in c for c in pip_calls) + + +def test_repair_skill_dependencies_for_installed_skill(tmp_path: Path, monkeypatch) -> None: + db = tmp_path / "ops.sqlite" + store = SqliteStore(str(db)) + src = tmp_path / "skill_repair" + src.mkdir(parents=True, exist_ok=True) + (src / "SKILL.md").write_text( + "---\nname: skill_repair\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n", + encoding="utf-8", + ) + (src / "main.py").write_text("import definitely_missing_pkg_abc\n", encoding="utf-8") + + calls: list[list[str]] = [] + + def _mock_run(cmd, **kwargs): # noqa: ANN001,ARG001 + calls.append([str(x) for x in cmd]) + return subprocess.CompletedProcess(args=cmd, returncode=0, stdout="", stderr="") + + monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run) + out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills") + assert out.ok + calls.clear() + result = repair_skill_dependencies(store=store, skill_name="skill_repair", skills_root=tmp_path / "skills") + assert bool(result.get("ok")) is True + assert any("definitely_missing_pkg_abc" in c for c in calls) + diff --git a/tests/test_skills_install_public_tool_visibility.py b/tests/test_skills_install_public_tool_visibility.py new file mode 100644 index 00000000..de0b5c47 --- /dev/null +++ b/tests/test_skills_install_public_tool_visibility.py @@ -0,0 +1,10 @@ +from __future__ import annotations + +from oclaw.runtime.tools.catalog import default_registry + + +def test_skill_install_public_tools_hidden_for_specialist_auto_only() -> None: + names = [t.name for t in default_registry(expert="network_ops+memory", specialist="ops").list()] + assert "skill_market_install" not in names + assert "skill_registry_install" not in names + diff --git a/tests/test_skills_install_tool_workspace_root.py b/tests/test_skills_install_tool_workspace_root.py new file mode 100644 index 00000000..394e2da3 --- /dev/null +++ b/tests/test_skills_install_tool_workspace_root.py @@ -0,0 +1,83 @@ +from __future__ import annotations + +from pathlib import Path + +from oclaw.runtime.tools.public.skills_install_tool import skill_market_install_tool, skill_registry_install_tool + + +def test_skill_registry_install_tool_forces_workspace_root(monkeypatch, tmp_path: Path) -> None: + captured: dict[str, str] = {} + + def _mock_default_root() -> Path: + return tmp_path / "skills" + + def _mock_install(**kwargs): # noqa: ANN003 + captured["skills_root"] = str(kwargs.get("skills_root") or "") + + class _Out: + ok = True + name = "demo" + target_dir = str((tmp_path / "skills" / "_workspace" / "demo").resolve()) + detail = "installed" + error_code = "ok" + retryable = False + + return _Out() + + monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.default_skills_root", _mock_default_root) + monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.install_skill_from_registry_archive", _mock_install) + tool = skill_registry_install_tool() + result = tool.handler({"archive_url": "https://example.com/demo.zip"}) + assert bool(result.get("ok")) is True + assert captured["skills_root"].replace("\\", "/").endswith("/skills/_workspace") + + +def test_skill_market_install_tool_provider_arg_overrides_setting(monkeypatch, tmp_path: Path) -> None: + captured: dict[str, str] = {} + + class _FakeStore: + def get_setting(self, key: str) -> str: + if key == "AIA_SKILL_MARKET_PROVIDER": + return "clawhub" + return "" + + class _FakeAdapter: + def resolve_archive_url(self, *, slug: str, version: str | None = None) -> tuple[str, str]: + captured["slug"] = slug + captured["version"] = str(version or "") + return "https://example.com/demo.zip", "1.0.0" + + def _mock_store() -> _FakeStore: + return _FakeStore() + + def _mock_default_root() -> Path: + return tmp_path / "skills" + + def _mock_get_market_adapter(provider: str): # noqa: ANN001 + captured["provider"] = provider + return _FakeAdapter() + + def _mock_install(**kwargs): # noqa: ANN003 + captured["skills_root"] = str(kwargs.get("skills_root") or "") + + class _Out: + ok = True + name = "demo" + target_dir = str((tmp_path / "skills" / "_workspace" / "demo").resolve()) + detail = "installed" + error_code = "ok" + retryable = False + + return _Out() + + monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool._store", _mock_store) + monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.default_skills_root", _mock_default_root) + monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.get_market_adapter", _mock_get_market_adapter) + monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.install_skill_from_registry_archive", _mock_install) + + tool = skill_market_install_tool() + result = tool.handler({"slug": "demo", "provider": "cocoloop", "version": "latest"}) + assert bool(result.get("ok")) is True + assert captured["provider"] == "cocoloop" + assert captured["skills_root"].replace("\\", "/").endswith("/skills/_workspace") + diff --git a/tests/test_skills_market_providers.py b/tests/test_skills_market_providers.py new file mode 100644 index 00000000..94ece27c --- /dev/null +++ b/tests/test_skills_market_providers.py @@ -0,0 +1,35 @@ +from __future__ import annotations + +from oclaw.runtime import skills_market + + +def test_get_market_adapter_clawhub_default() -> None: + a = skills_market.get_market_adapter("clawhub") + assert a.provider == "clawhub" + + +def test_get_market_adapter_cocoloop() -> None: + a = skills_market.get_market_adapter("cocoloop") + assert a.provider == "cocoloop" + + +def test_get_market_adapter_cocoloop_alias() -> None: + a = skills_market.get_market_adapter("cocoloop-cn") + assert a.provider == "cocoloop" + + +def test_cocoloop_resolve_archive_url(monkeypatch) -> None: + def _fake_detail(slug: str) -> dict: # noqa: ANN001 + return { + "source": "cocoloop", + "slug": slug, + "latestVersion": "1.0.0", + "archiveUrl": "https://dl.example/bss/skills/demo.zip", + "versions": [{"version": "1.0.0", "archiveUrl": "https://dl.example/bss/skills/demo.zip"}], + } + + monkeypatch.setattr("oclaw.runtime.skills_market.cocoloop_get_skill_detail", _fake_detail) + a = skills_market.CocoloopMarketAdapter() + url, ver = a.resolve_archive_url(slug="demo", version=None) + assert url.endswith("demo.zip") + assert ver == "1.0.0" diff --git a/tests/test_tool_loop_guard.py b/tests/test_tool_loop_guard.py index 042f50f8..ae2370d0 100644 --- a/tests/test_tool_loop_guard.py +++ b/tests/test_tool_loop_guard.py @@ -41,11 +41,50 @@ def test_tool_loop_guard_blocks_repeated_signature(tmp_path: Path) -> None: tool_uses=tool_uses, signature_budget=2, ) - assert calls["n"] == 2 + # Same-round duplicate calls now hit cache; only the first executes. + assert calls["n"] == 1 + second, _ = results["c2"] + assert bool(second.get("ok")) is True blocked, _ = results["c3"] assert blocked.get("error_code") == "tool_loop_guard" +def test_same_round_duplicate_tool_call_reuses_cached_result(tmp_path: Path) -> None: + store = SqliteStore(str(tmp_path / "dup.sqlite")) + sess = store.create_session("t") + calls = {"n": 0} + + def _handler(args): + calls["n"] += 1 + return {"ok": True, "echo": args, "counter": calls["n"]} + + reg = ToolRegistry( + [ + ToolSpec( + name="echo", + description="echo", + parameters={"type": "object", "properties": {"x": {"type": "integer"}}}, + handler=_handler, + read_only=True, + ) + ] + ) + tool_uses = [ + LLMToolCall(id="c1", name="echo", arguments={"x": 1}), + LLMToolCall(id="c2", name="echo", arguments={"x": 1}), + ] + _, results = ToolExecutor().execute_tool_uses( + ctx=ToolExecutionContext(store=store, tools=reg, session_id=sess.id), + assistant_msg_id=1, + tool_uses=tool_uses, + signature_budget=2, + ) + assert calls["n"] == 1 + r1, _ = results["c1"] + r2, _ = results["c2"] + assert r1 == r2 + + def test_repeated_tool_results_are_compacted_in_history(tmp_path: Path) -> None: store = SqliteStore(str(tmp_path / "g2.sqlite")) sess = store.create_session("t") diff --git a/tests/test_workspace_path_guard.py b/tests/test_workspace_path_guard.py index fabcd2a8..939e0c24 100644 --- a/tests/test_workspace_path_guard.py +++ b/tests/test_workspace_path_guard.py @@ -256,6 +256,43 @@ class WorkspacePathGuardTests(unittest.TestCase): self.assertEqual(str(r.get("error_code") or ""), "command_exit_nonzero") self.assertFalse(bool(r.get("output_truncated"))) + def test_run_command_blocks_cocoloop_install_cli(self) -> None: + with mock.patch.dict( + os.environ, + { + "OPS_WORKSPACE_ROOT": str(self.root), + "OPS_WORKSPACE_EXTRA_ROOTS": "", + "OPS_WORKSPACE_ALLOW_ANY_PATH": "", + "AIA_ENABLE_RUN_COMMAND": "1", + }, + clear=False, + ): + clear_workspace_path_access_for_tests() + spec = run_command_tool() + with workspace_path_access_scope(None, None): + r = spec.handler({"command": "cocoloop install 7288"}) + self.assertFalse(bool(r.get("ok")), r) + self.assertEqual("skill_install_cli_blocked", str(r.get("error_code") or "")) + self.assertIn("market/install", str(r.get("hint") or "")) + + def test_run_command_blocks_npx_clawhub_install_pattern(self) -> None: + with mock.patch.dict( + os.environ, + { + "OPS_WORKSPACE_ROOT": str(self.root), + "OPS_WORKSPACE_EXTRA_ROOTS": "", + "OPS_WORKSPACE_ALLOW_ANY_PATH": "", + "AIA_ENABLE_RUN_COMMAND": "1", + }, + clear=False, + ): + clear_workspace_path_access_for_tests() + spec = run_command_tool() + with workspace_path_access_scope(None, None): + r = spec.handler({"command": "npx -y clawhub@latest install foo"}) + self.assertFalse(bool(r.get("ok")), r) + self.assertEqual("skill_install_cli_blocked", str(r.get("error_code") or "")) + def test_run_command_rewrites_workspace_absolute_script_path_to_sandbox(self) -> None: with mock.patch.dict( os.environ,