mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 03:13:19 +08:00
统一技能安装与 Skills 市场接入链路。
新增公开安装工具与多来源 provider 支持,补齐管理端与目录加载逻辑,并同步更新相关测试与文档以保证可见性和路径安全。 Made-with: Cursor
This commit is contained in:
parent
d5e30542aa
commit
4d9232f3b3
66 changed files with 7369 additions and 1099 deletions
|
|
@ -15,6 +15,7 @@ from oclaw.runtime.skill_installer import (
|
|||
install_skill_from_local_dir,
|
||||
install_skill_from_registry_archive,
|
||||
list_skills_with_status,
|
||||
repair_skill_dependencies,
|
||||
set_skill_enabled,
|
||||
uninstall_skill,
|
||||
)
|
||||
|
|
@ -29,11 +30,13 @@ from oclaw.runtime.skill_role_binding import (
|
|||
)
|
||||
from oclaw.runtime.skills_prompt import collect_skill_catalog_entries
|
||||
from oclaw.runtime.skills import _allowed_tool_names_after_wire_policy, discover_workspace_skill_manifests
|
||||
from oclaw.runtime.skills_market import get_market_adapter
|
||||
from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting
|
||||
from oclaw.runtime.tools.skills_runtime.subprocess_exec import run_skill_runtime_entry
|
||||
from oclaw.platform.config.paths import db_path
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
|
||||
_SKILL_MARKET_PROVIDER_KEY = "AIA_SKILL_MARKET_PROVIDER"
|
||||
|
||||
|
||||
def include_skill_routes(
|
||||
router: APIRouter,
|
||||
|
|
@ -95,7 +98,13 @@ def include_skill_routes(
|
|||
raw_toolcall = str(store.get_setting("AIA_SKILL_TOOLCALL_ENABLED") or "").strip().lower()
|
||||
prompt_in_system = raw_prompt not in {"0", "false", "no", "off"}
|
||||
toolcall_enabled = raw_toolcall in {"1", "true", "yes", "on"}
|
||||
return {"ok": True, "prompt_in_system": bool(prompt_in_system), "toolcall_enabled": bool(toolcall_enabled)}
|
||||
market_provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or ""))
|
||||
return {
|
||||
"ok": True,
|
||||
"prompt_in_system": bool(prompt_in_system),
|
||||
"toolcall_enabled": bool(toolcall_enabled),
|
||||
"market_provider": market_provider,
|
||||
}
|
||||
|
||||
@sk.post("/mode")
|
||||
def api_skills_mode_save(
|
||||
|
|
@ -110,19 +119,31 @@ def include_skill_routes(
|
|||
store.set_setting("AIA_SKILLS_PROMPT_IN_SYSTEM", "1" if bool(payload.get("prompt_in_system")) else "0")
|
||||
if "toolcall_enabled" in payload:
|
||||
store.set_setting("AIA_SKILL_TOOLCALL_ENABLED", "1" if bool(payload.get("toolcall_enabled")) else "0")
|
||||
if "market_provider" in payload:
|
||||
store.set_setting(_SKILL_MARKET_PROVIDER_KEY, normalize_skill_market_provider_setting(str(payload.get("market_provider") or "")))
|
||||
raw_prompt = str(store.get_setting("AIA_SKILLS_PROMPT_IN_SYSTEM") or "").strip().lower()
|
||||
raw_toolcall = str(store.get_setting("AIA_SKILL_TOOLCALL_ENABLED") or "").strip().lower()
|
||||
prompt_in_system = raw_prompt not in {"0", "false", "no", "off"}
|
||||
toolcall_enabled = raw_toolcall in {"1", "true", "yes", "on"}
|
||||
market_provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or ""))
|
||||
_audit(
|
||||
store,
|
||||
ctx,
|
||||
action="skill_mode_update",
|
||||
target_id="skill_mode",
|
||||
status="ok",
|
||||
detail={"prompt_in_system": bool(prompt_in_system), "toolcall_enabled": bool(toolcall_enabled)},
|
||||
detail={
|
||||
"prompt_in_system": bool(prompt_in_system),
|
||||
"toolcall_enabled": bool(toolcall_enabled),
|
||||
"market_provider": market_provider,
|
||||
},
|
||||
)
|
||||
return {"ok": True, "prompt_in_system": bool(prompt_in_system), "toolcall_enabled": bool(toolcall_enabled)}
|
||||
return {
|
||||
"ok": True,
|
||||
"prompt_in_system": bool(prompt_in_system),
|
||||
"toolcall_enabled": bool(toolcall_enabled),
|
||||
"market_provider": market_provider,
|
||||
}
|
||||
|
||||
@sk.post("/install")
|
||||
def api_skills_install(
|
||||
|
|
@ -204,7 +225,7 @@ def include_skill_routes(
|
|||
query = str(q or "").strip()
|
||||
lim = int(limit) if isinstance(limit, int) and limit > 0 else 20
|
||||
lim = max(1, min(lim, 200))
|
||||
provider = str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub").strip()
|
||||
provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or ""))
|
||||
items = get_market_adapter(provider).search(query, limit=lim)
|
||||
return {"ok": True, "items": items}
|
||||
|
||||
|
|
@ -219,7 +240,7 @@ def include_skill_routes(
|
|||
s = str(slug or "").strip()
|
||||
if not s:
|
||||
raise HTTPException(status_code=400, detail="slug_required")
|
||||
provider = str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub").strip()
|
||||
provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or ""))
|
||||
detail = get_market_adapter(provider).detail(s)
|
||||
return {"ok": True, "detail": detail}
|
||||
|
||||
|
|
@ -238,7 +259,7 @@ def include_skill_routes(
|
|||
requested_version = str(payload.get("version") or "").strip()
|
||||
overwrite = bool(payload.get("overwrite"))
|
||||
|
||||
provider = str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub").strip()
|
||||
provider = normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or ""))
|
||||
adapter = get_market_adapter(provider)
|
||||
archive_url, chosen_version = adapter.resolve_archive_url(slug=s, version=requested_version or None)
|
||||
|
||||
|
|
@ -673,6 +694,98 @@ def include_skill_routes(
|
|||
},
|
||||
}
|
||||
|
||||
@sk.post("/repair-deps")
|
||||
def api_skills_repair_deps(
|
||||
payload: dict[str, Any] | None = Body(default=None),
|
||||
authorization: str | None = Header(default=None),
|
||||
) -> dict[str, Any]:
|
||||
payload = payload or {}
|
||||
store = SqliteStore(db_path())
|
||||
ctx = resolve_auth(store, authorization)
|
||||
_require_admin(ctx)
|
||||
name = str(payload.get("name") or "").strip()
|
||||
if not name:
|
||||
raise HTTPException(status_code=400, detail="name_required")
|
||||
_audit(
|
||||
store,
|
||||
ctx,
|
||||
action="skill_repair_deps_started",
|
||||
target_id=name,
|
||||
status="start",
|
||||
)
|
||||
out = repair_skill_dependencies(store=store, skill_name=name)
|
||||
_audit(
|
||||
store,
|
||||
ctx,
|
||||
action="skill_repair_deps_finished" if bool(out.get("ok")) else "skill_repair_deps_failed",
|
||||
target_id=name,
|
||||
status="ok" if bool(out.get("ok")) else "fail",
|
||||
detail=out,
|
||||
)
|
||||
return {"ok": bool(out.get("ok")), "result": out}
|
||||
|
||||
@sk.post("/repair-deps-all")
|
||||
def api_skills_repair_deps_all(
|
||||
payload: dict[str, Any] | None = Body(default=None),
|
||||
authorization: str | None = Header(default=None),
|
||||
) -> dict[str, Any]:
|
||||
_payload = payload or {}
|
||||
store = SqliteStore(db_path())
|
||||
ctx = resolve_auth(store, authorization)
|
||||
_require_admin(ctx)
|
||||
items = list_skills_with_status(store=store)
|
||||
results: list[dict[str, Any]] = []
|
||||
ok_count = 0
|
||||
warn_count = 0
|
||||
fail_count = 0
|
||||
for it in items:
|
||||
name = str((it or {}).get("name") or "").strip()
|
||||
if not name:
|
||||
continue
|
||||
_audit(
|
||||
store,
|
||||
ctx,
|
||||
action="skill_repair_deps_started",
|
||||
target_id=name,
|
||||
status="start",
|
||||
detail={"batch": True},
|
||||
)
|
||||
out = repair_skill_dependencies(store=store, skill_name=name)
|
||||
warnings = list(out.get("warnings") or []) if isinstance(out, dict) else []
|
||||
if bool(out.get("ok")):
|
||||
if warnings:
|
||||
warn_count += 1
|
||||
else:
|
||||
ok_count += 1
|
||||
else:
|
||||
fail_count += 1
|
||||
_audit(
|
||||
store,
|
||||
ctx,
|
||||
action="skill_repair_deps_finished" if bool(out.get("ok")) else "skill_repair_deps_failed",
|
||||
target_id=name,
|
||||
status="ok" if bool(out.get("ok")) else "fail",
|
||||
detail={"batch": True, **(out if isinstance(out, dict) else {})},
|
||||
)
|
||||
results.append(
|
||||
{
|
||||
"name": name,
|
||||
"ok": bool(out.get("ok")) if isinstance(out, dict) else False,
|
||||
"warnings": warnings,
|
||||
"detail": str((out or {}).get("detail") or "") if isinstance(out, dict) else "",
|
||||
}
|
||||
)
|
||||
return {
|
||||
"ok": True,
|
||||
"summary": {
|
||||
"total": len(results),
|
||||
"ok_count": ok_count,
|
||||
"warn_count": warn_count,
|
||||
"fail_count": fail_count,
|
||||
},
|
||||
"items": results,
|
||||
}
|
||||
|
||||
@sk.post("/test-run")
|
||||
def api_skills_test_run(
|
||||
payload: dict[str, Any] | None = Body(default=None),
|
||||
|
|
@ -778,7 +891,7 @@ def include_skill_routes(
|
|||
"execution_checked_total": len(execution_checks),
|
||||
"execution_checks": execution_checks,
|
||||
"classification_counts": classification_counts,
|
||||
"market_provider": str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "clawhub"),
|
||||
"market_provider": normalize_skill_market_provider_setting(str(store.get_setting(_SKILL_MARKET_PROVIDER_KEY) or "")),
|
||||
}
|
||||
|
||||
router.include_router(sk)
|
||||
|
|
|
|||
|
|
@ -1087,6 +1087,16 @@ async function apiPost(path, body) {
|
|||
return data ?? {};
|
||||
}
|
||||
|
||||
/** Skills endpoints often return HTTP 200 with `{ ok: false, result: {...} }` on failure — treat as error for UX. */
|
||||
function assertSkillMutationOk(r, fallbackMessage) {
|
||||
if (!r || r.ok !== false) return;
|
||||
const res = r.result && typeof r.result === "object" ? r.result : {};
|
||||
const code = res.error_code != null ? String(res.error_code).trim() : "";
|
||||
const detail = res.detail != null ? String(res.detail).trim() : "";
|
||||
const msg = [code, detail].filter(Boolean).join(": ") || String(fallbackMessage || "skill operation failed");
|
||||
throw new Error(msg);
|
||||
}
|
||||
|
||||
async function apiRequest(method, path, body) {
|
||||
const url = resolveAdminApiUrl(path);
|
||||
const token = getStoredAuthToken();
|
||||
|
|
@ -7363,7 +7373,11 @@ async function renderSkills() {
|
|||
const skillModeStatus = el("div", { class: "muted", text: "" });
|
||||
const skillPromptModeCb = el("input", { type: "checkbox" });
|
||||
const skillToolcallModeCb = el("input", { type: "checkbox" });
|
||||
const marketQ = el("input", { class: "input", placeholder: "search ClawHub skills" });
|
||||
const skillMarketProviderSelect = el("select", { class: "input", style: "min-width:160px;" }, [
|
||||
el("option", { value: "clawhub", text: "clawhub (ClawHub)" }),
|
||||
el("option", { value: "cocoloop", text: "cocoloop (CocoLoop)" }),
|
||||
]);
|
||||
const marketQ = el("input", { class: "input", placeholder: "search skills (keyword)" });
|
||||
const marketLimitInp = el("input", { class: "input", placeholder: "limit", value: "40", style: "max-width:120px;" });
|
||||
const marketTbody = el("tbody");
|
||||
const marketDetailPre = el("pre", { class: "muted pre", text: "" });
|
||||
|
|
@ -7374,6 +7388,8 @@ async function renderSkills() {
|
|||
const r = await apiGet("/admin/api/skills/mode");
|
||||
skillPromptModeCb.checked = !!r.prompt_in_system;
|
||||
skillToolcallModeCb.checked = !!r.toolcall_enabled;
|
||||
const mp = String(r.market_provider || "clawhub").trim().toLowerCase();
|
||||
skillMarketProviderSelect.value = mp === "cocoloop" ? "cocoloop" : "clawhub";
|
||||
skillModeStatus.textContent = "";
|
||||
} catch (e) {
|
||||
skillModeStatus.textContent = `mode: ${String(e && e.message ? e.message : e)}`;
|
||||
|
|
@ -7385,10 +7401,13 @@ async function renderSkills() {
|
|||
const r = await apiPost("/admin/api/skills/mode", {
|
||||
prompt_in_system: !!skillPromptModeCb.checked,
|
||||
toolcall_enabled: !!skillToolcallModeCb.checked,
|
||||
market_provider: String(skillMarketProviderSelect.value || "clawhub").trim(),
|
||||
});
|
||||
skillPromptModeCb.checked = !!r.prompt_in_system;
|
||||
skillToolcallModeCb.checked = !!r.toolcall_enabled;
|
||||
skillModeStatus.textContent = `saved: prompt=${String(!!r.prompt_in_system)} toolcall=${String(!!r.toolcall_enabled)}`;
|
||||
const mp = String(r.market_provider || "clawhub").trim().toLowerCase();
|
||||
skillMarketProviderSelect.value = mp === "cocoloop" ? "cocoloop" : "clawhub";
|
||||
skillModeStatus.textContent = `saved: prompt=${String(!!r.prompt_in_system)} toolcall=${String(!!r.toolcall_enabled)} market=${String(skillMarketProviderSelect.value)}`;
|
||||
} catch (e) {
|
||||
skillModeStatus.textContent = `mode: ${String(e && e.message ? e.message : e)}`;
|
||||
}
|
||||
|
|
@ -7429,7 +7448,8 @@ async function renderSkills() {
|
|||
openSkillInstallModal(`Installing ${s}...`);
|
||||
try {
|
||||
const r = await apiPost("/admin/api/skills/market/install", { slug: s, version: version ? String(version) : undefined, overwrite: false });
|
||||
status.textContent = `install-clawhub success: ${JSON.stringify(r.result || {})}`;
|
||||
assertSkillMutationOk(r, "Market install failed");
|
||||
status.textContent = `install-market success: ${JSON.stringify(r.result || {})}`;
|
||||
marketStatus.textContent = `installed: ${s}`;
|
||||
await refreshSkillsState();
|
||||
finishSkillInstallModal(true, `${s} installed successfully.`);
|
||||
|
|
@ -7446,14 +7466,13 @@ async function renderSkills() {
|
|||
const slug = String(x.slug || "");
|
||||
const ver = String(x.version || "");
|
||||
const btnDetail = el("button", { class: "btn btn--small", text: "Detail", onclick: async () => await loadMarketDetail(slug) });
|
||||
const btnInstall = el("button", { class: "btn btn--small btn--primary", text: "Install", onclick: async () => await installFromMarket(slug, ver || undefined) });
|
||||
marketTbody.appendChild(
|
||||
el("tr", {}, [
|
||||
el("td", { text: slug }),
|
||||
el("td", { text: String(x.name || "") }),
|
||||
el("td", { text: ver }),
|
||||
el("td", { text: shortText(String(x.description || ""), 80) }),
|
||||
el("td", {}, [btnDetail, el("span", { style: "display:inline-block;width:6px" }), btnInstall]),
|
||||
el("td", {}, [btnDetail]),
|
||||
]),
|
||||
);
|
||||
});
|
||||
|
|
@ -7462,7 +7481,7 @@ async function renderSkills() {
|
|||
const btnMarketSearch = el("button", { class: "btn", text: "Search", onclick: async () => await loadMarket(marketQ.value) });
|
||||
const btnMarketLatest = el("button", { class: "btn", text: "Latest", onclick: async () => await loadMarket("") });
|
||||
const marketBox = el("details", { style: "margin:10px 0 14px 0;" }, [
|
||||
el("summary", { text: "ClawHub Market", style: "cursor:pointer;user-select:none;" }),
|
||||
el("summary", { text: "Skill market (ClawHub / CocoLoop)", style: "cursor:pointer;user-select:none;" }),
|
||||
el("div", { style: "height:8px" }),
|
||||
el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [marketQ, marketLimitInp, btnMarketSearch, btnMarketLatest]),
|
||||
marketStatus,
|
||||
|
|
@ -8400,6 +8419,24 @@ async function renderSkills() {
|
|||
openSkillTestRunModal(name);
|
||||
},
|
||||
});
|
||||
const repairDepsBtn = el("button", {
|
||||
class: "chat-sess-menu-item",
|
||||
text: "Repair deps",
|
||||
onclick: async () => {
|
||||
closeSkillActionMenu();
|
||||
try {
|
||||
status.textContent = `repair deps: ${name}...`;
|
||||
const r = await apiPost("/admin/api/skills/repair-deps", { name });
|
||||
assertSkillMutationOk(r, "Repair deps failed");
|
||||
status.textContent = `repair deps: ${JSON.stringify((r && r.result) || {}, null, 0)}`;
|
||||
await loadRows();
|
||||
await loadAudits();
|
||||
repaint();
|
||||
} catch (e) {
|
||||
status.textContent = `repair deps failed: ${String(e && e.message ? e.message : e)}`;
|
||||
}
|
||||
},
|
||||
});
|
||||
const uninstallBtn = el("button", {
|
||||
class: "chat-sess-menu-item",
|
||||
text: "Uninstall",
|
||||
|
|
@ -8409,6 +8446,7 @@ async function renderSkills() {
|
|||
try {
|
||||
status.textContent = `uninstalling: ${name}...`;
|
||||
const r = await apiPost("/admin/api/skills/uninstall", { name });
|
||||
assertSkillMutationOk(r, "Uninstall failed");
|
||||
status.textContent = `uninstall success: ${JSON.stringify((r && r.result) || {}, null, 0)}`;
|
||||
await loadRows();
|
||||
await loadAudits();
|
||||
|
|
@ -8428,6 +8466,7 @@ async function renderSkills() {
|
|||
const menu = el("div", { class: "chat-sess-menu-pop", style: "position:fixed;z-index:250;" }, [
|
||||
toggleBtn,
|
||||
testRunBtn,
|
||||
repairDepsBtn,
|
||||
uninstallBtn,
|
||||
]);
|
||||
const rect = ev.currentTarget.getBoundingClientRect();
|
||||
|
|
@ -8483,6 +8522,7 @@ async function renderSkills() {
|
|||
return;
|
||||
}
|
||||
const r = await apiPost("/admin/api/skills/retry-install", { source: src, target });
|
||||
assertSkillMutationOk(r, "Retry install failed");
|
||||
status.textContent = `retry: ${JSON.stringify(r.result || {})}`;
|
||||
await loadRows();
|
||||
await loadAudits();
|
||||
|
|
@ -8544,6 +8584,7 @@ async function renderSkills() {
|
|||
description: String(descInp.value || "").trim(),
|
||||
body_markdown: String(bodyInp.value || ""),
|
||||
});
|
||||
assertSkillMutationOk(r, "Create skill failed");
|
||||
status.textContent = `create: ${JSON.stringify(r.result || {})}`;
|
||||
await loadRows();
|
||||
await loadAudits();
|
||||
|
|
@ -8566,6 +8607,7 @@ async function renderSkills() {
|
|||
const r = await apiPost("/admin/api/skills/install-registry", {
|
||||
archive_url: String(regInp.value || "").trim(),
|
||||
});
|
||||
assertSkillMutationOk(r, "Registry install failed");
|
||||
status.textContent = `install-registry success: ${JSON.stringify(r.result || {})}`;
|
||||
await refreshSkillsState();
|
||||
finishSkillInstallModal(true, "Registry skill installed successfully.");
|
||||
|
|
@ -8591,6 +8633,7 @@ async function renderSkills() {
|
|||
const r = await apiPost("/admin/api/skills/install", {
|
||||
source_dir: String(localDirInp.value || "").trim(),
|
||||
});
|
||||
assertSkillMutationOk(r, "Local install failed");
|
||||
status.textContent = `install-local success: ${JSON.stringify(r.result || {})}`;
|
||||
await refreshSkillsState();
|
||||
finishSkillInstallModal(true, "Local skill installed successfully.");
|
||||
|
|
@ -8615,6 +8658,26 @@ async function renderSkills() {
|
|||
}
|
||||
},
|
||||
});
|
||||
const btnRepairDepsAll = el("button", {
|
||||
class: "btn",
|
||||
text: "Repair all deps",
|
||||
onclick: async () => {
|
||||
const prev = btnRepairDepsAll.textContent;
|
||||
btnRepairDepsAll.disabled = true;
|
||||
btnRepairDepsAll.textContent = "Repairing...";
|
||||
try {
|
||||
const r = await apiPost("/admin/api/skills/repair-deps-all", {});
|
||||
const s = r && typeof r.summary === "object" ? r.summary : {};
|
||||
status.textContent = `repair all deps: total=${Number(s.total || 0)} ok=${Number(s.ok_count || 0)} warn=${Number(s.warn_count || 0)} fail=${Number(s.fail_count || 0)}`;
|
||||
await refreshSkillsState();
|
||||
} catch (e) {
|
||||
status.textContent = `repair all deps failed: ${String(e && e.message ? e.message : e)}`;
|
||||
} finally {
|
||||
btnRepairDepsAll.disabled = false;
|
||||
btnRepairDepsAll.textContent = prev;
|
||||
}
|
||||
},
|
||||
});
|
||||
retryableOnlyCb.addEventListener("change", () => {
|
||||
localStorage.setItem(SKILL_AUDIT_RETRYABLE_ONLY_KEY, retryableOnlyCb.checked ? "1" : "0");
|
||||
repaint();
|
||||
|
|
@ -8650,6 +8713,10 @@ async function renderSkills() {
|
|||
el("div", { class: "row", style: "gap:8px;align-items:center;flex-wrap:wrap;margin-bottom:8px;" }, [
|
||||
el("label", { class: "row", style: "gap:6px;align-items:center;" }, [skillPromptModeCb, el("span", { text: "Prompt mode (inject SKILL.md)" })]),
|
||||
el("label", { class: "row", style: "gap:6px;align-items:center;" }, [skillToolcallModeCb, el("span", { text: "Toolcall mode (runtime as tools)" })]),
|
||||
el("label", { class: "row", style: "gap:6px;align-items:center;flex-wrap:wrap;" }, [
|
||||
el("span", { text: "Market (AIA_SKILL_MARKET_PROVIDER)" }),
|
||||
skillMarketProviderSelect,
|
||||
]),
|
||||
el("button", { class: "btn", text: "Save skill mode", onclick: saveSkillMode }),
|
||||
skillModeStatus,
|
||||
]),
|
||||
|
|
@ -8663,8 +8730,7 @@ async function renderSkills() {
|
|||
el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [nameInp, descInp]),
|
||||
el("div", { style: "margin-bottom:8px;" }, [bodyInp]),
|
||||
el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [btnCreate]),
|
||||
el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [regInp, btnInstallRegistry, btnRefresh]),
|
||||
el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [localDirInp, btnInstallLocal]),
|
||||
el("div", { class: "row", style: "gap:8px;flex-wrap:wrap;margin-bottom:8px;" }, [btnRefresh, btnRepairDepsAll]),
|
||||
el("div", { class: "table-wrap" }, [
|
||||
el("table", { class: "table table--compact" }, [
|
||||
el("thead", {}, [el("tr", {}, [
|
||||
|
|
|
|||
|
|
@ -475,6 +475,7 @@ body.theme-ds-body .card {
|
|||
flex: 1 1 auto;
|
||||
width: auto;
|
||||
max-width: min(max(0px, calc(50% + 2cm - 58px)), 100%);
|
||||
align-items: flex-start;
|
||||
}
|
||||
|
||||
.chat-msg-col--user {
|
||||
|
|
@ -573,7 +574,8 @@ body.theme-ds-body .card {
|
|||
|
||||
.chat-msg--assistant {
|
||||
align-self: flex-start;
|
||||
width: 100%;
|
||||
width: auto;
|
||||
max-width: 100%;
|
||||
background: rgba(255, 255, 255, 0.04);
|
||||
border: 1px solid var(--ds-border, rgba(255, 255, 255, 0.08));
|
||||
}
|
||||
|
|
|
|||
|
|
@ -696,6 +696,8 @@ class ToolExecutor:
|
|||
|
||||
results_by_id: dict[str, tuple[dict[str, Any], int]] = {}
|
||||
runnable_tool_uses: list[LLMToolCall] = []
|
||||
dedupe_alias_to_source: dict[str, str] = {}
|
||||
first_tool_call_id_by_signature: dict[str, str] = {}
|
||||
sig_seen: dict[str, int] = {}
|
||||
budget = max(1, min(int(signature_budget or 2), 8))
|
||||
for tc in tool_uses:
|
||||
|
|
@ -796,6 +798,19 @@ class ToolExecutor:
|
|||
)
|
||||
continue
|
||||
sig_seen[sig] = count + 1
|
||||
source_tool_call_id = str(first_tool_call_id_by_signature.get(sig) or "").strip()
|
||||
if source_tool_call_id:
|
||||
dedupe_alias_to_source[str(tc.id or "")] = source_tool_call_id
|
||||
_trace(
|
||||
"tool_cache_hit_same_round",
|
||||
{
|
||||
"tool_name": tc.name,
|
||||
"tool_call_id": str(tc.id or ""),
|
||||
"source_tool_call_id": source_tool_call_id,
|
||||
},
|
||||
)
|
||||
continue
|
||||
first_tool_call_id_by_signature[sig] = str(tc.id or "")
|
||||
runnable_tool_uses.append(tc)
|
||||
|
||||
for batch in partition_tool_use_batches(runnable_tool_uses, ctx.tools):
|
||||
|
|
@ -825,6 +840,10 @@ class ToolExecutor:
|
|||
},
|
||||
)
|
||||
|
||||
for tool_call_id, source_tool_call_id in dedupe_alias_to_source.items():
|
||||
if source_tool_call_id in results_by_id:
|
||||
results_by_id[tool_call_id] = results_by_id[source_tool_call_id]
|
||||
|
||||
tool_messages: list[dict[str, Any]] = []
|
||||
for tc in tool_uses:
|
||||
_check_stop()
|
||||
|
|
|
|||
|
|
@ -1,10 +1,16 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import ast
|
||||
import importlib.util
|
||||
import json
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tarfile
|
||||
import tempfile
|
||||
import time
|
||||
import urllib.parse
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
import zipfile
|
||||
from dataclasses import dataclass
|
||||
|
|
@ -27,6 +33,33 @@ from oclaw.runtime.skills import (
|
|||
_DISABLED_SKILLS_KEY = "AIA_SKILL_DISABLED_NAMES"
|
||||
_AUTO_INSTALL_KEY = "AIA_SKILL_AUTO_INSTALL_ENABLED"
|
||||
_AUTO_ENABLE_TRUSTED_KEY = "AIA_SKILL_AUTO_ENABLE_TRUSTED"
|
||||
_AUTO_INSTALL_DEPS_KEY = "AIA_SKILL_AUTO_INSTALL_DEPS_ENABLED"
|
||||
_ARCHIVE_DOWNLOAD_TIMEOUT_S = 45
|
||||
_ARCHIVE_DOWNLOAD_MAX_RETRIES = 3
|
||||
|
||||
|
||||
def _download_archive_bytes(url: str) -> bytes:
|
||||
req = urllib.request.Request(
|
||||
url,
|
||||
headers={
|
||||
"User-Agent": "Oclaw-SkillInstaller/1.0 (+https://clawhub.ai)",
|
||||
"Accept": "*/*",
|
||||
},
|
||||
)
|
||||
last_exc: Exception | None = None
|
||||
for attempt in range(1, _ARCHIVE_DOWNLOAD_MAX_RETRIES + 1):
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=_ARCHIVE_DOWNLOAD_TIMEOUT_S) as resp:
|
||||
return resp.read()
|
||||
except Exception as exc:
|
||||
last_exc = exc
|
||||
code = getattr(exc, "code", None)
|
||||
retryable_http = isinstance(code, int) and code in {408, 429, 500, 502, 503, 504}
|
||||
retryable_net = isinstance(exc, urllib.error.URLError) or isinstance(exc, TimeoutError)
|
||||
if attempt >= _ARCHIVE_DOWNLOAD_MAX_RETRIES or not (retryable_http or retryable_net):
|
||||
raise
|
||||
time.sleep(0.8 * attempt)
|
||||
raise last_exc or RuntimeError("download_failed_unknown")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
|
|
@ -85,6 +118,144 @@ def skill_auto_enable_trusted_enabled(store: Any) -> bool:
|
|||
return _truthy(raw)
|
||||
|
||||
|
||||
def skill_auto_install_deps_enabled(store: Any) -> bool:
|
||||
try:
|
||||
raw = str(store.get_setting(_AUTO_INSTALL_DEPS_KEY) or "").strip()
|
||||
except Exception:
|
||||
raw = ""
|
||||
# Default ON to reduce first-run dependency issues for newly installed skills.
|
||||
if not raw:
|
||||
return True
|
||||
return _truthy(raw)
|
||||
|
||||
|
||||
def _scan_dependency_manifests(skill_dir: Path) -> dict[str, list[Path]]:
|
||||
req_files: list[Path] = []
|
||||
pkg_files: list[Path] = []
|
||||
try:
|
||||
for p in skill_dir.rglob("requirements.txt"):
|
||||
if p.is_file():
|
||||
req_files.append(p)
|
||||
for p in skill_dir.rglob("package.json"):
|
||||
if p.is_file():
|
||||
pkg_files.append(p)
|
||||
except Exception:
|
||||
return {"requirements": [], "package_json": []}
|
||||
req_files = sorted(req_files)
|
||||
pkg_files = sorted(pkg_files)
|
||||
return {"requirements": req_files, "package_json": pkg_files}
|
||||
|
||||
|
||||
def _run_dep_install(command: list[str], *, cwd: Path) -> tuple[bool, str]:
|
||||
try:
|
||||
cp = subprocess.run(
|
||||
command,
|
||||
cwd=str(cwd),
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=600,
|
||||
check=False,
|
||||
)
|
||||
except Exception as exc:
|
||||
return False, f"{type(exc).__name__}"
|
||||
if int(cp.returncode) == 0:
|
||||
return True, ""
|
||||
err = str(cp.stderr or cp.stdout or "").strip()
|
||||
if err:
|
||||
err = err[:160].replace("\n", " ")
|
||||
return False, err or f"exit_{cp.returncode}"
|
||||
|
||||
|
||||
def _auto_install_skill_dependencies(*, store: Any, skill_dir: Path) -> tuple[bool, str]:
|
||||
if not skill_auto_install_deps_enabled(store):
|
||||
return True, ""
|
||||
manifests = _scan_dependency_manifests(skill_dir)
|
||||
req_files = manifests.get("requirements") or []
|
||||
pkg_files = manifests.get("package_json") or []
|
||||
if not req_files and not pkg_files:
|
||||
return True, ""
|
||||
|
||||
failures: list[str] = []
|
||||
# Python dependencies
|
||||
for req in req_files:
|
||||
ok, detail = _run_dep_install([sys.executable, "-m", "pip", "install", "-r", str(req)], cwd=req.parent)
|
||||
if not ok:
|
||||
failures.append(f"pip:{req.name}:{detail}")
|
||||
|
||||
# Node dependencies (only when package.json declares deps)
|
||||
for pkg in pkg_files:
|
||||
try:
|
||||
obj = json.loads(pkg.read_text(encoding="utf-8"))
|
||||
except Exception:
|
||||
obj = {}
|
||||
deps = obj.get("dependencies") if isinstance(obj, dict) else None
|
||||
if not isinstance(deps, dict) or not deps:
|
||||
continue
|
||||
ok, detail = _run_dep_install(["npm", "install", "--omit=dev"], cwd=pkg.parent)
|
||||
if not ok:
|
||||
failures.append(f"npm:{pkg.name}:{detail}")
|
||||
if failures:
|
||||
return False, "; ".join(failures[:4])
|
||||
return True, ""
|
||||
|
||||
|
||||
def _collect_python_import_roots(skill_dir: Path) -> set[str]:
|
||||
names: set[str] = set()
|
||||
for py in skill_dir.rglob("*.py"):
|
||||
if not py.is_file():
|
||||
continue
|
||||
try:
|
||||
tree = ast.parse(py.read_text(encoding="utf-8"), filename=str(py))
|
||||
except Exception:
|
||||
continue
|
||||
for node in ast.walk(tree):
|
||||
if isinstance(node, ast.Import):
|
||||
for alias in node.names:
|
||||
root = str(alias.name or "").split(".", 1)[0].strip()
|
||||
if root:
|
||||
names.add(root)
|
||||
elif isinstance(node, ast.ImportFrom):
|
||||
if int(node.level or 0) > 0:
|
||||
continue
|
||||
root = str(node.module or "").split(".", 1)[0].strip()
|
||||
if root:
|
||||
names.add(root)
|
||||
return names
|
||||
|
||||
|
||||
def _local_python_module_roots(skill_dir: Path) -> set[str]:
|
||||
roots: set[str] = set()
|
||||
for py in skill_dir.rglob("*.py"):
|
||||
if py.is_file():
|
||||
roots.add(py.stem)
|
||||
for p in skill_dir.rglob("*"):
|
||||
if p.is_dir() and (p / "__init__.py").exists():
|
||||
roots.add(p.name)
|
||||
return {x for x in roots if x and x != "__init__"}
|
||||
|
||||
|
||||
def _auto_probe_and_install_python_imports(*, store: Any, skill_dir: Path) -> tuple[bool, str]:
|
||||
if not skill_auto_install_deps_enabled(store):
|
||||
return True, ""
|
||||
imports = _collect_python_import_roots(skill_dir)
|
||||
if not imports:
|
||||
return True, ""
|
||||
local_roots = _local_python_module_roots(skill_dir)
|
||||
stdlib = set(getattr(sys, "stdlib_module_names", set()) or set())
|
||||
missing: list[str] = []
|
||||
for name in sorted(imports):
|
||||
if name in local_roots or name in stdlib:
|
||||
continue
|
||||
if importlib.util.find_spec(name) is None:
|
||||
missing.append(name)
|
||||
if not missing:
|
||||
return True, ""
|
||||
ok, detail = _run_dep_install([sys.executable, "-m", "pip", "install", *missing], cwd=skill_dir)
|
||||
if not ok:
|
||||
return False, f"probe_pip:{detail}"
|
||||
return True, ""
|
||||
|
||||
|
||||
def _get_disabled_names(store: Any) -> set[str]:
|
||||
try:
|
||||
raw = str(store.get_setting(_DISABLED_SKILLS_KEY) or "").strip()
|
||||
|
|
@ -296,6 +467,7 @@ def install_skill_from_local_dir(
|
|||
source_dir: str | Path,
|
||||
overwrite: bool = False,
|
||||
skills_root: str | Path | None = None,
|
||||
auto_bind: bool = False,
|
||||
) -> SkillInstallResult:
|
||||
src = Path(source_dir).resolve()
|
||||
ok, detail = scan_skill_source_dir(src)
|
||||
|
|
@ -317,8 +489,49 @@ def install_skill_from_local_dir(
|
|||
shutil.rmtree(target)
|
||||
shutil.copytree(src, target)
|
||||
set_skill_enabled(store=store, skill_name=manifest.name, enabled=True)
|
||||
auto_enabled = False
|
||||
binding_roles: tuple[str, ...] = ()
|
||||
if auto_bind:
|
||||
binding_root = root.parent if str(root.name).strip().lower() == "_workspace" else root
|
||||
auto_enabled, binding_roles = _apply_auto_enable_binding(store=store, skill_name=manifest.name, skills_root=binding_root)
|
||||
deps_ok, deps_detail = _auto_install_skill_dependencies(store=store, skill_dir=target)
|
||||
if not deps_ok:
|
||||
# Keep install successful; surface dependency warning for operator follow-up.
|
||||
ec, rt = _classify_install_detail("installed")
|
||||
return SkillInstallResult(
|
||||
ok=True,
|
||||
name=manifest.name,
|
||||
target_dir=str(target),
|
||||
detail=f"installed_with_dependency_warnings:{deps_detail}",
|
||||
error_code=ec,
|
||||
retryable=rt,
|
||||
auto_enabled=auto_enabled,
|
||||
binding_applied_roles=binding_roles,
|
||||
)
|
||||
probe_ok, probe_detail = _auto_probe_and_install_python_imports(store=store, skill_dir=target)
|
||||
if not probe_ok:
|
||||
ec, rt = _classify_install_detail("installed")
|
||||
return SkillInstallResult(
|
||||
ok=True,
|
||||
name=manifest.name,
|
||||
target_dir=str(target),
|
||||
detail=f"installed_with_dependency_warnings:{probe_detail}",
|
||||
error_code=ec,
|
||||
retryable=rt,
|
||||
auto_enabled=auto_enabled,
|
||||
binding_applied_roles=binding_roles,
|
||||
)
|
||||
ec, rt = _classify_install_detail("installed")
|
||||
return SkillInstallResult(ok=True, name=manifest.name, target_dir=str(target), detail="installed", error_code=ec, retryable=rt)
|
||||
return SkillInstallResult(
|
||||
ok=True,
|
||||
name=manifest.name,
|
||||
target_dir=str(target),
|
||||
detail="installed",
|
||||
error_code=ec,
|
||||
retryable=rt,
|
||||
auto_enabled=auto_enabled,
|
||||
binding_applied_roles=binding_roles,
|
||||
)
|
||||
|
||||
|
||||
def install_skill_from_registry_archive(
|
||||
|
|
@ -327,6 +540,7 @@ def install_skill_from_registry_archive(
|
|||
archive_url: str,
|
||||
overwrite: bool = False,
|
||||
skills_root: str | Path | None = None,
|
||||
auto_bind: bool = False,
|
||||
) -> SkillInstallResult:
|
||||
url = str(archive_url or "").strip()
|
||||
if not url:
|
||||
|
|
@ -339,15 +553,7 @@ def install_skill_from_registry_archive(
|
|||
return SkillInstallResult(ok=False, name="", target_dir="", detail="unsupported_url_scheme", error_code=ec, retryable=rt)
|
||||
tmp_file = Path(tempfile.mkstemp(prefix="skill_pkg_", suffix=".bin")[1])
|
||||
try:
|
||||
req = urllib.request.Request(
|
||||
url,
|
||||
headers={
|
||||
"User-Agent": "Oclaw-SkillInstaller/1.0 (+https://clawhub.ai)",
|
||||
"Accept": "*/*",
|
||||
},
|
||||
)
|
||||
with urllib.request.urlopen(req, timeout=20) as resp:
|
||||
data = resp.read()
|
||||
data = _download_archive_bytes(url)
|
||||
if not data:
|
||||
ec, rt = _classify_install_detail("empty_archive")
|
||||
return SkillInstallResult(ok=False, name="", target_dir="", detail="empty_archive", error_code=ec, retryable=rt)
|
||||
|
|
@ -369,7 +575,13 @@ def install_skill_from_registry_archive(
|
|||
ec, rt = _classify_install_detail("skill_md_missing")
|
||||
return SkillInstallResult(ok=False, name="", target_dir="", detail="skill_md_missing", error_code=ec, retryable=rt)
|
||||
chosen = sorted(candidates, key=lambda x: len(x.parts))[0]
|
||||
return install_skill_from_local_dir(store=store, source_dir=chosen, overwrite=overwrite, skills_root=skills_root)
|
||||
return install_skill_from_local_dir(
|
||||
store=store,
|
||||
source_dir=chosen,
|
||||
overwrite=overwrite,
|
||||
skills_root=skills_root,
|
||||
auto_bind=auto_bind,
|
||||
)
|
||||
except Exception as exc:
|
||||
detail = f"download_failed:{type(exc).__name__}"
|
||||
code = getattr(exc, "code", None)
|
||||
|
|
@ -569,6 +781,36 @@ def auto_install_skill_from_payload(
|
|||
return SkillInstallResult(ok=False, name=name, target_dir=str(target), detail=detail, error_code=ec, retryable=rt)
|
||||
|
||||
|
||||
def repair_skill_dependencies(
|
||||
*,
|
||||
store: Any,
|
||||
skill_name: str,
|
||||
skills_root: str | Path | None = None,
|
||||
) -> dict[str, Any]:
|
||||
nm = str(skill_name or "").strip()
|
||||
if not nm:
|
||||
return {"ok": False, "error_code": "name_required", "error": "name_required"}
|
||||
manifests = list(discover_workspace_skill_manifests(skills_root))
|
||||
mf = next((m for m in manifests if str(m.name or "").strip() == nm), None)
|
||||
if mf is None:
|
||||
return {"ok": False, "error_code": "skill_not_found", "error": "skill_not_found", "name": nm}
|
||||
skill_dir = Path(str(mf.skill_dir or "")).resolve()
|
||||
deps_ok, deps_detail = _auto_install_skill_dependencies(store=store, skill_dir=skill_dir)
|
||||
probe_ok, probe_detail = _auto_probe_and_install_python_imports(store=store, skill_dir=skill_dir)
|
||||
warnings: list[str] = []
|
||||
if not deps_ok and deps_detail:
|
||||
warnings.append(f"manifest:{deps_detail}")
|
||||
if not probe_ok and probe_detail:
|
||||
warnings.append(f"probe:{probe_detail}")
|
||||
return {
|
||||
"ok": True,
|
||||
"name": nm,
|
||||
"skill_dir": str(skill_dir),
|
||||
"warnings": warnings,
|
||||
"detail": "ok" if not warnings else f"warnings:{'; '.join(warnings[:4])}",
|
||||
}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"SkillInstallResult",
|
||||
"auto_install_skill_from_payload",
|
||||
|
|
@ -577,6 +819,7 @@ __all__ = [
|
|||
"install_skill_from_local_dir",
|
||||
"install_skill_from_registry_archive",
|
||||
"list_skills_with_status",
|
||||
"repair_skill_dependencies",
|
||||
"set_skill_enabled",
|
||||
"skill_auto_install_enabled",
|
||||
"uninstall_skill",
|
||||
|
|
|
|||
|
|
@ -25,6 +25,12 @@
|
|||
- 若设置了环境变量 `AIA_SKILLS_ROOT`,以该变量为准。
|
||||
- 为兼容旧工程,仍可回退读取旧路径 `oclaw/runtime/skills/`(如存在)。
|
||||
|
||||
## 技能市场(安装来源)
|
||||
|
||||
- 租户设置 **`AIA_SKILL_MARKET_PROVIDER`**:`clawhub`(默认)或 **`cocoloop`**,由 `runtime/skills_market.get_market_adapter` 选择适配器;Admin「市场搜索 / 按 slug 安装」共用同一套路由。
|
||||
- ClawHub:见 `runtime/tools/skills/clawhub_client.py`(`AIA_CLAWHUB_*` / `CLAWHUB_*`)。公开 API 说明可参考 [openclaw/clawhub CLI 文档](https://github.com/openclaw/clawhub/blob/main/docs/cli.md)。
|
||||
- CocoLoop:`runtime/tools/skills/cocoloop_client.py`,默认 API 基址 `https://api.cocoloop.com`,可用 **`AIA_COCOLOOP_API_BASE`** 覆盖。
|
||||
|
||||
## 推荐实用 Skills(workspace)
|
||||
|
||||
以下为当前已落地并可直接在 Admin `Test run` 使用的实用技能:
|
||||
|
|
|
|||
97
runtime/skills/_workspace/SkillScan/SKILL.md
Normal file
97
runtime/skills/_workspace/SkillScan/SKILL.md
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
---
|
||||
name: SkillScan
|
||||
metadata:
|
||||
version: "1.1.6"
|
||||
description: >
|
||||
Security gate for skills. Every new skill MUST pass SkillScan before use.
|
||||
Activate on any install, load, add, evaluate, or safety question about a skill.
|
||||
On first load, run first-run to scan all existing skills.
|
||||
Blocks HIGH/CRITICAL skills. No exceptions.
|
||||
---
|
||||
|
||||
# SkillScan v1.1.6
|
||||
|
||||
Security check for skill packages. Every new skill must pass this scan before use.
|
||||
|
||||
Implemented in `scripts/scanner.py`. Supports Windows / macOS / Linux.
|
||||
|
||||
Do NOT generate report files inside the skill directory.
|
||||
|
||||
---
|
||||
|
||||
## When to Activate
|
||||
|
||||
Run SkillScan whenever a skill is installed, loaded, added, evaluated, or questioned for safety — regardless of method or phrasing.
|
||||
|
||||
Trigger examples: install, load, add, use, set up, "is this safe", upload .zip/.skill, paste SKILL.md, mention unknown skill name, install from hub/store.
|
||||
|
||||
---
|
||||
|
||||
## First Load
|
||||
|
||||
Check `scripts/.first_run_done`:
|
||||
|
||||
- **Not found** → **Tell user** SkillScan is installed, ask to scan all existing skills:
|
||||
```bash
|
||||
python <skill_scanner_dir>/scripts/scanner.py first-run
|
||||
```
|
||||
`.first_run_done` is created automatically after completion.
|
||||
|
||||
- **Found** → Skip, operate normally.
|
||||
|
||||
---
|
||||
|
||||
## Commands
|
||||
|
||||
| Command | Usage |
|
||||
|---------|-------|
|
||||
| `scanner.py scan <path>` | Scan a single skill (.zip or directory) |
|
||||
| `scanner.py scan-all` | Scan all installed skills |
|
||||
| `scanner.py first-run` | First-time full scan |
|
||||
| `scanner.py upgrade` | Manual upgrade |
|
||||
|
||||
---
|
||||
|
||||
## Scan Rules
|
||||
|
||||
- **.zip files** → Scan BEFORE installation. Block if fails.
|
||||
- **Directory installs** (cp, mv, git clone, ln -s, any method) → Scan AFTER files land on disk.
|
||||
- **Remote installs** (clawhub, skillhub, npx skills add, etc.) → Scan immediately after install.
|
||||
- **Unknown skills** → If user mentions a skill you haven't seen, scan it.
|
||||
|
||||
---
|
||||
|
||||
## Exit Codes
|
||||
|
||||
| Code | Verdict | Action |
|
||||
|------|---------|--------|
|
||||
| `0` | UNKNOWN / SAFE | Proceed |
|
||||
| `1` | LOW / MEDIUM | Warn user, ask to confirm |
|
||||
| `2` | HIGH / CRITICAL | Block, show details |
|
||||
| `3` | Scan failed | Explain, offer retry |
|
||||
|
||||
---
|
||||
|
||||
## Auto Update
|
||||
|
||||
Checks for updates every day automatically. Silent, no user action needed. Manual: `scanner.py upgrade`.
|
||||
|
||||
---
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `SKILL_SCANNER_UPDATE_URL` | Custom update source (optional) |
|
||||
|
||||
---
|
||||
|
||||
## API
|
||||
|
||||
Base URL: `https://skillscan.tokauth.com`
|
||||
|
||||
| Step | Method | Path |
|
||||
|------|--------|------|
|
||||
| ① Cache lookup | GET | `/oapi/v1/skill-scan/search?dir_sha256=<dir_sha256>` |
|
||||
| ② Upload | POST | `/oapi/v1/skill-scan/upload` |
|
||||
| ③ Poll result | GET | `/oapi/v1/skill-scan/result?task_no=<task_no>` (poll every 20s, max 180s) |
|
||||
6
runtime/skills/_workspace/SkillScan/_meta.json
Normal file
6
runtime/skills/_workspace/SkillScan/_meta.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn791cyx98pcsezkh5088g8jxn84c7mm",
|
||||
"slug": "skillscan",
|
||||
"version": "1.1.6",
|
||||
"publishedAt": 1776650587310
|
||||
}
|
||||
959
runtime/skills/_workspace/SkillScan/scripts/scanner.py
Normal file
959
runtime/skills/_workspace/SkillScan/scripts/scanner.py
Normal file
|
|
@ -0,0 +1,959 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
SkillScan v1.1.5 — OpenClaw Skill security scanner.
|
||||
Supports Windows / macOS / Linux. All temp files use the standard tempfile module.
|
||||
|
||||
Usage (invoked by the agent via bash):
|
||||
python scanner.py first-run # First install: list installed skills and ask to scan
|
||||
python scanner.py scan <path> # Scan a single skill (.zip or directory)
|
||||
python scanner.py scan-all # Scan all installed skills
|
||||
python scanner.py upgrade # Auto-upgrade
|
||||
"""
|
||||
|
||||
import sys, os, json, time, zipfile, hashlib, shutil, tempfile, uuid, platform, base64
|
||||
import urllib.request, urllib.error, urllib.parse
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Configuration
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
SCANNER_VERSION = "1.1.5"
|
||||
|
||||
BASE_URL = "https://skillscan.tokauth.com"
|
||||
API_SEARCH = f"{BASE_URL}/oapi/v1/skill-scan/search"
|
||||
API_UPLOAD = f"{BASE_URL}/oapi/v1/skill-scan/upload"
|
||||
API_RESULT = f"{BASE_URL}/oapi/v1/skill-scan/result"
|
||||
UPDATE_URL = os.environ.get("SKILL_SCANNER_UPDATE_URL",
|
||||
f"{BASE_URL}/downloads/SkillScan/manifest")
|
||||
|
||||
POLL_INTERVAL = 20 # Poll interval (seconds)
|
||||
POLL_TIMEOUT = 180 # Max wait time (seconds)
|
||||
|
||||
# First-run marker file (in the same directory as scanner.py)
|
||||
STATE_FILE = Path(__file__).parent / ".first_run_done"
|
||||
|
||||
# Auto-update check marker file and interval (7 days)
|
||||
LAST_UPDATE_CHECK_FILE = Path(__file__).parent / ".last_update_check"
|
||||
AUTO_UPDATE_INTERVAL = 1 * 24 * 3600 # 1 day (seconds)
|
||||
|
||||
# Client info file (generated on first run, reused afterwards)
|
||||
CLIENT_INFO_FILE = Path(__file__).parent / ".client_info"
|
||||
|
||||
# Files and directories to skip during scanning, hashing, and packing
|
||||
SKIP_FILES = {".first_run_done", ".last_update_check", ".client_info", "cloud_report.json", ".DS_Store"}
|
||||
SKIP_DIRS = {".git", "__pycache__", ".venv", "node_modules", ".idea", ".vscode", ".clawhub"}
|
||||
|
||||
# Resolve the root directory of SkillScan itself (parent of scripts/)
|
||||
SELF_ROOT = Path(__file__).parent.parent.resolve()
|
||||
|
||||
|
||||
# Skill installation paths (cross-platform)
|
||||
def skill_install_paths():
|
||||
# type: () -> list
|
||||
"""Auto-enumerate OpenClaw and local skill paths across platforms."""
|
||||
home = Path.home()
|
||||
oc_dir = home / ".openclaw"
|
||||
candidates = [
|
||||
# OpenClaw standard paths
|
||||
oc_dir / "skills",
|
||||
oc_dir / "workspace/skills",
|
||||
# Shared agent skill paths
|
||||
home / ".agents/skills",
|
||||
home / ".config/agents/skills",
|
||||
# Agent-specific global paths
|
||||
home / ".gemini/antigravity/skills",
|
||||
home / ".gemini/skills",
|
||||
home / ".augment/skills",
|
||||
home / ".claude/skills",
|
||||
home / ".codex/skills",
|
||||
home / ".commandcode/skills",
|
||||
home / ".continue/skills",
|
||||
home / ".snowflake/cortex/skills",
|
||||
home / ".config/crush/skills",
|
||||
home / ".cursor/skills",
|
||||
home / ".deepagents/agent/skills",
|
||||
home / ".factory/skills",
|
||||
home / ".firebender/skills",
|
||||
home / ".copilot/skills",
|
||||
home / ".config/goose/skills",
|
||||
home / ".junie/skills",
|
||||
home / ".iflow/skills",
|
||||
home / ".kilocode/skills",
|
||||
home / ".kiro/skills",
|
||||
home / ".kode/skills",
|
||||
home / ".mcpjam/skills",
|
||||
home / ".vibe/skills",
|
||||
home / ".mux/skills",
|
||||
home / ".config/opencode/skills",
|
||||
home / ".openhands/skills",
|
||||
home / ".pi/agent/skills",
|
||||
home / ".qoder/skills",
|
||||
home / ".qwen/skills",
|
||||
home / ".roo/skills",
|
||||
home / ".trae/skills",
|
||||
home / ".trae-cn/skills",
|
||||
home / ".codeium/windsurf/skills",
|
||||
home / ".zencoder/skills",
|
||||
home / ".neovate/skills",
|
||||
home / ".pochi/skills",
|
||||
home / ".adal/skills",
|
||||
home / ".npm-global/lib/node_modules/openclaw/skills",
|
||||
# Container default paths
|
||||
Path("/mnt/skills/public"),
|
||||
Path("/mnt/skills/private"),
|
||||
Path("/mnt/skills/user"),
|
||||
# User dev/download paths
|
||||
home / "Downloads/skills",
|
||||
]
|
||||
|
||||
# Windows-specific paths
|
||||
if os.name == "nt":
|
||||
appdata = os.environ.get("APPDATA")
|
||||
if appdata:
|
||||
candidates.append(Path(appdata) / "OpenClaw/skills")
|
||||
candidates.append(Path(appdata) / "Programs/LobsterAI/resources/SKILLs")
|
||||
|
||||
# Dynamically scan extensions: .openclaw/extensions/{xxxx}/skills
|
||||
if oc_dir.exists():
|
||||
ext_root = oc_dir / "extensions"
|
||||
if ext_root.exists():
|
||||
for sub in ext_root.iterdir():
|
||||
if sub.is_dir():
|
||||
s_dir = sub / "skills"
|
||||
if s_dir.exists():
|
||||
candidates.append(s_dir)
|
||||
|
||||
# Include script run path and workspace
|
||||
candidates.append(Path.cwd() / "skills")
|
||||
candidates.append(Path(__file__).parent.parent / "skills")
|
||||
|
||||
# Deduplicate and filter non-existent paths
|
||||
seen = set()
|
||||
result = []
|
||||
for p in candidates:
|
||||
try:
|
||||
abs_p = p.resolve()
|
||||
if abs_p.exists() and abs_p not in seen:
|
||||
result.append(p)
|
||||
seen.add(abs_p)
|
||||
except Exception:
|
||||
continue
|
||||
return result
|
||||
|
||||
RISK_EMOJI = {"SAFE":"✅","LOW":"⚠️ ","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️ "}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Client Info (X-Client-Info)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _get_mac_address():
|
||||
"""Try to get the MAC address; return empty string on failure."""
|
||||
try:
|
||||
import uuid as _uuid
|
||||
mac_int = _uuid.getnode()
|
||||
# getnode() returns a random value (bit 8 set) when it can't get the real MAC
|
||||
if (mac_int >> 40) & 1:
|
||||
return ""
|
||||
mac_str = ":".join(("%012X" % mac_int)[i:i+2] for i in range(0, 12, 2))
|
||||
return mac_str
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _build_client_info():
|
||||
"""Build client info dict and persist to file; reuse on subsequent runs."""
|
||||
# If a record file already exists, read it
|
||||
if CLIENT_INFO_FILE.exists():
|
||||
try:
|
||||
data = json.loads(CLIENT_INFO_FILE.read_text(encoding="utf-8"))
|
||||
if data.get("client_id"):
|
||||
return data
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# First run: generate new client info
|
||||
info = {
|
||||
"client_id": str(uuid.uuid4()),
|
||||
"os": platform.system() or "",
|
||||
"platform": platform.machine() or "",
|
||||
"os_version": platform.release() or "",
|
||||
"client": "SkillScanner/%s" % SCANNER_VERSION,
|
||||
}
|
||||
|
||||
mac = _get_mac_address()
|
||||
if mac:
|
||||
info["mac"] = mac
|
||||
|
||||
# Python version as extra
|
||||
info["extra"] = {
|
||||
"python": platform.python_version(),
|
||||
}
|
||||
|
||||
# Persist
|
||||
try:
|
||||
CLIENT_INFO_FILE.write_text(
|
||||
json.dumps(info, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8"
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return info
|
||||
|
||||
|
||||
def _get_client_info_header():
|
||||
"""Return Base64-encoded X-Client-Info header value; empty string on failure."""
|
||||
try:
|
||||
info = _build_client_info()
|
||||
json_str = json.dumps(info, ensure_ascii=False)
|
||||
encoded = base64.b64encode(json_str.encode("utf-8")).decode("ascii")
|
||||
return encoded
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Output Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def banner(title: str):
|
||||
w = 58
|
||||
print(f"\n{'═'*w}")
|
||||
print(f" {title}")
|
||||
print(f"{'═'*w}")
|
||||
|
||||
def divider(title: str = ""):
|
||||
if title:
|
||||
print(f"\n ── {title} {'─'*(48-len(title))}")
|
||||
else:
|
||||
print(f" {'─'*52}")
|
||||
|
||||
def log(msg: str):
|
||||
print(f" {msg}", flush=True)
|
||||
|
||||
def ask(prompt: str) -> str:
|
||||
"""Read user input (compatible with non-interactive environments)."""
|
||||
try:
|
||||
return input(f"\n {prompt} ").strip()
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# HTTP Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def http_get(url: str) -> dict:
|
||||
req = urllib.request.Request(url)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
def http_post(url: str, payload: dict) -> dict:
|
||||
headers = {"Content-Type": "application/json"}
|
||||
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Skill Utilities
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def skill_name_from_dir(skill_dir: Path) -> str:
|
||||
md = skill_dir / "SKILL.md"
|
||||
if md.exists():
|
||||
for line in md.read_text(encoding="utf-8", errors="replace").splitlines():
|
||||
s = line.strip()
|
||||
if s.startswith("name:"):
|
||||
return s.split(":", 1)[1].strip().strip("\"'")
|
||||
return skill_dir.name
|
||||
|
||||
def sha256_of(path: Path) -> str:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
def calculate_dir_sha256(directory: Path) -> str:
|
||||
"""Calculate SHA256 hash of a skill directory (based on all file contents + relative paths).
|
||||
Excludes _meta.json and files/dirs in SKIP_FILES/SKIP_DIRS."""
|
||||
file_hashes = []
|
||||
for file_path in sorted(directory.rglob('*')):
|
||||
if not file_path.is_file():
|
||||
continue
|
||||
if file_path.name == '_meta.json':
|
||||
continue
|
||||
rel = file_path.relative_to(directory)
|
||||
if any(part in SKIP_DIRS for part in rel.parts):
|
||||
continue
|
||||
if file_path.name in SKIP_FILES:
|
||||
continue
|
||||
rel_path = str(rel)
|
||||
file_hash = hashlib.sha256()
|
||||
file_hash.update(rel_path.encode('utf-8'))
|
||||
file_hash.update(b'\x00')
|
||||
with open(file_path, 'rb') as f:
|
||||
for chunk in iter(lambda: f.read(8192), b''):
|
||||
file_hash.update(chunk)
|
||||
file_hashes.append(file_hash.hexdigest())
|
||||
file_hashes.sort()
|
||||
final_hash = hashlib.sha256()
|
||||
for h in file_hashes:
|
||||
final_hash.update(h.encode('utf-8'))
|
||||
final_hash.update(b'\x00')
|
||||
return final_hash.hexdigest()
|
||||
|
||||
def collect_files(skill_dir: Path) -> dict:
|
||||
"""Collect files for scanning, skipping redundant or sensitive directories."""
|
||||
exts = {".md",".py",".js",".ts",".sh",".yaml",".yml",".json",".txt"}
|
||||
out = {}
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
if p.suffix.lower() in exts or p.name == "SKILL.md":
|
||||
try:
|
||||
out[str(p.relative_to(skill_dir))] = \
|
||||
p.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
def pack_zip(skill_dir: Path) -> bytes:
|
||||
"""Pack a skill directory into a zip byte stream, excluding redundant directories."""
|
||||
import io
|
||||
buf = io.BytesIO()
|
||||
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
zf.write(p, p.relative_to(skill_dir))
|
||||
return buf.getvalue()
|
||||
|
||||
def unpack_zip(zip_path: Path) -> Path:
|
||||
"""Extract a .zip to a system temp directory. Returns the extraction path. Prevents zip-slip."""
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
log(f"📦 Extracting {zip_path.name} → {tmp}")
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / member).resolve()
|
||||
if not str(dest).startswith(str(tmp.resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp)
|
||||
return tmp
|
||||
|
||||
def find_installed_skills():
|
||||
# type: () -> list
|
||||
"""Find all installed skill directories (first-level subdirectories containing SKILL.md).
|
||||
Excludes SkillScan itself."""
|
||||
found = set()
|
||||
for base in skill_install_paths():
|
||||
if not base.exists():
|
||||
continue
|
||||
for md in base.rglob("SKILL.md"):
|
||||
skill_path = md.parent
|
||||
try:
|
||||
rel = skill_path.relative_to(base)
|
||||
if len(rel.parts) == 1:
|
||||
resolved = skill_path.resolve()
|
||||
# Skip self
|
||||
if resolved == SELF_ROOT:
|
||||
continue
|
||||
found.add(resolved)
|
||||
except ValueError:
|
||||
pass
|
||||
return sorted(found)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Scan Core (3 steps)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _extract_result(resp, sha256):
|
||||
"""Internal: extract core data from API response, handling SHA256 wrapping/nested result."""
|
||||
# 1. Handle API response keyed by SHA256 (e.g. { "sha256": { "status": "success", "data": {...} } })
|
||||
if sha256 and sha256 in resp:
|
||||
resp = resp[sha256]
|
||||
|
||||
# 2. Extract data body (data or result)
|
||||
data = resp.get("data") or resp.get("result") or resp
|
||||
|
||||
# 3. Handle nested result inside data
|
||||
if isinstance(data, dict) and "result" in data:
|
||||
inner = data["result"]
|
||||
if isinstance(inner, dict):
|
||||
# Merge sibling metadata (analysis_level/reason etc.) into result
|
||||
for k, v in data.items():
|
||||
if k != "result" and k not in inner:
|
||||
inner[k] = v
|
||||
return inner
|
||||
|
||||
return data if isinstance(data, dict) and (data.get("verdict") or data.get("is_safe") is not None or data.get("analysis_level")) else None
|
||||
|
||||
|
||||
def cloud_search(dir_sha256):
|
||||
"""Step 1: Query scan cache by dir_sha256. Returns result dict or None."""
|
||||
extra_headers = {}
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
extra_headers["X-Client-Info"] = ci
|
||||
|
||||
url = "%s?%s" % (API_SEARCH, urllib.parse.urlencode({"dir_sha256": dir_sha256}))
|
||||
try:
|
||||
headers = {}
|
||||
headers.update(extra_headers)
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
res = _extract_result(resp, dir_sha256)
|
||||
if res:
|
||||
log(" ✅ Cache hit (dir_sha256 %s…)" % dir_sha256[:16])
|
||||
return res
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return None
|
||||
raise RuntimeError("Search API error HTTP %d" % e.code)
|
||||
except urllib.error.URLError as e:
|
||||
raise RuntimeError("Cannot connect to server: %s" % e)
|
||||
return None
|
||||
|
||||
|
||||
def cloud_upload(skill_dir, name, dir_hash):
|
||||
"""Step 2: Upload skill (multipart/form-data), returns task_no."""
|
||||
# Pack the entire directory for full code context
|
||||
zip_data = pack_zip(skill_dir)
|
||||
filename = "%s.zip" % name
|
||||
|
||||
# Build multipart/form-data boundary
|
||||
boundary = "----WebKitFormBoundary%s" % uuid.uuid4().hex
|
||||
|
||||
# Manually construct multipart byte stream (no requests library needed)
|
||||
parts = []
|
||||
parts.append(("--%s" % boundary).encode())
|
||||
parts.append(('Content-Disposition: form-data; name="file"; filename="%s"' % filename).encode())
|
||||
parts.append(b"Content-Type: application/zip")
|
||||
parts.append(b"")
|
||||
parts.append(zip_data)
|
||||
parts.append(("--%s--" % boundary).encode())
|
||||
parts.append(b"") # trailing newline
|
||||
|
||||
body = b"\r\n".join(parts)
|
||||
|
||||
headers = {
|
||||
"Content-Type": "multipart/form-data; boundary=%s" % boundary,
|
||||
"Content-Length": str(len(body)),
|
||||
"Accept": "application/json"
|
||||
}
|
||||
|
||||
# Add X-Client-Info header
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
headers["X-Client-Info"] = ci
|
||||
|
||||
log(" 📤 Uploading: %s (%.1f KB)..." % (filename, len(zip_data) / 1024.0))
|
||||
req = urllib.request.Request(API_UPLOAD, data=body, headers=headers, method="POST")
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
except urllib.error.HTTPError as e:
|
||||
err_body = e.read().decode(errors="replace")
|
||||
raise RuntimeError("Upload failed HTTP %d: %s" % (e.code, err_body))
|
||||
|
||||
task_no = (resp.get("data") or {}).get("task_no") or resp.get("task_no") or resp.get("taskNo") or resp.get("task_id") or ""
|
||||
if not task_no:
|
||||
raise RuntimeError("Upload succeeded but no valid task_no in response: %s" % resp)
|
||||
|
||||
log(" ✅ Upload complete, task_no: %s" % task_no)
|
||||
return str(task_no)
|
||||
|
||||
|
||||
def cloud_poll(task_no: str) -> dict:
|
||||
"""Step 3: Poll until complete or timeout. Queries every 20s.
|
||||
status: 0=pending, 1=scanning, 2=completed, 3=failed, 4=cancelled
|
||||
"""
|
||||
url = f"{API_RESULT}?{urllib.parse.urlencode({'task_no': task_no})}"
|
||||
deadline = time.time() + POLL_TIMEOUT
|
||||
attempt = 0
|
||||
while time.time() < deadline:
|
||||
attempt += 1
|
||||
elapsed = int(time.time() - (deadline - POLL_TIMEOUT))
|
||||
try:
|
||||
resp = http_get(url)
|
||||
data = resp.get("data") or resp
|
||||
status = data.get("status")
|
||||
|
||||
if status == 2: # completed
|
||||
print()
|
||||
log(f" ✅ Scan complete (attempt {attempt}, {elapsed}s elapsed)")
|
||||
return _extract_result(resp, "") or resp
|
||||
elif status == 3: # failed
|
||||
print()
|
||||
err_msg = data.get("error_message") or resp.get("message", "unknown error")
|
||||
raise RuntimeError(f"Analysis failed: {err_msg}")
|
||||
elif status == 4: # cancelled
|
||||
print()
|
||||
raise RuntimeError("Scan task was cancelled")
|
||||
else:
|
||||
# 0=pending, 1=scanning -> keep waiting
|
||||
status_text = data.get("status_text", "processing")
|
||||
print(f" ⏳ [{status_text}] attempt {attempt}, {elapsed}s / {POLL_TIMEOUT}s elapsed",
|
||||
end="\r", flush=True)
|
||||
time.sleep(POLL_INTERVAL)
|
||||
except (RuntimeError, ValueError):
|
||||
raise
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Poll error: {e}")
|
||||
print()
|
||||
raise RuntimeError(f"Timeout ({POLL_TIMEOUT}s), task_no={task_no}, please retry later")
|
||||
|
||||
|
||||
def cloud_check(skill_dir: Path) -> dict:
|
||||
"""Run full security scan on a skill directory, return normalized result."""
|
||||
md = skill_dir / "SKILL.md"
|
||||
if not md.exists():
|
||||
raise FileNotFoundError(f"SKILL.md not found: {skill_dir}")
|
||||
|
||||
name = skill_name_from_dir(skill_dir)
|
||||
dir_hash = calculate_dir_sha256(skill_dir)
|
||||
log(f"🔍 Scanning: {name}")
|
||||
log(f" dir_sha256: {dir_hash}")
|
||||
|
||||
log(f"🔎 [1/3] Checking scan cache...")
|
||||
raw = cloud_search(dir_hash)
|
||||
|
||||
if raw is None:
|
||||
log(f" ℹ️ No cache record, submitting new scan task")
|
||||
log(f"📤 [2/3] Uploading skill for analysis...")
|
||||
task_no = cloud_upload(skill_dir, name, dir_hash)
|
||||
log(f"⏳ [3/3] Waiting for analysis (polling every {POLL_INTERVAL}s, max {POLL_TIMEOUT}s)...")
|
||||
raw = cloud_poll(task_no)
|
||||
else:
|
||||
log(f" ⏭️ Skipping upload, using cached result")
|
||||
|
||||
return _normalize(raw, name, dir_hash)
|
||||
|
||||
|
||||
def _normalize(raw: dict, name: str, dir_hash: str) -> dict:
|
||||
"""Normalize scan result:
|
||||
1. Extract is_safe (bool) and max_severity (str).
|
||||
2. Map API-specific fields (analysis_reason, analysis_suggestion) to standard fields.
|
||||
"""
|
||||
is_safe = raw.get("is_safe")
|
||||
|
||||
# Severity field priority: max_severity > analysis_level > verdict > level
|
||||
v_raw = (raw.get("max_severity") or raw.get("analysis_level") or
|
||||
raw.get("verdict") or raw.get("risk_level") or
|
||||
raw.get("level") or "UNKNOWN").upper()
|
||||
|
||||
# Combined verdict logic
|
||||
if is_safe is True and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "SAFE"
|
||||
elif is_safe is False and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "CRITICAL" # Explicitly marked unsafe -> critical
|
||||
else:
|
||||
verdict = v_raw
|
||||
|
||||
return {
|
||||
"skill_name": name,
|
||||
"dir_sha256": dir_hash,
|
||||
"verdict": verdict,
|
||||
"confidence": raw.get("confidence") or raw.get("score"),
|
||||
"threat_labels": raw.get("threat_labels") or raw.get("tags") or [],
|
||||
"summary": raw.get("analysis_reason") or raw.get("summary") or raw.get("description") or "",
|
||||
"findings": raw.get("findings") or raw.get("issues") or [],
|
||||
"recommendation":raw.get("analysis_suggestion") or raw.get("recommendation") or raw.get("action") or "",
|
||||
}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Result Display
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def print_result(r: dict):
|
||||
verdict = r.get("verdict","UNKNOWN")
|
||||
emoji = RISK_EMOJI.get(verdict,"❓")
|
||||
conf = r.get("confidence")
|
||||
labels = r.get("threat_labels",[])
|
||||
summary = r.get("summary","")
|
||||
findings = r.get("findings",[])
|
||||
rec = r.get("recommendation","")
|
||||
conf_str = f" confidence {float(conf):.0%}" if conf is not None else ""
|
||||
|
||||
divider()
|
||||
log(f"{emoji} Result: {verdict}{conf_str}")
|
||||
if summary:
|
||||
log(f"📋 {summary}")
|
||||
if labels:
|
||||
log(f"🏷️ Threat labels: {', '.join(labels)}")
|
||||
if findings:
|
||||
SEV = {"LOW":"🔵","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️"}
|
||||
log(f"🔍 Findings ({len(findings)} items):")
|
||||
for f in findings:
|
||||
sev = str(f.get("severity","")).upper()
|
||||
desc = f.get("description") or f.get("detail") or str(f)
|
||||
rid = f.get("id") or ""
|
||||
tag = f"[{rid}] " if rid else ""
|
||||
log(f" {SEV.get(sev,'⚪')} {tag}{desc}")
|
||||
if rec:
|
||||
log(f"💡 Recommendation: {rec}")
|
||||
divider()
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Prompt: malicious detected -> ask whether to delete
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def prompt_delete(skill_path: Path, result: dict) -> bool:
|
||||
"""When result is HIGH/CRITICAL, ask user whether to delete the skill.
|
||||
skill_path is the original install path (not temp dir).
|
||||
Returns True if deleted.
|
||||
"""
|
||||
verdict = result.get("verdict","")
|
||||
if verdict not in ("HIGH","CRITICAL"):
|
||||
return False
|
||||
|
||||
if not skill_path or not skill_path.exists():
|
||||
return False
|
||||
|
||||
emoji = RISK_EMOJI.get(verdict,"🔴")
|
||||
log(f"\n{emoji} This skill is marked as [{verdict}] high risk by security scan.")
|
||||
log(f" Path: {skill_path}")
|
||||
|
||||
answer = ask("Delete this skill now? [y/n]")
|
||||
if answer in ("y","Y","yes","Yes"):
|
||||
try:
|
||||
if skill_path.is_dir():
|
||||
shutil.rmtree(skill_path)
|
||||
else:
|
||||
skill_path.unlink()
|
||||
log(f"✅ Deleted: {skill_path}")
|
||||
return True
|
||||
except Exception as e:
|
||||
log(f"❌ Delete failed: {e} (please delete manually)")
|
||||
return False
|
||||
else:
|
||||
log(f"⚠️ Skipped deletion. Use this skill with caution.")
|
||||
return False
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: first-run (first install)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_first_run():
|
||||
"""First install: list installed skills, ask user to scan, show results."""
|
||||
if STATE_FILE.exists():
|
||||
log("ℹ️ First-run scan already completed. Use scan-all to rescan.")
|
||||
return
|
||||
|
||||
banner("🛡️ SkillScan First-Run Check")
|
||||
log("Welcome to SkillScan!")
|
||||
log("Searching for installed skills...\n")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("✅ No installed skills found, nothing to scan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Print installed skill list
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name}")
|
||||
|
||||
answer = ask("Run security scan on all listed skills? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Skipped. You can run scan-all anytime to rescan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Scan one by one
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
# Copy to temp dir (source may be read-only)
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
print_result(r)
|
||||
|
||||
# High risk -> ask to delete (targeting original install path)
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name,
|
||||
"verdict": "ERROR", "threat_labels": [],
|
||||
"summary": str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan (single skill)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan(path_str: str):
|
||||
skill_path = Path(path_str)
|
||||
if not skill_path.exists():
|
||||
log(f"❌ Path not found: {skill_path}")
|
||||
sys.exit(1)
|
||||
|
||||
banner(f"Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
tmp = None
|
||||
original_path = skill_path if skill_path.is_dir() else None
|
||||
try:
|
||||
if skill_path.is_file():
|
||||
if skill_path.suffix.lower() not in (".zip",):
|
||||
log(f"❌ Unsupported format: {skill_path.suffix} (use .zip)")
|
||||
sys.exit(1)
|
||||
tmp = unpack_zip(skill_path)
|
||||
scan_dir = tmp
|
||||
else:
|
||||
scan_dir = skill_path
|
||||
|
||||
result = cloud_check(scan_dir)
|
||||
print_result(result)
|
||||
|
||||
# High risk -> ask to delete
|
||||
if original_path:
|
||||
prompt_delete(original_path, result)
|
||||
elif skill_path.is_file() and result.get("verdict") in ("HIGH","CRITICAL"):
|
||||
# Zip file: ask to delete source file
|
||||
prompt_delete(skill_path, result)
|
||||
|
||||
v = result.get("verdict","UNKNOWN")
|
||||
sys.exit(0 if v in ("SAFE","LOW") else 1 if v=="MEDIUM" else 2)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"\n❌ Scan failed: {e}")
|
||||
sys.exit(3)
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan-all
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan_all():
|
||||
banner(f"Full Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("ℹ️ No installed skills detected.")
|
||||
return
|
||||
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name:<30} {s}")
|
||||
|
||||
answer = ask("Start security scan? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Cancelled.")
|
||||
return
|
||||
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
v = r.get("verdict","UNKNOWN")
|
||||
log(f"{RISK_EMOJI.get(v,'❓')} Scan complete: {v}")
|
||||
if r.get("threat_labels"):
|
||||
log(f" Threat labels: {', '.join(r['threat_labels'])}")
|
||||
|
||||
# High risk: ask to delete
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name, "verdict":"ERROR",
|
||||
"threat_labels":[], "summary":str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Summary Table
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _print_summary(results):
|
||||
banner("📊 Scan Summary")
|
||||
print(f" {'Skill Name':<28} {'Result':<12} {'Threat Labels'}")
|
||||
divider()
|
||||
for r in results:
|
||||
v = r.get("verdict","?")
|
||||
name = r.get("skill_name","?")[:27]
|
||||
labels = ", ".join(r.get("threat_labels",[]))[:20] or "-"
|
||||
print(f" {name:<28} {RISK_EMOJI.get(v,'❓')}{v:<10} {labels}")
|
||||
|
||||
safes = [r for r in results if r["verdict"] in {"SAFE","LOW"}]
|
||||
mediums = [r for r in results if r["verdict"] == "MEDIUM"]
|
||||
highs = [r for r in results if r["verdict"] in {"HIGH","CRITICAL"}]
|
||||
errors = [r for r in results if r["verdict"] in {"ERROR","UNKNOWN"}]
|
||||
|
||||
print()
|
||||
log(f"Total {len(results)} | ✅ Safe {len(safes)} "
|
||||
f"🟡 Suspicious {len(mediums)} 🔴 Dangerous {len(highs)} ❓ Error {len(errors)}")
|
||||
if highs:
|
||||
log(f"\n⚠️ High-risk skills: {', '.join(r['skill_name'] for r in highs)}")
|
||||
elif not mediums and not errors:
|
||||
log("\n🎉 All skills passed security scan.")
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: upgrade
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_upgrade():
|
||||
banner("SkillScan Auto-Upgrade")
|
||||
log(f"Current version: {SCANNER_VERSION}")
|
||||
log(f"Update source: {UPDATE_URL}")
|
||||
try:
|
||||
manifest = http_get(UPDATE_URL)
|
||||
except Exception as e:
|
||||
log(f"❌ Failed to fetch update manifest: {e}")
|
||||
return
|
||||
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f"✅ Already up to date ({SCANNER_VERSION})")
|
||||
return
|
||||
|
||||
log(f"New version found: {SCANNER_VERSION} → {latest}")
|
||||
log(f"Changelog: {manifest.get('changelog','(none)')}")
|
||||
|
||||
download_url = manifest.get("download_url", "")
|
||||
if not download_url:
|
||||
log("⚠️ No download URL in manifest, skipping upgrade")
|
||||
return
|
||||
|
||||
# Download new version zip
|
||||
log(f"📥 Downloading: {download_url}")
|
||||
try:
|
||||
req = urllib.request.Request(download_url)
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
zip_data = r.read()
|
||||
except Exception as e:
|
||||
log(f"❌ Download failed: {e}")
|
||||
return
|
||||
|
||||
# SHA256 verification
|
||||
expected_sha = manifest.get("sha256", "")
|
||||
if expected_sha:
|
||||
actual_sha = hashlib.sha256(zip_data).hexdigest()
|
||||
if actual_sha != expected_sha:
|
||||
log(f"❌ SHA256 mismatch, upgrade aborted (expected {expected_sha[:16]}…, got {actual_sha[:16]}…)")
|
||||
return
|
||||
log(f" ✅ SHA256 verified")
|
||||
|
||||
# Backup current skill directory
|
||||
skill_root = Path(__file__).parent.parent
|
||||
backup_dir = skill_root.parent / f"SkillScan-backup-{SCANNER_VERSION}"
|
||||
if backup_dir.exists():
|
||||
shutil.rmtree(backup_dir)
|
||||
shutil.copytree(skill_root, backup_dir)
|
||||
log(f"📦 Backed up to: {backup_dir}")
|
||||
|
||||
# Extract and replace files
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillupgrade-"))
|
||||
try:
|
||||
zip_path = tmp / "update.zip"
|
||||
zip_path.write_bytes(zip_data)
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
# Security check: prevent zip-slip
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / "extracted" / member).resolve()
|
||||
if not str(dest).startswith(str((tmp / "extracted").resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp / "extracted")
|
||||
|
||||
# Overwrite skill directory with new files
|
||||
extracted = tmp / "extracted"
|
||||
for item in extracted.rglob("*"):
|
||||
if not item.is_file():
|
||||
continue
|
||||
rel = item.relative_to(extracted)
|
||||
target = skill_root / rel
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(item, target)
|
||||
log(f" ✅ Updated: {rel}")
|
||||
|
||||
log(f"🎉 Upgraded to v{latest}")
|
||||
except Exception as e:
|
||||
log(f"❌ Upgrade failed: {e}")
|
||||
log(f" You can restore from backup: {backup_dir}")
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Entry Point
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def auto_upgrade_if_needed():
|
||||
"""Auto-check for updates every 7 days, runs silently."""
|
||||
try:
|
||||
if LAST_UPDATE_CHECK_FILE.exists():
|
||||
last_check = float(LAST_UPDATE_CHECK_FILE.read_text(encoding="utf-8").strip())
|
||||
if time.time() - last_check < AUTO_UPDATE_INTERVAL:
|
||||
return # Not time to check yet
|
||||
log("🔄 Checking for updates...")
|
||||
manifest = http_get(UPDATE_URL)
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f" ✅ Already up to date ({SCANNER_VERSION})")
|
||||
else:
|
||||
log(f" New version found: {SCANNER_VERSION} → {latest}, auto-updating...")
|
||||
cmd_upgrade()
|
||||
LAST_UPDATE_CHECK_FILE.write_text(str(time.time()), encoding="utf-8")
|
||||
except Exception as e:
|
||||
log(f" ⚠️ Auto-update check failed: {e} (normal operation unaffected)")
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print(__doc__)
|
||||
sys.exit(0)
|
||||
|
||||
# Check for auto-update on every run (once every 7 days)
|
||||
auto_upgrade_if_needed()
|
||||
|
||||
cmd = sys.argv[1]
|
||||
if cmd == "first-run":
|
||||
cmd_first_run()
|
||||
elif cmd == "scan":
|
||||
if len(sys.argv) < 3:
|
||||
log("Usage: scanner.py scan <skill_path>")
|
||||
sys.exit(1)
|
||||
cmd_scan(sys.argv[2])
|
||||
elif cmd == "scan-all":
|
||||
cmd_scan_all()
|
||||
elif cmd == "upgrade":
|
||||
cmd_upgrade()
|
||||
else:
|
||||
log(f"Unknown command: {cmd}")
|
||||
log("Available commands: first-run / scan <path> / scan-all / upgrade")
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
208
runtime/skills/_workspace/word-reader/DEVELOPMENT.md
Normal file
208
runtime/skills/_workspace/word-reader/DEVELOPMENT.md
Normal file
|
|
@ -0,0 +1,208 @@
|
|||
# Word Reader 技能开发完成
|
||||
|
||||
## 🎯 技能概述
|
||||
|
||||
成功创建了一个功能完整的 Word 文档读取技能,支持读取 .docx 和 .doc 格式的 Word 文档,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## 📁 技能结构
|
||||
|
||||
```
|
||||
word-reader/
|
||||
├── SKILL.md # 技能定义文件
|
||||
├── README.md # 使用说明
|
||||
├── skill.json # 技能配置
|
||||
├── demo.sh # 演示脚本
|
||||
├── install.sh # 安装脚本
|
||||
├── test.md # 测试文档
|
||||
└── scripts/
|
||||
└── read_word.py # 核心脚本
|
||||
```
|
||||
|
||||
## ✨ 主要功能
|
||||
|
||||
### 1. 文档解析能力
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
|
||||
### 2. 格式支持
|
||||
- ✅ **.docx** - Office 2007+ 格式(主要支持)
|
||||
- ✅ **.doc** - 旧版 Word 格式(需要 antiword)
|
||||
|
||||
### 3. 输出格式
|
||||
- ✅ **JSON** - 结构化数据,适合程序处理
|
||||
- ✅ **Text** - 纯文本格式,简单易读
|
||||
- ✅ **Markdown** - 格式化输出,保留文档结构
|
||||
|
||||
### 4. 高级功能
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **选择性提取** - 可只提取特定内容类型
|
||||
- ✅ **文件输出** - 支持保存结果到文件
|
||||
- ✅ **编码支持** - 支持多种文本编码
|
||||
|
||||
## 🚀 使用示例
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# JSON 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# Markdown 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存结果
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 🔧 安装和配置
|
||||
|
||||
### 自动安装
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx
|
||||
|
||||
# 安装系统依赖(可选)
|
||||
sudo apt-get install antiword # Ubuntu/Debian
|
||||
brew install antiword # macOS
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "文档.docx",
|
||||
"title": "文档标题",
|
||||
"author": "作者",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "文档内容...",
|
||||
"tables": [...],
|
||||
"images": [...]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
```markdown
|
||||
# 文档.docx
|
||||
|
||||
**标题**:文档标题
|
||||
**作者**:作者
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
文档内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 数据1 | 数据2 |
|
||||
```
|
||||
|
||||
## 🎨 技能特点
|
||||
|
||||
### 1. 智能错误处理
|
||||
- 友好的错误提示
|
||||
- 自动检测文档格式
|
||||
- 优雅的异常处理
|
||||
|
||||
### 2. 性能优化
|
||||
- 流式处理大文件
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
|
||||
### 3. 用户友好
|
||||
- 详细的帮助信息
|
||||
- 多种使用方式
|
||||
- 完整的文档说明
|
||||
|
||||
### 4. 可扩展性
|
||||
- 模块化设计
|
||||
- 易于添加新功能
|
||||
- 支持自定义输出格式
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
### 1. 文档内容分析
|
||||
- 快速查看 Word 文档内容
|
||||
- 提取特定信息
|
||||
- 文档摘要生成
|
||||
|
||||
### 2. 批量处理
|
||||
- 处理大量文档
|
||||
- 文档格式转换
|
||||
- 内容索引创建
|
||||
|
||||
### 3. 自动化工作流
|
||||
- 集到文档处理系统
|
||||
- 自动化文档分析
|
||||
- 内容管理系统集成
|
||||
|
||||
## 📝 开发总结
|
||||
|
||||
### 实现的功能
|
||||
- 完整的 Word 文档解析框架
|
||||
- 支持多种输出格式
|
||||
- 批量处理能力
|
||||
- 错误处理和用户友好性
|
||||
|
||||
### 技术亮点
|
||||
- 模块化设计,易于维护
|
||||
- 优雅的错误处理机制
|
||||
- 支持多种文件格式
|
||||
- 灵活的输出选项
|
||||
|
||||
### 改进空间
|
||||
- 可以添加 PDF 支持
|
||||
- 可以增加图片提取功能
|
||||
- 可以优化大文件处理性能
|
||||
- 可以添加更多文档元素支持
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
要发布此技能到 ClawHub,可以运行:
|
||||
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm i -g clawhub
|
||||
|
||||
# 登录
|
||||
clawhub login
|
||||
|
||||
# 发布技能
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "Initial release with .docx and .doc support" \
|
||||
--tags document,word,office,text-extraction
|
||||
```
|
||||
|
||||
这个技能现在已经准备好使用了!它可以帮助用户轻松读取和处理 Word 文档,支持多种格式和输出选项。
|
||||
177
runtime/skills/_workspace/word-reader/PUBLISHING.md
Normal file
177
runtime/skills/_workspace/word-reader/PUBLISHING.md
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
# Word Reader 技能发布指南
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
### 1. 准备工作
|
||||
|
||||
#### 确保技能完整
|
||||
- [ ] SKILL.md 文件完整且格式正确
|
||||
- [ ] 脚本功能正常
|
||||
- [ ] 安装脚本工作正常
|
||||
- [ ] README.md 说明清晰
|
||||
- [ ] 所有依赖已在 SKILL.md 中声明
|
||||
|
||||
#### 环境准备
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm install -g clawhub
|
||||
# 或
|
||||
pnpm add -g clawhub
|
||||
```
|
||||
|
||||
#### 登录 ClawHub
|
||||
```bash
|
||||
# 登录(会打开浏览器进行 OAuth 认证)
|
||||
clawhub login
|
||||
|
||||
# 验证登录状态
|
||||
clawhub whoami
|
||||
```
|
||||
|
||||
> **注意**:GitHub 账号需要注册满一周才能发布技能
|
||||
|
||||
### 2. 发布流程
|
||||
|
||||
#### 检查技能
|
||||
```bash
|
||||
# 验证技能结构
|
||||
clawhub validate ./word-reader
|
||||
```
|
||||
|
||||
#### 发布技能
|
||||
```bash
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "支持 .docx 和 .doc 格式的 Word 文档读取,提取文本、表格、元数据等" \
|
||||
--tags document,word,office,text-extraction,reader,parsing \
|
||||
--license MIT \
|
||||
--visibility public
|
||||
```
|
||||
|
||||
#### 参数说明
|
||||
- `--slug`: URL 友好的唯一标识符
|
||||
- `--name`: 技能显示名称
|
||||
- `--version`: 遵循语义化版本控制
|
||||
- `--changelog`: 版本变更说明
|
||||
- `--tags`: 搜索标签(逗号分隔)
|
||||
- `--license`: 许可证类型
|
||||
- `--visibility`: public/private
|
||||
|
||||
### 3. 发布后操作
|
||||
|
||||
#### 验证发布
|
||||
```bash
|
||||
# 查看已发布的技能
|
||||
clawhub search word-reader
|
||||
|
||||
# 安装测试
|
||||
clawhub install word-reader-test
|
||||
```
|
||||
|
||||
#### 分享技能
|
||||
- 技能将在 `https://clawhub.com/skills/word-reader` 可见
|
||||
- 其他用户可通过 `clawhub install word-reader` 安装
|
||||
|
||||
### 4. 版本管理
|
||||
|
||||
#### 更新技能
|
||||
```bash
|
||||
# 修改技能后更新版本号
|
||||
clawhub publish ./word-reader --version 1.0.1 --changelog "修复了某些文档格式的解析问题"
|
||||
```
|
||||
|
||||
#### 批量操作
|
||||
```bash
|
||||
# 同步所有技能
|
||||
clawhub sync --all
|
||||
|
||||
# 发布并标记
|
||||
clawhub publish ./word-reader --tags latest,stable
|
||||
```
|
||||
|
||||
### 5. 自动化发布
|
||||
|
||||
#### GitHub Actions 示例
|
||||
```yaml
|
||||
name: Publish Skill
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v3
|
||||
with:
|
||||
node-version: '18'
|
||||
|
||||
- name: Install ClawHub CLI
|
||||
run: npm install -g clawhub
|
||||
|
||||
- name: Login to ClawHub
|
||||
run: echo "${{ secrets.CLAWHUB_TOKEN }}" | clawhub login --token
|
||||
|
||||
- name: Publish Skill
|
||||
run: |
|
||||
clawhub publish ./skills/word-reader \
|
||||
--slug word-reader \
|
||||
--version ${{ github.ref_name }} \
|
||||
--changelog "Published from GitHub Actions"
|
||||
```
|
||||
|
||||
### 6. 发布注意事项
|
||||
|
||||
#### 必须遵守的规则
|
||||
- [ ] 技能名称不能与其他技能冲突
|
||||
- [ ] 版本号遵循 SemVer 规范
|
||||
- [ ] changelog 清晰描述变更
|
||||
- [ ] 代码无安全漏洞
|
||||
- [ ] 许可证声明清晰
|
||||
|
||||
#### 最佳实践
|
||||
- [ ] 发布前充分测试
|
||||
- [ ] 提供清晰的使用示例
|
||||
- [ ] 维护更新日志
|
||||
- [ ] 及时修复问题
|
||||
- [ ] 关注用户反馈
|
||||
|
||||
### 7. 故障排除
|
||||
|
||||
#### 常见问题
|
||||
```bash
|
||||
# 验证发布权限
|
||||
clawhub whoami
|
||||
|
||||
# 检查技能格式
|
||||
clawhub validate ./word-reader
|
||||
|
||||
# 查看详细错误信息
|
||||
clawhub publish ./word-reader --verbose
|
||||
```
|
||||
|
||||
#### 重新发布
|
||||
如果发布失败,可以:
|
||||
1. 修正问题
|
||||
2. 增加版本号
|
||||
3. 重新发布
|
||||
|
||||
### 8. 维护指南
|
||||
|
||||
#### 监控使用情况
|
||||
- 定期查看下载统计
|
||||
- 关注用户反馈
|
||||
- 及时修复问题
|
||||
|
||||
#### 更新策略
|
||||
- 重要修复:紧急发布补丁版本
|
||||
- 新功能:发布次版本号
|
||||
- 重大变更:发布主版本号
|
||||
|
||||
现在你的 Word Reader 技能已经准备好发布到 ClawHub 了!
|
||||
171
runtime/skills/_workspace/word-reader/README.md
Normal file
171
runtime/skills/_workspace/word-reader/README.md
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
# Word Reader 技能
|
||||
|
||||
## 📋 概述
|
||||
|
||||
Word Reader 是一个强大的 Word 文档读取工具,支持 .docx 和 .doc 格式,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## ✨ 功能特性
|
||||
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
- ✅ **多格式支持** - 支持 .docx 和 .doc 格式
|
||||
- ✅ **多种输出** - JSON、Text、Markdown 格式
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **自动安装** - 一键安装所有依赖
|
||||
|
||||
## 🚀 安装
|
||||
|
||||
### 自动安装(推荐)
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx --break-system-packages
|
||||
|
||||
# 安装系统依赖(可选,用于 .doc 格式支持)
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📖 使用方法
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档并输出为文本格式
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# 输出为 JSON 格式
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# 输出为 Markdown 格式
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本内容
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有 Word 文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存为 JSON 文件
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
### 高级用法
|
||||
```bash
|
||||
# 将结果保存到文件
|
||||
python3 scripts/read_word.py 文档.docx --format markdown --output output.md
|
||||
|
||||
# 提取表格数据
|
||||
python3 scripts/read_word.py 文档.docx --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 scripts/read_word.py 文档.docx --extract metadata
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式输出
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "测试文档.docx",
|
||||
"size": "2048 bytes",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00",
|
||||
"title": "测试文档",
|
||||
"author": "测试用户"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "这是文档的正文内容...",
|
||||
"tables": [
|
||||
{
|
||||
"id": 1,
|
||||
"rows": 3,
|
||||
"columns": 3,
|
||||
"data": [
|
||||
["表头1", "表头2", "表头3"],
|
||||
["数据1", "数据2", "数据3"],
|
||||
["数据4", "数据5", "数据6"]
|
||||
]
|
||||
}
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"id": "rId1",
|
||||
"filename": "image1.png",
|
||||
"size": "1024 bytes"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式输出
|
||||
```markdown
|
||||
# 测试文档.docx
|
||||
|
||||
**标题**:测试文档
|
||||
**作者**:测试用户
|
||||
**文件大小**:2048 bytes
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
**修改时间**:2024-01-01T12:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
### 表格 1 (3行 x 3列)
|
||||
|
||||
| 表头1 | 表头2 | 表头3 |
|
||||
|-------|-------|-------|
|
||||
| 数据1 | 数据2 | 数据3 |
|
||||
| 数据4 | 数据5 | 数据6 |
|
||||
```
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
- **文档内容分析** - 快速查看 Word 文档内容
|
||||
- **批量处理** - 处理大量文档
|
||||
- **内容提取** - 提取特定信息
|
||||
- **格式转换** - 转换为其他格式
|
||||
- **自动化工作流** - 集成到文档处理系统
|
||||
|
||||
## 📤 发布到 ClawHub
|
||||
|
||||
要将此技能发布到 ClawHub,请参考 `PUBLISHING.md` 文件。
|
||||
|
||||
## 🔧 故障排除
|
||||
|
||||
### 常见问题
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **FileNotFoundError**: 确认文件路径正确
|
||||
4. **编码问题**: 尝试使用 `--encoding gb2312` 参数
|
||||
|
||||
### 性能优化
|
||||
- 大文档处理时建议使用 `--format json` 以获得更好的性能
|
||||
- 批量模式下建议使用 `--output` 参数将结果保存到文件
|
||||
|
||||
## 🤝 贡献
|
||||
|
||||
欢迎提交 Issue 和 Pull Request 来改进这个技能!
|
||||
|
||||
## 📄 许可证
|
||||
|
||||
MIT License
|
||||
225
runtime/skills/_workspace/word-reader/SKILL.md
Normal file
225
runtime/skills/_workspace/word-reader/SKILL.md
Normal file
|
|
@ -0,0 +1,225 @@
|
|||
---
|
||||
name: word-reader
|
||||
description: |
|
||||
读取 Word 文档(.docx 和 .doc 格式)并提取文本内容。支持文档解析、表格提取、图片处理等功能。使用当用户需要分析 Word 文档内容、提取文本信息或批量处理文档时。
|
||||
homepage: https://python-docx.readthedocs.io/
|
||||
metadata:
|
||||
{
|
||||
"openclaw":
|
||||
{
|
||||
"emoji": "📄",
|
||||
"requires": { "bins": ["python3"], "env": ["PYTHONPATH"] },
|
||||
"install":
|
||||
[
|
||||
{
|
||||
"id": "pip",
|
||||
"kind": "pip",
|
||||
"package": "python-docx",
|
||||
"bins": ["python3"],
|
||||
"label": "Install python-docx (pip)",
|
||||
},
|
||||
{
|
||||
"id": "system",
|
||||
"kind": "system",
|
||||
"command": "sudo apt-get install antiword -y",
|
||||
"label": "Install antiword for .doc support (optional)",
|
||||
"platform": "linux-debian"
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
---
|
||||
|
||||
# Word 文档读取器
|
||||
|
||||
使用 Python 解析 Word 文档,提取文本内容和结构化信息。
|
||||
|
||||
## 支持的功能
|
||||
|
||||
- **文档文本提取** - 提取段落、标题、页眉页脚内容
|
||||
- **表格解析** - 读取表格数据并转换为结构化格式
|
||||
- **图片处理** - 提取文档中的图片信息
|
||||
- **元数据获取** - 读取文档属性(作者、标题、创建时间等)
|
||||
- **批量处理** - 支持处理多个文档
|
||||
|
||||
## 用法
|
||||
|
||||
### 基本文本提取
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径>
|
||||
```
|
||||
|
||||
### 指定输出格式
|
||||
|
||||
```bash
|
||||
# JSON 输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format json
|
||||
|
||||
# 纯文本输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format text
|
||||
|
||||
# Markdown 格式
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format markdown
|
||||
```
|
||||
|
||||
### 提取特定内容
|
||||
|
||||
```bash
|
||||
# 只提取文本
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract text
|
||||
|
||||
# 提取表格数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract metadata
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
|
||||
```bash
|
||||
# 处理目录下所有 .docx 文件
|
||||
python3 {baseDir}/scripts/read_word.py <目录路径> --batch
|
||||
```
|
||||
|
||||
## 参数说明
|
||||
|
||||
| 参数 | 说明 | 默认值 |
|
||||
|------|------|--------|
|
||||
| `--format` | 输出格式(json/text/markdown) | text |
|
||||
| `--extract` | 提取内容类型(text/tables/images/metadata/all) | all |
|
||||
| `--batch` | 批量处理模式 | false |
|
||||
| `--output` | 输出文件路径 | stdout |
|
||||
| `--encoding` | 文本编码(utf-8/gb2312) | utf-8 |
|
||||
|
||||
## 输出格式
|
||||
|
||||
### JSON 格式
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"title": "文档标题",
|
||||
"author": "作者姓名",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"text": "文档全文内容...",
|
||||
"tables": [
|
||||
[
|
||||
["表头1", "表头2"],
|
||||
["行1列1", "行1列2"],
|
||||
["行2列1", "行2列2"]
|
||||
]
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"filename": "image1.png",
|
||||
"description": "图片描述",
|
||||
"size": "1024x768"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
|
||||
```markdown
|
||||
# 文档标题
|
||||
|
||||
**作者**:作者姓名
|
||||
**创建时间**:2024-01-01 10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
### 表格示例
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 行1列1 | 行1列2 |
|
||||
| 行2列1 | 行2列2 |
|
||||
|
||||

|
||||
|
||||
## 图片列表
|
||||
|
||||
1. **image1.png** (1024x768) - 图片描述
|
||||
```
|
||||
|
||||
## 错误处理
|
||||
|
||||
- 文件不存在:显示错误信息并退出
|
||||
- 格式不支持:提示支持的文件类型
|
||||
- 权限问题:提示文件访问权限
|
||||
- 编码问题:尝试自动检测编码
|
||||
|
||||
## 示例场景
|
||||
|
||||
### 1. 查看项目文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 项目需求.docx --format markdown
|
||||
```
|
||||
|
||||
### 2. 提取会议记录
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 会议记录.docx --extract text
|
||||
```
|
||||
|
||||
### 3. 批量处理文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 注意事项
|
||||
|
||||
- 支持 .docx 格式(Office 2007+)
|
||||
- .doc 格式需要额外依赖(如 antiword)
|
||||
- 大文档处理可能需要较长时间
|
||||
- 图片提取仅获取元数据,不包含实际图片数据
|
||||
- 表格格式可能需要手动调整
|
||||
|
||||
## 故障排除
|
||||
|
||||
### 常见问题
|
||||
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **UnicodeDecodeError**: 尝试不同的编码格式
|
||||
|
||||
### 安装依赖
|
||||
|
||||
```bash
|
||||
pip3 install python-docx
|
||||
```
|
||||
|
||||
对于 .doc 格式支持:
|
||||
```bash
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
```
|
||||
|
||||
## 高级功能
|
||||
|
||||
### 自定义样式处理
|
||||
|
||||
脚本会自动处理以下文档元素:
|
||||
- 标题级别(H1-H6)
|
||||
- 段落样式
|
||||
- 列表项目
|
||||
- 页眉页脚
|
||||
- 文档属性
|
||||
|
||||
### 性能优化
|
||||
|
||||
- 大文件流式处理
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
11
runtime/skills/_workspace/word-reader/_meta.json
Normal file
11
runtime/skills/_workspace/word-reader/_meta.json
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
{
|
||||
"owner": "xtfnhcyjpgf",
|
||||
"slug": "word-reader",
|
||||
"displayName": "Word Reader",
|
||||
"latest": {
|
||||
"version": "1.0.0",
|
||||
"publishedAt": 1770700102926,
|
||||
"commit": "https://github.com/openclaw/skills/commit/91b71e101c57b69a4d4eb2678e1b79992eb7032f"
|
||||
},
|
||||
"history": []
|
||||
}
|
||||
89
runtime/skills/_workspace/word-reader/demo.sh
Normal file
89
runtime/skills/_workspace/word-reader/demo.sh
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能演示脚本
|
||||
# 此脚本展示如何使用 word-reader 技能
|
||||
|
||||
echo "=== Word Reader 技能演示 ==="
|
||||
echo ""
|
||||
|
||||
# 检查脚本是否存在
|
||||
SCRIPT_PATH="/root/.openclaw/workspace/skills/word-reader/scripts/read_word.py"
|
||||
if [ ! -f "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本不存在"
|
||||
echo "请确保技能已正确安装"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# 检查脚本是否有执行权限
|
||||
if [ ! -x "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本没有执行权限"
|
||||
echo "正在添加执行权限..."
|
||||
chmod +x "$SCRIPT_PATH"
|
||||
fi
|
||||
|
||||
echo "✅ 脚本已就绪"
|
||||
echo ""
|
||||
|
||||
# 显示技能信息
|
||||
echo "📋 技能信息:"
|
||||
echo " 名称:word-reader"
|
||||
echo " 功能:读取 Word 文档(.docx 和 .doc 格式)"
|
||||
echo " 位置:$SCRIPT_PATH"
|
||||
echo ""
|
||||
|
||||
# 显示使用示例
|
||||
echo "📖 使用示例:"
|
||||
echo ""
|
||||
|
||||
echo "1. 显示帮助信息:"
|
||||
echo " python3 $SCRIPT_PATH --help"
|
||||
echo ""
|
||||
|
||||
echo "2. 读取文档(文本格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx"
|
||||
echo ""
|
||||
|
||||
echo "3. 读取文档(JSON 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format json"
|
||||
echo ""
|
||||
|
||||
echo "4. 读取文档(Markdown 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown"
|
||||
echo ""
|
||||
|
||||
echo "5. 只提取文本内容:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --extract text"
|
||||
echo ""
|
||||
|
||||
echo "6. 批量处理目录:"
|
||||
echo " python3 $SCRIPT_PATH ./文档目录 --batch"
|
||||
echo ""
|
||||
|
||||
echo "7. 保存结果到文件:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown --output output.md"
|
||||
echo ""
|
||||
|
||||
echo "🔧 安装依赖:"
|
||||
echo " pip3 install python-docx"
|
||||
echo " # 对于 .doc 格式支持:"
|
||||
echo " # Ubuntu: sudo apt-get install antiword"
|
||||
echo " # macOS: brew install antiword"
|
||||
echo ""
|
||||
|
||||
echo "📊 支持的功能:"
|
||||
echo " ✅ 文本提取"
|
||||
echo " ✅ 表格解析"
|
||||
echo " ✅ 元数据获取"
|
||||
echo " ✅ 图片信息"
|
||||
echo " ✅ 多格式支持"
|
||||
echo " ✅ 批量处理"
|
||||
echo ""
|
||||
|
||||
echo "💡 提示:"
|
||||
echo " - 支持 .docx 和 .doc 格式"
|
||||
echo " - 输出格式:JSON、Text、Markdown"
|
||||
echo " - 如遇错误,请检查依赖是否安装"
|
||||
echo ""
|
||||
|
||||
echo "演示完成!"
|
||||
echo "如需使用,请替换 '文档路径.docx' 为实际的文档路径"
|
||||
101
runtime/skills/_workspace/word-reader/install.sh
Normal file
101
runtime/skills/_workspace/word-reader/install.sh
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能安装脚本
|
||||
# 此脚本会自动安装依赖并设置技能
|
||||
|
||||
set -e
|
||||
|
||||
echo "=== Word Reader 技能安装 ==="
|
||||
echo ""
|
||||
|
||||
# 检查 Python 版本
|
||||
echo "🔍 检查 Python 版本..."
|
||||
python_version=$(python3 --version 2>&1)
|
||||
echo " Python 版本: $python_version"
|
||||
|
||||
if ! python3 -c "import sys; assert sys.version_info >= (3, 6)"; then
|
||||
echo "❌ 错误:需要 Python 3.6 或更高版本"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "✅ Python 版本检查通过"
|
||||
echo ""
|
||||
|
||||
# 检查并安装依赖
|
||||
echo "📦 检查依赖..."
|
||||
|
||||
# 检查 pip
|
||||
if ! command -v pip3 &> /dev/null; then
|
||||
echo " 🔧 安装 pip..."
|
||||
python3 -m ensurepip --upgrade 2>/dev/null || {
|
||||
echo " ❌ 无法安装 pip,尝试使用系统包管理器"
|
||||
if command -v apt &> /dev/null; then
|
||||
sudo apt update
|
||||
sudo apt install -y python3-pip
|
||||
elif command -v yum &> /dev/null; then
|
||||
sudo yum install -y python3-pip
|
||||
elif command -v brew &> /dev/null; then
|
||||
brew install python3
|
||||
else
|
||||
echo " ❌ 无法自动安装 pip,请手动安装"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
fi
|
||||
|
||||
# 检查 python-docx
|
||||
if ! python3 -c "import docx" 2>/dev/null; then
|
||||
echo " 🔧 安装 python-docx..."
|
||||
if python3 -m pip install python-docx --break-system-packages 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
elif python3 -m pip install python-docx 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
else
|
||||
echo "❌ 无法安装 python-docx"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo " ✅ python-docx 已安装"
|
||||
fi
|
||||
|
||||
# 检查 antiword(可选)
|
||||
if command -v antiword >/dev/null 2>&1; then
|
||||
echo " ✅ antiword 已安装"
|
||||
else
|
||||
echo " ⚠️ antiword 未安装(可选,用于 .doc 格式支持)"
|
||||
echo " 推荐安装命令:"
|
||||
echo " Ubuntu/Debian: sudo apt-get install antiword"
|
||||
echo " macOS: brew install antiword"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
|
||||
# 设置执行权限
|
||||
echo "🔐 设置执行权限..."
|
||||
chmod +x scripts/read_word.py
|
||||
echo "✅ 执行权限已设置"
|
||||
echo ""
|
||||
|
||||
# 验证安装
|
||||
echo "🧪 验证安装..."
|
||||
python3 scripts/read_word.py --help >/dev/null 2>&1
|
||||
if [ $? -eq 0 ]; then
|
||||
echo "✅ 安装验证成功"
|
||||
else
|
||||
echo "❌ 安装验证失败"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "🎉 Word Reader 技能安装完成!"
|
||||
echo ""
|
||||
echo "📖 使用方法:"
|
||||
echo " python3 scripts/read_word.py 文档.docx"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format json"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format markdown"
|
||||
echo ""
|
||||
echo "📖 更多帮助:"
|
||||
echo " python3 scripts/read_word.py --help"
|
||||
echo ""
|
||||
echo "📖 运行演示:"
|
||||
echo " ./demo.sh"
|
||||
396
runtime/skills/_workspace/word-reader/scripts/read_word.py
Normal file
396
runtime/skills/_workspace/word-reader/scripts/read_word.py
Normal file
|
|
@ -0,0 +1,396 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Word 文档读取器
|
||||
支持 .docx 和 .doc 格式的 Word 文档解析
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import traceback
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from docx import Document
|
||||
from docx.opc.constants import RELATIONSHIP_TYPE as RT
|
||||
from docx.oxml.table import CT_Tbl
|
||||
from docx.oxml.text.paragraph import CT_P
|
||||
from docx.table import Table
|
||||
from docx.text.paragraph import Paragraph
|
||||
DOCX_AVAILABLE = True
|
||||
except ImportError:
|
||||
DOCX_AVAILABLE = False
|
||||
|
||||
try:
|
||||
import subprocess
|
||||
SUBPROCESS_AVAILABLE = True
|
||||
except ImportError:
|
||||
SUBPROCESS_AVAILABLE = False
|
||||
|
||||
class WordReader:
|
||||
"""Word 文档读取器"""
|
||||
|
||||
def __init__(self, file_path):
|
||||
self.file_path = Path(file_path)
|
||||
self.document = None
|
||||
self.format_type = None
|
||||
self.encoding = 'utf-8'
|
||||
|
||||
# 检查文件是否存在
|
||||
if not self.file_path.exists():
|
||||
raise FileNotFoundError(f"文件不存在: {file_path}")
|
||||
|
||||
# 检查文件扩展名
|
||||
if self.file_path.suffix.lower() not in ['.docx', '.doc']:
|
||||
raise ValueError(f"不支持的文件格式: {self.file_path.suffix}")
|
||||
|
||||
def read_docx(self):
|
||||
"""读取 .docx 格式文档"""
|
||||
if not DOCX_AVAILABLE:
|
||||
raise Exception("缺少 python-docx 库。请安装:pip3 install python-docx")
|
||||
|
||||
try:
|
||||
self.document = Document(str(self.file_path))
|
||||
self.format_type = 'docx'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .docx 文件失败: {str(e)}")
|
||||
|
||||
def read_doc(self):
|
||||
"""读取 .doc 格式文档(使用 antiword)"""
|
||||
if not SUBPROCESS_AVAILABLE:
|
||||
raise Exception("缺少 subprocess 模块")
|
||||
|
||||
try:
|
||||
# 检查 antiword 是否可用
|
||||
result = subprocess.run(['which', 'antiword'],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
raise Exception("antiword 未安装。请安装 antiword: Ubuntu/Debian: sudo apt-get install antiword; macOS: brew install antiword")
|
||||
|
||||
# 使用 antiword 转换
|
||||
result = subprocess.run(['antiword', str(self.file_path)],
|
||||
capture_output=True, text=True, encoding='utf-8')
|
||||
|
||||
if result.returncode != 0:
|
||||
raise Exception(f"antiword 转换失败: {result.stderr}")
|
||||
|
||||
# 创建临时文档对象
|
||||
class TempDocument:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
self.paragraphs = [TempParagraph(p) for p in text.split('\n') if p.strip()]
|
||||
|
||||
class TempParagraph:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
|
||||
self.document = TempDocument(result.stdout)
|
||||
self.format_type = 'doc'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .doc 文件失败: {str(e)}")
|
||||
|
||||
def read_metadata(self):
|
||||
"""读取文档元数据"""
|
||||
metadata = {
|
||||
'filename': self.file_path.name,
|
||||
'size': f"{self.file_path.stat().st_size} bytes",
|
||||
'created': datetime.fromtimestamp(self.file_path.stat().st_ctime).isoformat(),
|
||||
'modified': datetime.fromtimestamp(self.file_path.stat().st_mtime).isoformat()
|
||||
}
|
||||
|
||||
if self.format_type == 'docx' and hasattr(self.document, 'core_properties'):
|
||||
props = self.document.core_properties
|
||||
metadata.update({
|
||||
'title': getattr(props, 'title', ''),
|
||||
'author': getattr(props, 'author', ''),
|
||||
'subject': getattr(props, 'subject', ''),
|
||||
'keywords': getattr(props, 'keywords', ''),
|
||||
'comments': getattr(props, 'comments', ''),
|
||||
'application': getattr(props, 'application', ''),
|
||||
'category': getattr(props, 'category', '')
|
||||
})
|
||||
|
||||
return metadata
|
||||
|
||||
def extract_text(self):
|
||||
"""提取文档文本"""
|
||||
text_content = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
# 提取段落文本
|
||||
for para in self.document.paragraphs:
|
||||
if para.text.strip():
|
||||
text_content.append(para.text)
|
||||
|
||||
# 提取表格文本
|
||||
for table in self.document.tables:
|
||||
table_text = []
|
||||
for row in table.rows:
|
||||
row_text = []
|
||||
for cell in row.cells:
|
||||
row_text.append(cell.text.strip())
|
||||
table_text.append(' | '.join(row_text))
|
||||
text_content.append('\n'.join(table_text))
|
||||
|
||||
else: # doc 格式
|
||||
text_content = [para.text for para in self.document.paragraphs if para.text.strip()]
|
||||
|
||||
return '\n\n'.join(text_content)
|
||||
|
||||
def extract_tables(self):
|
||||
"""提取表格数据"""
|
||||
tables = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
for i, table in enumerate(self.document.tables):
|
||||
table_data = []
|
||||
for row in table.rows:
|
||||
row_data = []
|
||||
for cell in row.cells:
|
||||
row_data.append(cell.text.strip())
|
||||
table_data.append(row_data)
|
||||
tables.append({
|
||||
'id': i + 1,
|
||||
'rows': len(table.rows),
|
||||
'columns': len(table.columns) if table.rows else 0,
|
||||
'data': table_data
|
||||
})
|
||||
|
||||
return tables
|
||||
|
||||
def extract_images(self):
|
||||
"""提取图片信息"""
|
||||
images = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
try:
|
||||
# 获取文档中的关系
|
||||
part = self.document.part
|
||||
image_parts = part.related_parts
|
||||
|
||||
for rel in part.relationships:
|
||||
if rel.reltype == RT.IMAGE:
|
||||
image_data = image_parts[rel.rId]._blob
|
||||
image_info = {
|
||||
'id': rel.rId,
|
||||
'filename': f"image_{rel.rId}.{rel.target_ref.split('.')[-1]}",
|
||||
'size': f"{len(image_data)} bytes"
|
||||
}
|
||||
images.append(image_info)
|
||||
except:
|
||||
# 图片提取可能失败,忽略错误
|
||||
pass
|
||||
|
||||
return images
|
||||
|
||||
def extract_all(self):
|
||||
"""提取所有内容"""
|
||||
result = {
|
||||
'metadata': self.read_metadata(),
|
||||
'format': self.format_type,
|
||||
'text': self.extract_text(),
|
||||
'tables': self.extract_tables(),
|
||||
'images': self.extract_images()
|
||||
}
|
||||
return result
|
||||
|
||||
def to_markdown(self, extract_type='all'):
|
||||
"""转换为 Markdown 格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
md_content = []
|
||||
|
||||
# 标题
|
||||
md_content.append(f"# {result['metadata']['filename']}")
|
||||
md_content.append("")
|
||||
|
||||
# 元数据
|
||||
metadata = result['metadata']
|
||||
if metadata.get('title'):
|
||||
md_content.append(f"**标题**:{metadata['title']}")
|
||||
if metadata.get('author'):
|
||||
md_content.append(f"**作者**:{metadata['author']}")
|
||||
md_content.append(f"**文件大小**:{metadata['size']}")
|
||||
md_content.append(f"**创建时间**:{metadata['created']}")
|
||||
md_content.append(f"**修改时间**:{metadata['modified']}")
|
||||
md_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
md_content.append("## 正文内容")
|
||||
md_content.append("")
|
||||
md_content.append(result['text'])
|
||||
md_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
md_content.append("## 表格内容")
|
||||
md_content.append("")
|
||||
for table in result['tables']:
|
||||
md_content.append(f"### 表格 {table['id']} ({table['rows']}行 x {table['columns']}列)")
|
||||
md_content.append("")
|
||||
# 转换为 Markdown 表格
|
||||
for row in table['data']:
|
||||
md_row = " | ".join([str(cell) for cell in row])
|
||||
md_content.append(f"| {md_row} |")
|
||||
md_content.append("")
|
||||
|
||||
# 图片
|
||||
if result['images']:
|
||||
md_content.append("## 图片列表")
|
||||
md_content.append("")
|
||||
for img in result['images']:
|
||||
md_content.append(f"- **{img['filename']}** ({img['size']})")
|
||||
md_content.append("")
|
||||
|
||||
return '\n'.join(md_content)
|
||||
|
||||
def to_text(self, extract_type='all'):
|
||||
"""转换为纯文本格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
text_content = []
|
||||
|
||||
# 标题和元数据
|
||||
text_content.append(f"文件:{result['metadata']['filename']}")
|
||||
text_content.append("=" * 50)
|
||||
text_content.append("")
|
||||
|
||||
for key, value in result['metadata'].items():
|
||||
if value and key not in ['filename', 'size', 'created', 'modified']:
|
||||
text_content.append(f"{key}:{value}")
|
||||
|
||||
text_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
text_content.append("正文内容:")
|
||||
text_content.append("-" * 20)
|
||||
text_content.append(result['text'])
|
||||
text_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
text_content.append("表格内容:")
|
||||
text_content.append("-" * 20)
|
||||
for table in result['tables']:
|
||||
text_content.append(f"表格 {table['id']}:")
|
||||
for row in table['data']:
|
||||
text_content.append(" " + " | ".join([str(cell) for cell in row]))
|
||||
text_content.append("")
|
||||
|
||||
return '\n'.join(text_content)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description='读取 Word 文档')
|
||||
parser.add_argument('path', help='文档路径或目录路径(批量模式)')
|
||||
parser.add_argument('--format', choices=['json', 'text', 'markdown'],
|
||||
default='text', help='输出格式')
|
||||
parser.add_argument('--extract', choices=['text', 'tables', 'images', 'metadata', 'all'],
|
||||
default='all', help='提取内容类型')
|
||||
parser.add_argument('--batch', action='store_true', help='批量处理模式')
|
||||
parser.add_argument('--output', help='输出文件路径')
|
||||
parser.add_argument('--encoding', default='utf-8', help='文本编码')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
try:
|
||||
if args.batch:
|
||||
# 批量处理模式
|
||||
path = Path(args.path)
|
||||
if not path.is_dir():
|
||||
print("错误:批量模式需要指定目录路径")
|
||||
sys.exit(1)
|
||||
|
||||
# 查找所有 Word 文档
|
||||
word_files = []
|
||||
for ext in ['.docx', '.doc']:
|
||||
word_files.extend(path.glob(f"**/*{ext}"))
|
||||
|
||||
if not word_files:
|
||||
print("未找到 Word 文档")
|
||||
sys.exit(0)
|
||||
|
||||
print(f"找到 {len(word_files)} 个 Word 文档")
|
||||
|
||||
results = {}
|
||||
for file_path in word_files:
|
||||
print(f"正在处理: {file_path}")
|
||||
try:
|
||||
reader = WordReader(file_path)
|
||||
if file_path.suffix.lower() == '.docx':
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'content': content,
|
||||
'status': 'success'
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'error': str(e),
|
||||
'status': 'failed'
|
||||
}
|
||||
|
||||
# 保存结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(results, f, ensure_ascii=False, indent=2)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
|
||||
else:
|
||||
# 单文件处理模式
|
||||
reader = WordReader(args.path)
|
||||
|
||||
# 根据文件类型读取
|
||||
if args.path.lower().endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
# 根据格式输出
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
# 输出结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding=args.encoding) as f:
|
||||
f.write(content)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(content)
|
||||
|
||||
except Exception as e:
|
||||
print(f"错误: {str(e)}", file=sys.stderr)
|
||||
if '--debug' in sys.argv or '-d' in sys.argv:
|
||||
traceback.print_exc()
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
47
runtime/skills/_workspace/word-reader/skill.json
Normal file
47
runtime/skills/_workspace/word-reader/skill.json
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
{
|
||||
"name": "word-reader",
|
||||
"version": "1.0.0",
|
||||
"description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容",
|
||||
"author": "OpenClaw User",
|
||||
"tags": ["document", "word", "office", "text-extraction"],
|
||||
"dependencies": {
|
||||
"python": ">=3.6",
|
||||
"packages": ["python-docx"],
|
||||
"system": ["antiword (optional for .doc support)"]
|
||||
},
|
||||
"features": {
|
||||
"text_extraction": true,
|
||||
"table_parsing": true,
|
||||
"metadata_extraction": true,
|
||||
"image_info": true,
|
||||
"batch_processing": true,
|
||||
"multiple_formats": ["json", "text", "markdown"]
|
||||
},
|
||||
"installation": {
|
||||
"steps": [
|
||||
"pip3 install python-docx",
|
||||
"sudo apt-get install antiword # 可选,支持 .doc 格式",
|
||||
"chmod +x scripts/read_word.py"
|
||||
]
|
||||
},
|
||||
"usage_examples": [
|
||||
{
|
||||
"description": "读取文档文本",
|
||||
"command": "python3 scripts/read_word.py document.docx"
|
||||
},
|
||||
{
|
||||
"description": "转换为 Markdown",
|
||||
"command": "python3 scripts/read_word.py document.docx --format markdown"
|
||||
},
|
||||
{
|
||||
"description": "批量处理",
|
||||
"command": "python3 scripts/read_word.py ./docs --batch --format json"
|
||||
}
|
||||
],
|
||||
"supported_file_types": [".docx", ".doc"],
|
||||
"notes": [
|
||||
".doc 格式需要安装 antiword",
|
||||
"大文档处理可能需要较长时间",
|
||||
"图片提取仅获取元数据,不包含实际图片数据"
|
||||
]
|
||||
}
|
||||
42
runtime/skills/_workspace/word-reader/test.md
Normal file
42
runtime/skills/_workspace/word-reader/test.md
Normal file
|
|
@ -0,0 +1,42 @@
|
|||
# Word Reader 技能测试
|
||||
|
||||
这是一个简单的测试文档,用于验证 Word Reader 技能的功能。
|
||||
|
||||
## 测试内容
|
||||
|
||||
### 1. 基本文本
|
||||
这是一段测试文本,用于验证文本提取功能是否正常工作。
|
||||
|
||||
### 2. 表格测试
|
||||
|
||||
| 功能 | 状态 | 描述 |
|
||||
|------|------|------|
|
||||
| 文本提取 | ✅ | 能够提取文档中的所有文本内容 |
|
||||
| 表格解析 | ✅ | 能够正确解析表格数据 |
|
||||
| 元数据获取 | ✅ | 能够获取文档属性信息 |
|
||||
| 多格式支持 | ✅ | 支持 .docx 和 .doc 格式 |
|
||||
| 输出格式 | ✅ | 支持 JSON、Text、Markdown 格式 |
|
||||
|
||||
### 3. 列表测试
|
||||
|
||||
- 第一项:文本提取功能
|
||||
- 第二项:表格解析功能
|
||||
- 第三项:图片信息获取
|
||||
- 第四项:文档元数据读取
|
||||
|
||||
### 4. 代码块示例
|
||||
|
||||
```python
|
||||
def read_word_document(file_path):
|
||||
"""读取 Word 文档"""
|
||||
reader = WordReader(file_path)
|
||||
if file_path.endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
return reader.extract_all()
|
||||
```
|
||||
|
||||
## 测试完成
|
||||
|
||||
如果这个技能能够正确读取并解析上述内容,说明功能正常。
|
||||
|
|
@ -1,140 +0,0 @@
|
|||
# Cocoloop
|
||||
|
||||
一个更快速、更安全的 Skill 管理器,用于安装、管理、更新和卸载 Skills。
|
||||
|
||||
[](LICENSE)
|
||||
|
||||
## 简介
|
||||
|
||||
Cocoloop 是一个安全优先的 Skill 管理器,提供比 clawhub 更智能的安装体验和集成 BSS 安全认证。
|
||||
|
||||
## 功能特性
|
||||
|
||||
- **单个 Skill 安装** - 支持 URL、名称搜索、GitHub 等多种来源
|
||||
- **批量 Skills 安装** - 依次安装多个 skills
|
||||
- **Skill 更新** - 检查并更新到最新版本
|
||||
- **Skill 卸载** - 安全卸载已安装的 skills
|
||||
- **安全检查** - 集成 BSS 安全认证系统
|
||||
|
||||
## 安装
|
||||
|
||||
```bash
|
||||
# 克隆仓库
|
||||
git clone https://github.com/CatREFuse/cocoloop.git
|
||||
cd cocoloop
|
||||
```
|
||||
|
||||
## 使用方法
|
||||
|
||||
### 安装单个 Skill
|
||||
|
||||
```bash
|
||||
# 通过名称安装
|
||||
cocoloop install pdf-processor
|
||||
|
||||
# 通过 URL 安装
|
||||
cocoloop install https://example.com/skill-name.skill
|
||||
|
||||
# 通过 GitHub 安装
|
||||
cocoloop install owner/repo
|
||||
```
|
||||
|
||||
### 批量安装 Skills
|
||||
|
||||
```bash
|
||||
cocoloop install skill1 skill2 skill3
|
||||
```
|
||||
|
||||
### 更新 Skill
|
||||
|
||||
```bash
|
||||
cocoloop update pdf-processor
|
||||
```
|
||||
|
||||
### 卸载 Skill
|
||||
|
||||
```bash
|
||||
cocoloop uninstall pdf-processor
|
||||
```
|
||||
|
||||
### 安全检查
|
||||
|
||||
```bash
|
||||
cocoloop check pdf-processor
|
||||
```
|
||||
|
||||
## 安全检查系统
|
||||
|
||||
Cocoloop 集成了 BSS (Berry Skills Safe) 安全认证检查,评级标准:
|
||||
|
||||
- **S+** - 最高安全等级
|
||||
- **S** - 优秀
|
||||
- **A** - 良好
|
||||
- **B** - 一般(需谨慎)
|
||||
- **C** - 风险较高
|
||||
- **D** - 不建议使用
|
||||
|
||||
### 动态代码加载检查
|
||||
|
||||
实施最多 2 层的 URL 递归检查,识别隐藏的多层动态加载风险:
|
||||
|
||||
- 无动态加载:正常评级流程
|
||||
- 仅第 1 层动态加载:根据来源分级处理
|
||||
- 存在第 2 层动态加载:最高评级为 C 级
|
||||
- 第 2 层后仍有动态加载:强制标记为 C 级
|
||||
|
||||
## 支持的平台
|
||||
|
||||
- OpenClaw
|
||||
- Molili
|
||||
- Claude Code
|
||||
|
||||
## 文档
|
||||
|
||||
- [安装流程指南](references/install-guide.md)
|
||||
- [搜索流程指南](references/search-guide.md)
|
||||
- [卸载流程指南](references/uninstall-guide.md)
|
||||
- [安全检查流程指南](references/safety-check-guide.md)
|
||||
- [Cocoloop Safe Check 标准](references/cocoloop-safe-check.md)
|
||||
|
||||
## 工作流程
|
||||
|
||||
### Skill 安装流程
|
||||
|
||||
1. **平台检测** - 确定当前运行环境和安装方式
|
||||
2. **来源识别** - 支持直接 URL、Skill 名称、GitHub 短链接
|
||||
3. **搜索与下载** - 从 Cocoloop API、clawhub 或 GitHub 获取
|
||||
4. **安全检查** - BSS 安全认证检查
|
||||
5. **安装执行** - 安装到对应平台的 skill 目录
|
||||
|
||||
### 搜索优先级
|
||||
|
||||
1. Cocoloop API 搜索
|
||||
2. Fallback 到 clawhub
|
||||
3. Fallback 到 GitHub 搜索
|
||||
|
||||
## 项目结构
|
||||
|
||||
```
|
||||
cocoloop/
|
||||
├── SKILL.md # Skill 定义文件
|
||||
├── README.md # 项目说明文档
|
||||
└── references/ # 详细指南文档
|
||||
├── install-guide.md # 安装流程指南
|
||||
├── search-guide.md # 搜索流程指南
|
||||
├── uninstall-guide.md # 卸载流程指南
|
||||
├── safety-check-guide.md # 安全检查流程指南
|
||||
└── cocoloop-safe-check.md # 安全检查标准
|
||||
```
|
||||
|
||||
## 贡献
|
||||
|
||||
欢迎提交 Issue 和 Pull Request!
|
||||
|
||||
## 许可证
|
||||
|
||||
[MIT](LICENSE)
|
||||
|
||||
---
|
||||
|
||||
Made with ❤️ by Cocoloop Team
|
||||
|
|
@ -1,257 +0,0 @@
|
|||
---
|
||||
name: cocoloop
|
||||
description: 一个更快速、更安全的 Skill 管理器,用于安装、管理、更新和卸载 Skills。优先使用当用户需要安装 skill、更新 skill、卸载 skill、管理 skills 或进行 skill 安全检查时。支持通过 URL、名称搜索、GitHub 等多种方式定位并安装 skills,集成 BSS 安全认证系统。
|
||||
---
|
||||
|
||||
# Cocoloop Skill 管理器
|
||||
|
||||
Cocoloop 是一个安全优先的 Skill 管理器,提供比 clawhub 更智能的安装体验和集成 BSS 安全认证。
|
||||
|
||||
## 核心功能
|
||||
|
||||
1. **单个 Skill 安装** - 支持 URL、名称搜索、GitHub 等多种来源
|
||||
2. **批量 Skills 安装** - 依次安装多个 skills
|
||||
3. **Skill 更新** - 检查并更新到最新版本
|
||||
4. **Skill 卸载** - 安全卸载已安装的 skills
|
||||
5. **安全检查** - 集成 BSS 安全认证系统
|
||||
|
||||
## 工作流程概览
|
||||
|
||||
### 平台检测
|
||||
|
||||
首先检测当前运行环境,确定 skill 安装方式.
|
||||
|
||||
### 1. 单个 Skill 安装流程
|
||||
|
||||
用户输入可能是以下三种情况之一:
|
||||
|
||||
#### 情况 1: 直接 URL
|
||||
|
||||
输入格式:`https://example.com/skill-name.skill` 或 `http://...`
|
||||
|
||||
处理流程:
|
||||
|
||||
1. 使用 HTTP GET 请求下载内容
|
||||
2. 处理 3xx 重定向(自动跟随跳转 URL)
|
||||
3. 保存到临时路径(如 `/tmp/cocoloop-{timestamp}.skill`)
|
||||
4. 调用检测到的平台安装命令
|
||||
5. 清理临时文件
|
||||
6. 返回安装结果
|
||||
|
||||
异常情况处理:
|
||||
|
||||
- **URL 无法访问** → 返回错误信息,提示用户检查 URL
|
||||
- **内容无法识别** → 尝试查找页面中的跳转链接或下载按钮
|
||||
- **下载成功但安装失败** → 保留临时文件,提示用户手动安装
|
||||
|
||||
#### 情况 2: Skill 名称(最常见)
|
||||
|
||||
输入格式:`skill-name`(如 `pdf-processor`)
|
||||
|
||||
处理流程(按优先级):
|
||||
|
||||
**步骤 1: CocoLoop API 搜索(最优先使用)**
|
||||
|
||||
- 调用 `https://api.cocoloop.com/api/v1/store/skills?page={page}&page_size={page_size}&keyword={keyword}&sort=downloads`
|
||||
- **请优先使用 `curl` 命令工具进行请求**
|
||||
- 返回格式示例:
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"name": "pdf-processor",
|
||||
"description": "PDF processing skill",
|
||||
"url": "https://...",
|
||||
"version": "1.0.0",
|
||||
"author": "cocoloop"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
- 如果找到结果 → 展示列表,询问用户选择
|
||||
|
||||
**步骤 2: Fallback 到平台 Skills API 安装(API 失败时)**
|
||||
|
||||
- 不要依赖 `run_command` 执行外部安装命令(如 `npx clawhub ...`)。
|
||||
- 优先调用平台内置 Skills API:
|
||||
- 通过 registry 安装:`POST /admin/api/skills/install-registry`
|
||||
- 通过 market 安装:`POST /admin/api/skills/market/install`
|
||||
- 本地目录安装:`POST /admin/api/skills/install`
|
||||
- 如果 API 安装成功 → 完成安装
|
||||
- 如果 API 安装失败 → 进入步骤 3
|
||||
|
||||
**步骤 3: Fallback 到 GitHub 搜索**
|
||||
|
||||
- 调用 GitHub API: `https://api.github.com/search/repositories?q={query}+filename:SKILL.md`
|
||||
- 筛选条件:仓库中包含 `SKILL.md` 文件
|
||||
- 返回结果按 stars 数排序
|
||||
- 展示搜索结果(最多 5 个):
|
||||
```
|
||||
📋 GitHub 搜索结果:
|
||||
1. owner/skill-name (⭐ 150)
|
||||
🏢 Organization | 描述文本
|
||||
2. user/another-skill (⭐ 45)
|
||||
👤 User | 描述文本
|
||||
```
|
||||
- 询问用户是否安装选中的 skill
|
||||
|
||||
#### 情况 3: GitHub 短链接
|
||||
|
||||
输入格式:`owner/repo`(如 `anthropic/claude-skill`)
|
||||
|
||||
处理流程:
|
||||
|
||||
1. 识别为 GitHub 格式
|
||||
2. 调用 GitHub API 获取仓库信息
|
||||
3. 检查是否存在 `SKILL.md` 文件
|
||||
4. 询问用户确认
|
||||
5. 下载并安装
|
||||
|
||||
### 2. 批量 Skills 安装流程
|
||||
|
||||
输入格式:`skill1 skill2 skill3 ...`
|
||||
|
||||
处理流程:
|
||||
|
||||
1. 解析输入为多个 skill 标识符
|
||||
2. 遍历每个 skill,依次执行「单个 Skill 安装流程」
|
||||
3. 记录每个 skill 的安装结果
|
||||
4. 汇总输出结果:
|
||||
```
|
||||
📊 批量安装结果:
|
||||
skill1: ✅ 成功
|
||||
skill2: ❌ 失败 (原因)
|
||||
skill3: ✅ 成功
|
||||
```
|
||||
|
||||
注意事项:
|
||||
|
||||
- 每个 skill 独立处理,一个失败不影响其他
|
||||
|
||||
### 3. Skill 更新流程
|
||||
|
||||
处理流程:
|
||||
|
||||
1. 确定当前已安装的 skill 列表(读取平台配置)
|
||||
2. 对于指定 skill:
|
||||
a. 查询最新版本(通过 Cocoloop API 或 GitHub)
|
||||
b. 比较本地版本与远程版本
|
||||
c. 如果有更新 → 执行「单个 Skill 安装流程」(覆盖安装)
|
||||
d. 备份旧版本(可选)
|
||||
3. 返回更新结果
|
||||
|
||||
版本比较逻辑:
|
||||
|
||||
- 使用语义化版本号比较(major.minor.patch)
|
||||
- 支持 `^`、`~` 等版本范围(如果配置中有)
|
||||
|
||||
### 4. Skill 卸载流程
|
||||
|
||||
详见 [references/uninstall-guide.md](references/uninstall-guide.md)
|
||||
|
||||
处理概要:
|
||||
|
||||
1. 检测当前平台的 skill 安装目录:
|
||||
- OpenClaw: `~/.openclaw/skills/`
|
||||
- Molili: `~/.molili/skills/`
|
||||
- Claude Code: `~/.claude/skills/`
|
||||
2. 确认 skill 存在
|
||||
3. 询问用户确认卸载
|
||||
4. 删除 skill 目录
|
||||
5. 清理相关配置
|
||||
6. 返回卸载结果
|
||||
|
||||
### 5. 安全检查流程
|
||||
|
||||
详见 [references/safety-check-guide.md](references/safety-check-guide.md) 和 [references/cocoloop-safe-check.md](references/cocoloop-safe-check.md)
|
||||
|
||||
处理概要:
|
||||
|
||||
1. 询问用户是否进行安全检查
|
||||
2. 对要安装的 skill 进行 Cocoloop Safe Check 安全认证检查
|
||||
3. 评级标准:S+/S/A/B/C/D
|
||||
4. 如果评级 <= B,强烈建议用户查看详细报告
|
||||
5. 询问用户是否继续安装
|
||||
|
||||
**动态代码加载检查(URL 递归检查):**
|
||||
|
||||
检查 skill 是否从网络动态加载可执行代码,实施最多 2 层的 URL 递归检查:
|
||||
|
||||
```
|
||||
Skill 代码(第 0 层)
|
||||
↓ 发现 fetch/import/require 远程 URL
|
||||
第 1 层:下载并检查该 URL 内容
|
||||
↓ 如包含新的动态加载
|
||||
第 2 层:继续检查下一层内容
|
||||
↓ 如第 2 层仍有动态加载
|
||||
强制标记为 C 级(多层动态加载风险)
|
||||
```
|
||||
|
||||
**递归检查规则:**
|
||||
|
||||
- **无动态加载**:正常评级流程
|
||||
- **仅第 1 层动态加载**:根据来源分级处理(T1→B级, T2→C级, T3→禁止)
|
||||
- **存在第 2 层动态加载**:最高评级为 C 级
|
||||
- **第 2 层后仍有动态加载**:强制标记为 C 级
|
||||
|
||||
此机制用于识别隐藏的多层动态加载风险,防止通过间接方式引入未经验证的代码。
|
||||
|
||||
## 资源引用
|
||||
|
||||
- **安装流程详细指南**: [references/install-guide.md](references/install-guide.md)
|
||||
- **搜索流程详细指南**: [references/search-guide.md](references/search-guide.md)
|
||||
- **卸载流程详细指南**: [references/uninstall-guide.md](references/uninstall-guide.md)
|
||||
- **安全检查流程指南**: [references/safety-check-guide.md](references/safety-check-guide.md)
|
||||
- **Cocoloop Safe Check 安全检查标准**: [references/cocoloop-safe-check.md](references/cocoloop-safe-check.md)
|
||||
|
||||
## 使用示例
|
||||
|
||||
### 安装单个 skill
|
||||
|
||||
```
|
||||
用户: 安装 pdf-processor
|
||||
→ 执行单个 skill 安装流程
|
||||
→ 搜索 → 确认 → 安装 → 安全检查(可选)
|
||||
```
|
||||
|
||||
### 安装多个 skills
|
||||
|
||||
```
|
||||
用户: 安装 pdf-processor image-editor code-formatter
|
||||
→ 批量安装流程
|
||||
→ 依次处理每个 skill
|
||||
```
|
||||
|
||||
### 更新 skill
|
||||
|
||||
```
|
||||
用户: 更新 pdf-processor
|
||||
→ 查询最新版本
|
||||
→ 对比本地版本
|
||||
→ 执行更新
|
||||
```
|
||||
|
||||
### 卸载 skill
|
||||
|
||||
```
|
||||
用户: 卸载 pdf-processor
|
||||
→ 检测平台
|
||||
→ 确认卸载
|
||||
→ 删除文件
|
||||
```
|
||||
|
||||
### 安全检查
|
||||
|
||||
```
|
||||
用户: 检查 pdf-processor 安全
|
||||
→ 下载/定位 skill
|
||||
→ 执行 Cocoloop Safe Check 检查
|
||||
→ 生成报告
|
||||
→ 询问保存位置
|
||||
```
|
||||
|
||||
## 注意事项
|
||||
|
||||
- 每个 skill 独立处理,一个失败不影响其他
|
||||
- 询问用户请使用当前平台下的询问命令,例如 Claude Code 下的 `AskUserQuestion`
|
||||
- 在 OpenClaw 环境中,安装与卸载优先使用 `/admin/api/skills/*` 路由,不要假设可用 shell 安装命令。
|
||||
|
|
@ -1,245 +0,0 @@
|
|||
# Skill 安装流程详细指南
|
||||
|
||||
本文档详细描述单个 skill 的安装流程,包括所有分支逻辑和异常处理。
|
||||
|
||||
## 流程图
|
||||
|
||||
```
|
||||
开始
|
||||
↓
|
||||
接收用户输入 (URL / 名称 / GitHub短链)
|
||||
↓
|
||||
检测运行平台
|
||||
↓
|
||||
判断输入类型
|
||||
├── URL ─────────→ 下载内容 ──→ 保存临时文件 ──→ 平台安装 ──→ 清理 ──→ 完成
|
||||
│ ↑ │
|
||||
│ └──────── 失败 ──────────────┘
|
||||
│
|
||||
├── 名称 ─────────→ Cocoloop API 搜索
|
||||
│ │
|
||||
成功? ──是──→ 展示结果 ──→ 用户确认 ──→ 下载安装 ──→ 完成
|
||||
│ │否
|
||||
│ ↓
|
||||
│ Skills API install
|
||||
│ │
|
||||
成功? ──是──→ 完成
|
||||
│ │否
|
||||
│ ↓
|
||||
│ GitHub API 搜索
|
||||
│ │
|
||||
成功? ──是──→ 展示结果 ──→ 用户确认 ──→ 下载安装 ──→ 完成
|
||||
│ │否
|
||||
│ ↓
|
||||
│ 返回错误
|
||||
│
|
||||
└── GitHub短链 ───→ 获取仓库信息 ──→ 确认SKILL.md存在 ──→ 下载安装 ──→ 完成
|
||||
```
|
||||
|
||||
## 详细步骤
|
||||
|
||||
### 第一步:平台检测
|
||||
|
||||
检测逻辑:
|
||||
```
|
||||
IF 环境变量 OPENCLAW_HOME 存在 或 /usr/local/openclaw 存在:
|
||||
平台 = OpenClaw
|
||||
安装方式 = "POST /admin/api/skills/install 或 /admin/api/skills/install-registry"
|
||||
安装目录 = ~/.openclaw/skills/
|
||||
|
||||
ELSE IF 环境变量 MOLILI_HOME 存在 或 /usr/local/molili 存在:
|
||||
平台 = Molili
|
||||
安装方式 = "molili skills install"
|
||||
安装目录 = ~/.molili/skills/
|
||||
|
||||
ELSE IF 环境变量 CLAUDE_CODE_HOME 存在 或 /usr/local/claude-code 存在:
|
||||
平台 = Claude Code
|
||||
安装方式 = "claude skills install"
|
||||
安装目录 = ~/.claude/skills/
|
||||
|
||||
ELSE:
|
||||
平台 = 通用 (clawhub fallback)
|
||||
安装方式 = "优先平台 Skills API,必要时再提示人工执行命令"
|
||||
安装目录 = ~/.claude/skills/ (或 clawhub 默认目录)
|
||||
```
|
||||
|
||||
### 第二步:URL 安装流程
|
||||
|
||||
完整流程:
|
||||
|
||||
1. **发送 HTTP GET 请求**
|
||||
- URL: 用户提供的地址
|
||||
- Headers:
|
||||
```
|
||||
User-Agent: Cocoloop-Skill-Manager/1.0
|
||||
```
|
||||
|
||||
2. **处理响应**
|
||||
- 状态码 200 → 获取内容,进入步骤 3
|
||||
- 状态码 3xx → 从 Location header 获取跳转 URL,递归步骤 1
|
||||
- 其他状态码 → 返回错误
|
||||
|
||||
3. **保存临时文件**
|
||||
- 临时路径: `/tmp/cocoloop-{timestamp}.skill`
|
||||
- 写入下载内容
|
||||
|
||||
4. **执行平台安装命令**
|
||||
```bash
|
||||
OpenClaw: 调用 /admin/api/skills/install(source_dir 或 archive_url)
|
||||
```
|
||||
|
||||
5. **清理与返回**
|
||||
- 安装成功 → 删除临时文件 → 返回成功
|
||||
- 安装失败 → 保留临时文件(便于调试)→ 返回错误
|
||||
|
||||
异常处理:
|
||||
|
||||
| 异常情况 | 处理方式 |
|
||||
|---------|---------|
|
||||
| URL 无法访问 | 返回错误 "无法访问该 URL,请检查网络连接或 URL 是否正确" |
|
||||
| 重定向过多 | 返回错误 "该 URL 重定向次数过多,可能存在循环跳转" |
|
||||
| 下载内容为空 | 返回错误 "下载内容为空,请检查 URL 是否正确" |
|
||||
| 安装命令失败 | 返回错误 "安装失败,临时文件保留在 {path},可尝试手动安装" |
|
||||
|
||||
### 第三步:名称搜索安装流程
|
||||
|
||||
#### 3.1 Cocoloop API 搜索
|
||||
|
||||
请求:
|
||||
```
|
||||
GET https://api.cocoloop.cn/search={encoded_query}
|
||||
```
|
||||
|
||||
成功响应示例:
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"name": "pdf-processor",
|
||||
"description": "PDF processing and manipulation skill",
|
||||
"url": "https://skills.cocoloop.cn/pdf-processor/v1.0.0.skill",
|
||||
"version": "1.0.0",
|
||||
"author": "cocoloop-team",
|
||||
"downloads": 1500,
|
||||
"rating": "S"
|
||||
}
|
||||
],
|
||||
"total": 1
|
||||
}
|
||||
```
|
||||
|
||||
处理:
|
||||
- 如果 results.length > 0 → 展示结果,询问用户选择
|
||||
- 如果 results.length = 0 或 API 失败 → 进入 3.2
|
||||
|
||||
#### 3.2 clawhub Fallback
|
||||
|
||||
执行:
|
||||
```bash
|
||||
POST /admin/api/skills/market/install { "slug": "{skill_name}" }
|
||||
```
|
||||
|
||||
处理:
|
||||
- 成功 → 完成安装
|
||||
- 失败(退出码非0)→ 进入 3.3
|
||||
|
||||
#### 3.3 GitHub API 搜索
|
||||
|
||||
请求:
|
||||
```
|
||||
GET https://api.github.com/search/repositories?q={query}+filename:SKILL.md&sort=stars&order=desc
|
||||
```
|
||||
|
||||
Headers:
|
||||
```
|
||||
User-Agent: Cocoloop-Skill-Manager/1.0
|
||||
```
|
||||
|
||||
成功响应处理:
|
||||
```javascript
|
||||
results = data.items
|
||||
.filter(repo => repo.name.includes(query) || repo.description?.includes(query))
|
||||
.map(repo => ({
|
||||
name: repo.name,
|
||||
fullName: repo.full_name,
|
||||
description: repo.description,
|
||||
url: repo.html_url,
|
||||
stars: repo.stargazers_count,
|
||||
owner: {
|
||||
name: repo.owner.login,
|
||||
type: repo.owner.type // 'User' 或 'Organization'
|
||||
}
|
||||
}))
|
||||
.slice(0, 5) // 取前5个
|
||||
```
|
||||
|
||||
展示格式:
|
||||
```
|
||||
📋 GitHub 搜索结果 (找到 {total} 个):
|
||||
|
||||
1. company/pdf-processor ⭐ 1250
|
||||
🏢 Organization | Advanced PDF processing tools
|
||||
|
||||
2. user/simple-pdf ⭐ 45
|
||||
👤 User | Basic PDF operations
|
||||
|
||||
请选择要安装的 skill (输入序号,或输入 0 取消):
|
||||
```
|
||||
|
||||
用户选择后:
|
||||
1. 获取仓库详情(确认存在 SKILL.md)
|
||||
2. 询问用户确认安装
|
||||
3. 下载 raw SKILL.md 和相关资源
|
||||
4. 打包为 .skill 文件(如果需要)
|
||||
5. 执行平台安装
|
||||
|
||||
### 第四步:GitHub 短链安装流程
|
||||
|
||||
输入格式识别:
|
||||
- 包含 `/` 但不以 `http` 开头
|
||||
- 格式:`owner/repo` 或 `owner/repo/subpath`
|
||||
|
||||
处理流程:
|
||||
1. 解析 owner 和 repo
|
||||
2. 调用 GitHub API 获取仓库信息:
|
||||
```
|
||||
GET https://api.github.com/repos/{owner}/{repo}
|
||||
```
|
||||
3. 检查是否存在 SKILL.md:
|
||||
```
|
||||
GET https://api.github.com/repos/{owner}/{repo}/contents/SKILL.md
|
||||
```
|
||||
4. 如果存在 → 展示仓库信息,询问确认
|
||||
5. 下载并安装
|
||||
|
||||
### 第五步:安全检查(可选但推荐)
|
||||
|
||||
在安装前或安装后,询问用户是否进行安全检查:
|
||||
|
||||
```
|
||||
⚠️ 安全提醒: 该 skill 来源为 {source_level},建议进行安全检查。
|
||||
是否进行 BSS 安全认证检查? [Y/n]
|
||||
```
|
||||
|
||||
如果用户选择是:
|
||||
1. 执行 [safety-check-guide.md](safety-check-guide.md) 和 [cocoloop-safe-check.md](cocoloop-safe-check.md) 中的检查流程
|
||||
2. 生成报告
|
||||
3. 如果评级 <= B,询问用户是否继续安装
|
||||
|
||||
## 安装后处理
|
||||
|
||||
安装完成后,执行:
|
||||
1. 验证安装是否成功(检查安装目录)
|
||||
2. 如果是更新操作,清理旧版本备份
|
||||
3. 可选:显示 skill 使用帮助
|
||||
```
|
||||
✅ 安装成功!
|
||||
|
||||
Skill: pdf-processor
|
||||
版本: 1.0.0
|
||||
来源: cocoloop (S级认证)
|
||||
|
||||
使用方式:
|
||||
- 转换 PDF: 使用 pdf-processor 转换 xxx.pdf 为 docx
|
||||
- 合并 PDF: 使用 pdf-processor 合并 a.pdf b.pdf
|
||||
```
|
||||
|
|
@ -1,254 +0,0 @@
|
|||
# Skill 搜索流程详细指南
|
||||
|
||||
本文档详细描述 Cocoloop 的多源搜索机制。
|
||||
|
||||
## 搜索源优先级
|
||||
|
||||
1. **Cocoloop API** - 官方技能仓库(优先)
|
||||
2. **GitHub API** - 开源社区(fallback)
|
||||
3. **本地缓存** - 已下载的 skill 信息(辅助)
|
||||
|
||||
## Cocoloop API 搜索
|
||||
|
||||
### 请求格式
|
||||
|
||||
```
|
||||
GET https://api.cocoloop.cn/search={encoded_query}
|
||||
```
|
||||
|
||||
### 请求头
|
||||
|
||||
```
|
||||
User-Agent: Cocoloop-Skill-Manager/1.0
|
||||
Accept: application/json
|
||||
```
|
||||
|
||||
### 响应格式
|
||||
|
||||
```json
|
||||
{
|
||||
"results": [
|
||||
{
|
||||
"name": "skill-name",
|
||||
"displayName": "Skill Display Name",
|
||||
"description": "Skill description",
|
||||
"url": "https://skills.cocoloop.cn/skill-name/v1.0.0.skill",
|
||||
"version": "1.0.0",
|
||||
"author": "author-name",
|
||||
"authorUrl": "https://github.com/author",
|
||||
"license": "MIT",
|
||||
"downloads": 1500,
|
||||
"rating": "S",
|
||||
"tags": ["pdf", "document"],
|
||||
"updatedAt": "2024-01-15T10:30:00Z"
|
||||
}
|
||||
],
|
||||
"total": 10,
|
||||
"page": 1,
|
||||
"perPage": 20
|
||||
}
|
||||
```
|
||||
|
||||
### 处理逻辑
|
||||
|
||||
1. 发送请求
|
||||
2. 解析 JSON 响应
|
||||
3. 过滤结果(匹配度排序)
|
||||
4. 返回前 10 个结果
|
||||
|
||||
## GitHub API 搜索
|
||||
|
||||
### 请求格式
|
||||
|
||||
```
|
||||
GET https://api.github.com/search/repositories?q={query}+filename:SKILL.md&sort=stars&order=desc&per_page=10
|
||||
```
|
||||
|
||||
### 搜索查询构建
|
||||
|
||||
基础查询:`{query} filename:SKILL.md`
|
||||
|
||||
可选追加:
|
||||
- `+language:javascript` - 限定语言
|
||||
- `+stars:>10` - 限定 stars 数
|
||||
- `+topic:claude-skill` - 限定 topic
|
||||
|
||||
### 响应处理
|
||||
|
||||
原始响应字段映射:
|
||||
|
||||
```javascript
|
||||
{
|
||||
name: item.name, // 仓库名
|
||||
fullName: item.full_name, // 完整名 owner/repo
|
||||
description: item.description, // 描述
|
||||
url: item.html_url, // GitHub 页面
|
||||
stars: item.stargazers_count, // stars 数
|
||||
forks: item.forks_count, // forks 数
|
||||
language: item.language, // 主要语言
|
||||
updatedAt: item.updated_at, // 更新时间
|
||||
owner: {
|
||||
name: item.owner.login, // 所有者名
|
||||
type: item.owner.type, // 'User' 或 'Organization'
|
||||
avatar: item.owner.avatar_url // 头像 URL
|
||||
},
|
||||
license: item.license?.name, // 许可证
|
||||
topics: item.topics // 标签数组
|
||||
}
|
||||
```
|
||||
|
||||
### 结果过滤与排序
|
||||
|
||||
过滤条件:
|
||||
1. 仓库名或描述包含查询词
|
||||
2. 不是 fork 的仓库(可选)
|
||||
3. 最近 2 年有更新(可选)
|
||||
|
||||
排序规则:
|
||||
1. 组织账号优先于个人账号
|
||||
2. stars 数高优先
|
||||
3. 最近更新优先
|
||||
|
||||
### 展示格式
|
||||
|
||||
```
|
||||
🐙 GitHub 搜索结果 (按 stars 排序):
|
||||
|
||||
1. company/skill-name ⭐ 1.2k
|
||||
🏢 Organization | MIT License
|
||||
📄 PDF processing and manipulation tools
|
||||
🏷️ pdf, document, converter
|
||||
|
||||
2. user/another-skill ⭐ 45
|
||||
👤 User | Apache-2.0
|
||||
📄 Simple PDF utilities
|
||||
🏷️ pdf, utils
|
||||
|
||||
3. ...
|
||||
```
|
||||
|
||||
## 综合搜索流程
|
||||
|
||||
当用户搜索时,执行以下流程:
|
||||
|
||||
```
|
||||
并行执行:
|
||||
├── Cocoloop API 搜索 ──────→ 结果 A
|
||||
└── GitHub API 搜索 ────────→ 结果 B
|
||||
|
||||
合并结果:
|
||||
1. 优先展示 Cocoloop 结果(官方源)
|
||||
2. 然后展示 GitHub 结果(社区源)
|
||||
3. 去重(相同 fullName 只保留一个)
|
||||
|
||||
展示:
|
||||
- 最多展示 10 个结果(可配置)
|
||||
- 标注来源(🌟 Cocoloop / 🐙 GitHub)
|
||||
- 显示关键信息(名称、描述、stars、来源类型)
|
||||
```
|
||||
|
||||
## 获取 Skill 详情
|
||||
|
||||
当用户选择某个 skill 后,获取详细信息:
|
||||
|
||||
### 对于 Cocoloop 源
|
||||
|
||||
直接读取 API 返回的完整信息。
|
||||
|
||||
### 对于 GitHub 源
|
||||
|
||||
1. **获取仓库详情**
|
||||
```
|
||||
GET https://api.github.com/repos/{owner}/{repo}
|
||||
```
|
||||
|
||||
2. **获取 SKILL.md 内容**
|
||||
```
|
||||
GET https://api.github.com/repos/{owner}/{repo}/contents/SKILL.md
|
||||
```
|
||||
响应中的 `content` 字段是 base64 编码的,需要解码。
|
||||
|
||||
3. **解析 SKILL.md**
|
||||
- 提取 frontmatter(name, description)
|
||||
- 提取前 500 字作为预览
|
||||
|
||||
4. **获取最新 release(可选)**
|
||||
```
|
||||
GET https://api.github.com/repos/{owner}/{repo}/releases/latest
|
||||
```
|
||||
|
||||
### 详情展示格式
|
||||
|
||||
```
|
||||
📋 Skill 详情
|
||||
|
||||
名称: pdf-processor
|
||||
版本: 1.0.0
|
||||
来源: 🐙 GitHub (Organization)
|
||||
⭐ Stars: 1250 | 🍴 Forks: 45
|
||||
📄 许可证: MIT
|
||||
🏷️ 标签: pdf, document, converter
|
||||
|
||||
描述:
|
||||
Advanced PDF processing and manipulation tools. Supports conversion,
|
||||
merging, splitting, and encryption.
|
||||
|
||||
SKILL.md 预览:
|
||||
---
|
||||
name: pdf-processor
|
||||
description: PDF processing skill...
|
||||
---
|
||||
# PDF Processor
|
||||
This skill provides tools for working with PDF files...
|
||||
|
||||
来源可信度: T2 (可信组织)
|
||||
安全评级: 待检查
|
||||
|
||||
是否安装此 skill? [Y/n]
|
||||
```
|
||||
|
||||
## 本地缓存搜索
|
||||
|
||||
为了提高重复搜索的速度,维护本地缓存:
|
||||
|
||||
### 缓存位置
|
||||
|
||||
`~/.cocoloop/cache/search.json`
|
||||
|
||||
### 缓存格式
|
||||
|
||||
```json
|
||||
{
|
||||
"query": "pdf",
|
||||
"timestamp": "2024-01-15T10:30:00Z",
|
||||
"results": [...],
|
||||
"expires": "2024-01-16T10:30:00Z"
|
||||
}
|
||||
```
|
||||
|
||||
### 缓存策略
|
||||
|
||||
- 缓存有效期:24 小时
|
||||
- 命中缓存时,询问用户是否使用缓存结果
|
||||
- 提供 `--fresh` 或 `-f` 参数强制刷新
|
||||
|
||||
## 错误处理
|
||||
|
||||
| 错误场景 | 处理方式 |
|
||||
|---------|---------|
|
||||
| Cocoloop API 超时 | 自动 fallback 到 GitHub |
|
||||
| GitHub API 限流 | 提示用户稍后重试,或使用本地缓存 |
|
||||
| 网络错误 | 显示错误信息,建议使用离线模式(如果有缓存)|
|
||||
| 解析错误 | 记录日志,跳过该结果,继续其他 |
|
||||
|
||||
## 高级搜索语法
|
||||
|
||||
支持以下搜索修饰符:
|
||||
|
||||
| 修饰符 | 含义 | 示例 |
|
||||
|-------|------|------|
|
||||
| `author:` | 限定作者 | `author:anthropic pdf` |
|
||||
| `lang:` | 限定语言 | `lang:javascript tool` |
|
||||
| `stars:>n` | stars 数大于 | `stars:>100 utility` |
|
||||
| `source:cocoloop` | 仅官方源 | `source:cocoloop document` |
|
||||
| `source:github` | 仅 GitHub | `source:github utility` |
|
||||
|
|
@ -1,163 +0,0 @@
|
|||
# Skill 卸载流程详细指南
|
||||
|
||||
本文档详细描述 skill 的卸载流程。
|
||||
|
||||
## 卸载前准备
|
||||
|
||||
### 1. 检测平台
|
||||
|
||||
使用与安装相同的平台检测逻辑:
|
||||
|
||||
```
|
||||
IF OpenClaw:
|
||||
安装目录 = ~/.openclaw/skills/
|
||||
配置文件 = ~/.openclaw/config.json
|
||||
|
||||
ELSE IF Molili:
|
||||
安装目录 = ~/.molili/skills/
|
||||
配置文件 = ~/.molili/config.json
|
||||
|
||||
ELSE IF Claude Code:
|
||||
安装目录 = ~/.claude/skills/
|
||||
配置文件 = ~/.claude/config.json
|
||||
|
||||
ELSE:
|
||||
安装目录 = ~/.claude/skills/ (clawhub 默认)
|
||||
配置文件 = ~/.claude/config.json
|
||||
```
|
||||
|
||||
### 2. 确认 Skill 存在
|
||||
|
||||
检查 skill 目录是否存在:
|
||||
|
||||
```
|
||||
{安装目录}/{skill-name}/
|
||||
├── SKILL.md
|
||||
├── scripts/
|
||||
├── references/
|
||||
└── assets/
|
||||
```
|
||||
|
||||
如果不存在:
|
||||
- 返回错误 "未找到该 skill,可能已卸载或名称错误"
|
||||
- 建议用户使用 `list` 命令查看已安装 skills
|
||||
|
||||
### 3. 获取 Skill 信息
|
||||
|
||||
读取 SKILL.md 获取基本信息:
|
||||
- name
|
||||
- description
|
||||
- version(如果有)
|
||||
|
||||
## 卸载流程
|
||||
|
||||
### 第一步:用户确认
|
||||
|
||||
展示将要卸载的 skill 信息,请求确认:
|
||||
|
||||
```
|
||||
⚠️ 即将卸载以下 skill:
|
||||
|
||||
名称: pdf-processor
|
||||
描述: PDF processing and manipulation tools
|
||||
安装路径: ~/.claude/skills/pdf-processor/
|
||||
|
||||
⚠️ 此操作将删除该 skill 的所有文件,不可恢复。
|
||||
|
||||
是否确认卸载? [y/N]
|
||||
```
|
||||
|
||||
可选:添加 `--force` 或 `-f` 参数跳过确认。
|
||||
|
||||
### 第二步:备份(可选)
|
||||
|
||||
如果用户指定 `--backup` 或 `-b` 参数:
|
||||
|
||||
1. 创建备份目录:`~/.cocoloop/backups/`
|
||||
2. 打包 skill 目录:`tar -czf ~/.cocoloop/backups/{skill-name}-{timestamp}.tar.gz {skill-path}/`
|
||||
3. 提示备份位置
|
||||
|
||||
### 第三步:执行卸载
|
||||
|
||||
1. **删除 skill 目录**
|
||||
```bash
|
||||
rm -rf {安装目录}/{skill-name}/
|
||||
```
|
||||
|
||||
2. **更新平台配置(如果需要)**
|
||||
- 某些平台维护已安装 skill 列表
|
||||
- 从列表中移除该 skill
|
||||
|
||||
3. **清理相关缓存**
|
||||
- 删除 Cocoloop 本地缓存中该 skill 的搜索记录
|
||||
- 删除安全检查缓存(如果有)
|
||||
|
||||
### 第四步:验证卸载
|
||||
|
||||
检查 skill 目录是否还存在:
|
||||
- 如果存在 → 返回错误 "卸载失败,请检查权限或手动删除"
|
||||
- 如果不存在 → 卸载成功
|
||||
|
||||
## 批量卸载
|
||||
|
||||
支持一次卸载多个 skills:
|
||||
|
||||
```
|
||||
卸载 skill1 skill2 skill3
|
||||
```
|
||||
|
||||
处理流程:
|
||||
1. 遍历每个 skill
|
||||
2. 执行单个卸载流程(不询问确认,或统一确认)
|
||||
3. 汇总结果:
|
||||
```
|
||||
📊 卸载结果:
|
||||
skill1: ✅ 已卸载
|
||||
skill2: ❌ 未找到
|
||||
skill3: ✅ 已卸载
|
||||
```
|
||||
|
||||
## 卸载后处理
|
||||
|
||||
### 依赖检查(可选)
|
||||
|
||||
检查是否有其他 skill 依赖被卸载的 skill:
|
||||
1. 遍历所有已安装 skills
|
||||
2. 检查它们的 dependencies(如果有记录)
|
||||
3. 如果有依赖关系,警告用户:
|
||||
```
|
||||
⚠️ 警告: 以下 skill 可能依赖 pdf-processor:
|
||||
- document-workflow
|
||||
|
||||
继续使用这些 skill 可能会出现问题。
|
||||
```
|
||||
|
||||
### 清理孤立依赖(高级)
|
||||
|
||||
如果 skill 安装了独立的依赖(如 node_modules),检查是否可以清理:
|
||||
- 如果其他 skill 不使用 → 可以删除
|
||||
- 如果有共享依赖 → 保留
|
||||
|
||||
## 错误处理
|
||||
|
||||
| 错误场景 | 处理方式 |
|
||||
|---------|---------|
|
||||
| 权限不足 | 提示使用 `sudo` 或检查目录权限 |
|
||||
| 文件被占用 | 提示关闭使用该 skill 的程序后重试 |
|
||||
| 目录非空但无法删除 | 保留日志,提示手动删除 |
|
||||
| 配置文件损坏 | 尝试修复或重建配置 |
|
||||
|
||||
## 恢复卸载
|
||||
|
||||
如果用户误卸载,提供恢复选项(前提是备份存在):
|
||||
|
||||
```
|
||||
恢复 pdf-processor
|
||||
```
|
||||
|
||||
流程:
|
||||
1. 查找备份目录:`~/.cocoloop/backups/pdf-processor-*.tar.gz`
|
||||
2. 列出可用备份(按时间排序)
|
||||
3. 询问用户选择恢复哪个版本
|
||||
4. 解压到安装目录
|
||||
5. 验证恢复
|
||||
21
runtime/skills/data-analyst/SKILL.md
Normal file
21
runtime/skills/data-analyst/SKILL.md
Normal file
|
|
@ -0,0 +1,21 @@
|
|||
---
|
||||
name: data-analyst
|
||||
description: Complete the data analysis tasks delegated by the user.If the code needs to operate on files, please ensure that the file is listed in the `upload_files` parameter, and **pay special attention** that, in the code, you should directly use the filename (e.g., `open('data.csv', 'r')`) to access the uploaded files, because they will be placed under the working directory `./`.
|
||||
---
|
||||
|
||||
# Data Analyst
|
||||
|
||||
## Overview
|
||||
|
||||
This skill provides specialized capabilities for data analyst.
|
||||
|
||||
## Instructions
|
||||
|
||||
Complete the data analysis tasks delegated by the user.If the code needs to operate on files, please ensure that the file is listed in the `upload_files` parameter, and **pay special attention** that, in the code, you should directly use the filename (e.g., `open('data.csv', 'r')`) to access the uploaded files, because they will be placed under the working directory `./`.
|
||||
|
||||
|
||||
## Usage Notes
|
||||
|
||||
- This skill is based on the data_analyst agent configuration
|
||||
- Template variables (if any) like $DATE$, $SESSION_GROUP_ID$ may require runtime substitution
|
||||
- Follow the instructions and guidelines provided in the content above
|
||||
6
runtime/skills/data-analyst/_meta.json
Normal file
6
runtime/skills/data-analyst/_meta.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn71xkrq2fawjvteej73gsx71s80p3kb",
|
||||
"slug": "data-analyst-pro",
|
||||
"version": "0.1.0",
|
||||
"publishedAt": 1771141367019
|
||||
}
|
||||
7
runtime/skills/oclaw-skill-manager/README.md
Normal file
7
runtime/skills/oclaw-skill-manager/README.md
Normal file
|
|
@ -0,0 +1,7 @@
|
|||
# oclaw-skill-manager
|
||||
|
||||
Oclaw **内置** Skill:说明如何在当前仓库中安装、更新、卸载技能,以及依赖与健康检查。
|
||||
|
||||
- 主文档:[SKILL.md](SKILL.md)
|
||||
- 本包**不是**任何外部「技能市场 CLI」的封装;平台不提供官方 shell 一键安装命令。
|
||||
- 模型安装策略:仅允许 `skill_auto_install`;失败时仅报告 `error_code/detail`,禁止绕路安装。
|
||||
87
runtime/skills/oclaw-skill-manager/SKILL.md
Normal file
87
runtime/skills/oclaw-skill-manager/SKILL.md
Normal file
|
|
@ -0,0 +1,87 @@
|
|||
---
|
||||
name: oclaw-skill-manager
|
||||
description: Oclaw 官方 Skill 生命周期手册:安装、更新、卸载、依赖与健康检查。仅使用本仓库 Admin API、安装器与内置工具;不提供也不推荐任何 shell「一键装技能」命令。
|
||||
---
|
||||
|
||||
# Oclaw Skill 管理器(官方手册)
|
||||
|
||||
本技能是 **Oclaw 自有** 的 Skill 运维说明,用于指导模型与运维在**同一套契约**下管理 `runtime/skills` 下的技能包。
|
||||
|
||||
## 系统强约束(模型必须遵守)
|
||||
|
||||
- 规范优先:安装决策只依据本文件,不依据临时推断或“经验性排障说法”。
|
||||
- 路径约束:当任务目标是“为当前会话/用户安装 skill”时,**只允许**走 `skill_auto_install`(`_workspace` lane)。
|
||||
- 禁止绕路:不得调用任何非 auto 安装路径(包括但不限于 `skill_market_install`、`skill_registry_install`、Admin 安装 API、手写下载/解压脚本落盘到 `runtime/skills`)。
|
||||
- 失败处理:`skill_auto_install` 失败后,**不得继续安装**;只报告 `error_code/detail` 与最小下一步,等待用户指示。
|
||||
- 禁止臆测前置条件:未在本规范显式声明的环境变量/端口/服务状态,不得被表述为“安装必经条件”。
|
||||
|
||||
## 平台事实(无「原生安装命令」)
|
||||
|
||||
- Oclaw **不提供**任何官方 shell「一键装技能」命令(包括市场 CLI、`npx …` 拉 CLI 再 `install` 等模式)。
|
||||
- 模型安装策略:默认仅使用 **`skill_auto_install`**;其他安装能力仅供管理员/后端运维链路使用。
|
||||
- 在沙箱里执行 `run_command` 时,**外部技能 CLI 安装模式会被拦截**(见 `shell_tools`),请改用下文 API。
|
||||
|
||||
## 目录策略
|
||||
|
||||
| 场景 | 路径 |
|
||||
|------|------|
|
||||
| 人工 / Admin 市场或 registry 安装 | `<skills_root>/<manifest_name>/` |
|
||||
| 智能体 payload 自动安装 | `<skills_root>/_workspace/<manifest_name>/` |
|
||||
|
||||
`<skills_root>` 默认 `runtime/skills/`,可被 **`AIA_SKILLS_ROOT`** 覆盖。
|
||||
|
||||
## 技能市场提供方(ClawHub + CocoLoop)
|
||||
|
||||
租户设置 **`AIA_SKILL_MARKET_PROVIDER`** 选择市场(网关 `get_market_adapter` 读取):
|
||||
|
||||
| 取值 | 说明 |
|
||||
|------|------|
|
||||
| **`clawhub`**(默认) | [ClawHub](https://clawhub.ai) 公开技能注册表;HTTP 形态与官方 CLI 一致,见上游文档 [CLI / Registry](https://github.com/openclaw/clawhub/blob/main/docs/cli.md)(`/api/v1/search`、`/api/v1/skills/{slug}`、`/api/v1/download?slug=&version=`)。本仓库客户端:`runtime/tools/skills/clawhub_client.py`,环境变量 **`AIA_CLAWHUB_SITE` / `AIA_CLAWHUB_REGISTRY` / `AIA_CLAWHUB_TOKEN`**(或 `CLAWHUB_*`)与官方 `CLAWHUB_*` 对齐。 |
|
||||
| **`cocoloop`** | [CocoLoop 技能商店](https://hub.cocoloop.cn) 开放列表接口:`GET {api}/api/v1/store/skills`(分页、`keyword`、`sort`),详情:`GET {api}/api/v1/store/skills/{id}`;列表项中的 **`download_url`** 为 zip 直链(常见域名 `dl.cocoloop.cn`)。实现:`runtime/tools/skills/cocoloop_client.py`;可选 **`AIA_COCOLOOP_API_BASE`**(默认 `https://api.cocoloop.com`)。别名:`cocoloop-cn`、`cocoloop_cn` 与 `cocoloop` 相同。 |
|
||||
|
||||
安装仍统一走 **`install_skill_from_registry_archive`**:对 ClawHub 与 CocoLoop 均为 **HTTPS zip 归档 URL**,无需在服务器上安装 `clawhub` / `cocoloop` CLI。
|
||||
|
||||
## 发现与安装(模型视角)
|
||||
|
||||
### 唯一安装路径(必须)
|
||||
|
||||
- **`skill_auto_install`**:仅写入 `_workspace` lane(见 `skill_installer.auto_install_skill_from_payload`)。
|
||||
- **非前置条件澄清**:`AIA_INTERNAL_BASE_URL`、Admin `market/search`、本地 5173 服务都不是 `skill_auto_install` 的必需前置。
|
||||
- 若安装失败,向用户返回:
|
||||
- `error_code`
|
||||
- `detail`
|
||||
- 建议下一步(例如补充 name/description、检查依赖、重试)
|
||||
- 不允许改走其它安装入口完成同一目标。
|
||||
|
||||
## 列表、开关、卸载
|
||||
|
||||
- **`skill_list`**(若已绑定到当前专家)
|
||||
- `GET /admin/api/skills`
|
||||
- `POST /admin/api/skills/enable` / `disable`,`{ "name": "..." }`
|
||||
- `POST /admin/api/skills/uninstall`,`{ "name": "..." }`
|
||||
|
||||
## 依赖与健康
|
||||
|
||||
- 安装后:`requirements.txt`、`package.json`(`dependencies`)、Python import 探测与补装(受 `AIA_SKILL_AUTO_INSTALL_DEPS_ENABLED` 控制)。
|
||||
- Admin:**Repair deps** / **Repair all deps**
|
||||
- `GET /admin/api/skills/self-check?include_execution=...`
|
||||
|
||||
## 更新
|
||||
|
||||
模型侧无独立 update 安装路径;如需更新,按用户指示走新的 `skill_auto_install` 版本草案或转人工管理员操作。
|
||||
|
||||
## 可选人工安全审查
|
||||
|
||||
- [references/safety-check-guide.md](references/safety-check-guide.md)
|
||||
- [references/skill-safety-rubric.md](references/skill-safety-rubric.md)
|
||||
均为**人工参考**,平台不自动执行远程认证。
|
||||
|
||||
## 子文档
|
||||
|
||||
- [references/install-guide.md](references/install-guide.md)
|
||||
- [references/search-guide.md](references/search-guide.md)
|
||||
- [references/uninstall-guide.md](references/uninstall-guide.md)
|
||||
|
||||
---
|
||||
|
||||
维护本技能时,请只增删 **Oclaw 已实现** 的行为,勿再引入第三方「技能 CLI」作为默认路径。
|
||||
|
|
@ -0,0 +1,53 @@
|
|||
# Oclaw — Skill 安装详细指南
|
||||
|
||||
本文档仅描述 **Oclaw** 内的安装行为,替代旧版「多平台检测 + Cocoloop API」流程。
|
||||
|
||||
## 模型执行硬规则
|
||||
|
||||
- 仅允许 `skill_auto_install`。
|
||||
- 安装失败时只允许“报告失败原因给用户”,禁止改走其它安装入口(market/install、install-registry、install、本地脚本解压落盘等)。
|
||||
- 不得把未在规范中声明的环境变量/端口/服务状态当作安装前置条件。
|
||||
|
||||
## 1. 技能根目录
|
||||
|
||||
- 默认:`runtime/skills/`(或环境变量 `AIA_SKILLS_ROOT` 指向的目录)
|
||||
- **主目录**:`<skills_root>/<name>/` — Admin 市场 / registry / 本地目录安装默认落点
|
||||
- **智能体自写目录**:`<skills_root>/_workspace/<name>/` — `skill_auto_install` / `auto_install_skill_from_payload` 等
|
||||
|
||||
具体以安装接口返回的 `target_dir` 为准。
|
||||
|
||||
## 2. 安装入口对照(运维参考)
|
||||
|
||||
| 场景 | 方式 | HTTP(Admin) |
|
||||
|------|------|----------------|
|
||||
| ClawHub slug | 解析 `archive_url` 后安装 | `POST /admin/api/skills/market/install` |
|
||||
| 已知归档 URL | 直接拉取 zip/tar | `POST /admin/api/skills/install-registry` |
|
||||
| 本地已展开目录 | 目录内含 `SKILL.md` | `POST /admin/api/skills/install` |
|
||||
| 从模板创建 | 生成新包 | `POST /admin/api/skills/create` |
|
||||
| Workspace 模板 | 带 runtime 桩 | `POST /admin/api/skills/create-workspace` |
|
||||
|
||||
认证:Admin 路由需带网关要求的 `Authorization`(与现有 Admin 一致)。
|
||||
|
||||
## 3. 依赖与自检
|
||||
|
||||
安装成功后,安装器会尽量:
|
||||
|
||||
1. 处理 `requirements.txt`、`package.json`(`dependencies` 非空)
|
||||
2. 扫描 `.py` 的 import,对缺失的第三方模块尝试 `pip install`
|
||||
|
||||
失败不一定会回滚整个目录,可能返回带 `installed_with_dependency_warnings` 的 `detail`。此时在 Admin 使用 **Repair deps** 或 **Repair all deps**。
|
||||
|
||||
## 4. 重试与覆盖
|
||||
|
||||
- 安装失败审计里若带 `retryable`,可用 `POST /admin/api/skills/retry-install`(见 Admin 实现)
|
||||
- 覆盖安装:`overwrite: true`
|
||||
|
||||
## 5. 禁止项
|
||||
|
||||
- **不要**使用任何「第三方技能 CLI + install」作为安装路径(`run_command` 会拦截常见模式)
|
||||
- **不要**把非本仓库契约的 HTTP 商店当作主源;技能发现以 **ClawHub 市场适配器**(`AIA_SKILL_MARKET_PROVIDER`)为准
|
||||
- **模型侧不要**在 `skill_auto_install` 失败后切换到 Admin 安装 API 或脚本直装。
|
||||
|
||||
## 6. slug 与包名
|
||||
|
||||
ClawHub 返回的 **slug** 可能与解压后 `SKILL.md` frontmatter 里的 **name** 不同。卸载、启用、绑定角色时以 **`skill_list` / API 返回的 `name`** 为准。
|
||||
|
|
@ -1,6 +1,8 @@
|
|||
# Cocoloop Safe Check 安全检查流程指南
|
||||
> **Oclaw 说明**:本文件为**可选的人工安全审查**参考流程。Oclaw **不会**自动调用 Cocoloop BSS 或远程「Safe Check」服务;安装与运行以平台自带沙箱、工具策略与租户配置为准。执行审查前请确认符合本地合规要求。
|
||||
|
||||
本文档详细描述 Cocoloop 安全检查的执行流程,基于 cocoloop-safe-check 安全认证体系。
|
||||
# Skill 安全检查流程指南(参考)
|
||||
|
||||
本文档描述一套**可参考**的人工安全检查执行流程,可与 [skill-safety-rubric.md](skill-safety-rubric.md) 配合阅读。Oclaw 不保证与任何第三方「安全认证产品」行为一致。
|
||||
|
||||
## 检查触发时机
|
||||
|
||||
|
|
@ -0,0 +1,41 @@
|
|||
# Oclaw — Skill 搜索指南
|
||||
|
||||
## 市场提供方
|
||||
|
||||
由设置 **`AIA_SKILL_MARKET_PROVIDER`** 决定:`clawhub`(默认)或 `cocoloop`(见主 `SKILL.md` 对照表)。Admin 的 `market/search` 与 `market/detail` 会调用当前提供方适配器。
|
||||
|
||||
## 主源:ClawHub
|
||||
|
||||
当 `AIA_SKILL_MARKET_PROVIDER=clawhub` 时使用。
|
||||
|
||||
### Admin HTTP(ClawHub 模式下)
|
||||
|
||||
- 搜索:`GET /admin/api/skills/market/search?q=<关键词>&limit=<n>`
|
||||
- 详情:`GET /admin/api/skills/market/detail?slug=<slug>`
|
||||
|
||||
从详情中读取:`slug`、`version`、描述、以及安装所需的 **`archiveUrl`**(ClawHub 下载链)。
|
||||
|
||||
## 主源:CocoLoop 商店
|
||||
|
||||
当 `AIA_SKILL_MARKET_PROVIDER=cocoloop` 时,同一组 Admin 路由背后走 **`cocoloop_client`**:关键词搜索商店列表,按技能 **`name` 字段** 匹配 slug;详情中的安装 URL 来自列表 **`download_url`**(或按 `asset_name` 拼 zip 直链)。商店前端:[hub.cocoloop.cn](https://hub.cocoloop.cn)。
|
||||
|
||||
### 模型侧
|
||||
|
||||
若无 Admin 权限,请用户代为搜索/安装,或提供准确 **slug** / **archive_url**。
|
||||
|
||||
> 重要:市场搜索不是安装前置条件。
|
||||
> 模型安装策略下只允许 `skill_auto_install`。若没有可用市场结果,应向用户索取可安装内容(如技能描述、`SKILL.md` 或源文件)并走 `skill_auto_install`,而不是要求先配置 `AIA_INTERNAL_BASE_URL` 或先启动本地 5173 服务。
|
||||
|
||||
## 辅助源:GitHub(可选)
|
||||
|
||||
当市场无结果或用户指定开源仓库时:
|
||||
|
||||
```
|
||||
GET https://api.github.com/search/repositories?q=<关键词>+filename:SKILL.md&sort=stars&order=desc
|
||||
```
|
||||
|
||||
需自备 `User-Agent`,注意 API 速率限制。找到仓库后仍需**可安装的归档 URL** 再走 `install-registry`。
|
||||
|
||||
## 合并展示建议
|
||||
|
||||
向用户展示时标注来源:`[ClawHub]` / `[GitHub]`,并给出 **slug** 或 **full_name**。
|
||||
|
|
@ -1,6 +1,8 @@
|
|||
# Cocoloop Safe Check 安全检查标准
|
||||
> **Oclaw 说明**:以下为**评级与检查维度参考**,供人工审阅 skill 时使用。Oclaw **不**根据本文件自动打分或拦截安装;实际风险控制依赖权限、审计、工具白名单与运行环境隔离。
|
||||
|
||||
本文件定义了 Cocoloop Skill 管理器的安全检查标准。
|
||||
# Skill 安全审查量表(参考)
|
||||
|
||||
本文件提供人工审阅时可用的检查维度与分级思路,**不**作为 Oclaw 运行时强制契约。
|
||||
|
||||
## 评级标准
|
||||
|
||||
|
|
@ -0,0 +1,32 @@
|
|||
# Oclaw — Skill 卸载指南
|
||||
|
||||
## 路径
|
||||
|
||||
卸载逻辑会依次查找(存在则删除):
|
||||
|
||||
1. `<skills_root>/<name>/`
|
||||
2. `<skills_root>/_workspace/<name>/`
|
||||
|
||||
其中 `<skills_root>` 为 `AIA_SKILLS_ROOT` 或默认 `runtime/skills/`。
|
||||
|
||||
`<name>` 为 **`SKILL.md` frontmatter 中的 `name`**(与 `skill_list` 中 `name` 字段一致),不一定等于 ClawHub **slug**。
|
||||
|
||||
## Admin
|
||||
|
||||
- `POST /admin/api/skills/uninstall`
|
||||
Body:`{ "name": "<skill manifest name>" }`
|
||||
|
||||
卸载前应在 UI 或对话中向用户确认;删除后不可恢复(除非有外部备份)。
|
||||
|
||||
## 启用状态
|
||||
|
||||
卸载实现中会尝试将技能从禁用列表恢复为可用状态(见 `skill_installer.uninstall_skill`);若需保留禁用记录,请在产品中另行约定(当前以代码为准)。
|
||||
|
||||
## 批量卸载
|
||||
|
||||
对每个名称依次调用卸载接口,独立汇总结果。
|
||||
|
||||
## 与旧版差异
|
||||
|
||||
- ~~`rm -rf ~/.openclaw/skills/`~~ 等路径不适用于本仓库默认布局
|
||||
- 不使用任何外部「技能卸载 CLI」;一律走 Admin `uninstall` 或安装器 API
|
||||
97
runtime/skills/skillscan/SKILL.md
Normal file
97
runtime/skills/skillscan/SKILL.md
Normal file
|
|
@ -0,0 +1,97 @@
|
|||
---
|
||||
name: SkillScan
|
||||
metadata:
|
||||
version: "1.1.6"
|
||||
description: >
|
||||
Security gate for skills. Every new skill MUST pass SkillScan before use.
|
||||
Activate on any install, load, add, evaluate, or safety question about a skill.
|
||||
On first load, run first-run to scan all existing skills.
|
||||
Blocks HIGH/CRITICAL skills. No exceptions.
|
||||
---
|
||||
|
||||
# SkillScan v1.1.6
|
||||
|
||||
Security check for skill packages. Every new skill must pass this scan before use.
|
||||
|
||||
Implemented in `scripts/scanner.py`. Supports Windows / macOS / Linux.
|
||||
|
||||
Do NOT generate report files inside the skill directory.
|
||||
|
||||
---
|
||||
|
||||
## When to Activate
|
||||
|
||||
Run SkillScan whenever a skill is installed, loaded, added, evaluated, or questioned for safety — regardless of method or phrasing.
|
||||
|
||||
Trigger examples: install, load, add, use, set up, "is this safe", upload .zip/.skill, paste SKILL.md, mention unknown skill name, install from hub/store.
|
||||
|
||||
---
|
||||
|
||||
## First Load
|
||||
|
||||
Check `scripts/.first_run_done`:
|
||||
|
||||
- **Not found** → **Tell user** SkillScan is installed, ask to scan all existing skills:
|
||||
```bash
|
||||
python <skill_scanner_dir>/scripts/scanner.py first-run
|
||||
```
|
||||
`.first_run_done` is created automatically after completion.
|
||||
|
||||
- **Found** → Skip, operate normally.
|
||||
|
||||
---
|
||||
|
||||
## Commands
|
||||
|
||||
| Command | Usage |
|
||||
|---------|-------|
|
||||
| `scanner.py scan <path>` | Scan a single skill (.zip or directory) |
|
||||
| `scanner.py scan-all` | Scan all installed skills |
|
||||
| `scanner.py first-run` | First-time full scan |
|
||||
| `scanner.py upgrade` | Manual upgrade |
|
||||
|
||||
---
|
||||
|
||||
## Scan Rules
|
||||
|
||||
- **.zip files** → Scan BEFORE installation. Block if fails.
|
||||
- **Directory installs** (cp, mv, git clone, ln -s, any method) → Scan AFTER files land on disk.
|
||||
- **Remote installs** (clawhub, skillhub, npx skills add, etc.) → Scan immediately after install.
|
||||
- **Unknown skills** → If user mentions a skill you haven't seen, scan it.
|
||||
|
||||
---
|
||||
|
||||
## Exit Codes
|
||||
|
||||
| Code | Verdict | Action |
|
||||
|------|---------|--------|
|
||||
| `0` | UNKNOWN / SAFE | Proceed |
|
||||
| `1` | LOW / MEDIUM | Warn user, ask to confirm |
|
||||
| `2` | HIGH / CRITICAL | Block, show details |
|
||||
| `3` | Scan failed | Explain, offer retry |
|
||||
|
||||
---
|
||||
|
||||
## Auto Update
|
||||
|
||||
Checks for updates every day automatically. Silent, no user action needed. Manual: `scanner.py upgrade`.
|
||||
|
||||
---
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `SKILL_SCANNER_UPDATE_URL` | Custom update source (optional) |
|
||||
|
||||
---
|
||||
|
||||
## API
|
||||
|
||||
Base URL: `https://skillscan.tokauth.com`
|
||||
|
||||
| Step | Method | Path |
|
||||
|------|--------|------|
|
||||
| ① Cache lookup | GET | `/oapi/v1/skill-scan/search?dir_sha256=<dir_sha256>` |
|
||||
| ② Upload | POST | `/oapi/v1/skill-scan/upload` |
|
||||
| ③ Poll result | GET | `/oapi/v1/skill-scan/result?task_no=<task_no>` (poll every 20s, max 180s) |
|
||||
6
runtime/skills/skillscan/_meta.json
Normal file
6
runtime/skills/skillscan/_meta.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn791cyx98pcsezkh5088g8jxn84c7mm",
|
||||
"slug": "skillscan",
|
||||
"version": "1.1.6",
|
||||
"publishedAt": 1776650587310
|
||||
}
|
||||
959
runtime/skills/skillscan/scripts/scanner.py
Normal file
959
runtime/skills/skillscan/scripts/scanner.py
Normal file
|
|
@ -0,0 +1,959 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
SkillScan v1.1.5 — OpenClaw Skill security scanner.
|
||||
Supports Windows / macOS / Linux. All temp files use the standard tempfile module.
|
||||
|
||||
Usage (invoked by the agent via bash):
|
||||
python scanner.py first-run # First install: list installed skills and ask to scan
|
||||
python scanner.py scan <path> # Scan a single skill (.zip or directory)
|
||||
python scanner.py scan-all # Scan all installed skills
|
||||
python scanner.py upgrade # Auto-upgrade
|
||||
"""
|
||||
|
||||
import sys, os, json, time, zipfile, hashlib, shutil, tempfile, uuid, platform, base64
|
||||
import urllib.request, urllib.error, urllib.parse
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Configuration
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
SCANNER_VERSION = "1.1.5"
|
||||
|
||||
BASE_URL = "https://skillscan.tokauth.com"
|
||||
API_SEARCH = f"{BASE_URL}/oapi/v1/skill-scan/search"
|
||||
API_UPLOAD = f"{BASE_URL}/oapi/v1/skill-scan/upload"
|
||||
API_RESULT = f"{BASE_URL}/oapi/v1/skill-scan/result"
|
||||
UPDATE_URL = os.environ.get("SKILL_SCANNER_UPDATE_URL",
|
||||
f"{BASE_URL}/downloads/SkillScan/manifest")
|
||||
|
||||
POLL_INTERVAL = 20 # Poll interval (seconds)
|
||||
POLL_TIMEOUT = 180 # Max wait time (seconds)
|
||||
|
||||
# First-run marker file (in the same directory as scanner.py)
|
||||
STATE_FILE = Path(__file__).parent / ".first_run_done"
|
||||
|
||||
# Auto-update check marker file and interval (7 days)
|
||||
LAST_UPDATE_CHECK_FILE = Path(__file__).parent / ".last_update_check"
|
||||
AUTO_UPDATE_INTERVAL = 1 * 24 * 3600 # 1 day (seconds)
|
||||
|
||||
# Client info file (generated on first run, reused afterwards)
|
||||
CLIENT_INFO_FILE = Path(__file__).parent / ".client_info"
|
||||
|
||||
# Files and directories to skip during scanning, hashing, and packing
|
||||
SKIP_FILES = {".first_run_done", ".last_update_check", ".client_info", "cloud_report.json", ".DS_Store"}
|
||||
SKIP_DIRS = {".git", "__pycache__", ".venv", "node_modules", ".idea", ".vscode", ".clawhub"}
|
||||
|
||||
# Resolve the root directory of SkillScan itself (parent of scripts/)
|
||||
SELF_ROOT = Path(__file__).parent.parent.resolve()
|
||||
|
||||
|
||||
# Skill installation paths (cross-platform)
|
||||
def skill_install_paths():
|
||||
# type: () -> list
|
||||
"""Auto-enumerate OpenClaw and local skill paths across platforms."""
|
||||
home = Path.home()
|
||||
oc_dir = home / ".openclaw"
|
||||
candidates = [
|
||||
# OpenClaw standard paths
|
||||
oc_dir / "skills",
|
||||
oc_dir / "workspace/skills",
|
||||
# Shared agent skill paths
|
||||
home / ".agents/skills",
|
||||
home / ".config/agents/skills",
|
||||
# Agent-specific global paths
|
||||
home / ".gemini/antigravity/skills",
|
||||
home / ".gemini/skills",
|
||||
home / ".augment/skills",
|
||||
home / ".claude/skills",
|
||||
home / ".codex/skills",
|
||||
home / ".commandcode/skills",
|
||||
home / ".continue/skills",
|
||||
home / ".snowflake/cortex/skills",
|
||||
home / ".config/crush/skills",
|
||||
home / ".cursor/skills",
|
||||
home / ".deepagents/agent/skills",
|
||||
home / ".factory/skills",
|
||||
home / ".firebender/skills",
|
||||
home / ".copilot/skills",
|
||||
home / ".config/goose/skills",
|
||||
home / ".junie/skills",
|
||||
home / ".iflow/skills",
|
||||
home / ".kilocode/skills",
|
||||
home / ".kiro/skills",
|
||||
home / ".kode/skills",
|
||||
home / ".mcpjam/skills",
|
||||
home / ".vibe/skills",
|
||||
home / ".mux/skills",
|
||||
home / ".config/opencode/skills",
|
||||
home / ".openhands/skills",
|
||||
home / ".pi/agent/skills",
|
||||
home / ".qoder/skills",
|
||||
home / ".qwen/skills",
|
||||
home / ".roo/skills",
|
||||
home / ".trae/skills",
|
||||
home / ".trae-cn/skills",
|
||||
home / ".codeium/windsurf/skills",
|
||||
home / ".zencoder/skills",
|
||||
home / ".neovate/skills",
|
||||
home / ".pochi/skills",
|
||||
home / ".adal/skills",
|
||||
home / ".npm-global/lib/node_modules/openclaw/skills",
|
||||
# Container default paths
|
||||
Path("/mnt/skills/public"),
|
||||
Path("/mnt/skills/private"),
|
||||
Path("/mnt/skills/user"),
|
||||
# User dev/download paths
|
||||
home / "Downloads/skills",
|
||||
]
|
||||
|
||||
# Windows-specific paths
|
||||
if os.name == "nt":
|
||||
appdata = os.environ.get("APPDATA")
|
||||
if appdata:
|
||||
candidates.append(Path(appdata) / "OpenClaw/skills")
|
||||
candidates.append(Path(appdata) / "Programs/LobsterAI/resources/SKILLs")
|
||||
|
||||
# Dynamically scan extensions: .openclaw/extensions/{xxxx}/skills
|
||||
if oc_dir.exists():
|
||||
ext_root = oc_dir / "extensions"
|
||||
if ext_root.exists():
|
||||
for sub in ext_root.iterdir():
|
||||
if sub.is_dir():
|
||||
s_dir = sub / "skills"
|
||||
if s_dir.exists():
|
||||
candidates.append(s_dir)
|
||||
|
||||
# Include script run path and workspace
|
||||
candidates.append(Path.cwd() / "skills")
|
||||
candidates.append(Path(__file__).parent.parent / "skills")
|
||||
|
||||
# Deduplicate and filter non-existent paths
|
||||
seen = set()
|
||||
result = []
|
||||
for p in candidates:
|
||||
try:
|
||||
abs_p = p.resolve()
|
||||
if abs_p.exists() and abs_p not in seen:
|
||||
result.append(p)
|
||||
seen.add(abs_p)
|
||||
except Exception:
|
||||
continue
|
||||
return result
|
||||
|
||||
RISK_EMOJI = {"SAFE":"✅","LOW":"⚠️ ","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️ "}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Client Info (X-Client-Info)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _get_mac_address():
|
||||
"""Try to get the MAC address; return empty string on failure."""
|
||||
try:
|
||||
import uuid as _uuid
|
||||
mac_int = _uuid.getnode()
|
||||
# getnode() returns a random value (bit 8 set) when it can't get the real MAC
|
||||
if (mac_int >> 40) & 1:
|
||||
return ""
|
||||
mac_str = ":".join(("%012X" % mac_int)[i:i+2] for i in range(0, 12, 2))
|
||||
return mac_str
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _build_client_info():
|
||||
"""Build client info dict and persist to file; reuse on subsequent runs."""
|
||||
# If a record file already exists, read it
|
||||
if CLIENT_INFO_FILE.exists():
|
||||
try:
|
||||
data = json.loads(CLIENT_INFO_FILE.read_text(encoding="utf-8"))
|
||||
if data.get("client_id"):
|
||||
return data
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# First run: generate new client info
|
||||
info = {
|
||||
"client_id": str(uuid.uuid4()),
|
||||
"os": platform.system() or "",
|
||||
"platform": platform.machine() or "",
|
||||
"os_version": platform.release() or "",
|
||||
"client": "SkillScanner/%s" % SCANNER_VERSION,
|
||||
}
|
||||
|
||||
mac = _get_mac_address()
|
||||
if mac:
|
||||
info["mac"] = mac
|
||||
|
||||
# Python version as extra
|
||||
info["extra"] = {
|
||||
"python": platform.python_version(),
|
||||
}
|
||||
|
||||
# Persist
|
||||
try:
|
||||
CLIENT_INFO_FILE.write_text(
|
||||
json.dumps(info, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8"
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return info
|
||||
|
||||
|
||||
def _get_client_info_header():
|
||||
"""Return Base64-encoded X-Client-Info header value; empty string on failure."""
|
||||
try:
|
||||
info = _build_client_info()
|
||||
json_str = json.dumps(info, ensure_ascii=False)
|
||||
encoded = base64.b64encode(json_str.encode("utf-8")).decode("ascii")
|
||||
return encoded
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Output Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def banner(title: str):
|
||||
w = 58
|
||||
print(f"\n{'═'*w}")
|
||||
print(f" {title}")
|
||||
print(f"{'═'*w}")
|
||||
|
||||
def divider(title: str = ""):
|
||||
if title:
|
||||
print(f"\n ── {title} {'─'*(48-len(title))}")
|
||||
else:
|
||||
print(f" {'─'*52}")
|
||||
|
||||
def log(msg: str):
|
||||
print(f" {msg}", flush=True)
|
||||
|
||||
def ask(prompt: str) -> str:
|
||||
"""Read user input (compatible with non-interactive environments)."""
|
||||
try:
|
||||
return input(f"\n {prompt} ").strip()
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# HTTP Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def http_get(url: str) -> dict:
|
||||
req = urllib.request.Request(url)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
def http_post(url: str, payload: dict) -> dict:
|
||||
headers = {"Content-Type": "application/json"}
|
||||
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Skill Utilities
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def skill_name_from_dir(skill_dir: Path) -> str:
|
||||
md = skill_dir / "SKILL.md"
|
||||
if md.exists():
|
||||
for line in md.read_text(encoding="utf-8", errors="replace").splitlines():
|
||||
s = line.strip()
|
||||
if s.startswith("name:"):
|
||||
return s.split(":", 1)[1].strip().strip("\"'")
|
||||
return skill_dir.name
|
||||
|
||||
def sha256_of(path: Path) -> str:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
def calculate_dir_sha256(directory: Path) -> str:
|
||||
"""Calculate SHA256 hash of a skill directory (based on all file contents + relative paths).
|
||||
Excludes _meta.json and files/dirs in SKIP_FILES/SKIP_DIRS."""
|
||||
file_hashes = []
|
||||
for file_path in sorted(directory.rglob('*')):
|
||||
if not file_path.is_file():
|
||||
continue
|
||||
if file_path.name == '_meta.json':
|
||||
continue
|
||||
rel = file_path.relative_to(directory)
|
||||
if any(part in SKIP_DIRS for part in rel.parts):
|
||||
continue
|
||||
if file_path.name in SKIP_FILES:
|
||||
continue
|
||||
rel_path = str(rel)
|
||||
file_hash = hashlib.sha256()
|
||||
file_hash.update(rel_path.encode('utf-8'))
|
||||
file_hash.update(b'\x00')
|
||||
with open(file_path, 'rb') as f:
|
||||
for chunk in iter(lambda: f.read(8192), b''):
|
||||
file_hash.update(chunk)
|
||||
file_hashes.append(file_hash.hexdigest())
|
||||
file_hashes.sort()
|
||||
final_hash = hashlib.sha256()
|
||||
for h in file_hashes:
|
||||
final_hash.update(h.encode('utf-8'))
|
||||
final_hash.update(b'\x00')
|
||||
return final_hash.hexdigest()
|
||||
|
||||
def collect_files(skill_dir: Path) -> dict:
|
||||
"""Collect files for scanning, skipping redundant or sensitive directories."""
|
||||
exts = {".md",".py",".js",".ts",".sh",".yaml",".yml",".json",".txt"}
|
||||
out = {}
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
if p.suffix.lower() in exts or p.name == "SKILL.md":
|
||||
try:
|
||||
out[str(p.relative_to(skill_dir))] = \
|
||||
p.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
def pack_zip(skill_dir: Path) -> bytes:
|
||||
"""Pack a skill directory into a zip byte stream, excluding redundant directories."""
|
||||
import io
|
||||
buf = io.BytesIO()
|
||||
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
zf.write(p, p.relative_to(skill_dir))
|
||||
return buf.getvalue()
|
||||
|
||||
def unpack_zip(zip_path: Path) -> Path:
|
||||
"""Extract a .zip to a system temp directory. Returns the extraction path. Prevents zip-slip."""
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
log(f"📦 Extracting {zip_path.name} → {tmp}")
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / member).resolve()
|
||||
if not str(dest).startswith(str(tmp.resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp)
|
||||
return tmp
|
||||
|
||||
def find_installed_skills():
|
||||
# type: () -> list
|
||||
"""Find all installed skill directories (first-level subdirectories containing SKILL.md).
|
||||
Excludes SkillScan itself."""
|
||||
found = set()
|
||||
for base in skill_install_paths():
|
||||
if not base.exists():
|
||||
continue
|
||||
for md in base.rglob("SKILL.md"):
|
||||
skill_path = md.parent
|
||||
try:
|
||||
rel = skill_path.relative_to(base)
|
||||
if len(rel.parts) == 1:
|
||||
resolved = skill_path.resolve()
|
||||
# Skip self
|
||||
if resolved == SELF_ROOT:
|
||||
continue
|
||||
found.add(resolved)
|
||||
except ValueError:
|
||||
pass
|
||||
return sorted(found)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Scan Core (3 steps)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _extract_result(resp, sha256):
|
||||
"""Internal: extract core data from API response, handling SHA256 wrapping/nested result."""
|
||||
# 1. Handle API response keyed by SHA256 (e.g. { "sha256": { "status": "success", "data": {...} } })
|
||||
if sha256 and sha256 in resp:
|
||||
resp = resp[sha256]
|
||||
|
||||
# 2. Extract data body (data or result)
|
||||
data = resp.get("data") or resp.get("result") or resp
|
||||
|
||||
# 3. Handle nested result inside data
|
||||
if isinstance(data, dict) and "result" in data:
|
||||
inner = data["result"]
|
||||
if isinstance(inner, dict):
|
||||
# Merge sibling metadata (analysis_level/reason etc.) into result
|
||||
for k, v in data.items():
|
||||
if k != "result" and k not in inner:
|
||||
inner[k] = v
|
||||
return inner
|
||||
|
||||
return data if isinstance(data, dict) and (data.get("verdict") or data.get("is_safe") is not None or data.get("analysis_level")) else None
|
||||
|
||||
|
||||
def cloud_search(dir_sha256):
|
||||
"""Step 1: Query scan cache by dir_sha256. Returns result dict or None."""
|
||||
extra_headers = {}
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
extra_headers["X-Client-Info"] = ci
|
||||
|
||||
url = "%s?%s" % (API_SEARCH, urllib.parse.urlencode({"dir_sha256": dir_sha256}))
|
||||
try:
|
||||
headers = {}
|
||||
headers.update(extra_headers)
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
res = _extract_result(resp, dir_sha256)
|
||||
if res:
|
||||
log(" ✅ Cache hit (dir_sha256 %s…)" % dir_sha256[:16])
|
||||
return res
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return None
|
||||
raise RuntimeError("Search API error HTTP %d" % e.code)
|
||||
except urllib.error.URLError as e:
|
||||
raise RuntimeError("Cannot connect to server: %s" % e)
|
||||
return None
|
||||
|
||||
|
||||
def cloud_upload(skill_dir, name, dir_hash):
|
||||
"""Step 2: Upload skill (multipart/form-data), returns task_no."""
|
||||
# Pack the entire directory for full code context
|
||||
zip_data = pack_zip(skill_dir)
|
||||
filename = "%s.zip" % name
|
||||
|
||||
# Build multipart/form-data boundary
|
||||
boundary = "----WebKitFormBoundary%s" % uuid.uuid4().hex
|
||||
|
||||
# Manually construct multipart byte stream (no requests library needed)
|
||||
parts = []
|
||||
parts.append(("--%s" % boundary).encode())
|
||||
parts.append(('Content-Disposition: form-data; name="file"; filename="%s"' % filename).encode())
|
||||
parts.append(b"Content-Type: application/zip")
|
||||
parts.append(b"")
|
||||
parts.append(zip_data)
|
||||
parts.append(("--%s--" % boundary).encode())
|
||||
parts.append(b"") # trailing newline
|
||||
|
||||
body = b"\r\n".join(parts)
|
||||
|
||||
headers = {
|
||||
"Content-Type": "multipart/form-data; boundary=%s" % boundary,
|
||||
"Content-Length": str(len(body)),
|
||||
"Accept": "application/json"
|
||||
}
|
||||
|
||||
# Add X-Client-Info header
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
headers["X-Client-Info"] = ci
|
||||
|
||||
log(" 📤 Uploading: %s (%.1f KB)..." % (filename, len(zip_data) / 1024.0))
|
||||
req = urllib.request.Request(API_UPLOAD, data=body, headers=headers, method="POST")
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
except urllib.error.HTTPError as e:
|
||||
err_body = e.read().decode(errors="replace")
|
||||
raise RuntimeError("Upload failed HTTP %d: %s" % (e.code, err_body))
|
||||
|
||||
task_no = (resp.get("data") or {}).get("task_no") or resp.get("task_no") or resp.get("taskNo") or resp.get("task_id") or ""
|
||||
if not task_no:
|
||||
raise RuntimeError("Upload succeeded but no valid task_no in response: %s" % resp)
|
||||
|
||||
log(" ✅ Upload complete, task_no: %s" % task_no)
|
||||
return str(task_no)
|
||||
|
||||
|
||||
def cloud_poll(task_no: str) -> dict:
|
||||
"""Step 3: Poll until complete or timeout. Queries every 20s.
|
||||
status: 0=pending, 1=scanning, 2=completed, 3=failed, 4=cancelled
|
||||
"""
|
||||
url = f"{API_RESULT}?{urllib.parse.urlencode({'task_no': task_no})}"
|
||||
deadline = time.time() + POLL_TIMEOUT
|
||||
attempt = 0
|
||||
while time.time() < deadline:
|
||||
attempt += 1
|
||||
elapsed = int(time.time() - (deadline - POLL_TIMEOUT))
|
||||
try:
|
||||
resp = http_get(url)
|
||||
data = resp.get("data") or resp
|
||||
status = data.get("status")
|
||||
|
||||
if status == 2: # completed
|
||||
print()
|
||||
log(f" ✅ Scan complete (attempt {attempt}, {elapsed}s elapsed)")
|
||||
return _extract_result(resp, "") or resp
|
||||
elif status == 3: # failed
|
||||
print()
|
||||
err_msg = data.get("error_message") or resp.get("message", "unknown error")
|
||||
raise RuntimeError(f"Analysis failed: {err_msg}")
|
||||
elif status == 4: # cancelled
|
||||
print()
|
||||
raise RuntimeError("Scan task was cancelled")
|
||||
else:
|
||||
# 0=pending, 1=scanning -> keep waiting
|
||||
status_text = data.get("status_text", "processing")
|
||||
print(f" ⏳ [{status_text}] attempt {attempt}, {elapsed}s / {POLL_TIMEOUT}s elapsed",
|
||||
end="\r", flush=True)
|
||||
time.sleep(POLL_INTERVAL)
|
||||
except (RuntimeError, ValueError):
|
||||
raise
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Poll error: {e}")
|
||||
print()
|
||||
raise RuntimeError(f"Timeout ({POLL_TIMEOUT}s), task_no={task_no}, please retry later")
|
||||
|
||||
|
||||
def cloud_check(skill_dir: Path) -> dict:
|
||||
"""Run full security scan on a skill directory, return normalized result."""
|
||||
md = skill_dir / "SKILL.md"
|
||||
if not md.exists():
|
||||
raise FileNotFoundError(f"SKILL.md not found: {skill_dir}")
|
||||
|
||||
name = skill_name_from_dir(skill_dir)
|
||||
dir_hash = calculate_dir_sha256(skill_dir)
|
||||
log(f"🔍 Scanning: {name}")
|
||||
log(f" dir_sha256: {dir_hash}")
|
||||
|
||||
log(f"🔎 [1/3] Checking scan cache...")
|
||||
raw = cloud_search(dir_hash)
|
||||
|
||||
if raw is None:
|
||||
log(f" ℹ️ No cache record, submitting new scan task")
|
||||
log(f"📤 [2/3] Uploading skill for analysis...")
|
||||
task_no = cloud_upload(skill_dir, name, dir_hash)
|
||||
log(f"⏳ [3/3] Waiting for analysis (polling every {POLL_INTERVAL}s, max {POLL_TIMEOUT}s)...")
|
||||
raw = cloud_poll(task_no)
|
||||
else:
|
||||
log(f" ⏭️ Skipping upload, using cached result")
|
||||
|
||||
return _normalize(raw, name, dir_hash)
|
||||
|
||||
|
||||
def _normalize(raw: dict, name: str, dir_hash: str) -> dict:
|
||||
"""Normalize scan result:
|
||||
1. Extract is_safe (bool) and max_severity (str).
|
||||
2. Map API-specific fields (analysis_reason, analysis_suggestion) to standard fields.
|
||||
"""
|
||||
is_safe = raw.get("is_safe")
|
||||
|
||||
# Severity field priority: max_severity > analysis_level > verdict > level
|
||||
v_raw = (raw.get("max_severity") or raw.get("analysis_level") or
|
||||
raw.get("verdict") or raw.get("risk_level") or
|
||||
raw.get("level") or "UNKNOWN").upper()
|
||||
|
||||
# Combined verdict logic
|
||||
if is_safe is True and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "SAFE"
|
||||
elif is_safe is False and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "CRITICAL" # Explicitly marked unsafe -> critical
|
||||
else:
|
||||
verdict = v_raw
|
||||
|
||||
return {
|
||||
"skill_name": name,
|
||||
"dir_sha256": dir_hash,
|
||||
"verdict": verdict,
|
||||
"confidence": raw.get("confidence") or raw.get("score"),
|
||||
"threat_labels": raw.get("threat_labels") or raw.get("tags") or [],
|
||||
"summary": raw.get("analysis_reason") or raw.get("summary") or raw.get("description") or "",
|
||||
"findings": raw.get("findings") or raw.get("issues") or [],
|
||||
"recommendation":raw.get("analysis_suggestion") or raw.get("recommendation") or raw.get("action") or "",
|
||||
}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Result Display
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def print_result(r: dict):
|
||||
verdict = r.get("verdict","UNKNOWN")
|
||||
emoji = RISK_EMOJI.get(verdict,"❓")
|
||||
conf = r.get("confidence")
|
||||
labels = r.get("threat_labels",[])
|
||||
summary = r.get("summary","")
|
||||
findings = r.get("findings",[])
|
||||
rec = r.get("recommendation","")
|
||||
conf_str = f" confidence {float(conf):.0%}" if conf is not None else ""
|
||||
|
||||
divider()
|
||||
log(f"{emoji} Result: {verdict}{conf_str}")
|
||||
if summary:
|
||||
log(f"📋 {summary}")
|
||||
if labels:
|
||||
log(f"🏷️ Threat labels: {', '.join(labels)}")
|
||||
if findings:
|
||||
SEV = {"LOW":"🔵","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️"}
|
||||
log(f"🔍 Findings ({len(findings)} items):")
|
||||
for f in findings:
|
||||
sev = str(f.get("severity","")).upper()
|
||||
desc = f.get("description") or f.get("detail") or str(f)
|
||||
rid = f.get("id") or ""
|
||||
tag = f"[{rid}] " if rid else ""
|
||||
log(f" {SEV.get(sev,'⚪')} {tag}{desc}")
|
||||
if rec:
|
||||
log(f"💡 Recommendation: {rec}")
|
||||
divider()
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Prompt: malicious detected -> ask whether to delete
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def prompt_delete(skill_path: Path, result: dict) -> bool:
|
||||
"""When result is HIGH/CRITICAL, ask user whether to delete the skill.
|
||||
skill_path is the original install path (not temp dir).
|
||||
Returns True if deleted.
|
||||
"""
|
||||
verdict = result.get("verdict","")
|
||||
if verdict not in ("HIGH","CRITICAL"):
|
||||
return False
|
||||
|
||||
if not skill_path or not skill_path.exists():
|
||||
return False
|
||||
|
||||
emoji = RISK_EMOJI.get(verdict,"🔴")
|
||||
log(f"\n{emoji} This skill is marked as [{verdict}] high risk by security scan.")
|
||||
log(f" Path: {skill_path}")
|
||||
|
||||
answer = ask("Delete this skill now? [y/n]")
|
||||
if answer in ("y","Y","yes","Yes"):
|
||||
try:
|
||||
if skill_path.is_dir():
|
||||
shutil.rmtree(skill_path)
|
||||
else:
|
||||
skill_path.unlink()
|
||||
log(f"✅ Deleted: {skill_path}")
|
||||
return True
|
||||
except Exception as e:
|
||||
log(f"❌ Delete failed: {e} (please delete manually)")
|
||||
return False
|
||||
else:
|
||||
log(f"⚠️ Skipped deletion. Use this skill with caution.")
|
||||
return False
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: first-run (first install)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_first_run():
|
||||
"""First install: list installed skills, ask user to scan, show results."""
|
||||
if STATE_FILE.exists():
|
||||
log("ℹ️ First-run scan already completed. Use scan-all to rescan.")
|
||||
return
|
||||
|
||||
banner("🛡️ SkillScan First-Run Check")
|
||||
log("Welcome to SkillScan!")
|
||||
log("Searching for installed skills...\n")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("✅ No installed skills found, nothing to scan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Print installed skill list
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name}")
|
||||
|
||||
answer = ask("Run security scan on all listed skills? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Skipped. You can run scan-all anytime to rescan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Scan one by one
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
# Copy to temp dir (source may be read-only)
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
print_result(r)
|
||||
|
||||
# High risk -> ask to delete (targeting original install path)
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name,
|
||||
"verdict": "ERROR", "threat_labels": [],
|
||||
"summary": str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan (single skill)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan(path_str: str):
|
||||
skill_path = Path(path_str)
|
||||
if not skill_path.exists():
|
||||
log(f"❌ Path not found: {skill_path}")
|
||||
sys.exit(1)
|
||||
|
||||
banner(f"Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
tmp = None
|
||||
original_path = skill_path if skill_path.is_dir() else None
|
||||
try:
|
||||
if skill_path.is_file():
|
||||
if skill_path.suffix.lower() not in (".zip",):
|
||||
log(f"❌ Unsupported format: {skill_path.suffix} (use .zip)")
|
||||
sys.exit(1)
|
||||
tmp = unpack_zip(skill_path)
|
||||
scan_dir = tmp
|
||||
else:
|
||||
scan_dir = skill_path
|
||||
|
||||
result = cloud_check(scan_dir)
|
||||
print_result(result)
|
||||
|
||||
# High risk -> ask to delete
|
||||
if original_path:
|
||||
prompt_delete(original_path, result)
|
||||
elif skill_path.is_file() and result.get("verdict") in ("HIGH","CRITICAL"):
|
||||
# Zip file: ask to delete source file
|
||||
prompt_delete(skill_path, result)
|
||||
|
||||
v = result.get("verdict","UNKNOWN")
|
||||
sys.exit(0 if v in ("SAFE","LOW") else 1 if v=="MEDIUM" else 2)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"\n❌ Scan failed: {e}")
|
||||
sys.exit(3)
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan-all
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan_all():
|
||||
banner(f"Full Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("ℹ️ No installed skills detected.")
|
||||
return
|
||||
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name:<30} {s}")
|
||||
|
||||
answer = ask("Start security scan? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Cancelled.")
|
||||
return
|
||||
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
v = r.get("verdict","UNKNOWN")
|
||||
log(f"{RISK_EMOJI.get(v,'❓')} Scan complete: {v}")
|
||||
if r.get("threat_labels"):
|
||||
log(f" Threat labels: {', '.join(r['threat_labels'])}")
|
||||
|
||||
# High risk: ask to delete
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name, "verdict":"ERROR",
|
||||
"threat_labels":[], "summary":str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Summary Table
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _print_summary(results):
|
||||
banner("📊 Scan Summary")
|
||||
print(f" {'Skill Name':<28} {'Result':<12} {'Threat Labels'}")
|
||||
divider()
|
||||
for r in results:
|
||||
v = r.get("verdict","?")
|
||||
name = r.get("skill_name","?")[:27]
|
||||
labels = ", ".join(r.get("threat_labels",[]))[:20] or "-"
|
||||
print(f" {name:<28} {RISK_EMOJI.get(v,'❓')}{v:<10} {labels}")
|
||||
|
||||
safes = [r for r in results if r["verdict"] in {"SAFE","LOW"}]
|
||||
mediums = [r for r in results if r["verdict"] == "MEDIUM"]
|
||||
highs = [r for r in results if r["verdict"] in {"HIGH","CRITICAL"}]
|
||||
errors = [r for r in results if r["verdict"] in {"ERROR","UNKNOWN"}]
|
||||
|
||||
print()
|
||||
log(f"Total {len(results)} | ✅ Safe {len(safes)} "
|
||||
f"🟡 Suspicious {len(mediums)} 🔴 Dangerous {len(highs)} ❓ Error {len(errors)}")
|
||||
if highs:
|
||||
log(f"\n⚠️ High-risk skills: {', '.join(r['skill_name'] for r in highs)}")
|
||||
elif not mediums and not errors:
|
||||
log("\n🎉 All skills passed security scan.")
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: upgrade
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_upgrade():
|
||||
banner("SkillScan Auto-Upgrade")
|
||||
log(f"Current version: {SCANNER_VERSION}")
|
||||
log(f"Update source: {UPDATE_URL}")
|
||||
try:
|
||||
manifest = http_get(UPDATE_URL)
|
||||
except Exception as e:
|
||||
log(f"❌ Failed to fetch update manifest: {e}")
|
||||
return
|
||||
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f"✅ Already up to date ({SCANNER_VERSION})")
|
||||
return
|
||||
|
||||
log(f"New version found: {SCANNER_VERSION} → {latest}")
|
||||
log(f"Changelog: {manifest.get('changelog','(none)')}")
|
||||
|
||||
download_url = manifest.get("download_url", "")
|
||||
if not download_url:
|
||||
log("⚠️ No download URL in manifest, skipping upgrade")
|
||||
return
|
||||
|
||||
# Download new version zip
|
||||
log(f"📥 Downloading: {download_url}")
|
||||
try:
|
||||
req = urllib.request.Request(download_url)
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
zip_data = r.read()
|
||||
except Exception as e:
|
||||
log(f"❌ Download failed: {e}")
|
||||
return
|
||||
|
||||
# SHA256 verification
|
||||
expected_sha = manifest.get("sha256", "")
|
||||
if expected_sha:
|
||||
actual_sha = hashlib.sha256(zip_data).hexdigest()
|
||||
if actual_sha != expected_sha:
|
||||
log(f"❌ SHA256 mismatch, upgrade aborted (expected {expected_sha[:16]}…, got {actual_sha[:16]}…)")
|
||||
return
|
||||
log(f" ✅ SHA256 verified")
|
||||
|
||||
# Backup current skill directory
|
||||
skill_root = Path(__file__).parent.parent
|
||||
backup_dir = skill_root.parent / f"SkillScan-backup-{SCANNER_VERSION}"
|
||||
if backup_dir.exists():
|
||||
shutil.rmtree(backup_dir)
|
||||
shutil.copytree(skill_root, backup_dir)
|
||||
log(f"📦 Backed up to: {backup_dir}")
|
||||
|
||||
# Extract and replace files
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillupgrade-"))
|
||||
try:
|
||||
zip_path = tmp / "update.zip"
|
||||
zip_path.write_bytes(zip_data)
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
# Security check: prevent zip-slip
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / "extracted" / member).resolve()
|
||||
if not str(dest).startswith(str((tmp / "extracted").resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp / "extracted")
|
||||
|
||||
# Overwrite skill directory with new files
|
||||
extracted = tmp / "extracted"
|
||||
for item in extracted.rglob("*"):
|
||||
if not item.is_file():
|
||||
continue
|
||||
rel = item.relative_to(extracted)
|
||||
target = skill_root / rel
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(item, target)
|
||||
log(f" ✅ Updated: {rel}")
|
||||
|
||||
log(f"🎉 Upgraded to v{latest}")
|
||||
except Exception as e:
|
||||
log(f"❌ Upgrade failed: {e}")
|
||||
log(f" You can restore from backup: {backup_dir}")
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Entry Point
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def auto_upgrade_if_needed():
|
||||
"""Auto-check for updates every 7 days, runs silently."""
|
||||
try:
|
||||
if LAST_UPDATE_CHECK_FILE.exists():
|
||||
last_check = float(LAST_UPDATE_CHECK_FILE.read_text(encoding="utf-8").strip())
|
||||
if time.time() - last_check < AUTO_UPDATE_INTERVAL:
|
||||
return # Not time to check yet
|
||||
log("🔄 Checking for updates...")
|
||||
manifest = http_get(UPDATE_URL)
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f" ✅ Already up to date ({SCANNER_VERSION})")
|
||||
else:
|
||||
log(f" New version found: {SCANNER_VERSION} → {latest}, auto-updating...")
|
||||
cmd_upgrade()
|
||||
LAST_UPDATE_CHECK_FILE.write_text(str(time.time()), encoding="utf-8")
|
||||
except Exception as e:
|
||||
log(f" ⚠️ Auto-update check failed: {e} (normal operation unaffected)")
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print(__doc__)
|
||||
sys.exit(0)
|
||||
|
||||
# Check for auto-update on every run (once every 7 days)
|
||||
auto_upgrade_if_needed()
|
||||
|
||||
cmd = sys.argv[1]
|
||||
if cmd == "first-run":
|
||||
cmd_first_run()
|
||||
elif cmd == "scan":
|
||||
if len(sys.argv) < 3:
|
||||
log("Usage: scanner.py scan <skill_path>")
|
||||
sys.exit(1)
|
||||
cmd_scan(sys.argv[2])
|
||||
elif cmd == "scan-all":
|
||||
cmd_scan_all()
|
||||
elif cmd == "upgrade":
|
||||
cmd_upgrade()
|
||||
else:
|
||||
log(f"Unknown command: {cmd}")
|
||||
log("Available commands: first-run / scan <path> / scan-all / upgrade")
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -11,7 +11,7 @@
|
|||
- `tavily-search-pro`:搜索与深度研究覆盖面广,能力强。
|
||||
- `weather`:基础可用,但能力边界较窄。
|
||||
- `self-improvement`:框架不错,但过去依赖手动触发。
|
||||
- `cocoloop`:管理类能力实用,但使用场景相对集中。
|
||||
- `oclaw-skill-manager`:技能安装与运维说明,使用场景相对集中。
|
||||
|
||||
核心缺口:
|
||||
|
||||
|
|
|
|||
208
runtime/skills/word-reader/DEVELOPMENT.md
Normal file
208
runtime/skills/word-reader/DEVELOPMENT.md
Normal file
|
|
@ -0,0 +1,208 @@
|
|||
# Word Reader 技能开发完成
|
||||
|
||||
## 🎯 技能概述
|
||||
|
||||
成功创建了一个功能完整的 Word 文档读取技能,支持读取 .docx 和 .doc 格式的 Word 文档,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## 📁 技能结构
|
||||
|
||||
```
|
||||
word-reader/
|
||||
├── SKILL.md # 技能定义文件
|
||||
├── README.md # 使用说明
|
||||
├── skill.json # 技能配置
|
||||
├── demo.sh # 演示脚本
|
||||
├── install.sh # 安装脚本
|
||||
├── test.md # 测试文档
|
||||
└── scripts/
|
||||
└── read_word.py # 核心脚本
|
||||
```
|
||||
|
||||
## ✨ 主要功能
|
||||
|
||||
### 1. 文档解析能力
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
|
||||
### 2. 格式支持
|
||||
- ✅ **.docx** - Office 2007+ 格式(主要支持)
|
||||
- ✅ **.doc** - 旧版 Word 格式(需要 antiword)
|
||||
|
||||
### 3. 输出格式
|
||||
- ✅ **JSON** - 结构化数据,适合程序处理
|
||||
- ✅ **Text** - 纯文本格式,简单易读
|
||||
- ✅ **Markdown** - 格式化输出,保留文档结构
|
||||
|
||||
### 4. 高级功能
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **选择性提取** - 可只提取特定内容类型
|
||||
- ✅ **文件输出** - 支持保存结果到文件
|
||||
- ✅ **编码支持** - 支持多种文本编码
|
||||
|
||||
## 🚀 使用示例
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# JSON 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# Markdown 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存结果
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 🔧 安装和配置
|
||||
|
||||
### 自动安装
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx
|
||||
|
||||
# 安装系统依赖(可选)
|
||||
sudo apt-get install antiword # Ubuntu/Debian
|
||||
brew install antiword # macOS
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "文档.docx",
|
||||
"title": "文档标题",
|
||||
"author": "作者",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "文档内容...",
|
||||
"tables": [...],
|
||||
"images": [...]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
```markdown
|
||||
# 文档.docx
|
||||
|
||||
**标题**:文档标题
|
||||
**作者**:作者
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
文档内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 数据1 | 数据2 |
|
||||
```
|
||||
|
||||
## 🎨 技能特点
|
||||
|
||||
### 1. 智能错误处理
|
||||
- 友好的错误提示
|
||||
- 自动检测文档格式
|
||||
- 优雅的异常处理
|
||||
|
||||
### 2. 性能优化
|
||||
- 流式处理大文件
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
|
||||
### 3. 用户友好
|
||||
- 详细的帮助信息
|
||||
- 多种使用方式
|
||||
- 完整的文档说明
|
||||
|
||||
### 4. 可扩展性
|
||||
- 模块化设计
|
||||
- 易于添加新功能
|
||||
- 支持自定义输出格式
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
### 1. 文档内容分析
|
||||
- 快速查看 Word 文档内容
|
||||
- 提取特定信息
|
||||
- 文档摘要生成
|
||||
|
||||
### 2. 批量处理
|
||||
- 处理大量文档
|
||||
- 文档格式转换
|
||||
- 内容索引创建
|
||||
|
||||
### 3. 自动化工作流
|
||||
- 集到文档处理系统
|
||||
- 自动化文档分析
|
||||
- 内容管理系统集成
|
||||
|
||||
## 📝 开发总结
|
||||
|
||||
### 实现的功能
|
||||
- 完整的 Word 文档解析框架
|
||||
- 支持多种输出格式
|
||||
- 批量处理能力
|
||||
- 错误处理和用户友好性
|
||||
|
||||
### 技术亮点
|
||||
- 模块化设计,易于维护
|
||||
- 优雅的错误处理机制
|
||||
- 支持多种文件格式
|
||||
- 灵活的输出选项
|
||||
|
||||
### 改进空间
|
||||
- 可以添加 PDF 支持
|
||||
- 可以增加图片提取功能
|
||||
- 可以优化大文件处理性能
|
||||
- 可以添加更多文档元素支持
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
要发布此技能到 ClawHub,可以运行:
|
||||
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm i -g clawhub
|
||||
|
||||
# 登录
|
||||
clawhub login
|
||||
|
||||
# 发布技能
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "Initial release with .docx and .doc support" \
|
||||
--tags document,word,office,text-extraction
|
||||
```
|
||||
|
||||
这个技能现在已经准备好使用了!它可以帮助用户轻松读取和处理 Word 文档,支持多种格式和输出选项。
|
||||
177
runtime/skills/word-reader/PUBLISHING.md
Normal file
177
runtime/skills/word-reader/PUBLISHING.md
Normal file
|
|
@ -0,0 +1,177 @@
|
|||
# Word Reader 技能发布指南
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
### 1. 准备工作
|
||||
|
||||
#### 确保技能完整
|
||||
- [ ] SKILL.md 文件完整且格式正确
|
||||
- [ ] 脚本功能正常
|
||||
- [ ] 安装脚本工作正常
|
||||
- [ ] README.md 说明清晰
|
||||
- [ ] 所有依赖已在 SKILL.md 中声明
|
||||
|
||||
#### 环境准备
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm install -g clawhub
|
||||
# 或
|
||||
pnpm add -g clawhub
|
||||
```
|
||||
|
||||
#### 登录 ClawHub
|
||||
```bash
|
||||
# 登录(会打开浏览器进行 OAuth 认证)
|
||||
clawhub login
|
||||
|
||||
# 验证登录状态
|
||||
clawhub whoami
|
||||
```
|
||||
|
||||
> **注意**:GitHub 账号需要注册满一周才能发布技能
|
||||
|
||||
### 2. 发布流程
|
||||
|
||||
#### 检查技能
|
||||
```bash
|
||||
# 验证技能结构
|
||||
clawhub validate ./word-reader
|
||||
```
|
||||
|
||||
#### 发布技能
|
||||
```bash
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "支持 .docx 和 .doc 格式的 Word 文档读取,提取文本、表格、元数据等" \
|
||||
--tags document,word,office,text-extraction,reader,parsing \
|
||||
--license MIT \
|
||||
--visibility public
|
||||
```
|
||||
|
||||
#### 参数说明
|
||||
- `--slug`: URL 友好的唯一标识符
|
||||
- `--name`: 技能显示名称
|
||||
- `--version`: 遵循语义化版本控制
|
||||
- `--changelog`: 版本变更说明
|
||||
- `--tags`: 搜索标签(逗号分隔)
|
||||
- `--license`: 许可证类型
|
||||
- `--visibility`: public/private
|
||||
|
||||
### 3. 发布后操作
|
||||
|
||||
#### 验证发布
|
||||
```bash
|
||||
# 查看已发布的技能
|
||||
clawhub search word-reader
|
||||
|
||||
# 安装测试
|
||||
clawhub install word-reader-test
|
||||
```
|
||||
|
||||
#### 分享技能
|
||||
- 技能将在 `https://clawhub.com/skills/word-reader` 可见
|
||||
- 其他用户可通过 `clawhub install word-reader` 安装
|
||||
|
||||
### 4. 版本管理
|
||||
|
||||
#### 更新技能
|
||||
```bash
|
||||
# 修改技能后更新版本号
|
||||
clawhub publish ./word-reader --version 1.0.1 --changelog "修复了某些文档格式的解析问题"
|
||||
```
|
||||
|
||||
#### 批量操作
|
||||
```bash
|
||||
# 同步所有技能
|
||||
clawhub sync --all
|
||||
|
||||
# 发布并标记
|
||||
clawhub publish ./word-reader --tags latest,stable
|
||||
```
|
||||
|
||||
### 5. 自动化发布
|
||||
|
||||
#### GitHub Actions 示例
|
||||
```yaml
|
||||
name: Publish Skill
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v3
|
||||
with:
|
||||
node-version: '18'
|
||||
|
||||
- name: Install ClawHub CLI
|
||||
run: npm install -g clawhub
|
||||
|
||||
- name: Login to ClawHub
|
||||
run: echo "${{ secrets.CLAWHUB_TOKEN }}" | clawhub login --token
|
||||
|
||||
- name: Publish Skill
|
||||
run: |
|
||||
clawhub publish ./skills/word-reader \
|
||||
--slug word-reader \
|
||||
--version ${{ github.ref_name }} \
|
||||
--changelog "Published from GitHub Actions"
|
||||
```
|
||||
|
||||
### 6. 发布注意事项
|
||||
|
||||
#### 必须遵守的规则
|
||||
- [ ] 技能名称不能与其他技能冲突
|
||||
- [ ] 版本号遵循 SemVer 规范
|
||||
- [ ] changelog 清晰描述变更
|
||||
- [ ] 代码无安全漏洞
|
||||
- [ ] 许可证声明清晰
|
||||
|
||||
#### 最佳实践
|
||||
- [ ] 发布前充分测试
|
||||
- [ ] 提供清晰的使用示例
|
||||
- [ ] 维护更新日志
|
||||
- [ ] 及时修复问题
|
||||
- [ ] 关注用户反馈
|
||||
|
||||
### 7. 故障排除
|
||||
|
||||
#### 常见问题
|
||||
```bash
|
||||
# 验证发布权限
|
||||
clawhub whoami
|
||||
|
||||
# 检查技能格式
|
||||
clawhub validate ./word-reader
|
||||
|
||||
# 查看详细错误信息
|
||||
clawhub publish ./word-reader --verbose
|
||||
```
|
||||
|
||||
#### 重新发布
|
||||
如果发布失败,可以:
|
||||
1. 修正问题
|
||||
2. 增加版本号
|
||||
3. 重新发布
|
||||
|
||||
### 8. 维护指南
|
||||
|
||||
#### 监控使用情况
|
||||
- 定期查看下载统计
|
||||
- 关注用户反馈
|
||||
- 及时修复问题
|
||||
|
||||
#### 更新策略
|
||||
- 重要修复:紧急发布补丁版本
|
||||
- 新功能:发布次版本号
|
||||
- 重大变更:发布主版本号
|
||||
|
||||
现在你的 Word Reader 技能已经准备好发布到 ClawHub 了!
|
||||
171
runtime/skills/word-reader/README.md
Normal file
171
runtime/skills/word-reader/README.md
Normal file
|
|
@ -0,0 +1,171 @@
|
|||
# Word Reader 技能
|
||||
|
||||
## 📋 概述
|
||||
|
||||
Word Reader 是一个强大的 Word 文档读取工具,支持 .docx 和 .doc 格式,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## ✨ 功能特性
|
||||
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
- ✅ **多格式支持** - 支持 .docx 和 .doc 格式
|
||||
- ✅ **多种输出** - JSON、Text、Markdown 格式
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **自动安装** - 一键安装所有依赖
|
||||
|
||||
## 🚀 安装
|
||||
|
||||
### 自动安装(推荐)
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx --break-system-packages
|
||||
|
||||
# 安装系统依赖(可选,用于 .doc 格式支持)
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📖 使用方法
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档并输出为文本格式
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# 输出为 JSON 格式
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# 输出为 Markdown 格式
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本内容
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有 Word 文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存为 JSON 文件
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
### 高级用法
|
||||
```bash
|
||||
# 将结果保存到文件
|
||||
python3 scripts/read_word.py 文档.docx --format markdown --output output.md
|
||||
|
||||
# 提取表格数据
|
||||
python3 scripts/read_word.py 文档.docx --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 scripts/read_word.py 文档.docx --extract metadata
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式输出
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "测试文档.docx",
|
||||
"size": "2048 bytes",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00",
|
||||
"title": "测试文档",
|
||||
"author": "测试用户"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "这是文档的正文内容...",
|
||||
"tables": [
|
||||
{
|
||||
"id": 1,
|
||||
"rows": 3,
|
||||
"columns": 3,
|
||||
"data": [
|
||||
["表头1", "表头2", "表头3"],
|
||||
["数据1", "数据2", "数据3"],
|
||||
["数据4", "数据5", "数据6"]
|
||||
]
|
||||
}
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"id": "rId1",
|
||||
"filename": "image1.png",
|
||||
"size": "1024 bytes"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式输出
|
||||
```markdown
|
||||
# 测试文档.docx
|
||||
|
||||
**标题**:测试文档
|
||||
**作者**:测试用户
|
||||
**文件大小**:2048 bytes
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
**修改时间**:2024-01-01T12:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
### 表格 1 (3行 x 3列)
|
||||
|
||||
| 表头1 | 表头2 | 表头3 |
|
||||
|-------|-------|-------|
|
||||
| 数据1 | 数据2 | 数据3 |
|
||||
| 数据4 | 数据5 | 数据6 |
|
||||
```
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
- **文档内容分析** - 快速查看 Word 文档内容
|
||||
- **批量处理** - 处理大量文档
|
||||
- **内容提取** - 提取特定信息
|
||||
- **格式转换** - 转换为其他格式
|
||||
- **自动化工作流** - 集成到文档处理系统
|
||||
|
||||
## 📤 发布到 ClawHub
|
||||
|
||||
要将此技能发布到 ClawHub,请参考 `PUBLISHING.md` 文件。
|
||||
|
||||
## 🔧 故障排除
|
||||
|
||||
### 常见问题
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **FileNotFoundError**: 确认文件路径正确
|
||||
4. **编码问题**: 尝试使用 `--encoding gb2312` 参数
|
||||
|
||||
### 性能优化
|
||||
- 大文档处理时建议使用 `--format json` 以获得更好的性能
|
||||
- 批量模式下建议使用 `--output` 参数将结果保存到文件
|
||||
|
||||
## 🤝 贡献
|
||||
|
||||
欢迎提交 Issue 和 Pull Request 来改进这个技能!
|
||||
|
||||
## 📄 许可证
|
||||
|
||||
MIT License
|
||||
225
runtime/skills/word-reader/SKILL.md
Normal file
225
runtime/skills/word-reader/SKILL.md
Normal file
|
|
@ -0,0 +1,225 @@
|
|||
---
|
||||
name: word-reader
|
||||
description: |
|
||||
读取 Word 文档(.docx 和 .doc 格式)并提取文本内容。支持文档解析、表格提取、图片处理等功能。使用当用户需要分析 Word 文档内容、提取文本信息或批量处理文档时。
|
||||
homepage: https://python-docx.readthedocs.io/
|
||||
metadata:
|
||||
{
|
||||
"openclaw":
|
||||
{
|
||||
"emoji": "📄",
|
||||
"requires": { "bins": ["python3"], "env": ["PYTHONPATH"] },
|
||||
"install":
|
||||
[
|
||||
{
|
||||
"id": "pip",
|
||||
"kind": "pip",
|
||||
"package": "python-docx",
|
||||
"bins": ["python3"],
|
||||
"label": "Install python-docx (pip)",
|
||||
},
|
||||
{
|
||||
"id": "system",
|
||||
"kind": "system",
|
||||
"command": "sudo apt-get install antiword -y",
|
||||
"label": "Install antiword for .doc support (optional)",
|
||||
"platform": "linux-debian"
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
---
|
||||
|
||||
# Word 文档读取器
|
||||
|
||||
使用 Python 解析 Word 文档,提取文本内容和结构化信息。
|
||||
|
||||
## 支持的功能
|
||||
|
||||
- **文档文本提取** - 提取段落、标题、页眉页脚内容
|
||||
- **表格解析** - 读取表格数据并转换为结构化格式
|
||||
- **图片处理** - 提取文档中的图片信息
|
||||
- **元数据获取** - 读取文档属性(作者、标题、创建时间等)
|
||||
- **批量处理** - 支持处理多个文档
|
||||
|
||||
## 用法
|
||||
|
||||
### 基本文本提取
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径>
|
||||
```
|
||||
|
||||
### 指定输出格式
|
||||
|
||||
```bash
|
||||
# JSON 输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format json
|
||||
|
||||
# 纯文本输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format text
|
||||
|
||||
# Markdown 格式
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format markdown
|
||||
```
|
||||
|
||||
### 提取特定内容
|
||||
|
||||
```bash
|
||||
# 只提取文本
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract text
|
||||
|
||||
# 提取表格数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract metadata
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
|
||||
```bash
|
||||
# 处理目录下所有 .docx 文件
|
||||
python3 {baseDir}/scripts/read_word.py <目录路径> --batch
|
||||
```
|
||||
|
||||
## 参数说明
|
||||
|
||||
| 参数 | 说明 | 默认值 |
|
||||
|------|------|--------|
|
||||
| `--format` | 输出格式(json/text/markdown) | text |
|
||||
| `--extract` | 提取内容类型(text/tables/images/metadata/all) | all |
|
||||
| `--batch` | 批量处理模式 | false |
|
||||
| `--output` | 输出文件路径 | stdout |
|
||||
| `--encoding` | 文本编码(utf-8/gb2312) | utf-8 |
|
||||
|
||||
## 输出格式
|
||||
|
||||
### JSON 格式
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"title": "文档标题",
|
||||
"author": "作者姓名",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"text": "文档全文内容...",
|
||||
"tables": [
|
||||
[
|
||||
["表头1", "表头2"],
|
||||
["行1列1", "行1列2"],
|
||||
["行2列1", "行2列2"]
|
||||
]
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"filename": "image1.png",
|
||||
"description": "图片描述",
|
||||
"size": "1024x768"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
|
||||
```markdown
|
||||
# 文档标题
|
||||
|
||||
**作者**:作者姓名
|
||||
**创建时间**:2024-01-01 10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
### 表格示例
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 行1列1 | 行1列2 |
|
||||
| 行2列1 | 行2列2 |
|
||||
|
||||

|
||||
|
||||
## 图片列表
|
||||
|
||||
1. **image1.png** (1024x768) - 图片描述
|
||||
```
|
||||
|
||||
## 错误处理
|
||||
|
||||
- 文件不存在:显示错误信息并退出
|
||||
- 格式不支持:提示支持的文件类型
|
||||
- 权限问题:提示文件访问权限
|
||||
- 编码问题:尝试自动检测编码
|
||||
|
||||
## 示例场景
|
||||
|
||||
### 1. 查看项目文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 项目需求.docx --format markdown
|
||||
```
|
||||
|
||||
### 2. 提取会议记录
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 会议记录.docx --extract text
|
||||
```
|
||||
|
||||
### 3. 批量处理文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 注意事项
|
||||
|
||||
- 支持 .docx 格式(Office 2007+)
|
||||
- .doc 格式需要额外依赖(如 antiword)
|
||||
- 大文档处理可能需要较长时间
|
||||
- 图片提取仅获取元数据,不包含实际图片数据
|
||||
- 表格格式可能需要手动调整
|
||||
|
||||
## 故障排除
|
||||
|
||||
### 常见问题
|
||||
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **UnicodeDecodeError**: 尝试不同的编码格式
|
||||
|
||||
### 安装依赖
|
||||
|
||||
```bash
|
||||
pip3 install python-docx
|
||||
```
|
||||
|
||||
对于 .doc 格式支持:
|
||||
```bash
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
```
|
||||
|
||||
## 高级功能
|
||||
|
||||
### 自定义样式处理
|
||||
|
||||
脚本会自动处理以下文档元素:
|
||||
- 标题级别(H1-H6)
|
||||
- 段落样式
|
||||
- 列表项目
|
||||
- 页眉页脚
|
||||
- 文档属性
|
||||
|
||||
### 性能优化
|
||||
|
||||
- 大文件流式处理
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
11
runtime/skills/word-reader/_meta.json
Normal file
11
runtime/skills/word-reader/_meta.json
Normal file
|
|
@ -0,0 +1,11 @@
|
|||
{
|
||||
"owner": "xtfnhcyjpgf",
|
||||
"slug": "word-reader",
|
||||
"displayName": "Word Reader",
|
||||
"latest": {
|
||||
"version": "1.0.0",
|
||||
"publishedAt": 1770700102926,
|
||||
"commit": "https://github.com/openclaw/skills/commit/91b71e101c57b69a4d4eb2678e1b79992eb7032f"
|
||||
},
|
||||
"history": []
|
||||
}
|
||||
89
runtime/skills/word-reader/demo.sh
Normal file
89
runtime/skills/word-reader/demo.sh
Normal file
|
|
@ -0,0 +1,89 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能演示脚本
|
||||
# 此脚本展示如何使用 word-reader 技能
|
||||
|
||||
echo "=== Word Reader 技能演示 ==="
|
||||
echo ""
|
||||
|
||||
# 检查脚本是否存在
|
||||
SCRIPT_PATH="/root/.openclaw/workspace/skills/word-reader/scripts/read_word.py"
|
||||
if [ ! -f "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本不存在"
|
||||
echo "请确保技能已正确安装"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# 检查脚本是否有执行权限
|
||||
if [ ! -x "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本没有执行权限"
|
||||
echo "正在添加执行权限..."
|
||||
chmod +x "$SCRIPT_PATH"
|
||||
fi
|
||||
|
||||
echo "✅ 脚本已就绪"
|
||||
echo ""
|
||||
|
||||
# 显示技能信息
|
||||
echo "📋 技能信息:"
|
||||
echo " 名称:word-reader"
|
||||
echo " 功能:读取 Word 文档(.docx 和 .doc 格式)"
|
||||
echo " 位置:$SCRIPT_PATH"
|
||||
echo ""
|
||||
|
||||
# 显示使用示例
|
||||
echo "📖 使用示例:"
|
||||
echo ""
|
||||
|
||||
echo "1. 显示帮助信息:"
|
||||
echo " python3 $SCRIPT_PATH --help"
|
||||
echo ""
|
||||
|
||||
echo "2. 读取文档(文本格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx"
|
||||
echo ""
|
||||
|
||||
echo "3. 读取文档(JSON 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format json"
|
||||
echo ""
|
||||
|
||||
echo "4. 读取文档(Markdown 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown"
|
||||
echo ""
|
||||
|
||||
echo "5. 只提取文本内容:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --extract text"
|
||||
echo ""
|
||||
|
||||
echo "6. 批量处理目录:"
|
||||
echo " python3 $SCRIPT_PATH ./文档目录 --batch"
|
||||
echo ""
|
||||
|
||||
echo "7. 保存结果到文件:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown --output output.md"
|
||||
echo ""
|
||||
|
||||
echo "🔧 安装依赖:"
|
||||
echo " pip3 install python-docx"
|
||||
echo " # 对于 .doc 格式支持:"
|
||||
echo " # Ubuntu: sudo apt-get install antiword"
|
||||
echo " # macOS: brew install antiword"
|
||||
echo ""
|
||||
|
||||
echo "📊 支持的功能:"
|
||||
echo " ✅ 文本提取"
|
||||
echo " ✅ 表格解析"
|
||||
echo " ✅ 元数据获取"
|
||||
echo " ✅ 图片信息"
|
||||
echo " ✅ 多格式支持"
|
||||
echo " ✅ 批量处理"
|
||||
echo ""
|
||||
|
||||
echo "💡 提示:"
|
||||
echo " - 支持 .docx 和 .doc 格式"
|
||||
echo " - 输出格式:JSON、Text、Markdown"
|
||||
echo " - 如遇错误,请检查依赖是否安装"
|
||||
echo ""
|
||||
|
||||
echo "演示完成!"
|
||||
echo "如需使用,请替换 '文档路径.docx' 为实际的文档路径"
|
||||
101
runtime/skills/word-reader/install.sh
Normal file
101
runtime/skills/word-reader/install.sh
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能安装脚本
|
||||
# 此脚本会自动安装依赖并设置技能
|
||||
|
||||
set -e
|
||||
|
||||
echo "=== Word Reader 技能安装 ==="
|
||||
echo ""
|
||||
|
||||
# 检查 Python 版本
|
||||
echo "🔍 检查 Python 版本..."
|
||||
python_version=$(python3 --version 2>&1)
|
||||
echo " Python 版本: $python_version"
|
||||
|
||||
if ! python3 -c "import sys; assert sys.version_info >= (3, 6)"; then
|
||||
echo "❌ 错误:需要 Python 3.6 或更高版本"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "✅ Python 版本检查通过"
|
||||
echo ""
|
||||
|
||||
# 检查并安装依赖
|
||||
echo "📦 检查依赖..."
|
||||
|
||||
# 检查 pip
|
||||
if ! command -v pip3 &> /dev/null; then
|
||||
echo " 🔧 安装 pip..."
|
||||
python3 -m ensurepip --upgrade 2>/dev/null || {
|
||||
echo " ❌ 无法安装 pip,尝试使用系统包管理器"
|
||||
if command -v apt &> /dev/null; then
|
||||
sudo apt update
|
||||
sudo apt install -y python3-pip
|
||||
elif command -v yum &> /dev/null; then
|
||||
sudo yum install -y python3-pip
|
||||
elif command -v brew &> /dev/null; then
|
||||
brew install python3
|
||||
else
|
||||
echo " ❌ 无法自动安装 pip,请手动安装"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
fi
|
||||
|
||||
# 检查 python-docx
|
||||
if ! python3 -c "import docx" 2>/dev/null; then
|
||||
echo " 🔧 安装 python-docx..."
|
||||
if python3 -m pip install python-docx --break-system-packages 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
elif python3 -m pip install python-docx 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
else
|
||||
echo "❌ 无法安装 python-docx"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo " ✅ python-docx 已安装"
|
||||
fi
|
||||
|
||||
# 检查 antiword(可选)
|
||||
if command -v antiword >/dev/null 2>&1; then
|
||||
echo " ✅ antiword 已安装"
|
||||
else
|
||||
echo " ⚠️ antiword 未安装(可选,用于 .doc 格式支持)"
|
||||
echo " 推荐安装命令:"
|
||||
echo " Ubuntu/Debian: sudo apt-get install antiword"
|
||||
echo " macOS: brew install antiword"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
|
||||
# 设置执行权限
|
||||
echo "🔐 设置执行权限..."
|
||||
chmod +x scripts/read_word.py
|
||||
echo "✅ 执行权限已设置"
|
||||
echo ""
|
||||
|
||||
# 验证安装
|
||||
echo "🧪 验证安装..."
|
||||
python3 scripts/read_word.py --help >/dev/null 2>&1
|
||||
if [ $? -eq 0 ]; then
|
||||
echo "✅ 安装验证成功"
|
||||
else
|
||||
echo "❌ 安装验证失败"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "🎉 Word Reader 技能安装完成!"
|
||||
echo ""
|
||||
echo "📖 使用方法:"
|
||||
echo " python3 scripts/read_word.py 文档.docx"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format json"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format markdown"
|
||||
echo ""
|
||||
echo "📖 更多帮助:"
|
||||
echo " python3 scripts/read_word.py --help"
|
||||
echo ""
|
||||
echo "📖 运行演示:"
|
||||
echo " ./demo.sh"
|
||||
396
runtime/skills/word-reader/scripts/read_word.py
Normal file
396
runtime/skills/word-reader/scripts/read_word.py
Normal file
|
|
@ -0,0 +1,396 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Word 文档读取器
|
||||
支持 .docx 和 .doc 格式的 Word 文档解析
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import traceback
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from docx import Document
|
||||
from docx.opc.constants import RELATIONSHIP_TYPE as RT
|
||||
from docx.oxml.table import CT_Tbl
|
||||
from docx.oxml.text.paragraph import CT_P
|
||||
from docx.table import Table
|
||||
from docx.text.paragraph import Paragraph
|
||||
DOCX_AVAILABLE = True
|
||||
except ImportError:
|
||||
DOCX_AVAILABLE = False
|
||||
|
||||
try:
|
||||
import subprocess
|
||||
SUBPROCESS_AVAILABLE = True
|
||||
except ImportError:
|
||||
SUBPROCESS_AVAILABLE = False
|
||||
|
||||
class WordReader:
|
||||
"""Word 文档读取器"""
|
||||
|
||||
def __init__(self, file_path):
|
||||
self.file_path = Path(file_path)
|
||||
self.document = None
|
||||
self.format_type = None
|
||||
self.encoding = 'utf-8'
|
||||
|
||||
# 检查文件是否存在
|
||||
if not self.file_path.exists():
|
||||
raise FileNotFoundError(f"文件不存在: {file_path}")
|
||||
|
||||
# 检查文件扩展名
|
||||
if self.file_path.suffix.lower() not in ['.docx', '.doc']:
|
||||
raise ValueError(f"不支持的文件格式: {self.file_path.suffix}")
|
||||
|
||||
def read_docx(self):
|
||||
"""读取 .docx 格式文档"""
|
||||
if not DOCX_AVAILABLE:
|
||||
raise Exception("缺少 python-docx 库。请安装:pip3 install python-docx")
|
||||
|
||||
try:
|
||||
self.document = Document(str(self.file_path))
|
||||
self.format_type = 'docx'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .docx 文件失败: {str(e)}")
|
||||
|
||||
def read_doc(self):
|
||||
"""读取 .doc 格式文档(使用 antiword)"""
|
||||
if not SUBPROCESS_AVAILABLE:
|
||||
raise Exception("缺少 subprocess 模块")
|
||||
|
||||
try:
|
||||
# 检查 antiword 是否可用
|
||||
result = subprocess.run(['which', 'antiword'],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
raise Exception("antiword 未安装。请安装 antiword: Ubuntu/Debian: sudo apt-get install antiword; macOS: brew install antiword")
|
||||
|
||||
# 使用 antiword 转换
|
||||
result = subprocess.run(['antiword', str(self.file_path)],
|
||||
capture_output=True, text=True, encoding='utf-8')
|
||||
|
||||
if result.returncode != 0:
|
||||
raise Exception(f"antiword 转换失败: {result.stderr}")
|
||||
|
||||
# 创建临时文档对象
|
||||
class TempDocument:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
self.paragraphs = [TempParagraph(p) for p in text.split('\n') if p.strip()]
|
||||
|
||||
class TempParagraph:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
|
||||
self.document = TempDocument(result.stdout)
|
||||
self.format_type = 'doc'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .doc 文件失败: {str(e)}")
|
||||
|
||||
def read_metadata(self):
|
||||
"""读取文档元数据"""
|
||||
metadata = {
|
||||
'filename': self.file_path.name,
|
||||
'size': f"{self.file_path.stat().st_size} bytes",
|
||||
'created': datetime.fromtimestamp(self.file_path.stat().st_ctime).isoformat(),
|
||||
'modified': datetime.fromtimestamp(self.file_path.stat().st_mtime).isoformat()
|
||||
}
|
||||
|
||||
if self.format_type == 'docx' and hasattr(self.document, 'core_properties'):
|
||||
props = self.document.core_properties
|
||||
metadata.update({
|
||||
'title': getattr(props, 'title', ''),
|
||||
'author': getattr(props, 'author', ''),
|
||||
'subject': getattr(props, 'subject', ''),
|
||||
'keywords': getattr(props, 'keywords', ''),
|
||||
'comments': getattr(props, 'comments', ''),
|
||||
'application': getattr(props, 'application', ''),
|
||||
'category': getattr(props, 'category', '')
|
||||
})
|
||||
|
||||
return metadata
|
||||
|
||||
def extract_text(self):
|
||||
"""提取文档文本"""
|
||||
text_content = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
# 提取段落文本
|
||||
for para in self.document.paragraphs:
|
||||
if para.text.strip():
|
||||
text_content.append(para.text)
|
||||
|
||||
# 提取表格文本
|
||||
for table in self.document.tables:
|
||||
table_text = []
|
||||
for row in table.rows:
|
||||
row_text = []
|
||||
for cell in row.cells:
|
||||
row_text.append(cell.text.strip())
|
||||
table_text.append(' | '.join(row_text))
|
||||
text_content.append('\n'.join(table_text))
|
||||
|
||||
else: # doc 格式
|
||||
text_content = [para.text for para in self.document.paragraphs if para.text.strip()]
|
||||
|
||||
return '\n\n'.join(text_content)
|
||||
|
||||
def extract_tables(self):
|
||||
"""提取表格数据"""
|
||||
tables = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
for i, table in enumerate(self.document.tables):
|
||||
table_data = []
|
||||
for row in table.rows:
|
||||
row_data = []
|
||||
for cell in row.cells:
|
||||
row_data.append(cell.text.strip())
|
||||
table_data.append(row_data)
|
||||
tables.append({
|
||||
'id': i + 1,
|
||||
'rows': len(table.rows),
|
||||
'columns': len(table.columns) if table.rows else 0,
|
||||
'data': table_data
|
||||
})
|
||||
|
||||
return tables
|
||||
|
||||
def extract_images(self):
|
||||
"""提取图片信息"""
|
||||
images = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
try:
|
||||
# 获取文档中的关系
|
||||
part = self.document.part
|
||||
image_parts = part.related_parts
|
||||
|
||||
for rel in part.relationships:
|
||||
if rel.reltype == RT.IMAGE:
|
||||
image_data = image_parts[rel.rId]._blob
|
||||
image_info = {
|
||||
'id': rel.rId,
|
||||
'filename': f"image_{rel.rId}.{rel.target_ref.split('.')[-1]}",
|
||||
'size': f"{len(image_data)} bytes"
|
||||
}
|
||||
images.append(image_info)
|
||||
except:
|
||||
# 图片提取可能失败,忽略错误
|
||||
pass
|
||||
|
||||
return images
|
||||
|
||||
def extract_all(self):
|
||||
"""提取所有内容"""
|
||||
result = {
|
||||
'metadata': self.read_metadata(),
|
||||
'format': self.format_type,
|
||||
'text': self.extract_text(),
|
||||
'tables': self.extract_tables(),
|
||||
'images': self.extract_images()
|
||||
}
|
||||
return result
|
||||
|
||||
def to_markdown(self, extract_type='all'):
|
||||
"""转换为 Markdown 格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
md_content = []
|
||||
|
||||
# 标题
|
||||
md_content.append(f"# {result['metadata']['filename']}")
|
||||
md_content.append("")
|
||||
|
||||
# 元数据
|
||||
metadata = result['metadata']
|
||||
if metadata.get('title'):
|
||||
md_content.append(f"**标题**:{metadata['title']}")
|
||||
if metadata.get('author'):
|
||||
md_content.append(f"**作者**:{metadata['author']}")
|
||||
md_content.append(f"**文件大小**:{metadata['size']}")
|
||||
md_content.append(f"**创建时间**:{metadata['created']}")
|
||||
md_content.append(f"**修改时间**:{metadata['modified']}")
|
||||
md_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
md_content.append("## 正文内容")
|
||||
md_content.append("")
|
||||
md_content.append(result['text'])
|
||||
md_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
md_content.append("## 表格内容")
|
||||
md_content.append("")
|
||||
for table in result['tables']:
|
||||
md_content.append(f"### 表格 {table['id']} ({table['rows']}行 x {table['columns']}列)")
|
||||
md_content.append("")
|
||||
# 转换为 Markdown 表格
|
||||
for row in table['data']:
|
||||
md_row = " | ".join([str(cell) for cell in row])
|
||||
md_content.append(f"| {md_row} |")
|
||||
md_content.append("")
|
||||
|
||||
# 图片
|
||||
if result['images']:
|
||||
md_content.append("## 图片列表")
|
||||
md_content.append("")
|
||||
for img in result['images']:
|
||||
md_content.append(f"- **{img['filename']}** ({img['size']})")
|
||||
md_content.append("")
|
||||
|
||||
return '\n'.join(md_content)
|
||||
|
||||
def to_text(self, extract_type='all'):
|
||||
"""转换为纯文本格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
text_content = []
|
||||
|
||||
# 标题和元数据
|
||||
text_content.append(f"文件:{result['metadata']['filename']}")
|
||||
text_content.append("=" * 50)
|
||||
text_content.append("")
|
||||
|
||||
for key, value in result['metadata'].items():
|
||||
if value and key not in ['filename', 'size', 'created', 'modified']:
|
||||
text_content.append(f"{key}:{value}")
|
||||
|
||||
text_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
text_content.append("正文内容:")
|
||||
text_content.append("-" * 20)
|
||||
text_content.append(result['text'])
|
||||
text_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
text_content.append("表格内容:")
|
||||
text_content.append("-" * 20)
|
||||
for table in result['tables']:
|
||||
text_content.append(f"表格 {table['id']}:")
|
||||
for row in table['data']:
|
||||
text_content.append(" " + " | ".join([str(cell) for cell in row]))
|
||||
text_content.append("")
|
||||
|
||||
return '\n'.join(text_content)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description='读取 Word 文档')
|
||||
parser.add_argument('path', help='文档路径或目录路径(批量模式)')
|
||||
parser.add_argument('--format', choices=['json', 'text', 'markdown'],
|
||||
default='text', help='输出格式')
|
||||
parser.add_argument('--extract', choices=['text', 'tables', 'images', 'metadata', 'all'],
|
||||
default='all', help='提取内容类型')
|
||||
parser.add_argument('--batch', action='store_true', help='批量处理模式')
|
||||
parser.add_argument('--output', help='输出文件路径')
|
||||
parser.add_argument('--encoding', default='utf-8', help='文本编码')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
try:
|
||||
if args.batch:
|
||||
# 批量处理模式
|
||||
path = Path(args.path)
|
||||
if not path.is_dir():
|
||||
print("错误:批量模式需要指定目录路径")
|
||||
sys.exit(1)
|
||||
|
||||
# 查找所有 Word 文档
|
||||
word_files = []
|
||||
for ext in ['.docx', '.doc']:
|
||||
word_files.extend(path.glob(f"**/*{ext}"))
|
||||
|
||||
if not word_files:
|
||||
print("未找到 Word 文档")
|
||||
sys.exit(0)
|
||||
|
||||
print(f"找到 {len(word_files)} 个 Word 文档")
|
||||
|
||||
results = {}
|
||||
for file_path in word_files:
|
||||
print(f"正在处理: {file_path}")
|
||||
try:
|
||||
reader = WordReader(file_path)
|
||||
if file_path.suffix.lower() == '.docx':
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'content': content,
|
||||
'status': 'success'
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'error': str(e),
|
||||
'status': 'failed'
|
||||
}
|
||||
|
||||
# 保存结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(results, f, ensure_ascii=False, indent=2)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
|
||||
else:
|
||||
# 单文件处理模式
|
||||
reader = WordReader(args.path)
|
||||
|
||||
# 根据文件类型读取
|
||||
if args.path.lower().endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
# 根据格式输出
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
# 输出结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding=args.encoding) as f:
|
||||
f.write(content)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(content)
|
||||
|
||||
except Exception as e:
|
||||
print(f"错误: {str(e)}", file=sys.stderr)
|
||||
if '--debug' in sys.argv or '-d' in sys.argv:
|
||||
traceback.print_exc()
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
47
runtime/skills/word-reader/skill.json
Normal file
47
runtime/skills/word-reader/skill.json
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
{
|
||||
"name": "word-reader",
|
||||
"version": "1.0.0",
|
||||
"description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容",
|
||||
"author": "OpenClaw User",
|
||||
"tags": ["document", "word", "office", "text-extraction"],
|
||||
"dependencies": {
|
||||
"python": ">=3.6",
|
||||
"packages": ["python-docx"],
|
||||
"system": ["antiword (optional for .doc support)"]
|
||||
},
|
||||
"features": {
|
||||
"text_extraction": true,
|
||||
"table_parsing": true,
|
||||
"metadata_extraction": true,
|
||||
"image_info": true,
|
||||
"batch_processing": true,
|
||||
"multiple_formats": ["json", "text", "markdown"]
|
||||
},
|
||||
"installation": {
|
||||
"steps": [
|
||||
"pip3 install python-docx",
|
||||
"sudo apt-get install antiword # 可选,支持 .doc 格式",
|
||||
"chmod +x scripts/read_word.py"
|
||||
]
|
||||
},
|
||||
"usage_examples": [
|
||||
{
|
||||
"description": "读取文档文本",
|
||||
"command": "python3 scripts/read_word.py document.docx"
|
||||
},
|
||||
{
|
||||
"description": "转换为 Markdown",
|
||||
"command": "python3 scripts/read_word.py document.docx --format markdown"
|
||||
},
|
||||
{
|
||||
"description": "批量处理",
|
||||
"command": "python3 scripts/read_word.py ./docs --batch --format json"
|
||||
}
|
||||
],
|
||||
"supported_file_types": [".docx", ".doc"],
|
||||
"notes": [
|
||||
".doc 格式需要安装 antiword",
|
||||
"大文档处理可能需要较长时间",
|
||||
"图片提取仅获取元数据,不包含实际图片数据"
|
||||
]
|
||||
}
|
||||
42
runtime/skills/word-reader/test.md
Normal file
42
runtime/skills/word-reader/test.md
Normal file
|
|
@ -0,0 +1,42 @@
|
|||
# Word Reader 技能测试
|
||||
|
||||
这是一个简单的测试文档,用于验证 Word Reader 技能的功能。
|
||||
|
||||
## 测试内容
|
||||
|
||||
### 1. 基本文本
|
||||
这是一段测试文本,用于验证文本提取功能是否正常工作。
|
||||
|
||||
### 2. 表格测试
|
||||
|
||||
| 功能 | 状态 | 描述 |
|
||||
|------|------|------|
|
||||
| 文本提取 | ✅ | 能够提取文档中的所有文本内容 |
|
||||
| 表格解析 | ✅ | 能够正确解析表格数据 |
|
||||
| 元数据获取 | ✅ | 能够获取文档属性信息 |
|
||||
| 多格式支持 | ✅ | 支持 .docx 和 .doc 格式 |
|
||||
| 输出格式 | ✅ | 支持 JSON、Text、Markdown 格式 |
|
||||
|
||||
### 3. 列表测试
|
||||
|
||||
- 第一项:文本提取功能
|
||||
- 第二项:表格解析功能
|
||||
- 第三项:图片信息获取
|
||||
- 第四项:文档元数据读取
|
||||
|
||||
### 4. 代码块示例
|
||||
|
||||
```python
|
||||
def read_word_document(file_path):
|
||||
"""读取 Word 文档"""
|
||||
reader = WordReader(file_path)
|
||||
if file_path.endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
return reader.extract_all()
|
||||
```
|
||||
|
||||
## 测试完成
|
||||
|
||||
如果这个技能能够正确读取并解析上述内容,说明功能正常。
|
||||
61
runtime/skills/xlsx-pro/README.md
Normal file
61
runtime/skills/xlsx-pro/README.md
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
# XLSX Pro
|
||||
|
||||
 
|
||||
|
||||
Un skill **Clawdbot / OpenClawd** pour générer et modifier des fichiers Excel **propres** (XLSX / XLSM / CSV / TSV) avec :
|
||||
- formatage “pro”
|
||||
- **formules Excel** (au lieu de valeurs hardcodées)
|
||||
- recalcul optionnel des formules via **LibreOffice headless**
|
||||
- contrôle qualité : détection des erreurs Excel (`#REF!`, `#DIV/0!`, `#VALUE!`, `#N/A`, `#NAME?`, …)
|
||||
|
||||
## Pourquoi ce skill ?
|
||||
|
||||
`openpyxl` sait **écrire** des formules, mais ne sait pas **calculer** leurs résultats. En production, ça crée des fichiers où les formules ne sont pas évaluées et où les erreurs ne sont pas détectées.
|
||||
|
||||
`XLSX Pro` ajoute une étape serveur fiable : **recalcul via LibreOffice** + scan d’erreurs.
|
||||
|
||||
## Prérequis
|
||||
|
||||
### Python
|
||||
```bash
|
||||
pip install openpyxl pandas xlrd xlwt
|
||||
```
|
||||
|
||||
### LibreOffice (uniquement si tu veux recalculer les formules)
|
||||
Ubuntu/Debian :
|
||||
```bash
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libreoffice-calc libreoffice-common
|
||||
```
|
||||
|
||||
## Quickstart
|
||||
|
||||
### 1) Générer un fichier Excel (avec openpyxl)
|
||||
Tu peux créer ton `.xlsx` comme d’habitude en Python, en mettant des **formules** dans les cellules.
|
||||
|
||||
### 2) Recalculer + valider
|
||||
```bash
|
||||
python scripts/recalc.py ton_fichier.xlsx 60
|
||||
```
|
||||
|
||||
Sortie JSON :
|
||||
- `status: success | errors_found`
|
||||
- `total_errors`
|
||||
- `error_summary` (types + emplacements)
|
||||
- `total_formulas`
|
||||
|
||||
## Bonnes pratiques (résumé)
|
||||
- **Préférer les formules Excel** plutôt que calculer en Python puis écrire des valeurs.
|
||||
- **Zéro erreur de formule** dans le livrable.
|
||||
- Si tu modifies un template existant : **respecte exactement** les styles/conventions.
|
||||
|
||||
## Troubleshooting
|
||||
- Si `soffice` est introuvable : installe LibreOffice (voir Prérequis).
|
||||
- Si le recalcul “timeout” : augmente le timeout (2e argument) et/ou teste sur un fichier plus petit.
|
||||
- Si erreur du type "macro mal configurée" / "macro not configured" : supprime le fichier de macro puis relance :
|
||||
- Linux : `~/.config/libreoffice/4/user/basic/Standard/Module1.xba`
|
||||
- macOS : `~/Library/Application Support/LibreOffice/4/user/basic/Standard/Module1.xba`
|
||||
- En conteneur (Docker) : ajoute la variable d'env `SAL_USE_VCLPLUGIN=svp` (ça évite des soucis d'UI en headless).
|
||||
|
||||
## Licence
|
||||
MIT (à ajuster si tu veux une autre licence).
|
||||
232
runtime/skills/xlsx-pro/SKILL.md
Normal file
232
runtime/skills/xlsx-pro/SKILL.md
Normal file
|
|
@ -0,0 +1,232 @@
|
|||
---
|
||||
name: xlsx-pro
|
||||
description: "Compétence pour manipuler les fichiers Excel (.xlsx, .xlsm, .csv, .tsv). Utiliser quand l'utilisateur veut : ouvrir, lire, éditer ou créer un fichier tableur ; ajouter des colonnes, calculer des formules, formater, créer des graphiques, nettoyer des données ; convertir entre formats tabulaires. Le livrable doit être un fichier tableur. NE PAS utiliser si le livrable est un document Word, HTML, script Python standalone, ou intégration Google Sheets."
|
||||
version: "1.0.1"
|
||||
author: "Eric Barotte"
|
||||
---
|
||||
|
||||
# Compétence Excel pour OpenClawd
|
||||
|
||||
## TL;DR
|
||||
- Génère/édite des fichiers Excel avec des **formules** (pas des valeurs hardcodées).
|
||||
- Optionnel: **recalcul** via LibreOffice headless + détection d’erreurs Excel.
|
||||
- Livrable attendu: un fichier tableur propre (XLSX/XLSM/CSV/TSV).
|
||||
|
||||
|
||||
## Prérequis
|
||||
|
||||
### Dépendances Python
|
||||
```bash
|
||||
pip install openpyxl pandas xlrd xlwt
|
||||
```
|
||||
|
||||
### LibreOffice (pour recalcul des formules)
|
||||
```bash
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install libreoffice-calc libreoffice-common
|
||||
```
|
||||
|
||||
## Règles de Qualité
|
||||
|
||||
### Police Professionnelle
|
||||
- Utiliser une police cohérente (Arial, Times New Roman) sauf instruction contraire
|
||||
|
||||
### Zéro Erreur de Formule
|
||||
- Tout fichier Excel DOIT être livré SANS erreurs (#REF!, #DIV/0!, #VALUE!, #N/A, #NAME?)
|
||||
|
||||
### Préservation des Templates
|
||||
- Respecter EXACTEMENT le format et style existants lors de modifications
|
||||
- Les conventions du template préexistant ont TOUJOURS priorité
|
||||
|
||||
## Standards pour Modèles Financiers
|
||||
|
||||
### Code Couleur (Standards Industrie)
|
||||
- **Texte bleu (RGB: 0,0,255)** : Inputs hardcodés, valeurs modifiables
|
||||
- **Texte noir (RGB: 0,0,0)** : TOUTES les formules et calculs
|
||||
- **Texte vert (RGB: 0,128,0)** : Liens vers autres feuilles du même classeur
|
||||
- **Texte rouge (RGB: 255,0,0)** : Liens externes vers autres fichiers
|
||||
- **Fond jaune (RGB: 255,255,0)** : Hypothèses clés ou cellules à mettre à jour
|
||||
|
||||
### Formatage des Nombres
|
||||
- **Années** : Format texte ("2024" pas "2,024")
|
||||
- **Devises** : Format $#,##0 ; spécifier unités dans les en-têtes ("Revenue ($mm)")
|
||||
- **Zéros** : Afficher comme "-" (format: "$#,##0;($#,##0);-")
|
||||
- **Pourcentages** : Format 0.0% par défaut
|
||||
- **Multiples** : Format 0.0x (EV/EBITDA, P/E)
|
||||
- **Négatifs** : Parenthèses (123) pas moins -123
|
||||
|
||||
## CRITIQUE : Utiliser des Formules, PAS des Valeurs Hardcodées
|
||||
|
||||
**TOUJOURS utiliser des formules Excel au lieu de calculer en Python et hardcoder.**
|
||||
|
||||
### ❌ MAUVAIS - Hardcoding
|
||||
```python
|
||||
# Mauvais: Calcul Python puis hardcode
|
||||
total = df['Sales'].sum()
|
||||
sheet['B10'] = total # Hardcode 5000
|
||||
|
||||
# Mauvais: Taux de croissance calculé en Python
|
||||
growth = (df.iloc[-1]['Revenue'] - df.iloc[0]['Revenue']) / df.iloc[0]['Revenue']
|
||||
sheet['C5'] = growth # Hardcode 0.15
|
||||
```
|
||||
|
||||
### ✅ CORRECT - Formules Excel
|
||||
```python
|
||||
# Bon: Laisser Excel calculer
|
||||
sheet['B10'] = '=SUM(B2:B9)'
|
||||
|
||||
# Bon: Taux de croissance en formule Excel
|
||||
sheet['C5'] = '=(C4-C2)/C2'
|
||||
|
||||
# Bon: Moyenne en fonction Excel
|
||||
sheet['D20'] = '=AVERAGE(D2:D19)'
|
||||
```
|
||||
|
||||
## Workflows
|
||||
|
||||
### Workflow Standard
|
||||
1. **Choisir l'outil** : pandas pour données, openpyxl pour formules/formatage
|
||||
2. **Créer/Charger** : Nouveau classeur ou fichier existant
|
||||
3. **Modifier** : Données, formules, formatage
|
||||
4. **Sauvegarder** : Écrire le fichier
|
||||
5. **Recalculer (OBLIGATOIRE si formules)** : `python scripts/recalc.py output.xlsx`
|
||||
6. **Vérifier et corriger** les erreurs détectées
|
||||
|
||||
### Lecture et Analyse avec pandas
|
||||
```python
|
||||
import pandas as pd
|
||||
|
||||
# Lire Excel
|
||||
df = pd.read_excel('file.xlsx') # Première feuille par défaut
|
||||
all_sheets = pd.read_excel('file.xlsx', sheet_name=None) # Dict de toutes les feuilles
|
||||
|
||||
# Analyser
|
||||
df.head() # Aperçu
|
||||
df.info() # Info colonnes
|
||||
df.describe() # Statistiques
|
||||
|
||||
# Écrire
|
||||
df.to_excel('output.xlsx', index=False)
|
||||
```
|
||||
|
||||
### Création de Fichiers Excel
|
||||
```python
|
||||
from openpyxl import Workbook
|
||||
from openpyxl.styles import Font, PatternFill, Alignment
|
||||
|
||||
wb = Workbook()
|
||||
sheet = wb.active
|
||||
|
||||
# Données
|
||||
sheet['A1'] = 'Hello'
|
||||
sheet['B1'] = 'World'
|
||||
sheet.append(['Row', 'of', 'data'])
|
||||
|
||||
# Formule
|
||||
sheet['B2'] = '=SUM(A1:A10)'
|
||||
|
||||
# Formatage
|
||||
sheet['A1'].font = Font(bold=True, color='FF0000')
|
||||
sheet['A1'].fill = PatternFill('solid', start_color='FFFF00')
|
||||
sheet['A1'].alignment = Alignment(horizontal='center')
|
||||
|
||||
# Largeur colonne
|
||||
sheet.column_dimensions['A'].width = 20
|
||||
|
||||
wb.save('output.xlsx')
|
||||
```
|
||||
|
||||
### Édition de Fichiers Existants
|
||||
```python
|
||||
from openpyxl import load_workbook
|
||||
|
||||
# Charger fichier existant
|
||||
wb = load_workbook('existing.xlsx')
|
||||
sheet = wb.active # ou wb['NomFeuille']
|
||||
|
||||
# Parcourir les feuilles
|
||||
for sheet_name in wb.sheetnames:
|
||||
sheet = wb[sheet_name]
|
||||
print(f"Feuille: {sheet_name}")
|
||||
|
||||
# Modifier
|
||||
sheet['A1'] = 'Nouvelle Valeur'
|
||||
sheet.insert_rows(2) # Insérer ligne
|
||||
sheet.delete_cols(3) # Supprimer colonne
|
||||
|
||||
# Ajouter feuille
|
||||
new_sheet = wb.create_sheet('NouvelleFeuille')
|
||||
new_sheet['A1'] = 'Data'
|
||||
|
||||
wb.save('modified.xlsx')
|
||||
```
|
||||
|
||||
## Recalcul des Formules
|
||||
|
||||
Les fichiers créés par openpyxl contiennent les formules comme chaînes mais pas les valeurs calculées. Utiliser le script `recalc.py` :
|
||||
|
||||
```bash
|
||||
python scripts/recalc.py <fichier_excel> [timeout_secondes]
|
||||
```
|
||||
|
||||
Le script :
|
||||
- Configure automatiquement la macro LibreOffice au premier lancement
|
||||
- Recalcule toutes les formules
|
||||
- Scanne TOUTES les cellules pour erreurs Excel
|
||||
- Retourne JSON avec détails et emplacements des erreurs
|
||||
|
||||
### Interprétation de la Sortie
|
||||
```json
|
||||
{
|
||||
"status": "success", // ou "errors_found"
|
||||
"total_errors": 0, // Nombre total d'erreurs
|
||||
"total_formulas": 42, // Nombre de formules
|
||||
"error_summary": { // Présent si erreurs
|
||||
"#REF!": {
|
||||
"count": 2,
|
||||
"locations": ["Sheet1!B5", "Sheet1!C10"]
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Checklist de Vérification
|
||||
|
||||
### Vérifications Essentielles
|
||||
- [ ] **Tester 2-3 références** : Vérifier qu'elles tirent les bonnes valeurs
|
||||
- [ ] **Mapping colonnes** : Confirmer correspondance (colonne 64 = BL, pas BK)
|
||||
- [ ] **Offset lignes** : Excel est 1-indexé (DataFrame row 5 = Excel row 6)
|
||||
|
||||
### Pièges Courants
|
||||
- [ ] **Gestion NaN** : Vérifier valeurs nulles avec `pd.notna()`
|
||||
- [ ] **Colonnes éloignées** : Données FY souvent en colonnes 50+
|
||||
- [ ] **Correspondances multiples** : Chercher toutes les occurrences
|
||||
- [ ] **Division par zéro** : Vérifier dénominateurs (#DIV/0!)
|
||||
- [ ] **Références invalides** : Vérifier que toutes pointent vers cellules existantes (#REF!)
|
||||
- [ ] **Références inter-feuilles** : Format correct (Sheet1!A1)
|
||||
|
||||
## Bonnes Pratiques
|
||||
|
||||
### Sélection de Bibliothèque
|
||||
- **pandas** : Analyse de données, opérations en masse, export simple
|
||||
- **openpyxl** : Formatage complexe, formules, fonctionnalités Excel spécifiques
|
||||
|
||||
### Avec openpyxl
|
||||
- Indices de cellules en base 1 (row=1, column=1 = cellule A1)
|
||||
- `data_only=True` pour lire valeurs calculées
|
||||
- **Attention** : Sauvegarder après `data_only=True` remplace définitivement les formules par les valeurs
|
||||
- Pour gros fichiers : `read_only=True` ou `write_only=True`
|
||||
|
||||
### Avec pandas
|
||||
- Spécifier types de données : `pd.read_excel('file.xlsx', dtype={'id': str})`
|
||||
- Pour gros fichiers, colonnes spécifiques : `usecols=['A', 'C', 'E']`
|
||||
- Gestion des dates : `parse_dates=['date_column']`
|
||||
|
||||
## Style de Code
|
||||
|
||||
**IMPORTANT** : Code Python minimal et concis, sans commentaires superflus.
|
||||
|
||||
**Pour les fichiers Excel** :
|
||||
- Commenter les cellules avec formules complexes
|
||||
- Documenter les sources des données hardcodées
|
||||
- Inclure notes pour calculs clés
|
||||
6
runtime/skills/xlsx-pro/_meta.json
Normal file
6
runtime/skills/xlsx-pro/_meta.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn7aga06tewbtydgnw3v1468x9804b6x",
|
||||
"slug": "xlsx-pro",
|
||||
"version": "1.0.1",
|
||||
"publishedAt": 1770059662515
|
||||
}
|
||||
8
runtime/skills/xlsx-pro/scripts/office/__init__.py
Normal file
8
runtime/skills/xlsx-pro/scripts/office/__init__.py
Normal file
|
|
@ -0,0 +1,8 @@
|
|||
"""
|
||||
Module office pour OpenClawd
|
||||
Gestion des opérations LibreOffice
|
||||
"""
|
||||
|
||||
from .soffice import get_soffice_env, run_soffice
|
||||
|
||||
__all__ = ['get_soffice_env', 'run_soffice']
|
||||
211
runtime/skills/xlsx-pro/scripts/office/soffice.py
Normal file
211
runtime/skills/xlsx-pro/scripts/office/soffice.py
Normal file
|
|
@ -0,0 +1,211 @@
|
|||
"""
|
||||
Helper pour exécuter LibreOffice (soffice) dans des environnements
|
||||
où les sockets AF_UNIX peuvent être bloqués (VMs sandboxées).
|
||||
|
||||
Usage:
|
||||
from office.soffice import run_soffice, get_soffice_env
|
||||
|
||||
# Option 1 – exécuter soffice directement
|
||||
result = run_soffice(["--headless", "--convert-to", "pdf", "input.docx"])
|
||||
|
||||
# Option 2 – obtenir env dict pour vos propres appels subprocess
|
||||
env = get_soffice_env()
|
||||
subprocess.run(["soffice", ...], env=env)
|
||||
"""
|
||||
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def get_soffice_env() -> dict:
|
||||
"""Retourne un env dict adapté pour exécuter soffice en headless.
|
||||
|
||||
Définit toujours SAL_USE_VCLPLUGIN=svp pour le rendu headless (pas de X11).
|
||||
Dans les environnements sandboxés où AF_UNIX est bloqué, ajoute aussi
|
||||
LD_PRELOAD (socket shim).
|
||||
"""
|
||||
env = os.environ.copy()
|
||||
env["SAL_USE_VCLPLUGIN"] = "svp"
|
||||
|
||||
if _needs_shim():
|
||||
shim = _ensure_shim()
|
||||
env["LD_PRELOAD"] = str(shim)
|
||||
|
||||
return env
|
||||
|
||||
|
||||
def run_soffice(args: list, **kwargs) -> subprocess.CompletedProcess:
|
||||
"""Exécute soffice avec les arguments donnés, appliquant le socket shim
|
||||
si nécessaire. Accepte les mêmes arguments que subprocess.run.
|
||||
|
||||
Dans les environnements sandboxés, le shim gère l'arrêt propre en appelant
|
||||
_exit(0) quand le socket listener de soffice.bin se ferme (après conversion).
|
||||
"""
|
||||
env = get_soffice_env()
|
||||
return subprocess.run(["soffice"] + args, env=env, **kwargs)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Internals
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SHIM_SO = Path(tempfile.gettempdir()) / "lo_socket_shim.so"
|
||||
|
||||
|
||||
def _needs_shim() -> bool:
|
||||
"""Vérifie si les sockets AF_UNIX sont bloqués."""
|
||||
try:
|
||||
s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
s.close()
|
||||
return False
|
||||
except OSError:
|
||||
return True
|
||||
|
||||
|
||||
def _ensure_shim() -> Path:
|
||||
"""Compile le shim .so s'il n'est pas déjà en cache."""
|
||||
if _SHIM_SO.exists():
|
||||
return _SHIM_SO
|
||||
|
||||
src = Path(tempfile.gettempdir()) / "lo_socket_shim.c"
|
||||
src.write_text(_SHIM_SOURCE)
|
||||
try:
|
||||
subprocess.run(
|
||||
["gcc", "-shared", "-fPIC", "-o", str(_SHIM_SO), str(src), "-ldl"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||
# Si gcc n'est pas disponible ou échoue, on continue sans shim
|
||||
pass
|
||||
finally:
|
||||
if src.exists():
|
||||
src.unlink()
|
||||
return _SHIM_SO
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# LD_PRELOAD shim – source C
|
||||
#
|
||||
# Problème
|
||||
# --------
|
||||
# LibreOffice utilise des sockets AF_UNIX pour la gestion single-instance
|
||||
# (OSL_PIPE). Dans les environnements sandboxés, le filtre seccomp bloque
|
||||
# socket(AF_UNIX) tout en permettant socketpair(AF_UNIX). Sans ce shim,
|
||||
# soffice crash ou reste bloqué après conversion.
|
||||
#
|
||||
# Solution
|
||||
# --------
|
||||
# Intercepte les appels concernés et fournit des substituts fonctionnels.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SHIM_SOURCE = r"""
|
||||
#define _GNU_SOURCE
|
||||
#include <dlfcn.h>
|
||||
#include <errno.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <sys/socket.h>
|
||||
#include <unistd.h>
|
||||
|
||||
static int (*real_socket)(int, int, int);
|
||||
static int (*real_socketpair)(int, int, int, int[2]);
|
||||
static int (*real_listen)(int, int);
|
||||
static int (*real_accept)(int, struct sockaddr *, socklen_t *);
|
||||
static int (*real_close)(int);
|
||||
static int (*real_read)(int, void *, size_t);
|
||||
|
||||
static int is_shimmed[1024];
|
||||
static int peer_of[1024];
|
||||
static int wake_r[1024];
|
||||
static int wake_w[1024];
|
||||
static int listener_fd = -1;
|
||||
|
||||
__attribute__((constructor))
|
||||
static void init(void) {
|
||||
real_socket = dlsym(RTLD_NEXT, "socket");
|
||||
real_socketpair = dlsym(RTLD_NEXT, "socketpair");
|
||||
real_listen = dlsym(RTLD_NEXT, "listen");
|
||||
real_accept = dlsym(RTLD_NEXT, "accept");
|
||||
real_close = dlsym(RTLD_NEXT, "close");
|
||||
real_read = dlsym(RTLD_NEXT, "read");
|
||||
for (int i = 0; i < 1024; i++) {
|
||||
peer_of[i] = -1;
|
||||
wake_r[i] = -1;
|
||||
wake_w[i] = -1;
|
||||
}
|
||||
}
|
||||
|
||||
int socket(int domain, int type, int protocol) {
|
||||
if (domain == AF_UNIX) {
|
||||
int fd = real_socket(domain, type, protocol);
|
||||
if (fd >= 0) return fd;
|
||||
int sv[2];
|
||||
if (real_socketpair(domain, type, protocol, sv) == 0) {
|
||||
if (sv[0] >= 0 && sv[0] < 1024) {
|
||||
is_shimmed[sv[0]] = 1;
|
||||
peer_of[sv[0]] = sv[1];
|
||||
int wp[2];
|
||||
if (pipe(wp) == 0) {
|
||||
wake_r[sv[0]] = wp[0];
|
||||
wake_w[sv[0]] = wp[1];
|
||||
}
|
||||
}
|
||||
return sv[0];
|
||||
}
|
||||
errno = EPERM;
|
||||
return -1;
|
||||
}
|
||||
return real_socket(domain, type, protocol);
|
||||
}
|
||||
|
||||
int listen(int sockfd, int backlog) {
|
||||
if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) {
|
||||
listener_fd = sockfd;
|
||||
return 0;
|
||||
}
|
||||
return real_listen(sockfd, backlog);
|
||||
}
|
||||
|
||||
int accept(int sockfd, struct sockaddr *addr, socklen_t *addrlen) {
|
||||
if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) {
|
||||
if (wake_r[sockfd] >= 0) {
|
||||
char buf;
|
||||
real_read(wake_r[sockfd], &buf, 1);
|
||||
}
|
||||
errno = ECONNABORTED;
|
||||
return -1;
|
||||
}
|
||||
return real_accept(sockfd, addr, addrlen);
|
||||
}
|
||||
|
||||
int close(int fd) {
|
||||
if (fd >= 0 && fd < 1024 && is_shimmed[fd]) {
|
||||
int was_listener = (fd == listener_fd);
|
||||
is_shimmed[fd] = 0;
|
||||
|
||||
if (wake_w[fd] >= 0) {
|
||||
char c = 0;
|
||||
write(wake_w[fd], &c, 1);
|
||||
real_close(wake_w[fd]);
|
||||
wake_w[fd] = -1;
|
||||
}
|
||||
if (wake_r[fd] >= 0) { real_close(wake_r[fd]); wake_r[fd] = -1; }
|
||||
if (peer_of[fd] >= 0) { real_close(peer_of[fd]); peer_of[fd] = -1; }
|
||||
|
||||
if (was_listener)
|
||||
_exit(0);
|
||||
}
|
||||
return real_close(fd);
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
result = run_soffice(sys.argv[1:])
|
||||
sys.exit(result.returncode)
|
||||
225
runtime/skills/xlsx-pro/scripts/recalc.py
Normal file
225
runtime/skills/xlsx-pro/scripts/recalc.py
Normal file
|
|
@ -0,0 +1,225 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Script de recalcul des formules Excel
|
||||
Recalcule toutes les formules d'un fichier Excel via LibreOffice
|
||||
Adapté pour OpenClawd
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from office.soffice import get_soffice_env
|
||||
except ImportError:
|
||||
# Fallback si le module office n'est pas disponible
|
||||
def get_soffice_env():
|
||||
env = os.environ.copy()
|
||||
env["SAL_USE_VCLPLUGIN"] = "svp"
|
||||
return env
|
||||
|
||||
try:
|
||||
from openpyxl import load_workbook
|
||||
except ImportError:
|
||||
print("Erreur: openpyxl non installé. Exécuter: pip install openpyxl")
|
||||
sys.exit(1)
|
||||
|
||||
# Répertoire macro LibreOffice selon plateforme
|
||||
MACRO_DIR_MACOS = "~/Library/Application Support/LibreOffice/4/user/basic/Standard"
|
||||
MACRO_DIR_LINUX = "~/.config/libreoffice/4/user/basic/Standard"
|
||||
MACRO_FILENAME = "Module1.xba"
|
||||
|
||||
# Macro LibreOffice Basic pour recalcul
|
||||
RECALCULATE_MACRO = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE script:module PUBLIC "-//OpenOffice.org//DTD OfficeDocument 1.0//EN" "module.dtd">
|
||||
<script:module xmlns:script="http://openoffice.org/2000/script" script:name="Module1" script:language="StarBasic">
|
||||
Sub RecalculateAndSave()
|
||||
ThisComponent.calculateAll()
|
||||
ThisComponent.store()
|
||||
ThisComponent.close(True)
|
||||
End Sub
|
||||
</script:module>"""
|
||||
|
||||
|
||||
def has_gtimeout():
|
||||
"""Vérifie si gtimeout est disponible sur macOS"""
|
||||
try:
|
||||
subprocess.run(
|
||||
["gtimeout", "--version"], capture_output=True, timeout=1, check=False
|
||||
)
|
||||
return True
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return False
|
||||
|
||||
|
||||
def setup_libreoffice_macro():
|
||||
"""Configure la macro LibreOffice si pas déjà fait"""
|
||||
macro_dir = os.path.expanduser(
|
||||
MACRO_DIR_MACOS if platform.system() == "Darwin" else MACRO_DIR_LINUX
|
||||
)
|
||||
macro_file = os.path.join(macro_dir, MACRO_FILENAME)
|
||||
|
||||
# Vérifier si macro existe déjà
|
||||
if (
|
||||
os.path.exists(macro_file)
|
||||
and "RecalculateAndSave" in Path(macro_file).read_text()
|
||||
):
|
||||
return True
|
||||
|
||||
# Créer répertoire macro si nécessaire
|
||||
if not os.path.exists(macro_dir):
|
||||
try:
|
||||
subprocess.run(
|
||||
["soffice", "--headless", "--terminate_after_init"],
|
||||
capture_output=True,
|
||||
timeout=10,
|
||||
env=get_soffice_env(),
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
os.makedirs(macro_dir, exist_ok=True)
|
||||
|
||||
# Écrire fichier macro
|
||||
try:
|
||||
Path(macro_file).write_text(RECALCULATE_MACRO)
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def recalc(filename, timeout=30):
|
||||
"""
|
||||
Recalcule les formules d'un fichier Excel et rapporte les erreurs
|
||||
|
||||
Args:
|
||||
filename: Chemin vers le fichier Excel
|
||||
timeout: Temps max d'attente pour le recalcul (secondes)
|
||||
|
||||
Returns:
|
||||
dict avec emplacements et compteurs d'erreurs
|
||||
"""
|
||||
if not Path(filename).exists():
|
||||
return {"error": f"Fichier {filename} inexistant"}
|
||||
|
||||
abs_path = str(Path(filename).absolute())
|
||||
|
||||
if not setup_libreoffice_macro():
|
||||
return {"error": "Échec configuration macro LibreOffice"}
|
||||
|
||||
cmd = [
|
||||
"soffice",
|
||||
"--headless",
|
||||
"--norestore",
|
||||
"vnd.sun.star.script:Standard.Module1.RecalculateAndSave?language=Basic&location=application",
|
||||
abs_path,
|
||||
]
|
||||
|
||||
# Encapsuler avec timeout si disponible
|
||||
if platform.system() == "Linux":
|
||||
cmd = ["timeout", str(timeout)] + cmd
|
||||
elif platform.system() == "Darwin" and has_gtimeout():
|
||||
cmd = ["gtimeout", str(timeout)] + cmd
|
||||
|
||||
try:
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, env=get_soffice_env(), timeout=timeout+10)
|
||||
except subprocess.TimeoutExpired:
|
||||
return {"error": "Timeout lors du recalcul"}
|
||||
|
||||
if result.returncode != 0 and result.returncode != 124: # 124 = code timeout
|
||||
error_msg = result.stderr or "Erreur inconnue lors du recalcul"
|
||||
if "Module1" in error_msg or "RecalculateAndSave" not in error_msg:
|
||||
return {"error": "Macro LibreOffice mal configurée"}
|
||||
return {"error": error_msg}
|
||||
|
||||
# Vérifier erreurs Excel dans le fichier recalculé
|
||||
try:
|
||||
wb = load_workbook(filename, data_only=True)
|
||||
|
||||
excel_errors = [
|
||||
"#VALUE!",
|
||||
"#DIV/0!",
|
||||
"#REF!",
|
||||
"#NAME?",
|
||||
"#NULL!",
|
||||
"#NUM!",
|
||||
"#N/A",
|
||||
]
|
||||
error_details = {err: [] for err in excel_errors}
|
||||
total_errors = 0
|
||||
|
||||
for sheet_name in wb.sheetnames:
|
||||
ws = wb[sheet_name]
|
||||
for row in ws.iter_rows():
|
||||
for cell in row:
|
||||
if cell.value is not None and isinstance(cell.value, str):
|
||||
for err in excel_errors:
|
||||
if err in cell.value:
|
||||
location = f"{sheet_name}!{cell.coordinate}"
|
||||
error_details[err].append(location)
|
||||
total_errors += 1
|
||||
break
|
||||
|
||||
wb.close()
|
||||
|
||||
# Construire résumé
|
||||
result = {
|
||||
"status": "success" if total_errors == 0 else "errors_found",
|
||||
"total_errors": total_errors,
|
||||
"error_summary": {},
|
||||
}
|
||||
|
||||
# Ajouter catégories d'erreurs non vides
|
||||
for err_type, locations in error_details.items():
|
||||
if locations:
|
||||
result["error_summary"][err_type] = {
|
||||
"count": len(locations),
|
||||
"locations": locations[:20], # Max 20 emplacements affichés
|
||||
}
|
||||
|
||||
# Ajouter compte de formules
|
||||
wb_formulas = load_workbook(filename, data_only=False)
|
||||
formula_count = 0
|
||||
for sheet_name in wb_formulas.sheetnames:
|
||||
ws = wb_formulas[sheet_name]
|
||||
for row in ws.iter_rows():
|
||||
for cell in row:
|
||||
if (
|
||||
cell.value
|
||||
and isinstance(cell.value, str)
|
||||
and cell.value.startswith("=")
|
||||
):
|
||||
formula_count += 1
|
||||
wb_formulas.close()
|
||||
|
||||
result["total_formulas"] = formula_count
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
return {"error": str(e)}
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python recalc.py <fichier_excel> [timeout_secondes]")
|
||||
print("\nRecalcule toutes les formules d'un fichier Excel via LibreOffice")
|
||||
print("\nRetourne JSON avec détails d'erreurs:")
|
||||
print(" - status: 'success' ou 'errors_found'")
|
||||
print(" - total_errors: Nombre total d'erreurs Excel")
|
||||
print(" - total_formulas: Nombre de formules dans le fichier")
|
||||
print(" - error_summary: Détail par type avec emplacements")
|
||||
print(" - #VALUE!, #DIV/0!, #REF!, #NAME?, #NULL!, #NUM!, #N/A")
|
||||
sys.exit(1)
|
||||
|
||||
filename = sys.argv[1]
|
||||
timeout = int(sys.argv[2]) if len(sys.argv) > 2 else 30
|
||||
|
||||
result = recalc(filename, timeout)
|
||||
print(json.dumps(result, indent=2, ensure_ascii=False))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -4,6 +4,16 @@ from dataclasses import dataclass
|
|||
from typing import Any, Protocol
|
||||
|
||||
from oclaw.runtime.tools.skills.clawhub_client import get_skill_detail, search_skills
|
||||
from oclaw.runtime.tools.skills.cocoloop_client import get_skill_detail_by_slug as cocoloop_get_skill_detail
|
||||
from oclaw.runtime.tools.skills.cocoloop_client import search_store_skills as cocoloop_search_skills
|
||||
|
||||
|
||||
def normalize_skill_market_provider_setting(raw: str | None) -> str:
|
||||
"""Tenant setting value for ``AIA_SKILL_MARKET_PROVIDER``: ``clawhub`` or ``cocoloop``."""
|
||||
p = str(raw or "").strip().lower()
|
||||
if p in {"cocoloop", "cocoloop-cn", "cocoloop_cn"}:
|
||||
return "cocoloop"
|
||||
return "clawhub"
|
||||
|
||||
|
||||
class SkillMarketAdapter(Protocol):
|
||||
|
|
@ -42,12 +52,45 @@ class ClawHubMarketAdapter:
|
|||
return str(detail.get("archiveUrl") or "").strip(), latest
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CocoloopMarketAdapter:
|
||||
provider: str = "cocoloop"
|
||||
|
||||
def search(self, query: str, *, limit: int = 20) -> list[dict[str, Any]]:
|
||||
return cocoloop_search_skills(query, limit=limit)
|
||||
|
||||
def detail(self, slug: str) -> dict[str, Any]:
|
||||
return cocoloop_get_skill_detail(slug)
|
||||
|
||||
def resolve_archive_url(self, *, slug: str, version: str | None = None) -> tuple[str, str]:
|
||||
detail = self.detail(slug)
|
||||
requested = str(version or "").strip().lstrip("vV")
|
||||
if requested:
|
||||
for row in detail.get("versions") or []:
|
||||
if not isinstance(row, dict):
|
||||
continue
|
||||
ver = str(row.get("version") or "").strip().lstrip("vV")
|
||||
if ver != requested:
|
||||
continue
|
||||
return str(row.get("archiveUrl") or "").strip(), str(row.get("version") or requested)
|
||||
latest = str(detail.get("latestVersion") or "").strip()
|
||||
return str(detail.get("archiveUrl") or "").strip(), latest
|
||||
|
||||
|
||||
def get_market_adapter(provider: str | None) -> SkillMarketAdapter:
|
||||
p = str(provider or "clawhub").strip().lower()
|
||||
p = normalize_skill_market_provider_setting(provider)
|
||||
if p in {"clawhub", "openclaw"}:
|
||||
return ClawHubMarketAdapter(provider="clawhub")
|
||||
if p in {"cocoloop", "cocoloop-cn", "cocoloop_cn"}:
|
||||
return CocoloopMarketAdapter(provider="cocoloop")
|
||||
raise ValueError(f"unsupported_market_provider:{p}")
|
||||
|
||||
|
||||
__all__ = ["SkillMarketAdapter", "ClawHubMarketAdapter", "get_market_adapter"]
|
||||
__all__ = [
|
||||
"SkillMarketAdapter",
|
||||
"ClawHubMarketAdapter",
|
||||
"CocoloopMarketAdapter",
|
||||
"get_market_adapter",
|
||||
"normalize_skill_market_provider_setting",
|
||||
]
|
||||
|
||||
|
|
|
|||
|
|
@ -50,6 +50,10 @@ def _unified_skill_policy_guidance() -> str:
|
|||
"- 如果脚本依赖相对路径(例如 `.learnings/`),请将工作目录设置为用户工作区。\n"
|
||||
"- 在没有显式工具调用成功结果前,不要假设脚本已经执行。\n"
|
||||
"- 在 Windows 上,`.sh` 可能需要 Git Bash、WSL 或等效环境。\n"
|
||||
"- 当用户目标是“安装 skill/技能”时,必须遵循 `oclaw-skill-manager` 的安装策略,并以其为唯一规范来源。\n"
|
||||
"- 安装路径强约束:仅允许 `skill_auto_install`(`_workspace` lane);不得改用任何非 auto 路径或脚本绕过。\n"
|
||||
"- 严禁臆测前置条件:不要把未在规范中声明的环境变量、端口、服务启动状态当作必需前提。\n"
|
||||
"- 若安装失败,仅输出可验证事实(至少包含 `error_code` 与 `detail`)和最小下一步,不得编造基础设施依赖。\n"
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -17,6 +17,13 @@ from oclaw.runtime.tools.skills_runtime.materialize_skill_tools import materiali
|
|||
from oclaw.runtime.skills import SkillSpec, materialize_skills_from_tool_specs
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
# Tools hidden from model-facing registry to enforce auto-install only policy.
|
||||
_MODEL_TOOLS_DENYLIST = frozenset(
|
||||
{
|
||||
"skill_market_install",
|
||||
"skill_registry_install",
|
||||
}
|
||||
)
|
||||
|
||||
# Legacy export: some modules still import TOOL_FACTORIES. Tools are now intentionally
|
||||
# restricted to a single safe builtin (`system_time`), so this is left empty.
|
||||
|
|
@ -114,6 +121,16 @@ def _resolve_tool_conflicts(collected: list[tuple[str, ToolSpec]]) -> list[ToolS
|
|||
|
||||
|
||||
def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]:
|
||||
# Lazy imports to avoid circular dependency at module import time.
|
||||
from oclaw.runtime.skill_installer import (
|
||||
auto_install_skill_from_payload,
|
||||
create_skill_from_template,
|
||||
install_skill_from_registry_archive,
|
||||
list_skills_with_status,
|
||||
)
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting
|
||||
|
||||
def _create_skill_handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
out = create_skill_from_template(
|
||||
store=store,
|
||||
|
|
@ -126,14 +143,60 @@ def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]:
|
|||
return {"ok": bool(out.ok), "name": out.name, "target_dir": out.target_dir, "detail": out.detail}
|
||||
|
||||
def _auto_install_skill_handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
archive_url = str(args.get("archive_url") or "").strip()
|
||||
slug = str(args.get("slug") or "").strip()
|
||||
version = str(args.get("version") or "").strip() or None
|
||||
overwrite = bool(args.get("overwrite"))
|
||||
provider = normalize_skill_market_provider_setting(
|
||||
str(args.get("provider") or store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "")
|
||||
)
|
||||
if not archive_url and slug:
|
||||
try:
|
||||
adapter = get_market_adapter(provider)
|
||||
archive_url, _chosen_version = adapter.resolve_archive_url(slug=slug, version=version)
|
||||
except Exception:
|
||||
archive_url = ""
|
||||
if archive_url:
|
||||
out = install_skill_from_registry_archive(
|
||||
store=store,
|
||||
archive_url=archive_url,
|
||||
overwrite=overwrite,
|
||||
skills_root=default_skills_root() / "_workspace",
|
||||
auto_bind=True,
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
"provider": provider,
|
||||
}
|
||||
|
||||
payload = {
|
||||
"name": str(args.get("name") or "").strip(),
|
||||
"description": str(args.get("description") or "").strip(),
|
||||
"body_markdown": str(args.get("body_markdown") or "").strip(),
|
||||
"metadata_oclaw": dict(args.get("metadata_oclaw") or {}) if isinstance(args.get("metadata_oclaw"), dict) else {},
|
||||
}
|
||||
if not payload["name"]:
|
||||
return {"ok": False, "error_code": "name_required", "error": "name_required"}
|
||||
if not payload["description"]:
|
||||
payload["description"] = f"{payload['name']} skill"
|
||||
out = auto_install_skill_from_payload(store=store, payload=payload)
|
||||
return {"ok": bool(out.ok), "name": out.name, "target_dir": out.target_dir, "detail": out.detail}
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
}
|
||||
|
||||
def _list_skills_handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
del args
|
||||
|
|
@ -146,9 +209,13 @@ def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]:
|
|||
"description": {"type": "string"},
|
||||
"body_markdown": {"type": "string"},
|
||||
"metadata_oclaw": {"type": "object"},
|
||||
"slug": {"type": "string"},
|
||||
"provider": {"type": "string"},
|
||||
"version": {"type": "string"},
|
||||
"archive_url": {"type": "string"},
|
||||
"overwrite": {"type": "boolean"},
|
||||
},
|
||||
"required": ["name", "description"],
|
||||
"required": [],
|
||||
}
|
||||
return [
|
||||
ToolSpec(
|
||||
|
|
@ -167,7 +234,7 @@ def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]:
|
|||
handler=_auto_install_skill_handler,
|
||||
tags=frozenset({"skill", "oclaw", "installer"}),
|
||||
risk_level="high",
|
||||
timeout_s=20.0,
|
||||
timeout_s=120.0,
|
||||
),
|
||||
ToolSpec(
|
||||
name="skill_list",
|
||||
|
|
@ -199,6 +266,9 @@ def materialize_tool_specs(
|
|||
_ = factories
|
||||
collected: list[tuple[str, ToolSpec]] = []
|
||||
|
||||
def _hidden_from_model(name: str) -> bool:
|
||||
return str(name or "").strip() in _MODEL_TOOLS_DENYLIST
|
||||
|
||||
def _risk_allowed(spec: ToolSpec) -> bool:
|
||||
# Optional safety gate for public tools.
|
||||
# Default: only allow low risk public tools to be visible to all roles.
|
||||
|
|
@ -213,6 +283,9 @@ def materialize_tool_specs(
|
|||
for spec in list(materialize_public_tools()):
|
||||
if not isinstance(spec, ToolSpec):
|
||||
continue
|
||||
if _hidden_from_model(str(spec.name or "")):
|
||||
logger.info("public tool hidden from model registry: %s", str(spec.name or ""))
|
||||
continue
|
||||
if not _risk_allowed(spec):
|
||||
logger.warning("public tool blocked by risk gate: %s", str(spec.name or ""))
|
||||
continue
|
||||
|
|
@ -225,6 +298,9 @@ def materialize_tool_specs(
|
|||
for spec in materialize_tools_for_expert(str(expert or "").strip() or None):
|
||||
if not isinstance(spec, ToolSpec):
|
||||
continue
|
||||
if _hidden_from_model(str(spec.name or "")):
|
||||
logger.info("expert tool hidden from model registry: %s", str(spec.name or ""))
|
||||
continue
|
||||
collected.append(("expert", spec))
|
||||
except Exception as exc:
|
||||
logger.warning("expert tool load skipped: %s", exc)
|
||||
|
|
@ -235,6 +311,9 @@ def materialize_tool_specs(
|
|||
for spec in materialize_executable_skill_tools(store=store):
|
||||
if not isinstance(spec, ToolSpec):
|
||||
continue
|
||||
if _hidden_from_model(str(spec.name or "")):
|
||||
logger.info("skill runtime tool hidden from model registry: %s", str(spec.name or ""))
|
||||
continue
|
||||
collected.append(("skill_runtime", spec))
|
||||
except Exception as exc:
|
||||
logger.warning("skill runtime tool load skipped: %s", exc)
|
||||
|
|
@ -260,6 +339,9 @@ def materialize_tool_specs(
|
|||
path_policy_user_id=path_policy_user_id,
|
||||
):
|
||||
if isinstance(spec, ToolSpec):
|
||||
if _hidden_from_model(str(spec.name or "")):
|
||||
logger.info("mcp tool hidden from model registry: %s", str(spec.name or ""))
|
||||
continue
|
||||
collected.append(("mcp", spec))
|
||||
except Exception as exc:
|
||||
logger.warning("mcp tool load skipped: %s", exc)
|
||||
|
|
@ -293,6 +375,9 @@ def materialize_tool_specs(
|
|||
continue
|
||||
tags_raw = row.get("tags")
|
||||
tags = frozenset(str(x).strip() for x in (tags_raw or []) if str(x).strip())
|
||||
if _hidden_from_model(name):
|
||||
logger.info("plugin tool hidden from model registry: %s", name)
|
||||
continue
|
||||
collected.append((
|
||||
"plugin",
|
||||
ToolSpec(
|
||||
|
|
|
|||
|
|
@ -172,6 +172,36 @@ def run_command_tool() -> ToolSpec:
|
|||
command, normalized_cd_removed = _strip_leading_cd_chain(command)
|
||||
command, command_rewritten = _rewrite_workspace_absolute_refs(command, workdir=workdir)
|
||||
command, script_path_rewritten = _rewrite_python_script_arg(command, workdir=workdir)
|
||||
|
||||
def _external_skill_install_cli_blocked(raw_cmd: str) -> bool:
|
||||
s = str(raw_cmd or "").strip()
|
||||
if not s:
|
||||
return False
|
||||
low = s.lower()
|
||||
if re.match(r"^\s*cocoloop(?:\.cmd|\.exe)?\s+install(?:\s|$)", s, flags=re.IGNORECASE):
|
||||
return True
|
||||
if re.match(r"^\s*clawhub(?:\.cmd|\.exe)?\s+install(?:\s|$)", s, flags=re.IGNORECASE):
|
||||
return True
|
||||
if re.search(r"\bnpx\b", low) and "clawhub" in low:
|
||||
return True
|
||||
if re.match(r"^\s*npm(?:\.cmd|\.exe)?\s+install\b", low) and "clawhub" in low:
|
||||
return True
|
||||
return False
|
||||
|
||||
if _external_skill_install_cli_blocked(str(command or "")):
|
||||
return {
|
||||
"ok": False,
|
||||
"error_code": "skill_install_cli_blocked",
|
||||
"error": "skill_install_cli_blocked",
|
||||
"hint": "Oclaw has no shell skill installer. Use Admin POST /admin/api/skills/market/install or install-registry, or skill_auto_install.",
|
||||
"command": command,
|
||||
"cwd": str(workdir),
|
||||
"normalized_cd_removed": bool(normalized_cd_removed),
|
||||
"cwd_redirected_to_sandbox": bool(cwd_redirected_to_sandbox),
|
||||
"command_rewritten": bool(command_rewritten),
|
||||
"script_path_rewritten": bool(script_path_rewritten),
|
||||
"original_command": original_command,
|
||||
}
|
||||
try:
|
||||
os.makedirs(workdir, exist_ok=True)
|
||||
except Exception:
|
||||
|
|
|
|||
141
runtime/tools/public/skills_install_tool.py
Normal file
141
runtime/tools/public/skills_install_tool.py
Normal file
|
|
@ -0,0 +1,141 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.config.paths import db_path
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
from oclaw.runtime.skill_installer import install_skill_from_registry_archive
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def _store() -> SqliteStore:
|
||||
return SqliteStore(db_path())
|
||||
|
||||
|
||||
def _agent_workspace_skills_root() -> Path:
|
||||
# Agent-origin installs are isolated under _workspace lane.
|
||||
return default_skills_root() / "_workspace"
|
||||
|
||||
|
||||
def skill_market_install_tool() -> ToolSpec:
|
||||
def _handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = args if isinstance(args, dict) else {}
|
||||
slug = str(payload.get("slug") or "").strip()
|
||||
if not slug:
|
||||
return {"ok": False, "error_code": "slug_required", "error": "slug_required"}
|
||||
version = str(payload.get("version") or "").strip() or None
|
||||
overwrite = bool(payload.get("overwrite"))
|
||||
store = _store()
|
||||
provider_arg = str(payload.get("provider") or "").strip()
|
||||
if provider_arg:
|
||||
provider = normalize_skill_market_provider_setting(provider_arg)
|
||||
else:
|
||||
provider = normalize_skill_market_provider_setting(str(store.get_setting("AIA_SKILL_MARKET_PROVIDER") or ""))
|
||||
try:
|
||||
adapter = get_market_adapter(provider)
|
||||
archive_url, chosen_version = adapter.resolve_archive_url(slug=slug, version=version)
|
||||
except Exception as exc:
|
||||
return {
|
||||
"ok": False,
|
||||
"error_code": "market_resolve_failed",
|
||||
"error": f"market_resolve_failed:{type(exc).__name__}",
|
||||
"provider": provider,
|
||||
"slug": slug,
|
||||
}
|
||||
if not str(archive_url or "").strip():
|
||||
return {
|
||||
"ok": False,
|
||||
"error_code": "archive_url_unavailable",
|
||||
"error": "archive_url_unavailable",
|
||||
"provider": provider,
|
||||
"slug": slug,
|
||||
}
|
||||
out = install_skill_from_registry_archive(
|
||||
store=store,
|
||||
archive_url=str(archive_url),
|
||||
overwrite=overwrite,
|
||||
skills_root=_agent_workspace_skills_root(),
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"result": {
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
},
|
||||
"provider": provider,
|
||||
"slug": slug,
|
||||
"version": str(chosen_version or version or ""),
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="skill_market_install",
|
||||
description="Install a skill from configured market by slug/version.",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"slug": {"type": "string"},
|
||||
"provider": {"type": "string", "description": "Optional provider override: clawhub or cocoloop."},
|
||||
"version": {"type": "string"},
|
||||
"overwrite": {"type": "boolean"},
|
||||
},
|
||||
"required": ["slug"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=_handler,
|
||||
tags=frozenset({"skill", "installer", "market"}),
|
||||
risk_level="medium",
|
||||
timeout_s=120.0,
|
||||
)
|
||||
|
||||
|
||||
def skill_registry_install_tool() -> ToolSpec:
|
||||
def _handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = args if isinstance(args, dict) else {}
|
||||
archive_url = str(payload.get("archive_url") or "").strip()
|
||||
if not archive_url:
|
||||
return {"ok": False, "error_code": "archive_url_required", "error": "archive_url_required"}
|
||||
overwrite = bool(payload.get("overwrite"))
|
||||
out = install_skill_from_registry_archive(
|
||||
store=_store(),
|
||||
archive_url=archive_url,
|
||||
overwrite=overwrite,
|
||||
skills_root=_agent_workspace_skills_root(),
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"result": {
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
},
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="skill_registry_install",
|
||||
description="Install a skill from archive URL (registry/market artifact).",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"archive_url": {"type": "string"},
|
||||
"overwrite": {"type": "boolean"},
|
||||
},
|
||||
"required": ["archive_url"],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=_handler,
|
||||
tags=frozenset({"skill", "installer", "registry"}),
|
||||
risk_level="medium",
|
||||
timeout_s=120.0,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["skill_market_install_tool", "skill_registry_install_tool"]
|
||||
|
||||
187
runtime/tools/skills/cocoloop_client.py
Normal file
187
runtime/tools/skills/cocoloop_client.py
Normal file
|
|
@ -0,0 +1,187 @@
|
|||
"""CocoLoop 技能商店 HTTP 客户端(与 ClawHub 并列,供 `skills_market` 使用)。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import httpx
|
||||
|
||||
|
||||
def _strip_trailing_slash(url: str) -> str:
|
||||
return str(url or "").strip().rstrip("/")
|
||||
|
||||
|
||||
def _join_url(base: str, path: str) -> str:
|
||||
b = _strip_trailing_slash(base)
|
||||
p = str(path or "").strip()
|
||||
if not p:
|
||||
return b
|
||||
if not p.startswith("/"):
|
||||
p = "/" + p
|
||||
return b + p
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CocoloopConfig:
|
||||
api_base_url: str = "https://api.cocoloop.com"
|
||||
|
||||
|
||||
def load_cocoloop_config() -> CocoloopConfig:
|
||||
base = str(os.getenv("AIA_COCOLOOP_API_BASE") or os.getenv("COCOLOOP_API_BASE") or "https://api.cocoloop.com").strip()
|
||||
return CocoloopConfig(api_base_url=_strip_trailing_slash(base))
|
||||
|
||||
|
||||
def _default_headers() -> dict[str, str]:
|
||||
return {
|
||||
"User-Agent": "Oclaw-SkillMarket/1.0 (+https://github.com/oclaw)",
|
||||
"Accept": "application/json",
|
||||
}
|
||||
|
||||
|
||||
def _get_json(url: str, *, params: dict[str, Any] | None = None) -> dict[str, Any]:
|
||||
try:
|
||||
with httpx.Client(timeout=12.0, follow_redirects=True) as c:
|
||||
r = c.get(url, params=params or {}, headers=_default_headers())
|
||||
if r.status_code != 200:
|
||||
return {}
|
||||
obj = r.json()
|
||||
return obj if isinstance(obj, dict) else {}
|
||||
except Exception:
|
||||
return {}
|
||||
|
||||
|
||||
def _list_items(cfg: CocoloopConfig, *, keyword: str, page: int, page_size: int) -> list[dict[str, Any]]:
|
||||
url = _join_url(cfg.api_base_url, "/api/v1/store/skills")
|
||||
blob = _get_json(
|
||||
url,
|
||||
params={
|
||||
"page": max(1, int(page)),
|
||||
"page_size": max(1, min(int(page_size), 100)),
|
||||
"keyword": str(keyword or "").strip(),
|
||||
"sort": "downloads",
|
||||
},
|
||||
)
|
||||
data = blob.get("data") if isinstance(blob.get("data"), dict) else {}
|
||||
items = data.get("items")
|
||||
if not isinstance(items, list):
|
||||
return []
|
||||
return [x for x in items if isinstance(x, dict)]
|
||||
|
||||
|
||||
def _normalize_list_row(raw: dict[str, Any]) -> dict[str, Any]:
|
||||
slug = str(raw.get("name") or "").strip()
|
||||
dl = str(raw.get("download_url") or "").strip()
|
||||
ver = str(raw.get("version") or "").strip() or "latest"
|
||||
return {
|
||||
"source": "cocoloop",
|
||||
"slug": slug,
|
||||
"name": str(raw.get("subtitle") or raw.get("summary") or slug),
|
||||
"description": str(raw.get("brief") or raw.get("summary") or raw.get("original_desc") or ""),
|
||||
"version": ver,
|
||||
"owner": str(raw.get("author") or ""),
|
||||
"updatedAt": "",
|
||||
"downloads": _parse_count(raw.get("downloads")),
|
||||
"stars": _parse_count(raw.get("github_stars")),
|
||||
"homepage": f"https://hub.cocoloop.cn/skills/{raw.get('id')}" if raw.get("id") else "",
|
||||
"archiveUrl": dl,
|
||||
"raw": raw,
|
||||
}
|
||||
|
||||
|
||||
def _parse_count(v: Any) -> int:
|
||||
if isinstance(v, int):
|
||||
return v
|
||||
s = str(v or "").strip().lower().replace(",", "")
|
||||
if not s:
|
||||
return 0
|
||||
mult = 1
|
||||
if s.endswith("k"):
|
||||
mult = 1000
|
||||
s = s[:-1]
|
||||
if s.endswith("m"):
|
||||
mult = 1_000_000
|
||||
s = s[:-1]
|
||||
try:
|
||||
return int(float(s) * mult)
|
||||
except ValueError:
|
||||
return 0
|
||||
|
||||
|
||||
def search_store_skills(query: str, *, limit: int = 20, cfg: CocoloopConfig | None = None) -> list[dict[str, Any]]:
|
||||
cfg = cfg or load_cocoloop_config()
|
||||
lim = max(1, min(int(limit or 20), 100))
|
||||
rows = _list_items(cfg, keyword=str(query or "").strip(), page=1, page_size=lim)
|
||||
return [_normalize_list_row(r) for r in rows if str(r.get("name") or "").strip()]
|
||||
|
||||
|
||||
def get_skill_detail_by_slug(slug: str, *, cfg: CocoloopConfig | None = None) -> dict[str, Any]:
|
||||
"""按商店 `name`(slug)解析技能;必要时用数字 id 直查。"""
|
||||
cfg = cfg or load_cocoloop_config()
|
||||
s = str(slug or "").strip()
|
||||
if not s:
|
||||
return {}
|
||||
if s.isdigit():
|
||||
return _detail_from_id(cfg, int(s))
|
||||
rows = _list_items(cfg, keyword=s, page=1, page_size=80)
|
||||
want = s.lower()
|
||||
hit: dict[str, Any] | None = None
|
||||
for r in rows:
|
||||
if str(r.get("name") or "").strip().lower() == want:
|
||||
hit = r
|
||||
break
|
||||
if hit is None:
|
||||
for r in rows:
|
||||
nm = str(r.get("name") or "").strip().lower()
|
||||
if want in nm or nm in want:
|
||||
hit = r
|
||||
break
|
||||
if hit is None:
|
||||
return {"slug": s, "source": "cocoloop"}
|
||||
return _detail_from_list_row(cfg, hit)
|
||||
|
||||
|
||||
def _detail_from_id(cfg: CocoloopConfig, skill_id: int) -> dict[str, Any]:
|
||||
url = _join_url(cfg.api_base_url, f"/api/v1/store/skills/{int(skill_id)}")
|
||||
blob = _get_json(url)
|
||||
data = blob.get("data") if isinstance(blob.get("data"), dict) else {}
|
||||
if not data:
|
||||
return {"slug": str(skill_id), "source": "cocoloop"}
|
||||
return _detail_from_list_row(cfg, data)
|
||||
|
||||
|
||||
def _detail_from_list_row(cfg: CocoloopConfig, row: dict[str, Any]) -> dict[str, Any]:
|
||||
slug = str(row.get("name") or "").strip()
|
||||
dl = str(row.get("download_url") or "").strip()
|
||||
if not dl and slug:
|
||||
asset = str(row.get("asset_name") or f"{slug}.zip").strip()
|
||||
if not asset.endswith(".zip"):
|
||||
asset = f"{asset}.zip"
|
||||
dl = f"https://dl.cocoloop.cn/bss/skills/{asset.lstrip('/')}"
|
||||
ver = str(row.get("version") or "").strip() or "latest"
|
||||
ver_clean = ver.lstrip("vV") if ver not in {"", "latest"} else ver
|
||||
versions: list[dict[str, Any]] = [{"version": ver_clean or "latest", "changelog": "", "createdAt": "", "archiveUrl": dl, "raw": row}]
|
||||
return {
|
||||
"source": "cocoloop",
|
||||
"slug": slug,
|
||||
"name": str(row.get("subtitle") or row.get("summary") or slug),
|
||||
"description": str(row.get("brief") or row.get("summary") or row.get("original_desc") or ""),
|
||||
"owner": str(row.get("author") or ""),
|
||||
"updatedAt": "",
|
||||
"homepage": f"https://hub.cocoloop.cn/skills/{row.get('id')}" if row.get("id") else "",
|
||||
"latestVersion": ver_clean if ver_clean else "latest",
|
||||
"archiveUrl": dl,
|
||||
"downloads": _parse_count(row.get("downloads")),
|
||||
"stars": _parse_count(row.get("github_stars")),
|
||||
"versions": versions,
|
||||
"raw": row,
|
||||
}
|
||||
|
||||
|
||||
__all__ = [
|
||||
"CocoloopConfig",
|
||||
"load_cocoloop_config",
|
||||
"search_store_skills",
|
||||
"get_skill_detail_by_slug",
|
||||
]
|
||||
|
|
@ -140,9 +140,11 @@ class AdminSkillsApiTests(unittest.TestCase):
|
|||
self.assertEqual(g.status_code, 200, g.text)
|
||||
gb = g.json() or {}
|
||||
self.assertTrue(gb.get("ok"))
|
||||
self.assertIn("market_provider", gb)
|
||||
self.assertIn(str(gb.get("market_provider") or ""), {"clawhub", "cocoloop"})
|
||||
s = self.client.post(
|
||||
"/admin/api/skills/mode",
|
||||
json={"prompt_in_system": True, "toolcall_enabled": False},
|
||||
json={"prompt_in_system": True, "toolcall_enabled": False, "market_provider": "cocoloop"},
|
||||
headers=self._h(),
|
||||
)
|
||||
self.assertEqual(s.status_code, 200, s.text)
|
||||
|
|
@ -150,6 +152,9 @@ class AdminSkillsApiTests(unittest.TestCase):
|
|||
self.assertTrue(sb.get("ok"))
|
||||
self.assertTrue(bool(sb.get("prompt_in_system")))
|
||||
self.assertFalse(bool(sb.get("toolcall_enabled")))
|
||||
self.assertEqual(str(sb.get("market_provider") or ""), "cocoloop")
|
||||
g2 = self.client.get("/admin/api/skills/mode", headers=self._h())
|
||||
self.assertEqual((g2.json() or {}).get("market_provider"), "cocoloop")
|
||||
|
||||
def test_skills_effective_dashboard(self) -> None:
|
||||
c = self.client.post(
|
||||
|
|
|
|||
|
|
@ -2,6 +2,7 @@ from __future__ import annotations
|
|||
|
||||
from pathlib import Path
|
||||
import zipfile
|
||||
import subprocess
|
||||
|
||||
from oclaw.runtime.skill_installer import (
|
||||
auto_install_skill_from_payload,
|
||||
|
|
@ -9,6 +10,7 @@ from oclaw.runtime.skill_installer import (
|
|||
install_skill_from_local_dir,
|
||||
install_skill_from_registry_archive,
|
||||
list_skills_with_status,
|
||||
repair_skill_dependencies,
|
||||
set_skill_enabled,
|
||||
)
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
|
|
@ -102,6 +104,33 @@ def test_install_skill_from_registry_archive_file_url(tmp_path: Path) -> None:
|
|||
assert out.name == "reg_demo"
|
||||
|
||||
|
||||
def test_install_skill_from_registry_archive_workspace_auto_bind(tmp_path: Path) -> None:
|
||||
db = tmp_path / "ops.sqlite"
|
||||
store = SqliteStore(str(db))
|
||||
pkg_dir = tmp_path / "pkg_ws"
|
||||
inner = pkg_dir / "demo"
|
||||
inner.mkdir(parents=True, exist_ok=True)
|
||||
(inner / "SKILL.md").write_text(
|
||||
"---\nname: reg_ws_demo\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
archive = tmp_path / "reg_ws.zip"
|
||||
with zipfile.ZipFile(archive, "w") as zf:
|
||||
zf.write(inner / "SKILL.md", arcname="demo/SKILL.md")
|
||||
root = tmp_path / "skills"
|
||||
out = install_skill_from_registry_archive(
|
||||
store=store,
|
||||
archive_url=archive.resolve().as_uri(),
|
||||
skills_root=root / "_workspace",
|
||||
auto_bind=True,
|
||||
)
|
||||
assert out.ok
|
||||
assert out.name == "reg_ws_demo"
|
||||
assert out.auto_enabled is True
|
||||
assert len(out.binding_applied_roles) >= 1
|
||||
assert (root / "_workspace" / "reg_ws_demo" / "SKILL.md").exists()
|
||||
|
||||
|
||||
def test_install_skill_from_clawhub_page_url(tmp_path: Path, monkeypatch) -> None:
|
||||
db = tmp_path / "ops.sqlite"
|
||||
store = SqliteStore(str(db))
|
||||
|
|
@ -149,3 +178,103 @@ def test_install_local_allows_sh_files(tmp_path: Path) -> None:
|
|||
assert out.ok
|
||||
assert out.name == "local_with_sh"
|
||||
|
||||
|
||||
def test_install_local_auto_installs_python_requirements(tmp_path: Path, monkeypatch) -> None:
|
||||
db = tmp_path / "ops.sqlite"
|
||||
store = SqliteStore(str(db))
|
||||
src = tmp_path / "skill_with_reqs"
|
||||
src.mkdir(parents=True, exist_ok=True)
|
||||
(src / "SKILL.md").write_text(
|
||||
"---\nname: with_reqs\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(src / "requirements.txt").write_text("requests>=2.0.0\n", encoding="utf-8")
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def _mock_run(cmd, **kwargs): # noqa: ANN001
|
||||
calls.append([str(x) for x in cmd])
|
||||
return subprocess.CompletedProcess(args=cmd, returncode=0, stdout="", stderr="")
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run)
|
||||
out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills")
|
||||
assert out.ok
|
||||
assert any(("pip" in " ".join(c) and "-r" in c) for c in calls)
|
||||
|
||||
|
||||
def test_install_local_dependency_install_failure_returns_warning(tmp_path: Path, monkeypatch) -> None:
|
||||
db = tmp_path / "ops.sqlite"
|
||||
store = SqliteStore(str(db))
|
||||
src = tmp_path / "skill_with_bad_reqs"
|
||||
src.mkdir(parents=True, exist_ok=True)
|
||||
(src / "SKILL.md").write_text(
|
||||
"---\nname: with_bad_reqs\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(src / "requirements.txt").write_text("not_a_real_pkg_zzz\n", encoding="utf-8")
|
||||
|
||||
def _mock_run(cmd, **kwargs): # noqa: ANN001,ARG001
|
||||
return subprocess.CompletedProcess(args=cmd, returncode=1, stdout="", stderr="install failed")
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run)
|
||||
out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills")
|
||||
assert out.ok
|
||||
assert out.detail.startswith("installed_with_dependency_warnings:")
|
||||
|
||||
|
||||
def test_install_local_probe_missing_imports_and_install(tmp_path: Path, monkeypatch) -> None:
|
||||
db = tmp_path / "ops.sqlite"
|
||||
store = SqliteStore(str(db))
|
||||
src = tmp_path / "skill_probe_imports"
|
||||
src.mkdir(parents=True, exist_ok=True)
|
||||
(src / "SKILL.md").write_text(
|
||||
"---\nname: probe_imports\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(src / "main.py").write_text(
|
||||
"import json\nimport office\nimport pandas\nimport totally_missing_pkg_xyz\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
office_dir = src / "office"
|
||||
office_dir.mkdir(parents=True, exist_ok=True)
|
||||
(office_dir / "__init__.py").write_text("", encoding="utf-8")
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def _mock_run(cmd, **kwargs): # noqa: ANN001,ARG001
|
||||
calls.append([str(x) for x in cmd])
|
||||
return subprocess.CompletedProcess(args=cmd, returncode=0, stdout="", stderr="")
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run)
|
||||
out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills")
|
||||
assert out.ok
|
||||
pip_calls = [c for c in calls if ("pip" in " ".join(c))]
|
||||
assert pip_calls
|
||||
assert any("totally_missing_pkg_xyz" in c for c in pip_calls)
|
||||
|
||||
|
||||
def test_repair_skill_dependencies_for_installed_skill(tmp_path: Path, monkeypatch) -> None:
|
||||
db = tmp_path / "ops.sqlite"
|
||||
store = SqliteStore(str(db))
|
||||
src = tmp_path / "skill_repair"
|
||||
src.mkdir(parents=True, exist_ok=True)
|
||||
(src / "SKILL.md").write_text(
|
||||
"---\nname: skill_repair\ndescription: x\nmetadata: {\"oclaw\":{}}\n---\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(src / "main.py").write_text("import definitely_missing_pkg_abc\n", encoding="utf-8")
|
||||
|
||||
calls: list[list[str]] = []
|
||||
|
||||
def _mock_run(cmd, **kwargs): # noqa: ANN001,ARG001
|
||||
calls.append([str(x) for x in cmd])
|
||||
return subprocess.CompletedProcess(args=cmd, returncode=0, stdout="", stderr="")
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.skill_installer.subprocess.run", _mock_run)
|
||||
out = install_skill_from_local_dir(store=store, source_dir=src, skills_root=tmp_path / "skills")
|
||||
assert out.ok
|
||||
calls.clear()
|
||||
result = repair_skill_dependencies(store=store, skill_name="skill_repair", skills_root=tmp_path / "skills")
|
||||
assert bool(result.get("ok")) is True
|
||||
assert any("definitely_missing_pkg_abc" in c for c in calls)
|
||||
|
||||
|
|
|
|||
10
tests/test_skills_install_public_tool_visibility.py
Normal file
10
tests/test_skills_install_public_tool_visibility.py
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from oclaw.runtime.tools.catalog import default_registry
|
||||
|
||||
|
||||
def test_skill_install_public_tools_hidden_for_specialist_auto_only() -> None:
|
||||
names = [t.name for t in default_registry(expert="network_ops+memory", specialist="ops").list()]
|
||||
assert "skill_market_install" not in names
|
||||
assert "skill_registry_install" not in names
|
||||
|
||||
83
tests/test_skills_install_tool_workspace_root.py
Normal file
83
tests/test_skills_install_tool_workspace_root.py
Normal file
|
|
@ -0,0 +1,83 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
from oclaw.runtime.tools.public.skills_install_tool import skill_market_install_tool, skill_registry_install_tool
|
||||
|
||||
|
||||
def test_skill_registry_install_tool_forces_workspace_root(monkeypatch, tmp_path: Path) -> None:
|
||||
captured: dict[str, str] = {}
|
||||
|
||||
def _mock_default_root() -> Path:
|
||||
return tmp_path / "skills"
|
||||
|
||||
def _mock_install(**kwargs): # noqa: ANN003
|
||||
captured["skills_root"] = str(kwargs.get("skills_root") or "")
|
||||
|
||||
class _Out:
|
||||
ok = True
|
||||
name = "demo"
|
||||
target_dir = str((tmp_path / "skills" / "_workspace" / "demo").resolve())
|
||||
detail = "installed"
|
||||
error_code = "ok"
|
||||
retryable = False
|
||||
|
||||
return _Out()
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.default_skills_root", _mock_default_root)
|
||||
monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.install_skill_from_registry_archive", _mock_install)
|
||||
tool = skill_registry_install_tool()
|
||||
result = tool.handler({"archive_url": "https://example.com/demo.zip"})
|
||||
assert bool(result.get("ok")) is True
|
||||
assert captured["skills_root"].replace("\\", "/").endswith("/skills/_workspace")
|
||||
|
||||
|
||||
def test_skill_market_install_tool_provider_arg_overrides_setting(monkeypatch, tmp_path: Path) -> None:
|
||||
captured: dict[str, str] = {}
|
||||
|
||||
class _FakeStore:
|
||||
def get_setting(self, key: str) -> str:
|
||||
if key == "AIA_SKILL_MARKET_PROVIDER":
|
||||
return "clawhub"
|
||||
return ""
|
||||
|
||||
class _FakeAdapter:
|
||||
def resolve_archive_url(self, *, slug: str, version: str | None = None) -> tuple[str, str]:
|
||||
captured["slug"] = slug
|
||||
captured["version"] = str(version or "")
|
||||
return "https://example.com/demo.zip", "1.0.0"
|
||||
|
||||
def _mock_store() -> _FakeStore:
|
||||
return _FakeStore()
|
||||
|
||||
def _mock_default_root() -> Path:
|
||||
return tmp_path / "skills"
|
||||
|
||||
def _mock_get_market_adapter(provider: str): # noqa: ANN001
|
||||
captured["provider"] = provider
|
||||
return _FakeAdapter()
|
||||
|
||||
def _mock_install(**kwargs): # noqa: ANN003
|
||||
captured["skills_root"] = str(kwargs.get("skills_root") or "")
|
||||
|
||||
class _Out:
|
||||
ok = True
|
||||
name = "demo"
|
||||
target_dir = str((tmp_path / "skills" / "_workspace" / "demo").resolve())
|
||||
detail = "installed"
|
||||
error_code = "ok"
|
||||
retryable = False
|
||||
|
||||
return _Out()
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool._store", _mock_store)
|
||||
monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.default_skills_root", _mock_default_root)
|
||||
monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.get_market_adapter", _mock_get_market_adapter)
|
||||
monkeypatch.setattr("oclaw.runtime.tools.public.skills_install_tool.install_skill_from_registry_archive", _mock_install)
|
||||
|
||||
tool = skill_market_install_tool()
|
||||
result = tool.handler({"slug": "demo", "provider": "cocoloop", "version": "latest"})
|
||||
assert bool(result.get("ok")) is True
|
||||
assert captured["provider"] == "cocoloop"
|
||||
assert captured["skills_root"].replace("\\", "/").endswith("/skills/_workspace")
|
||||
|
||||
35
tests/test_skills_market_providers.py
Normal file
35
tests/test_skills_market_providers.py
Normal file
|
|
@ -0,0 +1,35 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from oclaw.runtime import skills_market
|
||||
|
||||
|
||||
def test_get_market_adapter_clawhub_default() -> None:
|
||||
a = skills_market.get_market_adapter("clawhub")
|
||||
assert a.provider == "clawhub"
|
||||
|
||||
|
||||
def test_get_market_adapter_cocoloop() -> None:
|
||||
a = skills_market.get_market_adapter("cocoloop")
|
||||
assert a.provider == "cocoloop"
|
||||
|
||||
|
||||
def test_get_market_adapter_cocoloop_alias() -> None:
|
||||
a = skills_market.get_market_adapter("cocoloop-cn")
|
||||
assert a.provider == "cocoloop"
|
||||
|
||||
|
||||
def test_cocoloop_resolve_archive_url(monkeypatch) -> None:
|
||||
def _fake_detail(slug: str) -> dict: # noqa: ANN001
|
||||
return {
|
||||
"source": "cocoloop",
|
||||
"slug": slug,
|
||||
"latestVersion": "1.0.0",
|
||||
"archiveUrl": "https://dl.example/bss/skills/demo.zip",
|
||||
"versions": [{"version": "1.0.0", "archiveUrl": "https://dl.example/bss/skills/demo.zip"}],
|
||||
}
|
||||
|
||||
monkeypatch.setattr("oclaw.runtime.skills_market.cocoloop_get_skill_detail", _fake_detail)
|
||||
a = skills_market.CocoloopMarketAdapter()
|
||||
url, ver = a.resolve_archive_url(slug="demo", version=None)
|
||||
assert url.endswith("demo.zip")
|
||||
assert ver == "1.0.0"
|
||||
|
|
@ -41,11 +41,50 @@ def test_tool_loop_guard_blocks_repeated_signature(tmp_path: Path) -> None:
|
|||
tool_uses=tool_uses,
|
||||
signature_budget=2,
|
||||
)
|
||||
assert calls["n"] == 2
|
||||
# Same-round duplicate calls now hit cache; only the first executes.
|
||||
assert calls["n"] == 1
|
||||
second, _ = results["c2"]
|
||||
assert bool(second.get("ok")) is True
|
||||
blocked, _ = results["c3"]
|
||||
assert blocked.get("error_code") == "tool_loop_guard"
|
||||
|
||||
|
||||
def test_same_round_duplicate_tool_call_reuses_cached_result(tmp_path: Path) -> None:
|
||||
store = SqliteStore(str(tmp_path / "dup.sqlite"))
|
||||
sess = store.create_session("t")
|
||||
calls = {"n": 0}
|
||||
|
||||
def _handler(args):
|
||||
calls["n"] += 1
|
||||
return {"ok": True, "echo": args, "counter": calls["n"]}
|
||||
|
||||
reg = ToolRegistry(
|
||||
[
|
||||
ToolSpec(
|
||||
name="echo",
|
||||
description="echo",
|
||||
parameters={"type": "object", "properties": {"x": {"type": "integer"}}},
|
||||
handler=_handler,
|
||||
read_only=True,
|
||||
)
|
||||
]
|
||||
)
|
||||
tool_uses = [
|
||||
LLMToolCall(id="c1", name="echo", arguments={"x": 1}),
|
||||
LLMToolCall(id="c2", name="echo", arguments={"x": 1}),
|
||||
]
|
||||
_, results = ToolExecutor().execute_tool_uses(
|
||||
ctx=ToolExecutionContext(store=store, tools=reg, session_id=sess.id),
|
||||
assistant_msg_id=1,
|
||||
tool_uses=tool_uses,
|
||||
signature_budget=2,
|
||||
)
|
||||
assert calls["n"] == 1
|
||||
r1, _ = results["c1"]
|
||||
r2, _ = results["c2"]
|
||||
assert r1 == r2
|
||||
|
||||
|
||||
def test_repeated_tool_results_are_compacted_in_history(tmp_path: Path) -> None:
|
||||
store = SqliteStore(str(tmp_path / "g2.sqlite"))
|
||||
sess = store.create_session("t")
|
||||
|
|
|
|||
|
|
@ -256,6 +256,43 @@ class WorkspacePathGuardTests(unittest.TestCase):
|
|||
self.assertEqual(str(r.get("error_code") or ""), "command_exit_nonzero")
|
||||
self.assertFalse(bool(r.get("output_truncated")))
|
||||
|
||||
def test_run_command_blocks_cocoloop_install_cli(self) -> None:
|
||||
with mock.patch.dict(
|
||||
os.environ,
|
||||
{
|
||||
"OPS_WORKSPACE_ROOT": str(self.root),
|
||||
"OPS_WORKSPACE_EXTRA_ROOTS": "",
|
||||
"OPS_WORKSPACE_ALLOW_ANY_PATH": "",
|
||||
"AIA_ENABLE_RUN_COMMAND": "1",
|
||||
},
|
||||
clear=False,
|
||||
):
|
||||
clear_workspace_path_access_for_tests()
|
||||
spec = run_command_tool()
|
||||
with workspace_path_access_scope(None, None):
|
||||
r = spec.handler({"command": "cocoloop install 7288"})
|
||||
self.assertFalse(bool(r.get("ok")), r)
|
||||
self.assertEqual("skill_install_cli_blocked", str(r.get("error_code") or ""))
|
||||
self.assertIn("market/install", str(r.get("hint") or ""))
|
||||
|
||||
def test_run_command_blocks_npx_clawhub_install_pattern(self) -> None:
|
||||
with mock.patch.dict(
|
||||
os.environ,
|
||||
{
|
||||
"OPS_WORKSPACE_ROOT": str(self.root),
|
||||
"OPS_WORKSPACE_EXTRA_ROOTS": "",
|
||||
"OPS_WORKSPACE_ALLOW_ANY_PATH": "",
|
||||
"AIA_ENABLE_RUN_COMMAND": "1",
|
||||
},
|
||||
clear=False,
|
||||
):
|
||||
clear_workspace_path_access_for_tests()
|
||||
spec = run_command_tool()
|
||||
with workspace_path_access_scope(None, None):
|
||||
r = spec.handler({"command": "npx -y clawhub@latest install foo"})
|
||||
self.assertFalse(bool(r.get("ok")), r)
|
||||
self.assertEqual("skill_install_cli_blocked", str(r.get("error_code") or ""))
|
||||
|
||||
def test_run_command_rewrites_workspace_absolute_script_path_to_sandbox(self) -> None:
|
||||
with mock.patch.dict(
|
||||
os.environ,
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue