mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 07:20:44 +08:00
统一技能安装链路并提交当前skills变更。
新增 public 的 skill_auto_install 工具并移除 catalog 旧实现,同时按当前工作区状态提交 skills 精简与 skill-creator 新增内容。 Made-with: Cursor
This commit is contained in:
parent
1c71a29ec1
commit
329fb2239a
45 changed files with 1291 additions and 6100 deletions
|
|
@ -1,97 +0,0 @@
|
|||
---
|
||||
name: SkillScan
|
||||
metadata:
|
||||
version: "1.1.6"
|
||||
description: >
|
||||
Security gate for skills. Every new skill MUST pass SkillScan before use.
|
||||
Activate on any install, load, add, evaluate, or safety question about a skill.
|
||||
On first load, run first-run to scan all existing skills.
|
||||
Blocks HIGH/CRITICAL skills. No exceptions.
|
||||
---
|
||||
|
||||
# SkillScan v1.1.6
|
||||
|
||||
Security check for skill packages. Every new skill must pass this scan before use.
|
||||
|
||||
Implemented in `scripts/scanner.py`. Supports Windows / macOS / Linux.
|
||||
|
||||
Do NOT generate report files inside the skill directory.
|
||||
|
||||
---
|
||||
|
||||
## When to Activate
|
||||
|
||||
Run SkillScan whenever a skill is installed, loaded, added, evaluated, or questioned for safety — regardless of method or phrasing.
|
||||
|
||||
Trigger examples: install, load, add, use, set up, "is this safe", upload .zip/.skill, paste SKILL.md, mention unknown skill name, install from hub/store.
|
||||
|
||||
---
|
||||
|
||||
## First Load
|
||||
|
||||
Check `scripts/.first_run_done`:
|
||||
|
||||
- **Not found** → **Tell user** SkillScan is installed, ask to scan all existing skills:
|
||||
```bash
|
||||
python <skill_scanner_dir>/scripts/scanner.py first-run
|
||||
```
|
||||
`.first_run_done` is created automatically after completion.
|
||||
|
||||
- **Found** → Skip, operate normally.
|
||||
|
||||
---
|
||||
|
||||
## Commands
|
||||
|
||||
| Command | Usage |
|
||||
|---------|-------|
|
||||
| `scanner.py scan <path>` | Scan a single skill (.zip or directory) |
|
||||
| `scanner.py scan-all` | Scan all installed skills |
|
||||
| `scanner.py first-run` | First-time full scan |
|
||||
| `scanner.py upgrade` | Manual upgrade |
|
||||
|
||||
---
|
||||
|
||||
## Scan Rules
|
||||
|
||||
- **.zip files** → Scan BEFORE installation. Block if fails.
|
||||
- **Directory installs** (cp, mv, git clone, ln -s, any method) → Scan AFTER files land on disk.
|
||||
- **Remote installs** (clawhub, skillhub, npx skills add, etc.) → Scan immediately after install.
|
||||
- **Unknown skills** → If user mentions a skill you haven't seen, scan it.
|
||||
|
||||
---
|
||||
|
||||
## Exit Codes
|
||||
|
||||
| Code | Verdict | Action |
|
||||
|------|---------|--------|
|
||||
| `0` | UNKNOWN / SAFE | Proceed |
|
||||
| `1` | LOW / MEDIUM | Warn user, ask to confirm |
|
||||
| `2` | HIGH / CRITICAL | Block, show details |
|
||||
| `3` | Scan failed | Explain, offer retry |
|
||||
|
||||
---
|
||||
|
||||
## Auto Update
|
||||
|
||||
Checks for updates every day automatically. Silent, no user action needed. Manual: `scanner.py upgrade`.
|
||||
|
||||
---
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `SKILL_SCANNER_UPDATE_URL` | Custom update source (optional) |
|
||||
|
||||
---
|
||||
|
||||
## API
|
||||
|
||||
Base URL: `https://skillscan.tokauth.com`
|
||||
|
||||
| Step | Method | Path |
|
||||
|------|--------|------|
|
||||
| ① Cache lookup | GET | `/oapi/v1/skill-scan/search?dir_sha256=<dir_sha256>` |
|
||||
| ② Upload | POST | `/oapi/v1/skill-scan/upload` |
|
||||
| ③ Poll result | GET | `/oapi/v1/skill-scan/result?task_no=<task_no>` (poll every 20s, max 180s) |
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
{
|
||||
"ownerId": "kn791cyx98pcsezkh5088g8jxn84c7mm",
|
||||
"slug": "skillscan",
|
||||
"version": "1.1.6",
|
||||
"publishedAt": 1776650587310
|
||||
}
|
||||
|
|
@ -1,959 +0,0 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
SkillScan v1.1.5 — OpenClaw Skill security scanner.
|
||||
Supports Windows / macOS / Linux. All temp files use the standard tempfile module.
|
||||
|
||||
Usage (invoked by the agent via bash):
|
||||
python scanner.py first-run # First install: list installed skills and ask to scan
|
||||
python scanner.py scan <path> # Scan a single skill (.zip or directory)
|
||||
python scanner.py scan-all # Scan all installed skills
|
||||
python scanner.py upgrade # Auto-upgrade
|
||||
"""
|
||||
|
||||
import sys, os, json, time, zipfile, hashlib, shutil, tempfile, uuid, platform, base64
|
||||
import urllib.request, urllib.error, urllib.parse
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Configuration
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
SCANNER_VERSION = "1.1.5"
|
||||
|
||||
BASE_URL = "https://skillscan.tokauth.com"
|
||||
API_SEARCH = f"{BASE_URL}/oapi/v1/skill-scan/search"
|
||||
API_UPLOAD = f"{BASE_URL}/oapi/v1/skill-scan/upload"
|
||||
API_RESULT = f"{BASE_URL}/oapi/v1/skill-scan/result"
|
||||
UPDATE_URL = os.environ.get("SKILL_SCANNER_UPDATE_URL",
|
||||
f"{BASE_URL}/downloads/SkillScan/manifest")
|
||||
|
||||
POLL_INTERVAL = 20 # Poll interval (seconds)
|
||||
POLL_TIMEOUT = 180 # Max wait time (seconds)
|
||||
|
||||
# First-run marker file (in the same directory as scanner.py)
|
||||
STATE_FILE = Path(__file__).parent / ".first_run_done"
|
||||
|
||||
# Auto-update check marker file and interval (7 days)
|
||||
LAST_UPDATE_CHECK_FILE = Path(__file__).parent / ".last_update_check"
|
||||
AUTO_UPDATE_INTERVAL = 1 * 24 * 3600 # 1 day (seconds)
|
||||
|
||||
# Client info file (generated on first run, reused afterwards)
|
||||
CLIENT_INFO_FILE = Path(__file__).parent / ".client_info"
|
||||
|
||||
# Files and directories to skip during scanning, hashing, and packing
|
||||
SKIP_FILES = {".first_run_done", ".last_update_check", ".client_info", "cloud_report.json", ".DS_Store"}
|
||||
SKIP_DIRS = {".git", "__pycache__", ".venv", "node_modules", ".idea", ".vscode", ".clawhub"}
|
||||
|
||||
# Resolve the root directory of SkillScan itself (parent of scripts/)
|
||||
SELF_ROOT = Path(__file__).parent.parent.resolve()
|
||||
|
||||
|
||||
# Skill installation paths (cross-platform)
|
||||
def skill_install_paths():
|
||||
# type: () -> list
|
||||
"""Auto-enumerate OpenClaw and local skill paths across platforms."""
|
||||
home = Path.home()
|
||||
oc_dir = home / ".openclaw"
|
||||
candidates = [
|
||||
# OpenClaw standard paths
|
||||
oc_dir / "skills",
|
||||
oc_dir / "workspace/skills",
|
||||
# Shared agent skill paths
|
||||
home / ".agents/skills",
|
||||
home / ".config/agents/skills",
|
||||
# Agent-specific global paths
|
||||
home / ".gemini/antigravity/skills",
|
||||
home / ".gemini/skills",
|
||||
home / ".augment/skills",
|
||||
home / ".claude/skills",
|
||||
home / ".codex/skills",
|
||||
home / ".commandcode/skills",
|
||||
home / ".continue/skills",
|
||||
home / ".snowflake/cortex/skills",
|
||||
home / ".config/crush/skills",
|
||||
home / ".cursor/skills",
|
||||
home / ".deepagents/agent/skills",
|
||||
home / ".factory/skills",
|
||||
home / ".firebender/skills",
|
||||
home / ".copilot/skills",
|
||||
home / ".config/goose/skills",
|
||||
home / ".junie/skills",
|
||||
home / ".iflow/skills",
|
||||
home / ".kilocode/skills",
|
||||
home / ".kiro/skills",
|
||||
home / ".kode/skills",
|
||||
home / ".mcpjam/skills",
|
||||
home / ".vibe/skills",
|
||||
home / ".mux/skills",
|
||||
home / ".config/opencode/skills",
|
||||
home / ".oh/skills",
|
||||
home / ".pi/agent/skills",
|
||||
home / ".qoder/skills",
|
||||
home / ".qwen/skills",
|
||||
home / ".roo/skills",
|
||||
home / ".trae/skills",
|
||||
home / ".trae-cn/skills",
|
||||
home / ".codeium/windsurf/skills",
|
||||
home / ".zencoder/skills",
|
||||
home / ".neovate/skills",
|
||||
home / ".pochi/skills",
|
||||
home / ".adal/skills",
|
||||
home / ".npm-global/lib/node_modules/openclaw/skills",
|
||||
# Container default paths
|
||||
Path("/mnt/skills/public"),
|
||||
Path("/mnt/skills/private"),
|
||||
Path("/mnt/skills/user"),
|
||||
# User dev/download paths
|
||||
home / "Downloads/skills",
|
||||
]
|
||||
|
||||
# Windows-specific paths
|
||||
if os.name == "nt":
|
||||
appdata = os.environ.get("APPDATA")
|
||||
if appdata:
|
||||
candidates.append(Path(appdata) / "OpenClaw/skills")
|
||||
candidates.append(Path(appdata) / "Programs/LobsterAI/resources/SKILLs")
|
||||
|
||||
# Dynamically scan extensions: .openclaw/extensions/{xxxx}/skills
|
||||
if oc_dir.exists():
|
||||
ext_root = oc_dir / "extensions"
|
||||
if ext_root.exists():
|
||||
for sub in ext_root.iterdir():
|
||||
if sub.is_dir():
|
||||
s_dir = sub / "skills"
|
||||
if s_dir.exists():
|
||||
candidates.append(s_dir)
|
||||
|
||||
# Include script run path and workspace
|
||||
candidates.append(Path.cwd() / "skills")
|
||||
candidates.append(Path(__file__).parent.parent / "skills")
|
||||
|
||||
# Deduplicate and filter non-existent paths
|
||||
seen = set()
|
||||
result = []
|
||||
for p in candidates:
|
||||
try:
|
||||
abs_p = p.resolve()
|
||||
if abs_p.exists() and abs_p not in seen:
|
||||
result.append(p)
|
||||
seen.add(abs_p)
|
||||
except Exception:
|
||||
continue
|
||||
return result
|
||||
|
||||
RISK_EMOJI = {"SAFE":"✅","LOW":"⚠️ ","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️ "}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Client Info (X-Client-Info)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _get_mac_address():
|
||||
"""Try to get the MAC address; return empty string on failure."""
|
||||
try:
|
||||
import uuid as _uuid
|
||||
mac_int = _uuid.getnode()
|
||||
# getnode() returns a random value (bit 8 set) when it can't get the real MAC
|
||||
if (mac_int >> 40) & 1:
|
||||
return ""
|
||||
mac_str = ":".join(("%012X" % mac_int)[i:i+2] for i in range(0, 12, 2))
|
||||
return mac_str
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _build_client_info():
|
||||
"""Build client info dict and persist to file; reuse on subsequent runs."""
|
||||
# If a record file already exists, read it
|
||||
if CLIENT_INFO_FILE.exists():
|
||||
try:
|
||||
data = json.loads(CLIENT_INFO_FILE.read_text(encoding="utf-8"))
|
||||
if data.get("client_id"):
|
||||
return data
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# First run: generate new client info
|
||||
info = {
|
||||
"client_id": str(uuid.uuid4()),
|
||||
"os": platform.system() or "",
|
||||
"platform": platform.machine() or "",
|
||||
"os_version": platform.release() or "",
|
||||
"client": "SkillScanner/%s" % SCANNER_VERSION,
|
||||
}
|
||||
|
||||
mac = _get_mac_address()
|
||||
if mac:
|
||||
info["mac"] = mac
|
||||
|
||||
# Python version as extra
|
||||
info["extra"] = {
|
||||
"python": platform.python_version(),
|
||||
}
|
||||
|
||||
# Persist
|
||||
try:
|
||||
CLIENT_INFO_FILE.write_text(
|
||||
json.dumps(info, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8"
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return info
|
||||
|
||||
|
||||
def _get_client_info_header():
|
||||
"""Return Base64-encoded X-Client-Info header value; empty string on failure."""
|
||||
try:
|
||||
info = _build_client_info()
|
||||
json_str = json.dumps(info, ensure_ascii=False)
|
||||
encoded = base64.b64encode(json_str.encode("utf-8")).decode("ascii")
|
||||
return encoded
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Output Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def banner(title: str):
|
||||
w = 58
|
||||
print(f"\n{'═'*w}")
|
||||
print(f" {title}")
|
||||
print(f"{'═'*w}")
|
||||
|
||||
def divider(title: str = ""):
|
||||
if title:
|
||||
print(f"\n ── {title} {'─'*(48-len(title))}")
|
||||
else:
|
||||
print(f" {'─'*52}")
|
||||
|
||||
def log(msg: str):
|
||||
print(f" {msg}", flush=True)
|
||||
|
||||
def ask(prompt: str) -> str:
|
||||
"""Read user input (compatible with non-interactive environments)."""
|
||||
try:
|
||||
return input(f"\n {prompt} ").strip()
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# HTTP Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def http_get(url: str) -> dict:
|
||||
req = urllib.request.Request(url)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
def http_post(url: str, payload: dict) -> dict:
|
||||
headers = {"Content-Type": "application/json"}
|
||||
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Skill Utilities
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def skill_name_from_dir(skill_dir: Path) -> str:
|
||||
md = skill_dir / "SKILL.md"
|
||||
if md.exists():
|
||||
for line in md.read_text(encoding="utf-8", errors="replace").splitlines():
|
||||
s = line.strip()
|
||||
if s.startswith("name:"):
|
||||
return s.split(":", 1)[1].strip().strip("\"'")
|
||||
return skill_dir.name
|
||||
|
||||
def sha256_of(path: Path) -> str:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
def calculate_dir_sha256(directory: Path) -> str:
|
||||
"""Calculate SHA256 hash of a skill directory (based on all file contents + relative paths).
|
||||
Excludes _meta.json and files/dirs in SKIP_FILES/SKIP_DIRS."""
|
||||
file_hashes = []
|
||||
for file_path in sorted(directory.rglob('*')):
|
||||
if not file_path.is_file():
|
||||
continue
|
||||
if file_path.name == '_meta.json':
|
||||
continue
|
||||
rel = file_path.relative_to(directory)
|
||||
if any(part in SKIP_DIRS for part in rel.parts):
|
||||
continue
|
||||
if file_path.name in SKIP_FILES:
|
||||
continue
|
||||
rel_path = str(rel)
|
||||
file_hash = hashlib.sha256()
|
||||
file_hash.update(rel_path.encode('utf-8'))
|
||||
file_hash.update(b'\x00')
|
||||
with open(file_path, 'rb') as f:
|
||||
for chunk in iter(lambda: f.read(8192), b''):
|
||||
file_hash.update(chunk)
|
||||
file_hashes.append(file_hash.hexdigest())
|
||||
file_hashes.sort()
|
||||
final_hash = hashlib.sha256()
|
||||
for h in file_hashes:
|
||||
final_hash.update(h.encode('utf-8'))
|
||||
final_hash.update(b'\x00')
|
||||
return final_hash.hexdigest()
|
||||
|
||||
def collect_files(skill_dir: Path) -> dict:
|
||||
"""Collect files for scanning, skipping redundant or sensitive directories."""
|
||||
exts = {".md",".py",".js",".ts",".sh",".yaml",".yml",".json",".txt"}
|
||||
out = {}
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
if p.suffix.lower() in exts or p.name == "SKILL.md":
|
||||
try:
|
||||
out[str(p.relative_to(skill_dir))] = \
|
||||
p.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
def pack_zip(skill_dir: Path) -> bytes:
|
||||
"""Pack a skill directory into a zip byte stream, excluding redundant directories."""
|
||||
import io
|
||||
buf = io.BytesIO()
|
||||
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
zf.write(p, p.relative_to(skill_dir))
|
||||
return buf.getvalue()
|
||||
|
||||
def unpack_zip(zip_path: Path) -> Path:
|
||||
"""Extract a .zip to a system temp directory. Returns the extraction path. Prevents zip-slip."""
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
log(f"📦 Extracting {zip_path.name} → {tmp}")
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / member).resolve()
|
||||
if not str(dest).startswith(str(tmp.resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp)
|
||||
return tmp
|
||||
|
||||
def find_installed_skills():
|
||||
# type: () -> list
|
||||
"""Find all installed skill directories (first-level subdirectories containing SKILL.md).
|
||||
Excludes SkillScan itself."""
|
||||
found = set()
|
||||
for base in skill_install_paths():
|
||||
if not base.exists():
|
||||
continue
|
||||
for md in base.rglob("SKILL.md"):
|
||||
skill_path = md.parent
|
||||
try:
|
||||
rel = skill_path.relative_to(base)
|
||||
if len(rel.parts) == 1:
|
||||
resolved = skill_path.resolve()
|
||||
# Skip self
|
||||
if resolved == SELF_ROOT:
|
||||
continue
|
||||
found.add(resolved)
|
||||
except ValueError:
|
||||
pass
|
||||
return sorted(found)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Scan Core (3 steps)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _extract_result(resp, sha256):
|
||||
"""Internal: extract core data from API response, handling SHA256 wrapping/nested result."""
|
||||
# 1. Handle API response keyed by SHA256 (e.g. { "sha256": { "status": "success", "data": {...} } })
|
||||
if sha256 and sha256 in resp:
|
||||
resp = resp[sha256]
|
||||
|
||||
# 2. Extract data body (data or result)
|
||||
data = resp.get("data") or resp.get("result") or resp
|
||||
|
||||
# 3. Handle nested result inside data
|
||||
if isinstance(data, dict) and "result" in data:
|
||||
inner = data["result"]
|
||||
if isinstance(inner, dict):
|
||||
# Merge sibling metadata (analysis_level/reason etc.) into result
|
||||
for k, v in data.items():
|
||||
if k != "result" and k not in inner:
|
||||
inner[k] = v
|
||||
return inner
|
||||
|
||||
return data if isinstance(data, dict) and (data.get("verdict") or data.get("is_safe") is not None or data.get("analysis_level")) else None
|
||||
|
||||
|
||||
def cloud_search(dir_sha256):
|
||||
"""Step 1: Query scan cache by dir_sha256. Returns result dict or None."""
|
||||
extra_headers = {}
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
extra_headers["X-Client-Info"] = ci
|
||||
|
||||
url = "%s?%s" % (API_SEARCH, urllib.parse.urlencode({"dir_sha256": dir_sha256}))
|
||||
try:
|
||||
headers = {}
|
||||
headers.update(extra_headers)
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
res = _extract_result(resp, dir_sha256)
|
||||
if res:
|
||||
log(" ✅ Cache hit (dir_sha256 %s…)" % dir_sha256[:16])
|
||||
return res
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return None
|
||||
raise RuntimeError("Search API error HTTP %d" % e.code)
|
||||
except urllib.error.URLError as e:
|
||||
raise RuntimeError("Cannot connect to server: %s" % e)
|
||||
return None
|
||||
|
||||
|
||||
def cloud_upload(skill_dir, name, dir_hash):
|
||||
"""Step 2: Upload skill (multipart/form-data), returns task_no."""
|
||||
# Pack the entire directory for full code context
|
||||
zip_data = pack_zip(skill_dir)
|
||||
filename = "%s.zip" % name
|
||||
|
||||
# Build multipart/form-data boundary
|
||||
boundary = "----WebKitFormBoundary%s" % uuid.uuid4().hex
|
||||
|
||||
# Manually construct multipart byte stream (no requests library needed)
|
||||
parts = []
|
||||
parts.append(("--%s" % boundary).encode())
|
||||
parts.append(('Content-Disposition: form-data; name="file"; filename="%s"' % filename).encode())
|
||||
parts.append(b"Content-Type: application/zip")
|
||||
parts.append(b"")
|
||||
parts.append(zip_data)
|
||||
parts.append(("--%s--" % boundary).encode())
|
||||
parts.append(b"") # trailing newline
|
||||
|
||||
body = b"\r\n".join(parts)
|
||||
|
||||
headers = {
|
||||
"Content-Type": "multipart/form-data; boundary=%s" % boundary,
|
||||
"Content-Length": str(len(body)),
|
||||
"Accept": "application/json"
|
||||
}
|
||||
|
||||
# Add X-Client-Info header
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
headers["X-Client-Info"] = ci
|
||||
|
||||
log(" 📤 Uploading: %s (%.1f KB)..." % (filename, len(zip_data) / 1024.0))
|
||||
req = urllib.request.Request(API_UPLOAD, data=body, headers=headers, method="POST")
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
except urllib.error.HTTPError as e:
|
||||
err_body = e.read().decode(errors="replace")
|
||||
raise RuntimeError("Upload failed HTTP %d: %s" % (e.code, err_body))
|
||||
|
||||
task_no = (resp.get("data") or {}).get("task_no") or resp.get("task_no") or resp.get("taskNo") or resp.get("task_id") or ""
|
||||
if not task_no:
|
||||
raise RuntimeError("Upload succeeded but no valid task_no in response: %s" % resp)
|
||||
|
||||
log(" ✅ Upload complete, task_no: %s" % task_no)
|
||||
return str(task_no)
|
||||
|
||||
|
||||
def cloud_poll(task_no: str) -> dict:
|
||||
"""Step 3: Poll until complete or timeout. Queries every 20s.
|
||||
status: 0=pending, 1=scanning, 2=completed, 3=failed, 4=cancelled
|
||||
"""
|
||||
url = f"{API_RESULT}?{urllib.parse.urlencode({'task_no': task_no})}"
|
||||
deadline = time.time() + POLL_TIMEOUT
|
||||
attempt = 0
|
||||
while time.time() < deadline:
|
||||
attempt += 1
|
||||
elapsed = int(time.time() - (deadline - POLL_TIMEOUT))
|
||||
try:
|
||||
resp = http_get(url)
|
||||
data = resp.get("data") or resp
|
||||
status = data.get("status")
|
||||
|
||||
if status == 2: # completed
|
||||
print()
|
||||
log(f" ✅ Scan complete (attempt {attempt}, {elapsed}s elapsed)")
|
||||
return _extract_result(resp, "") or resp
|
||||
elif status == 3: # failed
|
||||
print()
|
||||
err_msg = data.get("error_message") or resp.get("message", "unknown error")
|
||||
raise RuntimeError(f"Analysis failed: {err_msg}")
|
||||
elif status == 4: # cancelled
|
||||
print()
|
||||
raise RuntimeError("Scan task was cancelled")
|
||||
else:
|
||||
# 0=pending, 1=scanning -> keep waiting
|
||||
status_text = data.get("status_text", "processing")
|
||||
print(f" ⏳ [{status_text}] attempt {attempt}, {elapsed}s / {POLL_TIMEOUT}s elapsed",
|
||||
end="\r", flush=True)
|
||||
time.sleep(POLL_INTERVAL)
|
||||
except (RuntimeError, ValueError):
|
||||
raise
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Poll error: {e}")
|
||||
print()
|
||||
raise RuntimeError(f"Timeout ({POLL_TIMEOUT}s), task_no={task_no}, please retry later")
|
||||
|
||||
|
||||
def cloud_check(skill_dir: Path) -> dict:
|
||||
"""Run full security scan on a skill directory, return normalized result."""
|
||||
md = skill_dir / "SKILL.md"
|
||||
if not md.exists():
|
||||
raise FileNotFoundError(f"SKILL.md not found: {skill_dir}")
|
||||
|
||||
name = skill_name_from_dir(skill_dir)
|
||||
dir_hash = calculate_dir_sha256(skill_dir)
|
||||
log(f"🔍 Scanning: {name}")
|
||||
log(f" dir_sha256: {dir_hash}")
|
||||
|
||||
log(f"🔎 [1/3] Checking scan cache...")
|
||||
raw = cloud_search(dir_hash)
|
||||
|
||||
if raw is None:
|
||||
log(f" ℹ️ No cache record, submitting new scan task")
|
||||
log(f"📤 [2/3] Uploading skill for analysis...")
|
||||
task_no = cloud_upload(skill_dir, name, dir_hash)
|
||||
log(f"⏳ [3/3] Waiting for analysis (polling every {POLL_INTERVAL}s, max {POLL_TIMEOUT}s)...")
|
||||
raw = cloud_poll(task_no)
|
||||
else:
|
||||
log(f" ⏭️ Skipping upload, using cached result")
|
||||
|
||||
return _normalize(raw, name, dir_hash)
|
||||
|
||||
|
||||
def _normalize(raw: dict, name: str, dir_hash: str) -> dict:
|
||||
"""Normalize scan result:
|
||||
1. Extract is_safe (bool) and max_severity (str).
|
||||
2. Map API-specific fields (analysis_reason, analysis_suggestion) to standard fields.
|
||||
"""
|
||||
is_safe = raw.get("is_safe")
|
||||
|
||||
# Severity field priority: max_severity > analysis_level > verdict > level
|
||||
v_raw = (raw.get("max_severity") or raw.get("analysis_level") or
|
||||
raw.get("verdict") or raw.get("risk_level") or
|
||||
raw.get("level") or "UNKNOWN").upper()
|
||||
|
||||
# Combined verdict logic
|
||||
if is_safe is True and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "SAFE"
|
||||
elif is_safe is False and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "CRITICAL" # Explicitly marked unsafe -> critical
|
||||
else:
|
||||
verdict = v_raw
|
||||
|
||||
return {
|
||||
"skill_name": name,
|
||||
"dir_sha256": dir_hash,
|
||||
"verdict": verdict,
|
||||
"confidence": raw.get("confidence") or raw.get("score"),
|
||||
"threat_labels": raw.get("threat_labels") or raw.get("tags") or [],
|
||||
"summary": raw.get("analysis_reason") or raw.get("summary") or raw.get("description") or "",
|
||||
"findings": raw.get("findings") or raw.get("issues") or [],
|
||||
"recommendation":raw.get("analysis_suggestion") or raw.get("recommendation") or raw.get("action") or "",
|
||||
}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Result Display
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def print_result(r: dict):
|
||||
verdict = r.get("verdict","UNKNOWN")
|
||||
emoji = RISK_EMOJI.get(verdict,"❓")
|
||||
conf = r.get("confidence")
|
||||
labels = r.get("threat_labels",[])
|
||||
summary = r.get("summary","")
|
||||
findings = r.get("findings",[])
|
||||
rec = r.get("recommendation","")
|
||||
conf_str = f" confidence {float(conf):.0%}" if conf is not None else ""
|
||||
|
||||
divider()
|
||||
log(f"{emoji} Result: {verdict}{conf_str}")
|
||||
if summary:
|
||||
log(f"📋 {summary}")
|
||||
if labels:
|
||||
log(f"🏷️ Threat labels: {', '.join(labels)}")
|
||||
if findings:
|
||||
SEV = {"LOW":"🔵","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️"}
|
||||
log(f"🔍 Findings ({len(findings)} items):")
|
||||
for f in findings:
|
||||
sev = str(f.get("severity","")).upper()
|
||||
desc = f.get("description") or f.get("detail") or str(f)
|
||||
rid = f.get("id") or ""
|
||||
tag = f"[{rid}] " if rid else ""
|
||||
log(f" {SEV.get(sev,'⚪')} {tag}{desc}")
|
||||
if rec:
|
||||
log(f"💡 Recommendation: {rec}")
|
||||
divider()
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Prompt: malicious detected -> ask whether to delete
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def prompt_delete(skill_path: Path, result: dict) -> bool:
|
||||
"""When result is HIGH/CRITICAL, ask user whether to delete the skill.
|
||||
skill_path is the original install path (not temp dir).
|
||||
Returns True if deleted.
|
||||
"""
|
||||
verdict = result.get("verdict","")
|
||||
if verdict not in ("HIGH","CRITICAL"):
|
||||
return False
|
||||
|
||||
if not skill_path or not skill_path.exists():
|
||||
return False
|
||||
|
||||
emoji = RISK_EMOJI.get(verdict,"🔴")
|
||||
log(f"\n{emoji} This skill is marked as [{verdict}] high risk by security scan.")
|
||||
log(f" Path: {skill_path}")
|
||||
|
||||
answer = ask("Delete this skill now? [y/n]")
|
||||
if answer in ("y","Y","yes","Yes"):
|
||||
try:
|
||||
if skill_path.is_dir():
|
||||
shutil.rmtree(skill_path)
|
||||
else:
|
||||
skill_path.unlink()
|
||||
log(f"✅ Deleted: {skill_path}")
|
||||
return True
|
||||
except Exception as e:
|
||||
log(f"❌ Delete failed: {e} (please delete manually)")
|
||||
return False
|
||||
else:
|
||||
log(f"⚠️ Skipped deletion. Use this skill with caution.")
|
||||
return False
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: first-run (first install)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_first_run():
|
||||
"""First install: list installed skills, ask user to scan, show results."""
|
||||
if STATE_FILE.exists():
|
||||
log("ℹ️ First-run scan already completed. Use scan-all to rescan.")
|
||||
return
|
||||
|
||||
banner("🛡️ SkillScan First-Run Check")
|
||||
log("Welcome to SkillScan!")
|
||||
log("Searching for installed skills...\n")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("✅ No installed skills found, nothing to scan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Print installed skill list
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name}")
|
||||
|
||||
answer = ask("Run security scan on all listed skills? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Skipped. You can run scan-all anytime to rescan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Scan one by one
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
# Copy to temp dir (source may be read-only)
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
print_result(r)
|
||||
|
||||
# High risk -> ask to delete (targeting original install path)
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name,
|
||||
"verdict": "ERROR", "threat_labels": [],
|
||||
"summary": str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan (single skill)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan(path_str: str):
|
||||
skill_path = Path(path_str)
|
||||
if not skill_path.exists():
|
||||
log(f"❌ Path not found: {skill_path}")
|
||||
sys.exit(1)
|
||||
|
||||
banner(f"Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
tmp = None
|
||||
original_path = skill_path if skill_path.is_dir() else None
|
||||
try:
|
||||
if skill_path.is_file():
|
||||
if skill_path.suffix.lower() not in (".zip",):
|
||||
log(f"❌ Unsupported format: {skill_path.suffix} (use .zip)")
|
||||
sys.exit(1)
|
||||
tmp = unpack_zip(skill_path)
|
||||
scan_dir = tmp
|
||||
else:
|
||||
scan_dir = skill_path
|
||||
|
||||
result = cloud_check(scan_dir)
|
||||
print_result(result)
|
||||
|
||||
# High risk -> ask to delete
|
||||
if original_path:
|
||||
prompt_delete(original_path, result)
|
||||
elif skill_path.is_file() and result.get("verdict") in ("HIGH","CRITICAL"):
|
||||
# Zip file: ask to delete source file
|
||||
prompt_delete(skill_path, result)
|
||||
|
||||
v = result.get("verdict","UNKNOWN")
|
||||
sys.exit(0 if v in ("SAFE","LOW") else 1 if v=="MEDIUM" else 2)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"\n❌ Scan failed: {e}")
|
||||
sys.exit(3)
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan-all
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan_all():
|
||||
banner(f"Full Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("ℹ️ No installed skills detected.")
|
||||
return
|
||||
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name:<30} {s}")
|
||||
|
||||
answer = ask("Start security scan? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Cancelled.")
|
||||
return
|
||||
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
v = r.get("verdict","UNKNOWN")
|
||||
log(f"{RISK_EMOJI.get(v,'❓')} Scan complete: {v}")
|
||||
if r.get("threat_labels"):
|
||||
log(f" Threat labels: {', '.join(r['threat_labels'])}")
|
||||
|
||||
# High risk: ask to delete
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name, "verdict":"ERROR",
|
||||
"threat_labels":[], "summary":str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Summary Table
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _print_summary(results):
|
||||
banner("📊 Scan Summary")
|
||||
print(f" {'Skill Name':<28} {'Result':<12} {'Threat Labels'}")
|
||||
divider()
|
||||
for r in results:
|
||||
v = r.get("verdict","?")
|
||||
name = r.get("skill_name","?")[:27]
|
||||
labels = ", ".join(r.get("threat_labels",[]))[:20] or "-"
|
||||
print(f" {name:<28} {RISK_EMOJI.get(v,'❓')}{v:<10} {labels}")
|
||||
|
||||
safes = [r for r in results if r["verdict"] in {"SAFE","LOW"}]
|
||||
mediums = [r for r in results if r["verdict"] == "MEDIUM"]
|
||||
highs = [r for r in results if r["verdict"] in {"HIGH","CRITICAL"}]
|
||||
errors = [r for r in results if r["verdict"] in {"ERROR","UNKNOWN"}]
|
||||
|
||||
print()
|
||||
log(f"Total {len(results)} | ✅ Safe {len(safes)} "
|
||||
f"🟡 Suspicious {len(mediums)} 🔴 Dangerous {len(highs)} ❓ Error {len(errors)}")
|
||||
if highs:
|
||||
log(f"\n⚠️ High-risk skills: {', '.join(r['skill_name'] for r in highs)}")
|
||||
elif not mediums and not errors:
|
||||
log("\n🎉 All skills passed security scan.")
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: upgrade
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_upgrade():
|
||||
banner("SkillScan Auto-Upgrade")
|
||||
log(f"Current version: {SCANNER_VERSION}")
|
||||
log(f"Update source: {UPDATE_URL}")
|
||||
try:
|
||||
manifest = http_get(UPDATE_URL)
|
||||
except Exception as e:
|
||||
log(f"❌ Failed to fetch update manifest: {e}")
|
||||
return
|
||||
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f"✅ Already up to date ({SCANNER_VERSION})")
|
||||
return
|
||||
|
||||
log(f"New version found: {SCANNER_VERSION} → {latest}")
|
||||
log(f"Changelog: {manifest.get('changelog','(none)')}")
|
||||
|
||||
download_url = manifest.get("download_url", "")
|
||||
if not download_url:
|
||||
log("⚠️ No download URL in manifest, skipping upgrade")
|
||||
return
|
||||
|
||||
# Download new version zip
|
||||
log(f"📥 Downloading: {download_url}")
|
||||
try:
|
||||
req = urllib.request.Request(download_url)
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
zip_data = r.read()
|
||||
except Exception as e:
|
||||
log(f"❌ Download failed: {e}")
|
||||
return
|
||||
|
||||
# SHA256 verification
|
||||
expected_sha = manifest.get("sha256", "")
|
||||
if expected_sha:
|
||||
actual_sha = hashlib.sha256(zip_data).hexdigest()
|
||||
if actual_sha != expected_sha:
|
||||
log(f"❌ SHA256 mismatch, upgrade aborted (expected {expected_sha[:16]}…, got {actual_sha[:16]}…)")
|
||||
return
|
||||
log(f" ✅ SHA256 verified")
|
||||
|
||||
# Backup current skill directory
|
||||
skill_root = Path(__file__).parent.parent
|
||||
backup_dir = skill_root.parent / f"SkillScan-backup-{SCANNER_VERSION}"
|
||||
if backup_dir.exists():
|
||||
shutil.rmtree(backup_dir)
|
||||
shutil.copytree(skill_root, backup_dir)
|
||||
log(f"📦 Backed up to: {backup_dir}")
|
||||
|
||||
# Extract and replace files
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillupgrade-"))
|
||||
try:
|
||||
zip_path = tmp / "update.zip"
|
||||
zip_path.write_bytes(zip_data)
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
# Security check: prevent zip-slip
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / "extracted" / member).resolve()
|
||||
if not str(dest).startswith(str((tmp / "extracted").resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp / "extracted")
|
||||
|
||||
# Overwrite skill directory with new files
|
||||
extracted = tmp / "extracted"
|
||||
for item in extracted.rglob("*"):
|
||||
if not item.is_file():
|
||||
continue
|
||||
rel = item.relative_to(extracted)
|
||||
target = skill_root / rel
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(item, target)
|
||||
log(f" ✅ Updated: {rel}")
|
||||
|
||||
log(f"🎉 Upgraded to v{latest}")
|
||||
except Exception as e:
|
||||
log(f"❌ Upgrade failed: {e}")
|
||||
log(f" You can restore from backup: {backup_dir}")
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Entry Point
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def auto_upgrade_if_needed():
|
||||
"""Auto-check for updates every 7 days, runs silently."""
|
||||
try:
|
||||
if LAST_UPDATE_CHECK_FILE.exists():
|
||||
last_check = float(LAST_UPDATE_CHECK_FILE.read_text(encoding="utf-8").strip())
|
||||
if time.time() - last_check < AUTO_UPDATE_INTERVAL:
|
||||
return # Not time to check yet
|
||||
log("🔄 Checking for updates...")
|
||||
manifest = http_get(UPDATE_URL)
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f" ✅ Already up to date ({SCANNER_VERSION})")
|
||||
else:
|
||||
log(f" New version found: {SCANNER_VERSION} → {latest}, auto-updating...")
|
||||
cmd_upgrade()
|
||||
LAST_UPDATE_CHECK_FILE.write_text(str(time.time()), encoding="utf-8")
|
||||
except Exception as e:
|
||||
log(f" ⚠️ Auto-update check failed: {e} (normal operation unaffected)")
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print(__doc__)
|
||||
sys.exit(0)
|
||||
|
||||
# Check for auto-update on every run (once every 7 days)
|
||||
auto_upgrade_if_needed()
|
||||
|
||||
cmd = sys.argv[1]
|
||||
if cmd == "first-run":
|
||||
cmd_first_run()
|
||||
elif cmd == "scan":
|
||||
if len(sys.argv) < 3:
|
||||
log("Usage: scanner.py scan <skill_path>")
|
||||
sys.exit(1)
|
||||
cmd_scan(sys.argv[2])
|
||||
elif cmd == "scan-all":
|
||||
cmd_scan_all()
|
||||
elif cmd == "upgrade":
|
||||
cmd_upgrade()
|
||||
else:
|
||||
log(f"Unknown command: {cmd}")
|
||||
log("Available commands: first-run / scan <path> / scan-all / upgrade")
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,208 +0,0 @@
|
|||
# Word Reader 技能开发完成
|
||||
|
||||
## 🎯 技能概述
|
||||
|
||||
成功创建了一个功能完整的 Word 文档读取技能,支持读取 .docx 和 .doc 格式的 Word 文档,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## 📁 技能结构
|
||||
|
||||
```
|
||||
word-reader/
|
||||
├── SKILL.md # 技能定义文件
|
||||
├── README.md # 使用说明
|
||||
├── skill.json # 技能配置
|
||||
├── demo.sh # 演示脚本
|
||||
├── install.sh # 安装脚本
|
||||
├── test.md # 测试文档
|
||||
└── scripts/
|
||||
└── read_word.py # 核心脚本
|
||||
```
|
||||
|
||||
## ✨ 主要功能
|
||||
|
||||
### 1. 文档解析能力
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
|
||||
### 2. 格式支持
|
||||
- ✅ **.docx** - Office 2007+ 格式(主要支持)
|
||||
- ✅ **.doc** - 旧版 Word 格式(需要 antiword)
|
||||
|
||||
### 3. 输出格式
|
||||
- ✅ **JSON** - 结构化数据,适合程序处理
|
||||
- ✅ **Text** - 纯文本格式,简单易读
|
||||
- ✅ **Markdown** - 格式化输出,保留文档结构
|
||||
|
||||
### 4. 高级功能
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **选择性提取** - 可只提取特定内容类型
|
||||
- ✅ **文件输出** - 支持保存结果到文件
|
||||
- ✅ **编码支持** - 支持多种文本编码
|
||||
|
||||
## 🚀 使用示例
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# JSON 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# Markdown 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存结果
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 🔧 安装和配置
|
||||
|
||||
### 自动安装
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx
|
||||
|
||||
# 安装系统依赖(可选)
|
||||
sudo apt-get install antiword # Ubuntu/Debian
|
||||
brew install antiword # macOS
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "文档.docx",
|
||||
"title": "文档标题",
|
||||
"author": "作者",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "文档内容...",
|
||||
"tables": [...],
|
||||
"images": [...]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
```markdown
|
||||
# 文档.docx
|
||||
|
||||
**标题**:文档标题
|
||||
**作者**:作者
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
文档内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 数据1 | 数据2 |
|
||||
```
|
||||
|
||||
## 🎨 技能特点
|
||||
|
||||
### 1. 智能错误处理
|
||||
- 友好的错误提示
|
||||
- 自动检测文档格式
|
||||
- 优雅的异常处理
|
||||
|
||||
### 2. 性能优化
|
||||
- 流式处理大文件
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
|
||||
### 3. 用户友好
|
||||
- 详细的帮助信息
|
||||
- 多种使用方式
|
||||
- 完整的文档说明
|
||||
|
||||
### 4. 可扩展性
|
||||
- 模块化设计
|
||||
- 易于添加新功能
|
||||
- 支持自定义输出格式
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
### 1. 文档内容分析
|
||||
- 快速查看 Word 文档内容
|
||||
- 提取特定信息
|
||||
- 文档摘要生成
|
||||
|
||||
### 2. 批量处理
|
||||
- 处理大量文档
|
||||
- 文档格式转换
|
||||
- 内容索引创建
|
||||
|
||||
### 3. 自动化工作流
|
||||
- 集到文档处理系统
|
||||
- 自动化文档分析
|
||||
- 内容管理系统集成
|
||||
|
||||
## 📝 开发总结
|
||||
|
||||
### 实现的功能
|
||||
- 完整的 Word 文档解析框架
|
||||
- 支持多种输出格式
|
||||
- 批量处理能力
|
||||
- 错误处理和用户友好性
|
||||
|
||||
### 技术亮点
|
||||
- 模块化设计,易于维护
|
||||
- 优雅的错误处理机制
|
||||
- 支持多种文件格式
|
||||
- 灵活的输出选项
|
||||
|
||||
### 改进空间
|
||||
- 可以添加 PDF 支持
|
||||
- 可以增加图片提取功能
|
||||
- 可以优化大文件处理性能
|
||||
- 可以添加更多文档元素支持
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
要发布此技能到 ClawHub,可以运行:
|
||||
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm i -g clawhub
|
||||
|
||||
# 登录
|
||||
clawhub login
|
||||
|
||||
# 发布技能
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "Initial release with .docx and .doc support" \
|
||||
--tags document,word,office,text-extraction
|
||||
```
|
||||
|
||||
这个技能现在已经准备好使用了!它可以帮助用户轻松读取和处理 Word 文档,支持多种格式和输出选项。
|
||||
|
|
@ -1,177 +0,0 @@
|
|||
# Word Reader 技能发布指南
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
### 1. 准备工作
|
||||
|
||||
#### 确保技能完整
|
||||
- [ ] SKILL.md 文件完整且格式正确
|
||||
- [ ] 脚本功能正常
|
||||
- [ ] 安装脚本工作正常
|
||||
- [ ] README.md 说明清晰
|
||||
- [ ] 所有依赖已在 SKILL.md 中声明
|
||||
|
||||
#### 环境准备
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm install -g clawhub
|
||||
# 或
|
||||
pnpm add -g clawhub
|
||||
```
|
||||
|
||||
#### 登录 ClawHub
|
||||
```bash
|
||||
# 登录(会打开浏览器进行 OAuth 认证)
|
||||
clawhub login
|
||||
|
||||
# 验证登录状态
|
||||
clawhub whoami
|
||||
```
|
||||
|
||||
> **注意**:GitHub 账号需要注册满一周才能发布技能
|
||||
|
||||
### 2. 发布流程
|
||||
|
||||
#### 检查技能
|
||||
```bash
|
||||
# 验证技能结构
|
||||
clawhub validate ./word-reader
|
||||
```
|
||||
|
||||
#### 发布技能
|
||||
```bash
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "支持 .docx 和 .doc 格式的 Word 文档读取,提取文本、表格、元数据等" \
|
||||
--tags document,word,office,text-extraction,reader,parsing \
|
||||
--license MIT \
|
||||
--visibility public
|
||||
```
|
||||
|
||||
#### 参数说明
|
||||
- `--slug`: URL 友好的唯一标识符
|
||||
- `--name`: 技能显示名称
|
||||
- `--version`: 遵循语义化版本控制
|
||||
- `--changelog`: 版本变更说明
|
||||
- `--tags`: 搜索标签(逗号分隔)
|
||||
- `--license`: 许可证类型
|
||||
- `--visibility`: public/private
|
||||
|
||||
### 3. 发布后操作
|
||||
|
||||
#### 验证发布
|
||||
```bash
|
||||
# 查看已发布的技能
|
||||
clawhub search word-reader
|
||||
|
||||
# 安装测试
|
||||
clawhub install word-reader-test
|
||||
```
|
||||
|
||||
#### 分享技能
|
||||
- 技能将在 `https://clawhub.com/skills/word-reader` 可见
|
||||
- 其他用户可通过 `clawhub install word-reader` 安装
|
||||
|
||||
### 4. 版本管理
|
||||
|
||||
#### 更新技能
|
||||
```bash
|
||||
# 修改技能后更新版本号
|
||||
clawhub publish ./word-reader --version 1.0.1 --changelog "修复了某些文档格式的解析问题"
|
||||
```
|
||||
|
||||
#### 批量操作
|
||||
```bash
|
||||
# 同步所有技能
|
||||
clawhub sync --all
|
||||
|
||||
# 发布并标记
|
||||
clawhub publish ./word-reader --tags latest,stable
|
||||
```
|
||||
|
||||
### 5. 自动化发布
|
||||
|
||||
#### GitHub Actions 示例
|
||||
```yaml
|
||||
name: Publish Skill
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v3
|
||||
with:
|
||||
node-version: '18'
|
||||
|
||||
- name: Install ClawHub CLI
|
||||
run: npm install -g clawhub
|
||||
|
||||
- name: Login to ClawHub
|
||||
run: echo "${{ secrets.CLAWHUB_TOKEN }}" | clawhub login --token
|
||||
|
||||
- name: Publish Skill
|
||||
run: |
|
||||
clawhub publish ./skills/word-reader \
|
||||
--slug word-reader \
|
||||
--version ${{ github.ref_name }} \
|
||||
--changelog "Published from GitHub Actions"
|
||||
```
|
||||
|
||||
### 6. 发布注意事项
|
||||
|
||||
#### 必须遵守的规则
|
||||
- [ ] 技能名称不能与其他技能冲突
|
||||
- [ ] 版本号遵循 SemVer 规范
|
||||
- [ ] changelog 清晰描述变更
|
||||
- [ ] 代码无安全漏洞
|
||||
- [ ] 许可证声明清晰
|
||||
|
||||
#### 最佳实践
|
||||
- [ ] 发布前充分测试
|
||||
- [ ] 提供清晰的使用示例
|
||||
- [ ] 维护更新日志
|
||||
- [ ] 及时修复问题
|
||||
- [ ] 关注用户反馈
|
||||
|
||||
### 7. 故障排除
|
||||
|
||||
#### 常见问题
|
||||
```bash
|
||||
# 验证发布权限
|
||||
clawhub whoami
|
||||
|
||||
# 检查技能格式
|
||||
clawhub validate ./word-reader
|
||||
|
||||
# 查看详细错误信息
|
||||
clawhub publish ./word-reader --verbose
|
||||
```
|
||||
|
||||
#### 重新发布
|
||||
如果发布失败,可以:
|
||||
1. 修正问题
|
||||
2. 增加版本号
|
||||
3. 重新发布
|
||||
|
||||
### 8. 维护指南
|
||||
|
||||
#### 监控使用情况
|
||||
- 定期查看下载统计
|
||||
- 关注用户反馈
|
||||
- 及时修复问题
|
||||
|
||||
#### 更新策略
|
||||
- 重要修复:紧急发布补丁版本
|
||||
- 新功能:发布次版本号
|
||||
- 重大变更:发布主版本号
|
||||
|
||||
现在你的 Word Reader 技能已经准备好发布到 ClawHub 了!
|
||||
|
|
@ -1,171 +0,0 @@
|
|||
# Word Reader 技能
|
||||
|
||||
## 📋 概述
|
||||
|
||||
Word Reader 是一个强大的 Word 文档读取工具,支持 .docx 和 .doc 格式,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## ✨ 功能特性
|
||||
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
- ✅ **多格式支持** - 支持 .docx 和 .doc 格式
|
||||
- ✅ **多种输出** - JSON、Text、Markdown 格式
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **自动安装** - 一键安装所有依赖
|
||||
|
||||
## 🚀 安装
|
||||
|
||||
### 自动安装(推荐)
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx --break-system-packages
|
||||
|
||||
# 安装系统依赖(可选,用于 .doc 格式支持)
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📖 使用方法
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档并输出为文本格式
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# 输出为 JSON 格式
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# 输出为 Markdown 格式
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本内容
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有 Word 文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存为 JSON 文件
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
### 高级用法
|
||||
```bash
|
||||
# 将结果保存到文件
|
||||
python3 scripts/read_word.py 文档.docx --format markdown --output output.md
|
||||
|
||||
# 提取表格数据
|
||||
python3 scripts/read_word.py 文档.docx --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 scripts/read_word.py 文档.docx --extract metadata
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式输出
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "测试文档.docx",
|
||||
"size": "2048 bytes",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00",
|
||||
"title": "测试文档",
|
||||
"author": "测试用户"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "这是文档的正文内容...",
|
||||
"tables": [
|
||||
{
|
||||
"id": 1,
|
||||
"rows": 3,
|
||||
"columns": 3,
|
||||
"data": [
|
||||
["表头1", "表头2", "表头3"],
|
||||
["数据1", "数据2", "数据3"],
|
||||
["数据4", "数据5", "数据6"]
|
||||
]
|
||||
}
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"id": "rId1",
|
||||
"filename": "image1.png",
|
||||
"size": "1024 bytes"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式输出
|
||||
```markdown
|
||||
# 测试文档.docx
|
||||
|
||||
**标题**:测试文档
|
||||
**作者**:测试用户
|
||||
**文件大小**:2048 bytes
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
**修改时间**:2024-01-01T12:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
### 表格 1 (3行 x 3列)
|
||||
|
||||
| 表头1 | 表头2 | 表头3 |
|
||||
|-------|-------|-------|
|
||||
| 数据1 | 数据2 | 数据3 |
|
||||
| 数据4 | 数据5 | 数据6 |
|
||||
```
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
- **文档内容分析** - 快速查看 Word 文档内容
|
||||
- **批量处理** - 处理大量文档
|
||||
- **内容提取** - 提取特定信息
|
||||
- **格式转换** - 转换为其他格式
|
||||
- **自动化工作流** - 集成到文档处理系统
|
||||
|
||||
## 📤 发布到 ClawHub
|
||||
|
||||
要将此技能发布到 ClawHub,请参考 `PUBLISHING.md` 文件。
|
||||
|
||||
## 🔧 故障排除
|
||||
|
||||
### 常见问题
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **FileNotFoundError**: 确认文件路径正确
|
||||
4. **编码问题**: 尝试使用 `--encoding gb2312` 参数
|
||||
|
||||
### 性能优化
|
||||
- 大文档处理时建议使用 `--format json` 以获得更好的性能
|
||||
- 批量模式下建议使用 `--output` 参数将结果保存到文件
|
||||
|
||||
## 🤝 贡献
|
||||
|
||||
欢迎提交 Issue 和 Pull Request 来改进这个技能!
|
||||
|
||||
## 📄 许可证
|
||||
|
||||
MIT License
|
||||
|
|
@ -1,225 +0,0 @@
|
|||
---
|
||||
name: word-reader
|
||||
description: |
|
||||
读取 Word 文档(.docx 和 .doc 格式)并提取文本内容。支持文档解析、表格提取、图片处理等功能。使用当用户需要分析 Word 文档内容、提取文本信息或批量处理文档时。
|
||||
homepage: https://python-docx.readthedocs.io/
|
||||
metadata:
|
||||
{
|
||||
"openclaw":
|
||||
{
|
||||
"emoji": "📄",
|
||||
"requires": { "bins": ["python3"], "env": ["PYTHONPATH"] },
|
||||
"install":
|
||||
[
|
||||
{
|
||||
"id": "pip",
|
||||
"kind": "pip",
|
||||
"package": "python-docx",
|
||||
"bins": ["python3"],
|
||||
"label": "Install python-docx (pip)",
|
||||
},
|
||||
{
|
||||
"id": "system",
|
||||
"kind": "system",
|
||||
"command": "sudo apt-get install antiword -y",
|
||||
"label": "Install antiword for .doc support (optional)",
|
||||
"platform": "linux-debian"
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
---
|
||||
|
||||
# Word 文档读取器
|
||||
|
||||
使用 Python 解析 Word 文档,提取文本内容和结构化信息。
|
||||
|
||||
## 支持的功能
|
||||
|
||||
- **文档文本提取** - 提取段落、标题、页眉页脚内容
|
||||
- **表格解析** - 读取表格数据并转换为结构化格式
|
||||
- **图片处理** - 提取文档中的图片信息
|
||||
- **元数据获取** - 读取文档属性(作者、标题、创建时间等)
|
||||
- **批量处理** - 支持处理多个文档
|
||||
|
||||
## 用法
|
||||
|
||||
### 基本文本提取
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径>
|
||||
```
|
||||
|
||||
### 指定输出格式
|
||||
|
||||
```bash
|
||||
# JSON 输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format json
|
||||
|
||||
# 纯文本输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format text
|
||||
|
||||
# Markdown 格式
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format markdown
|
||||
```
|
||||
|
||||
### 提取特定内容
|
||||
|
||||
```bash
|
||||
# 只提取文本
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract text
|
||||
|
||||
# 提取表格数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract metadata
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
|
||||
```bash
|
||||
# 处理目录下所有 .docx 文件
|
||||
python3 {baseDir}/scripts/read_word.py <目录路径> --batch
|
||||
```
|
||||
|
||||
## 参数说明
|
||||
|
||||
| 参数 | 说明 | 默认值 |
|
||||
|------|------|--------|
|
||||
| `--format` | 输出格式(json/text/markdown) | text |
|
||||
| `--extract` | 提取内容类型(text/tables/images/metadata/all) | all |
|
||||
| `--batch` | 批量处理模式 | false |
|
||||
| `--output` | 输出文件路径 | stdout |
|
||||
| `--encoding` | 文本编码(utf-8/gb2312) | utf-8 |
|
||||
|
||||
## 输出格式
|
||||
|
||||
### JSON 格式
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"title": "文档标题",
|
||||
"author": "作者姓名",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"text": "文档全文内容...",
|
||||
"tables": [
|
||||
[
|
||||
["表头1", "表头2"],
|
||||
["行1列1", "行1列2"],
|
||||
["行2列1", "行2列2"]
|
||||
]
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"filename": "image1.png",
|
||||
"description": "图片描述",
|
||||
"size": "1024x768"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
|
||||
```markdown
|
||||
# 文档标题
|
||||
|
||||
**作者**:作者姓名
|
||||
**创建时间**:2024-01-01 10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
### 表格示例
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 行1列1 | 行1列2 |
|
||||
| 行2列1 | 行2列2 |
|
||||
|
||||

|
||||
|
||||
## 图片列表
|
||||
|
||||
1. **image1.png** (1024x768) - 图片描述
|
||||
```
|
||||
|
||||
## 错误处理
|
||||
|
||||
- 文件不存在:显示错误信息并退出
|
||||
- 格式不支持:提示支持的文件类型
|
||||
- 权限问题:提示文件访问权限
|
||||
- 编码问题:尝试自动检测编码
|
||||
|
||||
## 示例场景
|
||||
|
||||
### 1. 查看项目文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 项目需求.docx --format markdown
|
||||
```
|
||||
|
||||
### 2. 提取会议记录
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 会议记录.docx --extract text
|
||||
```
|
||||
|
||||
### 3. 批量处理文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 注意事项
|
||||
|
||||
- 支持 .docx 格式(Office 2007+)
|
||||
- .doc 格式需要额外依赖(如 antiword)
|
||||
- 大文档处理可能需要较长时间
|
||||
- 图片提取仅获取元数据,不包含实际图片数据
|
||||
- 表格格式可能需要手动调整
|
||||
|
||||
## 故障排除
|
||||
|
||||
### 常见问题
|
||||
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **UnicodeDecodeError**: 尝试不同的编码格式
|
||||
|
||||
### 安装依赖
|
||||
|
||||
```bash
|
||||
pip3 install python-docx
|
||||
```
|
||||
|
||||
对于 .doc 格式支持:
|
||||
```bash
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
```
|
||||
|
||||
## 高级功能
|
||||
|
||||
### 自定义样式处理
|
||||
|
||||
脚本会自动处理以下文档元素:
|
||||
- 标题级别(H1-H6)
|
||||
- 段落样式
|
||||
- 列表项目
|
||||
- 页眉页脚
|
||||
- 文档属性
|
||||
|
||||
### 性能优化
|
||||
|
||||
- 大文件流式处理
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
|
|
@ -1,11 +0,0 @@
|
|||
{
|
||||
"owner": "xtfnhcyjpgf",
|
||||
"slug": "word-reader",
|
||||
"displayName": "Word Reader",
|
||||
"latest": {
|
||||
"version": "1.0.0",
|
||||
"publishedAt": 1770700102926,
|
||||
"commit": "https://github.com/openclaw/skills/commit/91b71e101c57b69a4d4eb2678e1b79992eb7032f"
|
||||
},
|
||||
"history": []
|
||||
}
|
||||
|
|
@ -1,89 +0,0 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能演示脚本
|
||||
# 此脚本展示如何使用 word-reader 技能
|
||||
|
||||
echo "=== Word Reader 技能演示 ==="
|
||||
echo ""
|
||||
|
||||
# 检查脚本是否存在
|
||||
SCRIPT_PATH="/root/.openclaw/workspace/skills/word-reader/scripts/read_word.py"
|
||||
if [ ! -f "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本不存在"
|
||||
echo "请确保技能已正确安装"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# 检查脚本是否有执行权限
|
||||
if [ ! -x "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本没有执行权限"
|
||||
echo "正在添加执行权限..."
|
||||
chmod +x "$SCRIPT_PATH"
|
||||
fi
|
||||
|
||||
echo "✅ 脚本已就绪"
|
||||
echo ""
|
||||
|
||||
# 显示技能信息
|
||||
echo "📋 技能信息:"
|
||||
echo " 名称:word-reader"
|
||||
echo " 功能:读取 Word 文档(.docx 和 .doc 格式)"
|
||||
echo " 位置:$SCRIPT_PATH"
|
||||
echo ""
|
||||
|
||||
# 显示使用示例
|
||||
echo "📖 使用示例:"
|
||||
echo ""
|
||||
|
||||
echo "1. 显示帮助信息:"
|
||||
echo " python3 $SCRIPT_PATH --help"
|
||||
echo ""
|
||||
|
||||
echo "2. 读取文档(文本格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx"
|
||||
echo ""
|
||||
|
||||
echo "3. 读取文档(JSON 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format json"
|
||||
echo ""
|
||||
|
||||
echo "4. 读取文档(Markdown 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown"
|
||||
echo ""
|
||||
|
||||
echo "5. 只提取文本内容:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --extract text"
|
||||
echo ""
|
||||
|
||||
echo "6. 批量处理目录:"
|
||||
echo " python3 $SCRIPT_PATH ./文档目录 --batch"
|
||||
echo ""
|
||||
|
||||
echo "7. 保存结果到文件:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown --output output.md"
|
||||
echo ""
|
||||
|
||||
echo "🔧 安装依赖:"
|
||||
echo " pip3 install python-docx"
|
||||
echo " # 对于 .doc 格式支持:"
|
||||
echo " # Ubuntu: sudo apt-get install antiword"
|
||||
echo " # macOS: brew install antiword"
|
||||
echo ""
|
||||
|
||||
echo "📊 支持的功能:"
|
||||
echo " ✅ 文本提取"
|
||||
echo " ✅ 表格解析"
|
||||
echo " ✅ 元数据获取"
|
||||
echo " ✅ 图片信息"
|
||||
echo " ✅ 多格式支持"
|
||||
echo " ✅ 批量处理"
|
||||
echo ""
|
||||
|
||||
echo "💡 提示:"
|
||||
echo " - 支持 .docx 和 .doc 格式"
|
||||
echo " - 输出格式:JSON、Text、Markdown"
|
||||
echo " - 如遇错误,请检查依赖是否安装"
|
||||
echo ""
|
||||
|
||||
echo "演示完成!"
|
||||
echo "如需使用,请替换 '文档路径.docx' 为实际的文档路径"
|
||||
|
|
@ -1,101 +0,0 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能安装脚本
|
||||
# 此脚本会自动安装依赖并设置技能
|
||||
|
||||
set -e
|
||||
|
||||
echo "=== Word Reader 技能安装 ==="
|
||||
echo ""
|
||||
|
||||
# 检查 Python 版本
|
||||
echo "🔍 检查 Python 版本..."
|
||||
python_version=$(python3 --version 2>&1)
|
||||
echo " Python 版本: $python_version"
|
||||
|
||||
if ! python3 -c "import sys; assert sys.version_info >= (3, 6)"; then
|
||||
echo "❌ 错误:需要 Python 3.6 或更高版本"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "✅ Python 版本检查通过"
|
||||
echo ""
|
||||
|
||||
# 检查并安装依赖
|
||||
echo "📦 检查依赖..."
|
||||
|
||||
# 检查 pip
|
||||
if ! command -v pip3 &> /dev/null; then
|
||||
echo " 🔧 安装 pip..."
|
||||
python3 -m ensurepip --upgrade 2>/dev/null || {
|
||||
echo " ❌ 无法安装 pip,尝试使用系统包管理器"
|
||||
if command -v apt &> /dev/null; then
|
||||
sudo apt update
|
||||
sudo apt install -y python3-pip
|
||||
elif command -v yum &> /dev/null; then
|
||||
sudo yum install -y python3-pip
|
||||
elif command -v brew &> /dev/null; then
|
||||
brew install python3
|
||||
else
|
||||
echo " ❌ 无法自动安装 pip,请手动安装"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
fi
|
||||
|
||||
# 检查 python-docx
|
||||
if ! python3 -c "import docx" 2>/dev/null; then
|
||||
echo " 🔧 安装 python-docx..."
|
||||
if python3 -m pip install python-docx --break-system-packages 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
elif python3 -m pip install python-docx 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
else
|
||||
echo "❌ 无法安装 python-docx"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo " ✅ python-docx 已安装"
|
||||
fi
|
||||
|
||||
# 检查 antiword(可选)
|
||||
if command -v antiword >/dev/null 2>&1; then
|
||||
echo " ✅ antiword 已安装"
|
||||
else
|
||||
echo " ⚠️ antiword 未安装(可选,用于 .doc 格式支持)"
|
||||
echo " 推荐安装命令:"
|
||||
echo " Ubuntu/Debian: sudo apt-get install antiword"
|
||||
echo " macOS: brew install antiword"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
|
||||
# 设置执行权限
|
||||
echo "🔐 设置执行权限..."
|
||||
chmod +x scripts/read_word.py
|
||||
echo "✅ 执行权限已设置"
|
||||
echo ""
|
||||
|
||||
# 验证安装
|
||||
echo "🧪 验证安装..."
|
||||
python3 scripts/read_word.py --help >/dev/null 2>&1
|
||||
if [ $? -eq 0 ]; then
|
||||
echo "✅ 安装验证成功"
|
||||
else
|
||||
echo "❌ 安装验证失败"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "🎉 Word Reader 技能安装完成!"
|
||||
echo ""
|
||||
echo "📖 使用方法:"
|
||||
echo " python3 scripts/read_word.py 文档.docx"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format json"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format markdown"
|
||||
echo ""
|
||||
echo "📖 更多帮助:"
|
||||
echo " python3 scripts/read_word.py --help"
|
||||
echo ""
|
||||
echo "📖 运行演示:"
|
||||
echo " ./demo.sh"
|
||||
|
|
@ -1,396 +0,0 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Word 文档读取器
|
||||
支持 .docx 和 .doc 格式的 Word 文档解析
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import traceback
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from docx import Document
|
||||
from docx.opc.constants import RELATIONSHIP_TYPE as RT
|
||||
from docx.oxml.table import CT_Tbl
|
||||
from docx.oxml.text.paragraph import CT_P
|
||||
from docx.table import Table
|
||||
from docx.text.paragraph import Paragraph
|
||||
DOCX_AVAILABLE = True
|
||||
except ImportError:
|
||||
DOCX_AVAILABLE = False
|
||||
|
||||
try:
|
||||
import subprocess
|
||||
SUBPROCESS_AVAILABLE = True
|
||||
except ImportError:
|
||||
SUBPROCESS_AVAILABLE = False
|
||||
|
||||
class WordReader:
|
||||
"""Word 文档读取器"""
|
||||
|
||||
def __init__(self, file_path):
|
||||
self.file_path = Path(file_path)
|
||||
self.document = None
|
||||
self.format_type = None
|
||||
self.encoding = 'utf-8'
|
||||
|
||||
# 检查文件是否存在
|
||||
if not self.file_path.exists():
|
||||
raise FileNotFoundError(f"文件不存在: {file_path}")
|
||||
|
||||
# 检查文件扩展名
|
||||
if self.file_path.suffix.lower() not in ['.docx', '.doc']:
|
||||
raise ValueError(f"不支持的文件格式: {self.file_path.suffix}")
|
||||
|
||||
def read_docx(self):
|
||||
"""读取 .docx 格式文档"""
|
||||
if not DOCX_AVAILABLE:
|
||||
raise Exception("缺少 python-docx 库。请安装:pip3 install python-docx")
|
||||
|
||||
try:
|
||||
self.document = Document(str(self.file_path))
|
||||
self.format_type = 'docx'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .docx 文件失败: {str(e)}")
|
||||
|
||||
def read_doc(self):
|
||||
"""读取 .doc 格式文档(使用 antiword)"""
|
||||
if not SUBPROCESS_AVAILABLE:
|
||||
raise Exception("缺少 subprocess 模块")
|
||||
|
||||
try:
|
||||
# 检查 antiword 是否可用
|
||||
result = subprocess.run(['which', 'antiword'],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
raise Exception("antiword 未安装。请安装 antiword: Ubuntu/Debian: sudo apt-get install antiword; macOS: brew install antiword")
|
||||
|
||||
# 使用 antiword 转换
|
||||
result = subprocess.run(['antiword', str(self.file_path)],
|
||||
capture_output=True, text=True, encoding='utf-8')
|
||||
|
||||
if result.returncode != 0:
|
||||
raise Exception(f"antiword 转换失败: {result.stderr}")
|
||||
|
||||
# 创建临时文档对象
|
||||
class TempDocument:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
self.paragraphs = [TempParagraph(p) for p in text.split('\n') if p.strip()]
|
||||
|
||||
class TempParagraph:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
|
||||
self.document = TempDocument(result.stdout)
|
||||
self.format_type = 'doc'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .doc 文件失败: {str(e)}")
|
||||
|
||||
def read_metadata(self):
|
||||
"""读取文档元数据"""
|
||||
metadata = {
|
||||
'filename': self.file_path.name,
|
||||
'size': f"{self.file_path.stat().st_size} bytes",
|
||||
'created': datetime.fromtimestamp(self.file_path.stat().st_ctime).isoformat(),
|
||||
'modified': datetime.fromtimestamp(self.file_path.stat().st_mtime).isoformat()
|
||||
}
|
||||
|
||||
if self.format_type == 'docx' and hasattr(self.document, 'core_properties'):
|
||||
props = self.document.core_properties
|
||||
metadata.update({
|
||||
'title': getattr(props, 'title', ''),
|
||||
'author': getattr(props, 'author', ''),
|
||||
'subject': getattr(props, 'subject', ''),
|
||||
'keywords': getattr(props, 'keywords', ''),
|
||||
'comments': getattr(props, 'comments', ''),
|
||||
'application': getattr(props, 'application', ''),
|
||||
'category': getattr(props, 'category', '')
|
||||
})
|
||||
|
||||
return metadata
|
||||
|
||||
def extract_text(self):
|
||||
"""提取文档文本"""
|
||||
text_content = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
# 提取段落文本
|
||||
for para in self.document.paragraphs:
|
||||
if para.text.strip():
|
||||
text_content.append(para.text)
|
||||
|
||||
# 提取表格文本
|
||||
for table in self.document.tables:
|
||||
table_text = []
|
||||
for row in table.rows:
|
||||
row_text = []
|
||||
for cell in row.cells:
|
||||
row_text.append(cell.text.strip())
|
||||
table_text.append(' | '.join(row_text))
|
||||
text_content.append('\n'.join(table_text))
|
||||
|
||||
else: # doc 格式
|
||||
text_content = [para.text for para in self.document.paragraphs if para.text.strip()]
|
||||
|
||||
return '\n\n'.join(text_content)
|
||||
|
||||
def extract_tables(self):
|
||||
"""提取表格数据"""
|
||||
tables = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
for i, table in enumerate(self.document.tables):
|
||||
table_data = []
|
||||
for row in table.rows:
|
||||
row_data = []
|
||||
for cell in row.cells:
|
||||
row_data.append(cell.text.strip())
|
||||
table_data.append(row_data)
|
||||
tables.append({
|
||||
'id': i + 1,
|
||||
'rows': len(table.rows),
|
||||
'columns': len(table.columns) if table.rows else 0,
|
||||
'data': table_data
|
||||
})
|
||||
|
||||
return tables
|
||||
|
||||
def extract_images(self):
|
||||
"""提取图片信息"""
|
||||
images = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
try:
|
||||
# 获取文档中的关系
|
||||
part = self.document.part
|
||||
image_parts = part.related_parts
|
||||
|
||||
for rel in part.relationships:
|
||||
if rel.reltype == RT.IMAGE:
|
||||
image_data = image_parts[rel.rId]._blob
|
||||
image_info = {
|
||||
'id': rel.rId,
|
||||
'filename': f"image_{rel.rId}.{rel.target_ref.split('.')[-1]}",
|
||||
'size': f"{len(image_data)} bytes"
|
||||
}
|
||||
images.append(image_info)
|
||||
except:
|
||||
# 图片提取可能失败,忽略错误
|
||||
pass
|
||||
|
||||
return images
|
||||
|
||||
def extract_all(self):
|
||||
"""提取所有内容"""
|
||||
result = {
|
||||
'metadata': self.read_metadata(),
|
||||
'format': self.format_type,
|
||||
'text': self.extract_text(),
|
||||
'tables': self.extract_tables(),
|
||||
'images': self.extract_images()
|
||||
}
|
||||
return result
|
||||
|
||||
def to_markdown(self, extract_type='all'):
|
||||
"""转换为 Markdown 格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
md_content = []
|
||||
|
||||
# 标题
|
||||
md_content.append(f"# {result['metadata']['filename']}")
|
||||
md_content.append("")
|
||||
|
||||
# 元数据
|
||||
metadata = result['metadata']
|
||||
if metadata.get('title'):
|
||||
md_content.append(f"**标题**:{metadata['title']}")
|
||||
if metadata.get('author'):
|
||||
md_content.append(f"**作者**:{metadata['author']}")
|
||||
md_content.append(f"**文件大小**:{metadata['size']}")
|
||||
md_content.append(f"**创建时间**:{metadata['created']}")
|
||||
md_content.append(f"**修改时间**:{metadata['modified']}")
|
||||
md_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
md_content.append("## 正文内容")
|
||||
md_content.append("")
|
||||
md_content.append(result['text'])
|
||||
md_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
md_content.append("## 表格内容")
|
||||
md_content.append("")
|
||||
for table in result['tables']:
|
||||
md_content.append(f"### 表格 {table['id']} ({table['rows']}行 x {table['columns']}列)")
|
||||
md_content.append("")
|
||||
# 转换为 Markdown 表格
|
||||
for row in table['data']:
|
||||
md_row = " | ".join([str(cell) for cell in row])
|
||||
md_content.append(f"| {md_row} |")
|
||||
md_content.append("")
|
||||
|
||||
# 图片
|
||||
if result['images']:
|
||||
md_content.append("## 图片列表")
|
||||
md_content.append("")
|
||||
for img in result['images']:
|
||||
md_content.append(f"- **{img['filename']}** ({img['size']})")
|
||||
md_content.append("")
|
||||
|
||||
return '\n'.join(md_content)
|
||||
|
||||
def to_text(self, extract_type='all'):
|
||||
"""转换为纯文本格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
text_content = []
|
||||
|
||||
# 标题和元数据
|
||||
text_content.append(f"文件:{result['metadata']['filename']}")
|
||||
text_content.append("=" * 50)
|
||||
text_content.append("")
|
||||
|
||||
for key, value in result['metadata'].items():
|
||||
if value and key not in ['filename', 'size', 'created', 'modified']:
|
||||
text_content.append(f"{key}:{value}")
|
||||
|
||||
text_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
text_content.append("正文内容:")
|
||||
text_content.append("-" * 20)
|
||||
text_content.append(result['text'])
|
||||
text_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
text_content.append("表格内容:")
|
||||
text_content.append("-" * 20)
|
||||
for table in result['tables']:
|
||||
text_content.append(f"表格 {table['id']}:")
|
||||
for row in table['data']:
|
||||
text_content.append(" " + " | ".join([str(cell) for cell in row]))
|
||||
text_content.append("")
|
||||
|
||||
return '\n'.join(text_content)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description='读取 Word 文档')
|
||||
parser.add_argument('path', help='文档路径或目录路径(批量模式)')
|
||||
parser.add_argument('--format', choices=['json', 'text', 'markdown'],
|
||||
default='text', help='输出格式')
|
||||
parser.add_argument('--extract', choices=['text', 'tables', 'images', 'metadata', 'all'],
|
||||
default='all', help='提取内容类型')
|
||||
parser.add_argument('--batch', action='store_true', help='批量处理模式')
|
||||
parser.add_argument('--output', help='输出文件路径')
|
||||
parser.add_argument('--encoding', default='utf-8', help='文本编码')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
try:
|
||||
if args.batch:
|
||||
# 批量处理模式
|
||||
path = Path(args.path)
|
||||
if not path.is_dir():
|
||||
print("错误:批量模式需要指定目录路径")
|
||||
sys.exit(1)
|
||||
|
||||
# 查找所有 Word 文档
|
||||
word_files = []
|
||||
for ext in ['.docx', '.doc']:
|
||||
word_files.extend(path.glob(f"**/*{ext}"))
|
||||
|
||||
if not word_files:
|
||||
print("未找到 Word 文档")
|
||||
sys.exit(0)
|
||||
|
||||
print(f"找到 {len(word_files)} 个 Word 文档")
|
||||
|
||||
results = {}
|
||||
for file_path in word_files:
|
||||
print(f"正在处理: {file_path}")
|
||||
try:
|
||||
reader = WordReader(file_path)
|
||||
if file_path.suffix.lower() == '.docx':
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'content': content,
|
||||
'status': 'success'
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'error': str(e),
|
||||
'status': 'failed'
|
||||
}
|
||||
|
||||
# 保存结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(results, f, ensure_ascii=False, indent=2)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
|
||||
else:
|
||||
# 单文件处理模式
|
||||
reader = WordReader(args.path)
|
||||
|
||||
# 根据文件类型读取
|
||||
if args.path.lower().endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
# 根据格式输出
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
# 输出结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding=args.encoding) as f:
|
||||
f.write(content)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(content)
|
||||
|
||||
except Exception as e:
|
||||
print(f"错误: {str(e)}", file=sys.stderr)
|
||||
if '--debug' in sys.argv or '-d' in sys.argv:
|
||||
traceback.print_exc()
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
|
@ -1,47 +0,0 @@
|
|||
{
|
||||
"name": "word-reader",
|
||||
"version": "1.0.0",
|
||||
"description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容",
|
||||
"author": "OpenClaw User",
|
||||
"tags": ["document", "word", "office", "text-extraction"],
|
||||
"dependencies": {
|
||||
"python": ">=3.6",
|
||||
"packages": ["python-docx"],
|
||||
"system": ["antiword (optional for .doc support)"]
|
||||
},
|
||||
"features": {
|
||||
"text_extraction": true,
|
||||
"table_parsing": true,
|
||||
"metadata_extraction": true,
|
||||
"image_info": true,
|
||||
"batch_processing": true,
|
||||
"multiple_formats": ["json", "text", "markdown"]
|
||||
},
|
||||
"installation": {
|
||||
"steps": [
|
||||
"pip3 install python-docx",
|
||||
"sudo apt-get install antiword # 可选,支持 .doc 格式",
|
||||
"chmod +x scripts/read_word.py"
|
||||
]
|
||||
},
|
||||
"usage_examples": [
|
||||
{
|
||||
"description": "读取文档文本",
|
||||
"command": "python3 scripts/read_word.py document.docx"
|
||||
},
|
||||
{
|
||||
"description": "转换为 Markdown",
|
||||
"command": "python3 scripts/read_word.py document.docx --format markdown"
|
||||
},
|
||||
{
|
||||
"description": "批量处理",
|
||||
"command": "python3 scripts/read_word.py ./docs --batch --format json"
|
||||
}
|
||||
],
|
||||
"supported_file_types": [".docx", ".doc"],
|
||||
"notes": [
|
||||
".doc 格式需要安装 antiword",
|
||||
"大文档处理可能需要较长时间",
|
||||
"图片提取仅获取元数据,不包含实际图片数据"
|
||||
]
|
||||
}
|
||||
|
|
@ -1,42 +0,0 @@
|
|||
# Word Reader 技能测试
|
||||
|
||||
这是一个简单的测试文档,用于验证 Word Reader 技能的功能。
|
||||
|
||||
## 测试内容
|
||||
|
||||
### 1. 基本文本
|
||||
这是一段测试文本,用于验证文本提取功能是否正常工作。
|
||||
|
||||
### 2. 表格测试
|
||||
|
||||
| 功能 | 状态 | 描述 |
|
||||
|------|------|------|
|
||||
| 文本提取 | ✅ | 能够提取文档中的所有文本内容 |
|
||||
| 表格解析 | ✅ | 能够正确解析表格数据 |
|
||||
| 元数据获取 | ✅ | 能够获取文档属性信息 |
|
||||
| 多格式支持 | ✅ | 支持 .docx 和 .doc 格式 |
|
||||
| 输出格式 | ✅ | 支持 JSON、Text、Markdown 格式 |
|
||||
|
||||
### 3. 列表测试
|
||||
|
||||
- 第一项:文本提取功能
|
||||
- 第二项:表格解析功能
|
||||
- 第三项:图片信息获取
|
||||
- 第四项:文档元数据读取
|
||||
|
||||
### 4. 代码块示例
|
||||
|
||||
```python
|
||||
def read_word_document(file_path):
|
||||
"""读取 Word 文档"""
|
||||
reader = WordReader(file_path)
|
||||
if file_path.endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
return reader.extract_all()
|
||||
```
|
||||
|
||||
## 测试完成
|
||||
|
||||
如果这个技能能够正确读取并解析上述内容,说明功能正常。
|
||||
|
|
@ -1,144 +0,0 @@
|
|||
# AIOps-电信数通智能运维技能包
|
||||
|
||||
## 概述
|
||||
面向电信/数通网络的 AI 智能运维技能包。支持告警日志智能分析、配置错漏检查、故障模式学习与根因推荐。适用于华为、中兴、Cisco、Juniper 等主流数通设备。
|
||||
|
||||
## 能力说明
|
||||
|
||||
### 1. 告警日志智能分析
|
||||
|
||||
#### 1.1 输入格式支持
|
||||
- Excel/CSV 格式的告警清单(如网管导出的告警报表)
|
||||
- 纯文本格式的 syslog/告警日志
|
||||
- 设备 CLI 输出的告警信息
|
||||
|
||||
#### 1.2 分析维度
|
||||
|
||||
| 维度 | 说明 |
|
||||
|------|------|
|
||||
| **告警分级统计** | Critical / Major / Warning 级别分布及占比 |
|
||||
| **设备维度聚合** | 按设备(ME/NE)统计告警数量 Top N |
|
||||
| **告警类型归类** | 物理层/链路层/网络层/路由层/安全层分类 |
|
||||
| **时间维度分析** | 告警爆发时间窗口、频次趋势 |
|
||||
| **关联分析** | 同设备/同链路的多层告警关联,推测根因 |
|
||||
| **影响评估** | 评估受影响业务(L3VPN/Tunnel/Ethernet)范围 |
|
||||
|
||||
#### 1.3 告警类型知识库
|
||||
|
||||
##### 物理层告警
|
||||
| 告警名称 | 级别 | 含义 | 常见根因 | 推荐操作 |
|
||||
|----------|------|------|---------|---------|
|
||||
| Ethernet Physical LOS | Critical | 光口信号丢失 | 光纤断/松、光模块故障、对端设备断电 | 检查光纤链路、光功率、更换光模块 |
|
||||
| Ethernet Physical Laser Temperature | Major | 激光器温度越限 | 光模块老化、设备散热不良、环境温度过高 | 检查设备风扇/温度、更换光模块 |
|
||||
| Ethernet Physical Input Optical Power | Major | 接收光功率越限 | 光纤衰减过大、接口脏污、对端发光异常 | 清洁光纤接头、检查光功率预算 |
|
||||
| Ethernet Physical CRC Error | Major | CRC 误码越限 | 光纤质量差、接头污染、电磁干扰 | 清洁光纤、检查物理链路、更换尾纤 |
|
||||
| SmartGroup RX CRC Error | Major | 聚合口 CRC 误码 | 聚合成员链路质量问题 | 排查各成员链路、更换问题链路 |
|
||||
|
||||
##### 路由/MPLS层告警
|
||||
| 告警名称 | 级别 | 含义 | 常见根因 | 推荐操作 |
|
||||
|----------|------|------|---------|---------|
|
||||
| OSPF Neighbor Down | Major | OSPF 邻居中断 | 物理链路故障、Hello 超时、配置不匹配 | 检查邻居间链路、验证 OSPF 配置参数 |
|
||||
| RSVP LSP BFD Session Down | Major | MPLS TE隧道 BFD 检测失败 | 隧道途经链路故障、节点故障 | 检查隧道路径、确认中间节点状态 |
|
||||
| BGP Peer Flap | Major | BGP 邻居震荡 | 链路不稳、Keepalive 超时、策略变更 | 检查 BGP 配置、链路稳定性 |
|
||||
|
||||
##### 安全/控制面告警
|
||||
| 告警名称 | 级别 | 含义 | 常见根因 | 推荐操作 |
|
||||
|----------|------|------|---------|---------|
|
||||
| CPSBP 超阈值 | Warning | 控制面报文速率超阈值 | 攻击流量、广播风暴、配置过低 | 分析流量源、调整 CP 保护阈值 |
|
||||
| ARP IP冲突 | Warning | 检测到 IP 地址冲突 | 私接设备、IP 配置错误 | 定位冲突源、释放/更改 IP 地址 |
|
||||
|
||||
### 2. 配置错漏检查
|
||||
|
||||
#### 2.1 支持检查项
|
||||
|
||||
| 检查类别 | 典型检查项 |
|
||||
|----------|-----------|
|
||||
| **接口配置** | VLAN 配置一致性、MTU 匹配、描述规范、端口模式(access/trunk) |
|
||||
| **路由协议** | OSPF area 一致性、BGP AS 号核对、Route-map 逻辑、邻居配置完整性 |
|
||||
| **MPLS/TE** | LSP 配置完整性、隧道保护策略、FRR 配置 |
|
||||
| **安全策略** | ACL 规则匹配、控制面保护(CoPP)、端口安全、MAC 漂移检测 |
|
||||
| **高可用** | VRRP/HSRP 配置一致性、BFD 联动、链路聚合(LACP)配置 |
|
||||
| **QoS** | 队列策略、带宽限制、优先级映射 |
|
||||
|
||||
#### 2.2 分析模式
|
||||
- **配置比对**:新旧配置 diff,快速定位变更点
|
||||
- **合规检查**:基于标准模板检查配置是否符合规范
|
||||
- **逻辑验证**:检查配置逻辑矛盾(如 ACL 冗余/冲突、路由黑洞)
|
||||
|
||||
### 3. 故障模式学习 (self-learning)
|
||||
|
||||
系统会记录每次分析过程中的:
|
||||
- 告警 → 修复措施 对应关系
|
||||
- 配置错误 → 正确配置 修正方案
|
||||
- 经用户确认的根因分析结论
|
||||
|
||||
这些 learnings 会存入 `.learnings/` 目录,在后续分析中自动参考,越用越准确。
|
||||
|
||||
## 使用示例
|
||||
|
||||
### 示例1:告警日志分析
|
||||
```
|
||||
用户: "分析这份告警日志"
|
||||
助手: 按以下格式输出分析报告:
|
||||
|
||||
📊 告警概览
|
||||
- 总告警数:24条
|
||||
- Critical:2条 (8.3%) ⛔
|
||||
- Major:9条 (37.5%) ⚠️
|
||||
- Warning:13条 (54.2%) ⚡
|
||||
|
||||
🔴 Critical 告警详情
|
||||
1. [SMD-LIIR-EN1-Z20HS] ETPI LOS → 物理光口信号丢失
|
||||
→ 该设备下联 SMD-SBRS-EN1-Z20HS,状态 Unack
|
||||
→ 建议:立即检查光纤/光模块
|
||||
|
||||
🟠 Major 告警详情
|
||||
1. [BKL-UNB-AN1-ZM3SP] 光功率越限 → 接收光功率过低
|
||||
2. [MDN-PMKN-EN1-Z20HS] 激光器温度越限
|
||||
3. [PAD-KBU-AN1-ZM8S] CRC误码越限(物理口+聚合口)
|
||||
4. RSVP LSP BFD Session Down ×3条
|
||||
5. OSPF邻居中断 ×2条
|
||||
|
||||
🔄 关联分析发现
|
||||
- PBR-RPKU 与 PBR-PYSK 之间:OSPF邻居中断 + CPSBP告警
|
||||
→ 推测该链路存在物理层问题
|
||||
- GRO 站点多条 RSVP LSP BFD Down
|
||||
→ 可能为 GRO 节点设备问题或出局光缆中断
|
||||
|
||||
💡 根因推荐
|
||||
1. 优先处理 SMD-LIIR-EN1-Z20HS 的 LOS(Critical)
|
||||
2. 检查 PBR-RPKU ↔ PBR-PYSK 链路光功率和光模块
|
||||
3. 排查 GRO 站点汇聚设备状态
|
||||
```
|
||||
|
||||
### 示例2:配置检查
|
||||
```
|
||||
用户: "帮我检查这两份配置文件"
|
||||
助手:
|
||||
🔍 配置检查报告
|
||||
1. OSPF 配置检查 ✅
|
||||
- Area 一致性:匹配
|
||||
- Hello/Dead 间隔:一致
|
||||
- 网络类型:一致
|
||||
|
||||
2. 接口配置检查 ⚠️
|
||||
- [GE0/0/1] MTU 不匹配:本端 1500,对端 9000
|
||||
- [GE0/0/2] 描述缺失
|
||||
|
||||
3. BGP 配置检查 ❌
|
||||
- AS 号不匹配:本端 AS65001,对端 AS65002
|
||||
```
|
||||
|
||||
## 数据来源说明
|
||||
|
||||
本技能的知识库基于以下标准构建:
|
||||
- 华为 NE40E/ME60 系列告警手册
|
||||
- 中兴 ZXR10 系列告警与配置规范
|
||||
- 3GPP 管理面标准(IRP/Solution)
|
||||
- ITU-T 光传输标准
|
||||
- RFC 相关协议标准
|
||||
|
||||
## 局限性与注意事项
|
||||
- 本技能不直接连接设备执行命令,不做配置变更操作
|
||||
- 分析结果基于提供的日志/配置数据,用户需确认数据准确性
|
||||
- 推荐操作为参考建议,重大操作需人工复核
|
||||
|
|
@ -1,21 +0,0 @@
|
|||
---
|
||||
name: data-analyst
|
||||
description: Complete the data analysis tasks delegated by the user.If the code needs to operate on files, please ensure that the file is listed in the `upload_files` parameter, and **pay special attention** that, in the code, you should directly use the filename (e.g., `open('data.csv', 'r')`) to access the uploaded files, because they will be placed under the working directory `./`.
|
||||
---
|
||||
|
||||
# Data Analyst
|
||||
|
||||
## Overview
|
||||
|
||||
This skill provides specialized capabilities for data analyst.
|
||||
|
||||
## Instructions
|
||||
|
||||
Complete the data analysis tasks delegated by the user.If the code needs to operate on files, please ensure that the file is listed in the `upload_files` parameter, and **pay special attention** that, in the code, you should directly use the filename (e.g., `open('data.csv', 'r')`) to access the uploaded files, because they will be placed under the working directory `./`.
|
||||
|
||||
|
||||
## Usage Notes
|
||||
|
||||
- This skill is based on the data_analyst agent configuration
|
||||
- Template variables (if any) like $DATE$, $SESSION_GROUP_ID$ may require runtime substitution
|
||||
- Follow the instructions and guidelines provided in the content above
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
{
|
||||
"ownerId": "kn71xkrq2fawjvteej73gsx71s80p3kb",
|
||||
"slug": "data-analyst-pro",
|
||||
"version": "0.1.0",
|
||||
"publishedAt": 1771141367019
|
||||
}
|
||||
202
runtime/skills/skill-creator/LICENSE.txt
Normal file
202
runtime/skills/skill-creator/LICENSE.txt
Normal file
|
|
@ -0,0 +1,202 @@
|
|||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
356
runtime/skills/skill-creator/SKILL.md
Normal file
356
runtime/skills/skill-creator/SKILL.md
Normal file
|
|
@ -0,0 +1,356 @@
|
|||
---
|
||||
name: skill-creator
|
||||
description: Guide for creating effective skills. This skill should be used when users want to create a new skill (or update an existing skill) that extends Claude's capabilities with specialized knowledge, workflows, or tool integrations.
|
||||
license: Complete terms in LICENSE.txt
|
||||
---
|
||||
|
||||
# Skill Creator
|
||||
|
||||
This skill provides guidance for creating effective skills.
|
||||
|
||||
## About Skills
|
||||
|
||||
Skills are modular, self-contained packages that extend Claude's capabilities by providing
|
||||
specialized knowledge, workflows, and tools. Think of them as "onboarding guides" for specific
|
||||
domains or tasks—they transform Claude from a general-purpose agent into a specialized agent
|
||||
equipped with procedural knowledge that no model can fully possess.
|
||||
|
||||
### What Skills Provide
|
||||
|
||||
1. Specialized workflows - Multi-step procedures for specific domains
|
||||
2. Tool integrations - Instructions for working with specific file formats or APIs
|
||||
3. Domain expertise - Company-specific knowledge, schemas, business logic
|
||||
4. Bundled resources - Scripts, references, and assets for complex and repetitive tasks
|
||||
|
||||
## Core Principles
|
||||
|
||||
### Concise is Key
|
||||
|
||||
The context window is a public good. Skills share the context window with everything else Claude needs: system prompt, conversation history, other Skills' metadata, and the actual user request.
|
||||
|
||||
**Default assumption: Claude is already very smart.** Only add context Claude doesn't already have. Challenge each piece of information: "Does Claude really need this explanation?" and "Does this paragraph justify its token cost?"
|
||||
|
||||
Prefer concise examples over verbose explanations.
|
||||
|
||||
### Set Appropriate Degrees of Freedom
|
||||
|
||||
Match the level of specificity to the task's fragility and variability:
|
||||
|
||||
**High freedom (text-based instructions)**: Use when multiple approaches are valid, decisions depend on context, or heuristics guide the approach.
|
||||
|
||||
**Medium freedom (pseudocode or scripts with parameters)**: Use when a preferred pattern exists, some variation is acceptable, or configuration affects behavior.
|
||||
|
||||
**Low freedom (specific scripts, few parameters)**: Use when operations are fragile and error-prone, consistency is critical, or a specific sequence must be followed.
|
||||
|
||||
Think of Claude as exploring a path: a narrow bridge with cliffs needs specific guardrails (low freedom), while an open field allows many routes (high freedom).
|
||||
|
||||
### Anatomy of a Skill
|
||||
|
||||
Every skill consists of a required SKILL.md file and optional bundled resources:
|
||||
|
||||
```
|
||||
skill-name/
|
||||
├── SKILL.md (required)
|
||||
│ ├── YAML frontmatter metadata (required)
|
||||
│ │ ├── name: (required)
|
||||
│ │ └── description: (required)
|
||||
│ └── Markdown instructions (required)
|
||||
└── Bundled Resources (optional)
|
||||
├── scripts/ - Executable code (Python/Bash/etc.)
|
||||
├── references/ - Documentation intended to be loaded into context as needed
|
||||
└── assets/ - Files used in output (templates, icons, fonts, etc.)
|
||||
```
|
||||
|
||||
#### SKILL.md (required)
|
||||
|
||||
Every SKILL.md consists of:
|
||||
|
||||
- **Frontmatter** (YAML): Contains `name` and `description` fields. These are the only fields that Claude reads to determine when the skill gets used, thus it is very important to be clear and comprehensive in describing what the skill is, and when it should be used.
|
||||
- **Body** (Markdown): Instructions and guidance for using the skill. Only loaded AFTER the skill triggers (if at all).
|
||||
|
||||
#### Bundled Resources (optional)
|
||||
|
||||
##### Scripts (`scripts/`)
|
||||
|
||||
Executable code (Python/Bash/etc.) for tasks that require deterministic reliability or are repeatedly rewritten.
|
||||
|
||||
- **When to include**: When the same code is being rewritten repeatedly or deterministic reliability is needed
|
||||
- **Example**: `scripts/rotate_pdf.py` for PDF rotation tasks
|
||||
- **Benefits**: Token efficient, deterministic, may be executed without loading into context
|
||||
- **Note**: Scripts may still need to be read by Claude for patching or environment-specific adjustments
|
||||
|
||||
##### References (`references/`)
|
||||
|
||||
Documentation and reference material intended to be loaded as needed into context to inform Claude's process and thinking.
|
||||
|
||||
- **When to include**: For documentation that Claude should reference while working
|
||||
- **Examples**: `references/finance.md` for financial schemas, `references/mnda.md` for company NDA template, `references/policies.md` for company policies, `references/api_docs.md` for API specifications
|
||||
- **Use cases**: Database schemas, API documentation, domain knowledge, company policies, detailed workflow guides
|
||||
- **Benefits**: Keeps SKILL.md lean, loaded only when Claude determines it's needed
|
||||
- **Best practice**: If files are large (>10k words), include grep search patterns in SKILL.md
|
||||
- **Avoid duplication**: Information should live in either SKILL.md or references files, not both. Prefer references files for detailed information unless it's truly core to the skill—this keeps SKILL.md lean while making information discoverable without hogging the context window. Keep only essential procedural instructions and workflow guidance in SKILL.md; move detailed reference material, schemas, and examples to references files.
|
||||
|
||||
##### Assets (`assets/`)
|
||||
|
||||
Files not intended to be loaded into context, but rather used within the output Claude produces.
|
||||
|
||||
- **When to include**: When the skill needs files that will be used in the final output
|
||||
- **Examples**: `assets/logo.png` for brand assets, `assets/slides.pptx` for PowerPoint templates, `assets/frontend-template/` for HTML/React boilerplate, `assets/font.ttf` for typography
|
||||
- **Use cases**: Templates, images, icons, boilerplate code, fonts, sample documents that get copied or modified
|
||||
- **Benefits**: Separates output resources from documentation, enables Claude to use files without loading them into context
|
||||
|
||||
#### What to Not Include in a Skill
|
||||
|
||||
A skill should only contain essential files that directly support its functionality. Do NOT create extraneous documentation or auxiliary files, including:
|
||||
|
||||
- README.md
|
||||
- INSTALLATION_GUIDE.md
|
||||
- QUICK_REFERENCE.md
|
||||
- CHANGELOG.md
|
||||
- etc.
|
||||
|
||||
The skill should only contain the information needed for an AI agent to do the job at hand. It should not contain auxilary context about the process that went into creating it, setup and testing procedures, user-facing documentation, etc. Creating additional documentation files just adds clutter and confusion.
|
||||
|
||||
### Progressive Disclosure Design Principle
|
||||
|
||||
Skills use a three-level loading system to manage context efficiently:
|
||||
|
||||
1. **Metadata (name + description)** - Always in context (~100 words)
|
||||
2. **SKILL.md body** - When skill triggers (<5k words)
|
||||
3. **Bundled resources** - As needed by Claude (Unlimited because scripts can be executed without reading into context window)
|
||||
|
||||
#### Progressive Disclosure Patterns
|
||||
|
||||
Keep SKILL.md body to the essentials and under 500 lines to minimize context bloat. Split content into separate files when approaching this limit. When splitting out content into other files, it is very important to reference them from SKILL.md and describe clearly when to read them, to ensure the reader of the skill knows they exist and when to use them.
|
||||
|
||||
**Key principle:** When a skill supports multiple variations, frameworks, or options, keep only the core workflow and selection guidance in SKILL.md. Move variant-specific details (patterns, examples, configuration) into separate reference files.
|
||||
|
||||
**Pattern 1: High-level guide with references**
|
||||
|
||||
```markdown
|
||||
# PDF Processing
|
||||
|
||||
## Quick start
|
||||
|
||||
Extract text with pdfplumber:
|
||||
[code example]
|
||||
|
||||
## Advanced features
|
||||
|
||||
- **Form filling**: See [FORMS.md](FORMS.md) for complete guide
|
||||
- **API reference**: See [REFERENCE.md](REFERENCE.md) for all methods
|
||||
- **Examples**: See [EXAMPLES.md](EXAMPLES.md) for common patterns
|
||||
```
|
||||
|
||||
Claude loads FORMS.md, REFERENCE.md, or EXAMPLES.md only when needed.
|
||||
|
||||
**Pattern 2: Domain-specific organization**
|
||||
|
||||
For Skills with multiple domains, organize content by domain to avoid loading irrelevant context:
|
||||
|
||||
```
|
||||
bigquery-skill/
|
||||
├── SKILL.md (overview and navigation)
|
||||
└── reference/
|
||||
├── finance.md (revenue, billing metrics)
|
||||
├── sales.md (opportunities, pipeline)
|
||||
├── product.md (API usage, features)
|
||||
└── marketing.md (campaigns, attribution)
|
||||
```
|
||||
|
||||
When a user asks about sales metrics, Claude only reads sales.md.
|
||||
|
||||
Similarly, for skills supporting multiple frameworks or variants, organize by variant:
|
||||
|
||||
```
|
||||
cloud-deploy/
|
||||
├── SKILL.md (workflow + provider selection)
|
||||
└── references/
|
||||
├── aws.md (AWS deployment patterns)
|
||||
├── gcp.md (GCP deployment patterns)
|
||||
└── azure.md (Azure deployment patterns)
|
||||
```
|
||||
|
||||
When the user chooses AWS, Claude only reads aws.md.
|
||||
|
||||
**Pattern 3: Conditional details**
|
||||
|
||||
Show basic content, link to advanced content:
|
||||
|
||||
```markdown
|
||||
# DOCX Processing
|
||||
|
||||
## Creating documents
|
||||
|
||||
Use docx-js for new documents. See [DOCX-JS.md](DOCX-JS.md).
|
||||
|
||||
## Editing documents
|
||||
|
||||
For simple edits, modify the XML directly.
|
||||
|
||||
**For tracked changes**: See [REDLINING.md](REDLINING.md)
|
||||
**For OOXML details**: See [OOXML.md](OOXML.md)
|
||||
```
|
||||
|
||||
Claude reads REDLINING.md or OOXML.md only when the user needs those features.
|
||||
|
||||
**Important guidelines:**
|
||||
|
||||
- **Avoid deeply nested references** - Keep references one level deep from SKILL.md. All reference files should link directly from SKILL.md.
|
||||
- **Structure longer reference files** - For files longer than 100 lines, include a table of contents at the top so Claude can see the full scope when previewing.
|
||||
|
||||
## Skill Creation Process
|
||||
|
||||
Skill creation involves these steps:
|
||||
|
||||
1. Understand the skill with concrete examples
|
||||
2. Plan reusable skill contents (scripts, references, assets)
|
||||
3. Initialize the skill (run init_skill.py)
|
||||
4. Edit the skill (implement resources and write SKILL.md)
|
||||
5. Package the skill (run package_skill.py)
|
||||
6. Iterate based on real usage
|
||||
|
||||
Follow these steps in order, skipping only if there is a clear reason why they are not applicable.
|
||||
|
||||
### Step 1: Understanding the Skill with Concrete Examples
|
||||
|
||||
Skip this step only when the skill's usage patterns are already clearly understood. It remains valuable even when working with an existing skill.
|
||||
|
||||
To create an effective skill, clearly understand concrete examples of how the skill will be used. This understanding can come from either direct user examples or generated examples that are validated with user feedback.
|
||||
|
||||
For example, when building an image-editor skill, relevant questions include:
|
||||
|
||||
- "What functionality should the image-editor skill support? Editing, rotating, anything else?"
|
||||
- "Can you give some examples of how this skill would be used?"
|
||||
- "I can imagine users asking for things like 'Remove the red-eye from this image' or 'Rotate this image'. Are there other ways you imagine this skill being used?"
|
||||
- "What would a user say that should trigger this skill?"
|
||||
|
||||
To avoid overwhelming users, avoid asking too many questions in a single message. Start with the most important questions and follow up as needed for better effectiveness.
|
||||
|
||||
Conclude this step when there is a clear sense of the functionality the skill should support.
|
||||
|
||||
### Step 2: Planning the Reusable Skill Contents
|
||||
|
||||
To turn concrete examples into an effective skill, analyze each example by:
|
||||
|
||||
1. Considering how to execute on the example from scratch
|
||||
2. Identifying what scripts, references, and assets would be helpful when executing these workflows repeatedly
|
||||
|
||||
Example: When building a `pdf-editor` skill to handle queries like "Help me rotate this PDF," the analysis shows:
|
||||
|
||||
1. Rotating a PDF requires re-writing the same code each time
|
||||
2. A `scripts/rotate_pdf.py` script would be helpful to store in the skill
|
||||
|
||||
Example: When designing a `frontend-webapp-builder` skill for queries like "Build me a todo app" or "Build me a dashboard to track my steps," the analysis shows:
|
||||
|
||||
1. Writing a frontend webapp requires the same boilerplate HTML/React each time
|
||||
2. An `assets/hello-world/` template containing the boilerplate HTML/React project files would be helpful to store in the skill
|
||||
|
||||
Example: When building a `big-query` skill to handle queries like "How many users have logged in today?" the analysis shows:
|
||||
|
||||
1. Querying BigQuery requires re-discovering the table schemas and relationships each time
|
||||
2. A `references/schema.md` file documenting the table schemas would be helpful to store in the skill
|
||||
|
||||
To establish the skill's contents, analyze each concrete example to create a list of the reusable resources to include: scripts, references, and assets.
|
||||
|
||||
### Step 3: Initializing the Skill
|
||||
|
||||
At this point, it is time to actually create the skill.
|
||||
|
||||
Skip this step only if the skill being developed already exists, and iteration or packaging is needed. In this case, continue to the next step.
|
||||
|
||||
When creating a new skill from scratch, always run the `init_skill.py` script. The script conveniently generates a new template skill directory that automatically includes everything a skill requires, making the skill creation process much more efficient and reliable.
|
||||
|
||||
Usage:
|
||||
|
||||
```bash
|
||||
scripts/init_skill.py <skill-name> --path <output-directory>
|
||||
```
|
||||
|
||||
The script:
|
||||
|
||||
- Creates the skill directory at the specified path
|
||||
- Generates a SKILL.md template with proper frontmatter and TODO placeholders
|
||||
- Creates example resource directories: `scripts/`, `references/`, and `assets/`
|
||||
- Adds example files in each directory that can be customized or deleted
|
||||
|
||||
After initialization, customize or remove the generated SKILL.md and example files as needed.
|
||||
|
||||
### Step 4: Edit the Skill
|
||||
|
||||
When editing the (newly-generated or existing) skill, remember that the skill is being created for another instance of Claude to use. Include information that would be beneficial and non-obvious to Claude. Consider what procedural knowledge, domain-specific details, or reusable assets would help another Claude instance execute these tasks more effectively.
|
||||
|
||||
#### Learn Proven Design Patterns
|
||||
|
||||
Consult these helpful guides based on your skill's needs:
|
||||
|
||||
- **Multi-step processes**: See references/workflows.md for sequential workflows and conditional logic
|
||||
- **Specific output formats or quality standards**: See references/output-patterns.md for template and example patterns
|
||||
|
||||
These files contain established best practices for effective skill design.
|
||||
|
||||
#### Start with Reusable Skill Contents
|
||||
|
||||
To begin implementation, start with the reusable resources identified above: `scripts/`, `references/`, and `assets/` files. Note that this step may require user input. For example, when implementing a `brand-guidelines` skill, the user may need to provide brand assets or templates to store in `assets/`, or documentation to store in `references/`.
|
||||
|
||||
Added scripts must be tested by actually running them to ensure there are no bugs and that the output matches what is expected. If there are many similar scripts, only a representative sample needs to be tested to ensure confidence that they all work while balancing time to completion.
|
||||
|
||||
Any example files and directories not needed for the skill should be deleted. The initialization script creates example files in `scripts/`, `references/`, and `assets/` to demonstrate structure, but most skills won't need all of them.
|
||||
|
||||
#### Update SKILL.md
|
||||
|
||||
**Writing Guidelines:** Always use imperative/infinitive form.
|
||||
|
||||
##### Frontmatter
|
||||
|
||||
Write the YAML frontmatter with `name` and `description`:
|
||||
|
||||
- `name`: The skill name
|
||||
- `description`: This is the primary triggering mechanism for your skill, and helps Claude understand when to use the skill.
|
||||
- Include both what the Skill does and specific triggers/contexts for when to use it.
|
||||
- Include all "when to use" information here - Not in the body. The body is only loaded after triggering, so "When to Use This Skill" sections in the body are not helpful to Claude.
|
||||
- Example description for a `docx` skill: "Comprehensive document creation, editing, and analysis with support for tracked changes, comments, formatting preservation, and text extraction. Use when Claude needs to work with professional documents (.docx files) for: (1) Creating new documents, (2) Modifying or editing content, (3) Working with tracked changes, (4) Adding comments, or any other document tasks"
|
||||
|
||||
Do not include any other fields in YAML frontmatter.
|
||||
|
||||
##### Body
|
||||
|
||||
Write instructions for using the skill and its bundled resources.
|
||||
|
||||
### Step 5: Packaging a Skill
|
||||
|
||||
Once development of the skill is complete, it must be packaged into a distributable .skill file that gets shared with the user. The packaging process automatically validates the skill first to ensure it meets all requirements:
|
||||
|
||||
```bash
|
||||
scripts/package_skill.py <path/to/skill-folder>
|
||||
```
|
||||
|
||||
Optional output directory specification:
|
||||
|
||||
```bash
|
||||
scripts/package_skill.py <path/to/skill-folder> ./dist
|
||||
```
|
||||
|
||||
The packaging script will:
|
||||
|
||||
1. **Validate** the skill automatically, checking:
|
||||
|
||||
- YAML frontmatter format and required fields
|
||||
- Skill naming conventions and directory structure
|
||||
- Description completeness and quality
|
||||
- File organization and resource references
|
||||
|
||||
2. **Package** the skill if validation passes, creating a .skill file named after the skill (e.g., `my-skill.skill`) that includes all files and maintains the proper directory structure for distribution. The .skill file is a zip file with a .skill extension.
|
||||
|
||||
If validation fails, the script will report the errors and exit without creating a package. Fix any validation errors and run the packaging command again.
|
||||
|
||||
### Step 6: Iterate
|
||||
|
||||
After testing the skill, users may request improvements. Often this happens right after using the skill, with fresh context of how the skill performed.
|
||||
|
||||
**Iteration workflow:**
|
||||
|
||||
1. Use the skill on real tasks
|
||||
2. Notice struggles or inefficiencies
|
||||
3. Identify how SKILL.md or bundled resources should be updated
|
||||
4. Implement changes and test again
|
||||
6
runtime/skills/skill-creator/_meta.json
Normal file
6
runtime/skills/skill-creator/_meta.json
Normal file
|
|
@ -0,0 +1,6 @@
|
|||
{
|
||||
"ownerId": "kn7ajm25k9s5t1ajwxp2yfjx818015pb",
|
||||
"slug": "skill-creator",
|
||||
"version": "0.1.0",
|
||||
"publishedAt": 1769522677376
|
||||
}
|
||||
82
runtime/skills/skill-creator/references/output-patterns.md
Normal file
82
runtime/skills/skill-creator/references/output-patterns.md
Normal file
|
|
@ -0,0 +1,82 @@
|
|||
# Output Patterns
|
||||
|
||||
Use these patterns when skills need to produce consistent, high-quality output.
|
||||
|
||||
## Template Pattern
|
||||
|
||||
Provide templates for output format. Match the level of strictness to your needs.
|
||||
|
||||
**For strict requirements (like API responses or data formats):**
|
||||
|
||||
```markdown
|
||||
## Report structure
|
||||
|
||||
ALWAYS use this exact template structure:
|
||||
|
||||
# [Analysis Title]
|
||||
|
||||
## Executive summary
|
||||
[One-paragraph overview of key findings]
|
||||
|
||||
## Key findings
|
||||
- Finding 1 with supporting data
|
||||
- Finding 2 with supporting data
|
||||
- Finding 3 with supporting data
|
||||
|
||||
## Recommendations
|
||||
1. Specific actionable recommendation
|
||||
2. Specific actionable recommendation
|
||||
```
|
||||
|
||||
**For flexible guidance (when adaptation is useful):**
|
||||
|
||||
```markdown
|
||||
## Report structure
|
||||
|
||||
Here is a sensible default format, but use your best judgment:
|
||||
|
||||
# [Analysis Title]
|
||||
|
||||
## Executive summary
|
||||
[Overview]
|
||||
|
||||
## Key findings
|
||||
[Adapt sections based on what you discover]
|
||||
|
||||
## Recommendations
|
||||
[Tailor to the specific context]
|
||||
|
||||
Adjust sections as needed for the specific analysis type.
|
||||
```
|
||||
|
||||
## Examples Pattern
|
||||
|
||||
For skills where output quality depends on seeing examples, provide input/output pairs:
|
||||
|
||||
```markdown
|
||||
## Commit message format
|
||||
|
||||
Generate commit messages following these examples:
|
||||
|
||||
**Example 1:**
|
||||
Input: Added user authentication with JWT tokens
|
||||
Output:
|
||||
```
|
||||
feat(auth): implement JWT-based authentication
|
||||
|
||||
Add login endpoint and token validation middleware
|
||||
```
|
||||
|
||||
**Example 2:**
|
||||
Input: Fixed bug where dates displayed incorrectly in reports
|
||||
Output:
|
||||
```
|
||||
fix(reports): correct date formatting in timezone conversion
|
||||
|
||||
Use UTC timestamps consistently across report generation
|
||||
```
|
||||
|
||||
Follow this style: type(scope): brief description, then detailed explanation.
|
||||
```
|
||||
|
||||
Examples help Claude understand the desired style and level of detail more clearly than descriptions alone.
|
||||
28
runtime/skills/skill-creator/references/workflows.md
Normal file
28
runtime/skills/skill-creator/references/workflows.md
Normal file
|
|
@ -0,0 +1,28 @@
|
|||
# Workflow Patterns
|
||||
|
||||
## Sequential Workflows
|
||||
|
||||
For complex tasks, break operations into clear, sequential steps. It is often helpful to give Claude an overview of the process towards the beginning of SKILL.md:
|
||||
|
||||
```markdown
|
||||
Filling a PDF form involves these steps:
|
||||
|
||||
1. Analyze the form (run analyze_form.py)
|
||||
2. Create field mapping (edit fields.json)
|
||||
3. Validate mapping (run validate_fields.py)
|
||||
4. Fill the form (run fill_form.py)
|
||||
5. Verify output (run verify_output.py)
|
||||
```
|
||||
|
||||
## Conditional Workflows
|
||||
|
||||
For tasks with branching logic, guide Claude through decision points:
|
||||
|
||||
```markdown
|
||||
1. Determine the modification type:
|
||||
**Creating new content?** → Follow "Creation workflow" below
|
||||
**Editing existing content?** → Follow "Editing workflow" below
|
||||
|
||||
2. Creation workflow: [steps]
|
||||
3. Editing workflow: [steps]
|
||||
```
|
||||
303
runtime/skills/skill-creator/scripts/init_skill.py
Normal file
303
runtime/skills/skill-creator/scripts/init_skill.py
Normal file
|
|
@ -0,0 +1,303 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Skill Initializer - Creates a new skill from template
|
||||
|
||||
Usage:
|
||||
init_skill.py <skill-name> --path <path>
|
||||
|
||||
Examples:
|
||||
init_skill.py my-new-skill --path skills/public
|
||||
init_skill.py my-api-helper --path skills/private
|
||||
init_skill.py custom-skill --path /custom/location
|
||||
"""
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
SKILL_TEMPLATE = """---
|
||||
name: {skill_name}
|
||||
description: [TODO: Complete and informative explanation of what the skill does and when to use it. Include WHEN to use this skill - specific scenarios, file types, or tasks that trigger it.]
|
||||
---
|
||||
|
||||
# {skill_title}
|
||||
|
||||
## Overview
|
||||
|
||||
[TODO: 1-2 sentences explaining what this skill enables]
|
||||
|
||||
## Structuring This Skill
|
||||
|
||||
[TODO: Choose the structure that best fits this skill's purpose. Common patterns:
|
||||
|
||||
**1. Workflow-Based** (best for sequential processes)
|
||||
- Works well when there are clear step-by-step procedures
|
||||
- Example: DOCX skill with "Workflow Decision Tree" → "Reading" → "Creating" → "Editing"
|
||||
- Structure: ## Overview → ## Workflow Decision Tree → ## Step 1 → ## Step 2...
|
||||
|
||||
**2. Task-Based** (best for tool collections)
|
||||
- Works well when the skill offers different operations/capabilities
|
||||
- Example: PDF skill with "Quick Start" → "Merge PDFs" → "Split PDFs" → "Extract Text"
|
||||
- Structure: ## Overview → ## Quick Start → ## Task Category 1 → ## Task Category 2...
|
||||
|
||||
**3. Reference/Guidelines** (best for standards or specifications)
|
||||
- Works well for brand guidelines, coding standards, or requirements
|
||||
- Example: Brand styling with "Brand Guidelines" → "Colors" → "Typography" → "Features"
|
||||
- Structure: ## Overview → ## Guidelines → ## Specifications → ## Usage...
|
||||
|
||||
**4. Capabilities-Based** (best for integrated systems)
|
||||
- Works well when the skill provides multiple interrelated features
|
||||
- Example: Product Management with "Core Capabilities" → numbered capability list
|
||||
- Structure: ## Overview → ## Core Capabilities → ### 1. Feature → ### 2. Feature...
|
||||
|
||||
Patterns can be mixed and matched as needed. Most skills combine patterns (e.g., start with task-based, add workflow for complex operations).
|
||||
|
||||
Delete this entire "Structuring This Skill" section when done - it's just guidance.]
|
||||
|
||||
## [TODO: Replace with the first main section based on chosen structure]
|
||||
|
||||
[TODO: Add content here. See examples in existing skills:
|
||||
- Code samples for technical skills
|
||||
- Decision trees for complex workflows
|
||||
- Concrete examples with realistic user requests
|
||||
- References to scripts/templates/references as needed]
|
||||
|
||||
## Resources
|
||||
|
||||
This skill includes example resource directories that demonstrate how to organize different types of bundled resources:
|
||||
|
||||
### scripts/
|
||||
Executable code (Python/Bash/etc.) that can be run directly to perform specific operations.
|
||||
|
||||
**Examples from other skills:**
|
||||
- PDF skill: `fill_fillable_fields.py`, `extract_form_field_info.py` - utilities for PDF manipulation
|
||||
- DOCX skill: `document.py`, `utilities.py` - Python modules for document processing
|
||||
|
||||
**Appropriate for:** Python scripts, shell scripts, or any executable code that performs automation, data processing, or specific operations.
|
||||
|
||||
**Note:** Scripts may be executed without loading into context, but can still be read by Claude for patching or environment adjustments.
|
||||
|
||||
### references/
|
||||
Documentation and reference material intended to be loaded into context to inform Claude's process and thinking.
|
||||
|
||||
**Examples from other skills:**
|
||||
- Product management: `communication.md`, `context_building.md` - detailed workflow guides
|
||||
- BigQuery: API reference documentation and query examples
|
||||
- Finance: Schema documentation, company policies
|
||||
|
||||
**Appropriate for:** In-depth documentation, API references, database schemas, comprehensive guides, or any detailed information that Claude should reference while working.
|
||||
|
||||
### assets/
|
||||
Files not intended to be loaded into context, but rather used within the output Claude produces.
|
||||
|
||||
**Examples from other skills:**
|
||||
- Brand styling: PowerPoint template files (.pptx), logo files
|
||||
- Frontend builder: HTML/React boilerplate project directories
|
||||
- Typography: Font files (.ttf, .woff2)
|
||||
|
||||
**Appropriate for:** Templates, boilerplate code, document templates, images, icons, fonts, or any files meant to be copied or used in the final output.
|
||||
|
||||
---
|
||||
|
||||
**Any unneeded directories can be deleted.** Not every skill requires all three types of resources.
|
||||
"""
|
||||
|
||||
EXAMPLE_SCRIPT = '''#!/usr/bin/env python3
|
||||
"""
|
||||
Example helper script for {skill_name}
|
||||
|
||||
This is a placeholder script that can be executed directly.
|
||||
Replace with actual implementation or delete if not needed.
|
||||
|
||||
Example real scripts from other skills:
|
||||
- pdf/scripts/fill_fillable_fields.py - Fills PDF form fields
|
||||
- pdf/scripts/convert_pdf_to_images.py - Converts PDF pages to images
|
||||
"""
|
||||
|
||||
def main():
|
||||
print("This is an example script for {skill_name}")
|
||||
# TODO: Add actual script logic here
|
||||
# This could be data processing, file conversion, API calls, etc.
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
'''
|
||||
|
||||
EXAMPLE_REFERENCE = """# Reference Documentation for {skill_title}
|
||||
|
||||
This is a placeholder for detailed reference documentation.
|
||||
Replace with actual reference content or delete if not needed.
|
||||
|
||||
Example real reference docs from other skills:
|
||||
- product-management/references/communication.md - Comprehensive guide for status updates
|
||||
- product-management/references/context_building.md - Deep-dive on gathering context
|
||||
- bigquery/references/ - API references and query examples
|
||||
|
||||
## When Reference Docs Are Useful
|
||||
|
||||
Reference docs are ideal for:
|
||||
- Comprehensive API documentation
|
||||
- Detailed workflow guides
|
||||
- Complex multi-step processes
|
||||
- Information too lengthy for main SKILL.md
|
||||
- Content that's only needed for specific use cases
|
||||
|
||||
## Structure Suggestions
|
||||
|
||||
### API Reference Example
|
||||
- Overview
|
||||
- Authentication
|
||||
- Endpoints with examples
|
||||
- Error codes
|
||||
- Rate limits
|
||||
|
||||
### Workflow Guide Example
|
||||
- Prerequisites
|
||||
- Step-by-step instructions
|
||||
- Common patterns
|
||||
- Troubleshooting
|
||||
- Best practices
|
||||
"""
|
||||
|
||||
EXAMPLE_ASSET = """# Example Asset File
|
||||
|
||||
This placeholder represents where asset files would be stored.
|
||||
Replace with actual asset files (templates, images, fonts, etc.) or delete if not needed.
|
||||
|
||||
Asset files are NOT intended to be loaded into context, but rather used within
|
||||
the output Claude produces.
|
||||
|
||||
Example asset files from other skills:
|
||||
- Brand guidelines: logo.png, slides_template.pptx
|
||||
- Frontend builder: hello-world/ directory with HTML/React boilerplate
|
||||
- Typography: custom-font.ttf, font-family.woff2
|
||||
- Data: sample_data.csv, test_dataset.json
|
||||
|
||||
## Common Asset Types
|
||||
|
||||
- Templates: .pptx, .docx, boilerplate directories
|
||||
- Images: .png, .jpg, .svg, .gif
|
||||
- Fonts: .ttf, .otf, .woff, .woff2
|
||||
- Boilerplate code: Project directories, starter files
|
||||
- Icons: .ico, .svg
|
||||
- Data files: .csv, .json, .xml, .yaml
|
||||
|
||||
Note: This is a text placeholder. Actual assets can be any file type.
|
||||
"""
|
||||
|
||||
|
||||
def title_case_skill_name(skill_name):
|
||||
"""Convert hyphenated skill name to Title Case for display."""
|
||||
return ' '.join(word.capitalize() for word in skill_name.split('-'))
|
||||
|
||||
|
||||
def init_skill(skill_name, path):
|
||||
"""
|
||||
Initialize a new skill directory with template SKILL.md.
|
||||
|
||||
Args:
|
||||
skill_name: Name of the skill
|
||||
path: Path where the skill directory should be created
|
||||
|
||||
Returns:
|
||||
Path to created skill directory, or None if error
|
||||
"""
|
||||
# Determine skill directory path
|
||||
skill_dir = Path(path).resolve() / skill_name
|
||||
|
||||
# Check if directory already exists
|
||||
if skill_dir.exists():
|
||||
print(f"❌ Error: Skill directory already exists: {skill_dir}")
|
||||
return None
|
||||
|
||||
# Create skill directory
|
||||
try:
|
||||
skill_dir.mkdir(parents=True, exist_ok=False)
|
||||
print(f"✅ Created skill directory: {skill_dir}")
|
||||
except Exception as e:
|
||||
print(f"❌ Error creating directory: {e}")
|
||||
return None
|
||||
|
||||
# Create SKILL.md from template
|
||||
skill_title = title_case_skill_name(skill_name)
|
||||
skill_content = SKILL_TEMPLATE.format(
|
||||
skill_name=skill_name,
|
||||
skill_title=skill_title
|
||||
)
|
||||
|
||||
skill_md_path = skill_dir / 'SKILL.md'
|
||||
try:
|
||||
skill_md_path.write_text(skill_content)
|
||||
print("✅ Created SKILL.md")
|
||||
except Exception as e:
|
||||
print(f"❌ Error creating SKILL.md: {e}")
|
||||
return None
|
||||
|
||||
# Create resource directories with example files
|
||||
try:
|
||||
# Create scripts/ directory with example script
|
||||
scripts_dir = skill_dir / 'scripts'
|
||||
scripts_dir.mkdir(exist_ok=True)
|
||||
example_script = scripts_dir / 'example.py'
|
||||
example_script.write_text(EXAMPLE_SCRIPT.format(skill_name=skill_name))
|
||||
example_script.chmod(0o755)
|
||||
print("✅ Created scripts/example.py")
|
||||
|
||||
# Create references/ directory with example reference doc
|
||||
references_dir = skill_dir / 'references'
|
||||
references_dir.mkdir(exist_ok=True)
|
||||
example_reference = references_dir / 'api_reference.md'
|
||||
example_reference.write_text(EXAMPLE_REFERENCE.format(skill_title=skill_title))
|
||||
print("✅ Created references/api_reference.md")
|
||||
|
||||
# Create assets/ directory with example asset placeholder
|
||||
assets_dir = skill_dir / 'assets'
|
||||
assets_dir.mkdir(exist_ok=True)
|
||||
example_asset = assets_dir / 'example_asset.txt'
|
||||
example_asset.write_text(EXAMPLE_ASSET)
|
||||
print("✅ Created assets/example_asset.txt")
|
||||
except Exception as e:
|
||||
print(f"❌ Error creating resource directories: {e}")
|
||||
return None
|
||||
|
||||
# Print next steps
|
||||
print(f"\n✅ Skill '{skill_name}' initialized successfully at {skill_dir}")
|
||||
print("\nNext steps:")
|
||||
print("1. Edit SKILL.md to complete the TODO items and update the description")
|
||||
print("2. Customize or delete the example files in scripts/, references/, and assets/")
|
||||
print("3. Run the validator when ready to check the skill structure")
|
||||
|
||||
return skill_dir
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 4 or sys.argv[2] != '--path':
|
||||
print("Usage: init_skill.py <skill-name> --path <path>")
|
||||
print("\nSkill name requirements:")
|
||||
print(" - Hyphen-case identifier (e.g., 'data-analyzer')")
|
||||
print(" - Lowercase letters, digits, and hyphens only")
|
||||
print(" - Max 40 characters")
|
||||
print(" - Must match directory name exactly")
|
||||
print("\nExamples:")
|
||||
print(" init_skill.py my-new-skill --path skills/public")
|
||||
print(" init_skill.py my-api-helper --path skills/private")
|
||||
print(" init_skill.py custom-skill --path /custom/location")
|
||||
sys.exit(1)
|
||||
|
||||
skill_name = sys.argv[1]
|
||||
path = sys.argv[3]
|
||||
|
||||
print(f"🚀 Initializing skill: {skill_name}")
|
||||
print(f" Location: {path}")
|
||||
print()
|
||||
|
||||
result = init_skill(skill_name, path)
|
||||
|
||||
if result:
|
||||
sys.exit(0)
|
||||
else:
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
110
runtime/skills/skill-creator/scripts/package_skill.py
Normal file
110
runtime/skills/skill-creator/scripts/package_skill.py
Normal file
|
|
@ -0,0 +1,110 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Skill Packager - Creates a distributable .skill file of a skill folder
|
||||
|
||||
Usage:
|
||||
python utils/package_skill.py <path/to/skill-folder> [output-directory]
|
||||
|
||||
Example:
|
||||
python utils/package_skill.py skills/public/my-skill
|
||||
python utils/package_skill.py skills/public/my-skill ./dist
|
||||
"""
|
||||
|
||||
import sys
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
from quick_validate import validate_skill
|
||||
|
||||
|
||||
def package_skill(skill_path, output_dir=None):
|
||||
"""
|
||||
Package a skill folder into a .skill file.
|
||||
|
||||
Args:
|
||||
skill_path: Path to the skill folder
|
||||
output_dir: Optional output directory for the .skill file (defaults to current directory)
|
||||
|
||||
Returns:
|
||||
Path to the created .skill file, or None if error
|
||||
"""
|
||||
skill_path = Path(skill_path).resolve()
|
||||
|
||||
# Validate skill folder exists
|
||||
if not skill_path.exists():
|
||||
print(f"❌ Error: Skill folder not found: {skill_path}")
|
||||
return None
|
||||
|
||||
if not skill_path.is_dir():
|
||||
print(f"❌ Error: Path is not a directory: {skill_path}")
|
||||
return None
|
||||
|
||||
# Validate SKILL.md exists
|
||||
skill_md = skill_path / "SKILL.md"
|
||||
if not skill_md.exists():
|
||||
print(f"❌ Error: SKILL.md not found in {skill_path}")
|
||||
return None
|
||||
|
||||
# Run validation before packaging
|
||||
print("🔍 Validating skill...")
|
||||
valid, message = validate_skill(skill_path)
|
||||
if not valid:
|
||||
print(f"❌ Validation failed: {message}")
|
||||
print(" Please fix the validation errors before packaging.")
|
||||
return None
|
||||
print(f"✅ {message}\n")
|
||||
|
||||
# Determine output location
|
||||
skill_name = skill_path.name
|
||||
if output_dir:
|
||||
output_path = Path(output_dir).resolve()
|
||||
output_path.mkdir(parents=True, exist_ok=True)
|
||||
else:
|
||||
output_path = Path.cwd()
|
||||
|
||||
skill_filename = output_path / f"{skill_name}.skill"
|
||||
|
||||
# Create the .skill file (zip format)
|
||||
try:
|
||||
with zipfile.ZipFile(skill_filename, 'w', zipfile.ZIP_DEFLATED) as zipf:
|
||||
# Walk through the skill directory
|
||||
for file_path in skill_path.rglob('*'):
|
||||
if file_path.is_file():
|
||||
# Calculate the relative path within the zip
|
||||
arcname = file_path.relative_to(skill_path.parent)
|
||||
zipf.write(file_path, arcname)
|
||||
print(f" Added: {arcname}")
|
||||
|
||||
print(f"\n✅ Successfully packaged skill to: {skill_filename}")
|
||||
return skill_filename
|
||||
|
||||
except Exception as e:
|
||||
print(f"❌ Error creating .skill file: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python utils/package_skill.py <path/to/skill-folder> [output-directory]")
|
||||
print("\nExample:")
|
||||
print(" python utils/package_skill.py skills/public/my-skill")
|
||||
print(" python utils/package_skill.py skills/public/my-skill ./dist")
|
||||
sys.exit(1)
|
||||
|
||||
skill_path = sys.argv[1]
|
||||
output_dir = sys.argv[2] if len(sys.argv) > 2 else None
|
||||
|
||||
print(f"📦 Packaging skill: {skill_path}")
|
||||
if output_dir:
|
||||
print(f" Output directory: {output_dir}")
|
||||
print()
|
||||
|
||||
result = package_skill(skill_path, output_dir)
|
||||
|
||||
if result:
|
||||
sys.exit(0)
|
||||
else:
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
95
runtime/skills/skill-creator/scripts/quick_validate.py
Normal file
95
runtime/skills/skill-creator/scripts/quick_validate.py
Normal file
|
|
@ -0,0 +1,95 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Quick validation script for skills - minimal version
|
||||
"""
|
||||
|
||||
import sys
|
||||
import os
|
||||
import re
|
||||
import yaml
|
||||
from pathlib import Path
|
||||
|
||||
def validate_skill(skill_path):
|
||||
"""Basic validation of a skill"""
|
||||
skill_path = Path(skill_path)
|
||||
|
||||
# Check SKILL.md exists
|
||||
skill_md = skill_path / 'SKILL.md'
|
||||
if not skill_md.exists():
|
||||
return False, "SKILL.md not found"
|
||||
|
||||
# Read and validate frontmatter
|
||||
content = skill_md.read_text()
|
||||
if not content.startswith('---'):
|
||||
return False, "No YAML frontmatter found"
|
||||
|
||||
# Extract frontmatter
|
||||
match = re.match(r'^---\n(.*?)\n---', content, re.DOTALL)
|
||||
if not match:
|
||||
return False, "Invalid frontmatter format"
|
||||
|
||||
frontmatter_text = match.group(1)
|
||||
|
||||
# Parse YAML frontmatter
|
||||
try:
|
||||
frontmatter = yaml.safe_load(frontmatter_text)
|
||||
if not isinstance(frontmatter, dict):
|
||||
return False, "Frontmatter must be a YAML dictionary"
|
||||
except yaml.YAMLError as e:
|
||||
return False, f"Invalid YAML in frontmatter: {e}"
|
||||
|
||||
# Define allowed properties
|
||||
ALLOWED_PROPERTIES = {'name', 'description', 'license', 'allowed-tools', 'metadata'}
|
||||
|
||||
# Check for unexpected properties (excluding nested keys under metadata)
|
||||
unexpected_keys = set(frontmatter.keys()) - ALLOWED_PROPERTIES
|
||||
if unexpected_keys:
|
||||
return False, (
|
||||
f"Unexpected key(s) in SKILL.md frontmatter: {', '.join(sorted(unexpected_keys))}. "
|
||||
f"Allowed properties are: {', '.join(sorted(ALLOWED_PROPERTIES))}"
|
||||
)
|
||||
|
||||
# Check required fields
|
||||
if 'name' not in frontmatter:
|
||||
return False, "Missing 'name' in frontmatter"
|
||||
if 'description' not in frontmatter:
|
||||
return False, "Missing 'description' in frontmatter"
|
||||
|
||||
# Extract name for validation
|
||||
name = frontmatter.get('name', '')
|
||||
if not isinstance(name, str):
|
||||
return False, f"Name must be a string, got {type(name).__name__}"
|
||||
name = name.strip()
|
||||
if name:
|
||||
# Check naming convention (hyphen-case: lowercase with hyphens)
|
||||
if not re.match(r'^[a-z0-9-]+$', name):
|
||||
return False, f"Name '{name}' should be hyphen-case (lowercase letters, digits, and hyphens only)"
|
||||
if name.startswith('-') or name.endswith('-') or '--' in name:
|
||||
return False, f"Name '{name}' cannot start/end with hyphen or contain consecutive hyphens"
|
||||
# Check name length (max 64 characters per spec)
|
||||
if len(name) > 64:
|
||||
return False, f"Name is too long ({len(name)} characters). Maximum is 64 characters."
|
||||
|
||||
# Extract and validate description
|
||||
description = frontmatter.get('description', '')
|
||||
if not isinstance(description, str):
|
||||
return False, f"Description must be a string, got {type(description).__name__}"
|
||||
description = description.strip()
|
||||
if description:
|
||||
# Check for angle brackets
|
||||
if '<' in description or '>' in description:
|
||||
return False, "Description cannot contain angle brackets (< or >)"
|
||||
# Check description length (max 1024 characters per spec)
|
||||
if len(description) > 1024:
|
||||
return False, f"Description is too long ({len(description)} characters). Maximum is 1024 characters."
|
||||
|
||||
return True, "Skill is valid!"
|
||||
|
||||
if __name__ == "__main__":
|
||||
if len(sys.argv) != 2:
|
||||
print("Usage: python quick_validate.py <skill_directory>")
|
||||
sys.exit(1)
|
||||
|
||||
valid, message = validate_skill(sys.argv[1])
|
||||
print(message)
|
||||
sys.exit(0 if valid else 1)
|
||||
|
|
@ -1,97 +0,0 @@
|
|||
---
|
||||
name: SkillScan
|
||||
metadata:
|
||||
version: "1.1.6"
|
||||
description: >
|
||||
Security gate for skills. Every new skill MUST pass SkillScan before use.
|
||||
Activate on any install, load, add, evaluate, or safety question about a skill.
|
||||
On first load, run first-run to scan all existing skills.
|
||||
Blocks HIGH/CRITICAL skills. No exceptions.
|
||||
---
|
||||
|
||||
# SkillScan v1.1.6
|
||||
|
||||
Security check for skill packages. Every new skill must pass this scan before use.
|
||||
|
||||
Implemented in `scripts/scanner.py`. Supports Windows / macOS / Linux.
|
||||
|
||||
Do NOT generate report files inside the skill directory.
|
||||
|
||||
---
|
||||
|
||||
## When to Activate
|
||||
|
||||
Run SkillScan whenever a skill is installed, loaded, added, evaluated, or questioned for safety — regardless of method or phrasing.
|
||||
|
||||
Trigger examples: install, load, add, use, set up, "is this safe", upload .zip/.skill, paste SKILL.md, mention unknown skill name, install from hub/store.
|
||||
|
||||
---
|
||||
|
||||
## First Load
|
||||
|
||||
Check `scripts/.first_run_done`:
|
||||
|
||||
- **Not found** → **Tell user** SkillScan is installed, ask to scan all existing skills:
|
||||
```bash
|
||||
python <skill_scanner_dir>/scripts/scanner.py first-run
|
||||
```
|
||||
`.first_run_done` is created automatically after completion.
|
||||
|
||||
- **Found** → Skip, operate normally.
|
||||
|
||||
---
|
||||
|
||||
## Commands
|
||||
|
||||
| Command | Usage |
|
||||
|---------|-------|
|
||||
| `scanner.py scan <path>` | Scan a single skill (.zip or directory) |
|
||||
| `scanner.py scan-all` | Scan all installed skills |
|
||||
| `scanner.py first-run` | First-time full scan |
|
||||
| `scanner.py upgrade` | Manual upgrade |
|
||||
|
||||
---
|
||||
|
||||
## Scan Rules
|
||||
|
||||
- **.zip files** → Scan BEFORE installation. Block if fails.
|
||||
- **Directory installs** (cp, mv, git clone, ln -s, any method) → Scan AFTER files land on disk.
|
||||
- **Remote installs** (clawhub, skillhub, npx skills add, etc.) → Scan immediately after install.
|
||||
- **Unknown skills** → If user mentions a skill you haven't seen, scan it.
|
||||
|
||||
---
|
||||
|
||||
## Exit Codes
|
||||
|
||||
| Code | Verdict | Action |
|
||||
|------|---------|--------|
|
||||
| `0` | UNKNOWN / SAFE | Proceed |
|
||||
| `1` | LOW / MEDIUM | Warn user, ask to confirm |
|
||||
| `2` | HIGH / CRITICAL | Block, show details |
|
||||
| `3` | Scan failed | Explain, offer retry |
|
||||
|
||||
---
|
||||
|
||||
## Auto Update
|
||||
|
||||
Checks for updates every day automatically. Silent, no user action needed. Manual: `scanner.py upgrade`.
|
||||
|
||||
---
|
||||
|
||||
## Environment Variables
|
||||
|
||||
| Variable | Description |
|
||||
|----------|-------------|
|
||||
| `SKILL_SCANNER_UPDATE_URL` | Custom update source (optional) |
|
||||
|
||||
---
|
||||
|
||||
## API
|
||||
|
||||
Base URL: `https://skillscan.tokauth.com`
|
||||
|
||||
| Step | Method | Path |
|
||||
|------|--------|------|
|
||||
| ① Cache lookup | GET | `/oapi/v1/skill-scan/search?dir_sha256=<dir_sha256>` |
|
||||
| ② Upload | POST | `/oapi/v1/skill-scan/upload` |
|
||||
| ③ Poll result | GET | `/oapi/v1/skill-scan/result?task_no=<task_no>` (poll every 20s, max 180s) |
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
{
|
||||
"ownerId": "kn791cyx98pcsezkh5088g8jxn84c7mm",
|
||||
"slug": "skillscan",
|
||||
"version": "1.1.6",
|
||||
"publishedAt": 1776650587310
|
||||
}
|
||||
|
|
@ -1,959 +0,0 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
SkillScan v1.1.5 — OpenClaw Skill security scanner.
|
||||
Supports Windows / macOS / Linux. All temp files use the standard tempfile module.
|
||||
|
||||
Usage (invoked by the agent via bash):
|
||||
python scanner.py first-run # First install: list installed skills and ask to scan
|
||||
python scanner.py scan <path> # Scan a single skill (.zip or directory)
|
||||
python scanner.py scan-all # Scan all installed skills
|
||||
python scanner.py upgrade # Auto-upgrade
|
||||
"""
|
||||
|
||||
import sys, os, json, time, zipfile, hashlib, shutil, tempfile, uuid, platform, base64
|
||||
import urllib.request, urllib.error, urllib.parse
|
||||
from pathlib import Path
|
||||
from datetime import datetime, timezone
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Configuration
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
SCANNER_VERSION = "1.1.5"
|
||||
|
||||
BASE_URL = "https://skillscan.tokauth.com"
|
||||
API_SEARCH = f"{BASE_URL}/oapi/v1/skill-scan/search"
|
||||
API_UPLOAD = f"{BASE_URL}/oapi/v1/skill-scan/upload"
|
||||
API_RESULT = f"{BASE_URL}/oapi/v1/skill-scan/result"
|
||||
UPDATE_URL = os.environ.get("SKILL_SCANNER_UPDATE_URL",
|
||||
f"{BASE_URL}/downloads/SkillScan/manifest")
|
||||
|
||||
POLL_INTERVAL = 20 # Poll interval (seconds)
|
||||
POLL_TIMEOUT = 180 # Max wait time (seconds)
|
||||
|
||||
# First-run marker file (in the same directory as scanner.py)
|
||||
STATE_FILE = Path(__file__).parent / ".first_run_done"
|
||||
|
||||
# Auto-update check marker file and interval (7 days)
|
||||
LAST_UPDATE_CHECK_FILE = Path(__file__).parent / ".last_update_check"
|
||||
AUTO_UPDATE_INTERVAL = 1 * 24 * 3600 # 1 day (seconds)
|
||||
|
||||
# Client info file (generated on first run, reused afterwards)
|
||||
CLIENT_INFO_FILE = Path(__file__).parent / ".client_info"
|
||||
|
||||
# Files and directories to skip during scanning, hashing, and packing
|
||||
SKIP_FILES = {".first_run_done", ".last_update_check", ".client_info", "cloud_report.json", ".DS_Store"}
|
||||
SKIP_DIRS = {".git", "__pycache__", ".venv", "node_modules", ".idea", ".vscode", ".clawhub"}
|
||||
|
||||
# Resolve the root directory of SkillScan itself (parent of scripts/)
|
||||
SELF_ROOT = Path(__file__).parent.parent.resolve()
|
||||
|
||||
|
||||
# Skill installation paths (cross-platform)
|
||||
def skill_install_paths():
|
||||
# type: () -> list
|
||||
"""Auto-enumerate OpenClaw and local skill paths across platforms."""
|
||||
home = Path.home()
|
||||
oc_dir = home / ".openclaw"
|
||||
candidates = [
|
||||
# OpenClaw standard paths
|
||||
oc_dir / "skills",
|
||||
oc_dir / "workspace/skills",
|
||||
# Shared agent skill paths
|
||||
home / ".agents/skills",
|
||||
home / ".config/agents/skills",
|
||||
# Agent-specific global paths
|
||||
home / ".gemini/antigravity/skills",
|
||||
home / ".gemini/skills",
|
||||
home / ".augment/skills",
|
||||
home / ".claude/skills",
|
||||
home / ".codex/skills",
|
||||
home / ".commandcode/skills",
|
||||
home / ".continue/skills",
|
||||
home / ".snowflake/cortex/skills",
|
||||
home / ".config/crush/skills",
|
||||
home / ".cursor/skills",
|
||||
home / ".deepagents/agent/skills",
|
||||
home / ".factory/skills",
|
||||
home / ".firebender/skills",
|
||||
home / ".copilot/skills",
|
||||
home / ".config/goose/skills",
|
||||
home / ".junie/skills",
|
||||
home / ".iflow/skills",
|
||||
home / ".kilocode/skills",
|
||||
home / ".kiro/skills",
|
||||
home / ".kode/skills",
|
||||
home / ".mcpjam/skills",
|
||||
home / ".vibe/skills",
|
||||
home / ".mux/skills",
|
||||
home / ".config/opencode/skills",
|
||||
home / ".oh/skills",
|
||||
home / ".pi/agent/skills",
|
||||
home / ".qoder/skills",
|
||||
home / ".qwen/skills",
|
||||
home / ".roo/skills",
|
||||
home / ".trae/skills",
|
||||
home / ".trae-cn/skills",
|
||||
home / ".codeium/windsurf/skills",
|
||||
home / ".zencoder/skills",
|
||||
home / ".neovate/skills",
|
||||
home / ".pochi/skills",
|
||||
home / ".adal/skills",
|
||||
home / ".npm-global/lib/node_modules/openclaw/skills",
|
||||
# Container default paths
|
||||
Path("/mnt/skills/public"),
|
||||
Path("/mnt/skills/private"),
|
||||
Path("/mnt/skills/user"),
|
||||
# User dev/download paths
|
||||
home / "Downloads/skills",
|
||||
]
|
||||
|
||||
# Windows-specific paths
|
||||
if os.name == "nt":
|
||||
appdata = os.environ.get("APPDATA")
|
||||
if appdata:
|
||||
candidates.append(Path(appdata) / "OpenClaw/skills")
|
||||
candidates.append(Path(appdata) / "Programs/LobsterAI/resources/SKILLs")
|
||||
|
||||
# Dynamically scan extensions: .openclaw/extensions/{xxxx}/skills
|
||||
if oc_dir.exists():
|
||||
ext_root = oc_dir / "extensions"
|
||||
if ext_root.exists():
|
||||
for sub in ext_root.iterdir():
|
||||
if sub.is_dir():
|
||||
s_dir = sub / "skills"
|
||||
if s_dir.exists():
|
||||
candidates.append(s_dir)
|
||||
|
||||
# Include script run path and workspace
|
||||
candidates.append(Path.cwd() / "skills")
|
||||
candidates.append(Path(__file__).parent.parent / "skills")
|
||||
|
||||
# Deduplicate and filter non-existent paths
|
||||
seen = set()
|
||||
result = []
|
||||
for p in candidates:
|
||||
try:
|
||||
abs_p = p.resolve()
|
||||
if abs_p.exists() and abs_p not in seen:
|
||||
result.append(p)
|
||||
seen.add(abs_p)
|
||||
except Exception:
|
||||
continue
|
||||
return result
|
||||
|
||||
RISK_EMOJI = {"SAFE":"✅","LOW":"⚠️ ","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️ "}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Client Info (X-Client-Info)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _get_mac_address():
|
||||
"""Try to get the MAC address; return empty string on failure."""
|
||||
try:
|
||||
import uuid as _uuid
|
||||
mac_int = _uuid.getnode()
|
||||
# getnode() returns a random value (bit 8 set) when it can't get the real MAC
|
||||
if (mac_int >> 40) & 1:
|
||||
return ""
|
||||
mac_str = ":".join(("%012X" % mac_int)[i:i+2] for i in range(0, 12, 2))
|
||||
return mac_str
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def _build_client_info():
|
||||
"""Build client info dict and persist to file; reuse on subsequent runs."""
|
||||
# If a record file already exists, read it
|
||||
if CLIENT_INFO_FILE.exists():
|
||||
try:
|
||||
data = json.loads(CLIENT_INFO_FILE.read_text(encoding="utf-8"))
|
||||
if data.get("client_id"):
|
||||
return data
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# First run: generate new client info
|
||||
info = {
|
||||
"client_id": str(uuid.uuid4()),
|
||||
"os": platform.system() or "",
|
||||
"platform": platform.machine() or "",
|
||||
"os_version": platform.release() or "",
|
||||
"client": "SkillScanner/%s" % SCANNER_VERSION,
|
||||
}
|
||||
|
||||
mac = _get_mac_address()
|
||||
if mac:
|
||||
info["mac"] = mac
|
||||
|
||||
# Python version as extra
|
||||
info["extra"] = {
|
||||
"python": platform.python_version(),
|
||||
}
|
||||
|
||||
# Persist
|
||||
try:
|
||||
CLIENT_INFO_FILE.write_text(
|
||||
json.dumps(info, ensure_ascii=False, indent=2),
|
||||
encoding="utf-8"
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
return info
|
||||
|
||||
|
||||
def _get_client_info_header():
|
||||
"""Return Base64-encoded X-Client-Info header value; empty string on failure."""
|
||||
try:
|
||||
info = _build_client_info()
|
||||
json_str = json.dumps(info, ensure_ascii=False)
|
||||
encoded = base64.b64encode(json_str.encode("utf-8")).decode("ascii")
|
||||
return encoded
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Output Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def banner(title: str):
|
||||
w = 58
|
||||
print(f"\n{'═'*w}")
|
||||
print(f" {title}")
|
||||
print(f"{'═'*w}")
|
||||
|
||||
def divider(title: str = ""):
|
||||
if title:
|
||||
print(f"\n ── {title} {'─'*(48-len(title))}")
|
||||
else:
|
||||
print(f" {'─'*52}")
|
||||
|
||||
def log(msg: str):
|
||||
print(f" {msg}", flush=True)
|
||||
|
||||
def ask(prompt: str) -> str:
|
||||
"""Read user input (compatible with non-interactive environments)."""
|
||||
try:
|
||||
return input(f"\n {prompt} ").strip()
|
||||
except (EOFError, KeyboardInterrupt):
|
||||
return ""
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# HTTP Helpers
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def http_get(url: str) -> dict:
|
||||
req = urllib.request.Request(url)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
def http_post(url: str, payload: dict) -> dict:
|
||||
headers = {"Content-Type": "application/json"}
|
||||
data = json.dumps(payload, ensure_ascii=False).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
return json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Skill Utilities
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def skill_name_from_dir(skill_dir: Path) -> str:
|
||||
md = skill_dir / "SKILL.md"
|
||||
if md.exists():
|
||||
for line in md.read_text(encoding="utf-8", errors="replace").splitlines():
|
||||
s = line.strip()
|
||||
if s.startswith("name:"):
|
||||
return s.split(":", 1)[1].strip().strip("\"'")
|
||||
return skill_dir.name
|
||||
|
||||
def sha256_of(path: Path) -> str:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
|
||||
def calculate_dir_sha256(directory: Path) -> str:
|
||||
"""Calculate SHA256 hash of a skill directory (based on all file contents + relative paths).
|
||||
Excludes _meta.json and files/dirs in SKIP_FILES/SKIP_DIRS."""
|
||||
file_hashes = []
|
||||
for file_path in sorted(directory.rglob('*')):
|
||||
if not file_path.is_file():
|
||||
continue
|
||||
if file_path.name == '_meta.json':
|
||||
continue
|
||||
rel = file_path.relative_to(directory)
|
||||
if any(part in SKIP_DIRS for part in rel.parts):
|
||||
continue
|
||||
if file_path.name in SKIP_FILES:
|
||||
continue
|
||||
rel_path = str(rel)
|
||||
file_hash = hashlib.sha256()
|
||||
file_hash.update(rel_path.encode('utf-8'))
|
||||
file_hash.update(b'\x00')
|
||||
with open(file_path, 'rb') as f:
|
||||
for chunk in iter(lambda: f.read(8192), b''):
|
||||
file_hash.update(chunk)
|
||||
file_hashes.append(file_hash.hexdigest())
|
||||
file_hashes.sort()
|
||||
final_hash = hashlib.sha256()
|
||||
for h in file_hashes:
|
||||
final_hash.update(h.encode('utf-8'))
|
||||
final_hash.update(b'\x00')
|
||||
return final_hash.hexdigest()
|
||||
|
||||
def collect_files(skill_dir: Path) -> dict:
|
||||
"""Collect files for scanning, skipping redundant or sensitive directories."""
|
||||
exts = {".md",".py",".js",".ts",".sh",".yaml",".yml",".json",".txt"}
|
||||
out = {}
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
if p.suffix.lower() in exts or p.name == "SKILL.md":
|
||||
try:
|
||||
out[str(p.relative_to(skill_dir))] = \
|
||||
p.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
def pack_zip(skill_dir: Path) -> bytes:
|
||||
"""Pack a skill directory into a zip byte stream, excluding redundant directories."""
|
||||
import io
|
||||
buf = io.BytesIO()
|
||||
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
||||
for p in sorted(skill_dir.rglob("*")):
|
||||
if any(part in SKIP_DIRS for part in p.relative_to(skill_dir).parts):
|
||||
continue
|
||||
if p.is_file() and p.name not in SKIP_FILES:
|
||||
zf.write(p, p.relative_to(skill_dir))
|
||||
return buf.getvalue()
|
||||
|
||||
def unpack_zip(zip_path: Path) -> Path:
|
||||
"""Extract a .zip to a system temp directory. Returns the extraction path. Prevents zip-slip."""
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
log(f"📦 Extracting {zip_path.name} → {tmp}")
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / member).resolve()
|
||||
if not str(dest).startswith(str(tmp.resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp)
|
||||
return tmp
|
||||
|
||||
def find_installed_skills():
|
||||
# type: () -> list
|
||||
"""Find all installed skill directories (first-level subdirectories containing SKILL.md).
|
||||
Excludes SkillScan itself."""
|
||||
found = set()
|
||||
for base in skill_install_paths():
|
||||
if not base.exists():
|
||||
continue
|
||||
for md in base.rglob("SKILL.md"):
|
||||
skill_path = md.parent
|
||||
try:
|
||||
rel = skill_path.relative_to(base)
|
||||
if len(rel.parts) == 1:
|
||||
resolved = skill_path.resolve()
|
||||
# Skip self
|
||||
if resolved == SELF_ROOT:
|
||||
continue
|
||||
found.add(resolved)
|
||||
except ValueError:
|
||||
pass
|
||||
return sorted(found)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Scan Core (3 steps)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _extract_result(resp, sha256):
|
||||
"""Internal: extract core data from API response, handling SHA256 wrapping/nested result."""
|
||||
# 1. Handle API response keyed by SHA256 (e.g. { "sha256": { "status": "success", "data": {...} } })
|
||||
if sha256 and sha256 in resp:
|
||||
resp = resp[sha256]
|
||||
|
||||
# 2. Extract data body (data or result)
|
||||
data = resp.get("data") or resp.get("result") or resp
|
||||
|
||||
# 3. Handle nested result inside data
|
||||
if isinstance(data, dict) and "result" in data:
|
||||
inner = data["result"]
|
||||
if isinstance(inner, dict):
|
||||
# Merge sibling metadata (analysis_level/reason etc.) into result
|
||||
for k, v in data.items():
|
||||
if k != "result" and k not in inner:
|
||||
inner[k] = v
|
||||
return inner
|
||||
|
||||
return data if isinstance(data, dict) and (data.get("verdict") or data.get("is_safe") is not None or data.get("analysis_level")) else None
|
||||
|
||||
|
||||
def cloud_search(dir_sha256):
|
||||
"""Step 1: Query scan cache by dir_sha256. Returns result dict or None."""
|
||||
extra_headers = {}
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
extra_headers["X-Client-Info"] = ci
|
||||
|
||||
url = "%s?%s" % (API_SEARCH, urllib.parse.urlencode({"dir_sha256": dir_sha256}))
|
||||
try:
|
||||
headers = {}
|
||||
headers.update(extra_headers)
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=30) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
res = _extract_result(resp, dir_sha256)
|
||||
if res:
|
||||
log(" ✅ Cache hit (dir_sha256 %s…)" % dir_sha256[:16])
|
||||
return res
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 404:
|
||||
return None
|
||||
raise RuntimeError("Search API error HTTP %d" % e.code)
|
||||
except urllib.error.URLError as e:
|
||||
raise RuntimeError("Cannot connect to server: %s" % e)
|
||||
return None
|
||||
|
||||
|
||||
def cloud_upload(skill_dir, name, dir_hash):
|
||||
"""Step 2: Upload skill (multipart/form-data), returns task_no."""
|
||||
# Pack the entire directory for full code context
|
||||
zip_data = pack_zip(skill_dir)
|
||||
filename = "%s.zip" % name
|
||||
|
||||
# Build multipart/form-data boundary
|
||||
boundary = "----WebKitFormBoundary%s" % uuid.uuid4().hex
|
||||
|
||||
# Manually construct multipart byte stream (no requests library needed)
|
||||
parts = []
|
||||
parts.append(("--%s" % boundary).encode())
|
||||
parts.append(('Content-Disposition: form-data; name="file"; filename="%s"' % filename).encode())
|
||||
parts.append(b"Content-Type: application/zip")
|
||||
parts.append(b"")
|
||||
parts.append(zip_data)
|
||||
parts.append(("--%s--" % boundary).encode())
|
||||
parts.append(b"") # trailing newline
|
||||
|
||||
body = b"\r\n".join(parts)
|
||||
|
||||
headers = {
|
||||
"Content-Type": "multipart/form-data; boundary=%s" % boundary,
|
||||
"Content-Length": str(len(body)),
|
||||
"Accept": "application/json"
|
||||
}
|
||||
|
||||
# Add X-Client-Info header
|
||||
ci = _get_client_info_header()
|
||||
if ci:
|
||||
headers["X-Client-Info"] = ci
|
||||
|
||||
log(" 📤 Uploading: %s (%.1f KB)..." % (filename, len(zip_data) / 1024.0))
|
||||
req = urllib.request.Request(API_UPLOAD, data=body, headers=headers, method="POST")
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
resp = json.loads(r.read().decode("utf-8", errors="replace"))
|
||||
except urllib.error.HTTPError as e:
|
||||
err_body = e.read().decode(errors="replace")
|
||||
raise RuntimeError("Upload failed HTTP %d: %s" % (e.code, err_body))
|
||||
|
||||
task_no = (resp.get("data") or {}).get("task_no") or resp.get("task_no") or resp.get("taskNo") or resp.get("task_id") or ""
|
||||
if not task_no:
|
||||
raise RuntimeError("Upload succeeded but no valid task_no in response: %s" % resp)
|
||||
|
||||
log(" ✅ Upload complete, task_no: %s" % task_no)
|
||||
return str(task_no)
|
||||
|
||||
|
||||
def cloud_poll(task_no: str) -> dict:
|
||||
"""Step 3: Poll until complete or timeout. Queries every 20s.
|
||||
status: 0=pending, 1=scanning, 2=completed, 3=failed, 4=cancelled
|
||||
"""
|
||||
url = f"{API_RESULT}?{urllib.parse.urlencode({'task_no': task_no})}"
|
||||
deadline = time.time() + POLL_TIMEOUT
|
||||
attempt = 0
|
||||
while time.time() < deadline:
|
||||
attempt += 1
|
||||
elapsed = int(time.time() - (deadline - POLL_TIMEOUT))
|
||||
try:
|
||||
resp = http_get(url)
|
||||
data = resp.get("data") or resp
|
||||
status = data.get("status")
|
||||
|
||||
if status == 2: # completed
|
||||
print()
|
||||
log(f" ✅ Scan complete (attempt {attempt}, {elapsed}s elapsed)")
|
||||
return _extract_result(resp, "") or resp
|
||||
elif status == 3: # failed
|
||||
print()
|
||||
err_msg = data.get("error_message") or resp.get("message", "unknown error")
|
||||
raise RuntimeError(f"Analysis failed: {err_msg}")
|
||||
elif status == 4: # cancelled
|
||||
print()
|
||||
raise RuntimeError("Scan task was cancelled")
|
||||
else:
|
||||
# 0=pending, 1=scanning -> keep waiting
|
||||
status_text = data.get("status_text", "processing")
|
||||
print(f" ⏳ [{status_text}] attempt {attempt}, {elapsed}s / {POLL_TIMEOUT}s elapsed",
|
||||
end="\r", flush=True)
|
||||
time.sleep(POLL_INTERVAL)
|
||||
except (RuntimeError, ValueError):
|
||||
raise
|
||||
except Exception as e:
|
||||
raise RuntimeError(f"Poll error: {e}")
|
||||
print()
|
||||
raise RuntimeError(f"Timeout ({POLL_TIMEOUT}s), task_no={task_no}, please retry later")
|
||||
|
||||
|
||||
def cloud_check(skill_dir: Path) -> dict:
|
||||
"""Run full security scan on a skill directory, return normalized result."""
|
||||
md = skill_dir / "SKILL.md"
|
||||
if not md.exists():
|
||||
raise FileNotFoundError(f"SKILL.md not found: {skill_dir}")
|
||||
|
||||
name = skill_name_from_dir(skill_dir)
|
||||
dir_hash = calculate_dir_sha256(skill_dir)
|
||||
log(f"🔍 Scanning: {name}")
|
||||
log(f" dir_sha256: {dir_hash}")
|
||||
|
||||
log(f"🔎 [1/3] Checking scan cache...")
|
||||
raw = cloud_search(dir_hash)
|
||||
|
||||
if raw is None:
|
||||
log(f" ℹ️ No cache record, submitting new scan task")
|
||||
log(f"📤 [2/3] Uploading skill for analysis...")
|
||||
task_no = cloud_upload(skill_dir, name, dir_hash)
|
||||
log(f"⏳ [3/3] Waiting for analysis (polling every {POLL_INTERVAL}s, max {POLL_TIMEOUT}s)...")
|
||||
raw = cloud_poll(task_no)
|
||||
else:
|
||||
log(f" ⏭️ Skipping upload, using cached result")
|
||||
|
||||
return _normalize(raw, name, dir_hash)
|
||||
|
||||
|
||||
def _normalize(raw: dict, name: str, dir_hash: str) -> dict:
|
||||
"""Normalize scan result:
|
||||
1. Extract is_safe (bool) and max_severity (str).
|
||||
2. Map API-specific fields (analysis_reason, analysis_suggestion) to standard fields.
|
||||
"""
|
||||
is_safe = raw.get("is_safe")
|
||||
|
||||
# Severity field priority: max_severity > analysis_level > verdict > level
|
||||
v_raw = (raw.get("max_severity") or raw.get("analysis_level") or
|
||||
raw.get("verdict") or raw.get("risk_level") or
|
||||
raw.get("level") or "UNKNOWN").upper()
|
||||
|
||||
# Combined verdict logic
|
||||
if is_safe is True and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "SAFE"
|
||||
elif is_safe is False and v_raw in ("UNKNOWN", "SAFE"):
|
||||
verdict = "CRITICAL" # Explicitly marked unsafe -> critical
|
||||
else:
|
||||
verdict = v_raw
|
||||
|
||||
return {
|
||||
"skill_name": name,
|
||||
"dir_sha256": dir_hash,
|
||||
"verdict": verdict,
|
||||
"confidence": raw.get("confidence") or raw.get("score"),
|
||||
"threat_labels": raw.get("threat_labels") or raw.get("tags") or [],
|
||||
"summary": raw.get("analysis_reason") or raw.get("summary") or raw.get("description") or "",
|
||||
"findings": raw.get("findings") or raw.get("issues") or [],
|
||||
"recommendation":raw.get("analysis_suggestion") or raw.get("recommendation") or raw.get("action") or "",
|
||||
}
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Result Display
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def print_result(r: dict):
|
||||
verdict = r.get("verdict","UNKNOWN")
|
||||
emoji = RISK_EMOJI.get(verdict,"❓")
|
||||
conf = r.get("confidence")
|
||||
labels = r.get("threat_labels",[])
|
||||
summary = r.get("summary","")
|
||||
findings = r.get("findings",[])
|
||||
rec = r.get("recommendation","")
|
||||
conf_str = f" confidence {float(conf):.0%}" if conf is not None else ""
|
||||
|
||||
divider()
|
||||
log(f"{emoji} Result: {verdict}{conf_str}")
|
||||
if summary:
|
||||
log(f"📋 {summary}")
|
||||
if labels:
|
||||
log(f"🏷️ Threat labels: {', '.join(labels)}")
|
||||
if findings:
|
||||
SEV = {"LOW":"🔵","MEDIUM":"🟡","HIGH":"🔴","CRITICAL":"☠️"}
|
||||
log(f"🔍 Findings ({len(findings)} items):")
|
||||
for f in findings:
|
||||
sev = str(f.get("severity","")).upper()
|
||||
desc = f.get("description") or f.get("detail") or str(f)
|
||||
rid = f.get("id") or ""
|
||||
tag = f"[{rid}] " if rid else ""
|
||||
log(f" {SEV.get(sev,'⚪')} {tag}{desc}")
|
||||
if rec:
|
||||
log(f"💡 Recommendation: {rec}")
|
||||
divider()
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Prompt: malicious detected -> ask whether to delete
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def prompt_delete(skill_path: Path, result: dict) -> bool:
|
||||
"""When result is HIGH/CRITICAL, ask user whether to delete the skill.
|
||||
skill_path is the original install path (not temp dir).
|
||||
Returns True if deleted.
|
||||
"""
|
||||
verdict = result.get("verdict","")
|
||||
if verdict not in ("HIGH","CRITICAL"):
|
||||
return False
|
||||
|
||||
if not skill_path or not skill_path.exists():
|
||||
return False
|
||||
|
||||
emoji = RISK_EMOJI.get(verdict,"🔴")
|
||||
log(f"\n{emoji} This skill is marked as [{verdict}] high risk by security scan.")
|
||||
log(f" Path: {skill_path}")
|
||||
|
||||
answer = ask("Delete this skill now? [y/n]")
|
||||
if answer in ("y","Y","yes","Yes"):
|
||||
try:
|
||||
if skill_path.is_dir():
|
||||
shutil.rmtree(skill_path)
|
||||
else:
|
||||
skill_path.unlink()
|
||||
log(f"✅ Deleted: {skill_path}")
|
||||
return True
|
||||
except Exception as e:
|
||||
log(f"❌ Delete failed: {e} (please delete manually)")
|
||||
return False
|
||||
else:
|
||||
log(f"⚠️ Skipped deletion. Use this skill with caution.")
|
||||
return False
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: first-run (first install)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_first_run():
|
||||
"""First install: list installed skills, ask user to scan, show results."""
|
||||
if STATE_FILE.exists():
|
||||
log("ℹ️ First-run scan already completed. Use scan-all to rescan.")
|
||||
return
|
||||
|
||||
banner("🛡️ SkillScan First-Run Check")
|
||||
log("Welcome to SkillScan!")
|
||||
log("Searching for installed skills...\n")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("✅ No installed skills found, nothing to scan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Print installed skill list
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name}")
|
||||
|
||||
answer = ask("Run security scan on all listed skills? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Skipped. You can run scan-all anytime to rescan.")
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
return
|
||||
|
||||
# Scan one by one
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
# Copy to temp dir (source may be read-only)
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
print_result(r)
|
||||
|
||||
# High risk -> ask to delete (targeting original install path)
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name,
|
||||
"verdict": "ERROR", "threat_labels": [],
|
||||
"summary": str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
STATE_FILE.write_text(datetime.now(timezone.utc).isoformat(), encoding="utf-8")
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan (single skill)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan(path_str: str):
|
||||
skill_path = Path(path_str)
|
||||
if not skill_path.exists():
|
||||
log(f"❌ Path not found: {skill_path}")
|
||||
sys.exit(1)
|
||||
|
||||
banner(f"Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
tmp = None
|
||||
original_path = skill_path if skill_path.is_dir() else None
|
||||
try:
|
||||
if skill_path.is_file():
|
||||
if skill_path.suffix.lower() not in (".zip",):
|
||||
log(f"❌ Unsupported format: {skill_path.suffix} (use .zip)")
|
||||
sys.exit(1)
|
||||
tmp = unpack_zip(skill_path)
|
||||
scan_dir = tmp
|
||||
else:
|
||||
scan_dir = skill_path
|
||||
|
||||
result = cloud_check(scan_dir)
|
||||
print_result(result)
|
||||
|
||||
# High risk -> ask to delete
|
||||
if original_path:
|
||||
prompt_delete(original_path, result)
|
||||
elif skill_path.is_file() and result.get("verdict") in ("HIGH","CRITICAL"):
|
||||
# Zip file: ask to delete source file
|
||||
prompt_delete(skill_path, result)
|
||||
|
||||
v = result.get("verdict","UNKNOWN")
|
||||
sys.exit(0 if v in ("SAFE","LOW") else 1 if v=="MEDIUM" else 2)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"\n❌ Scan failed: {e}")
|
||||
sys.exit(3)
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: scan-all
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_scan_all():
|
||||
banner(f"Full Skill Security Scan v{SCANNER_VERSION}")
|
||||
|
||||
skills = find_installed_skills()
|
||||
if not skills:
|
||||
log("ℹ️ No installed skills detected.")
|
||||
return
|
||||
|
||||
log(f"Found {len(skills)} installed skill(s):\n")
|
||||
for i, s in enumerate(skills, 1):
|
||||
log(f" {i:2d}. {s.name:<30} {s}")
|
||||
|
||||
answer = ask("Start security scan? [y/n]")
|
||||
if answer not in ("y","Y","yes","Yes"):
|
||||
log("Cancelled.")
|
||||
return
|
||||
|
||||
results = []
|
||||
for idx, skill_path in enumerate(skills, 1):
|
||||
divider(f"[{idx}/{len(skills)}] {skill_path.name}")
|
||||
tmp = None
|
||||
try:
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillscan-"))
|
||||
scan_dir = tmp / skill_path.name
|
||||
shutil.copytree(skill_path, scan_dir)
|
||||
|
||||
r = cloud_check(scan_dir)
|
||||
v = r.get("verdict","UNKNOWN")
|
||||
log(f"{RISK_EMOJI.get(v,'❓')} Scan complete: {v}")
|
||||
if r.get("threat_labels"):
|
||||
log(f" Threat labels: {', '.join(r['threat_labels'])}")
|
||||
|
||||
# High risk: ask to delete
|
||||
prompt_delete(skill_path, r)
|
||||
results.append(r)
|
||||
|
||||
except RuntimeError as e:
|
||||
log(f"❌ Scan failed: {e}")
|
||||
results.append({"skill_name": skill_path.name, "verdict":"ERROR",
|
||||
"threat_labels":[], "summary":str(e)[:100]})
|
||||
finally:
|
||||
if tmp:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
_print_summary(results)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Summary Table
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def _print_summary(results):
|
||||
banner("📊 Scan Summary")
|
||||
print(f" {'Skill Name':<28} {'Result':<12} {'Threat Labels'}")
|
||||
divider()
|
||||
for r in results:
|
||||
v = r.get("verdict","?")
|
||||
name = r.get("skill_name","?")[:27]
|
||||
labels = ", ".join(r.get("threat_labels",[]))[:20] or "-"
|
||||
print(f" {name:<28} {RISK_EMOJI.get(v,'❓')}{v:<10} {labels}")
|
||||
|
||||
safes = [r for r in results if r["verdict"] in {"SAFE","LOW"}]
|
||||
mediums = [r for r in results if r["verdict"] == "MEDIUM"]
|
||||
highs = [r for r in results if r["verdict"] in {"HIGH","CRITICAL"}]
|
||||
errors = [r for r in results if r["verdict"] in {"ERROR","UNKNOWN"}]
|
||||
|
||||
print()
|
||||
log(f"Total {len(results)} | ✅ Safe {len(safes)} "
|
||||
f"🟡 Suspicious {len(mediums)} 🔴 Dangerous {len(highs)} ❓ Error {len(errors)}")
|
||||
if highs:
|
||||
log(f"\n⚠️ High-risk skills: {', '.join(r['skill_name'] for r in highs)}")
|
||||
elif not mediums and not errors:
|
||||
log("\n🎉 All skills passed security scan.")
|
||||
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Subcommand: upgrade
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def cmd_upgrade():
|
||||
banner("SkillScan Auto-Upgrade")
|
||||
log(f"Current version: {SCANNER_VERSION}")
|
||||
log(f"Update source: {UPDATE_URL}")
|
||||
try:
|
||||
manifest = http_get(UPDATE_URL)
|
||||
except Exception as e:
|
||||
log(f"❌ Failed to fetch update manifest: {e}")
|
||||
return
|
||||
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f"✅ Already up to date ({SCANNER_VERSION})")
|
||||
return
|
||||
|
||||
log(f"New version found: {SCANNER_VERSION} → {latest}")
|
||||
log(f"Changelog: {manifest.get('changelog','(none)')}")
|
||||
|
||||
download_url = manifest.get("download_url", "")
|
||||
if not download_url:
|
||||
log("⚠️ No download URL in manifest, skipping upgrade")
|
||||
return
|
||||
|
||||
# Download new version zip
|
||||
log(f"📥 Downloading: {download_url}")
|
||||
try:
|
||||
req = urllib.request.Request(download_url)
|
||||
with urllib.request.urlopen(req, timeout=60) as r:
|
||||
zip_data = r.read()
|
||||
except Exception as e:
|
||||
log(f"❌ Download failed: {e}")
|
||||
return
|
||||
|
||||
# SHA256 verification
|
||||
expected_sha = manifest.get("sha256", "")
|
||||
if expected_sha:
|
||||
actual_sha = hashlib.sha256(zip_data).hexdigest()
|
||||
if actual_sha != expected_sha:
|
||||
log(f"❌ SHA256 mismatch, upgrade aborted (expected {expected_sha[:16]}…, got {actual_sha[:16]}…)")
|
||||
return
|
||||
log(f" ✅ SHA256 verified")
|
||||
|
||||
# Backup current skill directory
|
||||
skill_root = Path(__file__).parent.parent
|
||||
backup_dir = skill_root.parent / f"SkillScan-backup-{SCANNER_VERSION}"
|
||||
if backup_dir.exists():
|
||||
shutil.rmtree(backup_dir)
|
||||
shutil.copytree(skill_root, backup_dir)
|
||||
log(f"📦 Backed up to: {backup_dir}")
|
||||
|
||||
# Extract and replace files
|
||||
tmp = Path(tempfile.mkdtemp(prefix="skillupgrade-"))
|
||||
try:
|
||||
zip_path = tmp / "update.zip"
|
||||
zip_path.write_bytes(zip_data)
|
||||
with zipfile.ZipFile(zip_path, "r") as zf:
|
||||
# Security check: prevent zip-slip
|
||||
for member in zf.namelist():
|
||||
dest = (tmp / "extracted" / member).resolve()
|
||||
if not str(dest).startswith(str((tmp / "extracted").resolve())):
|
||||
raise ValueError(f"zip-slip path rejected: {member}")
|
||||
zf.extractall(tmp / "extracted")
|
||||
|
||||
# Overwrite skill directory with new files
|
||||
extracted = tmp / "extracted"
|
||||
for item in extracted.rglob("*"):
|
||||
if not item.is_file():
|
||||
continue
|
||||
rel = item.relative_to(extracted)
|
||||
target = skill_root / rel
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
shutil.copy2(item, target)
|
||||
log(f" ✅ Updated: {rel}")
|
||||
|
||||
log(f"🎉 Upgraded to v{latest}")
|
||||
except Exception as e:
|
||||
log(f"❌ Upgrade failed: {e}")
|
||||
log(f" You can restore from backup: {backup_dir}")
|
||||
finally:
|
||||
shutil.rmtree(tmp, ignore_errors=True)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Entry Point
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def auto_upgrade_if_needed():
|
||||
"""Auto-check for updates every 7 days, runs silently."""
|
||||
try:
|
||||
if LAST_UPDATE_CHECK_FILE.exists():
|
||||
last_check = float(LAST_UPDATE_CHECK_FILE.read_text(encoding="utf-8").strip())
|
||||
if time.time() - last_check < AUTO_UPDATE_INTERVAL:
|
||||
return # Not time to check yet
|
||||
log("🔄 Checking for updates...")
|
||||
manifest = http_get(UPDATE_URL)
|
||||
latest = manifest.get("version", SCANNER_VERSION)
|
||||
if (tuple(int(x) for x in latest.split(".")) <=
|
||||
tuple(int(x) for x in SCANNER_VERSION.split("."))):
|
||||
log(f" ✅ Already up to date ({SCANNER_VERSION})")
|
||||
else:
|
||||
log(f" New version found: {SCANNER_VERSION} → {latest}, auto-updating...")
|
||||
cmd_upgrade()
|
||||
LAST_UPDATE_CHECK_FILE.write_text(str(time.time()), encoding="utf-8")
|
||||
except Exception as e:
|
||||
log(f" ⚠️ Auto-update check failed: {e} (normal operation unaffected)")
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print(__doc__)
|
||||
sys.exit(0)
|
||||
|
||||
# Check for auto-update on every run (once every 7 days)
|
||||
auto_upgrade_if_needed()
|
||||
|
||||
cmd = sys.argv[1]
|
||||
if cmd == "first-run":
|
||||
cmd_first_run()
|
||||
elif cmd == "scan":
|
||||
if len(sys.argv) < 3:
|
||||
log("Usage: scanner.py scan <skill_path>")
|
||||
sys.exit(1)
|
||||
cmd_scan(sys.argv[2])
|
||||
elif cmd == "scan-all":
|
||||
cmd_scan_all()
|
||||
elif cmd == "upgrade":
|
||||
cmd_upgrade()
|
||||
else:
|
||||
log(f"Unknown command: {cmd}")
|
||||
log("Available commands: first-run / scan <path> / scan-all / upgrade")
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,208 +0,0 @@
|
|||
# Word Reader 技能开发完成
|
||||
|
||||
## 🎯 技能概述
|
||||
|
||||
成功创建了一个功能完整的 Word 文档读取技能,支持读取 .docx 和 .doc 格式的 Word 文档,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## 📁 技能结构
|
||||
|
||||
```
|
||||
word-reader/
|
||||
├── SKILL.md # 技能定义文件
|
||||
├── README.md # 使用说明
|
||||
├── skill.json # 技能配置
|
||||
├── demo.sh # 演示脚本
|
||||
├── install.sh # 安装脚本
|
||||
├── test.md # 测试文档
|
||||
└── scripts/
|
||||
└── read_word.py # 核心脚本
|
||||
```
|
||||
|
||||
## ✨ 主要功能
|
||||
|
||||
### 1. 文档解析能力
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
|
||||
### 2. 格式支持
|
||||
- ✅ **.docx** - Office 2007+ 格式(主要支持)
|
||||
- ✅ **.doc** - 旧版 Word 格式(需要 antiword)
|
||||
|
||||
### 3. 输出格式
|
||||
- ✅ **JSON** - 结构化数据,适合程序处理
|
||||
- ✅ **Text** - 纯文本格式,简单易读
|
||||
- ✅ **Markdown** - 格式化输出,保留文档结构
|
||||
|
||||
### 4. 高级功能
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **选择性提取** - 可只提取特定内容类型
|
||||
- ✅ **文件输出** - 支持保存结果到文件
|
||||
- ✅ **编码支持** - 支持多种文本编码
|
||||
|
||||
## 🚀 使用示例
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# JSON 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# Markdown 格式输出
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存结果
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 🔧 安装和配置
|
||||
|
||||
### 自动安装
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx
|
||||
|
||||
# 安装系统依赖(可选)
|
||||
sudo apt-get install antiword # Ubuntu/Debian
|
||||
brew install antiword # macOS
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "文档.docx",
|
||||
"title": "文档标题",
|
||||
"author": "作者",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "文档内容...",
|
||||
"tables": [...],
|
||||
"images": [...]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
```markdown
|
||||
# 文档.docx
|
||||
|
||||
**标题**:文档标题
|
||||
**作者**:作者
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
文档内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 数据1 | 数据2 |
|
||||
```
|
||||
|
||||
## 🎨 技能特点
|
||||
|
||||
### 1. 智能错误处理
|
||||
- 友好的错误提示
|
||||
- 自动检测文档格式
|
||||
- 优雅的异常处理
|
||||
|
||||
### 2. 性能优化
|
||||
- 流式处理大文件
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
|
||||
### 3. 用户友好
|
||||
- 详细的帮助信息
|
||||
- 多种使用方式
|
||||
- 完整的文档说明
|
||||
|
||||
### 4. 可扩展性
|
||||
- 模块化设计
|
||||
- 易于添加新功能
|
||||
- 支持自定义输出格式
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
### 1. 文档内容分析
|
||||
- 快速查看 Word 文档内容
|
||||
- 提取特定信息
|
||||
- 文档摘要生成
|
||||
|
||||
### 2. 批量处理
|
||||
- 处理大量文档
|
||||
- 文档格式转换
|
||||
- 内容索引创建
|
||||
|
||||
### 3. 自动化工作流
|
||||
- 集到文档处理系统
|
||||
- 自动化文档分析
|
||||
- 内容管理系统集成
|
||||
|
||||
## 📝 开发总结
|
||||
|
||||
### 实现的功能
|
||||
- 完整的 Word 文档解析框架
|
||||
- 支持多种输出格式
|
||||
- 批量处理能力
|
||||
- 错误处理和用户友好性
|
||||
|
||||
### 技术亮点
|
||||
- 模块化设计,易于维护
|
||||
- 优雅的错误处理机制
|
||||
- 支持多种文件格式
|
||||
- 灵活的输出选项
|
||||
|
||||
### 改进空间
|
||||
- 可以添加 PDF 支持
|
||||
- 可以增加图片提取功能
|
||||
- 可以优化大文件处理性能
|
||||
- 可以添加更多文档元素支持
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
要发布此技能到 ClawHub,可以运行:
|
||||
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm i -g clawhub
|
||||
|
||||
# 登录
|
||||
clawhub login
|
||||
|
||||
# 发布技能
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "Initial release with .docx and .doc support" \
|
||||
--tags document,word,office,text-extraction
|
||||
```
|
||||
|
||||
这个技能现在已经准备好使用了!它可以帮助用户轻松读取和处理 Word 文档,支持多种格式和输出选项。
|
||||
|
|
@ -1,177 +0,0 @@
|
|||
# Word Reader 技能发布指南
|
||||
|
||||
## 🚀 发布到 ClawHub
|
||||
|
||||
### 1. 准备工作
|
||||
|
||||
#### 确保技能完整
|
||||
- [ ] SKILL.md 文件完整且格式正确
|
||||
- [ ] 脚本功能正常
|
||||
- [ ] 安装脚本工作正常
|
||||
- [ ] README.md 说明清晰
|
||||
- [ ] 所有依赖已在 SKILL.md 中声明
|
||||
|
||||
#### 环境准备
|
||||
```bash
|
||||
# 安装 ClawHub CLI
|
||||
npm install -g clawhub
|
||||
# 或
|
||||
pnpm add -g clawhub
|
||||
```
|
||||
|
||||
#### 登录 ClawHub
|
||||
```bash
|
||||
# 登录(会打开浏览器进行 OAuth 认证)
|
||||
clawhub login
|
||||
|
||||
# 验证登录状态
|
||||
clawhub whoami
|
||||
```
|
||||
|
||||
> **注意**:GitHub 账号需要注册满一周才能发布技能
|
||||
|
||||
### 2. 发布流程
|
||||
|
||||
#### 检查技能
|
||||
```bash
|
||||
# 验证技能结构
|
||||
clawhub validate ./word-reader
|
||||
```
|
||||
|
||||
#### 发布技能
|
||||
```bash
|
||||
clawhub publish ./word-reader \
|
||||
--slug word-reader \
|
||||
--name "Word Reader" \
|
||||
--version 1.0.0 \
|
||||
--changelog "支持 .docx 和 .doc 格式的 Word 文档读取,提取文本、表格、元数据等" \
|
||||
--tags document,word,office,text-extraction,reader,parsing \
|
||||
--license MIT \
|
||||
--visibility public
|
||||
```
|
||||
|
||||
#### 参数说明
|
||||
- `--slug`: URL 友好的唯一标识符
|
||||
- `--name`: 技能显示名称
|
||||
- `--version`: 遵循语义化版本控制
|
||||
- `--changelog`: 版本变更说明
|
||||
- `--tags`: 搜索标签(逗号分隔)
|
||||
- `--license`: 许可证类型
|
||||
- `--visibility`: public/private
|
||||
|
||||
### 3. 发布后操作
|
||||
|
||||
#### 验证发布
|
||||
```bash
|
||||
# 查看已发布的技能
|
||||
clawhub search word-reader
|
||||
|
||||
# 安装测试
|
||||
clawhub install word-reader-test
|
||||
```
|
||||
|
||||
#### 分享技能
|
||||
- 技能将在 `https://clawhub.com/skills/word-reader` 可见
|
||||
- 其他用户可通过 `clawhub install word-reader` 安装
|
||||
|
||||
### 4. 版本管理
|
||||
|
||||
#### 更新技能
|
||||
```bash
|
||||
# 修改技能后更新版本号
|
||||
clawhub publish ./word-reader --version 1.0.1 --changelog "修复了某些文档格式的解析问题"
|
||||
```
|
||||
|
||||
#### 批量操作
|
||||
```bash
|
||||
# 同步所有技能
|
||||
clawhub sync --all
|
||||
|
||||
# 发布并标记
|
||||
clawhub publish ./word-reader --tags latest,stable
|
||||
```
|
||||
|
||||
### 5. 自动化发布
|
||||
|
||||
#### GitHub Actions 示例
|
||||
```yaml
|
||||
name: Publish Skill
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v3
|
||||
with:
|
||||
node-version: '18'
|
||||
|
||||
- name: Install ClawHub CLI
|
||||
run: npm install -g clawhub
|
||||
|
||||
- name: Login to ClawHub
|
||||
run: echo "${{ secrets.CLAWHUB_TOKEN }}" | clawhub login --token
|
||||
|
||||
- name: Publish Skill
|
||||
run: |
|
||||
clawhub publish ./skills/word-reader \
|
||||
--slug word-reader \
|
||||
--version ${{ github.ref_name }} \
|
||||
--changelog "Published from GitHub Actions"
|
||||
```
|
||||
|
||||
### 6. 发布注意事项
|
||||
|
||||
#### 必须遵守的规则
|
||||
- [ ] 技能名称不能与其他技能冲突
|
||||
- [ ] 版本号遵循 SemVer 规范
|
||||
- [ ] changelog 清晰描述变更
|
||||
- [ ] 代码无安全漏洞
|
||||
- [ ] 许可证声明清晰
|
||||
|
||||
#### 最佳实践
|
||||
- [ ] 发布前充分测试
|
||||
- [ ] 提供清晰的使用示例
|
||||
- [ ] 维护更新日志
|
||||
- [ ] 及时修复问题
|
||||
- [ ] 关注用户反馈
|
||||
|
||||
### 7. 故障排除
|
||||
|
||||
#### 常见问题
|
||||
```bash
|
||||
# 验证发布权限
|
||||
clawhub whoami
|
||||
|
||||
# 检查技能格式
|
||||
clawhub validate ./word-reader
|
||||
|
||||
# 查看详细错误信息
|
||||
clawhub publish ./word-reader --verbose
|
||||
```
|
||||
|
||||
#### 重新发布
|
||||
如果发布失败,可以:
|
||||
1. 修正问题
|
||||
2. 增加版本号
|
||||
3. 重新发布
|
||||
|
||||
### 8. 维护指南
|
||||
|
||||
#### 监控使用情况
|
||||
- 定期查看下载统计
|
||||
- 关注用户反馈
|
||||
- 及时修复问题
|
||||
|
||||
#### 更新策略
|
||||
- 重要修复:紧急发布补丁版本
|
||||
- 新功能:发布次版本号
|
||||
- 重大变更:发布主版本号
|
||||
|
||||
现在你的 Word Reader 技能已经准备好发布到 ClawHub 了!
|
||||
|
|
@ -1,171 +0,0 @@
|
|||
# Word Reader 技能
|
||||
|
||||
## 📋 概述
|
||||
|
||||
Word Reader 是一个强大的 Word 文档读取工具,支持 .docx 和 .doc 格式,能够提取文本内容、表格数据、文档元信息,并提供多种输出格式。
|
||||
|
||||
## ✨ 功能特性
|
||||
|
||||
- ✅ **文本提取** - 提取文档中的所有段落文本
|
||||
- ✅ **表格解析** - 解析表格数据并转换为结构化格式
|
||||
- ✅ **元数据获取** - 读取文档属性(标题、作者、创建时间等)
|
||||
- ✅ **图片信息** - 获取文档中图片的基本信息
|
||||
- ✅ **多格式支持** - 支持 .docx 和 .doc 格式
|
||||
- ✅ **多种输出** - JSON、Text、Markdown 格式
|
||||
- ✅ **批量处理** - 支持处理整个目录的文档
|
||||
- ✅ **自动安装** - 一键安装所有依赖
|
||||
|
||||
## 🚀 安装
|
||||
|
||||
### 自动安装(推荐)
|
||||
```bash
|
||||
cd word-reader/
|
||||
./install.sh
|
||||
```
|
||||
|
||||
### 手动安装
|
||||
```bash
|
||||
# 安装 Python 依赖
|
||||
pip3 install python-docx --break-system-packages
|
||||
|
||||
# 安装系统依赖(可选,用于 .doc 格式支持)
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
|
||||
# 设置执行权限
|
||||
chmod +x scripts/read_word.py
|
||||
```
|
||||
|
||||
## 📖 使用方法
|
||||
|
||||
### 基本用法
|
||||
```bash
|
||||
# 读取文档并输出为文本格式
|
||||
python3 scripts/read_word.py 文档.docx
|
||||
|
||||
# 输出为 JSON 格式
|
||||
python3 scripts/read_word.py 文档.docx --format json
|
||||
|
||||
# 输出为 Markdown 格式
|
||||
python3 scripts/read_word.py 文档.docx --format markdown
|
||||
|
||||
# 只提取文本内容
|
||||
python3 scripts/read_word.py 文档.docx --extract text
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
```bash
|
||||
# 批量处理目录下所有 Word 文档
|
||||
python3 scripts/read_word.py ./文档目录 --batch
|
||||
|
||||
# 批量处理并保存为 JSON 文件
|
||||
python3 scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
### 高级用法
|
||||
```bash
|
||||
# 将结果保存到文件
|
||||
python3 scripts/read_word.py 文档.docx --format markdown --output output.md
|
||||
|
||||
# 提取表格数据
|
||||
python3 scripts/read_word.py 文档.docx --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 scripts/read_word.py 文档.docx --extract metadata
|
||||
```
|
||||
|
||||
## 📊 输出示例
|
||||
|
||||
### JSON 格式输出
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"filename": "测试文档.docx",
|
||||
"size": "2048 bytes",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00",
|
||||
"title": "测试文档",
|
||||
"author": "测试用户"
|
||||
},
|
||||
"format": "docx",
|
||||
"text": "这是文档的正文内容...",
|
||||
"tables": [
|
||||
{
|
||||
"id": 1,
|
||||
"rows": 3,
|
||||
"columns": 3,
|
||||
"data": [
|
||||
["表头1", "表头2", "表头3"],
|
||||
["数据1", "数据2", "数据3"],
|
||||
["数据4", "数据5", "数据6"]
|
||||
]
|
||||
}
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"id": "rId1",
|
||||
"filename": "image1.png",
|
||||
"size": "1024 bytes"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式输出
|
||||
```markdown
|
||||
# 测试文档.docx
|
||||
|
||||
**标题**:测试文档
|
||||
**作者**:测试用户
|
||||
**文件大小**:2048 bytes
|
||||
**创建时间**:2024-01-01T10:00:00
|
||||
**修改时间**:2024-01-01T12:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
## 表格内容
|
||||
|
||||
### 表格 1 (3行 x 3列)
|
||||
|
||||
| 表头1 | 表头2 | 表头3 |
|
||||
|-------|-------|-------|
|
||||
| 数据1 | 数据2 | 数据3 |
|
||||
| 数据4 | 数据5 | 数据6 |
|
||||
```
|
||||
|
||||
## 🎯 应用场景
|
||||
|
||||
- **文档内容分析** - 快速查看 Word 文档内容
|
||||
- **批量处理** - 处理大量文档
|
||||
- **内容提取** - 提取特定信息
|
||||
- **格式转换** - 转换为其他格式
|
||||
- **自动化工作流** - 集成到文档处理系统
|
||||
|
||||
## 📤 发布到 ClawHub
|
||||
|
||||
要将此技能发布到 ClawHub,请参考 `PUBLISHING.md` 文件。
|
||||
|
||||
## 🔧 故障排除
|
||||
|
||||
### 常见问题
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **FileNotFoundError**: 确认文件路径正确
|
||||
4. **编码问题**: 尝试使用 `--encoding gb2312` 参数
|
||||
|
||||
### 性能优化
|
||||
- 大文档处理时建议使用 `--format json` 以获得更好的性能
|
||||
- 批量模式下建议使用 `--output` 参数将结果保存到文件
|
||||
|
||||
## 🤝 贡献
|
||||
|
||||
欢迎提交 Issue 和 Pull Request 来改进这个技能!
|
||||
|
||||
## 📄 许可证
|
||||
|
||||
MIT License
|
||||
|
|
@ -1,225 +0,0 @@
|
|||
---
|
||||
name: word-reader
|
||||
description: |
|
||||
读取 Word 文档(.docx 和 .doc 格式)并提取文本内容。支持文档解析、表格提取、图片处理等功能。使用当用户需要分析 Word 文档内容、提取文本信息或批量处理文档时。
|
||||
homepage: https://python-docx.readthedocs.io/
|
||||
metadata:
|
||||
{
|
||||
"openclaw":
|
||||
{
|
||||
"emoji": "📄",
|
||||
"requires": { "bins": ["python3"], "env": ["PYTHONPATH"] },
|
||||
"install":
|
||||
[
|
||||
{
|
||||
"id": "pip",
|
||||
"kind": "pip",
|
||||
"package": "python-docx",
|
||||
"bins": ["python3"],
|
||||
"label": "Install python-docx (pip)",
|
||||
},
|
||||
{
|
||||
"id": "system",
|
||||
"kind": "system",
|
||||
"command": "sudo apt-get install antiword -y",
|
||||
"label": "Install antiword for .doc support (optional)",
|
||||
"platform": "linux-debian"
|
||||
}
|
||||
],
|
||||
},
|
||||
}
|
||||
---
|
||||
|
||||
# Word 文档读取器
|
||||
|
||||
使用 Python 解析 Word 文档,提取文本内容和结构化信息。
|
||||
|
||||
## 支持的功能
|
||||
|
||||
- **文档文本提取** - 提取段落、标题、页眉页脚内容
|
||||
- **表格解析** - 读取表格数据并转换为结构化格式
|
||||
- **图片处理** - 提取文档中的图片信息
|
||||
- **元数据获取** - 读取文档属性(作者、标题、创建时间等)
|
||||
- **批量处理** - 支持处理多个文档
|
||||
|
||||
## 用法
|
||||
|
||||
### 基本文本提取
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径>
|
||||
```
|
||||
|
||||
### 指定输出格式
|
||||
|
||||
```bash
|
||||
# JSON 输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format json
|
||||
|
||||
# 纯文本输出
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format text
|
||||
|
||||
# Markdown 格式
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --format markdown
|
||||
```
|
||||
|
||||
### 提取特定内容
|
||||
|
||||
```bash
|
||||
# 只提取文本
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract text
|
||||
|
||||
# 提取表格数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract tables
|
||||
|
||||
# 获取文档元数据
|
||||
python3 {baseDir}/scripts/read_word.py <文件路径> --extract metadata
|
||||
```
|
||||
|
||||
### 批量处理
|
||||
|
||||
```bash
|
||||
# 处理目录下所有 .docx 文件
|
||||
python3 {baseDir}/scripts/read_word.py <目录路径> --batch
|
||||
```
|
||||
|
||||
## 参数说明
|
||||
|
||||
| 参数 | 说明 | 默认值 |
|
||||
|------|------|--------|
|
||||
| `--format` | 输出格式(json/text/markdown) | text |
|
||||
| `--extract` | 提取内容类型(text/tables/images/metadata/all) | all |
|
||||
| `--batch` | 批量处理模式 | false |
|
||||
| `--output` | 输出文件路径 | stdout |
|
||||
| `--encoding` | 文本编码(utf-8/gb2312) | utf-8 |
|
||||
|
||||
## 输出格式
|
||||
|
||||
### JSON 格式
|
||||
|
||||
```json
|
||||
{
|
||||
"metadata": {
|
||||
"title": "文档标题",
|
||||
"author": "作者姓名",
|
||||
"created": "2024-01-01T10:00:00",
|
||||
"modified": "2024-01-01T12:00:00"
|
||||
},
|
||||
"text": "文档全文内容...",
|
||||
"tables": [
|
||||
[
|
||||
["表头1", "表头2"],
|
||||
["行1列1", "行1列2"],
|
||||
["行2列1", "行2列2"]
|
||||
]
|
||||
],
|
||||
"images": [
|
||||
{
|
||||
"filename": "image1.png",
|
||||
"description": "图片描述",
|
||||
"size": "1024x768"
|
||||
}
|
||||
]
|
||||
}
|
||||
```
|
||||
|
||||
### Markdown 格式
|
||||
|
||||
```markdown
|
||||
# 文档标题
|
||||
|
||||
**作者**:作者姓名
|
||||
**创建时间**:2024-01-01 10:00:00
|
||||
|
||||
## 正文内容
|
||||
|
||||
这是文档的正文内容...
|
||||
|
||||
### 表格示例
|
||||
|
||||
| 表头1 | 表头2 |
|
||||
|-------|-------|
|
||||
| 行1列1 | 行1列2 |
|
||||
| 行2列1 | 行2列2 |
|
||||
|
||||

|
||||
|
||||
## 图片列表
|
||||
|
||||
1. **image1.png** (1024x768) - 图片描述
|
||||
```
|
||||
|
||||
## 错误处理
|
||||
|
||||
- 文件不存在:显示错误信息并退出
|
||||
- 格式不支持:提示支持的文件类型
|
||||
- 权限问题:提示文件访问权限
|
||||
- 编码问题:尝试自动检测编码
|
||||
|
||||
## 示例场景
|
||||
|
||||
### 1. 查看项目文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 项目需求.docx --format markdown
|
||||
```
|
||||
|
||||
### 2. 提取会议记录
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py 会议记录.docx --extract text
|
||||
```
|
||||
|
||||
### 3. 批量处理文档
|
||||
|
||||
```bash
|
||||
python3 {baseDir}/scripts/read_word.py ./文档目录 --batch --format json --output results.json
|
||||
```
|
||||
|
||||
## 注意事项
|
||||
|
||||
- 支持 .docx 格式(Office 2007+)
|
||||
- .doc 格式需要额外依赖(如 antiword)
|
||||
- 大文档处理可能需要较长时间
|
||||
- 图片提取仅获取元数据,不包含实际图片数据
|
||||
- 表格格式可能需要手动调整
|
||||
|
||||
## 故障排除
|
||||
|
||||
### 常见问题
|
||||
|
||||
1. **ModuleNotFoundError**: 确保已安装 python-docx
|
||||
2. **PermissionError**: 检查文件读取权限
|
||||
3. **UnicodeDecodeError**: 尝试不同的编码格式
|
||||
|
||||
### 安装依赖
|
||||
|
||||
```bash
|
||||
pip3 install python-docx
|
||||
```
|
||||
|
||||
对于 .doc 格式支持:
|
||||
```bash
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install antiword
|
||||
|
||||
# macOS
|
||||
brew install antiword
|
||||
```
|
||||
|
||||
## 高级功能
|
||||
|
||||
### 自定义样式处理
|
||||
|
||||
脚本会自动处理以下文档元素:
|
||||
- 标题级别(H1-H6)
|
||||
- 段落样式
|
||||
- 列表项目
|
||||
- 页眉页脚
|
||||
- 文档属性
|
||||
|
||||
### 性能优化
|
||||
|
||||
- 大文件流式处理
|
||||
- 内存使用优化
|
||||
- 进度显示(批量模式)
|
||||
|
|
@ -1,11 +0,0 @@
|
|||
{
|
||||
"owner": "xtfnhcyjpgf",
|
||||
"slug": "word-reader",
|
||||
"displayName": "Word Reader",
|
||||
"latest": {
|
||||
"version": "1.0.0",
|
||||
"publishedAt": 1770700102926,
|
||||
"commit": "https://github.com/openclaw/skills/commit/91b71e101c57b69a4d4eb2678e1b79992eb7032f"
|
||||
},
|
||||
"history": []
|
||||
}
|
||||
|
|
@ -1,89 +0,0 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能演示脚本
|
||||
# 此脚本展示如何使用 word-reader 技能
|
||||
|
||||
echo "=== Word Reader 技能演示 ==="
|
||||
echo ""
|
||||
|
||||
# 检查脚本是否存在
|
||||
SCRIPT_PATH="/root/.openclaw/workspace/skills/word-reader/scripts/read_word.py"
|
||||
if [ ! -f "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本不存在"
|
||||
echo "请确保技能已正确安装"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# 检查脚本是否有执行权限
|
||||
if [ ! -x "$SCRIPT_PATH" ]; then
|
||||
echo "❌ 错误:脚本没有执行权限"
|
||||
echo "正在添加执行权限..."
|
||||
chmod +x "$SCRIPT_PATH"
|
||||
fi
|
||||
|
||||
echo "✅ 脚本已就绪"
|
||||
echo ""
|
||||
|
||||
# 显示技能信息
|
||||
echo "📋 技能信息:"
|
||||
echo " 名称:word-reader"
|
||||
echo " 功能:读取 Word 文档(.docx 和 .doc 格式)"
|
||||
echo " 位置:$SCRIPT_PATH"
|
||||
echo ""
|
||||
|
||||
# 显示使用示例
|
||||
echo "📖 使用示例:"
|
||||
echo ""
|
||||
|
||||
echo "1. 显示帮助信息:"
|
||||
echo " python3 $SCRIPT_PATH --help"
|
||||
echo ""
|
||||
|
||||
echo "2. 读取文档(文本格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx"
|
||||
echo ""
|
||||
|
||||
echo "3. 读取文档(JSON 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format json"
|
||||
echo ""
|
||||
|
||||
echo "4. 读取文档(Markdown 格式):"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown"
|
||||
echo ""
|
||||
|
||||
echo "5. 只提取文本内容:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --extract text"
|
||||
echo ""
|
||||
|
||||
echo "6. 批量处理目录:"
|
||||
echo " python3 $SCRIPT_PATH ./文档目录 --batch"
|
||||
echo ""
|
||||
|
||||
echo "7. 保存结果到文件:"
|
||||
echo " python3 $SCRIPT_PATH 文档路径.docx --format markdown --output output.md"
|
||||
echo ""
|
||||
|
||||
echo "🔧 安装依赖:"
|
||||
echo " pip3 install python-docx"
|
||||
echo " # 对于 .doc 格式支持:"
|
||||
echo " # Ubuntu: sudo apt-get install antiword"
|
||||
echo " # macOS: brew install antiword"
|
||||
echo ""
|
||||
|
||||
echo "📊 支持的功能:"
|
||||
echo " ✅ 文本提取"
|
||||
echo " ✅ 表格解析"
|
||||
echo " ✅ 元数据获取"
|
||||
echo " ✅ 图片信息"
|
||||
echo " ✅ 多格式支持"
|
||||
echo " ✅ 批量处理"
|
||||
echo ""
|
||||
|
||||
echo "💡 提示:"
|
||||
echo " - 支持 .docx 和 .doc 格式"
|
||||
echo " - 输出格式:JSON、Text、Markdown"
|
||||
echo " - 如遇错误,请检查依赖是否安装"
|
||||
echo ""
|
||||
|
||||
echo "演示完成!"
|
||||
echo "如需使用,请替换 '文档路径.docx' 为实际的文档路径"
|
||||
|
|
@ -1,101 +0,0 @@
|
|||
#!/bin/bash
|
||||
|
||||
# Word Reader 技能安装脚本
|
||||
# 此脚本会自动安装依赖并设置技能
|
||||
|
||||
set -e
|
||||
|
||||
echo "=== Word Reader 技能安装 ==="
|
||||
echo ""
|
||||
|
||||
# 检查 Python 版本
|
||||
echo "🔍 检查 Python 版本..."
|
||||
python_version=$(python3 --version 2>&1)
|
||||
echo " Python 版本: $python_version"
|
||||
|
||||
if ! python3 -c "import sys; assert sys.version_info >= (3, 6)"; then
|
||||
echo "❌ 错误:需要 Python 3.6 或更高版本"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "✅ Python 版本检查通过"
|
||||
echo ""
|
||||
|
||||
# 检查并安装依赖
|
||||
echo "📦 检查依赖..."
|
||||
|
||||
# 检查 pip
|
||||
if ! command -v pip3 &> /dev/null; then
|
||||
echo " 🔧 安装 pip..."
|
||||
python3 -m ensurepip --upgrade 2>/dev/null || {
|
||||
echo " ❌ 无法安装 pip,尝试使用系统包管理器"
|
||||
if command -v apt &> /dev/null; then
|
||||
sudo apt update
|
||||
sudo apt install -y python3-pip
|
||||
elif command -v yum &> /dev/null; then
|
||||
sudo yum install -y python3-pip
|
||||
elif command -v brew &> /dev/null; then
|
||||
brew install python3
|
||||
else
|
||||
echo " ❌ 无法自动安装 pip,请手动安装"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
fi
|
||||
|
||||
# 检查 python-docx
|
||||
if ! python3 -c "import docx" 2>/dev/null; then
|
||||
echo " 🔧 安装 python-docx..."
|
||||
if python3 -m pip install python-docx --break-system-packages 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
elif python3 -m pip install python-docx 2>/dev/null; then
|
||||
echo " ✅ python-docx 安装完成"
|
||||
else
|
||||
echo "❌ 无法安装 python-docx"
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo " ✅ python-docx 已安装"
|
||||
fi
|
||||
|
||||
# 检查 antiword(可选)
|
||||
if command -v antiword >/dev/null 2>&1; then
|
||||
echo " ✅ antiword 已安装"
|
||||
else
|
||||
echo " ⚠️ antiword 未安装(可选,用于 .doc 格式支持)"
|
||||
echo " 推荐安装命令:"
|
||||
echo " Ubuntu/Debian: sudo apt-get install antiword"
|
||||
echo " macOS: brew install antiword"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
|
||||
# 设置执行权限
|
||||
echo "🔐 设置执行权限..."
|
||||
chmod +x scripts/read_word.py
|
||||
echo "✅ 执行权限已设置"
|
||||
echo ""
|
||||
|
||||
# 验证安装
|
||||
echo "🧪 验证安装..."
|
||||
python3 scripts/read_word.py --help >/dev/null 2>&1
|
||||
if [ $? -eq 0 ]; then
|
||||
echo "✅ 安装验证成功"
|
||||
else
|
||||
echo "❌ 安装验证失败"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "🎉 Word Reader 技能安装完成!"
|
||||
echo ""
|
||||
echo "📖 使用方法:"
|
||||
echo " python3 scripts/read_word.py 文档.docx"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format json"
|
||||
echo " python3 scripts/read_word.py 文档.docx --format markdown"
|
||||
echo ""
|
||||
echo "📖 更多帮助:"
|
||||
echo " python3 scripts/read_word.py --help"
|
||||
echo ""
|
||||
echo "📖 运行演示:"
|
||||
echo " ./demo.sh"
|
||||
|
|
@ -1,396 +0,0 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Word 文档读取器
|
||||
支持 .docx 和 .doc 格式的 Word 文档解析
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import re
|
||||
import traceback
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from docx import Document
|
||||
from docx.opc.constants import RELATIONSHIP_TYPE as RT
|
||||
from docx.oxml.table import CT_Tbl
|
||||
from docx.oxml.text.paragraph import CT_P
|
||||
from docx.table import Table
|
||||
from docx.text.paragraph import Paragraph
|
||||
DOCX_AVAILABLE = True
|
||||
except ImportError:
|
||||
DOCX_AVAILABLE = False
|
||||
|
||||
try:
|
||||
import subprocess
|
||||
SUBPROCESS_AVAILABLE = True
|
||||
except ImportError:
|
||||
SUBPROCESS_AVAILABLE = False
|
||||
|
||||
class WordReader:
|
||||
"""Word 文档读取器"""
|
||||
|
||||
def __init__(self, file_path):
|
||||
self.file_path = Path(file_path)
|
||||
self.document = None
|
||||
self.format_type = None
|
||||
self.encoding = 'utf-8'
|
||||
|
||||
# 检查文件是否存在
|
||||
if not self.file_path.exists():
|
||||
raise FileNotFoundError(f"文件不存在: {file_path}")
|
||||
|
||||
# 检查文件扩展名
|
||||
if self.file_path.suffix.lower() not in ['.docx', '.doc']:
|
||||
raise ValueError(f"不支持的文件格式: {self.file_path.suffix}")
|
||||
|
||||
def read_docx(self):
|
||||
"""读取 .docx 格式文档"""
|
||||
if not DOCX_AVAILABLE:
|
||||
raise Exception("缺少 python-docx 库。请安装:pip3 install python-docx")
|
||||
|
||||
try:
|
||||
self.document = Document(str(self.file_path))
|
||||
self.format_type = 'docx'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .docx 文件失败: {str(e)}")
|
||||
|
||||
def read_doc(self):
|
||||
"""读取 .doc 格式文档(使用 antiword)"""
|
||||
if not SUBPROCESS_AVAILABLE:
|
||||
raise Exception("缺少 subprocess 模块")
|
||||
|
||||
try:
|
||||
# 检查 antiword 是否可用
|
||||
result = subprocess.run(['which', 'antiword'],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
raise Exception("antiword 未安装。请安装 antiword: Ubuntu/Debian: sudo apt-get install antiword; macOS: brew install antiword")
|
||||
|
||||
# 使用 antiword 转换
|
||||
result = subprocess.run(['antiword', str(self.file_path)],
|
||||
capture_output=True, text=True, encoding='utf-8')
|
||||
|
||||
if result.returncode != 0:
|
||||
raise Exception(f"antiword 转换失败: {result.stderr}")
|
||||
|
||||
# 创建临时文档对象
|
||||
class TempDocument:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
self.paragraphs = [TempParagraph(p) for p in text.split('\n') if p.strip()]
|
||||
|
||||
class TempParagraph:
|
||||
def __init__(self, text):
|
||||
self.text = text
|
||||
|
||||
self.document = TempDocument(result.stdout)
|
||||
self.format_type = 'doc'
|
||||
return True
|
||||
except Exception as e:
|
||||
raise Exception(f"读取 .doc 文件失败: {str(e)}")
|
||||
|
||||
def read_metadata(self):
|
||||
"""读取文档元数据"""
|
||||
metadata = {
|
||||
'filename': self.file_path.name,
|
||||
'size': f"{self.file_path.stat().st_size} bytes",
|
||||
'created': datetime.fromtimestamp(self.file_path.stat().st_ctime).isoformat(),
|
||||
'modified': datetime.fromtimestamp(self.file_path.stat().st_mtime).isoformat()
|
||||
}
|
||||
|
||||
if self.format_type == 'docx' and hasattr(self.document, 'core_properties'):
|
||||
props = self.document.core_properties
|
||||
metadata.update({
|
||||
'title': getattr(props, 'title', ''),
|
||||
'author': getattr(props, 'author', ''),
|
||||
'subject': getattr(props, 'subject', ''),
|
||||
'keywords': getattr(props, 'keywords', ''),
|
||||
'comments': getattr(props, 'comments', ''),
|
||||
'application': getattr(props, 'application', ''),
|
||||
'category': getattr(props, 'category', '')
|
||||
})
|
||||
|
||||
return metadata
|
||||
|
||||
def extract_text(self):
|
||||
"""提取文档文本"""
|
||||
text_content = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
# 提取段落文本
|
||||
for para in self.document.paragraphs:
|
||||
if para.text.strip():
|
||||
text_content.append(para.text)
|
||||
|
||||
# 提取表格文本
|
||||
for table in self.document.tables:
|
||||
table_text = []
|
||||
for row in table.rows:
|
||||
row_text = []
|
||||
for cell in row.cells:
|
||||
row_text.append(cell.text.strip())
|
||||
table_text.append(' | '.join(row_text))
|
||||
text_content.append('\n'.join(table_text))
|
||||
|
||||
else: # doc 格式
|
||||
text_content = [para.text for para in self.document.paragraphs if para.text.strip()]
|
||||
|
||||
return '\n\n'.join(text_content)
|
||||
|
||||
def extract_tables(self):
|
||||
"""提取表格数据"""
|
||||
tables = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
for i, table in enumerate(self.document.tables):
|
||||
table_data = []
|
||||
for row in table.rows:
|
||||
row_data = []
|
||||
for cell in row.cells:
|
||||
row_data.append(cell.text.strip())
|
||||
table_data.append(row_data)
|
||||
tables.append({
|
||||
'id': i + 1,
|
||||
'rows': len(table.rows),
|
||||
'columns': len(table.columns) if table.rows else 0,
|
||||
'data': table_data
|
||||
})
|
||||
|
||||
return tables
|
||||
|
||||
def extract_images(self):
|
||||
"""提取图片信息"""
|
||||
images = []
|
||||
|
||||
if self.format_type == 'docx':
|
||||
try:
|
||||
# 获取文档中的关系
|
||||
part = self.document.part
|
||||
image_parts = part.related_parts
|
||||
|
||||
for rel in part.relationships:
|
||||
if rel.reltype == RT.IMAGE:
|
||||
image_data = image_parts[rel.rId]._blob
|
||||
image_info = {
|
||||
'id': rel.rId,
|
||||
'filename': f"image_{rel.rId}.{rel.target_ref.split('.')[-1]}",
|
||||
'size': f"{len(image_data)} bytes"
|
||||
}
|
||||
images.append(image_info)
|
||||
except:
|
||||
# 图片提取可能失败,忽略错误
|
||||
pass
|
||||
|
||||
return images
|
||||
|
||||
def extract_all(self):
|
||||
"""提取所有内容"""
|
||||
result = {
|
||||
'metadata': self.read_metadata(),
|
||||
'format': self.format_type,
|
||||
'text': self.extract_text(),
|
||||
'tables': self.extract_tables(),
|
||||
'images': self.extract_images()
|
||||
}
|
||||
return result
|
||||
|
||||
def to_markdown(self, extract_type='all'):
|
||||
"""转换为 Markdown 格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
md_content = []
|
||||
|
||||
# 标题
|
||||
md_content.append(f"# {result['metadata']['filename']}")
|
||||
md_content.append("")
|
||||
|
||||
# 元数据
|
||||
metadata = result['metadata']
|
||||
if metadata.get('title'):
|
||||
md_content.append(f"**标题**:{metadata['title']}")
|
||||
if metadata.get('author'):
|
||||
md_content.append(f"**作者**:{metadata['author']}")
|
||||
md_content.append(f"**文件大小**:{metadata['size']}")
|
||||
md_content.append(f"**创建时间**:{metadata['created']}")
|
||||
md_content.append(f"**修改时间**:{metadata['modified']}")
|
||||
md_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
md_content.append("## 正文内容")
|
||||
md_content.append("")
|
||||
md_content.append(result['text'])
|
||||
md_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
md_content.append("## 表格内容")
|
||||
md_content.append("")
|
||||
for table in result['tables']:
|
||||
md_content.append(f"### 表格 {table['id']} ({table['rows']}行 x {table['columns']}列)")
|
||||
md_content.append("")
|
||||
# 转换为 Markdown 表格
|
||||
for row in table['data']:
|
||||
md_row = " | ".join([str(cell) for cell in row])
|
||||
md_content.append(f"| {md_row} |")
|
||||
md_content.append("")
|
||||
|
||||
# 图片
|
||||
if result['images']:
|
||||
md_content.append("## 图片列表")
|
||||
md_content.append("")
|
||||
for img in result['images']:
|
||||
md_content.append(f"- **{img['filename']}** ({img['size']})")
|
||||
md_content.append("")
|
||||
|
||||
return '\n'.join(md_content)
|
||||
|
||||
def to_text(self, extract_type='all'):
|
||||
"""转换为纯文本格式"""
|
||||
if extract_type == 'text':
|
||||
return self.extract_text()
|
||||
|
||||
result = self.extract_all()
|
||||
text_content = []
|
||||
|
||||
# 标题和元数据
|
||||
text_content.append(f"文件:{result['metadata']['filename']}")
|
||||
text_content.append("=" * 50)
|
||||
text_content.append("")
|
||||
|
||||
for key, value in result['metadata'].items():
|
||||
if value and key not in ['filename', 'size', 'created', 'modified']:
|
||||
text_content.append(f"{key}:{value}")
|
||||
|
||||
text_content.append("")
|
||||
|
||||
# 文本内容
|
||||
if result['text']:
|
||||
text_content.append("正文内容:")
|
||||
text_content.append("-" * 20)
|
||||
text_content.append(result['text'])
|
||||
text_content.append("")
|
||||
|
||||
# 表格
|
||||
if result['tables']:
|
||||
text_content.append("表格内容:")
|
||||
text_content.append("-" * 20)
|
||||
for table in result['tables']:
|
||||
text_content.append(f"表格 {table['id']}:")
|
||||
for row in table['data']:
|
||||
text_content.append(" " + " | ".join([str(cell) for cell in row]))
|
||||
text_content.append("")
|
||||
|
||||
return '\n'.join(text_content)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description='读取 Word 文档')
|
||||
parser.add_argument('path', help='文档路径或目录路径(批量模式)')
|
||||
parser.add_argument('--format', choices=['json', 'text', 'markdown'],
|
||||
default='text', help='输出格式')
|
||||
parser.add_argument('--extract', choices=['text', 'tables', 'images', 'metadata', 'all'],
|
||||
default='all', help='提取内容类型')
|
||||
parser.add_argument('--batch', action='store_true', help='批量处理模式')
|
||||
parser.add_argument('--output', help='输出文件路径')
|
||||
parser.add_argument('--encoding', default='utf-8', help='文本编码')
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
try:
|
||||
if args.batch:
|
||||
# 批量处理模式
|
||||
path = Path(args.path)
|
||||
if not path.is_dir():
|
||||
print("错误:批量模式需要指定目录路径")
|
||||
sys.exit(1)
|
||||
|
||||
# 查找所有 Word 文档
|
||||
word_files = []
|
||||
for ext in ['.docx', '.doc']:
|
||||
word_files.extend(path.glob(f"**/*{ext}"))
|
||||
|
||||
if not word_files:
|
||||
print("未找到 Word 文档")
|
||||
sys.exit(0)
|
||||
|
||||
print(f"找到 {len(word_files)} 个 Word 文档")
|
||||
|
||||
results = {}
|
||||
for file_path in word_files:
|
||||
print(f"正在处理: {file_path}")
|
||||
try:
|
||||
reader = WordReader(file_path)
|
||||
if file_path.suffix.lower() == '.docx':
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'content': content,
|
||||
'status': 'success'
|
||||
}
|
||||
|
||||
except Exception as e:
|
||||
results[str(file_path)] = {
|
||||
'filename': file_path.name,
|
||||
'error': str(e),
|
||||
'status': 'failed'
|
||||
}
|
||||
|
||||
# 保存结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding='utf-8') as f:
|
||||
json.dump(results, f, ensure_ascii=False, indent=2)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(json.dumps(results, ensure_ascii=False, indent=2))
|
||||
|
||||
else:
|
||||
# 单文件处理模式
|
||||
reader = WordReader(args.path)
|
||||
|
||||
# 根据文件类型读取
|
||||
if args.path.lower().endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
|
||||
# 根据格式输出
|
||||
if args.format == 'json':
|
||||
content = reader.extract_all()
|
||||
elif args.format == 'markdown':
|
||||
content = reader.to_markdown(args.extract)
|
||||
else:
|
||||
content = reader.to_text(args.extract)
|
||||
|
||||
# 输出结果
|
||||
if args.output:
|
||||
with open(args.output, 'w', encoding=args.encoding) as f:
|
||||
f.write(content)
|
||||
print(f"结果已保存到: {args.output}")
|
||||
else:
|
||||
print(content)
|
||||
|
||||
except Exception as e:
|
||||
print(f"错误: {str(e)}", file=sys.stderr)
|
||||
if '--debug' in sys.argv or '-d' in sys.argv:
|
||||
traceback.print_exc()
|
||||
sys.exit(1)
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
|
|
@ -1,47 +0,0 @@
|
|||
{
|
||||
"name": "word-reader",
|
||||
"version": "1.0.0",
|
||||
"description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容",
|
||||
"author": "OpenClaw User",
|
||||
"tags": ["document", "word", "office", "text-extraction"],
|
||||
"dependencies": {
|
||||
"python": ">=3.6",
|
||||
"packages": ["python-docx"],
|
||||
"system": ["antiword (optional for .doc support)"]
|
||||
},
|
||||
"features": {
|
||||
"text_extraction": true,
|
||||
"table_parsing": true,
|
||||
"metadata_extraction": true,
|
||||
"image_info": true,
|
||||
"batch_processing": true,
|
||||
"multiple_formats": ["json", "text", "markdown"]
|
||||
},
|
||||
"installation": {
|
||||
"steps": [
|
||||
"pip3 install python-docx",
|
||||
"sudo apt-get install antiword # 可选,支持 .doc 格式",
|
||||
"chmod +x scripts/read_word.py"
|
||||
]
|
||||
},
|
||||
"usage_examples": [
|
||||
{
|
||||
"description": "读取文档文本",
|
||||
"command": "python3 scripts/read_word.py document.docx"
|
||||
},
|
||||
{
|
||||
"description": "转换为 Markdown",
|
||||
"command": "python3 scripts/read_word.py document.docx --format markdown"
|
||||
},
|
||||
{
|
||||
"description": "批量处理",
|
||||
"command": "python3 scripts/read_word.py ./docs --batch --format json"
|
||||
}
|
||||
],
|
||||
"supported_file_types": [".docx", ".doc"],
|
||||
"notes": [
|
||||
".doc 格式需要安装 antiword",
|
||||
"大文档处理可能需要较长时间",
|
||||
"图片提取仅获取元数据,不包含实际图片数据"
|
||||
]
|
||||
}
|
||||
|
|
@ -1,42 +0,0 @@
|
|||
# Word Reader 技能测试
|
||||
|
||||
这是一个简单的测试文档,用于验证 Word Reader 技能的功能。
|
||||
|
||||
## 测试内容
|
||||
|
||||
### 1. 基本文本
|
||||
这是一段测试文本,用于验证文本提取功能是否正常工作。
|
||||
|
||||
### 2. 表格测试
|
||||
|
||||
| 功能 | 状态 | 描述 |
|
||||
|------|------|------|
|
||||
| 文本提取 | ✅ | 能够提取文档中的所有文本内容 |
|
||||
| 表格解析 | ✅ | 能够正确解析表格数据 |
|
||||
| 元数据获取 | ✅ | 能够获取文档属性信息 |
|
||||
| 多格式支持 | ✅ | 支持 .docx 和 .doc 格式 |
|
||||
| 输出格式 | ✅ | 支持 JSON、Text、Markdown 格式 |
|
||||
|
||||
### 3. 列表测试
|
||||
|
||||
- 第一项:文本提取功能
|
||||
- 第二项:表格解析功能
|
||||
- 第三项:图片信息获取
|
||||
- 第四项:文档元数据读取
|
||||
|
||||
### 4. 代码块示例
|
||||
|
||||
```python
|
||||
def read_word_document(file_path):
|
||||
"""读取 Word 文档"""
|
||||
reader = WordReader(file_path)
|
||||
if file_path.endswith('.docx'):
|
||||
reader.read_docx()
|
||||
else:
|
||||
reader.read_doc()
|
||||
return reader.extract_all()
|
||||
```
|
||||
|
||||
## 测试完成
|
||||
|
||||
如果这个技能能够正确读取并解析上述内容,说明功能正常。
|
||||
|
|
@ -1,61 +0,0 @@
|
|||
# XLSX Pro
|
||||
|
||||
 
|
||||
|
||||
Un skill **Clawdbot / OpenClawd** pour générer et modifier des fichiers Excel **propres** (XLSX / XLSM / CSV / TSV) avec :
|
||||
- formatage “pro”
|
||||
- **formules Excel** (au lieu de valeurs hardcodées)
|
||||
- recalcul optionnel des formules via **LibreOffice headless**
|
||||
- contrôle qualité : détection des erreurs Excel (`#REF!`, `#DIV/0!`, `#VALUE!`, `#N/A`, `#NAME?`, …)
|
||||
|
||||
## Pourquoi ce skill ?
|
||||
|
||||
`openpyxl` sait **écrire** des formules, mais ne sait pas **calculer** leurs résultats. En production, ça crée des fichiers où les formules ne sont pas évaluées et où les erreurs ne sont pas détectées.
|
||||
|
||||
`XLSX Pro` ajoute une étape serveur fiable : **recalcul via LibreOffice** + scan d’erreurs.
|
||||
|
||||
## Prérequis
|
||||
|
||||
### Python
|
||||
```bash
|
||||
pip install openpyxl pandas xlrd xlwt
|
||||
```
|
||||
|
||||
### LibreOffice (uniquement si tu veux recalculer les formules)
|
||||
Ubuntu/Debian :
|
||||
```bash
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y libreoffice-calc libreoffice-common
|
||||
```
|
||||
|
||||
## Quickstart
|
||||
|
||||
### 1) Générer un fichier Excel (avec openpyxl)
|
||||
Tu peux créer ton `.xlsx` comme d’habitude en Python, en mettant des **formules** dans les cellules.
|
||||
|
||||
### 2) Recalculer + valider
|
||||
```bash
|
||||
python scripts/recalc.py ton_fichier.xlsx 60
|
||||
```
|
||||
|
||||
Sortie JSON :
|
||||
- `status: success | errors_found`
|
||||
- `total_errors`
|
||||
- `error_summary` (types + emplacements)
|
||||
- `total_formulas`
|
||||
|
||||
## Bonnes pratiques (résumé)
|
||||
- **Préférer les formules Excel** plutôt que calculer en Python puis écrire des valeurs.
|
||||
- **Zéro erreur de formule** dans le livrable.
|
||||
- Si tu modifies un template existant : **respecte exactement** les styles/conventions.
|
||||
|
||||
## Troubleshooting
|
||||
- Si `soffice` est introuvable : installe LibreOffice (voir Prérequis).
|
||||
- Si le recalcul “timeout” : augmente le timeout (2e argument) et/ou teste sur un fichier plus petit.
|
||||
- Si erreur du type "macro mal configurée" / "macro not configured" : supprime le fichier de macro puis relance :
|
||||
- Linux : `~/.config/libreoffice/4/user/basic/Standard/Module1.xba`
|
||||
- macOS : `~/Library/Application Support/LibreOffice/4/user/basic/Standard/Module1.xba`
|
||||
- En conteneur (Docker) : ajoute la variable d'env `SAL_USE_VCLPLUGIN=svp` (ça évite des soucis d'UI en headless).
|
||||
|
||||
## Licence
|
||||
MIT (à ajuster si tu veux une autre licence).
|
||||
|
|
@ -1,232 +0,0 @@
|
|||
---
|
||||
name: xlsx-pro
|
||||
description: "Compétence pour manipuler les fichiers Excel (.xlsx, .xlsm, .csv, .tsv). Utiliser quand l'utilisateur veut : ouvrir, lire, éditer ou créer un fichier tableur ; ajouter des colonnes, calculer des formules, formater, créer des graphiques, nettoyer des données ; convertir entre formats tabulaires. Le livrable doit être un fichier tableur. NE PAS utiliser si le livrable est un document Word, HTML, script Python standalone, ou intégration Google Sheets."
|
||||
version: "1.0.1"
|
||||
author: "Eric Barotte"
|
||||
---
|
||||
|
||||
# Compétence Excel pour OpenClawd
|
||||
|
||||
## TL;DR
|
||||
- Génère/édite des fichiers Excel avec des **formules** (pas des valeurs hardcodées).
|
||||
- Optionnel: **recalcul** via LibreOffice headless + détection d’erreurs Excel.
|
||||
- Livrable attendu: un fichier tableur propre (XLSX/XLSM/CSV/TSV).
|
||||
|
||||
|
||||
## Prérequis
|
||||
|
||||
### Dépendances Python
|
||||
```bash
|
||||
pip install openpyxl pandas xlrd xlwt
|
||||
```
|
||||
|
||||
### LibreOffice (pour recalcul des formules)
|
||||
```bash
|
||||
# Ubuntu/Debian
|
||||
sudo apt-get install libreoffice-calc libreoffice-common
|
||||
```
|
||||
|
||||
## Règles de Qualité
|
||||
|
||||
### Police Professionnelle
|
||||
- Utiliser une police cohérente (Arial, Times New Roman) sauf instruction contraire
|
||||
|
||||
### Zéro Erreur de Formule
|
||||
- Tout fichier Excel DOIT être livré SANS erreurs (#REF!, #DIV/0!, #VALUE!, #N/A, #NAME?)
|
||||
|
||||
### Préservation des Templates
|
||||
- Respecter EXACTEMENT le format et style existants lors de modifications
|
||||
- Les conventions du template préexistant ont TOUJOURS priorité
|
||||
|
||||
## Standards pour Modèles Financiers
|
||||
|
||||
### Code Couleur (Standards Industrie)
|
||||
- **Texte bleu (RGB: 0,0,255)** : Inputs hardcodés, valeurs modifiables
|
||||
- **Texte noir (RGB: 0,0,0)** : TOUTES les formules et calculs
|
||||
- **Texte vert (RGB: 0,128,0)** : Liens vers autres feuilles du même classeur
|
||||
- **Texte rouge (RGB: 255,0,0)** : Liens externes vers autres fichiers
|
||||
- **Fond jaune (RGB: 255,255,0)** : Hypothèses clés ou cellules à mettre à jour
|
||||
|
||||
### Formatage des Nombres
|
||||
- **Années** : Format texte ("2024" pas "2,024")
|
||||
- **Devises** : Format $#,##0 ; spécifier unités dans les en-têtes ("Revenue ($mm)")
|
||||
- **Zéros** : Afficher comme "-" (format: "$#,##0;($#,##0);-")
|
||||
- **Pourcentages** : Format 0.0% par défaut
|
||||
- **Multiples** : Format 0.0x (EV/EBITDA, P/E)
|
||||
- **Négatifs** : Parenthèses (123) pas moins -123
|
||||
|
||||
## CRITIQUE : Utiliser des Formules, PAS des Valeurs Hardcodées
|
||||
|
||||
**TOUJOURS utiliser des formules Excel au lieu de calculer en Python et hardcoder.**
|
||||
|
||||
### ❌ MAUVAIS - Hardcoding
|
||||
```python
|
||||
# Mauvais: Calcul Python puis hardcode
|
||||
total = df['Sales'].sum()
|
||||
sheet['B10'] = total # Hardcode 5000
|
||||
|
||||
# Mauvais: Taux de croissance calculé en Python
|
||||
growth = (df.iloc[-1]['Revenue'] - df.iloc[0]['Revenue']) / df.iloc[0]['Revenue']
|
||||
sheet['C5'] = growth # Hardcode 0.15
|
||||
```
|
||||
|
||||
### ✅ CORRECT - Formules Excel
|
||||
```python
|
||||
# Bon: Laisser Excel calculer
|
||||
sheet['B10'] = '=SUM(B2:B9)'
|
||||
|
||||
# Bon: Taux de croissance en formule Excel
|
||||
sheet['C5'] = '=(C4-C2)/C2'
|
||||
|
||||
# Bon: Moyenne en fonction Excel
|
||||
sheet['D20'] = '=AVERAGE(D2:D19)'
|
||||
```
|
||||
|
||||
## Workflows
|
||||
|
||||
### Workflow Standard
|
||||
1. **Choisir l'outil** : pandas pour données, openpyxl pour formules/formatage
|
||||
2. **Créer/Charger** : Nouveau classeur ou fichier existant
|
||||
3. **Modifier** : Données, formules, formatage
|
||||
4. **Sauvegarder** : Écrire le fichier
|
||||
5. **Recalculer (OBLIGATOIRE si formules)** : `python scripts/recalc.py output.xlsx`
|
||||
6. **Vérifier et corriger** les erreurs détectées
|
||||
|
||||
### Lecture et Analyse avec pandas
|
||||
```python
|
||||
import pandas as pd
|
||||
|
||||
# Lire Excel
|
||||
df = pd.read_excel('file.xlsx') # Première feuille par défaut
|
||||
all_sheets = pd.read_excel('file.xlsx', sheet_name=None) # Dict de toutes les feuilles
|
||||
|
||||
# Analyser
|
||||
df.head() # Aperçu
|
||||
df.info() # Info colonnes
|
||||
df.describe() # Statistiques
|
||||
|
||||
# Écrire
|
||||
df.to_excel('output.xlsx', index=False)
|
||||
```
|
||||
|
||||
### Création de Fichiers Excel
|
||||
```python
|
||||
from openpyxl import Workbook
|
||||
from openpyxl.styles import Font, PatternFill, Alignment
|
||||
|
||||
wb = Workbook()
|
||||
sheet = wb.active
|
||||
|
||||
# Données
|
||||
sheet['A1'] = 'Hello'
|
||||
sheet['B1'] = 'World'
|
||||
sheet.append(['Row', 'of', 'data'])
|
||||
|
||||
# Formule
|
||||
sheet['B2'] = '=SUM(A1:A10)'
|
||||
|
||||
# Formatage
|
||||
sheet['A1'].font = Font(bold=True, color='FF0000')
|
||||
sheet['A1'].fill = PatternFill('solid', start_color='FFFF00')
|
||||
sheet['A1'].alignment = Alignment(horizontal='center')
|
||||
|
||||
# Largeur colonne
|
||||
sheet.column_dimensions['A'].width = 20
|
||||
|
||||
wb.save('output.xlsx')
|
||||
```
|
||||
|
||||
### Édition de Fichiers Existants
|
||||
```python
|
||||
from openpyxl import load_workbook
|
||||
|
||||
# Charger fichier existant
|
||||
wb = load_workbook('existing.xlsx')
|
||||
sheet = wb.active # ou wb['NomFeuille']
|
||||
|
||||
# Parcourir les feuilles
|
||||
for sheet_name in wb.sheetnames:
|
||||
sheet = wb[sheet_name]
|
||||
print(f"Feuille: {sheet_name}")
|
||||
|
||||
# Modifier
|
||||
sheet['A1'] = 'Nouvelle Valeur'
|
||||
sheet.insert_rows(2) # Insérer ligne
|
||||
sheet.delete_cols(3) # Supprimer colonne
|
||||
|
||||
# Ajouter feuille
|
||||
new_sheet = wb.create_sheet('NouvelleFeuille')
|
||||
new_sheet['A1'] = 'Data'
|
||||
|
||||
wb.save('modified.xlsx')
|
||||
```
|
||||
|
||||
## Recalcul des Formules
|
||||
|
||||
Les fichiers créés par openpyxl contiennent les formules comme chaînes mais pas les valeurs calculées. Utiliser le script `recalc.py` :
|
||||
|
||||
```bash
|
||||
python scripts/recalc.py <fichier_excel> [timeout_secondes]
|
||||
```
|
||||
|
||||
Le script :
|
||||
- Configure automatiquement la macro LibreOffice au premier lancement
|
||||
- Recalcule toutes les formules
|
||||
- Scanne TOUTES les cellules pour erreurs Excel
|
||||
- Retourne JSON avec détails et emplacements des erreurs
|
||||
|
||||
### Interprétation de la Sortie
|
||||
```json
|
||||
{
|
||||
"status": "success", // ou "errors_found"
|
||||
"total_errors": 0, // Nombre total d'erreurs
|
||||
"total_formulas": 42, // Nombre de formules
|
||||
"error_summary": { // Présent si erreurs
|
||||
"#REF!": {
|
||||
"count": 2,
|
||||
"locations": ["Sheet1!B5", "Sheet1!C10"]
|
||||
}
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
## Checklist de Vérification
|
||||
|
||||
### Vérifications Essentielles
|
||||
- [ ] **Tester 2-3 références** : Vérifier qu'elles tirent les bonnes valeurs
|
||||
- [ ] **Mapping colonnes** : Confirmer correspondance (colonne 64 = BL, pas BK)
|
||||
- [ ] **Offset lignes** : Excel est 1-indexé (DataFrame row 5 = Excel row 6)
|
||||
|
||||
### Pièges Courants
|
||||
- [ ] **Gestion NaN** : Vérifier valeurs nulles avec `pd.notna()`
|
||||
- [ ] **Colonnes éloignées** : Données FY souvent en colonnes 50+
|
||||
- [ ] **Correspondances multiples** : Chercher toutes les occurrences
|
||||
- [ ] **Division par zéro** : Vérifier dénominateurs (#DIV/0!)
|
||||
- [ ] **Références invalides** : Vérifier que toutes pointent vers cellules existantes (#REF!)
|
||||
- [ ] **Références inter-feuilles** : Format correct (Sheet1!A1)
|
||||
|
||||
## Bonnes Pratiques
|
||||
|
||||
### Sélection de Bibliothèque
|
||||
- **pandas** : Analyse de données, opérations en masse, export simple
|
||||
- **openpyxl** : Formatage complexe, formules, fonctionnalités Excel spécifiques
|
||||
|
||||
### Avec openpyxl
|
||||
- Indices de cellules en base 1 (row=1, column=1 = cellule A1)
|
||||
- `data_only=True` pour lire valeurs calculées
|
||||
- **Attention** : Sauvegarder après `data_only=True` remplace définitivement les formules par les valeurs
|
||||
- Pour gros fichiers : `read_only=True` ou `write_only=True`
|
||||
|
||||
### Avec pandas
|
||||
- Spécifier types de données : `pd.read_excel('file.xlsx', dtype={'id': str})`
|
||||
- Pour gros fichiers, colonnes spécifiques : `usecols=['A', 'C', 'E']`
|
||||
- Gestion des dates : `parse_dates=['date_column']`
|
||||
|
||||
## Style de Code
|
||||
|
||||
**IMPORTANT** : Code Python minimal et concis, sans commentaires superflus.
|
||||
|
||||
**Pour les fichiers Excel** :
|
||||
- Commenter les cellules avec formules complexes
|
||||
- Documenter les sources des données hardcodées
|
||||
- Inclure notes pour calculs clés
|
||||
|
|
@ -1,6 +0,0 @@
|
|||
{
|
||||
"ownerId": "kn7aga06tewbtydgnw3v1468x9804b6x",
|
||||
"slug": "xlsx-pro",
|
||||
"version": "1.0.1",
|
||||
"publishedAt": 1770059662515
|
||||
}
|
||||
|
|
@ -1,8 +0,0 @@
|
|||
"""
|
||||
Module office pour OpenClawd
|
||||
Gestion des opérations LibreOffice
|
||||
"""
|
||||
|
||||
from .soffice import get_soffice_env, run_soffice
|
||||
|
||||
__all__ = ['get_soffice_env', 'run_soffice']
|
||||
|
|
@ -1,211 +0,0 @@
|
|||
"""
|
||||
Helper pour exécuter LibreOffice (soffice) dans des environnements
|
||||
où les sockets AF_UNIX peuvent être bloqués (VMs sandboxées).
|
||||
|
||||
Usage:
|
||||
from office.soffice import run_soffice, get_soffice_env
|
||||
|
||||
# Option 1 – exécuter soffice directement
|
||||
result = run_soffice(["--headless", "--convert-to", "pdf", "input.docx"])
|
||||
|
||||
# Option 2 – obtenir env dict pour vos propres appels subprocess
|
||||
env = get_soffice_env()
|
||||
subprocess.run(["soffice", ...], env=env)
|
||||
"""
|
||||
|
||||
import os
|
||||
import socket
|
||||
import subprocess
|
||||
import tempfile
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def get_soffice_env() -> dict:
|
||||
"""Retourne un env dict adapté pour exécuter soffice en headless.
|
||||
|
||||
Définit toujours SAL_USE_VCLPLUGIN=svp pour le rendu headless (pas de X11).
|
||||
Dans les environnements sandboxés où AF_UNIX est bloqué, ajoute aussi
|
||||
LD_PRELOAD (socket shim).
|
||||
"""
|
||||
env = os.environ.copy()
|
||||
env["SAL_USE_VCLPLUGIN"] = "svp"
|
||||
|
||||
if _needs_shim():
|
||||
shim = _ensure_shim()
|
||||
env["LD_PRELOAD"] = str(shim)
|
||||
|
||||
return env
|
||||
|
||||
|
||||
def run_soffice(args: list, **kwargs) -> subprocess.CompletedProcess:
|
||||
"""Exécute soffice avec les arguments donnés, appliquant le socket shim
|
||||
si nécessaire. Accepte les mêmes arguments que subprocess.run.
|
||||
|
||||
Dans les environnements sandboxés, le shim gère l'arrêt propre en appelant
|
||||
_exit(0) quand le socket listener de soffice.bin se ferme (après conversion).
|
||||
"""
|
||||
env = get_soffice_env()
|
||||
return subprocess.run(["soffice"] + args, env=env, **kwargs)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Internals
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SHIM_SO = Path(tempfile.gettempdir()) / "lo_socket_shim.so"
|
||||
|
||||
|
||||
def _needs_shim() -> bool:
|
||||
"""Vérifie si les sockets AF_UNIX sont bloqués."""
|
||||
try:
|
||||
s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
s.close()
|
||||
return False
|
||||
except OSError:
|
||||
return True
|
||||
|
||||
|
||||
def _ensure_shim() -> Path:
|
||||
"""Compile le shim .so s'il n'est pas déjà en cache."""
|
||||
if _SHIM_SO.exists():
|
||||
return _SHIM_SO
|
||||
|
||||
src = Path(tempfile.gettempdir()) / "lo_socket_shim.c"
|
||||
src.write_text(_SHIM_SOURCE)
|
||||
try:
|
||||
subprocess.run(
|
||||
["gcc", "-shared", "-fPIC", "-o", str(_SHIM_SO), str(src), "-ldl"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
)
|
||||
except (subprocess.CalledProcessError, FileNotFoundError):
|
||||
# Si gcc n'est pas disponible ou échoue, on continue sans shim
|
||||
pass
|
||||
finally:
|
||||
if src.exists():
|
||||
src.unlink()
|
||||
return _SHIM_SO
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# LD_PRELOAD shim – source C
|
||||
#
|
||||
# Problème
|
||||
# --------
|
||||
# LibreOffice utilise des sockets AF_UNIX pour la gestion single-instance
|
||||
# (OSL_PIPE). Dans les environnements sandboxés, le filtre seccomp bloque
|
||||
# socket(AF_UNIX) tout en permettant socketpair(AF_UNIX). Sans ce shim,
|
||||
# soffice crash ou reste bloqué après conversion.
|
||||
#
|
||||
# Solution
|
||||
# --------
|
||||
# Intercepte les appels concernés et fournit des substituts fonctionnels.
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
_SHIM_SOURCE = r"""
|
||||
#define _GNU_SOURCE
|
||||
#include <dlfcn.h>
|
||||
#include <errno.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <sys/socket.h>
|
||||
#include <unistd.h>
|
||||
|
||||
static int (*real_socket)(int, int, int);
|
||||
static int (*real_socketpair)(int, int, int, int[2]);
|
||||
static int (*real_listen)(int, int);
|
||||
static int (*real_accept)(int, struct sockaddr *, socklen_t *);
|
||||
static int (*real_close)(int);
|
||||
static int (*real_read)(int, void *, size_t);
|
||||
|
||||
static int is_shimmed[1024];
|
||||
static int peer_of[1024];
|
||||
static int wake_r[1024];
|
||||
static int wake_w[1024];
|
||||
static int listener_fd = -1;
|
||||
|
||||
__attribute__((constructor))
|
||||
static void init(void) {
|
||||
real_socket = dlsym(RTLD_NEXT, "socket");
|
||||
real_socketpair = dlsym(RTLD_NEXT, "socketpair");
|
||||
real_listen = dlsym(RTLD_NEXT, "listen");
|
||||
real_accept = dlsym(RTLD_NEXT, "accept");
|
||||
real_close = dlsym(RTLD_NEXT, "close");
|
||||
real_read = dlsym(RTLD_NEXT, "read");
|
||||
for (int i = 0; i < 1024; i++) {
|
||||
peer_of[i] = -1;
|
||||
wake_r[i] = -1;
|
||||
wake_w[i] = -1;
|
||||
}
|
||||
}
|
||||
|
||||
int socket(int domain, int type, int protocol) {
|
||||
if (domain == AF_UNIX) {
|
||||
int fd = real_socket(domain, type, protocol);
|
||||
if (fd >= 0) return fd;
|
||||
int sv[2];
|
||||
if (real_socketpair(domain, type, protocol, sv) == 0) {
|
||||
if (sv[0] >= 0 && sv[0] < 1024) {
|
||||
is_shimmed[sv[0]] = 1;
|
||||
peer_of[sv[0]] = sv[1];
|
||||
int wp[2];
|
||||
if (pipe(wp) == 0) {
|
||||
wake_r[sv[0]] = wp[0];
|
||||
wake_w[sv[0]] = wp[1];
|
||||
}
|
||||
}
|
||||
return sv[0];
|
||||
}
|
||||
errno = EPERM;
|
||||
return -1;
|
||||
}
|
||||
return real_socket(domain, type, protocol);
|
||||
}
|
||||
|
||||
int listen(int sockfd, int backlog) {
|
||||
if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) {
|
||||
listener_fd = sockfd;
|
||||
return 0;
|
||||
}
|
||||
return real_listen(sockfd, backlog);
|
||||
}
|
||||
|
||||
int accept(int sockfd, struct sockaddr *addr, socklen_t *addrlen) {
|
||||
if (sockfd >= 0 && sockfd < 1024 && is_shimmed[sockfd]) {
|
||||
if (wake_r[sockfd] >= 0) {
|
||||
char buf;
|
||||
real_read(wake_r[sockfd], &buf, 1);
|
||||
}
|
||||
errno = ECONNABORTED;
|
||||
return -1;
|
||||
}
|
||||
return real_accept(sockfd, addr, addrlen);
|
||||
}
|
||||
|
||||
int close(int fd) {
|
||||
if (fd >= 0 && fd < 1024 && is_shimmed[fd]) {
|
||||
int was_listener = (fd == listener_fd);
|
||||
is_shimmed[fd] = 0;
|
||||
|
||||
if (wake_w[fd] >= 0) {
|
||||
char c = 0;
|
||||
write(wake_w[fd], &c, 1);
|
||||
real_close(wake_w[fd]);
|
||||
wake_w[fd] = -1;
|
||||
}
|
||||
if (wake_r[fd] >= 0) { real_close(wake_r[fd]); wake_r[fd] = -1; }
|
||||
if (peer_of[fd] >= 0) { real_close(peer_of[fd]); peer_of[fd] = -1; }
|
||||
|
||||
if (was_listener)
|
||||
_exit(0);
|
||||
}
|
||||
return real_close(fd);
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import sys
|
||||
result = run_soffice(sys.argv[1:])
|
||||
sys.exit(result.returncode)
|
||||
|
|
@ -1,225 +0,0 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Script de recalcul des formules Excel
|
||||
Recalcule toutes les formules d'un fichier Excel via LibreOffice
|
||||
Adapté pour OpenClawd
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
import platform
|
||||
import subprocess
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from office.soffice import get_soffice_env
|
||||
except ImportError:
|
||||
# Fallback si le module office n'est pas disponible
|
||||
def get_soffice_env():
|
||||
env = os.environ.copy()
|
||||
env["SAL_USE_VCLPLUGIN"] = "svp"
|
||||
return env
|
||||
|
||||
try:
|
||||
from openpyxl import load_workbook
|
||||
except ImportError:
|
||||
print("Erreur: openpyxl non installé. Exécuter: pip install openpyxl")
|
||||
sys.exit(1)
|
||||
|
||||
# Répertoire macro LibreOffice selon plateforme
|
||||
MACRO_DIR_MACOS = "~/Library/Application Support/LibreOffice/4/user/basic/Standard"
|
||||
MACRO_DIR_LINUX = "~/.config/libreoffice/4/user/basic/Standard"
|
||||
MACRO_FILENAME = "Module1.xba"
|
||||
|
||||
# Macro LibreOffice Basic pour recalcul
|
||||
RECALCULATE_MACRO = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE script:module PUBLIC "-//OpenOffice.org//DTD OfficeDocument 1.0//EN" "module.dtd">
|
||||
<script:module xmlns:script="http://openoffice.org/2000/script" script:name="Module1" script:language="StarBasic">
|
||||
Sub RecalculateAndSave()
|
||||
ThisComponent.calculateAll()
|
||||
ThisComponent.store()
|
||||
ThisComponent.close(True)
|
||||
End Sub
|
||||
</script:module>"""
|
||||
|
||||
|
||||
def has_gtimeout():
|
||||
"""Vérifie si gtimeout est disponible sur macOS"""
|
||||
try:
|
||||
subprocess.run(
|
||||
["gtimeout", "--version"], capture_output=True, timeout=1, check=False
|
||||
)
|
||||
return True
|
||||
except (FileNotFoundError, subprocess.TimeoutExpired):
|
||||
return False
|
||||
|
||||
|
||||
def setup_libreoffice_macro():
|
||||
"""Configure la macro LibreOffice si pas déjà fait"""
|
||||
macro_dir = os.path.expanduser(
|
||||
MACRO_DIR_MACOS if platform.system() == "Darwin" else MACRO_DIR_LINUX
|
||||
)
|
||||
macro_file = os.path.join(macro_dir, MACRO_FILENAME)
|
||||
|
||||
# Vérifier si macro existe déjà
|
||||
if (
|
||||
os.path.exists(macro_file)
|
||||
and "RecalculateAndSave" in Path(macro_file).read_text()
|
||||
):
|
||||
return True
|
||||
|
||||
# Créer répertoire macro si nécessaire
|
||||
if not os.path.exists(macro_dir):
|
||||
try:
|
||||
subprocess.run(
|
||||
["soffice", "--headless", "--terminate_after_init"],
|
||||
capture_output=True,
|
||||
timeout=10,
|
||||
env=get_soffice_env(),
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
os.makedirs(macro_dir, exist_ok=True)
|
||||
|
||||
# Écrire fichier macro
|
||||
try:
|
||||
Path(macro_file).write_text(RECALCULATE_MACRO)
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def recalc(filename, timeout=30):
|
||||
"""
|
||||
Recalcule les formules d'un fichier Excel et rapporte les erreurs
|
||||
|
||||
Args:
|
||||
filename: Chemin vers le fichier Excel
|
||||
timeout: Temps max d'attente pour le recalcul (secondes)
|
||||
|
||||
Returns:
|
||||
dict avec emplacements et compteurs d'erreurs
|
||||
"""
|
||||
if not Path(filename).exists():
|
||||
return {"error": f"Fichier {filename} inexistant"}
|
||||
|
||||
abs_path = str(Path(filename).absolute())
|
||||
|
||||
if not setup_libreoffice_macro():
|
||||
return {"error": "Échec configuration macro LibreOffice"}
|
||||
|
||||
cmd = [
|
||||
"soffice",
|
||||
"--headless",
|
||||
"--norestore",
|
||||
"vnd.sun.star.script:Standard.Module1.RecalculateAndSave?language=Basic&location=application",
|
||||
abs_path,
|
||||
]
|
||||
|
||||
# Encapsuler avec timeout si disponible
|
||||
if platform.system() == "Linux":
|
||||
cmd = ["timeout", str(timeout)] + cmd
|
||||
elif platform.system() == "Darwin" and has_gtimeout():
|
||||
cmd = ["gtimeout", str(timeout)] + cmd
|
||||
|
||||
try:
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, env=get_soffice_env(), timeout=timeout+10)
|
||||
except subprocess.TimeoutExpired:
|
||||
return {"error": "Timeout lors du recalcul"}
|
||||
|
||||
if result.returncode != 0 and result.returncode != 124: # 124 = code timeout
|
||||
error_msg = result.stderr or "Erreur inconnue lors du recalcul"
|
||||
if "Module1" in error_msg or "RecalculateAndSave" not in error_msg:
|
||||
return {"error": "Macro LibreOffice mal configurée"}
|
||||
return {"error": error_msg}
|
||||
|
||||
# Vérifier erreurs Excel dans le fichier recalculé
|
||||
try:
|
||||
wb = load_workbook(filename, data_only=True)
|
||||
|
||||
excel_errors = [
|
||||
"#VALUE!",
|
||||
"#DIV/0!",
|
||||
"#REF!",
|
||||
"#NAME?",
|
||||
"#NULL!",
|
||||
"#NUM!",
|
||||
"#N/A",
|
||||
]
|
||||
error_details = {err: [] for err in excel_errors}
|
||||
total_errors = 0
|
||||
|
||||
for sheet_name in wb.sheetnames:
|
||||
ws = wb[sheet_name]
|
||||
for row in ws.iter_rows():
|
||||
for cell in row:
|
||||
if cell.value is not None and isinstance(cell.value, str):
|
||||
for err in excel_errors:
|
||||
if err in cell.value:
|
||||
location = f"{sheet_name}!{cell.coordinate}"
|
||||
error_details[err].append(location)
|
||||
total_errors += 1
|
||||
break
|
||||
|
||||
wb.close()
|
||||
|
||||
# Construire résumé
|
||||
result = {
|
||||
"status": "success" if total_errors == 0 else "errors_found",
|
||||
"total_errors": total_errors,
|
||||
"error_summary": {},
|
||||
}
|
||||
|
||||
# Ajouter catégories d'erreurs non vides
|
||||
for err_type, locations in error_details.items():
|
||||
if locations:
|
||||
result["error_summary"][err_type] = {
|
||||
"count": len(locations),
|
||||
"locations": locations[:20], # Max 20 emplacements affichés
|
||||
}
|
||||
|
||||
# Ajouter compte de formules
|
||||
wb_formulas = load_workbook(filename, data_only=False)
|
||||
formula_count = 0
|
||||
for sheet_name in wb_formulas.sheetnames:
|
||||
ws = wb_formulas[sheet_name]
|
||||
for row in ws.iter_rows():
|
||||
for cell in row:
|
||||
if (
|
||||
cell.value
|
||||
and isinstance(cell.value, str)
|
||||
and cell.value.startswith("=")
|
||||
):
|
||||
formula_count += 1
|
||||
wb_formulas.close()
|
||||
|
||||
result["total_formulas"] = formula_count
|
||||
|
||||
return result
|
||||
|
||||
except Exception as e:
|
||||
return {"error": str(e)}
|
||||
|
||||
|
||||
def main():
|
||||
if len(sys.argv) < 2:
|
||||
print("Usage: python recalc.py <fichier_excel> [timeout_secondes]")
|
||||
print("\nRecalcule toutes les formules d'un fichier Excel via LibreOffice")
|
||||
print("\nRetourne JSON avec détails d'erreurs:")
|
||||
print(" - status: 'success' ou 'errors_found'")
|
||||
print(" - total_errors: Nombre total d'erreurs Excel")
|
||||
print(" - total_formulas: Nombre de formules dans le fichier")
|
||||
print(" - error_summary: Détail par type avec emplacements")
|
||||
print(" - #VALUE!, #DIV/0!, #REF!, #NAME?, #NULL!, #NUM!, #N/A")
|
||||
sys.exit(1)
|
||||
|
||||
filename = sys.argv[1]
|
||||
timeout = int(sys.argv[2]) if len(sys.argv) > 2 else 30
|
||||
|
||||
result = recalc(filename, timeout)
|
||||
print(json.dumps(result, indent=2, ensure_ascii=False))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -120,134 +120,6 @@ def _resolve_tool_conflicts(collected: list[tuple[str, ToolSpec]]) -> list[ToolS
|
|||
return output
|
||||
|
||||
|
||||
def _skill_management_tools(store: SqliteStore) -> list[ToolSpec]:
|
||||
# Lazy imports to avoid circular dependency at module import time.
|
||||
from oclaw.runtime.skill_installer import (
|
||||
auto_install_skill_from_payload,
|
||||
create_skill_from_template,
|
||||
install_skill_from_registry_archive,
|
||||
list_skills_with_status,
|
||||
)
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting
|
||||
|
||||
def _create_skill_handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
out = create_skill_from_template(
|
||||
store=store,
|
||||
name=str(args.get("name") or "").strip(),
|
||||
description=str(args.get("description") or "").strip(),
|
||||
body_markdown=str(args.get("body_markdown") or "").strip(),
|
||||
metadata_oclaw=dict(args.get("metadata_oclaw") or {}) if isinstance(args.get("metadata_oclaw"), dict) else {},
|
||||
overwrite=bool(args.get("overwrite")),
|
||||
)
|
||||
return {"ok": bool(out.ok), "name": out.name, "target_dir": out.target_dir, "detail": out.detail}
|
||||
|
||||
def _auto_install_skill_handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
archive_url = str(args.get("archive_url") or "").strip()
|
||||
slug = str(args.get("slug") or "").strip()
|
||||
version = str(args.get("version") or "").strip() or None
|
||||
overwrite = bool(args.get("overwrite"))
|
||||
provider = normalize_skill_market_provider_setting(
|
||||
str(args.get("provider") or store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "")
|
||||
)
|
||||
if not archive_url and slug:
|
||||
try:
|
||||
adapter = get_market_adapter(provider)
|
||||
archive_url, _chosen_version = adapter.resolve_archive_url(slug=slug, version=version)
|
||||
except Exception:
|
||||
archive_url = ""
|
||||
if archive_url:
|
||||
out = install_skill_from_registry_archive(
|
||||
store=store,
|
||||
archive_url=archive_url,
|
||||
overwrite=overwrite,
|
||||
skills_root=default_skills_root() / "_workspace",
|
||||
auto_bind=True,
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
"provider": provider,
|
||||
}
|
||||
|
||||
payload = {
|
||||
"name": str(args.get("name") or "").strip(),
|
||||
"description": str(args.get("description") or "").strip(),
|
||||
"body_markdown": str(args.get("body_markdown") or "").strip(),
|
||||
"metadata_oclaw": dict(args.get("metadata_oclaw") or {}) if isinstance(args.get("metadata_oclaw"), dict) else {},
|
||||
}
|
||||
if not payload["name"]:
|
||||
return {"ok": False, "error_code": "name_required", "error": "name_required"}
|
||||
if not payload["description"]:
|
||||
payload["description"] = f"{payload['name']} skill"
|
||||
out = auto_install_skill_from_payload(store=store, payload=payload)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
}
|
||||
|
||||
def _list_skills_handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
del args
|
||||
return {"ok": True, "items": list_skills_with_status(store=store)}
|
||||
|
||||
schema = {
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {"type": "string"},
|
||||
"description": {"type": "string"},
|
||||
"body_markdown": {"type": "string"},
|
||||
"metadata_oclaw": {"type": "object"},
|
||||
"slug": {"type": "string"},
|
||||
"provider": {"type": "string"},
|
||||
"version": {"type": "string"},
|
||||
"archive_url": {"type": "string"},
|
||||
"overwrite": {"type": "boolean"},
|
||||
},
|
||||
"required": [],
|
||||
}
|
||||
return [
|
||||
ToolSpec(
|
||||
name="skill_create",
|
||||
description="Create a local oclaw skill package from template.",
|
||||
parameters=schema,
|
||||
handler=_create_skill_handler,
|
||||
tags=frozenset({"skill", "oclaw", "builder"}),
|
||||
risk_level="high",
|
||||
timeout_s=20.0,
|
||||
),
|
||||
ToolSpec(
|
||||
name="skill_auto_install",
|
||||
description="Auto install a generated oclaw skill package.",
|
||||
parameters=schema,
|
||||
handler=_auto_install_skill_handler,
|
||||
tags=frozenset({"skill", "oclaw", "installer"}),
|
||||
risk_level="high",
|
||||
timeout_s=120.0,
|
||||
),
|
||||
ToolSpec(
|
||||
name="skill_list",
|
||||
description="List installed skills and status.",
|
||||
parameters={"type": "object", "properties": {}, "additionalProperties": False},
|
||||
handler=_list_skills_handler,
|
||||
tags=frozenset({"skill", "oclaw", "read"}),
|
||||
read_only=True,
|
||||
timeout_s=10.0,
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
def materialize_tool_specs(
|
||||
factories: tuple[ToolFactory, ...] | None = None,
|
||||
*,
|
||||
|
|
|
|||
109
runtime/tools/public/skill_auto_install_tool.py
Normal file
109
runtime/tools/public/skill_auto_install_tool.py
Normal file
|
|
@ -0,0 +1,109 @@
|
|||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.config.paths import db_path
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
from oclaw.runtime.skill_installer import auto_install_skill_from_payload, install_skill_from_registry_archive
|
||||
from oclaw.runtime.skills import default_skills_root
|
||||
from oclaw.runtime.skills_market import get_market_adapter, normalize_skill_market_provider_setting
|
||||
from oclaw.runtime.tools.base import ToolSpec
|
||||
|
||||
|
||||
def _store() -> SqliteStore:
|
||||
return SqliteStore(db_path())
|
||||
|
||||
|
||||
def _agent_workspace_skills_root() -> Path:
|
||||
return default_skills_root() / "_workspace"
|
||||
|
||||
|
||||
def skill_auto_install_tool() -> ToolSpec:
|
||||
def _handler(args: dict[str, Any]) -> dict[str, Any]:
|
||||
payload = args if isinstance(args, dict) else {}
|
||||
store = _store()
|
||||
archive_url = str(payload.get("archive_url") or "").strip()
|
||||
slug = str(payload.get("slug") or "").strip()
|
||||
version = str(payload.get("version") or "").strip() or None
|
||||
overwrite = bool(payload.get("overwrite"))
|
||||
provider = normalize_skill_market_provider_setting(
|
||||
str(payload.get("provider") or store.get_setting("AIA_SKILL_MARKET_PROVIDER") or "")
|
||||
)
|
||||
if not archive_url and slug:
|
||||
try:
|
||||
adapter = get_market_adapter(provider)
|
||||
archive_url, _chosen_version = adapter.resolve_archive_url(slug=slug, version=version)
|
||||
except Exception:
|
||||
archive_url = ""
|
||||
if archive_url:
|
||||
out = install_skill_from_registry_archive(
|
||||
store=store,
|
||||
archive_url=archive_url,
|
||||
overwrite=overwrite,
|
||||
skills_root=_agent_workspace_skills_root(),
|
||||
auto_bind=True,
|
||||
)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
"provider": provider,
|
||||
}
|
||||
|
||||
skill_payload = {
|
||||
"name": str(payload.get("name") or "").strip(),
|
||||
"description": str(payload.get("description") or "").strip(),
|
||||
"body_markdown": str(payload.get("body_markdown") or "").strip(),
|
||||
"metadata_oclaw": dict(payload.get("metadata_oclaw") or {})
|
||||
if isinstance(payload.get("metadata_oclaw"), dict)
|
||||
else {},
|
||||
}
|
||||
if not skill_payload["name"]:
|
||||
return {"ok": False, "error_code": "name_required", "error": "name_required"}
|
||||
if not skill_payload["description"]:
|
||||
skill_payload["description"] = f"{skill_payload['name']} skill"
|
||||
out = auto_install_skill_from_payload(store=store, payload=skill_payload)
|
||||
return {
|
||||
"ok": bool(out.ok),
|
||||
"name": out.name,
|
||||
"target_dir": out.target_dir,
|
||||
"detail": out.detail,
|
||||
"error_code": out.error_code,
|
||||
"retryable": bool(out.retryable),
|
||||
"auto_enabled": bool(getattr(out, "auto_enabled", False)),
|
||||
"binding_applied_roles": list(getattr(out, "binding_applied_roles", ()) or []),
|
||||
}
|
||||
|
||||
return ToolSpec(
|
||||
name="skill_auto_install",
|
||||
description="Auto install a skill into _workspace lane (payload or market/archive).",
|
||||
parameters={
|
||||
"type": "object",
|
||||
"properties": {
|
||||
"name": {"type": "string"},
|
||||
"description": {"type": "string"},
|
||||
"body_markdown": {"type": "string"},
|
||||
"metadata_oclaw": {"type": "object"},
|
||||
"slug": {"type": "string"},
|
||||
"provider": {"type": "string"},
|
||||
"version": {"type": "string"},
|
||||
"archive_url": {"type": "string"},
|
||||
"overwrite": {"type": "boolean"},
|
||||
},
|
||||
"required": [],
|
||||
"additionalProperties": False,
|
||||
},
|
||||
handler=_handler,
|
||||
tags=frozenset({"skill", "oclaw", "installer"}),
|
||||
risk_level="high",
|
||||
timeout_s=120.0,
|
||||
)
|
||||
|
||||
|
||||
__all__ = ["skill_auto_install_tool"]
|
||||
Loading…
Add table
Add a link
Reference in a new issue