mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-09 07:13:17 +08:00
47 lines
No EOL
1.4 KiB
JSON
47 lines
No EOL
1.4 KiB
JSON
{
|
|
"name": "word-reader",
|
|
"version": "1.0.0",
|
|
"description": "读取 Word 文档(.docx 和 .doc 格式)并提取文本内容",
|
|
"author": "OpenClaw User",
|
|
"tags": ["document", "word", "office", "text-extraction"],
|
|
"dependencies": {
|
|
"python": ">=3.6",
|
|
"packages": ["python-docx"],
|
|
"system": ["antiword (optional for .doc support)"]
|
|
},
|
|
"features": {
|
|
"text_extraction": true,
|
|
"table_parsing": true,
|
|
"metadata_extraction": true,
|
|
"image_info": true,
|
|
"batch_processing": true,
|
|
"multiple_formats": ["json", "text", "markdown"]
|
|
},
|
|
"installation": {
|
|
"steps": [
|
|
"pip3 install python-docx",
|
|
"sudo apt-get install antiword # 可选,支持 .doc 格式",
|
|
"chmod +x scripts/read_word.py"
|
|
]
|
|
},
|
|
"usage_examples": [
|
|
{
|
|
"description": "读取文档文本",
|
|
"command": "python3 scripts/read_word.py document.docx"
|
|
},
|
|
{
|
|
"description": "转换为 Markdown",
|
|
"command": "python3 scripts/read_word.py document.docx --format markdown"
|
|
},
|
|
{
|
|
"description": "批量处理",
|
|
"command": "python3 scripts/read_word.py ./docs --batch --format json"
|
|
}
|
|
],
|
|
"supported_file_types": [".docx", ".doc"],
|
|
"notes": [
|
|
".doc 格式需要安装 antiword",
|
|
"大文档处理可能需要较长时间",
|
|
"图片提取仅获取元数据,不包含实际图片数据"
|
|
]
|
|
} |