统一公共工具并移除 workspace 专家链路。

将文件/命令/Git/检索能力统一迁入 public,去除重复工具与 local_ 命名前缀,并重命名底层路径守卫模块为 path_guard 以提升语义清晰度。

Made-with: Cursor
This commit is contained in:
oliver 2026-05-01 00:38:31 +08:00
parent 75aec2abe4
commit d0a66d190d
21 changed files with 326 additions and 1012 deletions

View file

@ -1,4 +0,0 @@
from __future__ import annotations
__all__ = []

View file

@ -1,178 +0,0 @@
from __future__ import annotations
import hashlib
from pathlib import Path
from typing import Any
from oclaw.runtime.tools.base import ToolSpec
from oclaw.runtime.tools.experts.workspace.workspace_base import (
resolve_workspace_path,
)
def read_file_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
path = str(args.get("path") or "").strip()
offset = int(args.get("offset") or 1)
limit = int(args.get("limit") or 400)
if offset == 0:
offset = 1
if limit <= 0:
limit = 1
p = resolve_workspace_path(path)
if not p.exists() or not p.is_file():
return {"ok": False, "error": "file_not_found", "path": str(p)}
text = p.read_text(encoding="utf-8", errors="replace").splitlines()
# 1-indexed offsets; negative counts from end
if offset < 0:
start = max(0, len(text) + offset)
else:
start = max(0, offset - 1)
end = min(len(text), start + min(limit, 2000))
out_lines = [f"{i+1}|{text[i]}" for i in range(start, end)]
blob = p.read_bytes()
sha = hashlib.sha256(blob).hexdigest()
return {
"ok": True,
"path": str(p),
"start_line": start + 1,
"end_line": end,
"total_lines": len(text),
"sha256": sha,
"content": "\n".join(out_lines),
}
return ToolSpec(
name="read_file",
description="Read a text file from the workspace with line numbers.",
parameters={
"type": "object",
"properties": {
"path": {"type": "string", "description": "File path, relative to workspace root."},
"offset": {"type": "integer", "description": "1-indexed start line; negative counts from end.", "default": 1},
"limit": {"type": "integer", "description": "Max lines to return (capped).", "default": 400},
},
"required": ["path"],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace"}),
read_only=True,
)
def write_file_tool() -> ToolSpec:
def _sandbox_base_dir() -> Path:
return Path("data") / "workspace"
def _normalize_write_path(path: str) -> str:
raw = str(path or "").strip().strip('"').strip("'")
if not raw:
raise ValueError("path_required")
p = Path(raw)
base = _sandbox_base_dir()
# Enforce sandbox for absolute paths as well.
if p.is_absolute():
# Collapse absolute user path into sandbox-relative target to prevent
# writes to repo root or arbitrary host locations.
name = str(p.name or "").strip()
if not name:
raise ValueError("path_required")
return str(base / name)
# Keep generated files out of repo root: default relative writes go under data/workspace/...
rel = raw.lstrip("./\\")
if not rel:
raise ValueError("path_required")
return str(base / rel)
def handler(args: dict[str, Any]) -> dict[str, Any]:
path = str(args.get("path") or "").strip()
content = str(args.get("content") or "")
mode = str(args.get("mode") or "overwrite").strip().lower()
try:
normalized = _normalize_write_path(path)
except ValueError as exc:
return {"ok": False, "error": str(exc)}
p = resolve_workspace_path(normalized)
p.parent.mkdir(parents=True, exist_ok=True)
if mode not in ("overwrite", "append"):
return {"ok": False, "error": "invalid_mode", "allowed": ["overwrite", "append"]}
if mode == "append":
p.write_text(p.read_text(encoding="utf-8", errors="replace") + content, encoding="utf-8")
else:
p.write_text(content, encoding="utf-8")
return {"ok": True, "path": str(p), "bytes": p.stat().st_size}
return ToolSpec(
name="write_file",
description="Write text content to a workspace file (overwrite or append).",
parameters={
"type": "object",
"properties": {
"path": {"type": "string", "description": "File path, relative to workspace root."},
"content": {"type": "string", "description": "Full text content to write."},
"mode": {"type": "string", "enum": ["overwrite", "append"], "default": "overwrite"},
},
"required": ["path", "content"],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "write"}),
)
def list_files_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
pattern = str(args.get("pattern") or "**/*").strip() or "**/*"
max_results = int(args.get("max_results") or 200)
root_arg = str(args.get("root") or "").strip()
if not root_arg:
base = resolve_workspace_path(".")
else:
base = resolve_workspace_path(root_arg)
if not base.is_dir():
return {"ok": False, "error": "not_a_directory", "path": str(base)}
out: list[str] = []
for p in base.glob(pattern):
if p.is_dir():
continue
rel = str(p.relative_to(base))
out.append(rel)
if len(out) >= max(1, min(max_results, 2000)):
break
return {
"ok": True,
"root": str(base),
"pattern": pattern,
"count": len(out),
"files": out,
}
return ToolSpec(
name="glob",
description=(
"List files under a directory matching a glob pattern. "
"Default root is the workspace root; set `root` to an absolute path (e.g. D:\\\\download) when the user names a folder outside the repo — "
"this respects gateway workspace path policy. Prefer this over MCP filesystem list_directory when the user path may be outside MCP's configured roots."
),
parameters={
"type": "object",
"properties": {
"pattern": {"type": "string", "description": "Glob pattern relative to root, e.g. '**/*' or '*.pdf'.", "default": "**/*"},
"root": {
"type": "string",
"description": "Optional directory to search under (absolute or workspace-relative). If omitted, uses workspace root.",
},
"max_results": {"type": "integer", "default": 200, "description": "Max number of files to return."},
},
"required": [],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace"}),
read_only=True,
)
__all__ = ["read_file_tool", "write_file_tool", "list_files_tool"]

View file

@ -1,154 +0,0 @@
from __future__ import annotations
import subprocess
from typing import Any
from oclaw.runtime.tools.base import ToolSpec
from oclaw.runtime.tools.experts.workspace.workspace_base import resolve_workspace_path, truncate_text, sanitize_git_ref
def _git(command: str, *, cwd: str) -> dict[str, Any]:
workdir = resolve_workspace_path(cwd or ".")
cp = subprocess.run(
f"git {command}",
cwd=str(workdir),
shell=True,
capture_output=True,
text=True,
timeout=60.0,
)
out = (cp.stdout or "") + (("\n" + cp.stderr) if cp.stderr else "")
return {"exit_code": int(cp.returncode), "output": truncate_text(out, limit=20000), "cwd": str(workdir)}
def git_status_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
cwd = str(args.get("cwd") or ".").strip()
res = _git("status --porcelain=v1 -b", cwd=cwd)
ok = res["exit_code"] == 0
return {"ok": ok, **res}
return ToolSpec(
name="git_status",
description="Show git status (porcelain).",
parameters={
"type": "object",
"properties": {"cwd": {"type": "string", "default": ".", "description": "Repo directory."}},
"required": [],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "git"}),
)
def git_diff_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
cwd = str(args.get("cwd") or ".").strip()
ref = sanitize_git_ref(str(args.get("ref") or "").strip()) if args.get("ref") else ""
cmd = "diff" if not ref else f"diff {ref}...HEAD"
res = _git(cmd, cwd=cwd)
ok = res["exit_code"] == 0
return {"ok": ok, **res}
return ToolSpec(
name="git_diff",
description="Show git diff (default: working tree; optional ref...HEAD).",
parameters={
"type": "object",
"properties": {
"cwd": {"type": "string", "default": ".", "description": "Repo directory."},
"ref": {"type": "string", "description": "Optional ref for ref...HEAD diff."},
},
"required": [],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "git"}),
)
def git_log_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
cwd = str(args.get("cwd") or ".").strip()
n = int(args.get("n") or 10)
n = max(1, min(n, 50))
res = _git(f"log -{n} --oneline --decorate", cwd=cwd)
ok = res["exit_code"] == 0
return {"ok": ok, **res}
return ToolSpec(
name="git_log",
description="Show recent git commits (oneline).",
parameters={
"type": "object",
"properties": {"cwd": {"type": "string", "default": ".", "description": "Repo directory."}, "n": {"type": "integer", "default": 10}},
"required": [],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "git"}),
)
def git_commit_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
cwd = str(args.get("cwd") or ".").strip()
message = str(args.get("message") or "").strip()
if not message:
return {"ok": False, "error": "message_required"}
# stage all changes (simple default)
s1 = _git("add -A", cwd=cwd)
if s1["exit_code"] != 0:
return {"ok": False, "error": "git_add_failed", **s1}
msg_esc = message.replace('"', '\\"')
s2 = _git(f'commit -m "{msg_esc}"', cwd=cwd)
ok = s2["exit_code"] == 0
return {"ok": ok, **s2}
return ToolSpec(
name="git_commit",
description="Stage all and create a git commit (requires confirmation by policy).",
parameters={
"type": "object",
"properties": {
"cwd": {"type": "string", "default": ".", "description": "Repo directory."},
"message": {"type": "string", "description": "Commit message."},
},
"required": ["message"],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "git", "write"}),
)
def git_push_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
cwd = str(args.get("cwd") or ".").strip()
remote = str(args.get("remote") or "origin").strip() or "origin"
refspec = str(args.get("refspec") or "HEAD").strip() or "HEAD"
res = _git(f"push {remote} {refspec}", cwd=cwd)
ok = res["exit_code"] == 0
return {"ok": ok, **res}
return ToolSpec(
name="git_push",
description="Push current branch (requires confirmation by policy).",
parameters={
"type": "object",
"properties": {
"cwd": {"type": "string", "default": ".", "description": "Repo directory."},
"remote": {"type": "string", "default": "origin"},
"refspec": {"type": "string", "default": "HEAD"},
},
"required": [],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "git", "write"}),
)
__all__ = ["git_status_tool", "git_diff_tool", "git_log_tool", "git_commit_tool", "git_push_tool"]

View file

@ -1,53 +0,0 @@
from __future__ import annotations
import hashlib
from typing import Any
from oclaw.runtime.tools.base import ToolSpec
from oclaw.runtime.tools.experts.workspace.workspace_base import resolve_workspace_path
def apply_patch_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
path = str(args.get("path") or "").strip()
new_content = str(args.get("new_content") or "")
expected_sha256 = str(args.get("expected_sha256") or "").strip()
p = resolve_workspace_path(path)
if p.exists() and p.is_file() and expected_sha256:
cur = hashlib.sha256(p.read_bytes()).hexdigest()
if cur != expected_sha256:
return {
"ok": False,
"error": "sha_mismatch",
"path": str(p),
"expected_sha256": expected_sha256,
"current_sha256": cur,
}
p.parent.mkdir(parents=True, exist_ok=True)
p.write_text(new_content, encoding="utf-8")
sha = hashlib.sha256(p.read_bytes()).hexdigest()
return {"ok": True, "path": str(p), "sha256": sha, "bytes": p.stat().st_size}
return ToolSpec(
name="apply_patch",
description="Apply a full-file patch by overwriting a file with new content (optional sha256 precondition).",
parameters={
"type": "object",
"properties": {
"path": {"type": "string", "description": "File path, relative to workspace root."},
"new_content": {"type": "string", "description": "New full file content."},
"expected_sha256": {
"type": "string",
"description": "If provided, the current file sha256 must match (precondition).",
},
},
"required": ["path", "new_content"],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "write"}),
)
__all__ = ["apply_patch_tool"]

View file

@ -1,92 +0,0 @@
from __future__ import annotations
import re
from typing import Any
from oclaw.runtime.tools.base import ToolSpec
from oclaw.runtime.tools.experts.workspace.workspace_base import resolve_workspace_path
def grep_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
pattern = str(args.get("pattern") or "").strip()
file_glob = str(args.get("file_glob") or "**/*").strip() or "**/*"
max_matches = int(args.get("max_matches") or 200)
if not pattern:
return {"ok": False, "error": "pattern_required"}
root = resolve_workspace_path(".")
try:
rx = re.compile(pattern)
except re.error as e:
return {"ok": False, "error": "invalid_regex", "detail": str(e)}
matches: list[dict[str, Any]] = []
for p in root.glob(file_glob):
if p.is_dir():
continue
try:
text = p.read_text(encoding="utf-8", errors="replace").splitlines()
except Exception:
continue
for i, line in enumerate(text, start=1):
if rx.search(line):
matches.append({"file": str(p.relative_to(root)), "line": i, "text": line[:400]})
if len(matches) >= max(1, min(max_matches, 5000)):
return {"ok": True, "pattern": pattern, "count": len(matches), "matches": matches}
return {"ok": True, "pattern": pattern, "count": len(matches), "matches": matches}
return ToolSpec(
name="grep",
description="Search files in the workspace for a regex pattern.",
parameters={
"type": "object",
"properties": {
"pattern": {"type": "string", "description": "Regex pattern."},
"file_glob": {"type": "string", "default": "**/*", "description": "Glob of files to search."},
"max_matches": {"type": "integer", "default": 200, "description": "Max number of matches."},
},
"required": ["pattern"],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace"}),
read_only=True,
)
def index_workspace_tool() -> ToolSpec:
def handler(args: dict[str, Any]) -> dict[str, Any]:
max_files = int(args.get("max_files") or 120)
try:
# Lazy import to avoid heavy deps during tool discovery.
from oclaw.platform.config.paths import db_path
except Exception:
pass
# Indexer uses store passed via closure? ToolSpec doesn't carry store.
# We index using the global SqliteStore path (same as app runtime).
try:
from oclaw.platform.persistence.sqlite_store import SqliteStore
from oclaw.platform.config.paths import db_path
from oclaw.runtime.tools.workspace_indexer import index_workspace
store = SqliteStore(db_path())
st = index_workspace(store, max_files=max(1, min(max_files, 800)))
return {"ok": True, "files_seen": st.files_seen, "chunks_upserted": st.chunks_upserted, "embeddings_upserted": st.embeddings_upserted}
except Exception as e:
return {"ok": False, "error": f"{type(e).__name__}: {e}"}
return ToolSpec(
name="index_workspace",
description="Index workspace files into the vector knowledge base for RAG (may be slow).",
parameters={
"type": "object",
"properties": {"max_files": {"type": "integer", "default": 120, "description": "Max files to index."}},
"required": [],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "rag"}),
)
__all__ = ["grep_tool", "index_workspace_tool"]

View file

@ -1,315 +0,0 @@
from __future__ import annotations
import subprocess
import re
from pathlib import Path
from typing import Any
from oclaw.platform.config.paths import db_path
from oclaw.platform.persistence.sqlite_store import SqliteStore
from oclaw.runtime.tools.base import ToolSpec
from oclaw.runtime.tools.experts.workspace.workspace_base import (
resolve_workspace_path,
truncate_text,
workspace_root,
)
def run_command_tool() -> ToolSpec:
_LEADING_CD_CHAIN_RE = re.compile(
r"^\s*(?:(?:[A-Za-z]:)\s*&&\s*)?(?:@echo\s+off\s*&&\s*)?cd\s+(?:/d\s+)?(?:\"[^\"]+\"|[^&]+?)\s*&&\s*(.+)$",
re.IGNORECASE | re.DOTALL,
)
def _run_command_enabled() -> bool:
import os
try:
raw_setting = str(SqliteStore(db_path()).get_setting("AIA_ENABLE_RUN_COMMAND") or "").strip().lower()
if raw_setting in ("0", "false", "no", "off"):
return False
if raw_setting in ("1", "true", "yes", "on"):
return True
except Exception:
pass
raw_env = str(os.getenv("AIA_ENABLE_RUN_COMMAND") or "").strip().lower()
if raw_env in ("0", "false", "no", "off"):
return False
if raw_env in ("1", "true", "yes", "on"):
return True
# Default disabled when unset (explicit opt-in only).
return False
def handler(args: dict[str, Any]) -> dict[str, Any]:
import os
def _default_exec_dir() -> str:
# Keep command execution in the same sandbox namespace as write_file.
try:
return str(resolve_workspace_path("data/workspace"))
except Exception:
return str(workspace_root())
def _strip_leading_cd_chain(cmd: str) -> tuple[str, bool]:
raw = str(cmd or "")
changed = False
out = raw
# Strip repeated leading "cd ... &&" so default sandbox cwd cannot be bypassed by habit.
for _ in range(3):
m = _LEADING_CD_CHAIN_RE.match(out)
if not m:
break
tail = str(m.group(1) or "").strip()
if not tail:
break
out = tail
changed = True
return out, changed
def _rewrite_workspace_absolute_refs(cmd: str, *, workdir: str) -> tuple[str, bool]:
raw = str(cmd or "")
root = str(workspace_root())
if not raw or not root:
return raw, False
root_norm = root.rstrip("\\/")
changed = False
out = raw
marker = root_norm + "\\"
if marker.lower() not in out.lower():
return out, False
idx = 0
rebuilt = []
low = out.lower()
marker_low = marker.lower()
while True:
pos = low.find(marker_low, idx)
if pos < 0:
rebuilt.append(out[idx:])
break
rebuilt.append(out[idx:pos])
tail_start = pos + len(marker)
tail_end = tail_start
while tail_end < len(out) and out[tail_end] not in ('"', "'", " ", "\t", "\r", "\n"):
tail_end += 1
rel_tail = out[tail_start:tail_end]
candidate = str(Path(workdir) / rel_tail)
if Path(candidate).exists():
rebuilt.append(candidate)
changed = True
else:
rebuilt.append(out[pos:tail_end])
idx = tail_end
return "".join(rebuilt), changed
def _rewrite_python_script_arg(cmd: str, *, workdir: str) -> tuple[str, bool]:
raw = str(cmd or "").strip()
if not raw:
return raw, False
m = re.match(r'^\s*(python|py)\s+("([^"]+\.py)"|([^\s]+\.py))(\s+.*)?$', raw, flags=re.IGNORECASE)
if not m:
return raw, False
script = str(m.group(3) or m.group(4) or "").strip()
if not script:
return raw, False
# Absolute path is handled by workspace-absolute rewrite already.
sp = Path(script)
if sp.is_absolute():
return raw, False
base = str(Path(script).name or "").strip()
if not base:
return raw, False
# Deterministic policy: always bind python script arg to sandbox root.
rel = base
quote = '"' if " " in rel else ""
prefix = str(m.group(1) or "python")
rest = str(m.group(5) or "")
return f"{prefix} {quote}{rel}{quote}{rest}", True
if not _run_command_enabled():
return {
"ok": False,
"error": "disabled",
"hint": "Enable run_command in Admin -> Plugins -> Tool Policy.",
}
command = str(args.get("command") or "").strip()
cwd = str(args.get("cwd") or "").strip()
timeout_s = float(args.get("timeout_s") or 30.0)
max_output_chars = int(args.get("max_output_chars") or 20000)
if not command:
return {"ok": False, "error": "command_required"}
normalized_cd_removed = False
command_rewritten = False
cwd_redirected_to_sandbox = False
script_path_rewritten = False
original_command = command
# Execute in caller-provided cwd (guarded by resolve_workspace_path), otherwise workspace root.
try:
workdir = str(resolve_workspace_path(cwd)) if cwd else _default_exec_dir()
except Exception:
# If caller path is rejected by workspace guard, fall back to workspace root.
workdir = _default_exec_dir()
if cwd:
cwd_redirected_to_sandbox = True
if cwd:
try:
requested_path = Path(cwd).expanduser().resolve()
if requested_path == workspace_root().resolve():
# Explicit repo-root cwd still gets sandboxed to avoid writes/exec at repo root.
workdir = _default_exec_dir()
cwd_redirected_to_sandbox = True
except Exception:
pass
if cwd:
try:
requested = str(Path(cwd).expanduser().resolve())
resolved = str(Path(workdir).expanduser().resolve())
if requested != resolved:
cwd_redirected_to_sandbox = True
except Exception:
# Conservative signal: explicit cwd provided but could not preserve same path.
cwd_redirected_to_sandbox = True
command, normalized_cd_removed = _strip_leading_cd_chain(command)
command, command_rewritten = _rewrite_workspace_absolute_refs(command, workdir=workdir)
command, script_path_rewritten = _rewrite_python_script_arg(command, workdir=workdir)
def _external_skill_install_cli_blocked(raw_cmd: str) -> bool:
s = str(raw_cmd or "").strip()
if not s:
return False
low = s.lower()
if re.match(r"^\s*cocoloop(?:\.cmd|\.exe)?\s+install(?:\s|$)", s, flags=re.IGNORECASE):
return True
if re.match(r"^\s*clawhub(?:\.cmd|\.exe)?\s+install(?:\s|$)", s, flags=re.IGNORECASE):
return True
if re.search(r"\bnpx\b", low) and "clawhub" in low:
return True
if re.match(r"^\s*npm(?:\.cmd|\.exe)?\s+install\b", low) and "clawhub" in low:
return True
return False
if _external_skill_install_cli_blocked(str(command or "")):
return {
"ok": False,
"error_code": "skill_install_cli_blocked",
"error": "skill_install_cli_blocked",
"hint": "Oclaw has no shell skill installer. Use Admin POST /admin/api/skills/market/install or install-registry, or skill_auto_install.",
"command": command,
"cwd": str(workdir),
"normalized_cd_removed": bool(normalized_cd_removed),
"cwd_redirected_to_sandbox": bool(cwd_redirected_to_sandbox),
"command_rewritten": bool(command_rewritten),
"script_path_rewritten": bool(script_path_rewritten),
"original_command": original_command,
}
try:
os.makedirs(workdir, exist_ok=True)
except Exception:
pass
try:
run_kwargs: dict[str, Any] = {
"cwd": str(workdir),
"shell": True,
"capture_output": True,
"text": True,
"timeout": max(1.0, min(timeout_s, 600.0)),
}
if os.name == "nt":
startupinfo = subprocess.STARTUPINFO()
startupinfo.dwFlags |= subprocess.STARTF_USESHOWWINDOW
startupinfo.wShowWindow = 0 # SW_HIDE
run_kwargs["startupinfo"] = startupinfo
run_kwargs["creationflags"] = subprocess.CREATE_NO_WINDOW
cp = subprocess.run(
command,
**run_kwargs,
)
stdout_text = str(cp.stdout or "")
stderr_text = str(cp.stderr or "")
out_raw = stdout_text + (("\n" + stderr_text) if stderr_text else "")
out_limit = max(1000, min(max_output_chars, 200000))
out = truncate_text(out_raw, limit=out_limit)
out_truncated = len(out_raw) > out_limit
out_empty = (len(str(out_raw or "").strip()) == 0)
exit_code = int(cp.returncode)
ok_flag = exit_code == 0
return {
"ok": bool(ok_flag),
"command": command,
"cwd": str(workdir),
"exit_code": exit_code,
"stdout": stdout_text,
"stderr": stderr_text,
"output": out,
"output_chars": int(len(out_raw)),
"output_empty": bool(out_empty),
"output_truncated": bool(out_truncated),
"output_not_truncated": bool(not out_truncated),
"normalized_cd_removed": bool(normalized_cd_removed),
"cwd_redirected_to_sandbox": bool(cwd_redirected_to_sandbox),
"command_rewritten": bool(command_rewritten),
"script_path_rewritten": bool(script_path_rewritten),
"original_command": original_command,
"error_code": ("" if ok_flag else "command_exit_nonzero"),
"output_hint": (
"Command produced empty stdout/stderr; this is not system truncation. "
"Do not claim truncation unless output_truncated=true."
if out_empty
else ""
),
}
except subprocess.TimeoutExpired as e:
partial = ""
try:
partial = ((e.stdout or "") + ("\n" + (e.stderr or "") if e.stderr else "")).strip()
except Exception:
partial = ""
return {
"ok": False,
"error": "timeout",
"command": command,
"cwd": str(workdir),
"timeout_s": timeout_s,
"output": truncate_text(partial, limit=max_output_chars),
"output_empty": len(str(partial or "").strip()) == 0,
"output_truncated": len(str(partial or "")) > int(max_output_chars or 0),
"normalized_cd_removed": bool(normalized_cd_removed),
"cwd_redirected_to_sandbox": bool(cwd_redirected_to_sandbox),
"command_rewritten": bool(command_rewritten),
"script_path_rewritten": bool(script_path_rewritten),
"original_command": original_command,
}
except Exception as e:
return {
"ok": False,
"error": f"{type(e).__name__}: {e}",
"command": command,
"cwd": str(workdir),
"normalized_cd_removed": bool(normalized_cd_removed),
"cwd_redirected_to_sandbox": bool(cwd_redirected_to_sandbox),
"command_rewritten": bool(command_rewritten),
"script_path_rewritten": bool(script_path_rewritten),
"original_command": original_command,
}
return ToolSpec(
name="run_command",
description="Run a shell command inside the workspace (captured output, timeout).",
parameters={
"type": "object",
"properties": {
"command": {"type": "string", "description": "Shell command to run."},
"cwd": {"type": "string", "description": "Working directory relative to workspace.", "default": "."},
"timeout_s": {"type": "number", "default": 30.0, "description": "Command timeout in seconds."},
"max_output_chars": {"type": "integer", "default": 20000, "description": "Max characters to return."},
},
"required": ["command"],
"additionalProperties": False,
},
handler=handler,
tags=frozenset({"workspace", "exec"}),
)
__all__ = ["run_command_tool"]

View file

@ -1,287 +0,0 @@
from __future__ import annotations
import os
import re
import threading
from contextlib import contextmanager
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Iterator
from oclaw.platform.config.paths import PROJECT_ROOT
_TLS = threading.local()
def _env_truthy(name: str) -> bool:
return str(os.getenv(name) or "").strip().lower() in ("1", "true", "yes", "on")
def workspace_root() -> Path:
# Allow explicit override (recommended when running as a packaged app)
# Prefer legacy OPS_* overrides when explicitly provided (tests + backwards compatibility).
override = (os.getenv("OPS_WORKSPACE_ROOT") or os.getenv("AIA_WORKSPACE_ROOT") or "").strip()
if override:
p = Path(override).expanduser()
return p.resolve()
return Path(PROJECT_ROOT).resolve()
def _parse_pipe_separated_roots(raw: str) -> list[Path]:
out: list[Path] = []
for part in (raw or "").split("|"):
p = part.strip().strip('"').strip("'")
if not p:
continue
try:
rp = Path(p).expanduser().resolve()
if rp.is_absolute():
out.append(rp)
except Exception:
continue
return out
@dataclass(frozen=True)
class WorkspacePathAccess:
"""Effective path guard for the current tool invocation (env + optional per-user DB)."""
extra_roots: tuple[Path, ...]
allow_any_path: bool
def access_from_env() -> WorkspacePathAccess:
raw_extra = os.getenv("OPS_WORKSPACE_EXTRA_ROOTS") or os.getenv("AIA_WORKSPACE_EXTRA_ROOTS") or ""
extra = _parse_pipe_separated_roots(raw_extra)
allow = _env_truthy("OPS_WORKSPACE_ALLOW_ANY_PATH") or _env_truthy("AIA_WORKSPACE_ALLOW_ANY_PATH")
return WorkspacePathAccess(extra_roots=tuple(extra), allow_any_path=allow)
def _merge_access(a: WorkspacePathAccess, b: WorkspacePathAccess) -> WorkspacePathAccess:
merged: dict[str, Path] = {}
for p in (*a.extra_roots, *b.extra_roots):
try:
k = str(p.resolve())
except Exception:
k = str(p)
merged.setdefault(k, p)
return WorkspacePathAccess(
extra_roots=tuple(merged.values()),
allow_any_path=bool(a.allow_any_path or b.allow_any_path),
)
def build_workspace_path_access(
store: Any,
session_id: str | None,
*,
owner_fallback_session_id: str | None = None,
allowlist_tenant_id: str | None = None,
allowlist_user_id: str | None = None,
) -> WorkspacePathAccess:
"""Resolve per-user ``extra_roots`` / ``allow_any_path`` from ``user_workspace_path_allowlist``.
``session_id`` is usually the chat row messages are written to (may be a specialist temp session
without ``ui_session_owner``). In that case pass ``owner_fallback_session_id`` = the user's
UI-owned session id so DB allowlist still applies.
If ``get_ui_session_owner`` yields nothing, ``allowlist_tenant_id`` + ``allowlist_user_id``
(from the authenticated user / request metadata) can be used to load the same allowlist, so
a missing ``ui_session_owner`` row does not drop per-user extra roots.
"""
base = access_from_env()
if store is None:
return base
picked_owner: dict[str, Any] | None = None
for cand in (str(session_id or "").strip(), str(owner_fallback_session_id or "").strip()):
if not cand:
continue
try:
own = store.get_ui_session_owner(session_id=cand)
except Exception:
own = None
if not own:
continue
tid = str(own.get("tenant_id") or "").strip()
uid = str(own.get("user_id") or "").strip()
if tid and uid:
picked_owner = own
break
if picked_owner:
tid = str(picked_owner.get("tenant_id") or "").strip()
uid = str(picked_owner.get("user_id") or "").strip()
try:
row = store.get_user_workspace_path_allowlist(tenant_id=tid, user_id=uid)
except Exception:
row = None
if not row:
return base
db_extras = _parse_pipe_separated_roots(str(row.get("extra_roots") or ""))
db_access = WorkspacePathAccess(
extra_roots=tuple(db_extras),
allow_any_path=bool(row.get("allow_any_path")),
)
return _merge_access(base, db_access)
# Fallback: use explicit tenant / user (e.g. wecom or admin ``metadata``) when session is not
# linked in ``ui_session_owner`` (legacy session or data repair in progress).
t2 = str(allowlist_tenant_id or "").strip()
u2 = str(allowlist_user_id or "").strip()
if not t2 or not u2:
return base
try:
row = store.get_user_workspace_path_allowlist(tenant_id=t2, user_id=u2)
except Exception:
row = None
if not row:
return base
db_extras = _parse_pipe_separated_roots(str(row.get("extra_roots") or ""))
db_access = WorkspacePathAccess(
extra_roots=tuple(db_extras),
allow_any_path=bool(row.get("allow_any_path")),
)
return _merge_access(base, db_access)
@contextmanager
def workspace_path_access_scope(
store: Any,
session_id: str | None,
*,
owner_fallback_session_id: str | None = None,
allowlist_tenant_id: str | None = None,
allowlist_user_id: str | None = None,
) -> Iterator[WorkspacePathAccess]:
acc = build_workspace_path_access(
store,
session_id,
owner_fallback_session_id=owner_fallback_session_id,
allowlist_tenant_id=allowlist_tenant_id,
allowlist_user_id=allowlist_user_id,
)
prev = getattr(_TLS, "access", None)
_TLS.access = acc
try:
yield acc
finally:
if prev is None:
if hasattr(_TLS, "access"):
delattr(_TLS, "access")
else:
_TLS.access = prev
def current_workspace_path_access() -> WorkspacePathAccess:
a = getattr(_TLS, "access", None)
if isinstance(a, WorkspacePathAccess):
return a
return access_from_env()
@contextmanager
def workspace_write_namespace_scope(namespace: str | None) -> Iterator[str]:
prev = getattr(_TLS, "write_namespace", None)
ns = str(namespace or "").strip()
_TLS.write_namespace = ns
try:
yield ns
finally:
if prev is None:
if hasattr(_TLS, "write_namespace"):
delattr(_TLS, "write_namespace")
else:
_TLS.write_namespace = prev
def current_workspace_write_namespace() -> str:
ns = str(getattr(_TLS, "write_namespace", "") or "").strip()
if ns:
return ns
root = workspace_root()
return str(root.name or "workspace").strip() or "workspace"
def clear_workspace_path_access_for_tests() -> None:
if hasattr(_TLS, "access"):
delattr(_TLS, "access")
def _is_subpath(path: Path, root: Path) -> bool:
"""``path`` is under ``root`` (treated as a directory), including the root itself.
On Windows, comparison is case- and path-separator-insensitive; ``resolve`` may
not normalize casing consistently across all drives, so we use normcase.
"""
try:
pr = path.resolve()
rr = root.resolve()
except (OSError, ValueError, RuntimeError):
return False
if os.name == "nt":
np = os.path.normcase(str(pr))
nroot = os.path.normcase(str(rr))
if np == nroot:
return True
sep = os.sep
if not nroot.endswith(sep):
nroot = nroot + sep
return np.startswith(nroot) or (np + sep).startswith(nroot)
try:
pr.relative_to(rr)
return True
except (ValueError, OSError, RuntimeError):
return False
def resolve_workspace_path(user_path: str) -> Path:
p = Path(str(user_path or "").strip().strip('"').strip("'") or "")
if not p:
raise ValueError("path is required")
root = workspace_root()
abs_path = p if p.is_absolute() else (root / p)
abs_path = abs_path.resolve()
access = current_workspace_path_access()
if access.allow_any_path:
return abs_path
roots = (root,) + access.extra_roots
if any(_is_subpath(abs_path, r) for r in roots):
return abs_path
raise ValueError("path escapes workspace root")
def truncate_text(s: str, *, limit: int = 20000) -> str:
s = s or ""
if len(s) <= limit:
return s
return s[: max(0, limit - 12)] + "\n...<truncated>"
# NOTE: put '-' at end or escape it to avoid "bad character range" on Windows Python regex.
_SAFE_GIT_REF_RE = re.compile(r"^[A-Za-z0-9._/\\-]{1,80}$")
def sanitize_git_ref(ref: str) -> str:
r = (ref or "").strip()
if not r:
return ""
if not _SAFE_GIT_REF_RE.match(r):
raise ValueError("invalid git ref")
return r
__all__ = [
"WorkspacePathAccess",
"access_from_env",
"build_workspace_path_access",
"clear_workspace_path_access_for_tests",
"current_workspace_path_access",
"current_workspace_write_namespace",
"resolve_workspace_path",
"sanitize_git_ref",
"truncate_text",
"workspace_write_namespace_scope",
"workspace_path_access_scope",
"workspace_root",
]