mirror of
https://github.com/hansjone/oclaw.git
synced 2026-10-10 13:00:45 +08:00
重构仓库目录为统一的 runtime 分层并清理历史 openclaw 残留。
本次迁移将网关/通道/工具/技能/脚本与协议资源集中到新结构,统一路径常量与脚本转发机制,减少顶层噪音并保证运行与测试行为一致。 Made-with: Cursor
This commit is contained in:
parent
ba3836f00f
commit
4a23b715a2
498 changed files with 2760 additions and 2200 deletions
101
runtime/tools/workspace_indexer.py
Normal file
101
runtime/tools/workspace_indexer.py
Normal file
|
|
@ -0,0 +1,101 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
from oclaw.platform.embeddings.embedding_client import build_default_embedding_client
|
||||
from oclaw.platform.persistence.sqlite_store import SqliteStore
|
||||
from oclaw.platform.config.paths import PROJECT_ROOT
|
||||
|
||||
|
||||
def _default_workspace_root() -> Path:
|
||||
import os
|
||||
|
||||
override = (os.getenv("AIA_WORKSPACE_ROOT") or "").strip()
|
||||
if override:
|
||||
return Path(override).expanduser().resolve()
|
||||
return Path(PROJECT_ROOT).resolve()
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class IndexStats:
|
||||
files_seen: int
|
||||
chunks_upserted: int
|
||||
embeddings_upserted: int
|
||||
|
||||
|
||||
def _chunk_id(source: str, text: str) -> str:
|
||||
raw = f"{source}\n{text}".encode("utf-8", errors="ignore")
|
||||
return hashlib.sha1(raw).hexdigest()
|
||||
|
||||
|
||||
def _iter_files(root: Path, patterns: list[str], *, max_files: int) -> list[Path]:
|
||||
files: list[Path] = []
|
||||
for pat in patterns:
|
||||
for p in root.glob(pat):
|
||||
if p.is_dir():
|
||||
continue
|
||||
files.append(p)
|
||||
if len(files) >= max_files:
|
||||
return files
|
||||
# de-dup by path
|
||||
uniq = []
|
||||
seen = set()
|
||||
for p in files:
|
||||
rp = str(p)
|
||||
if rp in seen:
|
||||
continue
|
||||
seen.add(rp)
|
||||
uniq.append(p)
|
||||
return uniq[:max_files]
|
||||
|
||||
|
||||
def index_workspace(
|
||||
store: SqliteStore,
|
||||
*,
|
||||
root: str | None = None,
|
||||
patterns: list[str] | None = None,
|
||||
max_files: int = 200,
|
||||
max_chars_per_chunk: int = 1600,
|
||||
) -> IndexStats:
|
||||
r = Path(root).resolve() if root else _default_workspace_root()
|
||||
pats = patterns or ["oclaw/**/*.py", "README*.md", "**/*.md"]
|
||||
files = _iter_files(r, pats, max_files=max(1, int(max_files)))
|
||||
client = build_default_embedding_client()
|
||||
chunks_upserted = 0
|
||||
embeds_upserted = 0
|
||||
|
||||
for p in files:
|
||||
try:
|
||||
text = p.read_text(encoding="utf-8", errors="replace")
|
||||
except Exception:
|
||||
continue
|
||||
rel = str(p.relative_to(r))
|
||||
# naive chunking by chars (good enough for MVP)
|
||||
for idx in range(0, len(text), max(200, int(max_chars_per_chunk))):
|
||||
chunk = text[idx : idx + int(max_chars_per_chunk)].strip()
|
||||
if not chunk:
|
||||
continue
|
||||
source = f"workspace:{rel}#c{idx//max_chars_per_chunk:04d}"
|
||||
cid = _chunk_id(source, chunk)
|
||||
store.upsert_knowledge_chunk(
|
||||
chunk_id=cid,
|
||||
source=source,
|
||||
content=chunk,
|
||||
metadata={"source": source, "path": rel, "offset": idx},
|
||||
)
|
||||
chunks_upserted += 1
|
||||
try:
|
||||
emb = client.embed(chunk[:8000])
|
||||
store.upsert_knowledge_embedding(chunk_id=cid, model=emb.model, vector=emb.vector)
|
||||
embeds_upserted += 1
|
||||
except Exception:
|
||||
continue
|
||||
|
||||
return IndexStats(files_seen=len(files), chunks_upserted=chunks_upserted, embeddings_upserted=embeds_upserted)
|
||||
|
||||
|
||||
__all__ = ["IndexStats", "index_workspace"]
|
||||
|
||||
Loading…
Add table
Add a link
Reference in a new issue