feat(migration): 建立 Skill 迁移工作流与追踪工具
This commit is contained in:
@@ -0,0 +1,79 @@
|
||||
#!/usr/bin/env python3
|
||||
"""检查 CraftKit Skill 的基础结构、引用和敏感内容。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
from pathlib import Path
|
||||
|
||||
NAME_RE = re.compile(r"^[a-z0-9]+(?:-[a-z0-9]+)*$")
|
||||
LINK_RE = re.compile(r"\[[^]]+]\((?!https?://|#)([^)]+)\)")
|
||||
SECRET_RE = re.compile(
|
||||
r"(?i)(api[_-]?key|password|private[_-]?key|access[_-]?token)\s*[:=]\s*[^\s]+"
|
||||
)
|
||||
|
||||
|
||||
def frontmatter(text: str) -> dict[str, str]:
|
||||
"""解析当前校验所需的简单顶层 frontmatter 字段。"""
|
||||
if not text.startswith("---\n"):
|
||||
return {}
|
||||
end = text.find("\n---", 4)
|
||||
if end < 0:
|
||||
return {}
|
||||
values: dict[str, str] = {}
|
||||
for line in text[4:end].splitlines():
|
||||
if ":" in line and not line.startswith((" ", "\t")):
|
||||
key, value = line.split(":", 1)
|
||||
values[key.strip()] = value.strip().strip("'\"")
|
||||
return values
|
||||
|
||||
|
||||
def check_skill(root: Path, forbidden: list[str]) -> dict:
|
||||
"""返回结构化问题列表,不修改被检查目录。"""
|
||||
issues: list[dict[str, str]] = []
|
||||
skill_md = root / "SKILL.md"
|
||||
if not skill_md.is_file():
|
||||
return {"skill": root.name, "issues": [{"code": "missing-skill-md", "path": "SKILL.md"}]}
|
||||
text = skill_md.read_text(encoding="utf-8-sig")
|
||||
metadata = frontmatter(text)
|
||||
name = metadata.get("name", "")
|
||||
if not NAME_RE.fullmatch(name):
|
||||
issues.append({"code": "invalid-name", "path": "SKILL.md"})
|
||||
if name != root.name:
|
||||
issues.append({"code": "name-path-mismatch", "path": "SKILL.md"})
|
||||
if not metadata.get("description"):
|
||||
issues.append({"code": "missing-description", "path": "SKILL.md"})
|
||||
|
||||
for path in sorted(item for item in root.rglob("*") if item.is_file()):
|
||||
relative = path.relative_to(root).as_posix()
|
||||
content = path.read_text(encoding="utf-8-sig", errors="replace")
|
||||
if "[TODO:" in content:
|
||||
issues.append({"code": "todo-placeholder", "path": relative})
|
||||
if SECRET_RE.search(content):
|
||||
issues.append({"code": "possible-secret", "path": relative})
|
||||
lowered = content.casefold()
|
||||
for term in forbidden:
|
||||
if term.casefold() in lowered:
|
||||
issues.append({"code": f"forbidden:{term}", "path": relative})
|
||||
if path.suffix.lower() == ".md":
|
||||
for link in LINK_RE.findall(content):
|
||||
target = link.split("#", 1)[0]
|
||||
if target and not (path.parent / target).resolve().is_file():
|
||||
issues.append({"code": "broken-link", "path": f"{relative}:{link}"})
|
||||
return {"skill": root.name, "issues": issues}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="检查 CraftKit Skill")
|
||||
parser.add_argument("skill", nargs="+", type=Path)
|
||||
parser.add_argument("--forbid", action="append", default=[])
|
||||
args = parser.parse_args()
|
||||
results = [check_skill(path.resolve(), args.forbid) for path in args.skill]
|
||||
print(json.dumps({"results": results}, ensure_ascii=False, indent=2))
|
||||
raise SystemExit(1 if any(item["issues"] for item in results) else 0)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,111 @@
|
||||
#!/usr/bin/env python3
|
||||
"""只读扫描来源目录中的 Skill,并输出稳定的 JSON 清单。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import subprocess
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def file_hash(path: Path) -> str:
|
||||
"""按二进制内容计算单文件 SHA-256。"""
|
||||
digest = hashlib.sha256()
|
||||
with path.open("rb") as stream:
|
||||
for block in iter(lambda: stream.read(65536), b""):
|
||||
digest.update(block)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def directory_hash(root: Path) -> tuple[str, int]:
|
||||
"""把相对路径和文件内容共同纳入哈希,确保结构变化也可被识别。"""
|
||||
digest = hashlib.sha256()
|
||||
files = sorted(path for path in root.rglob("*") if path.is_file())
|
||||
for path in files:
|
||||
relative = path.relative_to(root).as_posix()
|
||||
digest.update(relative.encode("utf-8"))
|
||||
digest.update(b"\0")
|
||||
digest.update(bytes.fromhex(file_hash(path)))
|
||||
return digest.hexdigest(), len(files)
|
||||
|
||||
|
||||
def read_skill_name(skill_md: Path) -> str:
|
||||
"""从 YAML frontmatter 中读取 name;解析失败时回退为目录名。"""
|
||||
for line in skill_md.read_text(encoding="utf-8-sig").splitlines():
|
||||
if line.startswith("name:"):
|
||||
return line.split(":", 1)[1].strip().strip("'\"") or skill_md.parent.name
|
||||
return skill_md.parent.name
|
||||
|
||||
|
||||
def git_commit(root: Path) -> str | None:
|
||||
"""尽力读取来源提交;非 Git 目录时返回空值,不阻塞文件扫描。"""
|
||||
result = subprocess.run(
|
||||
["git", "-c", f"safe.directory={root.as_posix()}", "-C", str(root), "rev-parse", "HEAD"],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
check=False,
|
||||
)
|
||||
return result.stdout.strip() if result.returncode == 0 else None
|
||||
|
||||
|
||||
def scan_source(source_id: str, root: Path) -> dict:
|
||||
"""扫描单个来源,结果仅包含相对路径和指纹,不包含本地绝对路径。"""
|
||||
if not root.is_dir():
|
||||
raise ValueError(f"来源目录不存在:{root}")
|
||||
skills = []
|
||||
for skill_md in sorted(root.rglob("SKILL.md")):
|
||||
skill_root = skill_md.parent
|
||||
digest, file_count = directory_hash(skill_root)
|
||||
children = sorted(
|
||||
child.name for child in skill_root.iterdir() if child.is_dir()
|
||||
)
|
||||
skills.append(
|
||||
{
|
||||
"name": read_skill_name(skill_md),
|
||||
"relativePath": skill_root.relative_to(root).as_posix(),
|
||||
"sha256": digest,
|
||||
"fileCount": file_count,
|
||||
"resources": children,
|
||||
}
|
||||
)
|
||||
return {
|
||||
"id": source_id,
|
||||
"commit": git_commit(root),
|
||||
"skillCount": len(skills),
|
||||
"skills": skills,
|
||||
}
|
||||
|
||||
|
||||
def parse_source(value: str) -> tuple[str, Path]:
|
||||
"""解析 id=path 参数,避免把本地路径写入输出文件。"""
|
||||
if "=" not in value:
|
||||
raise argparse.ArgumentTypeError("来源参数必须使用 id=path 格式")
|
||||
source_id, raw_path = value.split("=", 1)
|
||||
if not source_id.strip() or not raw_path.strip():
|
||||
raise argparse.ArgumentTypeError("来源标识和路径均不能为空")
|
||||
return source_id.strip(), Path(raw_path).expanduser().resolve()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="只读扫描 Skill 来源目录")
|
||||
parser.add_argument("--source", action="append", required=True, type=parse_source)
|
||||
parser.add_argument("--output", type=Path, help="可选 JSON 输出路径;省略时输出到终端")
|
||||
args = parser.parse_args()
|
||||
payload = {
|
||||
"schemaVersion": 1,
|
||||
"generatedAt": datetime.now(timezone.utc).isoformat(),
|
||||
"sources": [scan_source(source_id, root) for source_id, root in args.source],
|
||||
}
|
||||
content = json.dumps(payload, ensure_ascii=False, indent=2) + "\n"
|
||||
if args.output:
|
||||
args.output.parent.mkdir(parents=True, exist_ok=True)
|
||||
args.output.write_text(content, encoding="utf-8")
|
||||
else:
|
||||
print(content, end="")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,65 @@
|
||||
#!/usr/bin/env python3
|
||||
"""根据扫描报告更新迁移台账;默认预览,显式 --apply 才写入。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
from datetime import date
|
||||
from pathlib import Path
|
||||
|
||||
VALID_STATUS = {
|
||||
"pending", "specified", "rewriting", "review",
|
||||
"migrated", "excluded", "superseded",
|
||||
}
|
||||
|
||||
|
||||
def merge(lock: dict, report: dict) -> dict:
|
||||
"""保留人工维护字段,仅同步来源提交、数量和 Skill 指纹。"""
|
||||
lock.setdefault("schemaVersion", 1)
|
||||
lock["updatedAt"] = date.today().isoformat()
|
||||
lock.setdefault("sources", {})
|
||||
lock.setdefault("skills", {})
|
||||
for source in report.get("sources", []):
|
||||
source_id = source["id"]
|
||||
lock["sources"][source_id] = {
|
||||
"commit": source.get("commit"),
|
||||
"skillCount": source["skillCount"],
|
||||
}
|
||||
for skill in source["skills"]:
|
||||
# 发布台账不保存来源名称和路径;本地扫描报告负责提供可读映射。
|
||||
path_hash = hashlib.sha256(skill["relativePath"].encode("utf-8")).hexdigest()
|
||||
key = f"{source_id}:{path_hash[:16]}"
|
||||
current = lock["skills"].get(key, {})
|
||||
status = current.get("status", "pending")
|
||||
if status not in VALID_STATUS:
|
||||
raise ValueError(f"非法迁移状态:{key}={status}")
|
||||
lock["skills"][key] = {
|
||||
**current,
|
||||
"sourcePathHash": path_hash,
|
||||
"sourceSha256": skill["sha256"],
|
||||
"status": status,
|
||||
}
|
||||
return lock
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="更新 Skill 迁移台账")
|
||||
parser.add_argument("--lock", required=True, type=Path)
|
||||
parser.add_argument("--report", required=True, type=Path)
|
||||
parser.add_argument("--apply", action="store_true", help="确认写入台账")
|
||||
args = parser.parse_args()
|
||||
lock = json.loads(args.lock.read_text(encoding="utf-8"))
|
||||
report = json.loads(args.report.read_text(encoding="utf-8"))
|
||||
content = json.dumps(merge(lock, report), ensure_ascii=False, indent=2) + "\n"
|
||||
if args.apply:
|
||||
temporary = args.lock.with_suffix(args.lock.suffix + ".tmp")
|
||||
temporary.write_text(content, encoding="utf-8")
|
||||
temporary.replace(args.lock)
|
||||
else:
|
||||
print(content, end="")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user