feat(doc): 新增文档转换与规则化归档 Skill

This commit is contained in:
zhiye.sun
2026-08-25 15:32:12 +08:00
parent dbc2f65154
commit d255bdd524
20 changed files with 735 additions and 20 deletions
+39
View File
@@ -0,0 +1,39 @@
---
name: archive
description: 按显式配置将项目文档复制、重命名、清理 Frontmatter 或抽取 Markdown 章节到归档目录,并生成操作报告。适用于需要可重复文档归档规则的项目;删除来源、猜测业务目录或隐式查询内部系统不应触发本 Skill。
---
# 文档归档
使用 `scripts/archive.py` 执行中性的规则化归档。规则由当前项目维护,不携带固定组织目录、业务名称、数据库表或包结构。
## 两阶段执行
1. 读取 `.craftkit/project.json` 的 `documents` 配置,或使用用户指定的规则文件。
2. 检查每条规则的来源、目标模板、转换和校验条件。
3. 不带 `--apply` 执行预演,向用户展示复制、跳过、冲突和失败项。
4. 用户确认后使用相同参数加 `--apply` 执行。
5. 检查 JSON 报告和目标文件,并确认来源文件仍然存在。
```text
python scripts/archive.py --root <project> --config <rules.json> [--report <report.json>]
python scripts/archive.py --root <project> --config <rules.json> --apply [--report <report.json>]
```
## 中性平替原则
- 固定公司目录改为 `archiveRoot` 与规则级 `target` 模板。
- 固定业务文件名改为 `{name}`、`{stem}`、`{suffix}`、`{relative}` 占位符。
- 专有文档拆分逻辑改为通用 Markdown 标题章节抽取。
- 内部数据库或服务校验改为显式 `requiredText` 内容校验;需要外部事实时由用户先提供结果,不隐式连接系统。
- 来源专属元数据清理改为可选 `stripFrontmatter`,不会默认删除内容。
## 安全边界
- 默认只预演;`--apply` 才写入。
- 永不删除或移动来源文件。
- 来源必须位于项目根内,目标必须位于归档根内;拒绝绝对路径和 `..` 逃逸。
- 冲突策略仅允许 `skip`、`overwrite`、`append`,默认 `skip`。覆盖或追加必须在用户确认的配置中明确出现。
- 规则不执行 Shell、SQL、模板代码或网络请求。
完整配置见 `references/config.md`。
@@ -0,0 +1,4 @@
interface:
display_name: "Archive Documents"
short_description: "按项目规则安全预演并归档文档"
default_prompt: "使用 $archive 根据项目规则预演文档归档,确认后再执行写入。"
@@ -0,0 +1,32 @@
# 归档配置
```json
{
"version": 1,
"archiveRoot": "docs/archive",
"defaults": { "onConflict": "skip" },
"rules": [
{
"match": "design/**/*.md",
"target": "design/{relative}",
"stripFrontmatter": true,
"requiredText": ["#"]
},
{
"match": "specs/*.md",
"target": "reference/{stem}-api.md",
"section": "API"
}
]
}
```
- `archiveRoot`:相对项目根的归档目录。
- `match`:相对项目根的 glob,只处理普通文件。
- `target`:可用 `{name}`、`{stem}`、`{suffix}`、`{relative}`、`{parent}`。
- `section`:可选 Markdown 标题文本,输出该标题及下属内容。
- `stripFrontmatter`:仅移除开头完整闭合的 Frontmatter。
- `requiredText`:归档前必须出现的文字,用于替代隐式外部校验。
- `onConflict`:`skip`、`overwrite` 或 `append`。
目标不得是绝对路径或逃逸归档根。`append` 仅适合文本文件。
@@ -0,0 +1,137 @@
#!/usr/bin/env python3
"""按 JSON 规则预演或执行保留来源的文档归档。"""
from __future__ import annotations
import argparse
import json
import re
import sys
from pathlib import Path
def inside(path: Path, root: Path) -> bool:
"""判断规范化路径是否位于指定根目录内。"""
try:
path.relative_to(root); return True
except ValueError:
return False
def strip_frontmatter(text: str) -> str:
"""只移除文档开头完整闭合的 YAML Frontmatter。"""
return re.sub(r"\A---\s*\r?\n.*?\r?\n---\s*\r?\n?", "", text, count=1, flags=re.S)
def extract_section(text: str, title: str) -> str | None:
"""按标题文本抽取该 Markdown 章节及其子标题。"""
lines = text.splitlines(); start = level = None
for index, line in enumerate(lines):
match = re.match(r"^(#{1,6})\s+(.+?)\s*$", line)
if match and match.group(2).strip() == title:
start, level = index, len(match.group(1)); break
if start is None or level is None:
return None
end = len(lines)
for index in range(start + 1, len(lines)):
match = re.match(r"^(#{1,6})\s+", lines[index])
if match and len(match.group(1)) <= level:
end = index; break
return "\n".join(lines[start:end]).rstrip() + "\n"
def target_for(template: str, source: Path, root: Path, archive_root: Path) -> Path:
"""展开有限占位符,并拒绝绝对路径与目录逃逸。"""
relative = source.relative_to(root)
values = {"name": source.name, "stem": source.stem, "suffix": source.suffix,
"relative": relative.as_posix(),
"parent": relative.parent.as_posix() if relative.parent != Path(".") else ""}
try:
candidate = Path(template.format(**values))
except KeyError as error:
raise ValueError(f"未知目标占位符:{error.args[0]}") from error
if candidate.is_absolute():
raise ValueError("目标模板不得生成绝对路径")
resolved = (archive_root / candidate).resolve()
if not inside(resolved, archive_root):
raise ValueError("目标路径逃逸归档目录")
return resolved
def process(root: Path, config: dict, apply: bool) -> list[dict]:
"""应用全部规则并返回可持久化的操作清单。"""
archive_value = config.get("archiveRoot", "docs/archive")
archive_root = (root / archive_value).resolve()
if Path(archive_value).is_absolute() or not inside(archive_root, root):
raise ValueError("archiveRoot 必须是项目根内的相对路径")
default_conflict = config.get("defaults", {}).get("onConflict", "skip")
actions: list[dict] = []
for rule in config.get("rules", []):
conflict = rule.get("onConflict", default_conflict)
if conflict not in {"skip", "overwrite", "append"}:
raise ValueError(f"不支持的冲突策略:{conflict}")
for source in sorted(root.glob(rule["match"])):
# 归档目录自身不能再次成为来源,避免宽泛 glob 在重复执行时形成递归副本。
if not source.is_file() or not inside(source.resolve(), root) or inside(source.resolve(), archive_root):
continue
target = target_for(rule["target"], source, root, archive_root)
text = source.read_text(encoding="utf-8-sig")
missing = [value for value in rule.get("requiredText", []) if value not in text]
if missing:
actions.append({"status": "failed", "source": str(source), "target": str(target), "reason": f"缺少必需文本:{missing}"}); continue
if rule.get("stripFrontmatter"):
text = strip_frontmatter(text)
if rule.get("section"):
section = extract_section(text, rule["section"])
if section is None:
actions.append({"status": "failed", "source": str(source), "target": str(target), "reason": "未找到指定章节"}); continue
text = section
if target.exists() and conflict == "skip":
actions.append({"status": "skipped", "source": str(source), "target": str(target), "reason": "目标已存在"}); continue
status = "planned"
if apply:
target.parent.mkdir(parents=True, exist_ok=True)
if target.exists() and conflict == "append":
prior = target.read_text(encoding="utf-8")
target.write_text(prior.rstrip() + "\n\n" + text.lstrip(), encoding="utf-8")
else:
target.write_text(text, encoding="utf-8")
status = "archived"
actions.append({"status": status, "source": str(source), "target": str(target)})
return actions
def main(argv: list[str] | None = None) -> int:
"""读取配置并输出 JSON 报告;默认不写归档文件。"""
parser = argparse.ArgumentParser(description="按规则预演或执行文档归档")
parser.add_argument("--root", type=Path, required=True)
parser.add_argument("--config", type=Path, required=True)
parser.add_argument("--report", type=Path)
parser.add_argument("--apply", action="store_true")
args = parser.parse_args(argv)
root, config_path = args.root.resolve(), args.config.resolve()
if not root.is_dir() or not config_path.is_file():
print("错误:项目根或配置文件不存在", file=sys.stderr); return 2
try:
config = json.loads(config_path.read_text(encoding="utf-8-sig"))
if config.get("version") != 1 or not isinstance(config.get("rules"), list):
raise ValueError("配置必须使用 version 1 并包含 rules 数组")
actions = process(root, config, args.apply)
encoded = json.dumps({"mode": "apply" if args.apply else "dry-run", "actions": actions}, ensure_ascii=False, indent=2)
if args.report:
report_path = args.report.resolve(); report_path.parent.mkdir(parents=True, exist_ok=True)
report_path.write_text(encoded + "\n", encoding="utf-8")
print(encoded)
return 1 if any(item["status"] == "failed" for item in actions) else 0
except (OSError, ValueError, KeyError, json.JSONDecodeError) as error:
print(f"错误:{error}", file=sys.stderr); return 1
if __name__ == "__main__":
raise SystemExit(main())