feat(doc): 新增文档转换与规则化归档 Skill
This commit is contained in:
@@ -0,0 +1,137 @@
|
||||
#!/usr/bin/env python3
|
||||
"""按 JSON 规则预演或执行保留来源的文档归档。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
def inside(path: Path, root: Path) -> bool:
|
||||
"""判断规范化路径是否位于指定根目录内。"""
|
||||
|
||||
try:
|
||||
path.relative_to(root); return True
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
|
||||
def strip_frontmatter(text: str) -> str:
|
||||
"""只移除文档开头完整闭合的 YAML Frontmatter。"""
|
||||
|
||||
return re.sub(r"\A---\s*\r?\n.*?\r?\n---\s*\r?\n?", "", text, count=1, flags=re.S)
|
||||
|
||||
|
||||
def extract_section(text: str, title: str) -> str | None:
|
||||
"""按标题文本抽取该 Markdown 章节及其子标题。"""
|
||||
|
||||
lines = text.splitlines(); start = level = None
|
||||
for index, line in enumerate(lines):
|
||||
match = re.match(r"^(#{1,6})\s+(.+?)\s*$", line)
|
||||
if match and match.group(2).strip() == title:
|
||||
start, level = index, len(match.group(1)); break
|
||||
if start is None or level is None:
|
||||
return None
|
||||
end = len(lines)
|
||||
for index in range(start + 1, len(lines)):
|
||||
match = re.match(r"^(#{1,6})\s+", lines[index])
|
||||
if match and len(match.group(1)) <= level:
|
||||
end = index; break
|
||||
return "\n".join(lines[start:end]).rstrip() + "\n"
|
||||
|
||||
|
||||
def target_for(template: str, source: Path, root: Path, archive_root: Path) -> Path:
|
||||
"""展开有限占位符,并拒绝绝对路径与目录逃逸。"""
|
||||
|
||||
relative = source.relative_to(root)
|
||||
values = {"name": source.name, "stem": source.stem, "suffix": source.suffix,
|
||||
"relative": relative.as_posix(),
|
||||
"parent": relative.parent.as_posix() if relative.parent != Path(".") else ""}
|
||||
try:
|
||||
candidate = Path(template.format(**values))
|
||||
except KeyError as error:
|
||||
raise ValueError(f"未知目标占位符:{error.args[0]}") from error
|
||||
if candidate.is_absolute():
|
||||
raise ValueError("目标模板不得生成绝对路径")
|
||||
resolved = (archive_root / candidate).resolve()
|
||||
if not inside(resolved, archive_root):
|
||||
raise ValueError("目标路径逃逸归档目录")
|
||||
return resolved
|
||||
|
||||
|
||||
def process(root: Path, config: dict, apply: bool) -> list[dict]:
|
||||
"""应用全部规则并返回可持久化的操作清单。"""
|
||||
|
||||
archive_value = config.get("archiveRoot", "docs/archive")
|
||||
archive_root = (root / archive_value).resolve()
|
||||
if Path(archive_value).is_absolute() or not inside(archive_root, root):
|
||||
raise ValueError("archiveRoot 必须是项目根内的相对路径")
|
||||
default_conflict = config.get("defaults", {}).get("onConflict", "skip")
|
||||
actions: list[dict] = []
|
||||
for rule in config.get("rules", []):
|
||||
conflict = rule.get("onConflict", default_conflict)
|
||||
if conflict not in {"skip", "overwrite", "append"}:
|
||||
raise ValueError(f"不支持的冲突策略:{conflict}")
|
||||
for source in sorted(root.glob(rule["match"])):
|
||||
# 归档目录自身不能再次成为来源,避免宽泛 glob 在重复执行时形成递归副本。
|
||||
if not source.is_file() or not inside(source.resolve(), root) or inside(source.resolve(), archive_root):
|
||||
continue
|
||||
target = target_for(rule["target"], source, root, archive_root)
|
||||
text = source.read_text(encoding="utf-8-sig")
|
||||
missing = [value for value in rule.get("requiredText", []) if value not in text]
|
||||
if missing:
|
||||
actions.append({"status": "failed", "source": str(source), "target": str(target), "reason": f"缺少必需文本:{missing}"}); continue
|
||||
if rule.get("stripFrontmatter"):
|
||||
text = strip_frontmatter(text)
|
||||
if rule.get("section"):
|
||||
section = extract_section(text, rule["section"])
|
||||
if section is None:
|
||||
actions.append({"status": "failed", "source": str(source), "target": str(target), "reason": "未找到指定章节"}); continue
|
||||
text = section
|
||||
if target.exists() and conflict == "skip":
|
||||
actions.append({"status": "skipped", "source": str(source), "target": str(target), "reason": "目标已存在"}); continue
|
||||
status = "planned"
|
||||
if apply:
|
||||
target.parent.mkdir(parents=True, exist_ok=True)
|
||||
if target.exists() and conflict == "append":
|
||||
prior = target.read_text(encoding="utf-8")
|
||||
target.write_text(prior.rstrip() + "\n\n" + text.lstrip(), encoding="utf-8")
|
||||
else:
|
||||
target.write_text(text, encoding="utf-8")
|
||||
status = "archived"
|
||||
actions.append({"status": status, "source": str(source), "target": str(target)})
|
||||
return actions
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
"""读取配置并输出 JSON 报告;默认不写归档文件。"""
|
||||
|
||||
parser = argparse.ArgumentParser(description="按规则预演或执行文档归档")
|
||||
parser.add_argument("--root", type=Path, required=True)
|
||||
parser.add_argument("--config", type=Path, required=True)
|
||||
parser.add_argument("--report", type=Path)
|
||||
parser.add_argument("--apply", action="store_true")
|
||||
args = parser.parse_args(argv)
|
||||
root, config_path = args.root.resolve(), args.config.resolve()
|
||||
if not root.is_dir() or not config_path.is_file():
|
||||
print("错误:项目根或配置文件不存在", file=sys.stderr); return 2
|
||||
try:
|
||||
config = json.loads(config_path.read_text(encoding="utf-8-sig"))
|
||||
if config.get("version") != 1 or not isinstance(config.get("rules"), list):
|
||||
raise ValueError("配置必须使用 version 1 并包含 rules 数组")
|
||||
actions = process(root, config, args.apply)
|
||||
encoded = json.dumps({"mode": "apply" if args.apply else "dry-run", "actions": actions}, ensure_ascii=False, indent=2)
|
||||
if args.report:
|
||||
report_path = args.report.resolve(); report_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
report_path.write_text(encoded + "\n", encoding="utf-8")
|
||||
print(encoded)
|
||||
return 1 if any(item["status"] == "failed" for item in actions) else 0
|
||||
except (OSError, ValueError, KeyError, json.JSONDecodeError) as error:
|
||||
print(f"错误:{error}", file=sys.stderr); return 1
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user