feat(doc): 新增文档转换与规则化归档 Skill
This commit is contained in:
@@ -0,0 +1,208 @@
|
||||
#!/usr/bin/env python3
|
||||
"""将常见 Markdown 结构转换为可编辑 DOCX。"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
from docx import Document
|
||||
from docx.shared import Inches, Pt
|
||||
|
||||
|
||||
@dataclass
|
||||
class Result:
|
||||
"""记录生成物与转换统计。"""
|
||||
|
||||
output: Path
|
||||
headings: int = 0
|
||||
paragraphs: int = 0
|
||||
lists: int = 0
|
||||
tables: int = 0
|
||||
images: int = 0
|
||||
warnings: list[str] = field(default_factory=list)
|
||||
|
||||
|
||||
def cells(line: str) -> list[str]:
|
||||
"""拆分基础表格行并还原转义竖线。"""
|
||||
|
||||
return [item.strip().replace(r"\|", "|") for item in re.split(r"(?<!\\)\|", line.strip().strip("|"))]
|
||||
|
||||
|
||||
def separator(line: str) -> bool:
|
||||
"""判断 Markdown 表格分隔行。"""
|
||||
|
||||
values = cells(line)
|
||||
return bool(values) and all(re.fullmatch(r":?-{3,}:?", item) for item in values)
|
||||
|
||||
|
||||
class Converter:
|
||||
"""使用可预测的小型解析器转换常见 Markdown。"""
|
||||
|
||||
def __init__(self, source: Path, output: Path, template: Path | None) -> None:
|
||||
self.source = source
|
||||
self.doc = Document(template) if template else Document()
|
||||
self.result = Result(output)
|
||||
|
||||
def convert(self) -> Result:
|
||||
"""按块转换、保存并重新打开产物完成结构校验。"""
|
||||
|
||||
lines = self.source.read_text(encoding="utf-8-sig").splitlines()
|
||||
index = 0
|
||||
while index < len(lines):
|
||||
if lines[index].lstrip().startswith("```"):
|
||||
index = self.add_code(lines, index)
|
||||
elif index + 1 < len(lines) and "|" in lines[index] and separator(lines[index + 1]):
|
||||
index = self.add_table(lines, index)
|
||||
else:
|
||||
self.add_line(lines[index])
|
||||
index += 1
|
||||
self.doc.save(self.result.output)
|
||||
Document(self.result.output)
|
||||
return self.result
|
||||
|
||||
def add_code(self, lines: list[str], start: int) -> int:
|
||||
"""读取围栏代码块;未闭合时输出其余内容并记录警告。"""
|
||||
|
||||
index = start + 1
|
||||
content: list[str] = []
|
||||
while index < len(lines) and not lines[index].lstrip().startswith("```"):
|
||||
content.append(lines[index])
|
||||
index += 1
|
||||
if index == len(lines):
|
||||
self.result.warnings.append("代码围栏未闭合")
|
||||
paragraph = self.doc.add_paragraph()
|
||||
run = paragraph.add_run("\n".join(content))
|
||||
run.font.name = "Consolas"
|
||||
run.font.size = Pt(9)
|
||||
self.result.paragraphs += 1
|
||||
return min(index + 1, len(lines))
|
||||
|
||||
def add_table(self, lines: list[str], start: int) -> int:
|
||||
"""将连续表格行转换为统一列数的 Word 表格。"""
|
||||
|
||||
rows = [cells(lines[start])]
|
||||
index = start + 2
|
||||
while index < len(lines) and lines[index].strip() and "|" in lines[index]:
|
||||
rows.append(cells(lines[index]))
|
||||
index += 1
|
||||
width = max(map(len, rows))
|
||||
table = self.doc.add_table(rows=len(rows), cols=width)
|
||||
table.style = "Table Grid"
|
||||
for row_no, values in enumerate(rows):
|
||||
for col_no in range(width):
|
||||
table.cell(row_no, col_no).text = (values[col_no] if col_no < len(values) else "").replace("<br>", "\n")
|
||||
self.result.tables += 1
|
||||
return index
|
||||
|
||||
def add_line(self, line: str) -> None:
|
||||
"""识别标题、列表、引用、分隔线和普通段落。"""
|
||||
|
||||
value = line.strip()
|
||||
if not value:
|
||||
return
|
||||
heading = re.match(r"^(#{1,6})\s+(.+)$", value)
|
||||
if heading:
|
||||
self.add_inline(self.doc.add_heading(level=len(heading.group(1))), heading.group(2))
|
||||
self.result.headings += 1
|
||||
return
|
||||
unordered = re.match(r"^(\s*)[-+*]\s+(.+)$", line)
|
||||
ordered = re.match(r"^(\s*)\d+[.)]\s+(.+)$", line)
|
||||
if unordered or ordered:
|
||||
match = unordered or ordered
|
||||
paragraph = self.doc.add_paragraph(style="List Bullet" if unordered else "List Number")
|
||||
paragraph.paragraph_format.left_indent = Inches(min(len(match.group(1)) // 2, 4) * 0.25)
|
||||
self.add_inline(paragraph, match.group(2))
|
||||
self.result.lists += 1
|
||||
return
|
||||
if value.startswith(">"):
|
||||
paragraph = self.doc.add_paragraph(value.lstrip("> "))
|
||||
paragraph.paragraph_format.left_indent = Inches(0.3)
|
||||
for run in paragraph.runs:
|
||||
run.italic = True
|
||||
self.result.paragraphs += 1
|
||||
return
|
||||
if re.fullmatch(r"(?:-{3,}|\*{3,}|_{3,})", value):
|
||||
self.doc.add_paragraph("────────")
|
||||
return
|
||||
paragraph = self.doc.add_paragraph()
|
||||
self.add_inline(paragraph, value)
|
||||
self.result.paragraphs += 1
|
||||
|
||||
def add_inline(self, paragraph, text: str) -> None:
|
||||
"""转换图片与基础强调;链接保留为可读地址文本。"""
|
||||
|
||||
pattern = re.compile(r"(!?\[[^\]]*\]\([^)]*\)|\*\*[^*]+\*\*|`[^`]+`|\*[^*]+\*)")
|
||||
cursor = 0
|
||||
for match in pattern.finditer(text):
|
||||
paragraph.add_run(text[cursor:match.start()])
|
||||
token = match.group(0)
|
||||
image = re.fullmatch(r"!\[([^\]]*)\]\(([^)]+)\)", token)
|
||||
link = re.fullmatch(r"\[([^\]]+)\]\(([^)]+)\)", token)
|
||||
if image:
|
||||
self.add_image(paragraph, image.group(1), image.group(2))
|
||||
elif link:
|
||||
paragraph.add_run(f"{link.group(1)}({link.group(2)})")
|
||||
else:
|
||||
run = paragraph.add_run(token[2:-2] if token.startswith("**") else token[1:-1])
|
||||
run.bold = token.startswith("**")
|
||||
run.italic = token.startswith("*") and not token.startswith("**")
|
||||
if token.startswith("`"):
|
||||
run.font.name = "Consolas"
|
||||
cursor = match.end()
|
||||
paragraph.add_run(text[cursor:])
|
||||
|
||||
def add_image(self, paragraph, alt: str, target: str) -> None:
|
||||
"""只处理本地图片,防止转换过程产生隐式网络访问。"""
|
||||
|
||||
if re.match(r"^[a-z][a-z0-9+.-]*://", target, re.I):
|
||||
paragraph.add_run(f"[远程图片:{alt or target}]")
|
||||
self.result.warnings.append(f"未下载远程图片:{target}")
|
||||
return
|
||||
path = (self.source.parent / target).resolve()
|
||||
if not path.is_file():
|
||||
paragraph.add_run(f"[缺失图片:{alt or target}]")
|
||||
self.result.warnings.append(f"本地图片不存在:{target}")
|
||||
return
|
||||
try:
|
||||
paragraph.add_run().add_picture(str(path), width=Inches(5.8))
|
||||
self.result.images += 1
|
||||
except (OSError, ValueError) as error:
|
||||
paragraph.add_run(f"[无法嵌入图片:{alt or target}]")
|
||||
self.result.warnings.append(f"图片无法嵌入:{target}({error})")
|
||||
|
||||
|
||||
def main(argv: list[str] | None = None) -> int:
|
||||
"""校验输入、覆盖权限与模板后执行转换。"""
|
||||
|
||||
parser = argparse.ArgumentParser(description="将 Markdown 转换为 DOCX")
|
||||
parser.add_argument("input", type=Path)
|
||||
parser.add_argument("--output", type=Path)
|
||||
parser.add_argument("--template", type=Path)
|
||||
parser.add_argument("--force", action="store_true")
|
||||
args = parser.parse_args(argv)
|
||||
source = args.input.resolve()
|
||||
output = (args.output or source.with_suffix(".docx")).resolve()
|
||||
template = args.template.resolve() if args.template else None
|
||||
if not source.is_file() or source.suffix.lower() not in {".md", ".markdown"}:
|
||||
print("错误:输入必须是存在的 .md 或 .markdown 文件", file=sys.stderr); return 2
|
||||
if output.suffix.lower() != ".docx" or (output.exists() and not args.force):
|
||||
print("错误:输出必须是可写的 .docx;覆盖需使用 --force", file=sys.stderr); return 1
|
||||
if template and (not template.is_file() or template.suffix.lower() != ".docx"):
|
||||
print("错误:模板必须是存在的 .docx 文件", file=sys.stderr); return 2
|
||||
output.parent.mkdir(parents=True, exist_ok=True)
|
||||
try:
|
||||
result = Converter(source, output, template).convert()
|
||||
except (OSError, ValueError) as error:
|
||||
print(f"错误:{error}", file=sys.stderr); return 1
|
||||
print(f"DOCX:{result.output}")
|
||||
print(f"标题:{result.headings},段落:{result.paragraphs},列表:{result.lists},表格:{result.tables},图片:{result.images}")
|
||||
for warning in dict.fromkeys(result.warnings): print(f"警告:{warning}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user