Files
CraftKit/plugins/doc/skills/md-to-docx/scripts/convert.py
T

228 lines
9.2 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""将常见 Markdown 结构转换为可编辑 DOCX。"""
from __future__ import annotations
import argparse
import re
import sys
from dataclasses import dataclass, field
from pathlib import Path
@dataclass
class Result:
"""记录生成物与转换统计。"""
output: Path
headings: int = 0
paragraphs: int = 0
lists: int = 0
tables: int = 0
images: int = 0
warnings: list[str] = field(default_factory=list)
def cells(line: str) -> list[str]:
"""拆分基础表格行并还原转义竖线。"""
return [item.strip().replace(r"\|", "|") for item in re.split(r"(?<!\\)\|", line.strip().strip("|"))]
def separator(line: str) -> bool:
"""判断 Markdown 表格分隔行。"""
values = cells(line)
return bool(values) and all(re.fullmatch(r":?-{3,}:?", item) for item in values)
class Converter:
"""使用可预测的小型解析器转换常见 Markdown。"""
def __init__(self, source: Path, output: Path, template: Path | None) -> None:
# 第三方库在真正转换时才加载,使 --help、参数错误和能力探测不依赖本机预装包。
from docx import Document
self.document_type = Document
self.source = source
self.doc = Document(template) if template else Document()
self.result = Result(output)
def convert(self) -> Result:
"""按块转换、保存并重新打开产物完成结构校验。"""
lines = self.source.read_text(encoding="utf-8-sig").splitlines()
index = 0
while index < len(lines):
if lines[index].lstrip().startswith("```"):
index = self.add_code(lines, index)
elif index + 1 < len(lines) and "|" in lines[index] and separator(lines[index + 1]):
index = self.add_table(lines, index)
else:
self.add_line(lines[index])
index += 1
self.doc.save(self.result.output)
self.document_type(self.result.output)
return self.result
def add_code(self, lines: list[str], start: int) -> int:
"""读取围栏代码块;未闭合时输出其余内容并记录警告。"""
from docx.shared import Pt
index = start + 1
content: list[str] = []
while index < len(lines) and not lines[index].lstrip().startswith("```"):
content.append(lines[index])
index += 1
if index == len(lines):
self.result.warnings.append("代码围栏未闭合")
paragraph = self.doc.add_paragraph()
run = paragraph.add_run("\n".join(content))
run.font.name = "Consolas"
run.font.size = Pt(9)
self.result.paragraphs += 1
return min(index + 1, len(lines))
def add_table(self, lines: list[str], start: int) -> int:
"""将连续表格行转换为统一列数的 Word 表格。"""
rows = [cells(lines[start])]
index = start + 2
while index < len(lines) and lines[index].strip() and "|" in lines[index]:
rows.append(cells(lines[index]))
index += 1
width = max(map(len, rows))
table = self.doc.add_table(rows=len(rows), cols=width)
table.style = "Table Grid"
for row_no, values in enumerate(rows):
for col_no in range(width):
table.cell(row_no, col_no).text = (values[col_no] if col_no < len(values) else "").replace("<br>", "\n")
self.result.tables += 1
return index
def add_line(self, line: str) -> None:
"""识别标题、列表、引用、分隔线和普通段落。"""
from docx.shared import Inches
value = line.strip()
if not value:
return
heading = re.match(r"^(#{1,6})\s+(.+)$", value)
if heading:
self.add_inline(self.doc.add_heading(level=len(heading.group(1))), heading.group(2))
self.result.headings += 1
return
unordered = re.match(r"^(\s*)[-+*]\s+(.+)$", line)
ordered = re.match(r"^(\s*)\d+[.)]\s+(.+)$", line)
if unordered or ordered:
match = unordered or ordered
paragraph = self.doc.add_paragraph(style="List Bullet" if unordered else "List Number")
paragraph.paragraph_format.left_indent = Inches(min(len(match.group(1)) // 2, 4) * 0.25)
self.add_inline(paragraph, match.group(2))
self.result.lists += 1
return
if value.startswith(">"):
paragraph = self.doc.add_paragraph(value.lstrip("> "))
paragraph.paragraph_format.left_indent = Inches(0.3)
for run in paragraph.runs:
run.italic = True
self.result.paragraphs += 1
return
if re.fullmatch(r"(?:-{3,}|\*{3,}|_{3,})", value):
self.doc.add_paragraph("────────")
return
paragraph = self.doc.add_paragraph()
self.add_inline(paragraph, value)
self.result.paragraphs += 1
def add_inline(self, paragraph, text: str) -> None:
"""转换图片与基础强调;链接保留为可读地址文本。"""
pattern = re.compile(r"(!?\[[^\]]*\]\([^)]*\)|\*\*[^*]+\*\*|`[^`]+`|\*[^*]+\*)")
cursor = 0
for match in pattern.finditer(text):
paragraph.add_run(text[cursor:match.start()])
token = match.group(0)
image = re.fullmatch(r"!\[([^\]]*)\]\(([^)]+)\)", token)
link = re.fullmatch(r"\[([^\]]+)\]\(([^)]+)\)", token)
if image:
self.add_image(paragraph, image.group(1), image.group(2))
elif link:
paragraph.add_run(f"{link.group(1)}({link.group(2)})")
else:
run = paragraph.add_run(token[2:-2] if token.startswith("**") else token[1:-1])
run.bold = token.startswith("**")
run.italic = token.startswith("*") and not token.startswith("**")
if token.startswith("`"):
run.font.name = "Consolas"
cursor = match.end()
paragraph.add_run(text[cursor:])
def add_image(self, paragraph, alt: str, target: str) -> None:
"""只处理本地图片,防止转换过程产生隐式网络访问。"""
from docx.shared import Inches
if re.match(r"^[a-z][a-z0-9+.-]*://", target, re.I):
paragraph.add_run(f"[远程图片:{alt or target}]")
self.result.warnings.append(f"未下载远程图片:{target}")
return
path = (self.source.parent / target).resolve()
if not path.is_file():
paragraph.add_run(f"[缺失图片:{alt or target}]")
self.result.warnings.append(f"本地图片不存在:{target}")
return
try:
paragraph.add_run().add_picture(str(path), width=Inches(5.8))
self.result.images += 1
except (OSError, ValueError) as error:
paragraph.add_run(f"[无法嵌入图片:{alt or target}]")
self.result.warnings.append(f"图片无法嵌入:{target}({error})")
def main(argv: list[str] | None = None) -> int:
"""校验输入、覆盖权限与模板后执行转换。"""
parser = argparse.ArgumentParser(description="将 Markdown 转换为 DOCX")
parser.add_argument("input", type=Path)
parser.add_argument("--output", type=Path)
parser.add_argument("--template", type=Path)
parser.add_argument("--force", action="store_true")
args = parser.parse_args(argv)
source = args.input.resolve()
output = (args.output or source.with_suffix(".docx")).resolve()
template = args.template.resolve() if args.template else None
if not source.is_file() or source.suffix.lower() not in {".md", ".markdown"}:
print("错误:输入必须是存在的 .md 或 .markdown 文件", file=sys.stderr); return 2
if output.suffix.lower() != ".docx" or (output.exists() and not args.force):
print("错误:输出必须是可写的 .docx;覆盖需使用 --force", file=sys.stderr); return 1
if template and (not template.is_file() or template.suffix.lower() != ".docx"):
print("错误:模板必须是存在的 .docx 文件", file=sys.stderr); return 2
try:
converter = Converter(source, output, template)
except ModuleNotFoundError as error:
if error.name == "docx":
print(
"错误:缺少 python-docx。请使用 Codex 工作区依赖运行时,"
"或在已获授权的本地 Python 环境中安装 python-docx 后重试。",
file=sys.stderr,
)
return 3
raise
except (OSError, ValueError) as error:
print(f"错误:{error}", file=sys.stderr); return 1
output.parent.mkdir(parents=True, exist_ok=True)
try:
result = converter.convert()
except (OSError, ValueError) as error:
print(f"错误:{error}", file=sys.stderr); return 1
print(f"DOCX:{result.output}")
print(f"标题:{result.headings},段落:{result.paragraphs},列表:{result.lists},表格:{result.tables},图片:{result.images}")
for warning in dict.fromkeys(result.warnings): print(f"警告:{warning}")
return 0
if __name__ == "__main__":
raise SystemExit(main())