209 lines
8.4 KiB
Python
209 lines
8.4 KiB
Python
#!/usr/bin/env python3
|
||
"""将常见 Markdown 结构转换为可编辑 DOCX。"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import argparse
|
||
import re
|
||
import sys
|
||
from dataclasses import dataclass, field
|
||
from pathlib import Path
|
||
|
||
from docx import Document
|
||
from docx.shared import Inches, Pt
|
||
|
||
|
||
@dataclass
|
||
class Result:
|
||
"""记录生成物与转换统计。"""
|
||
|
||
output: Path
|
||
headings: int = 0
|
||
paragraphs: int = 0
|
||
lists: int = 0
|
||
tables: int = 0
|
||
images: int = 0
|
||
warnings: list[str] = field(default_factory=list)
|
||
|
||
|
||
def cells(line: str) -> list[str]:
|
||
"""拆分基础表格行并还原转义竖线。"""
|
||
|
||
return [item.strip().replace(r"\|", "|") for item in re.split(r"(?<!\\)\|", line.strip().strip("|"))]
|
||
|
||
|
||
def separator(line: str) -> bool:
|
||
"""判断 Markdown 表格分隔行。"""
|
||
|
||
values = cells(line)
|
||
return bool(values) and all(re.fullmatch(r":?-{3,}:?", item) for item in values)
|
||
|
||
|
||
class Converter:
|
||
"""使用可预测的小型解析器转换常见 Markdown。"""
|
||
|
||
def __init__(self, source: Path, output: Path, template: Path | None) -> None:
|
||
self.source = source
|
||
self.doc = Document(template) if template else Document()
|
||
self.result = Result(output)
|
||
|
||
def convert(self) -> Result:
|
||
"""按块转换、保存并重新打开产物完成结构校验。"""
|
||
|
||
lines = self.source.read_text(encoding="utf-8-sig").splitlines()
|
||
index = 0
|
||
while index < len(lines):
|
||
if lines[index].lstrip().startswith("```"):
|
||
index = self.add_code(lines, index)
|
||
elif index + 1 < len(lines) and "|" in lines[index] and separator(lines[index + 1]):
|
||
index = self.add_table(lines, index)
|
||
else:
|
||
self.add_line(lines[index])
|
||
index += 1
|
||
self.doc.save(self.result.output)
|
||
Document(self.result.output)
|
||
return self.result
|
||
|
||
def add_code(self, lines: list[str], start: int) -> int:
|
||
"""读取围栏代码块;未闭合时输出其余内容并记录警告。"""
|
||
|
||
index = start + 1
|
||
content: list[str] = []
|
||
while index < len(lines) and not lines[index].lstrip().startswith("```"):
|
||
content.append(lines[index])
|
||
index += 1
|
||
if index == len(lines):
|
||
self.result.warnings.append("代码围栏未闭合")
|
||
paragraph = self.doc.add_paragraph()
|
||
run = paragraph.add_run("\n".join(content))
|
||
run.font.name = "Consolas"
|
||
run.font.size = Pt(9)
|
||
self.result.paragraphs += 1
|
||
return min(index + 1, len(lines))
|
||
|
||
def add_table(self, lines: list[str], start: int) -> int:
|
||
"""将连续表格行转换为统一列数的 Word 表格。"""
|
||
|
||
rows = [cells(lines[start])]
|
||
index = start + 2
|
||
while index < len(lines) and lines[index].strip() and "|" in lines[index]:
|
||
rows.append(cells(lines[index]))
|
||
index += 1
|
||
width = max(map(len, rows))
|
||
table = self.doc.add_table(rows=len(rows), cols=width)
|
||
table.style = "Table Grid"
|
||
for row_no, values in enumerate(rows):
|
||
for col_no in range(width):
|
||
table.cell(row_no, col_no).text = (values[col_no] if col_no < len(values) else "").replace("<br>", "\n")
|
||
self.result.tables += 1
|
||
return index
|
||
|
||
def add_line(self, line: str) -> None:
|
||
"""识别标题、列表、引用、分隔线和普通段落。"""
|
||
|
||
value = line.strip()
|
||
if not value:
|
||
return
|
||
heading = re.match(r"^(#{1,6})\s+(.+)$", value)
|
||
if heading:
|
||
self.add_inline(self.doc.add_heading(level=len(heading.group(1))), heading.group(2))
|
||
self.result.headings += 1
|
||
return
|
||
unordered = re.match(r"^(\s*)[-+*]\s+(.+)$", line)
|
||
ordered = re.match(r"^(\s*)\d+[.)]\s+(.+)$", line)
|
||
if unordered or ordered:
|
||
match = unordered or ordered
|
||
paragraph = self.doc.add_paragraph(style="List Bullet" if unordered else "List Number")
|
||
paragraph.paragraph_format.left_indent = Inches(min(len(match.group(1)) // 2, 4) * 0.25)
|
||
self.add_inline(paragraph, match.group(2))
|
||
self.result.lists += 1
|
||
return
|
||
if value.startswith(">"):
|
||
paragraph = self.doc.add_paragraph(value.lstrip("> "))
|
||
paragraph.paragraph_format.left_indent = Inches(0.3)
|
||
for run in paragraph.runs:
|
||
run.italic = True
|
||
self.result.paragraphs += 1
|
||
return
|
||
if re.fullmatch(r"(?:-{3,}|\*{3,}|_{3,})", value):
|
||
self.doc.add_paragraph("────────")
|
||
return
|
||
paragraph = self.doc.add_paragraph()
|
||
self.add_inline(paragraph, value)
|
||
self.result.paragraphs += 1
|
||
|
||
def add_inline(self, paragraph, text: str) -> None:
|
||
"""转换图片与基础强调;链接保留为可读地址文本。"""
|
||
|
||
pattern = re.compile(r"(!?\[[^\]]*\]\([^)]*\)|\*\*[^*]+\*\*|`[^`]+`|\*[^*]+\*)")
|
||
cursor = 0
|
||
for match in pattern.finditer(text):
|
||
paragraph.add_run(text[cursor:match.start()])
|
||
token = match.group(0)
|
||
image = re.fullmatch(r"!\[([^\]]*)\]\(([^)]+)\)", token)
|
||
link = re.fullmatch(r"\[([^\]]+)\]\(([^)]+)\)", token)
|
||
if image:
|
||
self.add_image(paragraph, image.group(1), image.group(2))
|
||
elif link:
|
||
paragraph.add_run(f"{link.group(1)}({link.group(2)})")
|
||
else:
|
||
run = paragraph.add_run(token[2:-2] if token.startswith("**") else token[1:-1])
|
||
run.bold = token.startswith("**")
|
||
run.italic = token.startswith("*") and not token.startswith("**")
|
||
if token.startswith("`"):
|
||
run.font.name = "Consolas"
|
||
cursor = match.end()
|
||
paragraph.add_run(text[cursor:])
|
||
|
||
def add_image(self, paragraph, alt: str, target: str) -> None:
|
||
"""只处理本地图片,防止转换过程产生隐式网络访问。"""
|
||
|
||
if re.match(r"^[a-z][a-z0-9+.-]*://", target, re.I):
|
||
paragraph.add_run(f"[远程图片:{alt or target}]")
|
||
self.result.warnings.append(f"未下载远程图片:{target}")
|
||
return
|
||
path = (self.source.parent / target).resolve()
|
||
if not path.is_file():
|
||
paragraph.add_run(f"[缺失图片:{alt or target}]")
|
||
self.result.warnings.append(f"本地图片不存在:{target}")
|
||
return
|
||
try:
|
||
paragraph.add_run().add_picture(str(path), width=Inches(5.8))
|
||
self.result.images += 1
|
||
except (OSError, ValueError) as error:
|
||
paragraph.add_run(f"[无法嵌入图片:{alt or target}]")
|
||
self.result.warnings.append(f"图片无法嵌入:{target}({error})")
|
||
|
||
|
||
def main(argv: list[str] | None = None) -> int:
|
||
"""校验输入、覆盖权限与模板后执行转换。"""
|
||
|
||
parser = argparse.ArgumentParser(description="将 Markdown 转换为 DOCX")
|
||
parser.add_argument("input", type=Path)
|
||
parser.add_argument("--output", type=Path)
|
||
parser.add_argument("--template", type=Path)
|
||
parser.add_argument("--force", action="store_true")
|
||
args = parser.parse_args(argv)
|
||
source = args.input.resolve()
|
||
output = (args.output or source.with_suffix(".docx")).resolve()
|
||
template = args.template.resolve() if args.template else None
|
||
if not source.is_file() or source.suffix.lower() not in {".md", ".markdown"}:
|
||
print("错误:输入必须是存在的 .md 或 .markdown 文件", file=sys.stderr); return 2
|
||
if output.suffix.lower() != ".docx" or (output.exists() and not args.force):
|
||
print("错误:输出必须是可写的 .docx;覆盖需使用 --force", file=sys.stderr); return 1
|
||
if template and (not template.is_file() or template.suffix.lower() != ".docx"):
|
||
print("错误:模板必须是存在的 .docx 文件", file=sys.stderr); return 2
|
||
output.parent.mkdir(parents=True, exist_ok=True)
|
||
try:
|
||
result = Converter(source, output, template).convert()
|
||
except (OSError, ValueError) as error:
|
||
print(f"错误:{error}", file=sys.stderr); return 1
|
||
print(f"DOCX:{result.output}")
|
||
print(f"标题:{result.headings},段落:{result.paragraphs},列表:{result.lists},表格:{result.tables},图片:{result.images}")
|
||
for warning in dict.fromkeys(result.warnings): print(f"警告:{warning}")
|
||
return 0
|
||
|
||
|
||
if __name__ == "__main__":
|
||
raise SystemExit(main())
|