@@ -0,0 +1,208 @@
#!/usr/bin/env python3
""" 将常见 Markdown 结构转换为可编辑 DOCX。 """
from __future__ import annotations
import argparse
import re
import sys
from dataclasses import dataclass , field
from pathlib import Path
from docx import Document
from docx . shared import Inches , Pt
@dataclass
class Result :
""" 记录生成物与转换统计。 """
output : Path
headings : int = 0
paragraphs : int = 0
lists : int = 0
tables : int = 0
images : int = 0
warnings : list [ str ] = field ( default_factory = list )
def cells ( line : str ) - > list [ str ] :
""" 拆分基础表格行并还原转义竖线。 """
return [ item . strip ( ) . replace ( r " \ | " , " | " ) for item in re . split ( r " (?<! \\ ) \ | " , line . strip ( ) . strip ( " | " ) ) ]
def separator ( line : str ) - > bool :
""" 判断 Markdown 表格分隔行。 """
values = cells ( line )
return bool ( values ) and all ( re . fullmatch ( r " :?- { 3,}:? " , item ) for item in values )
class Converter :
""" 使用可预测的小型解析器转换常见 Markdown。 """
def __init__ ( self , source : Path , output : Path , template : Path | None ) - > None :
self . source = source
self . doc = Document ( template ) if template else Document ( )
self . result = Result ( output )
def convert ( self ) - > Result :
""" 按块转换、保存并重新打开产物完成结构校验。 """
lines = self . source . read_text ( encoding = " utf-8-sig " ) . splitlines ( )
index = 0
while index < len ( lines ) :
if lines [ index ] . lstrip ( ) . startswith ( " ``` " ) :
index = self . add_code ( lines , index )
elif index + 1 < len ( lines ) and " | " in lines [ index ] and separator ( lines [ index + 1 ] ) :
index = self . add_table ( lines , index )
else :
self . add_line ( lines [ index ] )
index + = 1
self . doc . save ( self . result . output )
Document ( self . result . output )
return self . result
def add_code ( self , lines : list [ str ] , start : int ) - > int :
""" 读取围栏代码块;未闭合时输出其余内容并记录警告。 """
index = start + 1
content : list [ str ] = [ ]
while index < len ( lines ) and not lines [ index ] . lstrip ( ) . startswith ( " ``` " ) :
content . append ( lines [ index ] )
index + = 1
if index == len ( lines ) :
self . result . warnings . append ( " 代码围栏未闭合 " )
paragraph = self . doc . add_paragraph ( )
run = paragraph . add_run ( " \n " . join ( content ) )
run . font . name = " Consolas "
run . font . size = Pt ( 9 )
self . result . paragraphs + = 1
return min ( index + 1 , len ( lines ) )
def add_table ( self , lines : list [ str ] , start : int ) - > int :
""" 将连续表格行转换为统一列数的 Word 表格。 """
rows = [ cells ( lines [ start ] ) ]
index = start + 2
while index < len ( lines ) and lines [ index ] . strip ( ) and " | " in lines [ index ] :
rows . append ( cells ( lines [ index ] ) )
index + = 1
width = max ( map ( len , rows ) )
table = self . doc . add_table ( rows = len ( rows ) , cols = width )
table . style = " Table Grid "
for row_no , values in enumerate ( rows ) :
for col_no in range ( width ) :
table . cell ( row_no , col_no ) . text = ( values [ col_no ] if col_no < len ( values ) else " " ) . replace ( " <br> " , " \n " )
self . result . tables + = 1
return index
def add_line ( self , line : str ) - > None :
""" 识别标题、列表、引用、分隔线和普通段落。 """
value = line . strip ( )
if not value :
return
heading = re . match ( r " ^(# { 1,6}) \ s+(.+)$ " , value )
if heading :
self . add_inline ( self . doc . add_heading ( level = len ( heading . group ( 1 ) ) ) , heading . group ( 2 ) )
self . result . headings + = 1
return
unordered = re . match ( r " ^( \ s*)[-+*] \ s+(.+)$ " , line )
ordered = re . match ( r " ^( \ s*) \ d+[.)] \ s+(.+)$ " , line )
if unordered or ordered :
match = unordered or ordered
paragraph = self . doc . add_paragraph ( style = " List Bullet " if unordered else " List Number " )
paragraph . paragraph_format . left_indent = Inches ( min ( len ( match . group ( 1 ) ) / / 2 , 4 ) * 0.25 )
self . add_inline ( paragraph , match . group ( 2 ) )
self . result . lists + = 1
return
if value . startswith ( " > " ) :
paragraph = self . doc . add_paragraph ( value . lstrip ( " > " ) )
paragraph . paragraph_format . left_indent = Inches ( 0.3 )
for run in paragraph . runs :
run . italic = True
self . result . paragraphs + = 1
return
if re . fullmatch ( r " (?:- { 3,}| \ * { 3,}|_ { 3,}) " , value ) :
self . doc . add_paragraph ( " ──────── " )
return
paragraph = self . doc . add_paragraph ( )
self . add_inline ( paragraph , value )
self . result . paragraphs + = 1
def add_inline ( self , paragraph , text : str ) - > None :
""" 转换图片与基础强调;链接保留为可读地址文本。 """
pattern = re . compile ( r " (!? \ [[^ \ ]]* \ ] \ ([^)]* \ )| \ * \ *[^*]+ \ * \ *|`[^`]+`| \ *[^*]+ \ *) " )
cursor = 0
for match in pattern . finditer ( text ) :
paragraph . add_run ( text [ cursor : match . start ( ) ] )
token = match . group ( 0 )
image = re . fullmatch ( r " ! \ [([^ \ ]]*) \ ] \ (([^)]+) \ ) " , token )
link = re . fullmatch ( r " \ [([^ \ ]]+) \ ] \ (([^)]+) \ ) " , token )
if image :
self . add_image ( paragraph , image . group ( 1 ) , image . group ( 2 ) )
elif link :
paragraph . add_run ( f " { link . group ( 1 ) } ( { link . group ( 2 ) } ) " )
else :
run = paragraph . add_run ( token [ 2 : - 2 ] if token . startswith ( " ** " ) else token [ 1 : - 1 ] )
run . bold = token . startswith ( " ** " )
run . italic = token . startswith ( " * " ) and not token . startswith ( " ** " )
if token . startswith ( " ` " ) :
run . font . name = " Consolas "
cursor = match . end ( )
paragraph . add_run ( text [ cursor : ] )
def add_image ( self , paragraph , alt : str , target : str ) - > None :
""" 只处理本地图片,防止转换过程产生隐式网络访问。 """
if re . match ( r " ^[a-z][a-z0-9+.-]*:// " , target , re . I ) :
paragraph . add_run ( f " [远程图片: { alt or target } ] " )
self . result . warnings . append ( f " 未下载远程图片: { target } " )
return
path = ( self . source . parent / target ) . resolve ( )
if not path . is_file ( ) :
paragraph . add_run ( f " [缺失图片: { alt or target } ] " )
self . result . warnings . append ( f " 本地图片不存在: { target } " )
return
try :
paragraph . add_run ( ) . add_picture ( str ( path ) , width = Inches ( 5.8 ) )
self . result . images + = 1
except ( OSError , ValueError ) as error :
paragraph . add_run ( f " [无法嵌入图片: { alt or target } ] " )
self . result . warnings . append ( f " 图片无法嵌入: { target } ( { error } ) " )
def main ( argv : list [ str ] | None = None ) - > int :
""" 校验输入、覆盖权限与模板后执行转换。 """
parser = argparse . ArgumentParser ( description = " 将 Markdown 转换为 DOCX " )
parser . add_argument ( " input " , type = Path )
parser . add_argument ( " --output " , type = Path )
parser . add_argument ( " --template " , type = Path )
parser . add_argument ( " --force " , action = " store_true " )
args = parser . parse_args ( argv )
source = args . input . resolve ( )
output = ( args . output or source . with_suffix ( " .docx " ) ) . resolve ( )
template = args . template . resolve ( ) if args . template else None
if not source . is_file ( ) or source . suffix . lower ( ) not in { " .md " , " .markdown " } :
print ( " 错误:输入必须是存在的 .md 或 .markdown 文件 " , file = sys . stderr ) ; return 2
if output . suffix . lower ( ) != " .docx " or ( output . exists ( ) and not args . force ) :
print ( " 错误:输出必须是可写的 .docx;覆盖需使用 --force " , file = sys . stderr ) ; return 1
if template and ( not template . is_file ( ) or template . suffix . lower ( ) != " .docx " ) :
print ( " 错误:模板必须是存在的 .docx 文件 " , file = sys . stderr ) ; return 2
output . parent . mkdir ( parents = True , exist_ok = True )
try :
result = Converter ( source , output , template ) . convert ( )
except ( OSError , ValueError ) as error :
print ( f " 错误: { error } " , file = sys . stderr ) ; return 1
print ( f " DOCX: { result . output } " )
print ( f " 标题: { result . headings } ,段落: { result . paragraphs } ,列表: { result . lists } ,表格: { result . tables } ,图片: { result . images } " )
for warning in dict . fromkeys ( result . warnings ) : print ( f " 警告: { warning } " )
return 0
if __name__ == " __main__ " :
raise SystemExit ( main ( ) )