Files
CraftKit/migration/tests/test_docx_to_md.py
T

118 lines
5.1 KiB
Python

import importlib.util
import sys
import tempfile
import unittest
import zipfile
from pathlib import Path
SCRIPT = (
Path(__file__).parents[2]
/ "plugins"
/ "doc"
/ "skills"
/ "docx-to-md"
/ "scripts"
/ "convert.py"
)
SPEC = importlib.util.spec_from_file_location("docx_to_md_convert", SCRIPT)
MODULE = importlib.util.module_from_spec(SPEC)
assert SPEC.loader is not None
sys.modules[SPEC.name] = MODULE
SPEC.loader.exec_module(MODULE)
class DocxToMarkdownTest(unittest.TestCase):
"""验证转换器的核心内容、覆盖保护和输入边界。"""
def make_docx(self, path: Path) -> None:
"""创建只包含公开 OOXML 结构的最小测试文档。"""
document = """<?xml version="1.0" encoding="UTF-8"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"
xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships"
xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
<w:body>
<w:p><w:pPr><w:pStyle w:val="Heading1"/></w:pPr><w:r><w:t>测试标题</w:t></w:r></w:p>
<w:p><w:r><w:rPr><w:b/></w:rPr><w:t>加粗正文</w:t></w:r>
<w:hyperlink r:id="rLink"><w:r><w:t>示例链接</w:t></w:r></w:hyperlink></w:p>
<w:p><w:pPr><w:numPr><w:ilvl w:val="0"/><w:numId w:val="1"/></w:numPr></w:pPr>
<w:r><w:t>列表项目</w:t></w:r></w:p>
<w:p><w:r><w:drawing><a:blip r:embed="rImage"/></w:drawing></w:r></w:p>
<w:tbl>
<w:tr><w:tc><w:p><w:r><w:t>名称</w:t></w:r></w:p></w:tc><w:tc><w:p><w:r><w:t>值</w:t></w:r></w:p></w:tc></w:tr>
<w:tr><w:tc><w:p><w:r><w:t>A</w:t></w:r></w:p></w:tc><w:tc><w:p><w:r><w:t>1</w:t></w:r></w:p></w:tc></w:tr>
</w:tbl>
</w:body>
</w:document>"""
styles = """<?xml version="1.0" encoding="UTF-8"?>
<w:styles xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:style w:type="paragraph" w:styleId="Heading1"><w:name w:val="heading 1"/></w:style>
</w:styles>"""
numbering = """<?xml version="1.0" encoding="UTF-8"?>
<w:numbering xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:abstractNum w:abstractNumId="0"><w:lvl w:ilvl="0"><w:numFmt w:val="bullet"/></w:lvl></w:abstractNum>
<w:num w:numId="1"><w:abstractNumId w:val="0"/></w:num>
</w:numbering>"""
relationships = """<?xml version="1.0" encoding="UTF-8"?>
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
<Relationship Id="rLink" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="https://example.com" TargetMode="External"/>
<Relationship Id="rImage" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/image" Target="media/test.png"/>
</Relationships>"""
with zipfile.ZipFile(path, "w") as archive:
archive.writestr("word/document.xml", document)
archive.writestr("word/styles.xml", styles)
archive.writestr("word/numbering.xml", numbering)
archive.writestr("word/_rels/document.xml.rels", relationships)
archive.writestr("word/media/test.png", b"\x89PNG\r\n\x1a\n")
archive.writestr("word/comments.xml", "<comments/>")
def test_convert_common_content_and_report_warning(self) -> None:
"""常见结构应转换,无法处理的批注应明确告警。"""
with tempfile.TemporaryDirectory() as temp:
root = Path(temp)
source = root / "sample.docx"
output = root / "result"
self.make_docx(source)
result = MODULE.DocxConverter(source, output).convert()
markdown = result.output_file.read_text(encoding="utf-8")
self.assertIn("# 测试标题", markdown)
self.assertIn("**加粗正文**", markdown)
self.assertIn("[示例链接](https://example.com)", markdown)
self.assertIn("- 列表项目", markdown)
self.assertIn("| 名称 | 值 |", markdown)
self.assertIn("![图片](images/image-", markdown)
self.assertEqual(result.images, 1)
self.assertTrue(any("批注" in warning for warning in result.warnings))
def test_existing_output_requires_force(self) -> None:
"""默认不得写入已有输出目录,显式覆盖后才可继续。"""
with tempfile.TemporaryDirectory() as temp:
root = Path(temp)
source = root / "sample.docx"
output = root / "result"
self.make_docx(source)
output.mkdir()
with self.assertRaises(FileExistsError):
MODULE.DocxConverter(source, output).convert()
result = MODULE.DocxConverter(source, output, force=True).convert()
self.assertTrue(result.output_file.exists())
def test_cli_rejects_non_docx(self) -> None:
"""命令行入口应拒绝扩展名不正确的文件。"""
with tempfile.TemporaryDirectory() as temp:
source = Path(temp) / "sample.txt"
source.write_text("not docx", encoding="utf-8")
self.assertEqual(MODULE.main([str(source)]), 2)
if __name__ == "__main__":
unittest.main()