feat(doc): 新增 DOCX 转 Markdown Skill
This commit is contained in:
@@ -0,0 +1,117 @@
|
||||
import importlib.util
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
SCRIPT = (
|
||||
Path(__file__).parents[2]
|
||||
/ "plugins"
|
||||
/ "doc"
|
||||
/ "skills"
|
||||
/ "docx-to-md"
|
||||
/ "scripts"
|
||||
/ "convert.py"
|
||||
)
|
||||
SPEC = importlib.util.spec_from_file_location("docx_to_md_convert", SCRIPT)
|
||||
MODULE = importlib.util.module_from_spec(SPEC)
|
||||
assert SPEC.loader is not None
|
||||
sys.modules[SPEC.name] = MODULE
|
||||
SPEC.loader.exec_module(MODULE)
|
||||
|
||||
|
||||
class DocxToMarkdownTest(unittest.TestCase):
|
||||
"""验证转换器的核心内容、覆盖保护和输入边界。"""
|
||||
|
||||
def make_docx(self, path: Path) -> None:
|
||||
"""创建只包含公开 OOXML 结构的最小测试文档。"""
|
||||
|
||||
document = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
||||
xmlns:r="http://schemas.openxmlformats.org/officeDocument/2006/relationships"
|
||||
xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
|
||||
<w:body>
|
||||
<w:p><w:pPr><w:pStyle w:val="Heading1"/></w:pPr><w:r><w:t>测试标题</w:t></w:r></w:p>
|
||||
<w:p><w:r><w:rPr><w:b/></w:rPr><w:t>加粗正文</w:t></w:r>
|
||||
<w:hyperlink r:id="rLink"><w:r><w:t>示例链接</w:t></w:r></w:hyperlink></w:p>
|
||||
<w:p><w:pPr><w:numPr><w:ilvl w:val="0"/><w:numId w:val="1"/></w:numPr></w:pPr>
|
||||
<w:r><w:t>列表项目</w:t></w:r></w:p>
|
||||
<w:p><w:r><w:drawing><a:blip r:embed="rImage"/></w:drawing></w:r></w:p>
|
||||
<w:tbl>
|
||||
<w:tr><w:tc><w:p><w:r><w:t>名称</w:t></w:r></w:p></w:tc><w:tc><w:p><w:r><w:t>值</w:t></w:r></w:p></w:tc></w:tr>
|
||||
<w:tr><w:tc><w:p><w:r><w:t>A</w:t></w:r></w:p></w:tc><w:tc><w:p><w:r><w:t>1</w:t></w:r></w:p></w:tc></w:tr>
|
||||
</w:tbl>
|
||||
</w:body>
|
||||
</w:document>"""
|
||||
styles = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<w:styles xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:style w:type="paragraph" w:styleId="Heading1"><w:name w:val="heading 1"/></w:style>
|
||||
</w:styles>"""
|
||||
numbering = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<w:numbering xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:abstractNum w:abstractNumId="0"><w:lvl w:ilvl="0"><w:numFmt w:val="bullet"/></w:lvl></w:abstractNum>
|
||||
<w:num w:numId="1"><w:abstractNumId w:val="0"/></w:num>
|
||||
</w:numbering>"""
|
||||
relationships = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<Relationships xmlns="http://schemas.openxmlformats.org/package/2006/relationships">
|
||||
<Relationship Id="rLink" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/hyperlink" Target="https://example.com" TargetMode="External"/>
|
||||
<Relationship Id="rImage" Type="http://schemas.openxmlformats.org/officeDocument/2006/relationships/image" Target="media/test.png"/>
|
||||
</Relationships>"""
|
||||
with zipfile.ZipFile(path, "w") as archive:
|
||||
archive.writestr("word/document.xml", document)
|
||||
archive.writestr("word/styles.xml", styles)
|
||||
archive.writestr("word/numbering.xml", numbering)
|
||||
archive.writestr("word/_rels/document.xml.rels", relationships)
|
||||
archive.writestr("word/media/test.png", b"\x89PNG\r\n\x1a\n")
|
||||
archive.writestr("word/comments.xml", "<comments/>")
|
||||
|
||||
def test_convert_common_content_and_report_warning(self) -> None:
|
||||
"""常见结构应转换,无法处理的批注应明确告警。"""
|
||||
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
root = Path(temp)
|
||||
source = root / "sample.docx"
|
||||
output = root / "result"
|
||||
self.make_docx(source)
|
||||
|
||||
result = MODULE.DocxConverter(source, output).convert()
|
||||
markdown = result.output_file.read_text(encoding="utf-8")
|
||||
|
||||
self.assertIn("# 测试标题", markdown)
|
||||
self.assertIn("**加粗正文**", markdown)
|
||||
self.assertIn("[示例链接](https://example.com)", markdown)
|
||||
self.assertIn("- 列表项目", markdown)
|
||||
self.assertIn("| 名称 | 值 |", markdown)
|
||||
self.assertIn("
|
||||
self.assertEqual(result.images, 1)
|
||||
self.assertTrue(any("批注" in warning for warning in result.warnings))
|
||||
|
||||
def test_existing_output_requires_force(self) -> None:
|
||||
"""默认不得写入已有输出目录,显式覆盖后才可继续。"""
|
||||
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
root = Path(temp)
|
||||
source = root / "sample.docx"
|
||||
output = root / "result"
|
||||
self.make_docx(source)
|
||||
output.mkdir()
|
||||
|
||||
with self.assertRaises(FileExistsError):
|
||||
MODULE.DocxConverter(source, output).convert()
|
||||
|
||||
result = MODULE.DocxConverter(source, output, force=True).convert()
|
||||
self.assertTrue(result.output_file.exists())
|
||||
|
||||
def test_cli_rejects_non_docx(self) -> None:
|
||||
"""命令行入口应拒绝扩展名不正确的文件。"""
|
||||
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
source = Path(temp) / "sample.txt"
|
||||
source.write_text("not docx", encoding="utf-8")
|
||||
self.assertEqual(MODULE.main([str(source)]), 2)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user