govdoc-md-cleaner v0.1.0: 政务文档 Markdown 清洗工具(规则引擎 + CLI + 测试)

- 8 条 YAML 声明规则:页码/页眉页脚/目录点线/图片/HTML表格/散落标签/行尾空白/空行
- 防误伤设计:protect 正则 + 内容形态豁免 + OCR burst 检测
- md-clean single/batch CLI,JSON 清洗报告
- 18 个单元测试

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
2026-08-20 17:04:10 +08:00
commit c19e9fcebf
10 changed files with 814 additions and 0 deletions
+154
View File
@@ -0,0 +1,154 @@
"""清洗规则单元测试。
fixtures 里是各脏数据模式的最小样例,跑一遍断言规则命中且正文无损。
"""
import sys
import unittest
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from cleaner.cleaner import MarkdownCleaner
from cleaner.rules import (
_drop_images,
_normalize_tables,
_rstrip_lines,
_strip_page_lines,
_strip_repeated_short_lines,
_strip_toc_dots,
load_rules,
)
S = {} # 每个用例独立 stats
def st():
return {}
class TestPageLines(unittest.TestCase):
def test_cn_page(self):
text = "正文A\n第 3 页 共 53 页\n正文B\n第4页共4页\n结尾"
out = _strip_page_lines(text, {}, st())
self.assertEqual(out, "正文A\n正文B\n结尾")
def test_en_page(self):
text = "foo\nPage 3 of 10\nbar\n- 4 -\nbaz"
out = _strip_page_lines(text, {"keep_bare_numbers": False}, st())
self.assertNotIn("Page 3", out)
self.assertNotIn("- 4 -", out)
def test_bare_number_kept_by_default(self):
text = "条款\n12\n下文"
out = _strip_page_lines(text, {}, st())
self.assertIn("12", out)
class TestRepeatedLines(unittest.TestCase):
def test_header_removed(self):
lines = ["某某采购项目招标文件"] + ["内容%d" % i for i in range(5)]
text = "\n".join(("某某采购项目招标文件 \n" + l) for l in lines)
out = _strip_repeated_short_lines(text, {"threshold": 3, "max_len": 40}, st())
self.assertNotIn("某某采购项目招标文件", out)
self.assertIn("内容1", out)
def test_protected_signature_kept(self):
text = "\n".join(["投标人:(公章)"] * 4 + ["正文"])
out = _strip_repeated_short_lines(
text, {"threshold": 3, "max_len": 40, "protect": ["公章"]}, st()
)
self.assertIn("投标人:(公章)", out)
def test_numbered_clause_kept(self):
# 编号条款是正文不是页眉
text = "\n".join(["1)乙方须接受甲方监督。"] * 4 + ["正文"])
out = _strip_repeated_short_lines(text, {"threshold": 3, "max_len": 40}, st())
self.assertIn("1)乙方须接受甲方监督。", out)
def test_ocr_burst_removed(self):
text = "\n".join(["审计程序"] * 30)
out = _strip_repeated_short_lines(
text,
{"threshold": 999, "max_len": 40, "burst_limit": 5, "burst_gap": 2},
st(),
)
self.assertNotIn("审计程序", out)
class TestTocDots(unittest.TestCase):
def test_toc_line(self):
text = "第一章 投标邀请函 ……………………………………… 2"
out = _strip_toc_dots(text, {}, st())
self.assertEqual(out, "第一章 投标邀请函")
def test_plain_toc_line_keeps_text(self):
# 无标题目录行:去掉点线页码,保留"序号+标题"文字
text = "21 迷交的投标文件 ………………………………………………… 25"
out = _strip_toc_dots(text, {}, st())
self.assertEqual(out, "21 迷交的投标文件")
def test_body_with_ellipsis_kept(self):
text = "此处省略部分内容……后续"
out = _strip_toc_dots(text, {}, st())
self.assertIn("后续", out)
class TestImages(unittest.TestCase):
def test_image_line_dropped(self):
text = "![](images/abc.jpg) \n正文"
out = _drop_images(text, {}, st())
self.assertEqual(out, "正文")
def test_placeholder(self):
text = "![](images/abc.jpg)\n正文"
out = _drop_images(text, {"placeholder": "[图]"}, st())
self.assertIn("[图]", out)
def test_base64_image_dropped(self):
text = "![名称](data:image/png;base64,AAAA)\n正文"
out = _drop_images(text, {}, st())
self.assertNotIn("base64", out)
class TestTables(unittest.TestCase):
def test_table_to_pipe(self):
html = "<table><tr><td>序号</td><td>名称</td></tr><tr><td>1</td><td>保洁</td></tr></table>"
out = _normalize_tables(html, {}, st())
self.assertIn("| 序号 | 名称 |", out)
self.assertIn("| 1 | 保洁 |", out)
self.assertIn("|---|---|", out)
def test_pipe_escaped(self):
html = "<table><tr><td>a|b</td></tr></table>"
out = _normalize_tables(html, {}, st())
self.assertIn("a\\|b", out)
def test_br_in_cell(self):
html = "<table><tr><td>品<br/>目</td></tr></table>"
out = _normalize_tables(html, {}, st())
self.assertIn("品 目", out)
class TestEngine(unittest.TestCase):
def test_full_pipeline(self):
cleaner = MarkdownCleaner(rules=load_rules())
raw = (
"# 标题\n![](images/x.jpg) \n第 1 页 共 2 页 \n"
"正文一段。 \n<table><tr><td>a</td></tr></table> \n\n\n\n尾部\n"
)
result = cleaner.clean_text(raw)
self.assertNotIn("images/", result.text)
self.assertNotIn("第 1 页", result.text)
self.assertNotIn("<table>", result.text)
self.assertIn("| a |", result.text)
self.assertNotIn("\n\n\n", result.text)
def test_default_rules_load(self):
rules = load_rules()
self.assertTrue(len(rules) >= 8)
self.assertEqual(rules, sorted(rules, key=lambda r: r.order))
if __name__ == "__main__":
unittest.main(verbosity=2)