Files
md_to_gost/tests/test_word2md.py
T
Igor20264 510f7e7adf
Python application / build (push) Waiting to run
v0.5.2
Что то сделал
2026-09-08 19:37:54 +03:00

379 lines
14 KiB
Python

"""Tests for word2md (DOCX → md2gost Markdown)."""
from __future__ import annotations
from pathlib import Path
from docx import Document
from word2md.captions import caption_unique_id, parse_any_caption
from word2md.classify import classify_heading_text, is_body_start_heading, is_special_title
from word2md.emit import emit_markdown
from word2md.pipeline import ImportRequest, convert_docx
from word2md.refs import apply_reference_pass, rewrite_text
from word2md.blocks import CaptionBlock, HeadingBlock, ParagraphBlock, InlineSpan, TableBlock, ImageBlock
from word2md.walker import postprocess_blocks
from word2md.tables import table_to_grid
def test_parse_captions():
fig = parse_any_caption("Рисунок 1.1 — Архитектура")
assert fig is not None
assert fig.kind == "figure"
assert fig.number == "1.1"
assert fig.text == "Архитектура"
tbl = parse_any_caption("Таблица 2 — Сравнение")
assert tbl is not None and tbl.kind == "table" and tbl.number == "2"
cont = parse_any_caption("Продолжение Таблицы 2.1")
assert cont is not None and cont.is_continuation and cont.kind == "table"
lst = parse_any_caption("Листинг 1 — Код")
assert lst is not None and lst.kind == "listing"
assert caption_unique_id("figure", "1.1") == "fig1_1"
def test_special_headings():
assert is_special_title("ВВЕДЕНИЕ")
assert is_special_title("Список использованных источников")
level, title, numbered = classify_heading_text("ВВЕДЕНИЕ", 1)
assert level == 1 and not numbered and title == "ВВЕДЕНИЕ"
assert is_body_start_heading("ВВЕДЕНИЕ", "Heading 1")
def test_rewrite_refs():
maps = {
"figure": {"1.1": "fig1_1", "3.1": "fig3_1"},
"table": {"2": "tbl2", "1.1": "tbl1_1"},
"listing": {},
"equation": {"1": "eq1"},
}
text = "См. Рисунок 1.1 и табл. 2, а также [1]."
out = rewrite_text(text, maps)
assert "@Рисунок:fig1_1" in out
assert "@Таблица:tbl2" in out
assert "[1]" in out
# trailing sentence period must not stick to the number
assert rewrite_text("сведены в Таблице 1.1.", maps) == "сведены в @Таблица:tbl1_1."
assert rewrite_text("показан на Рисунке 3.1.", maps) == "показан на @Рисунок:fig3_1."
def test_emit_special_and_table():
blocks = [
HeadingBlock(1, "СОДЕРЖАНИЕ", numbered=False),
HeadingBlock(1, "ВВЕДЕНИЕ", numbered=False),
ParagraphBlock(spans=[InlineSpan("Текст со ссылкой на Рисунок 1.")]),
CaptionBlock(kind="table", number="1", text="Сравнение", unique_id="tbl1"),
TableBlock(rows=[["A", "B"], ["1", "2"]], caption_id="tbl1", caption_text="Сравнение"),
]
blocks = apply_reference_pass(blocks)
# no figure map → leave as is for figure
md = emit_markdown(blocks)
assert "# *СОДЕРЖАНИЕ" in md
assert "# *ВВЕДЕНИЕ" in md
assert "%tbl1 Сравнение" in md
assert "| A | B |" in md
def test_escape_literal_asterisk():
"""DOCX stores bare '*'; MD footnotes need '\\*' so Marko won't eat them."""
from word2md.emit import escape_md_inline
assert escape_md_inline("* Для 2025") == "\\* Для 2025"
assert escape_md_inline("1,50*") == "1,50\\*"
assert escape_md_inline("@Рисунок:fig1_1") == "@Рисунок:fig1_1"
md = emit_markdown(
[
ParagraphBlock(spans=[InlineSpan("* Для 2025 года приведён документооборот.")]),
TableBlock(rows=[["Показатель", "2025"], ["ЭДО", "1,50*"]]),
]
)
assert "\\* Для 2025 года" in md
assert "1,50\\*" in md
# dialect heading star stays structural
assert "# *ВВЕДЕНИЕ" == emit_markdown(
[HeadingBlock(1, "ВВЕДЕНИЕ", numbered=False)]
).strip()
def test_postprocess_continuation_tables():
blocks = [
CaptionBlock(kind="table", number="1", text="T", unique_id="tbl1"),
TableBlock(rows=[["A"], ["1"]], caption_id="tbl1"),
CaptionBlock(kind="table", number="1", is_continuation=True, unique_id="tbl1"),
TableBlock(rows=[["2"]]),
]
out = postprocess_blocks(blocks)
tables = [b for b in out if isinstance(b, TableBlock)]
assert len(tables) == 1
assert tables[0].rows == [["A"], ["1"], ["2"]]
def test_postprocess_skips_repeated_header():
blocks = [
CaptionBlock(kind="table", number="1", text="T", unique_id="tbl1"),
TableBlock(rows=[["H1", "H2"], ["1", "a"]], caption_id="tbl1"),
CaptionBlock(kind="table", number="1", is_continuation=True, unique_id="tbl1"),
TableBlock(rows=[["H1", "H2"], ["2", "b"]]),
]
out = postprocess_blocks(blocks)
tables = [b for b in out if isinstance(b, TableBlock)]
assert len(tables) == 1
assert tables[0].rows == [["H1", "H2"], ["1", "a"], ["2", "b"]]
def test_fold_figure_caption_under_image():
"""GOST: picture then «Рисунок N — …» → one image with %id in title."""
blocks = [
ParagraphBlock(spans=[InlineSpan("См. Рисунок 1.1.")]),
ImageBlock(rel_path="media/a.png"),
CaptionBlock(kind="figure", number="1.1", text="Схема", unique_id="fig1_1"),
]
out = postprocess_blocks(blocks)
assert len([b for b in out if isinstance(b, CaptionBlock)]) == 0
imgs = [b for b in out if isinstance(b, ImageBlock)]
assert len(imgs) == 1
assert imgs[0].caption_id == "fig1_1"
assert imgs[0].caption_text == "Схема"
out = apply_reference_pass(out)
md = emit_markdown(out)
assert "@Рисунок:fig1_1" in md
img_line = [l for l in md.splitlines() if l.startswith("![")][0]
assert "%fig1_1" in img_line
def test_sequential_figures_keep_distinct_ids():
"""Caption under fig N must not leak onto figure N+1 via pending."""
blocks = [
ImageBlock(rel_path="a.png"),
CaptionBlock(kind="figure", number="5.3", text="Архитектура", unique_id="fig5_3"),
ImageBlock(rel_path="b.png"),
CaptionBlock(kind="figure", number="5.2", text="DFD", unique_id="fig5_2"),
]
out = postprocess_blocks(blocks)
imgs = [b for b in out if isinstance(b, ImageBlock)]
assert [i.caption_id for i in imgs] == ["fig5_3", "fig5_2"]
def test_docx_continuation_tables_merge(tmp_path: Path):
"""«Продолжение Таблицы N» + fragment must become one markdown table."""
doc = Document()
_ensure_styles(doc)
h = doc.add_paragraph("ВВЕДЕНИЕ")
try:
h.style = "Heading 1"
except KeyError:
pass
cap = doc.add_paragraph("Таблица 1 — Демо")
try:
cap.style = "Название таблицы"
except KeyError:
pass
t1 = doc.add_table(rows=3, cols=2)
t1.cell(0, 0).text = "A"
t1.cell(0, 1).text = "B"
t1.cell(1, 0).text = "1"
t1.cell(1, 1).text = "x"
t1.cell(2, 0).text = "2"
t1.cell(2, 1).text = "y"
cont = doc.add_paragraph("Продолжение Таблицы 1")
try:
cont.style = "Название таблицы"
except KeyError:
pass
t2 = doc.add_table(rows=2, cols=2)
t2.cell(0, 0).text = "3"
t2.cell(0, 1).text = "z"
t2.cell(1, 0).text = "4"
t2.cell(1, 1).text = "w"
path = tmp_path / "cont.docx"
doc.save(path)
out = tmp_path / "cont.md"
result = convert_docx(ImportRequest(filename=str(path), output=str(out)))
assert result.ok, result.message
md = out.read_text(encoding="utf-8")
assert md.count("%tbl1") == 1
assert "Продолжение" not in md
# one header separator only
assert md.count("| --- | --- |") == 1 or md.count("| --- | --- |") == 1
assert "| 3 | z |" in md
assert "| 4 | w |" in md
assert "| 1 | x |" in md
def _ensure_styles(doc: Document) -> None:
from docx.enum.style import WD_STYLE_TYPE
for name, base in (
("Caption Figure", "Caption"),
("Название таблицы", "Caption"),
("Caption Listing", "Caption"),
("Code", "Normal"),
("Bibliography", "Normal"),
):
try:
doc.styles[name]
except KeyError:
try:
st = doc.styles.add_style(name, WD_STYLE_TYPE.PARAGRAPH)
try:
st.base_style = doc.styles[base]
except KeyError:
pass
except Exception:
pass
def test_convert_minimal_docx(tmp_path: Path):
doc = Document()
_ensure_styles(doc)
doc.add_heading("ВВЕДЕНИЕ", level=1)
# unnumbered look: we'll rely on special title text
p = doc.paragraphs[-1]
p.text = "ВВЕДЕНИЕ"
try:
p.style = doc.styles["Heading 1"]
except KeyError:
pass
doc.add_paragraph("Абзац с жирным текстом и ссылкой [1].")
cap = doc.add_paragraph("Таблица 1 — Демо")
try:
cap.style = "Название таблицы"
except KeyError:
pass
table = doc.add_table(rows=2, cols=2)
table.cell(0, 0).text = "A"
table.cell(0, 1).text = "B"
table.cell(1, 0).text = "1"
table.cell(1, 1).text = "2"
doc.add_heading("ЗАКЛЮЧЕНИЕ", level=1)
doc.paragraphs[-1].text = "ЗАКЛЮЧЕНИЕ"
doc.add_heading("СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ", level=1)
doc.paragraphs[-1].text = "СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ"
bib = doc.add_paragraph("[1]: Иванов И. И. Книга. — М., 2023.")
try:
bib.style = "Bibliography"
except KeyError:
pass
docx_path = tmp_path / "mini.docx"
doc.save(docx_path)
result = convert_docx(ImportRequest(filename=str(docx_path), output=str(tmp_path / "mini.md")))
assert result.ok, result.message
md = Path(result.output_path).read_text(encoding="utf-8")
assert "# *ВВЕДЕНИЕ" in md
assert "%tbl1" in md
assert "| A | B |" in md
assert "[1]:" in md
assert "# *ЗАКЛЮЧЕНИЕ" in md
def test_roundtrip_structure(tmp_path: Path):
"""MD → DOCX (md2gost) → MD (word2md): keep headings / table / biblio shape."""
from md2gost.pipeline import ConvertRequest, convert
md_src = tmp_path / "sample.md"
md_src.write_text(
"""# *ВВЕДЕНИЕ
Текст введения со ссылкой [1].
%cmp Сравнение
| A | B |
|---|---|
| 1 | 2 |
См. Таблица 1 — будет заменена после нумерации.
# *ЗАКЛЮЧЕНИЕ
Итог.
# *СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ
[1]: Источник один. — М., 2023.
""",
encoding="utf-8",
)
docx_out = tmp_path / "sample.docx"
conv = convert(
ConvertRequest(
filename=str(md_src),
output=str(docx_out),
table_continuation="off",
listing_continuation="off",
check=False,
)
)
assert conv.ok, conv.message
md2 = tmp_path / "back.md"
imp = convert_docx(ImportRequest(filename=str(docx_out), output=str(md2)))
assert imp.ok, imp.message
text = md2.read_text(encoding="utf-8")
assert "# *ВВЕДЕНИЕ" in text
assert "# *ЗАКЛЮЧЕНИЕ" in text
assert "| A | B |" in text or "| A |" in text
assert "[1]" in text
assert "%tbl" in text or "Таблица" in text or "|" in text
def test_table_grid_merge_markers():
doc = Document()
table = doc.add_table(rows=2, cols=2)
table.cell(0, 0).text = "X"
table.cell(0, 1).text = "Y"
table.cell(1, 0).text = "Z"
table.cell(1, 1).text = "W"
# merge horizontally first row
table.cell(0, 0).merge(table.cell(0, 1))
rows, merges = table_to_grid(table)
assert len(rows) == 2
# first row should have continue marker for span
assert any(cell == ">" for cell in rows[0]) or merges[0][1][1] == "continue"
def test_nobreak_hyphen_preserved(tmp_path: Path):
"""md2gost writes '-' as w:noBreakHyphen; word2md must restore ASCII hyphen."""
from md2gost.pipeline import ConvertRequest, convert
from md2gost.util import create_element
from docx.oxml.ns import qn
md_src = tmp_path / "hyphen.md"
md_src.write_text(
"# *ВВЕДЕНИЕ\n\n"
"Во-первых, ИТ-компания и 63-ФЗ. Организационно-правовая форма.\n",
encoding="utf-8",
)
docx_out = tmp_path / "hyphen.docx"
conv = convert(
ConvertRequest(
filename=str(md_src),
output=str(docx_out),
table_continuation="off",
listing_continuation="off",
)
)
assert conv.ok, conv.message
# Sanity: DOCX really has noBreakHyphen
doc = Document(str(docx_out))
xml = doc.element.body.xml
assert "noBreakHyphen" in xml
# python-docx .text drops them
joined = " ".join(p.text or "" for p in doc.paragraphs)
assert "Вопервых" in joined.replace(" ", "") or "Во-первых" not in joined
md2 = tmp_path / "back.md"
imp = convert_docx(ImportRequest(filename=str(docx_out), output=str(md2)))
assert imp.ok, imp.message
text = md2.read_text(encoding="utf-8")
assert "Во-первых" in text
assert "ИТ-компания" in text
assert "63-ФЗ" in text
assert "Организационно-правовая" in text