379 lines
14 KiB
Python
379 lines
14 KiB
Python
"""Tests for word2md (DOCX → md2gost Markdown)."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
from docx import Document
|
|
|
|
from word2md.captions import caption_unique_id, parse_any_caption
|
|
from word2md.classify import classify_heading_text, is_body_start_heading, is_special_title
|
|
from word2md.emit import emit_markdown
|
|
from word2md.pipeline import ImportRequest, convert_docx
|
|
from word2md.refs import apply_reference_pass, rewrite_text
|
|
from word2md.blocks import CaptionBlock, HeadingBlock, ParagraphBlock, InlineSpan, TableBlock, ImageBlock
|
|
from word2md.walker import postprocess_blocks
|
|
from word2md.tables import table_to_grid
|
|
|
|
|
|
def test_parse_captions():
|
|
fig = parse_any_caption("Рисунок 1.1 — Архитектура")
|
|
assert fig is not None
|
|
assert fig.kind == "figure"
|
|
assert fig.number == "1.1"
|
|
assert fig.text == "Архитектура"
|
|
|
|
tbl = parse_any_caption("Таблица 2 — Сравнение")
|
|
assert tbl is not None and tbl.kind == "table" and tbl.number == "2"
|
|
|
|
cont = parse_any_caption("Продолжение Таблицы 2.1")
|
|
assert cont is not None and cont.is_continuation and cont.kind == "table"
|
|
|
|
lst = parse_any_caption("Листинг 1 — Код")
|
|
assert lst is not None and lst.kind == "listing"
|
|
assert caption_unique_id("figure", "1.1") == "fig1_1"
|
|
|
|
|
|
def test_special_headings():
|
|
assert is_special_title("ВВЕДЕНИЕ")
|
|
assert is_special_title("Список использованных источников")
|
|
level, title, numbered = classify_heading_text("ВВЕДЕНИЕ", 1)
|
|
assert level == 1 and not numbered and title == "ВВЕДЕНИЕ"
|
|
assert is_body_start_heading("ВВЕДЕНИЕ", "Heading 1")
|
|
|
|
|
|
def test_rewrite_refs():
|
|
maps = {
|
|
"figure": {"1.1": "fig1_1", "3.1": "fig3_1"},
|
|
"table": {"2": "tbl2", "1.1": "tbl1_1"},
|
|
"listing": {},
|
|
"equation": {"1": "eq1"},
|
|
}
|
|
text = "См. Рисунок 1.1 и табл. 2, а также [1]."
|
|
out = rewrite_text(text, maps)
|
|
assert "@Рисунок:fig1_1" in out
|
|
assert "@Таблица:tbl2" in out
|
|
assert "[1]" in out
|
|
# trailing sentence period must not stick to the number
|
|
assert rewrite_text("сведены в Таблице 1.1.", maps) == "сведены в @Таблица:tbl1_1."
|
|
assert rewrite_text("показан на Рисунке 3.1.", maps) == "показан на @Рисунок:fig3_1."
|
|
|
|
|
|
def test_emit_special_and_table():
|
|
blocks = [
|
|
HeadingBlock(1, "СОДЕРЖАНИЕ", numbered=False),
|
|
HeadingBlock(1, "ВВЕДЕНИЕ", numbered=False),
|
|
ParagraphBlock(spans=[InlineSpan("Текст со ссылкой на Рисунок 1.")]),
|
|
CaptionBlock(kind="table", number="1", text="Сравнение", unique_id="tbl1"),
|
|
TableBlock(rows=[["A", "B"], ["1", "2"]], caption_id="tbl1", caption_text="Сравнение"),
|
|
]
|
|
blocks = apply_reference_pass(blocks)
|
|
# no figure map → leave as is for figure
|
|
md = emit_markdown(blocks)
|
|
assert "# *СОДЕРЖАНИЕ" in md
|
|
assert "# *ВВЕДЕНИЕ" in md
|
|
assert "%tbl1 Сравнение" in md
|
|
assert "| A | B |" in md
|
|
|
|
|
|
def test_escape_literal_asterisk():
|
|
"""DOCX stores bare '*'; MD footnotes need '\\*' so Marko won't eat them."""
|
|
from word2md.emit import escape_md_inline
|
|
|
|
assert escape_md_inline("* Для 2025") == "\\* Для 2025"
|
|
assert escape_md_inline("1,50*") == "1,50\\*"
|
|
assert escape_md_inline("@Рисунок:fig1_1") == "@Рисунок:fig1_1"
|
|
|
|
md = emit_markdown(
|
|
[
|
|
ParagraphBlock(spans=[InlineSpan("* Для 2025 года приведён документооборот.")]),
|
|
TableBlock(rows=[["Показатель", "2025"], ["ЭДО", "1,50*"]]),
|
|
]
|
|
)
|
|
assert "\\* Для 2025 года" in md
|
|
assert "1,50\\*" in md
|
|
# dialect heading star stays structural
|
|
assert "# *ВВЕДЕНИЕ" == emit_markdown(
|
|
[HeadingBlock(1, "ВВЕДЕНИЕ", numbered=False)]
|
|
).strip()
|
|
|
|
|
|
def test_postprocess_continuation_tables():
|
|
blocks = [
|
|
CaptionBlock(kind="table", number="1", text="T", unique_id="tbl1"),
|
|
TableBlock(rows=[["A"], ["1"]], caption_id="tbl1"),
|
|
CaptionBlock(kind="table", number="1", is_continuation=True, unique_id="tbl1"),
|
|
TableBlock(rows=[["2"]]),
|
|
]
|
|
out = postprocess_blocks(blocks)
|
|
tables = [b for b in out if isinstance(b, TableBlock)]
|
|
assert len(tables) == 1
|
|
assert tables[0].rows == [["A"], ["1"], ["2"]]
|
|
|
|
|
|
def test_postprocess_skips_repeated_header():
|
|
blocks = [
|
|
CaptionBlock(kind="table", number="1", text="T", unique_id="tbl1"),
|
|
TableBlock(rows=[["H1", "H2"], ["1", "a"]], caption_id="tbl1"),
|
|
CaptionBlock(kind="table", number="1", is_continuation=True, unique_id="tbl1"),
|
|
TableBlock(rows=[["H1", "H2"], ["2", "b"]]),
|
|
]
|
|
out = postprocess_blocks(blocks)
|
|
tables = [b for b in out if isinstance(b, TableBlock)]
|
|
assert len(tables) == 1
|
|
assert tables[0].rows == [["H1", "H2"], ["1", "a"], ["2", "b"]]
|
|
|
|
|
|
def test_fold_figure_caption_under_image():
|
|
"""GOST: picture then «Рисунок N — …» → one image with %id in title."""
|
|
blocks = [
|
|
ParagraphBlock(spans=[InlineSpan("См. Рисунок 1.1.")]),
|
|
ImageBlock(rel_path="media/a.png"),
|
|
CaptionBlock(kind="figure", number="1.1", text="Схема", unique_id="fig1_1"),
|
|
]
|
|
out = postprocess_blocks(blocks)
|
|
assert len([b for b in out if isinstance(b, CaptionBlock)]) == 0
|
|
imgs = [b for b in out if isinstance(b, ImageBlock)]
|
|
assert len(imgs) == 1
|
|
assert imgs[0].caption_id == "fig1_1"
|
|
assert imgs[0].caption_text == "Схема"
|
|
out = apply_reference_pass(out)
|
|
md = emit_markdown(out)
|
|
assert "@Рисунок:fig1_1" in md
|
|
img_line = [l for l in md.splitlines() if l.startswith("![")][0]
|
|
assert "%fig1_1" in img_line
|
|
|
|
|
|
def test_sequential_figures_keep_distinct_ids():
|
|
"""Caption under fig N must not leak onto figure N+1 via pending."""
|
|
blocks = [
|
|
ImageBlock(rel_path="a.png"),
|
|
CaptionBlock(kind="figure", number="5.3", text="Архитектура", unique_id="fig5_3"),
|
|
ImageBlock(rel_path="b.png"),
|
|
CaptionBlock(kind="figure", number="5.2", text="DFD", unique_id="fig5_2"),
|
|
]
|
|
out = postprocess_blocks(blocks)
|
|
imgs = [b for b in out if isinstance(b, ImageBlock)]
|
|
assert [i.caption_id for i in imgs] == ["fig5_3", "fig5_2"]
|
|
|
|
|
|
def test_docx_continuation_tables_merge(tmp_path: Path):
|
|
"""«Продолжение Таблицы N» + fragment must become one markdown table."""
|
|
doc = Document()
|
|
_ensure_styles(doc)
|
|
h = doc.add_paragraph("ВВЕДЕНИЕ")
|
|
try:
|
|
h.style = "Heading 1"
|
|
except KeyError:
|
|
pass
|
|
cap = doc.add_paragraph("Таблица 1 — Демо")
|
|
try:
|
|
cap.style = "Название таблицы"
|
|
except KeyError:
|
|
pass
|
|
t1 = doc.add_table(rows=3, cols=2)
|
|
t1.cell(0, 0).text = "A"
|
|
t1.cell(0, 1).text = "B"
|
|
t1.cell(1, 0).text = "1"
|
|
t1.cell(1, 1).text = "x"
|
|
t1.cell(2, 0).text = "2"
|
|
t1.cell(2, 1).text = "y"
|
|
cont = doc.add_paragraph("Продолжение Таблицы 1")
|
|
try:
|
|
cont.style = "Название таблицы"
|
|
except KeyError:
|
|
pass
|
|
t2 = doc.add_table(rows=2, cols=2)
|
|
t2.cell(0, 0).text = "3"
|
|
t2.cell(0, 1).text = "z"
|
|
t2.cell(1, 0).text = "4"
|
|
t2.cell(1, 1).text = "w"
|
|
path = tmp_path / "cont.docx"
|
|
doc.save(path)
|
|
out = tmp_path / "cont.md"
|
|
result = convert_docx(ImportRequest(filename=str(path), output=str(out)))
|
|
assert result.ok, result.message
|
|
md = out.read_text(encoding="utf-8")
|
|
assert md.count("%tbl1") == 1
|
|
assert "Продолжение" not in md
|
|
# one header separator only
|
|
assert md.count("| --- | --- |") == 1 or md.count("| --- | --- |") == 1
|
|
assert "| 3 | z |" in md
|
|
assert "| 4 | w |" in md
|
|
assert "| 1 | x |" in md
|
|
|
|
|
|
def _ensure_styles(doc: Document) -> None:
|
|
from docx.enum.style import WD_STYLE_TYPE
|
|
|
|
for name, base in (
|
|
("Caption Figure", "Caption"),
|
|
("Название таблицы", "Caption"),
|
|
("Caption Listing", "Caption"),
|
|
("Code", "Normal"),
|
|
("Bibliography", "Normal"),
|
|
):
|
|
try:
|
|
doc.styles[name]
|
|
except KeyError:
|
|
try:
|
|
st = doc.styles.add_style(name, WD_STYLE_TYPE.PARAGRAPH)
|
|
try:
|
|
st.base_style = doc.styles[base]
|
|
except KeyError:
|
|
pass
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def test_convert_minimal_docx(tmp_path: Path):
|
|
doc = Document()
|
|
_ensure_styles(doc)
|
|
doc.add_heading("ВВЕДЕНИЕ", level=1)
|
|
# unnumbered look: we'll rely on special title text
|
|
p = doc.paragraphs[-1]
|
|
p.text = "ВВЕДЕНИЕ"
|
|
try:
|
|
p.style = doc.styles["Heading 1"]
|
|
except KeyError:
|
|
pass
|
|
doc.add_paragraph("Абзац с жирным текстом и ссылкой [1].")
|
|
cap = doc.add_paragraph("Таблица 1 — Демо")
|
|
try:
|
|
cap.style = "Название таблицы"
|
|
except KeyError:
|
|
pass
|
|
table = doc.add_table(rows=2, cols=2)
|
|
table.cell(0, 0).text = "A"
|
|
table.cell(0, 1).text = "B"
|
|
table.cell(1, 0).text = "1"
|
|
table.cell(1, 1).text = "2"
|
|
doc.add_heading("ЗАКЛЮЧЕНИЕ", level=1)
|
|
doc.paragraphs[-1].text = "ЗАКЛЮЧЕНИЕ"
|
|
doc.add_heading("СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ", level=1)
|
|
doc.paragraphs[-1].text = "СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ"
|
|
bib = doc.add_paragraph("[1]: Иванов И. И. Книга. — М., 2023.")
|
|
try:
|
|
bib.style = "Bibliography"
|
|
except KeyError:
|
|
pass
|
|
|
|
docx_path = tmp_path / "mini.docx"
|
|
doc.save(docx_path)
|
|
|
|
result = convert_docx(ImportRequest(filename=str(docx_path), output=str(tmp_path / "mini.md")))
|
|
assert result.ok, result.message
|
|
md = Path(result.output_path).read_text(encoding="utf-8")
|
|
assert "# *ВВЕДЕНИЕ" in md
|
|
assert "%tbl1" in md
|
|
assert "| A | B |" in md
|
|
assert "[1]:" in md
|
|
assert "# *ЗАКЛЮЧЕНИЕ" in md
|
|
|
|
|
|
def test_roundtrip_structure(tmp_path: Path):
|
|
"""MD → DOCX (md2gost) → MD (word2md): keep headings / table / biblio shape."""
|
|
from md2gost.pipeline import ConvertRequest, convert
|
|
|
|
md_src = tmp_path / "sample.md"
|
|
md_src.write_text(
|
|
"""# *ВВЕДЕНИЕ
|
|
|
|
Текст введения со ссылкой [1].
|
|
|
|
%cmp Сравнение
|
|
|
|
| A | B |
|
|
|---|---|
|
|
| 1 | 2 |
|
|
|
|
См. Таблица 1 — будет заменена после нумерации.
|
|
|
|
# *ЗАКЛЮЧЕНИЕ
|
|
|
|
Итог.
|
|
|
|
# *СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ
|
|
|
|
[1]: Источник один. — М., 2023.
|
|
""",
|
|
encoding="utf-8",
|
|
)
|
|
docx_out = tmp_path / "sample.docx"
|
|
conv = convert(
|
|
ConvertRequest(
|
|
filename=str(md_src),
|
|
output=str(docx_out),
|
|
table_continuation="off",
|
|
listing_continuation="off",
|
|
check=False,
|
|
)
|
|
)
|
|
assert conv.ok, conv.message
|
|
|
|
md2 = tmp_path / "back.md"
|
|
imp = convert_docx(ImportRequest(filename=str(docx_out), output=str(md2)))
|
|
assert imp.ok, imp.message
|
|
text = md2.read_text(encoding="utf-8")
|
|
assert "# *ВВЕДЕНИЕ" in text
|
|
assert "# *ЗАКЛЮЧЕНИЕ" in text
|
|
assert "| A | B |" in text or "| A |" in text
|
|
assert "[1]" in text
|
|
assert "%tbl" in text or "Таблица" in text or "|" in text
|
|
|
|
|
|
def test_table_grid_merge_markers():
|
|
doc = Document()
|
|
table = doc.add_table(rows=2, cols=2)
|
|
table.cell(0, 0).text = "X"
|
|
table.cell(0, 1).text = "Y"
|
|
table.cell(1, 0).text = "Z"
|
|
table.cell(1, 1).text = "W"
|
|
# merge horizontally first row
|
|
table.cell(0, 0).merge(table.cell(0, 1))
|
|
rows, merges = table_to_grid(table)
|
|
assert len(rows) == 2
|
|
# first row should have continue marker for span
|
|
assert any(cell == ">" for cell in rows[0]) or merges[0][1][1] == "continue"
|
|
|
|
|
|
def test_nobreak_hyphen_preserved(tmp_path: Path):
|
|
"""md2gost writes '-' as w:noBreakHyphen; word2md must restore ASCII hyphen."""
|
|
from md2gost.pipeline import ConvertRequest, convert
|
|
from md2gost.util import create_element
|
|
from docx.oxml.ns import qn
|
|
|
|
md_src = tmp_path / "hyphen.md"
|
|
md_src.write_text(
|
|
"# *ВВЕДЕНИЕ\n\n"
|
|
"Во-первых, ИТ-компания и 63-ФЗ. Организационно-правовая форма.\n",
|
|
encoding="utf-8",
|
|
)
|
|
docx_out = tmp_path / "hyphen.docx"
|
|
conv = convert(
|
|
ConvertRequest(
|
|
filename=str(md_src),
|
|
output=str(docx_out),
|
|
table_continuation="off",
|
|
listing_continuation="off",
|
|
)
|
|
)
|
|
assert conv.ok, conv.message
|
|
|
|
# Sanity: DOCX really has noBreakHyphen
|
|
doc = Document(str(docx_out))
|
|
xml = doc.element.body.xml
|
|
assert "noBreakHyphen" in xml
|
|
# python-docx .text drops them
|
|
joined = " ".join(p.text or "" for p in doc.paragraphs)
|
|
assert "Вопервых" in joined.replace(" ", "") or "Во-первых" not in joined
|
|
|
|
md2 = tmp_path / "back.md"
|
|
imp = convert_docx(ImportRequest(filename=str(docx_out), output=str(md2)))
|
|
assert imp.ok, imp.message
|
|
text = md2.read_text(encoding="utf-8")
|
|
assert "Во-первых" in text
|
|
assert "ИТ-компания" in text
|
|
assert "63-ФЗ" in text
|
|
assert "Организационно-правовая" in text
|