"""Convert md2gost markdown AST → Flat ODF Text (.fodt) body + document."""
from __future__ import annotations
import base64
import os
import re
from dataclasses import dataclass, field
from pathlib import Path
from md2gost.extended_markdown import markdown
from md2gost.diagram_renderer import DIAGRAM_LANGS
from md2gost.profiles import preprocess_markdown
from .escape import escape_text, xml_id
from .styles import FODT_NS, automatic_styles_xml, styles_xml
@dataclass
class EmitState:
body: list[str] = field(default_factory=list)
pending_caption: tuple[str | None, str | None, bool] | None = None
biblio: list[tuple[str, str]] = field(default_factory=list)
work_dir: Path | None = None
fig_n: int = 0
tab_n: int = 0
lst_n: int = 0
labels: dict[str, str] = field(default_factory=dict) # id -> display number text
list_style_i: int = 0
def w(self, s: str = "") -> None:
self.body.append(s)
def _inline(children, st: EmitState | None = None) -> str:
parts: list[str] = []
for child in children or []:
t = type(child).__name__
if t in ("RawText", "Literal"):
parts.append(escape_text(getattr(child, "children", "") or ""))
elif t == "CodeSpan":
parts.append(
f'{escape_text(child.children)}'
)
elif t == "Emphasis":
parts.append(
f'{_inline(child.children, st)}'
)
elif t == "StrongEmphasis":
parts.append(
f'{_inline(child.children, st)}'
)
elif t == "Link":
href = escape_text(child.dest)
parts.append(
f''
f"{_inline(child.children, st)}"
)
elif t == "Reference":
full = {
"рисунок": "Рисунок",
"таблица": "Таблица",
"листинг": "Листинг",
"формула": "формула",
}.get(child.type.lower(), child.type)
key = f"{child.type.lower()}:{child.name}"
num = (st.labels.get(key) if st else None) or "?"
parts.append(escape_text(f"{full} {num}"))
elif t == "InlineEquation":
parts.append(
f''
f"{escape_text(child.latex_equation)}"
)
elif t == "Image":
parts.append("")
elif t == "LineBreak":
parts.append("")
elif hasattr(child, "children"):
parts.append(_inline(child.children, st))
else:
parts.append(escape_text(str(child)))
return "".join(parts)
def _plain_inline(children) -> str:
parts: list[str] = []
for child in children or []:
t = type(child).__name__
if t in ("RawText", "Literal"):
parts.append(getattr(child, "children", "") or "")
elif t == "CodeSpan":
parts.append(child.children or "")
elif hasattr(child, "children"):
parts.append(_plain_inline(child.children))
return "".join(parts)
def _p(style: str, content: str) -> str:
return f'{content}'
def _h(level: int, content: str) -> str:
style = {1: "Heading_20_1", 2: "Heading_20_2", 3: "Heading_20_3"}.get(
level, "Heading_20_3"
)
return f'{content}'
def _emit_heading(st: EmitState, heading) -> None:
text = _inline(heading.children, st).strip()
plain = _plain_inline(heading.children).strip()
upper = plain.upper()
level = min(int(heading.level), 3)
if upper == "СОДЕРЖАНИЕ" or upper.startswith("СОДЕРЖАНИЕ"):
st.w(_p("Centered", escape_text("СОДЕРЖАНИЕ")))
st.w(
''
''
''
"Содержание"
""
""
+ _p("Standard", escape_text(
"(Обновите содержание в LibreOffice Writer: ПКМ → Обновить индекс)"
))
+ ""
)
return
st.w(_h(level, text))
def _emit_paragraph(st: EmitState, para) -> None:
text_probe = ""
for ch in para.children or []:
if type(ch).__name__ in ("RawText", "Literal"):
text_probe += ch.children or ""
m = re.match(r"^\[([^\]]+)\]:\s*(.+)$", text_probe.strip())
if m:
st.biblio.append((m.group(1), m.group(2).strip()))
return
has_image = any(type(c).__name__ == "Image" for c in (para.children or []))
if has_image:
for c in para.children:
if type(c).__name__ == "Image":
_emit_image(st, c)
others = [c for c in para.children if type(c).__name__ != "Image"]
t = _inline(others, st).strip()
if t:
st.w(_p("Standard", t))
return
raw = _inline(para.children, st).strip()
if raw:
st.w(_p("Standard", raw))
def _mime_for(path: Path) -> str:
ext = path.suffix.lower()
return {
".png": "image/png",
".jpg": "image/jpeg",
".jpeg": "image/jpeg",
".gif": "image/gif",
".svg": "image/svg+xml",
".bmp": "image/bmp",
".webp": "image/webp",
}.get(ext, "application/octet-stream")
def _emit_image(st: EmitState, img) -> None:
dest = img.dest
title = getattr(img, "title", None) or ""
caption_id = getattr(img, "unique_name", None)
caption_text = title
if title.startswith("%"):
parts = title[1:].split(None, 1)
caption_id = parts[0]
caption_text = parts[1] if len(parts) > 1 else ""
if st.pending_caption and not caption_text:
caption_id = caption_id or st.pending_caption[0]
caption_text = st.pending_caption[1] or ""
st.pending_caption = None
src = dest
if st.work_dir and not os.path.isabs(src):
src = str(st.work_dir / dest)
st.fig_n += 1
num = str(st.fig_n)
if caption_id:
st.labels[f"рисунок:{caption_id}"] = num
frame_inner = ""
if os.path.isfile(src):
data = Path(src).read_bytes()
b64 = base64.b64encode(data).decode("ascii")
name = Path(dest).name.replace(" ", "_")
mime = _mime_for(Path(src))
# width ~ 16cm (page content ≈ 17cm)
frame_inner = (
f''
f''
f"{b64}"
f""
)
else:
frame_inner = escape_text(f"[изображение не найдено: {dest}]")
st.w(_p("Centered", frame_inner))
cap = caption_text or "Рисунок"
label = f"Рисунок {num} — {cap}" if caption_text else f"Рисунок {num}"
bookmark = ""
if caption_id:
bid = xml_id(caption_id, "fig")
bookmark = f''
st.w(_p("Caption", bookmark + escape_text(label)))
def _cell_spans(rows, row: int, col: int) -> tuple[int, int] | None:
"""Return (rowspan, colspan) or None if cell is covered by a merge."""
cell = rows[row].children[col]
if getattr(cell, "merge_v", "none") == "continue":
return None
if getattr(cell, "merge_h", "none") == "continue":
return None
rowspan = 1
if getattr(cell, "merge_v", "none") == "restart":
r = row + 1
while r < len(rows) and getattr(rows[r].children[col], "merge_v", "none") == "continue":
rowspan += 1
r += 1
colspan = 1
n_cols = len(rows[row].children)
if getattr(cell, "merge_h", "none") == "restart":
c = col + 1
while c < n_cols and getattr(rows[row].children[c], "merge_h", "none") == "continue":
colspan += 1
c += 1
return rowspan, colspan
def _emit_table(st: EmitState, table, caption_id: str | None, caption_text: str | None) -> None:
rows = table.children
if not rows:
return
st.tab_n += 1
num = str(st.tab_n)
if caption_id:
st.labels[f"таблица:{caption_id}"] = num
if caption_text or caption_id:
bookmark = ""
if caption_id:
bookmark = f''
label = f"Таблица {num}"
if caption_text:
label = f"Таблица {num} — {caption_text}"
st.w(_p("Caption", bookmark + escape_text(label)))
n_cols = len(rows[0].children)
col_w = max(1, 1000 // n_cols)
covered: set[tuple[int, int]] = set()
parts: list[str] = [
f''
]
for _ in range(n_cols):
parts.append(
f''
)
# Header row: repeat on page break in Writer
parts.append("")
parts.append(_emit_table_row(rows, 0, n_cols, st, covered))
parts.append("")
for ri in range(1, len(rows)):
parts.append(_emit_table_row(rows, ri, n_cols, st, covered))
parts.append("")
st.w("".join(parts))
def _emit_table_row(
rows, ri: int, n_cols: int, st: EmitState, covered: set[tuple[int, int]]
) -> str:
cells_xml: list[str] = [""]
row = rows[ri]
ci = 0
while ci < n_cols:
if (ri, ci) in covered:
cells_xml.append("")
ci += 1
continue
if ci >= len(row.children):
cells_xml.append(
''
f"{_p('Pcell', '')}"
)
ci += 1
continue
spans = _cell_spans(rows, ri, ci)
if spans is None:
cells_xml.append("")
ci += 1
continue
rowspan, colspan = spans
attrs = ' table:style-name="Table1.1"'
if rowspan > 1:
attrs += f' table:number-rows-spanned="{rowspan}"'
if colspan > 1:
attrs += f' table:number-columns-spanned="{colspan}"'
content = _inline(row.children[ci].children, st)
cells_xml.append(
f''
f"{_p('Pcell', content)}"
f""
)
for dr in range(rowspan):
for dc in range(colspan):
if dr == 0 and dc == 0:
continue
covered.add((ri + dr, ci + dc))
for dc in range(1, colspan):
cells_xml.append("")
covered.discard((ri, ci + dc))
ci += colspan
cells_xml.append("")
return "".join(cells_xml)
def _emit_code(st: EmitState, block, lang: str) -> None:
source = block.children[0].children if block.children else ""
cap_id, cap_text, _with_listing = (None, None, False)
if st.pending_caption:
cap_id, cap_text, _with_listing = st.pending_caption
st.pending_caption = None
lang_l = (lang or "").strip().lower()
if lang_l in DIAGRAM_LANGS:
pass # v1: source listing only
st.lst_n += 1
num = str(st.lst_n)
if cap_id:
st.labels[f"листинг:{cap_id}"] = num
if cap_text or cap_id:
bookmark = ""
if cap_id:
bookmark = f''
label = f"Листинг {num}"
if cap_text:
label = f"Листинг {num} — {cap_text}"
st.w(_p("Caption", bookmark + escape_text(label)))
for line in source.rstrip("\n").split("\n"):
spaced = escape_text(line)
def _spaces(m: re.Match) -> str:
n = len(m.group(0))
return "" if n == 1 else f''
spaced = re.sub(r" +", _spaces, spaced)
st.w(_p("Preformatted_20_Text", spaced if spaced else ""))
def _emit_list(st: EmitState, lst) -> None:
st.list_style_i += 1
style = "L1"
attrs = f' text:style-name="{style}"'
if lst.ordered:
# use continue numbering via list level number style — simplified
pass
st.w(f"")
for item in lst.children:
st.w("")
for ch in item.children:
name = type(ch).__name__
if name == "Paragraph":
st.w(_p("Standard", _inline(ch.children, st)))
elif name == "List":
_emit_list(st, ch)
else:
st.w(_p("Standard", _inline(getattr(ch, "children", []) or [], st)))
st.w("")
st.w("")
def _emit_equation(st: EmitState, eq) -> None:
cap_id = None
if st.pending_caption:
cap_id = st.pending_caption[0]
st.pending_caption = None
latex = eq.latex_equation.strip()
bookmark = ""
if cap_id:
bookmark = f''
st.w(_p("Centered", bookmark + f'{escape_text(latex)}'))
def emit_body(md_text: str, *, work_dir: Path, emdash_to_hyphen: bool = False) -> EmitState:
st = EmitState(work_dir=work_dir)
pre = preprocess_markdown(md_text, emdash_to_hyphen=emdash_to_hyphen)
# Two-pass: first collect captions/labels roughly by walking once for numbers,
# then emit. Simple approach: single pass (refs before definition show "?").
# Better: first pass assign numbers, second emit.
doc = markdown.parse(pre)
# Pass 1 — assign numbers for captions attached to tables/images/listings
pending = None
for el in doc.children:
name = type(el).__name__
if name == "Caption":
pending = (el.unique_name, el.text, getattr(el, "with_listing", False))
continue
if name == "Table" and pending:
st.tab_n += 1
st.labels[f"таблица:{pending[0]}"] = str(st.tab_n)
pending = None
elif name in ("FencedCode", "CodeBlock") and pending:
st.lst_n += 1
st.labels[f"листинг:{pending[0]}"] = str(st.lst_n)
pending = None
elif name == "Paragraph" and pending:
if any(type(c).__name__ == "Image" for c in (el.children or [])):
for c in el.children or []:
if type(c).__name__ == "Image":
st.fig_n += 1
cid = pending[0]
title = getattr(c, "title", "") or ""
if title.startswith("%"):
cid = title[1:].split(None, 1)[0]
st.labels[f"рисунок:{cid}"] = str(st.fig_n)
pending = None
elif name == "Equation" and pending:
pending = None
# reset counters for emit
st.fig_n = 0
st.tab_n = 0
st.lst_n = 0
st.pending_caption = None
for el in doc.children:
name = type(el).__name__
if name == "BlankLine":
continue
if name == "Caption":
st.pending_caption = (
el.unique_name,
el.text,
getattr(el, "with_listing", False),
)
continue
if name == "Heading":
_emit_heading(st, el)
continue
if name == "Paragraph":
_emit_paragraph(st, el)
continue
if name == "Table":
cid = ct = None
if st.pending_caption:
cid, ct, _ = st.pending_caption
st.pending_caption = None
_emit_table(st, el, cid, ct)
continue
if name in ("FencedCode", "CodeBlock"):
lang = getattr(el, "lang", "") or ""
_emit_code(st, el, lang)
continue
if name == "List":
_emit_list(st, el)
continue
if name == "Equation":
_emit_equation(st, el)
continue
if name == "TOC":
continue
if name == "Quote":
for ch in el.children:
if type(ch).__name__ == "Paragraph":
st.w(_p("Standard", _inline(ch.children, st)))
continue
if hasattr(el, "children"):
t = _inline(el.children, st).strip()
if t:
st.w(_p("Standard", t))
if st.biblio:
st.w(_h(1, escape_text("СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ")))
for key, txt in st.biblio:
st.w(_p("Standard", escape_text(f"[{key}] {txt}")))
return st
def emit_fodt(md_text: str, *, work_dir: Path, emdash_to_hyphen: bool = False) -> str:
st = emit_body(md_text, work_dir=work_dir, emdash_to_hyphen=emdash_to_hyphen)
body = "\n".join(st.body)
return (
'\n'
f"\n"
" \n"
" md2fodt\n"
" md2fodt\n"
" \n"
f"{styles_xml()}"
f"{automatic_styles_xml()}"
" \n"
" \n"
f"{body}\n"
" \n"
" \n"
"\n"
)