"""Walk DOCX body and produce IR blocks.""" from __future__ import annotations import re from typing import Callable from docx import Document from docx.oxml.ns import qn from docx.table import Table as DocxTable from docx.text.paragraph import Paragraph as DocxParagraph from .blocks import ( Block, CaptionBlock, EquationBlock, HeadingBlock, ImageBlock, InlineSpan, ListBlock, ListingBlock, PageBreakBlock, ParagraphBlock, SkipNote, TableBlock, TocPlaceholder, ) from .captions import caption_unique_id, parse_any_caption from .classify import ( caption_kind_from_style, classify_heading_text, heading_level_from_style, is_bibliography_style, is_body_start_heading, is_caption_style, is_code_style, is_special_title, is_toc_style, look_like_list_item, parse_biblio_line, paragraph_has_numpr, style_name, ) from .media import MediaExtractor, iter_blip_rids from .omml import find_omml, omml_to_latex, paragraph_has_omml from .tables import formula_table_parts, is_formula_table, table_to_grid from .text import ( hyperlink_element_text, paragraph_plain, paragraph_raw, run_element_text, ) LogFn = Callable[[str], None] def _para_plain(paragraph: DocxParagraph) -> str: return paragraph_plain(paragraph) def _iter_body_items(document: Document): body = document.element.body for child in body.iterchildren(): tag = child.tag if tag == qn("w:p"): yield "p", DocxParagraph(child, document) elif tag == qn("w:tbl"): yield "tbl", DocxTable(child, document) elif tag == qn("w:sectPr"): continue else: yield "other", child def _inline_spans(paragraph: DocxParagraph) -> list[InlineSpan]: """Build inline spans from runs + hyperlinks + inline math.""" spans: list[InlineSpan] = [] p = paragraph._element for child in p: tag = child.tag if tag == qn("w:r"): text = run_element_text(child) if child.find(qn("w:br")) is not None and not text.strip(): # lone page/line break run continue if not text: continue rPr = child.find(qn("w:rPr")) bold = italic = strike = False if rPr is not None: bold = rPr.find(qn("w:b")) is not None italic = rPr.find(qn("w:i")) is not None strike = rPr.find(qn("w:strike")) is not None or rPr.find(qn("w:dstrike")) is not None spans.append(InlineSpan(text=text, bold=bold, italic=italic, strike=strike)) elif tag == qn("w:hyperlink"): anchor = child.get(qn("w:anchor")) rid = child.get(qn("r:id")) href = None if rid: try: rel = paragraph.part.rels[rid] href = rel.target_ref except Exception: href = None inner = hyperlink_element_text(child) if not inner: continue # Internal biblio links → keep plain [n] text (no md link) if anchor and str(anchor).startswith("biblio_"): spans.append(InlineSpan(text=inner)) elif href and not str(href).startswith("#"): spans.append(InlineSpan(text=inner, href=href)) else: spans.append(InlineSpan(text=inner)) elif tag.endswith("}oMath") or tag.endswith("}oMathPara"): latex = omml_to_latex(child) if latex: spans.append(InlineSpan(text="", math=latex)) # Fallback if empty but paragraph has text if not spans: plain = _para_plain(paragraph) if plain: spans.append(InlineSpan(text=plain)) return spans def _section_landscape(document: Document, element) -> bool: """Best-effort: check nearest following sectPr or document default.""" try: # Walk forward siblings for sectPr parent = element.getparent() if parent is None: return False el = element.getnext() while el is not None: if el.tag == qn("w:sectPr"): pgSz = el.find(qn("w:pgSz")) if pgSz is not None: w = int(pgSz.get(qn("w:w") or "0") or 0) h = int(pgSz.get(qn("w:h") or "0") or 0) orient = pgSz.get(qn("w:orient")) if orient == "landscape" or (w and h and w > h): return True return False el = el.getnext() # last section sect = document.sections[-1] return bool(sect.page_width and sect.page_height and sect.page_width > sect.page_height) except Exception: return False class DocxWalker: def __init__( self, document: Document, media: MediaExtractor, *, keep_toc_pages: bool = False, emit_page_breaks: bool = False, log: LogFn | None = None, ): self.document = document self.media = media self.keep_toc_pages = keep_toc_pages self.emit_page_breaks = emit_page_breaks self.log = log or (lambda _m: None) self._started = False self._saw_toc_heading = False self._pending_caption: CaptionBlock | None = None self._code_buf: list[str] = [] self._code_caption: CaptionBlock | None = None self._list_buf: list[str] | None = None self._list_ordered: bool | None = None def walk(self) -> list[Block]: blocks: list[Block] = [] for kind, item in _iter_body_items(self.document): if kind == "p": blocks.extend(self._handle_paragraph(item)) elif kind == "tbl": self._flush_code(blocks) self._flush_list(blocks) blocks.extend(self._handle_table(item)) else: self.log(f"Пропуск неизвестного элемента body: {getattr(item, 'tag', item)}") blocks.append(SkipNote(message=f"unknown body child {getattr(item, 'tag', '')}")) self._flush_code(blocks) self._flush_list(blocks) return [b for b in blocks if not isinstance(b, SkipNote)] def _flush_code(self, blocks: list[Block]) -> None: if not self._code_buf: self._code_caption = None return cap = self._code_caption uid = cap.unique_id if cap else None text = cap.text if cap else None blocks.append( ListingBlock( lines=list(self._code_buf), language="", caption_id=uid, caption_text=text, ) ) self._code_buf.clear() self._code_caption = None self._pending_caption = None def _flush_list(self, blocks: list[Block]) -> None: if self._list_buf is None: return blocks.append( ListBlock( ordered=bool(self._list_ordered), items=list(self._list_buf), ) ) self._list_buf = None self._list_ordered = None def _handle_paragraph(self, paragraph: DocxParagraph) -> list[Block]: out: list[Block] = [] style = style_name(paragraph) text = _para_plain(paragraph) el = paragraph._element # Page break if self.emit_page_breaks: for br in el.findall(".//" + qn("w:br")): if br.get(qn("w:type")) == "page": self._flush_code(out) self._flush_list(out) out.append(PageBreakBlock()) # Skip until body start (title / assignment) if not self._started: if is_body_start_heading(text, style) or heading_level_from_style(style): if is_body_start_heading(text, style) or heading_level_from_style(style) == 1: self._started = True elif not text: return out else: return out else: return out # TOC field lines if is_toc_style(style): if not self.keep_toc_pages: return out # keep as plain paragraphs if requested self._flush_code(out) self._flush_list(out) out.append(ParagraphBlock(spans=_inline_spans(paragraph), style=style)) return out # Native TOC field detection if self._has_toc_field(el): self._flush_code(out) self._flush_list(out) if not self.keep_toc_pages: # only placeholder once after СОДЕРЖАНИЕ heading return out return out level = heading_level_from_style(style) if level is not None or (text and is_special_title(text) and not text.startswith("[")): self._flush_code(out) self._flush_list(out) if not text: return out lvl, title, numbered = classify_heading_text(text, level or 1) if title.upper() == "СОДЕРЖАНИЕ" or title.upper().startswith("СОДЕРЖАНИЕ"): out.append(HeadingBlock(level=1, text="СОДЕРЖАНИЕ", numbered=False)) out.append(TocPlaceholder()) self._saw_toc_heading = True self._pending_caption = None return out out.append(HeadingBlock(level=lvl, text=title, numbered=numbered)) self._pending_caption = None return out # Caption styles / heuristic captions if is_caption_style(style) or parse_any_caption(text): parsed = parse_any_caption(text) kind = caption_kind_from_style(style) if is_caption_style(style) else None if parsed is None and kind: # style known but text odd — treat whole text as title parsed = parse_any_caption(text) # may still fail if parsed is not None: self._flush_code(out) self._flush_list(out) if kind and parsed.kind != kind and kind != "figure": # prefer text kind when style is generic Caption pass uid = caption_unique_id(parsed.kind, parsed.number) cap = CaptionBlock( kind=parsed.kind, number=parsed.number, text=parsed.text, is_continuation=parsed.is_continuation, unique_id=uid, ) if parsed.is_continuation: # Emit marker for postprocess_blocks to splice into previous table/listing. # Do NOT stash in _pending_caption — that created a second %tbl / table. self._pending_caption = None out.append(cap) return out # Tables/listings: caption precedes object → keep pending. # Figures (GOST): caption is under the picture → only CaptionBlock; # postprocess folds Image+Caption. Stashing pending would glue this # caption onto the *next* figure. if parsed.kind in ("table", "listing"): self._pending_caption = cap else: self._pending_caption = None out.append(cap) return out # Code / listing lines if is_code_style(style): self._flush_list(out) if self._pending_caption and self._pending_caption.kind == "listing": self._code_caption = self._pending_caption self._pending_caption = None self._code_buf.append(paragraph_raw(paragraph)) return out # Images in paragraph rids = iter_blip_rids(el) if rids: self._flush_code(out) self._flush_list(out) # Caption-before-image is rare; GOST puts caption under the picture # (folded later). Only consume pending if it is explicitly a figure. cap = self._pending_caption cid = text_c = None if cap and cap.kind == "figure": cid = cap.unique_id text_c = cap.text self._pending_caption = None for rid in rids: rel = self.media.extract_rid(paragraph.part, rid) if not rel: self.log(f"Не удалось извлечь картинку {rid}") continue out.append( ImageBlock( rel_path=rel, alt=text or "", caption_id=cid, caption_text=text_c, ) ) cid = text_c = None # only first drawing gets pre-caption return out # Bibliography if is_bibliography_style(style) or parse_biblio_line(text): self._flush_code(out) self._flush_list(out) biblio = parse_biblio_line(text) if biblio: key, body = biblio out.append( ParagraphBlock( spans=[InlineSpan(text=f"[{key}]: {body}")], style=style, is_bibliography=True, biblio_key=key, ) ) return out # Lists (Word numPr or md2gost markers) if paragraph_has_numpr(paragraph): self._flush_code(out) body = text # strip auto number text if present at start — keep as item body ordered = True try: # python-docx doesn't expose easily; assume numbered if numPr m = re.match(r"^\d+\.\s*(.*)$", text) if m: body = m.group(1) else: # bullet ordered = False body = re.sub(r"^[–—\-•▪]\s*", "", text) except Exception: pass if self._list_buf is not None and self._list_ordered == ordered: self._list_buf.append(body) else: self._flush_list(out) self._list_buf = [body] self._list_ordered = ordered return [] list_hit = look_like_list_item(text) if list_hit and (paragraph.paragraph_format.left_indent or "\t" in (paragraph.text or "")): ordered, _ok, body = list_hit self._flush_code(out) if self._list_buf is not None and self._list_ordered == ordered: self._list_buf.append(body) else: self._flush_list(out) self._list_buf = [body] self._list_ordered = ordered return [] # Plain paragraph (maybe with inline math) self._flush_code(out) self._flush_list(out) if not text and not paragraph_has_omml(el): return out spans = _inline_spans(paragraph) # If whole paragraph is only omml (block-ish), emit equation oms = find_omml(el) if oms and not text.replace(" ", ""): latex = omml_to_latex(oms[0]) if latex: out.append(EquationBlock(latex=latex)) return out out.append(ParagraphBlock(spans=spans, style=style)) self._pending_caption = None return out def _handle_table(self, table: DocxTable) -> list[Block]: out: list[Block] = [] if not self._started: # title-page layout tables — skip return out if is_formula_table(table): latex, number = formula_table_parts(table) uid = None if number: uid = caption_unique_id("figure", number).replace("fig", "eq") # better: eq + number uid = "eq" + str(number).replace(".", "_") out.append(EquationBlock(latex=latex or "?", number=number, unique_id=uid)) self._pending_caption = None return out landscape = _section_landscape(self.document, table._tbl) rows, merges = table_to_grid(table) if not rows: return out cap = self._pending_caption cid = text_c = None if cap and cap.kind == "table" and not cap.is_continuation: cid = cap.unique_id text_c = cap.text landscape = landscape or cap.landscape self._pending_caption = None elif cap and cap.is_continuation: self._pending_caption = None out.append( TableBlock( rows=rows, merges=merges, caption_id=cid, caption_text=text_c, landscape=landscape, ) ) return out @staticmethod def _has_toc_field(p_element) -> bool: for instr in p_element.findall(".//" + qn("w:instrText")): raw = (instr.text or "").upper() if "TOC" in raw: return True return False def postprocess_blocks(blocks: list[Block]) -> list[Block]: """Merge continuation tables/listings; fold figure captions under images; drop dup TOC.""" result: list[Block] = [] saw_toc = False i = 0 while i < len(blocks): b = blocks[i] if isinstance(b, TocPlaceholder): if saw_toc: i += 1 continue saw_toc = True result.append(b) i += 1 continue if isinstance(b, CaptionBlock) and b.is_continuation: kind = b.kind i += 1 while i < len(blocks) and isinstance(blocks[i], SkipNote): i += 1 if i >= len(blocks): break nxt = blocks[i] prev = _last_block_of_type( result, TableBlock if kind == "table" else ListingBlock if kind == "listing" else None, ) if kind == "table" and isinstance(nxt, TableBlock) and isinstance(prev, TableBlock): _merge_table_continuation(prev, nxt) i += 1 continue if kind == "listing" and isinstance(nxt, ListingBlock) and isinstance(prev, ListingBlock): prev.lines.extend(nxt.lines) i += 1 continue result.append(nxt) i += 1 continue if isinstance(b, CaptionBlock) and not b.is_continuation: result.append(b) i += 1 continue result.append(b) i += 1 return _fold_figure_captions_after_images(result) def _fold_figure_captions_after_images(blocks: list[Block]) -> list[Block]: """ In DOCX/GOST the caption is under the picture; in md2gost %id / title must precede the image. Fold «Image + Caption(figure)» into one ImageBlock. """ out: list[Block] = [] i = 0 while i < len(blocks): b = blocks[i] nxt = blocks[i + 1] if i + 1 < len(blocks) else None if ( isinstance(b, ImageBlock) and isinstance(nxt, CaptionBlock) and nxt.kind == "figure" and not nxt.is_continuation ): b.caption_id = b.caption_id or nxt.unique_id b.caption_text = b.caption_text or nxt.text out.append(b) i += 2 continue # Drop a lone figure caption that duplicates the previous image's id if ( isinstance(b, CaptionBlock) and b.kind == "figure" and not b.is_continuation and out and isinstance(out[-1], ImageBlock) and out[-1].caption_id == b.unique_id ): if not out[-1].caption_text and b.text: out[-1].caption_text = b.text i += 1 continue out.append(b) i += 1 return out def _last_block_of_type(blocks: list[Block], typ) -> Block | None: if typ is None: return None for b in reversed(blocks): if isinstance(b, typ): return b return None def _merge_table_continuation(prev: TableBlock, nxt: TableBlock) -> None: """Append continuation rows; drop a repeated header row if Word duplicated it.""" rows = list(nxt.rows or []) merges = list(nxt.merges) if nxt.merges is not None else None if prev.rows and rows and prev.rows[0] == rows[0]: rows = rows[1:] if merges is not None: merges = merges[1:] prev.rows.extend(rows) if prev.merges is not None and merges is not None: prev.merges.extend(merges) elif prev.merges is None and merges is not None and prev.rows: # keep merges only if both sides had them; otherwise drop pass