Files
md_to_gost/md2gost/biblio_processor.py
Igor20264 510f7e7adf
Python application / build (push) Waiting to run
v0.5.2
Что то сделал
2026-09-08 19:37:54 +03:00

170 lines
5.9 KiB
Python

"""Convert [n]: source lines under bibliography heading into Bibliography paragraphs."""
from __future__ import annotations
import re
from .bibliography import BIBLIO_LINE_RE, BiblioEntry, extract_year
from .bibliography_renderable import Bibliography
from .renderable import Renderable
from .renderable.heading import Heading
from .renderable.paragraph import Paragraph
BIBLIO_HEADING = re.compile(
r"СПИСОК\s+ИСПОЛЬЗОВАНН?ЫХ\s+ИСТОЧНИКОВ",
re.IGNORECASE,
)
APPENDIX_HEADING = re.compile(r"^ПРИЛОЖЕН", re.IGNORECASE)
# Marko merges consecutive [n]: lines into one paragraph (soft breaks ignored),
# so we must find every entry inside the paragraph text.
BIBLIO_FIND_RE = re.compile(
r"\[(\d+(?:\.\d+)?)\]:\s*(.*?)(?=\s*\[\d+(?:\.\d+)?\]:|\s*$)",
re.DOTALL,
)
def entries_from_text(text: str) -> list[BiblioEntry]:
"""Pull all [n]: … entries from a (possibly concatenated) paragraph."""
entries: list[BiblioEntry] = []
if not text or not text.strip():
return entries
for m in BIBLIO_FIND_RE.finditer(text.strip()):
key = m.group(1)
body = re.sub(r"\s+", " ", m.group(2)).strip()
if not body:
continue
entries.append(BiblioEntry(key=key, text=body, year=extract_year(body)))
return entries
# After citation linking, `[5]: …` often becomes `[]: …` in paragraph.text
# (the digits live in a hyperlink). Still a biblio source line to drop.
BIBLIO_RESIDUE_RE = re.compile(r"^\[\d*(?:\.\d+)?\]:\s*\S")
def is_biblio_source_paragraph(text: str) -> bool:
"""True if paragraph is a [n]: / []: bibliography source line (or empty)."""
stripped = (text or "").strip()
if not stripped:
return True
if entries_from_text(stripped):
return True
return bool(BIBLIO_RESIDUE_RE.match(stripped))
def extract_bibliography_from_markdown(md: str) -> list[BiblioEntry] | list[tuple[str, list[BiblioEntry]]]:
"""
Parse bibliography from raw markdown (reliable even if Marko merges lines).
Returns either a flat list of entries, or a list of (section_title, entries) for VKR.
"""
m = re.search(
r"^#\s*\*?\s*СПИСОК\s+ИСПОЛЬЗОВАНН?ЫХ\s+ИСТОЧНИКОВ\s*$",
md,
re.M | re.I,
)
if not m:
return []
start = m.end()
rest = md[start:]
next_h = re.search(r"^#\s+", rest, re.M)
block = rest[: next_h.start()] if next_h else rest
sections: list[tuple[str, list[BiblioEntry]]] = []
current_title: str | None = None
current: list[BiblioEntry] = []
flat: list[BiblioEntry] = []
for line in block.splitlines():
hm = re.match(r"^#{2,6}\s+(\*?)(.+)$", line)
if hm:
title = hm.group(2).strip()
if current_title is not None:
sections.append((current_title, current))
elif current:
flat.extend(current)
current_title = title
current = []
continue
entries = entries_from_text(line)
if entries:
if current_title is not None:
current.extend(entries)
else:
flat.extend(entries)
continue
# concatenated line with several [n]:
if "[" in line and "]:" in line:
more = entries_from_text(line)
if current_title is not None:
current.extend(more)
else:
flat.extend(more)
if current_title is not None:
sections.append((current_title, current))
elif current:
flat.extend(current)
if sections:
return sections
return flat
def fold_bibliography(renderables: list[Renderable], parent,
raw_markdown: str | None = None) -> list[Renderable]:
"""Replace bibliography source paragraphs with a Bibliography renderable."""
# Prefer raw markdown extraction (handles Marko soft-break merge)
raw_entries = None
if raw_markdown:
raw_entries = extract_bibliography_from_markdown(raw_markdown)
result: list[Renderable] = []
i = 0
while i < len(renderables):
r = renderables[i]
result.append(r)
if isinstance(r, Heading) and BIBLIO_HEADING.search(r.text or ""):
i += 1
# Skip / drop following paragraphs that are biblio lines (already extracted)
consumed = 0
while i + consumed < len(renderables):
item = renderables[i + consumed]
if isinstance(item, Heading):
title = (item.text or "").strip()
if APPENDIX_HEADING.match(title):
break
if item.level == 1:
break
# subsection inside biblio — skip heading, content comes from raw
consumed += 1
continue
if isinstance(item, Paragraph):
text = item._docx_paragraph.text or ""
if is_biblio_source_paragraph(text):
consumed += 1
continue
break
break
if raw_entries:
if raw_entries and isinstance(raw_entries[0], tuple):
result.append(Bibliography(parent, [], sections=raw_entries)) # type: ignore
else:
result.append(Bibliography(parent, raw_entries)) # type: ignore
else:
# Fallback: parse from renderable paragraphs
entries: list[BiblioEntry] = []
for j in range(consumed):
item = renderables[i + j]
if isinstance(item, Paragraph):
entries.extend(entries_from_text(item._docx_paragraph.text or ""))
if entries:
result.append(Bibliography(parent, entries))
i += consumed
continue
i += 1
return result