170 lines
5.9 KiB
Python
170 lines
5.9 KiB
Python
"""Convert [n]: source lines under bibliography heading into Bibliography paragraphs."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import re
|
|
|
|
from .bibliography import BIBLIO_LINE_RE, BiblioEntry, extract_year
|
|
from .bibliography_renderable import Bibliography
|
|
from .renderable import Renderable
|
|
from .renderable.heading import Heading
|
|
from .renderable.paragraph import Paragraph
|
|
|
|
|
|
BIBLIO_HEADING = re.compile(
|
|
r"СПИСОК\s+ИСПОЛЬЗОВАНН?ЫХ\s+ИСТОЧНИКОВ",
|
|
re.IGNORECASE,
|
|
)
|
|
APPENDIX_HEADING = re.compile(r"^ПРИЛОЖЕН", re.IGNORECASE)
|
|
|
|
# Marko merges consecutive [n]: lines into one paragraph (soft breaks ignored),
|
|
# so we must find every entry inside the paragraph text.
|
|
BIBLIO_FIND_RE = re.compile(
|
|
r"\[(\d+(?:\.\d+)?)\]:\s*(.*?)(?=\s*\[\d+(?:\.\d+)?\]:|\s*$)",
|
|
re.DOTALL,
|
|
)
|
|
|
|
|
|
def entries_from_text(text: str) -> list[BiblioEntry]:
|
|
"""Pull all [n]: … entries from a (possibly concatenated) paragraph."""
|
|
entries: list[BiblioEntry] = []
|
|
if not text or not text.strip():
|
|
return entries
|
|
for m in BIBLIO_FIND_RE.finditer(text.strip()):
|
|
key = m.group(1)
|
|
body = re.sub(r"\s+", " ", m.group(2)).strip()
|
|
if not body:
|
|
continue
|
|
entries.append(BiblioEntry(key=key, text=body, year=extract_year(body)))
|
|
return entries
|
|
|
|
|
|
# After citation linking, `[5]: …` often becomes `[]: …` in paragraph.text
|
|
# (the digits live in a hyperlink). Still a biblio source line to drop.
|
|
BIBLIO_RESIDUE_RE = re.compile(r"^\[\d*(?:\.\d+)?\]:\s*\S")
|
|
|
|
|
|
def is_biblio_source_paragraph(text: str) -> bool:
|
|
"""True if paragraph is a [n]: / []: bibliography source line (or empty)."""
|
|
stripped = (text or "").strip()
|
|
if not stripped:
|
|
return True
|
|
if entries_from_text(stripped):
|
|
return True
|
|
return bool(BIBLIO_RESIDUE_RE.match(stripped))
|
|
|
|
|
|
def extract_bibliography_from_markdown(md: str) -> list[BiblioEntry] | list[tuple[str, list[BiblioEntry]]]:
|
|
"""
|
|
Parse bibliography from raw markdown (reliable even if Marko merges lines).
|
|
Returns either a flat list of entries, or a list of (section_title, entries) for VKR.
|
|
"""
|
|
m = re.search(
|
|
r"^#\s*\*?\s*СПИСОК\s+ИСПОЛЬЗОВАНН?ЫХ\s+ИСТОЧНИКОВ\s*$",
|
|
md,
|
|
re.M | re.I,
|
|
)
|
|
if not m:
|
|
return []
|
|
start = m.end()
|
|
rest = md[start:]
|
|
next_h = re.search(r"^#\s+", rest, re.M)
|
|
block = rest[: next_h.start()] if next_h else rest
|
|
|
|
sections: list[tuple[str, list[BiblioEntry]]] = []
|
|
current_title: str | None = None
|
|
current: list[BiblioEntry] = []
|
|
flat: list[BiblioEntry] = []
|
|
|
|
for line in block.splitlines():
|
|
hm = re.match(r"^#{2,6}\s+(\*?)(.+)$", line)
|
|
if hm:
|
|
title = hm.group(2).strip()
|
|
if current_title is not None:
|
|
sections.append((current_title, current))
|
|
elif current:
|
|
flat.extend(current)
|
|
current_title = title
|
|
current = []
|
|
continue
|
|
entries = entries_from_text(line)
|
|
if entries:
|
|
if current_title is not None:
|
|
current.extend(entries)
|
|
else:
|
|
flat.extend(entries)
|
|
continue
|
|
# concatenated line with several [n]:
|
|
if "[" in line and "]:" in line:
|
|
more = entries_from_text(line)
|
|
if current_title is not None:
|
|
current.extend(more)
|
|
else:
|
|
flat.extend(more)
|
|
|
|
if current_title is not None:
|
|
sections.append((current_title, current))
|
|
elif current:
|
|
flat.extend(current)
|
|
|
|
if sections:
|
|
return sections
|
|
return flat
|
|
|
|
|
|
def fold_bibliography(renderables: list[Renderable], parent,
|
|
raw_markdown: str | None = None) -> list[Renderable]:
|
|
"""Replace bibliography source paragraphs with a Bibliography renderable."""
|
|
# Prefer raw markdown extraction (handles Marko soft-break merge)
|
|
raw_entries = None
|
|
if raw_markdown:
|
|
raw_entries = extract_bibliography_from_markdown(raw_markdown)
|
|
|
|
result: list[Renderable] = []
|
|
i = 0
|
|
while i < len(renderables):
|
|
r = renderables[i]
|
|
result.append(r)
|
|
if isinstance(r, Heading) and BIBLIO_HEADING.search(r.text or ""):
|
|
i += 1
|
|
# Skip / drop following paragraphs that are biblio lines (already extracted)
|
|
consumed = 0
|
|
while i + consumed < len(renderables):
|
|
item = renderables[i + consumed]
|
|
if isinstance(item, Heading):
|
|
title = (item.text or "").strip()
|
|
if APPENDIX_HEADING.match(title):
|
|
break
|
|
if item.level == 1:
|
|
break
|
|
# subsection inside biblio — skip heading, content comes from raw
|
|
consumed += 1
|
|
continue
|
|
if isinstance(item, Paragraph):
|
|
text = item._docx_paragraph.text or ""
|
|
if is_biblio_source_paragraph(text):
|
|
consumed += 1
|
|
continue
|
|
break
|
|
break
|
|
|
|
if raw_entries:
|
|
if raw_entries and isinstance(raw_entries[0], tuple):
|
|
result.append(Bibliography(parent, [], sections=raw_entries)) # type: ignore
|
|
else:
|
|
result.append(Bibliography(parent, raw_entries)) # type: ignore
|
|
else:
|
|
# Fallback: parse from renderable paragraphs
|
|
entries: list[BiblioEntry] = []
|
|
for j in range(consumed):
|
|
item = renderables[i + j]
|
|
if isinstance(item, Paragraph):
|
|
entries.extend(entries_from_text(item._docx_paragraph.text or ""))
|
|
if entries:
|
|
result.append(Bibliography(parent, entries))
|
|
|
|
i += consumed
|
|
continue
|
|
i += 1
|
|
return result
|