Files
md_to_gost/md2gost/checker/__init__.py
T
Igor20264 b38661f588
Python application / build (push) Has been cancelled
Update 0.4.0
- Add\Rework UI
- Add Split Table and Listing
- Add Support Customazeble schems
2026-09-04 22:28:39 +03:00

523 lines
22 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""MIREA TZ markdown checker."""
from __future__ import annotations
import re
from dataclasses import dataclass
from typing import Iterable
from ..bibliography import CITE_RE, BIBLIO_LINE_RE, parse_biblio_block, find_citations, extract_year
from ..profiles import get_profile, current_year, DocProfile
@dataclass
class Issue:
id: str
severity: str # error | warning
message: str
line: int | None = None
SPECIAL_UNNUMBERED = {
"СОДЕРЖАНИЕ", "ВВЕДЕНИЕ", "ЗАКЛЮЧЕНИЕ",
"СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ",
"СПИСОК ИСПОЛЬЗУЕМЫХ ИСТОЧНИКОВ",
"ПРИЛОЖЕНИЕ", "ПРИЛОЖЕНИЯ",
}
HEADING_RE = re.compile(r"^(#{1,6})\s+(\*?)(.+)$", re.M)
FOOTNOTE_RE = re.compile(r"\[\^[^\]]+\]")
RIS_RE = re.compile(r"(?i)\bрис\.")
TABL_RE = re.compile(r"(?i)\bтабл\.")
CAPTION_HYPHEN_RE = re.compile(r"^%\w+\s+.+\s+-\s+", re.M)
NOPP_RE = re.compile(r"\s*п\s*/\s*п", re.I)
IMAGE_RE = re.compile(r"!\[.*?\]\([^)]+\)")
TABLE_CAPTION_RE = re.compile(r"^%(\w+)", re.M)
APPENDIX_LETTER_RE = re.compile(r"(?i)приложение\s+([А-ЯA-ZЁ])")
FORBIDDEN_APPENDIX = set("ЁЗЙОЧЬЫЪ")
def _line_of(text: str, pos: int) -> int:
return text.count("\n", 0, pos) + 1
def _sections(text: str) -> list[tuple[str, str, int]]:
"""Return list of (heading_upper, body, start_line)."""
parts = []
matches = list(HEADING_RE.finditer(text))
for i, m in enumerate(matches):
start = m.end()
end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
title = m.group(3).strip()
parts.append((title.upper(), text[start:end], _line_of(text, m.start())))
return parts
PRACTICAL_H1_RE = re.compile(
r"^#\s+(?!\*)(.+)$",
re.M,
)
PRACTICAL_TITLE_RE = re.compile(
r"(?i)практическая\s+работа\s*№?\s*\d+",
)
def check_structure(text: str, profile: DocProfile) -> list[Issue]:
issues = []
upper = text.upper()
if profile.require_practical_works:
# PIS_custom: содержание + практические работы как H1
if "СОДЕРЖАНИЕ" not in upper:
issues.append(Issue(
"structure.missing", "error",
"Отсутствует раздел «СОДЕРЖАНИЕ»",
))
if "[TOC]" not in text and "[toc]" not in text:
issues.append(Issue(
"structure.toc", "error",
"Нет маркера [TOC] для оглавления (обязательно для итогового отчёта)",
))
h1_titles = [m.group(1).strip() for m in PRACTICAL_H1_RE.finditer(text)]
practicals = [t for t in h1_titles if PRACTICAL_TITLE_RE.search(t)]
if not practicals:
issues.append(Issue(
"structure.practical", "error",
"Нет заголовков вида «Практическая работа №N. …» "
"(H1 без *, каждая работа — отдельный раздел)",
))
for t in practicals:
if t.rstrip().endswith("."):
issues.append(Issue(
"heading.trailing_dot", "warning",
f"Заголовок раздела без точки в конце: «{t.rstrip('.')}»",
))
return issues
required = ["СОДЕРЖАНИЕ", "ВВЕДЕНИЕ", "ЗАКЛЮЧЕНИЕ", "СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ"]
if not profile.require_intro_conclusion:
required = ["СОДЕРЖАНИЕ"]
if profile.require_bibliography and "СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ" not in required:
required.append("СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ")
# allow alternate spelling
if "СПИСОК ИСПОЛЬЗОВАННЫХ ИСТОЧНИКОВ" not in upper and "СПИСОК ИСПОЛЬЗУЕМЫХ ИСТОЧНИКОВ" in upper:
required = [
("СПИСОК ИСПОЛЬЗУЕМЫХ ИСТОЧНИКОВ" if r.startswith("СПИСОК") else r)
for r in required
]
for name in required:
if name not in upper:
issues.append(Issue("structure.missing", "error",
f"Отсутствует обязательный раздел «{name}»"))
if "[TOC]" not in text and "[toc]" not in text:
issues.append(Issue("structure.toc", "warning",
"Нет маркера [TOC] для автособираемого содержания"))
# numbered chapter exists
if not re.search(r"^#\s+[^*\n]", text, re.M):
issues.append(Issue("structure.body", "error",
"Нет нумерованного раздела основной части (# заголовок без *)"))
if profile.require_graphic_appendix:
if "ГРАФИЧЕСКИЙ МАТЕРИАЛ" not in upper:
issues.append(Issue(
"structure.graphic", "error",
"Для ВКР обязательно приложение «Графический материал»",
))
if profile.require_apid_kr_structure:
issues.extend(check_apid_kr_structure(text))
return issues
APID_REQUIRED_HEADING_PATTERNS = [
(r"теоретическ\w+\s+аспект", "1. Теоретические аспекты разработки архитектуры приложений и данных: …"),
(r"прикладн\w+\s+аспект", "2. Прикладные аспекты разработки архитектуры приложений и данных …"),
(r"2\.1\b.{0,40}описани[ея]\s+проекта", "2.1 Описание проекта команды и разрабатываемого программного приложения"),
(r"2\.2\b.{0,40}описани[ея]\s+роли", "2.2 Описание роли … в приложении …"),
(r"2\.3\b.{0,80}архитектур", "2.3 Описание архитектуры программного приложения и данных для роли"),
(r"2\.4\b.{0,40}вариант", "2.4 Варианты развития архитектуры программного приложения"),
]
def check_apid_kr_structure(text: str) -> list[Issue]:
"""Обязательные разделы КР по методичке АПИД (Аншина, Лагунова)."""
issues: list[Issue] = []
headings = [m.group(3).strip() for m in HEADING_RE.finditer(text)]
blob = "\n".join(headings)
for pat, title in APID_REQUIRED_HEADING_PATTERNS:
if not re.search(pat, blob, re.I):
issues.append(Issue(
"structure.apid", "error",
f"Для КР АПИД отсутствует обязательный раздел: «{title}»",
))
return issues
def check_pis_headings(text: str, profile: DocProfile) -> list[Issue]:
"""PIS: подразделы с прописной, без точки; H1 — практические работы."""
if not profile.require_practical_works:
return []
issues = []
for m in HEADING_RE.finditer(text):
level = len(m.group(1))
title = m.group(3).strip()
line = _line_of(text, m.start())
if title.endswith("."):
issues.append(Issue(
"heading.trailing_dot", "warning",
f"Заголовок без точки в конце: «{title.rstrip('.')}»",
line,
))
if level >= 2 and title and title[0].islower():
issues.append(Issue(
"heading.case", "warning",
f"Подраздел с прописной буквы: «{title[0].upper() + title[1:]}»",
line,
))
return issues
def check_special_headings(text: str) -> list[Issue]:
issues = []
for m in HEADING_RE.finditer(text):
starred = bool(m.group(2))
title = m.group(3).strip()
upper = title.upper()
line = _line_of(text, m.start())
for special in SPECIAL_UNNUMBERED:
if upper == special or (special.startswith("ПРИЛОЖЕН") and upper.startswith("ПРИЛОЖЕН")):
if not starred and m.group(1) == "#":
issues.append(Issue(
"heading.numbered_special", "error",
f"Спецраздел «{title}» должен быть без номера: # *{title}",
line,
))
if title != title.upper() and upper in SPECIAL_UNNUMBERED:
issues.append(Issue(
"heading.case", "warning",
f"Спецраздел лучше писать ПРОПИСНЫМИ: «{upper}»",
line,
))
return issues
def check_forbidden_abbreviations(text: str) -> list[Issue]:
issues = []
for m in RIS_RE.finditer(text):
issues.append(Issue("ref.ris", "error",
"Используйте «Рисунок», не «рис.»", _line_of(text, m.start())))
for m in TABL_RE.finditer(text):
issues.append(Issue("ref.tabl", "error",
"Используйте «Таблица», не «табл.»", _line_of(text, m.start())))
for m in FOOTNOTE_RE.finditer(text):
issues.append(Issue("footnote", "error",
"Сноски в конце страницы не допускаются", _line_of(text, m.start())))
for m in NOPP_RE.finditer(text):
issues.append(Issue("table.nopp", "error",
"Графу «№ п/п» в таблицы не включают", _line_of(text, m.start())))
return issues
def check_citations_in_intro_conclusion(text: str) -> list[Issue]:
issues = []
for title, body, line in _sections(text):
if title in ("ВВЕДЕНИЕ", "ЗАКЛЮЧЕНИЕ"):
for m in CITE_RE.finditer(body):
issues.append(Issue(
"cite.intro", "error",
f"Во разделе «{title}» ссылки на источники не указываются",
line + body[:m.start()].count("\n"),
))
return issues
def check_bibliography(text: str, profile: DocProfile) -> list[Issue]:
issues = []
if profile.min_sources <= 0 and not profile.require_bibliography:
# PIS_custom и др.: список источников не обязателен
m_probe = re.search(
r"^#\s*\*?\s*СПИСОК\s+ИСПОЛЬЗОВАНН?ЫХ\s+ИСТОЧНИКОВ\s*$",
text, re.M | re.I,
)
if not m_probe:
return issues
# Extract biblio block
m = re.search(
r"^#\s*\*?\s*СПИСОК\s+ИСПОЛЬЗОВАНН?ЫХ\s+ИСТОЧНИКОВ\s*$",
text, re.M | re.I,
)
if not m:
return issues
start = m.end()
next_h = re.search(r"^#\s+", text[start:], re.M)
block = text[start: start + next_h.start()] if next_h else text[start:]
entries = []
for line in block.splitlines():
bm = BIBLIO_LINE_RE.match(line.strip())
if bm:
entries.append((bm.group(1), bm.group(2), extract_year(bm.group(2))))
n = len(entries)
if profile.sectional_biblio:
# Count per ## section inside bibliography
sections: list[tuple[str, int]] = []
current_title: str | None = None
current_count = 0
for line in block.splitlines():
hm = re.match(r"^#{2,6}\s+(\*?)(.+)$", line)
if hm:
if current_title is not None:
sections.append((current_title, current_count))
current_title = hm.group(2).strip()
current_count = 0
continue
if BIBLIO_LINE_RE.match(line.strip()):
current_count += 1
if current_title is not None:
sections.append((current_title, current_count))
if sections:
for title, cnt in sections:
if cnt < profile.min_sources:
issues.append(Issue(
"biblio.count", "warning",
f"ВКР: в разделе списка «{title}» желательно ≥{profile.min_sources} "
f"источников (сейчас {cnt})",
))
elif n < profile.min_sources:
issues.append(Issue(
"biblio.count", "warning",
f"ВКР: список должен делиться на разделы; в каждом ≥{profile.min_sources} "
f"источников (сейчас всего {n}, разделов нет)",
))
else:
if n < profile.min_sources:
issues.append(Issue(
"biblio.count", "error",
f"Источников должно быть от {profile.min_sources} до {profile.max_sources} "
f"(сейчас {n})",
))
elif profile.max_sources and n > profile.max_sources:
issues.append(Issue(
"biblio.count", "error",
f"Источников должно быть от {profile.min_sources} до {profile.max_sources} "
f"(сейчас {n})",
))
year_now = current_year()
for key, txt, year in entries:
if year and year < year_now - profile.max_source_age_years:
issues.append(Issue(
"biblio.age", "error",
f"Источник [{key}] старше {profile.max_source_age_years} лет (год {year})",
))
# Citation order vs first appearance
body_for_cites = text
cites = find_citations(body_for_cites)
keys = [e[0] for e in entries]
if cites and keys:
if not profile.sectional_biblio:
expected = cites[:len(keys)]
if keys != expected and set(keys) == set(expected):
issues.append(Issue(
"biblio.order", "warning",
"Порядок источников в списке должен совпадать с порядком первого упоминания",
))
for c in cites:
if c not in keys:
issues.append(Issue(
"biblio.missing", "error",
f"Ссылка [{c}] есть в тексте, но нет в списке источников",
))
return issues
def check_object_refs(text: str) -> list[Issue]:
issues = []
labels = set(re.findall(r"^%(\w+)", text, re.M))
labels.update(re.findall(r'!\[[^\]]*\]\([^)]*%(\w+)', text))
refs = set(re.findall(r"@[\wА-Яа-я]+:(\w+)", text))
for lab in labels:
if lab not in refs:
issues.append(Issue(
"ref.unused", "warning",
f"Метка «{lab}» объявлена, но в тексте нет ссылки @…:{lab}",
))
for ref in refs:
if ref not in labels:
issues.append(Issue(
"ref.dangling", "error",
f"Ссылка на несуществующую метку «{ref}»",
))
return issues
def check_appendices(text: str) -> list[Issue]:
issues = []
for m in APPENDIX_LETTER_RE.finditer(text):
letter = m.group(1).upper()
if letter in FORBIDDEN_APPENDIX:
issues.append(Issue(
"appendix.letter", "error",
f"Буква «{letter}» не используется для обозначения приложений",
_line_of(text, m.start()),
))
# Multiple lettered appendices should have a list after # *ПРИЛОЖЕНИЯ
apps = list(re.finditer(
r"^#{1,6}\s+\*?Приложение\s+[А-ЯA-ZЁ]\b",
text, re.M | re.I,
))
# Filter out the section header «ПРИЛОЖЕНИЯ» / bare «ПРИЛОЖЕНИЕ»
lettered = []
for m in apps:
line = m.group(0)
if re.search(r"(?i)приложение\s+[А-ЯA-ZЁ]", line):
lettered.append(m)
if len(lettered) >= 2:
sec = re.search(
r"^#\s*\*?\s*ПРИЛОЖЕНИЯ?\s*$",
text, re.M | re.I,
)
if sec:
between = text[sec.end(): lettered[0].start()]
# Expect a plain-text list (not only blank / headings)
has_list = bool(re.search(r"(?im)^\s*[-–—*]|\bприложение\s+[А-ЯA-ZЁ]\b", between))
has_prose = bool(re.search(r"[А-Яа-яA-Za-z]{3,}", between))
if not (has_list or has_prose):
issues.append(Issue(
"appendix.toc", "warning",
"При нескольких приложениях после «ПРИЛОЖЕНИЯ» нужен перечень "
"(основной текст) с номерами и названиями",
_line_of(text, sec.start()),
))
return issues
def check_continuation_hints(
text: str,
*,
table_continuation: str = "off",
listing_continuation: str = "off",
) -> list[Issue]:
"""Warn that Word cannot auto-insert «Продолжение…» when mode is off/soft."""
issues: list[Issue] = []
soft = {"off", "soft"}
if table_continuation in soft and _TABLE_BLOCK_RE.search(text):
issues.append(Issue(
"table.continuation", "warning",
"Word сам не вставит «Продолжение Таблицы N» при переносе. "
"Варианты: --table-continuation word (точный разрыв через Word COM) "
"или caption (оценка высоты ≠ вёрстка Word, возможны артефакты).",
))
# Long fenced code blocks (likely listings)
long_listing = False
for m in re.finditer(r"^```[^\n]*\n([\s\S]*?)^```", text, re.M):
if m.group(1).count("\n") >= 40:
long_listing = True
break
if listing_continuation in soft and long_listing:
issues.append(Issue(
"listing.continuation", "warning",
"Длинный листинг: Word сам не вставит «Продолжение Листинга N». "
"Варианты: --listing-continuation word или caption.",
))
return issues
_TABLE_BLOCK_RE = re.compile(
r"(?:^[ \t]*\|.+\|[ \t]*\n)+"
r"(?:^[ \t]*\|[-:| ]+\|[ \t]*\n)"
r"(?:^[ \t]*\|.+\|[ \t]*\n?)*",
re.M,
)
_MERGE_ONLY_RE = re.compile(r"^\s*(\^{1,2}|>{1,2})\s*$")
def _split_table_row(line: str) -> list[str]:
parts = re.split(r"\s*(?<!\\)\|\s*", line.strip())
if parts and not parts[0]:
parts.pop(0)
if parts and not parts[-1]:
parts.pop()
return parts
def check_table_merge(text: str) -> list[Issue]:
"""Validate ^ / > merge markers in pipe tables."""
issues: list[Issue] = []
for tm in _TABLE_BLOCK_RE.finditer(text):
block = tm.group(0)
block_start = tm.start()
lines = [ln for ln in block.splitlines() if ln.strip()]
if len(lines) < 2:
continue
header = _split_table_row(lines[0])
data_rows = [_split_table_row(ln) for ln in lines[2:]]
n_cols = len(header)
header_line = _line_of(text, block_start)
for cell in header:
if _MERGE_ONLY_RE.match(cell):
issues.append(Issue(
"table.merge_header", "error",
"Маркер склеивания ^/> нельзя использовать в первой (заголовочной) строке таблицы",
header_line,
))
break
for r_i, raw_cells in enumerate(data_rows):
line_no = header_line + 2 + r_i
cells = (raw_cells + [""] * n_cols)[:n_cols]
for c_i, cell in enumerate(cells):
m = _MERGE_ONLY_RE.match(cell)
if not m:
continue
if m.group(1).startswith(">") and c_i == 0:
issues.append(Issue(
"table.merge_col", "error",
"Маркер «>» нельзя ставить в первом столбце (нет ячейки слева)",
line_no,
))
return issues
def check_markdown(
text: str,
doc_type: str = "coursework",
*,
table_continuation: str = "off",
listing_continuation: str = "off",
) -> list[Issue]:
profile = get_profile(doc_type)
issues: list[Issue] = []
issues.extend(check_structure(text, profile))
issues.extend(check_special_headings(text))
issues.extend(check_pis_headings(text, profile))
issues.extend(check_forbidden_abbreviations(text))
if profile.require_intro_conclusion:
issues.extend(check_citations_in_intro_conclusion(text))
issues.extend(check_bibliography(text, profile))
issues.extend(check_object_refs(text))
issues.extend(check_appendices(text))
issues.extend(check_table_merge(text))
issues.extend(check_continuation_hints(
text,
table_continuation=table_continuation,
listing_continuation=listing_continuation,
))
return issues
def format_report(issues: Iterable[Issue]) -> str:
issues = list(issues)
if not issues:
return "Проверка ТЗ: замечаний нет."
lines = [f"Проверка ТЗ: найдено замечаний — {len(issues)}"]
for i in issues:
loc = f" (стр. md:{i.line})" if i.line else ""
lines.append(f" [{i.severity}] {i.id}{loc}: {i.message}")
return "\n".join(lines)