This commit is contained in:
2026-09-14 10:54:55 +03:00
parent 3eb0d24e7f
commit 458571576e
6 changed files with 307 additions and 51 deletions

View File

@@ -562,15 +562,33 @@ _HEADING_TOKEN_RE = re.compile(
r"(\d+)?$",
re.I,
)
# Word „Title“ / MsoTitle ≠ Heading 1 — корица, не граница на секция.
_COVER_TITLE_TOKENS = frozenset({
"title", "mstotitle", "наименование", "doctitle", "documenttitle",
})
_HTML_HEADING_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
r"(?:^|[\s_-])(?:heading|заглавие|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
re.I,
)
_HTML_COVER_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:mso)?title(?:\d+)?(?:$|[\s_-])|"
r"(?:^|[\s_-])наименование(?:\d+)?(?:$|[\s_-])",
re.I,
)
_HTML_SUBTITLE_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:subtitle|подзаглавие)(?:\d+)?(?:$|[\s_-])",
re.I,
)
_TOC_HEADING_RE = re.compile(
r"^(съдържание|съдържанието|contents|table of contents|toc|"
r"inhaltsverzeichnis|оглавление|содержание)\s*:?\s*$",
re.I,
)
_HEADING_LEVEL = {
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
"title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
"subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
"подзаглавие": 2, "наименование": 1,
"подзаглавие": 2,
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
}
@@ -579,10 +597,20 @@ def _compact_style_token(s: str) -> str:
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
def _is_cover_title_token(token: str) -> bool:
return _compact_style_token(token) in _COVER_TITLE_TOKENS
def _is_toc_heading(text: str) -> bool:
return bool(_TOC_HEADING_RE.match((text or "").strip()))
def _heading_level_from_token(token: str) -> Optional[int]:
t = _compact_style_token(token)
if not t:
return None
if _is_cover_title_token(t):
return None
if t in _HEADING_LEVEL:
return _HEADING_LEVEL[t]
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
@@ -594,21 +622,38 @@ def _heading_level_from_token(token: str) -> Optional[int]:
kind = (m.group(1) or "").lower()
if kind in ("subtitle", "подзаглавие"):
return 2
if kind in ("title", "наименование"):
return None
return 1
def _docx_heading_level(para) -> Optional[int]:
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word."""
def _docx_style_tokens(para) -> list[str]:
style = getattr(para, "style", None)
tokens: list[str] = []
seen: set[int] = set()
cur = style
while cur is not None and id(cur) not in seen:
seen.add(id(cur))
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
lvl = _heading_level_from_token(str(token or ""))
if lvl:
return lvl
if token:
tokens.append(str(token))
cur = getattr(cur, "base_style", None)
return tokens
def _docx_is_cover_title(para) -> bool:
return any(_is_cover_title_token(t) for t in _docx_style_tokens(para))
def _docx_heading_level(para) -> Optional[int]:
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word.
Word Title / Наименование не са граница (виж _docx_is_cover_title)."""
if _docx_is_cover_title(para):
return None
for token in _docx_style_tokens(para):
lvl = _heading_level_from_token(token)
if lvl:
return lvl
try:
pPr = para._element.pPr
if pPr is not None and pPr.outlineLvl is not None:
@@ -628,11 +673,28 @@ def _is_bold_heading_text(text: str, runs) -> bool:
return bool(useful) and all(bool(r.bold) for r in useful)
def _html_is_cover_title(el) -> bool:
raw = el.get("class") if hasattr(el, "get") else None
classes: list[str]
if isinstance(raw, str):
classes = raw.split()
else:
classes = list(raw or [])
for c in classes:
if _is_cover_title_token(c):
return True
return bool(_HTML_COVER_CLASS_RE.search(" ".join(classes)))
def _html_heading_level(el) -> Optional[int]:
name = (getattr(el, "name", None) or "").lower()
if name in _HTML_HEADING_MAP:
return _HTML_HEADING_MAP[name]
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
if _HTML_COVER_CLASS_RE.search(cls):
return None
if _HTML_SUBTITLE_CLASS_RE.search(cls):
return 2
m = _HTML_HEADING_CLASS_RE.search(cls)
if m:
n = m.group(1)
@@ -703,7 +765,7 @@ def parse_html(path: Path) -> list[Section]:
blocks = []
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
name = (el.name or "").lower()
if name == "div" and not _html_heading_level(el):
if name == "div" and not _html_heading_level(el) and not _html_is_cover_title(el):
continue
if any(id(par) in consumed for par in el.parents):
continue
@@ -725,12 +787,38 @@ def parse_html(path: Path) -> list[Section]:
sec.html_text = "\n".join(sec_html) if sec_html else None
sections.append(sec)
def append_block_as_body(el):
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
_strip_attrs(el)
txt = _html_block_plain_text(el)
if txt:
sec_text.append(txt)
try:
sec_html.append(str(el))
except Exception:
pass
for el in blocks:
txt = el.get_text(" ", strip=True)
is_cover = _html_is_cover_title(el)
heading_lvl = _html_heading_level(el)
# Корица (Word Title / class Title): заглавие на преамбюла, без нова секция
if is_cover and txt:
if not current_title and not sec_text and not sec_html and not sec_images:
current_title = txt
current_level = 1
else:
append_block_as_body(el)
continue
if heading_lvl:
txt = el.get_text(" ", strip=True)
if not txt:
continue
# TOC и H2+ остават в тялото — граница само при H1 / Заглавие 1
if _is_toc_heading(txt) or heading_lvl >= 2:
append_block_as_body(el)
continue
flush()
current_title = txt
current_level = heading_lvl
@@ -742,21 +830,13 @@ def parse_html(path: Path) -> list[Section]:
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
base_dir, sec_images, img_counter)
# ако е заменен с placeholder, добавяме като текст
txt = el.get_text(" ", strip=True) if el.name else ""
if txt:
sec_text.append(txt)
sec_html.append(f"<p>{txt}</p>")
img_txt = el.get_text(" ", strip=True) if el.name else ""
if img_txt:
sec_text.append(img_txt)
sec_html.append(f"<p>{img_txt}</p>")
continue
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
_strip_attrs(el)
txt = _html_block_plain_text(el)
if txt:
sec_text.append(txt)
try:
sec_html.append(str(el))
except Exception:
pass
append_block_as_body(el)
flush()
@@ -852,19 +932,55 @@ def parse_docx(path: Path) -> list[Section]:
if not text and not para_imgs:
continue
is_cover = _docx_is_cover_title(para)
level = _docx_heading_level(para)
is_bold_heading = (
not level
and not is_cover
and _is_bold_heading_text(text, para.runs)
and not style_name.startswith("list")
and not para_imgs
)
if level or is_bold_heading:
# Корица (Word Title): заглавие на преамбюла, без нова секция
if is_cover and text:
if not current_title and not buf and not sec_images:
current_title = text
current_level = 1
else:
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
continue
# TOC и H2+/bold остават в тялото — граница само при Heading 1 / Заглавие 1
if text and _is_toc_heading(text):
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
continue
if (level or 0) >= 2 or is_bold_heading:
if text:
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
continue
if level == 1:
flush()
buf, sec_images = [], []
current_title = text
current_level = level or 2
current_level = 1
continue
if text:
@@ -1268,6 +1384,29 @@ def merge_short_sections(sections: list[Section]) -> list[Section]:
return result
def merge_preamble_sections(sections: list[Section]) -> list[Section]:
"""Слива корица (Title) + „Съдържание“/TOC в един преамбюл, ако парсерът ги е разделил."""
if len(sections) < 2:
return sections
first, second = sections[0], sections[1]
first_words = len((first.text or "").split())
if not (first.title or "").strip():
return sections
if first_words >= MIN_SECTION_TOKENS:
return sections
if not _is_toc_heading(second.title or ""):
return sections
body_parts = [p for p in (second.title, second.text) if (p or "").strip()]
if (first.text or "").strip():
body_parts.append(first.text.strip())
merged = Section(first.title, "\n".join(body_parts).strip(), first.level)
merged.images = (first.images or []) + (second.images or [])
html_parts = [h for h in (first.html_text, second.html_text) if h]
merged.html_text = "\n".join(html_parts) if html_parts else None
return [merged] + list(sections[2:])
def clean_text(text: str) -> str:
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
text = text.replace("\r\n", "\n").replace("\r", "\n")
@@ -1504,6 +1643,7 @@ def process_file(
return _file_result(rel, file_index, existing_codes, saved=0)
sections = merge_short_sections(sections)
sections = merge_preamble_sections(sections)
remove_section_outputs(
output_dir,