This commit is contained in:
2026-09-14 11:10:05 +03:00
parent 458571576e
commit 181bbc3603
6 changed files with 242 additions and 68 deletions

View File

@@ -605,6 +605,76 @@ def _is_toc_heading(text: str) -> bool:
return bool(_TOC_HEADING_RE.match((text or "").strip()))
# Top-level chapter: "1. Title" / "2) Title" — short line, not body/table/figure.
_NUMBERED_CHAPTER_RE = re.compile(
r"^(\d{1,2})[\.\)]\s+(\S.{0,100})$"
)
_FIGURE_CAPTION_RE = re.compile(
r"^(фигура|figure|abb\.?|рис\.?)\s*\d+",
re.I,
)
def _normalize_chapter_key(text: str) -> str:
return re.sub(r"\s+", " ", (text or "").strip().lower())
def _is_numbered_chapter_heading(text: str) -> bool:
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
t = (text or "").strip()
if not t or len(t) >= 120:
return False
if "|" in t:
return False
if _is_toc_heading(t):
return False
if _FIGURE_CAPTION_RE.match(t):
return False
m = _NUMBERED_CHAPTER_RE.match(t)
if not m:
return False
rest = m.group(2).strip()
words = rest.split()
if not words or len(words) > 12:
return False
# Body-like numbered sentences: "1. Copy the files into the folder."
if rest.endswith((".", "!", "?")) and len(words) > 4:
return False
return True
class _TocSplitState:
"""Tracks Съдържание / TOC so first '1. Title' stays in preamble, second starts a section."""
__slots__ = ("phase", "had_toc", "titles")
def __init__(self) -> None:
self.phase = False
self.had_toc = False
self.titles: set[str] = set()
def note_toc_heading(self) -> None:
self.phase = True
self.had_toc = True
def leave_phase(self) -> None:
self.phase = False
def handle_numbered(self, text: str) -> str:
"""Return 'toc' (keep in body), 'split' (new section), or 'no' (not numbered chapter)."""
if not _is_numbered_chapter_heading(text):
return "no"
key = _normalize_chapter_key(text)
if self.phase:
if key in self.titles:
self.phase = False
return "split"
self.titles.add(key)
return "toc"
# After TOC list/block, or docs without TOC: numbered chapter opens a section.
return "split"
def _heading_level_from_token(token: str) -> Optional[int]:
t = _compact_style_token(token)
if not t:
@@ -779,6 +849,7 @@ def parse_html(path: Path) -> list[Section]:
sec_html: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
toc_state = _TocSplitState()
def flush():
if current_title or sec_text or sec_html or sec_images:
@@ -798,6 +869,13 @@ def parse_html(path: Path) -> list[Section]:
except Exception:
pass
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, sec_text, sec_html, sec_images
flush()
current_title = title
current_level = level
sec_text, sec_html, sec_images = [], [], []
for el in blocks:
txt = el.get_text(" ", strip=True)
is_cover = _html_is_cover_title(el)
@@ -809,23 +887,42 @@ def parse_html(path: Path) -> list[Section]:
current_title = txt
current_level = 1
else:
if toc_state.phase:
toc_state.leave_phase()
append_block_as_body(el)
continue
if txt and _is_toc_heading(txt):
toc_state.note_toc_heading()
append_block_as_body(el)
continue
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
if numbered_action == "toc":
append_block_as_body(el)
continue
if numbered_action == "split":
start_section(txt, heading_lvl or 1)
continue
if heading_lvl:
if not txt:
continue
# TOC и H2+ остават в тялото — граница само при H1 / Заглавие 1
if _is_toc_heading(txt) or heading_lvl >= 2:
# H2+ без номерация остават в тялото; H1 / Заглавие 1 режат
if heading_lvl >= 2:
if toc_state.phase:
toc_state.leave_phase()
append_block_as_body(el)
continue
flush()
current_title = txt
current_level = heading_lvl
sec_text, sec_html, sec_images = [], [], []
if toc_state.phase:
toc_state.leave_phase()
start_section(txt, heading_lvl)
continue
if el.name == "img":
if toc_state.phase:
toc_state.leave_phase()
# самостоятелен <img> (не вътре в блок)
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
base_dir, sec_images, img_counter)
@@ -836,6 +933,11 @@ def parse_html(path: Path) -> list[Section]:
sec_html.append(f"<p>{img_txt}</p>")
continue
if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS:
# <ol>/<ul> TOC списък — остава в преамбюла, приключва TOC фазата
toc_state.leave_phase()
elif toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
toc_state.leave_phase()
append_block_as_body(el)
flush()
@@ -911,6 +1013,7 @@ def parse_docx(path: Path) -> list[Section]:
buf: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
toc_state = _TocSplitState()
def flush():
if current_title or buf or sec_images:
@@ -918,8 +1021,26 @@ def parse_docx(path: Path) -> list[Section]:
sec.images = list(sec_images)
sections.append(sec)
def append_para(text: str, para_imgs: list[ImageRef]):
if text:
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, buf, sec_images
flush()
buf, sec_images = [], []
current_title = title
current_level = level
for kind, block in _iter_docx_blocks(doc):
if kind == "tbl":
if toc_state.phase:
toc_state.leave_phase()
for line in _table_lines(block):
buf.append(line)
continue
@@ -948,48 +1069,40 @@ def parse_docx(path: Path) -> list[Section]:
current_title = text
current_level = 1
else:
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
if toc_state.phase:
toc_state.leave_phase()
append_para(text, para_imgs)
continue
# TOC и H2+/bold остават в тялото — граница само при Heading 1 / Заглавие 1
if text and _is_toc_heading(text):
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
toc_state.note_toc_heading()
append_para(text, para_imgs)
continue
numbered_action = toc_state.handle_numbered(text) if text else "no"
if numbered_action == "toc":
append_para(text, para_imgs)
continue
if numbered_action == "split":
start_section(text, level or 1)
continue
# H2+/bold без номерация на глава — в тялото
if (level or 0) >= 2 or is_bold_heading:
if text:
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
if toc_state.phase:
toc_state.leave_phase()
append_para(text, para_imgs)
continue
if level == 1:
flush()
buf, sec_images = [], []
current_title = text
current_level = 1
if toc_state.phase:
toc_state.leave_phase()
start_section(text, 1)
continue
if text:
buf.append(text)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
toc_state.leave_phase()
append_para(text, para_imgs)
flush()