.
This commit is contained in:
@@ -605,6 +605,76 @@ def _is_toc_heading(text: str) -> bool:
|
||||
return bool(_TOC_HEADING_RE.match((text or "").strip()))
|
||||
|
||||
|
||||
# Top-level chapter: "1. Title" / "2) Title" — short line, not body/table/figure.
|
||||
_NUMBERED_CHAPTER_RE = re.compile(
|
||||
r"^(\d{1,2})[\.\)]\s+(\S.{0,100})$"
|
||||
)
|
||||
_FIGURE_CAPTION_RE = re.compile(
|
||||
r"^(фигура|figure|abb\.?|рис\.?)\s*\d+",
|
||||
re.I,
|
||||
)
|
||||
|
||||
|
||||
def _normalize_chapter_key(text: str) -> str:
|
||||
return re.sub(r"\s+", " ", (text or "").strip().lower())
|
||||
|
||||
|
||||
def _is_numbered_chapter_heading(text: str) -> bool:
|
||||
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
|
||||
t = (text or "").strip()
|
||||
if not t or len(t) >= 120:
|
||||
return False
|
||||
if "|" in t:
|
||||
return False
|
||||
if _is_toc_heading(t):
|
||||
return False
|
||||
if _FIGURE_CAPTION_RE.match(t):
|
||||
return False
|
||||
m = _NUMBERED_CHAPTER_RE.match(t)
|
||||
if not m:
|
||||
return False
|
||||
rest = m.group(2).strip()
|
||||
words = rest.split()
|
||||
if not words or len(words) > 12:
|
||||
return False
|
||||
# Body-like numbered sentences: "1. Copy the files into the folder."
|
||||
if rest.endswith((".", "!", "?")) and len(words) > 4:
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
class _TocSplitState:
|
||||
"""Tracks Съдържание / TOC so first '1. Title' stays in preamble, second starts a section."""
|
||||
|
||||
__slots__ = ("phase", "had_toc", "titles")
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.phase = False
|
||||
self.had_toc = False
|
||||
self.titles: set[str] = set()
|
||||
|
||||
def note_toc_heading(self) -> None:
|
||||
self.phase = True
|
||||
self.had_toc = True
|
||||
|
||||
def leave_phase(self) -> None:
|
||||
self.phase = False
|
||||
|
||||
def handle_numbered(self, text: str) -> str:
|
||||
"""Return 'toc' (keep in body), 'split' (new section), or 'no' (not numbered chapter)."""
|
||||
if not _is_numbered_chapter_heading(text):
|
||||
return "no"
|
||||
key = _normalize_chapter_key(text)
|
||||
if self.phase:
|
||||
if key in self.titles:
|
||||
self.phase = False
|
||||
return "split"
|
||||
self.titles.add(key)
|
||||
return "toc"
|
||||
# After TOC list/block, or docs without TOC: numbered chapter opens a section.
|
||||
return "split"
|
||||
|
||||
|
||||
def _heading_level_from_token(token: str) -> Optional[int]:
|
||||
t = _compact_style_token(token)
|
||||
if not t:
|
||||
@@ -779,6 +849,7 @@ def parse_html(path: Path) -> list[Section]:
|
||||
sec_html: list[str] = []
|
||||
sec_images: list[ImageRef] = []
|
||||
img_counter = [0]
|
||||
toc_state = _TocSplitState()
|
||||
|
||||
def flush():
|
||||
if current_title or sec_text or sec_html or sec_images:
|
||||
@@ -798,6 +869,13 @@ def parse_html(path: Path) -> list[Section]:
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def start_section(title: str, level: int = 1):
|
||||
nonlocal current_title, current_level, sec_text, sec_html, sec_images
|
||||
flush()
|
||||
current_title = title
|
||||
current_level = level
|
||||
sec_text, sec_html, sec_images = [], [], []
|
||||
|
||||
for el in blocks:
|
||||
txt = el.get_text(" ", strip=True)
|
||||
is_cover = _html_is_cover_title(el)
|
||||
@@ -809,23 +887,42 @@ def parse_html(path: Path) -> list[Section]:
|
||||
current_title = txt
|
||||
current_level = 1
|
||||
else:
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
append_block_as_body(el)
|
||||
continue
|
||||
|
||||
if txt and _is_toc_heading(txt):
|
||||
toc_state.note_toc_heading()
|
||||
append_block_as_body(el)
|
||||
continue
|
||||
|
||||
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
|
||||
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
|
||||
if numbered_action == "toc":
|
||||
append_block_as_body(el)
|
||||
continue
|
||||
if numbered_action == "split":
|
||||
start_section(txt, heading_lvl or 1)
|
||||
continue
|
||||
|
||||
if heading_lvl:
|
||||
if not txt:
|
||||
continue
|
||||
# TOC и H2+ остават в тялото — граница само при H1 / Заглавие 1
|
||||
if _is_toc_heading(txt) or heading_lvl >= 2:
|
||||
# H2+ без номерация остават в тялото; H1 / Заглавие 1 режат
|
||||
if heading_lvl >= 2:
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
append_block_as_body(el)
|
||||
continue
|
||||
flush()
|
||||
current_title = txt
|
||||
current_level = heading_lvl
|
||||
sec_text, sec_html, sec_images = [], [], []
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
start_section(txt, heading_lvl)
|
||||
continue
|
||||
|
||||
if el.name == "img":
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
# самостоятелен <img> (не вътре в блок)
|
||||
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
|
||||
base_dir, sec_images, img_counter)
|
||||
@@ -836,6 +933,11 @@ def parse_html(path: Path) -> list[Section]:
|
||||
sec_html.append(f"<p>{img_txt}</p>")
|
||||
continue
|
||||
|
||||
if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS:
|
||||
# <ol>/<ul> TOC списък — остава в преамбюла, приключва TOC фазата
|
||||
toc_state.leave_phase()
|
||||
elif toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
|
||||
toc_state.leave_phase()
|
||||
append_block_as_body(el)
|
||||
|
||||
flush()
|
||||
@@ -911,6 +1013,7 @@ def parse_docx(path: Path) -> list[Section]:
|
||||
buf: list[str] = []
|
||||
sec_images: list[ImageRef] = []
|
||||
img_counter = [0]
|
||||
toc_state = _TocSplitState()
|
||||
|
||||
def flush():
|
||||
if current_title or buf or sec_images:
|
||||
@@ -918,8 +1021,26 @@ def parse_docx(path: Path) -> list[Section]:
|
||||
sec.images = list(sec_images)
|
||||
sections.append(sec)
|
||||
|
||||
def append_para(text: str, para_imgs: list[ImageRef]):
|
||||
if text:
|
||||
buf.append(text)
|
||||
for im in para_imgs:
|
||||
img_counter[0] += 1
|
||||
im.placeholder = f"img_{img_counter[0]:02d}"
|
||||
sec_images.append(im)
|
||||
buf.append(f"[IMG: {im.placeholder}]")
|
||||
|
||||
def start_section(title: str, level: int = 1):
|
||||
nonlocal current_title, current_level, buf, sec_images
|
||||
flush()
|
||||
buf, sec_images = [], []
|
||||
current_title = title
|
||||
current_level = level
|
||||
|
||||
for kind, block in _iter_docx_blocks(doc):
|
||||
if kind == "tbl":
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
for line in _table_lines(block):
|
||||
buf.append(line)
|
||||
continue
|
||||
@@ -948,48 +1069,40 @@ def parse_docx(path: Path) -> list[Section]:
|
||||
current_title = text
|
||||
current_level = 1
|
||||
else:
|
||||
buf.append(text)
|
||||
for im in para_imgs:
|
||||
img_counter[0] += 1
|
||||
im.placeholder = f"img_{img_counter[0]:02d}"
|
||||
sec_images.append(im)
|
||||
buf.append(f"[IMG: {im.placeholder}]")
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
append_para(text, para_imgs)
|
||||
continue
|
||||
|
||||
# TOC и H2+/bold остават в тялото — граница само при Heading 1 / Заглавие 1
|
||||
if text and _is_toc_heading(text):
|
||||
buf.append(text)
|
||||
for im in para_imgs:
|
||||
img_counter[0] += 1
|
||||
im.placeholder = f"img_{img_counter[0]:02d}"
|
||||
sec_images.append(im)
|
||||
buf.append(f"[IMG: {im.placeholder}]")
|
||||
toc_state.note_toc_heading()
|
||||
append_para(text, para_imgs)
|
||||
continue
|
||||
|
||||
numbered_action = toc_state.handle_numbered(text) if text else "no"
|
||||
if numbered_action == "toc":
|
||||
append_para(text, para_imgs)
|
||||
continue
|
||||
if numbered_action == "split":
|
||||
start_section(text, level or 1)
|
||||
continue
|
||||
|
||||
# H2+/bold без номерация на глава — в тялото
|
||||
if (level or 0) >= 2 or is_bold_heading:
|
||||
if text:
|
||||
buf.append(text)
|
||||
for im in para_imgs:
|
||||
img_counter[0] += 1
|
||||
im.placeholder = f"img_{img_counter[0]:02d}"
|
||||
sec_images.append(im)
|
||||
buf.append(f"[IMG: {im.placeholder}]")
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
append_para(text, para_imgs)
|
||||
continue
|
||||
|
||||
if level == 1:
|
||||
flush()
|
||||
buf, sec_images = [], []
|
||||
current_title = text
|
||||
current_level = 1
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
start_section(text, 1)
|
||||
continue
|
||||
|
||||
if text:
|
||||
buf.append(text)
|
||||
for im in para_imgs:
|
||||
img_counter[0] += 1
|
||||
im.placeholder = f"img_{img_counter[0]:02d}"
|
||||
sec_images.append(im)
|
||||
buf.append(f"[IMG: {im.placeholder}]")
|
||||
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
|
||||
toc_state.leave_phase()
|
||||
append_para(text, para_imgs)
|
||||
|
||||
flush()
|
||||
|
||||
|
||||
Reference in New Issue
Block a user