This commit is contained in:
2026-09-14 11:24:21 +03:00
parent 181bbc3603
commit 971f60652d
5 changed files with 243 additions and 18 deletions

View File

@@ -22,6 +22,7 @@ help_processor.py
import os
import re
import sys
import html as html_lib
import json
import hashlib
import logging
@@ -164,7 +165,7 @@ class ProcessedSection:
keywords: str # "кл1, кл2, кл3"
text: str
images_json: str = "[]" # JSON масив с относителни пътища
html_text: str = "" # rich HTML (само за HTML-source файлове)
html_text: str = "" # rich HTML (HTML + DOCX; bold/цветове/картинки)
char_count: int = 0
def __post_init__(self):
@@ -556,6 +557,12 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
"valign", "width", "height", "bgcolor", "border")
# От style пазим само inline форматиране, нужно за viewer (bold/italic/color).
_HTML_STYLE_KEEP_RES = (
("color", re.compile(r"(?:^|;)\s*color\s*:\s*([^;]+)", re.I)),
("font-weight", re.compile(r"(?:^|;)\s*font-weight\s*:\s*([^;]+)", re.I)),
("font-style", re.compile(r"(?:^|;)\s*font-style\s*:\s*([^;]+)", re.I)),
)
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
_HEADING_TOKEN_RE = re.compile(
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
@@ -786,12 +793,42 @@ def _html_block_plain_text(el) -> str:
return el.get_text(sep, strip=True)
def _kept_inline_style(style: str) -> str:
"""Извлича color / bold / italic от CSS style; останалото се маха."""
if not style:
return ""
parts: list[str] = []
for prop, rx in _HTML_STYLE_KEEP_RES:
m = rx.search(style)
if not m:
continue
val = m.group(1).strip()
if not val:
continue
low = val.lower()
if prop == "font-weight" and low not in (
"bold", "bolder", "600", "700", "800", "900"
):
continue
if prop == "font-style" and "italic" not in low and "oblique" not in low:
continue
parts.append(f"{prop}:{val}")
return ";".join(parts)
def _strip_attrs(el):
"""Премахва decorative атрибути (class, style, on*, data-*)."""
"""Премахва decorative атрибути; пази color/bold/italic от style и font color=."""
for t in el.find_all(True):
kept_style = ""
raw_style = t.attrs.get("style") if t.attrs else None
if isinstance(raw_style, str):
kept_style = _kept_inline_style(raw_style)
for a in list(t.attrs):
if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"):
del t[a]
# <font color="..."> остава (color не е в DROP)
if kept_style:
t["style"] = kept_style
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
@@ -975,6 +1012,69 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
return imgs
def _docx_run_color_css(run) -> Optional[str]:
"""RGB от run.font.color → '#RRGGBB'; theme/auto цветове се пропускат."""
try:
color = run.font.color
if color is None or color.rgb is None:
return None
return f"#{color.rgb}"
except Exception:
return None
def _docx_run_to_html(run) -> str:
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
text = run.text or ""
if not text:
return ""
s = html_lib.escape(text, quote=False).replace("\n", "<br>")
if run.bold:
s = f"<b>{s}</b>"
if run.italic:
s = f"<i>{s}</i>"
if run.underline:
s = f"<u>{s}</u>"
css_color = _docx_run_color_css(run)
if css_color:
s = f'<span style="color:{css_color}">{s}</span>'
return s
def _docx_para_to_html(para) -> str:
"""Параграф → <p>…</p> с inline форматиране от runs."""
parts = [_docx_run_to_html(r) for r in para.runs]
inner = "".join(parts)
if not inner.strip():
return ""
return f"<p>{inner}</p>"
def _table_to_html(table) -> str:
"""Таблица → прост HTML <table> (plain клетки, без вложен rich text)."""
rows_html: list[str] = []
try:
rows = table.rows
except Exception:
return ""
for row in rows:
cells_html: list[str] = []
try:
cells = row.cells
except Exception:
continue
for cell in cells:
cell_text = " ".join((cell.text or "").split())
if not cell_text:
continue
cells_html.append(f"<td>{html_lib.escape(cell_text, quote=False)}</td>")
if cells_html:
rows_html.append("<tr>" + "".join(cells_html) + "</tr>")
if not rows_html:
return ""
return "<table>" + "".join(rows_html) + "</table>"
def _iter_docx_blocks(doc):
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
from docx.oxml.ns import qn
@@ -1011,6 +1111,7 @@ def parse_docx(path: Path) -> list[Section]:
sections: list[Section] = []
current_title, current_level = "", 1
buf: list[str] = []
buf_html: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
toc_state = _TocSplitState()
@@ -1019,21 +1120,30 @@ def parse_docx(path: Path) -> list[Section]:
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), current_level)
sec.images = list(sec_images)
sec.html_text = "\n".join(buf_html) if buf_html else None
sections.append(sec)
def append_para(text: str, para_imgs: list[ImageRef]):
def append_para(text: str, para_imgs: list[ImageRef], html_frag: str = ""):
if text:
buf.append(text)
if html_frag:
buf_html.append(html_frag)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
buf.append(f"[IMG: {im.placeholder}]")
ph = f"[IMG: {im.placeholder}]"
buf.append(ph)
buf_html.append(f"<p>{ph}</p>")
def append_from_para(para, text: str, para_imgs: list[ImageRef]):
html_frag = _docx_para_to_html(para) if text else ""
append_para(text, para_imgs, html_frag)
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, buf, sec_images
nonlocal current_title, current_level, buf, buf_html, sec_images
flush()
buf, sec_images = [], []
buf, buf_html, sec_images = [], [], []
current_title = title
current_level = level
@@ -1043,6 +1153,9 @@ def parse_docx(path: Path) -> list[Section]:
toc_state.leave_phase()
for line in _table_lines(block):
buf.append(line)
th = _table_to_html(block)
if th:
buf_html.append(th)
continue
para = block
@@ -1071,17 +1184,17 @@ def parse_docx(path: Path) -> list[Section]:
else:
if toc_state.phase:
toc_state.leave_phase()
append_para(text, para_imgs)
append_from_para(para, text, para_imgs)
continue
if text and _is_toc_heading(text):
toc_state.note_toc_heading()
append_para(text, para_imgs)
append_from_para(para, text, para_imgs)
continue
numbered_action = toc_state.handle_numbered(text) if text else "no"
if numbered_action == "toc":
append_para(text, para_imgs)
append_from_para(para, text, para_imgs)
continue
if numbered_action == "split":
start_section(text, level or 1)
@@ -1091,7 +1204,7 @@ def parse_docx(path: Path) -> list[Section]:
if (level or 0) >= 2 or is_bold_heading:
if toc_state.phase:
toc_state.leave_phase()
append_para(text, para_imgs)
append_from_para(para, text, para_imgs)
continue
if level == 1:
@@ -1102,7 +1215,7 @@ def parse_docx(path: Path) -> list[Section]:
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
toc_state.leave_phase()
append_para(text, para_imgs)
append_from_para(para, text, para_imgs)
flush()
@@ -1521,8 +1634,13 @@ def merge_preamble_sections(sections: list[Section]) -> list[Section]:
def clean_text(text: str) -> str:
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
"""Collapse spaces/tabs but keep newlines (lists, paragraphs).
Маха опасни C0 контроли (NUL и др.); пази Unicode символи (•, →, NBSP→space).
"""
text = text.replace("\r\n", "\n").replace("\r", "\n")
text = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", "", text)
text = text.replace("\u00a0", " ")
text = re.sub(r"[^\S\n]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
text = "\n".join(line.strip() for line in text.split("\n"))