From 971f60652d8779d0c6d6f4797b831bbacdcf8bb8 Mon Sep 17 00:00:00 2001 From: sabo Date: Mon, 14 Sep 2026 11:24:21 +0300 Subject: [PATCH] . --- README.md | 2 +- VERSION | 2 +- docs/razdeljane-na-sekcii.md | 10 ++- help_processor.py | 142 ++++++++++++++++++++++++++++++++--- tests/test_rich_content.py | 105 ++++++++++++++++++++++++++ 5 files changed, 243 insertions(+), 18 deletions(-) create mode 100644 tests/test_rich_content.py diff --git a/README.md b/README.md index 4bb2dee..3ee6fe5 100644 --- a/README.md +++ b/README.md @@ -92,7 +92,7 @@ UNIQUE constraint: `(prefix, file_path)` | char_count | INT | Размер на чистия текст | | output_path | NVARCHAR(1000) | Път до `.txt` файла | | images | NVARCHAR(MAX) | JSON масив с относителни пътища | -| html_text | NVARCHAR(MAX) | Rich HTML с форматиране (само за `.html` източници) | +| html_text | NVARCHAR(MAX) | Rich HTML (`.html` и `.docx`: bold/цветове/картинки) | | created_at, updated_at | DATETIME2 | | ## HTML Viewer — 3 / 4 таба diff --git a/VERSION b/VERSION index 1c09c74..42045ac 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -0.3.3 +0.3.4 diff --git a/docs/razdeljane-na-sekcii.md b/docs/razdeljane-na-sekcii.md index 84bba0b..70734ae 100644 --- a/docs/razdeljane-na-sekcii.md +++ b/docs/razdeljane-na-sekcii.md @@ -98,9 +98,11 @@ TXT и PDF запазват своите правила (номерирани р ### 3.3. Таблици и картинки -Таблица: всеки ред се сплесква до `клетка | клетка | клетка` и редовете се добавят в тялото на текущата секция. +Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост ``. -Картинки в параграф: записват се като `[IMG: img_NN]` в същото тяло. +Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с ``). + +**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `` / `` / ``, `` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. Празен параграф без картинка се пропуска. @@ -132,7 +134,7 @@ TXT и PDF запазват своите правила (номерирани р ### 5.2. Съдържание -Списъци и таблици влизат в текущата секция. В plain text редовете им са с нов ред (не сплескани в един ред). Декоративни атрибути (`class`, `style`, `on*`, `data-*` …) се махат от запазения HTML. +Списъци и таблици влизат в текущата секция. В plain text редовете им са с нов ред (не сплескани в един ред). Декоративни атрибути (`class`, `id`, `on*`, `data-*` …) се махат; от `style` се **пазят** `color`, `font-weight` (bold) и `font-style` (italic). Тагове ``/``/``/`` и `` остават. Картинки: локални и `data:` URI; HTTP(S) се пропускат. Дребни иконки под 50×50 px се изхвърлят. @@ -192,7 +194,7 @@ TXT и PDF запазват своите правила (номерирани р - Ако няма текст, картинки и HTML, но има заглавие → тялото става самото заглавие. Затова „голо“ заглавие оцелява като секция. - Ако няма нито текст, нито заглавие, нито картинки → секцията се пропуска. -`clean_text()` срива поредици от интервали/табове, но пази нови редове (списъци и абзаци). +`clean_text()` срива поредици от интервали/табове, но пази нови редове (списъци и абзаци). Маха опасни C0 контроли (NUL и др.); Unicode символи (•, →) остават; NBSP става обикновен интервал. --- diff --git a/help_processor.py b/help_processor.py index 68122dd..f24ef9d 100644 --- a/help_processor.py +++ b/help_processor.py @@ -22,6 +22,7 @@ help_processor.py import os import re import sys +import html as html_lib import json import hashlib import logging @@ -164,7 +165,7 @@ class ProcessedSection: keywords: str # "кл1, кл2, кл3" text: str images_json: str = "[]" # JSON масив с относителни пътища - html_text: str = "" # rich HTML (само за HTML-source файлове) + html_text: str = "" # rich HTML (HTML + DOCX; bold/цветове/картинки) char_count: int = 0 def __post_init__(self): @@ -556,6 +557,12 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6", _HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"}) _HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align", "valign", "width", "height", "bgcolor", "border") +# От style пазим само inline форматиране, нужно за viewer (bold/italic/color). +_HTML_STYLE_KEEP_RES = ( + ("color", re.compile(r"(?:^|;)\s*color\s*:\s*([^;]+)", re.I)), + ("font-weight", re.compile(r"(?:^|;)\s*font-weight\s*:\s*([^;]+)", re.I)), + ("font-style", re.compile(r"(?:^|;)\s*font-style\s*:\s*([^;]+)", re.I)), +) _HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3} _HEADING_TOKEN_RE = re.compile( r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)" @@ -786,12 +793,42 @@ def _html_block_plain_text(el) -> str: return el.get_text(sep, strip=True) +def _kept_inline_style(style: str) -> str: + """Извлича color / bold / italic от CSS style; останалото се маха.""" + if not style: + return "" + parts: list[str] = [] + for prop, rx in _HTML_STYLE_KEEP_RES: + m = rx.search(style) + if not m: + continue + val = m.group(1).strip() + if not val: + continue + low = val.lower() + if prop == "font-weight" and low not in ( + "bold", "bolder", "600", "700", "800", "900" + ): + continue + if prop == "font-style" and "italic" not in low and "oblique" not in low: + continue + parts.append(f"{prop}:{val}") + return ";".join(parts) + + def _strip_attrs(el): - """Премахва decorative атрибути (class, style, on*, data-*).""" + """Премахва decorative атрибути; пази color/bold/italic от style и font color=.""" for t in el.find_all(True): + kept_style = "" + raw_style = t.attrs.get("style") if t.attrs else None + if isinstance(raw_style, str): + kept_style = _kept_inline_style(raw_style) for a in list(t.attrs): if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"): del t[a] + # остава (color не е в DROP) + if kept_style: + t["style"] = kept_style def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None: @@ -975,6 +1012,69 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]: return imgs +def _docx_run_color_css(run) -> Optional[str]: + """RGB от run.font.color → '#RRGGBB'; theme/auto цветове се пропускат.""" + try: + color = run.font.color + if color is None or color.rgb is None: + return None + return f"#{color.rgb}" + except Exception: + return None + + +def _docx_run_to_html(run) -> str: + """Един Word run → HTML с // и color span. Спец. символи се escape-ват.""" + text = run.text or "" + if not text: + return "" + s = html_lib.escape(text, quote=False).replace("\n", "
") + if run.bold: + s = f"{s}" + if run.italic: + s = f"{s}" + if run.underline: + s = f"{s}" + css_color = _docx_run_color_css(run) + if css_color: + s = f'{s}' + return s + + +def _docx_para_to_html(para) -> str: + """Параграф →

…

с inline форматиране от runs.""" + parts = [_docx_run_to_html(r) for r in para.runs] + inner = "".join(parts) + if not inner.strip(): + return "" + return f"

{inner}

" + + +def _table_to_html(table) -> str: + """Таблица → прост HTML
(plain клетки, без вложен rich text).""" + rows_html: list[str] = [] + try: + rows = table.rows + except Exception: + return "" + for row in rows: + cells_html: list[str] = [] + try: + cells = row.cells + except Exception: + continue + for cell in cells: + cell_text = " ".join((cell.text or "").split()) + if not cell_text: + continue + cells_html.append(f"") + if cells_html: + rows_html.append("" + "".join(cells_html) + "") + if not rows_html: + return "" + return "
{html_lib.escape(cell_text, quote=False)}
" + "".join(rows_html) + "
" + + def _iter_docx_blocks(doc): """Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците).""" from docx.oxml.ns import qn @@ -1011,6 +1111,7 @@ def parse_docx(path: Path) -> list[Section]: sections: list[Section] = [] current_title, current_level = "", 1 buf: list[str] = [] + buf_html: list[str] = [] sec_images: list[ImageRef] = [] img_counter = [0] toc_state = _TocSplitState() @@ -1019,21 +1120,30 @@ def parse_docx(path: Path) -> list[Section]: if current_title or buf or sec_images: sec = Section(current_title, "\n".join(buf), current_level) sec.images = list(sec_images) + sec.html_text = "\n".join(buf_html) if buf_html else None sections.append(sec) - def append_para(text: str, para_imgs: list[ImageRef]): + def append_para(text: str, para_imgs: list[ImageRef], html_frag: str = ""): if text: buf.append(text) + if html_frag: + buf_html.append(html_frag) for im in para_imgs: img_counter[0] += 1 im.placeholder = f"img_{img_counter[0]:02d}" sec_images.append(im) - buf.append(f"[IMG: {im.placeholder}]") + ph = f"[IMG: {im.placeholder}]" + buf.append(ph) + buf_html.append(f"

{ph}

") + + def append_from_para(para, text: str, para_imgs: list[ImageRef]): + html_frag = _docx_para_to_html(para) if text else "" + append_para(text, para_imgs, html_frag) def start_section(title: str, level: int = 1): - nonlocal current_title, current_level, buf, sec_images + nonlocal current_title, current_level, buf, buf_html, sec_images flush() - buf, sec_images = [], [] + buf, buf_html, sec_images = [], [], [] current_title = title current_level = level @@ -1043,6 +1153,9 @@ def parse_docx(path: Path) -> list[Section]: toc_state.leave_phase() for line in _table_lines(block): buf.append(line) + th = _table_to_html(block) + if th: + buf_html.append(th) continue para = block @@ -1071,17 +1184,17 @@ def parse_docx(path: Path) -> list[Section]: else: if toc_state.phase: toc_state.leave_phase() - append_para(text, para_imgs) + append_from_para(para, text, para_imgs) continue if text and _is_toc_heading(text): toc_state.note_toc_heading() - append_para(text, para_imgs) + append_from_para(para, text, para_imgs) continue numbered_action = toc_state.handle_numbered(text) if text else "no" if numbered_action == "toc": - append_para(text, para_imgs) + append_from_para(para, text, para_imgs) continue if numbered_action == "split": start_section(text, level or 1) @@ -1091,7 +1204,7 @@ def parse_docx(path: Path) -> list[Section]: if (level or 0) >= 2 or is_bold_heading: if toc_state.phase: toc_state.leave_phase() - append_para(text, para_imgs) + append_from_para(para, text, para_imgs) continue if level == 1: @@ -1102,7 +1215,7 @@ def parse_docx(path: Path) -> list[Section]: if toc_state.phase and text and not _is_numbered_chapter_heading(text): toc_state.leave_phase() - append_para(text, para_imgs) + append_from_para(para, text, para_imgs) flush() @@ -1521,8 +1634,13 @@ def merge_preamble_sections(sections: list[Section]) -> list[Section]: def clean_text(text: str) -> str: - """Collapse spaces/tabs but keep newlines (lists, paragraphs).""" + """Collapse spaces/tabs but keep newlines (lists, paragraphs). + + Маха опасни C0 контроли (NUL и др.); пази Unicode символи (•, →, NBSP→space). + """ text = text.replace("\r\n", "\n").replace("\r", "\n") + text = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", "", text) + text = text.replace("\u00a0", " ") text = re.sub(r"[^\S\n]+", " ", text) text = re.sub(r"\n{3,}", "\n\n", text) text = "\n".join(line.strip() for line in text.split("\n")) diff --git a/tests/test_rich_content.py b/tests/test_rich_content.py new file mode 100644 index 0000000..e13735f --- /dev/null +++ b/tests/test_rich_content.py @@ -0,0 +1,105 @@ +"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML).""" +from pathlib import Path + +from docx import Document +from docx.shared import RGBColor + +from help_processor import ( + clean_text, + merge_preamble_sections, + merge_short_sections, + parse_docx, + parse_html, +) + +FIXTURES = Path(__file__).parent / "fixtures" + + +def _mini_png(w: int = 64, h: int = 64) -> bytes: + """Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста.""" + try: + from io import BytesIO + from PIL import Image + + buf = BytesIO() + Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG") + return buf.getvalue() + except Exception: + # 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично + return ( + b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01" + b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00" + b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N" + b"\x00\x00\x00\x00IEND\xaeB`\x82" + ) + + +def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path): + doc = Document() + doc.add_heading("1. Форматиране", level=1) + p = doc.add_paragraph() + r1 = p.add_run("Bold ") + r1.bold = True + r2 = p.add_run("и червено") + r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00) + p2 = doc.add_paragraph() + p2.add_run("Стрелка → и булет • в текста") + + img_path = tmp_path / "pic.png" + png = _mini_png() + img_path.write_bytes(png) + try: + from io import BytesIO + from PIL import Image + + assert Image.open(BytesIO(png)).size[0] >= 50 + doc.add_picture(str(img_path)) + expect_img = True + except Exception: + expect_img = False + + out = tmp_path / "rich.docx" + doc.save(out) + + sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) + assert len(sections) >= 1 + body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text)) + assert "Bold" in body.text + assert "→" in body.text + assert "•" in body.text + assert "" in (body.html_text or "") + assert "color:#C00000" in (body.html_text or "") + # Plain text без HTML тагове — за Claude / keywords + assert "" not in body.text + if expect_img: + assert body.images + assert "[IMG:" in body.text + assert "[IMG:" in (body.html_text or "") + + +def test_html_preserves_bold_and_style_color(tmp_path: Path): + html = tmp_path / "rich.html" + html.write_text( + """ +

1. Цветове

+

Обикновен bold и + зелен + плюс син.

+ """, + encoding="utf-8", + ) + sections = merge_preamble_sections(merge_short_sections(parse_html(html))) + assert sections + sec = sections[0] + assert "bold" in sec.text + html_body = sec.html_text or "" + assert "" in html_body or "" in html_body + assert "color:#00aa00" in html_body or "color: #00aa00" in html_body + assert "font-size" not in html_body # декоративното се маха + assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body + + +def test_clean_text_strips_nul_keeps_bullets(): + assert "\x00" not in clean_text("a\x00b\tc") + assert "•" in clean_text("елемент • едно") + assert "→" in clean_text("A → B")