diff --git a/docs/razdeljane-na-sekcii.md b/docs/razdeljane-na-sekcii.md index 70734ae..0ee0770 100644 --- a/docs/razdeljane-na-sekcii.md +++ b/docs/razdeljane-na-sekcii.md @@ -100,9 +100,9 @@ TXT и PDF запазват своите правила (номерирани р Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост ``. -Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с ``). +Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с ``). Поддържат се `r:embed` (вградени) и `r:link` (външни `file:///` / UNC), ако файлът е достъпен при сканиране. -**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `` / `` / ``, `` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. +**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `` / `` / `` (изричен run **или** наследен от paragraph/character style), `` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Runs вътре в `w:hyperlink` (TOC редове) също влизат в `html_text`; ако runs липсват, има fallback към escaped `para.text`. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. Viewer показва `html_text` когато е наличен — затова TOC/bold/картинки трябва да са в него, не само в plain `text`. Празен параграф без картинка се пропуска. diff --git a/help_processor.py b/help_processor.py index f24ef9d..edcd326 100644 --- a/help_processor.py +++ b/help_processor.py @@ -985,8 +985,69 @@ def parse_html(path: Path) -> list[Section]: return sections +def _file_url_to_path(url: str) -> Optional[Path]: + """file:///… или обикновен път → Path (вкл. UNC от Word r:link).""" + from urllib.parse import unquote + + s = unquote((url or "").strip()) + if not s: + return None + if s.lower().startswith("file:"): + s = s[5:] + while s.startswith("/"): + s = s[1:] + if not s: + return None + return Path(s) + + +def _load_external_image_bytes(url: str) -> Optional[tuple[bytes, str]]: + """Чете външна картинка (r:link / TargetMode=External); връща (data, ext).""" + path = _file_url_to_path(url) + if path is None: + return None + try: + if not path.is_file(): + return None + data = path.read_bytes() + except OSError: + return None + if not data: + return None + ext = (path.suffix or "").lstrip(".").lower() or "png" + if ext == "jpeg": + ext = "jpg" + return data, ext + + +def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]: + """r:embed → related_parts; r:link → външен файл. Връща (data, ext).""" + if not rId: + return None + try: + part = doc.part.related_parts[rId] + data = part.blob + ct = getattr(part, "content_type", "") or "" + if data: + return data, _ext_from_content_type(ct) + except Exception: + pass + try: + rel = doc.part.rels[rId] + except Exception: + return None + if not getattr(rel, "is_external", False): + return None + target = getattr(rel, "target_ref", None) or "" + loaded = _load_external_image_bytes(target) + return loaded + + def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]: - """Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.""" + """Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер. + + Поддържа r:embed (вградени) и r:link (външни file:/// / UNC пътища). + """ from docx.oxml.ns import qn imgs: list[ImageRef] = [] try: @@ -995,19 +1056,17 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]: return imgs embed_attr = qn("r:embed") + link_attr = qn("r:link") for blip in blips: - rId = blip.get(embed_attr) + rId = blip.get(embed_attr) or blip.get(link_attr) if not rId: continue - try: - part = doc.part.related_parts[rId] - data = part.blob - ct = getattr(part, "content_type", "") or "" - except Exception: + resolved = _resolve_docx_image_rid(doc, rId) + if not resolved: continue + data, ext = resolved if not _should_keep_image(data): continue - ext = _ext_from_content_type(ct) imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext)) return imgs @@ -1023,17 +1082,72 @@ def _docx_run_color_css(run) -> Optional[str]: return None -def _docx_run_to_html(run) -> str: +def _docx_style_font_flag(style, attr: str) -> Optional[bool]: + """True/False от style.font.; None ако стилът/атрибутът липсва.""" + if style is None: + return None + try: + font = style.font + if font is None: + return None + val = getattr(font, attr, None) + if val is None: + return None + return bool(val) + except Exception: + return None + + +def _docx_run_effective_flag(run, para, attr: str) -> bool: + """Ефективен bold/italic/underline: изричен run → char style → paragraph style.""" + try: + direct = getattr(run, attr, None) + except Exception: + direct = None + if direct is True: + return True + if direct is False: + return False + # None = наследяване + try: + char_style = run.style + except Exception: + char_style = None + flag = _docx_style_font_flag(char_style, attr) + if flag is not None: + return flag + try: + p_style = para.style if para is not None else None + except Exception: + p_style = None + flag = _docx_style_font_flag(p_style, attr) + return bool(flag) + + +def _iter_docx_para_runs(para): + """Runs в параграф, вкл. вътре в w:hyperlink (python-docx.para.runs ги пропуска).""" + from docx.oxml.ns import qn + from docx.text.run import Run + + for child in para._element.iterchildren(): + if child.tag == qn("w:r"): + yield Run(child, para) + elif child.tag == qn("w:hyperlink"): + for r_elem in child.findall(qn("w:r")): + yield Run(r_elem, para) + + +def _docx_run_to_html(run, para=None) -> str: """Един Word run → HTML с // и color span. Спец. символи се escape-ват.""" text = run.text or "" if not text: return "" s = html_lib.escape(text, quote=False).replace("\n", "
") - if run.bold: + if _docx_run_effective_flag(run, para, "bold"): s = f"{s}" - if run.italic: + if _docx_run_effective_flag(run, para, "italic"): s = f"{s}" - if run.underline: + if _docx_run_effective_flag(run, para, "underline"): s = f"{s}" css_color = _docx_run_color_css(run) if css_color: @@ -1042,11 +1156,15 @@ def _docx_run_to_html(run) -> str: def _docx_para_to_html(para) -> str: - """Параграф →

…

с inline форматиране от runs.""" - parts = [_docx_run_to_html(r) for r in para.runs] + """Параграф →

…

с inline форматиране от runs (вкл. hyperlink TOC).""" + parts = [_docx_run_to_html(r, para) for r in _iter_docx_para_runs(para)] inner = "".join(parts) if not inner.strip(): - return "" + # Fallback: para.text вижда hyperlink текст, но без runs в .runs + plain = (para.text or "").strip() + if not plain: + return "" + inner = html_lib.escape(plain, quote=False).replace("\n", "
") return f"

{inner}

" diff --git a/tests/test_rich_content.py b/tests/test_rich_content.py index e13735f..c5ad5c0 100644 --- a/tests/test_rich_content.py +++ b/tests/test_rich_content.py @@ -1,7 +1,10 @@ """Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML).""" from pathlib import Path +import pytest from docx import Document +from docx.oxml import OxmlElement +from docx.oxml.ns import qn from docx.shared import RGBColor from help_processor import ( @@ -13,6 +16,7 @@ from help_processor import ( ) FIXTURES = Path(__file__).parent / "fixtures" +ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx") def _mini_png(w: int = 64, h: int = 64) -> bytes: @@ -34,6 +38,79 @@ def _mini_png(w: int = 64, h: int = 64) -> bytes: ) +def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"): + """TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == []).""" + p = doc.add_paragraph() + for child in list(p._element): + if child.tag == qn("w:r"): + p._element.remove(child) + hl = OxmlElement("w:hyperlink") + hl.set(qn("w:anchor"), anchor) + r = OxmlElement("w:r") + t = OxmlElement("w:t") + t.set(qn("xml:space"), "preserve") + t.text = text + r.append(t) + hl.append(r) + p._element.append(hl) + assert len(p.runs) == 0 + assert text in (p.text or "") + return p + + +def _add_external_blip_paragraph(doc: Document, image_path: Path): + """Параграф с a:blip/@r:link към външен файл (като atra-manual.docx).""" + from docx.opc.constants import RELATIONSHIP_TYPE as RT + + rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri()) + rId = rel if isinstance(rel, str) else rel.rId + + p = doc.add_paragraph() + r = OxmlElement("w:r") + drawing = OxmlElement("w:drawing") + inline = OxmlElement("wp:inline") + inline.set("distT", "0") + inline.set("distB", "0") + inline.set("distL", "0") + inline.set("distR", "0") + extent = OxmlElement("wp:extent") + extent.set("cx", "914400") + extent.set("cy", "914400") + inline.append(extent) + docPr = OxmlElement("wp:docPr") + docPr.set("id", "1") + docPr.set("name", "Picture") + inline.append(docPr) + graphic = OxmlElement("a:graphic") + graphicData = OxmlElement("a:graphicData") + graphicData.set( + "uri", "http://schemas.openxmlformats.org/drawingml/2006/picture" + ) + pic = OxmlElement("pic:pic") + nvPicPr = OxmlElement("pic:nvPicPr") + cNvPr = OxmlElement("pic:cNvPr") + cNvPr.set("id", "0") + cNvPr.set("name", "pic") + nvPicPr.append(cNvPr) + nvPicPr.append(OxmlElement("pic:cNvPicPr")) + pic.append(nvPicPr) + blipFill = OxmlElement("pic:blipFill") + blip = OxmlElement("a:blip") + blip.set(qn("r:link"), rId) + blipFill.append(blip) + blipFill.append(OxmlElement("a:stretch")) + pic.append(blipFill) + spPr = OxmlElement("pic:spPr") + pic.append(spPr) + graphicData.append(pic) + graphic.append(graphicData) + inline.append(graphic) + drawing.append(inline) + r.append(drawing) + p._element.append(r) + return p + + def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path): doc = Document() doc.add_heading("1. Форматиране", level=1) @@ -77,6 +154,95 @@ def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path): assert "[IMG:" in (body.html_text or "") +def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path): + """TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази.""" + doc = Document() + doc.add_heading("NESPERTCAM Launcher - тест", level=1) + doc.add_heading("Съдържание", level=3) + _add_hyperlink_paragraph(doc, "1. Преглед на приложението") + _add_hyperlink_paragraph(doc, "2. Инсталация и настройка") + doc.add_heading("1. Преглед на приложението", level=2) + h3 = doc.add_heading("Какво е NESPERTCAM?", level=3) + # Heading style → run.bold is None, style.font.bold True + assert h3.runs and h3.runs[0].bold is not True + p = doc.add_paragraph() + p.add_run("Тяло с достатъчно думи за секцията на ръководството.") + + out = tmp_path / "toc_hyperlink.docx" + doc.save(out) + + sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) + assert sections + preamble = sections[0] + assert "Съдържание" in (preamble.text or "") + assert "1. Преглед" in (preamble.text or "") + html = preamble.html_text or "" + assert "Съдържание" in html + assert "1. Преглед" in html + assert "2. Инсталация" in html + + overview = next(s for s in sections if s.title.startswith("1. Преглед")) + assert "" in (overview.html_text or "") + assert "Какво е NESPERTCAM?" in (overview.html_text or "") + + +def test_docx_external_r_link_image(tmp_path: Path): + png = _mini_png() + img_path = tmp_path / "ext.png" + img_path.write_bytes(png) + try: + from io import BytesIO + from PIL import Image + + assert Image.open(BytesIO(png)).size[0] >= 50 + except Exception: + pytest.skip("PIL unavailable for size check") + + doc = Document() + doc.add_heading("1. С картинка", level=1) + doc.add_paragraph("Преди фигурата.") + try: + _add_external_blip_paragraph(doc, img_path) + except Exception as exc: + pytest.skip(f"cannot build r:link drawing: {exc}") + doc.add_paragraph("След фигурата.") + + out = tmp_path / "ext_img.docx" + doc.save(out) + + sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) + body = sections[0] + assert body.images, "r:link external image must be loaded" + assert "[IMG:" in (body.text or "") + assert "[IMG:" in (body.html_text or "") + + +@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted") +def test_atra_manual_toc_bold_and_image_in_html(): + """Регресия върху реалния NESPERTCAM Launcher manual.""" + sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL))) + assert len(sections) >= 3 + first = sections[0] + assert "NESPERTCAM" in first.title + assert "Съдържание" in (first.text or "") + assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "") + html0 = first.html_text or "" + assert "Съдържание" in html0 + assert "1. Преглед" in html0 or "Инсталация" in html0 + + # Style-bold подзаглавия + изричен bold по-нататък + assert any("" in (s.html_text or "") for s in sections) + + # Фигура 1 е външен UNC линк — ако share-ът е достъпен, трябва [IMG:] + overview = next((s for s in sections if "Преглед" in s.title), None) + assert overview is not None + if overview.images or "[IMG:" in (overview.text or ""): + assert "[IMG:" in (overview.html_text or "") + else: + # Документът сочи към външен файл; ако UNC липсва, plain caption остава + assert "Фигура 1" in (overview.text or "") + + def test_html_preserves_bold_and_style_color(tmp_path: Path): html = tmp_path / "rich.html" html.write_text(