"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML).""" from pathlib import Path import pytest from docx import Document from docx.oxml import OxmlElement from docx.oxml.ns import qn from docx.shared import RGBColor from help_processor import ( clean_text, merge_preamble_sections, merge_short_sections, parse_docx, parse_html, ) FIXTURES = Path(__file__).parent / "fixtures" ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx") def _mini_png(w: int = 64, h: int = 64) -> bytes: """Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста.""" try: from io import BytesIO from PIL import Image buf = BytesIO() Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG") return buf.getvalue() except Exception: # 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично return ( b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01" b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00" b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N" b"\x00\x00\x00\x00IEND\xaeB`\x82" ) def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"): """TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == []).""" p = doc.add_paragraph() for child in list(p._element): if child.tag == qn("w:r"): p._element.remove(child) hl = OxmlElement("w:hyperlink") hl.set(qn("w:anchor"), anchor) r = OxmlElement("w:r") t = OxmlElement("w:t") t.set(qn("xml:space"), "preserve") t.text = text r.append(t) hl.append(r) p._element.append(hl) assert len(p.runs) == 0 assert text in (p.text or "") return p def _add_external_blip_paragraph(doc: Document, image_path: Path): """Параграф с a:blip/@r:link към външен файл (като atra-manual.docx).""" from docx.opc.constants import RELATIONSHIP_TYPE as RT rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri()) rId = rel if isinstance(rel, str) else rel.rId p = doc.add_paragraph() r = OxmlElement("w:r") drawing = OxmlElement("w:drawing") inline = OxmlElement("wp:inline") inline.set("distT", "0") inline.set("distB", "0") inline.set("distL", "0") inline.set("distR", "0") extent = OxmlElement("wp:extent") extent.set("cx", "914400") extent.set("cy", "914400") inline.append(extent) docPr = OxmlElement("wp:docPr") docPr.set("id", "1") docPr.set("name", "Picture") inline.append(docPr) graphic = OxmlElement("a:graphic") graphicData = OxmlElement("a:graphicData") graphicData.set( "uri", "http://schemas.openxmlformats.org/drawingml/2006/picture" ) pic = OxmlElement("pic:pic") nvPicPr = OxmlElement("pic:nvPicPr") cNvPr = OxmlElement("pic:cNvPr") cNvPr.set("id", "0") cNvPr.set("name", "pic") nvPicPr.append(cNvPr) nvPicPr.append(OxmlElement("pic:cNvPicPr")) pic.append(nvPicPr) blipFill = OxmlElement("pic:blipFill") blip = OxmlElement("a:blip") blip.set(qn("r:link"), rId) blipFill.append(blip) blipFill.append(OxmlElement("a:stretch")) pic.append(blipFill) spPr = OxmlElement("pic:spPr") pic.append(spPr) graphicData.append(pic) graphic.append(graphicData) inline.append(graphic) drawing.append(inline) r.append(drawing) p._element.append(r) return p def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path): doc = Document() doc.add_heading("1. Форматиране", level=1) p = doc.add_paragraph() r1 = p.add_run("Bold ") r1.bold = True r2 = p.add_run("и червено") r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00) p2 = doc.add_paragraph() p2.add_run("Стрелка → и булет • в текста") img_path = tmp_path / "pic.png" png = _mini_png() img_path.write_bytes(png) try: from io import BytesIO from PIL import Image assert Image.open(BytesIO(png)).size[0] >= 50 doc.add_picture(str(img_path)) expect_img = True except Exception: expect_img = False out = tmp_path / "rich.docx" doc.save(out) sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) assert len(sections) >= 1 body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text)) assert "Bold" in body.text assert "→" in body.text assert "•" in body.text assert "" in (body.html_text or "") assert "color:#C00000" in (body.html_text or "") # Plain text без HTML тагове — за Claude / keywords assert "" not in body.text if expect_img: assert body.images assert "[IMG:" in body.text assert "[IMG:" in (body.html_text or "") def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path): """TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази.""" doc = Document() doc.add_heading("NESPERTCAM Launcher - тест", level=1) doc.add_heading("Съдържание", level=3) _add_hyperlink_paragraph(doc, "1. Преглед на приложението") _add_hyperlink_paragraph(doc, "2. Инсталация и настройка") doc.add_heading("1. Преглед на приложението", level=2) h3 = doc.add_heading("Какво е NESPERTCAM?", level=3) # Heading style → run.bold is None, style.font.bold True assert h3.runs and h3.runs[0].bold is not True p = doc.add_paragraph() p.add_run("Тяло с достатъчно думи за секцията на ръководството.") out = tmp_path / "toc_hyperlink.docx" doc.save(out) sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) assert sections preamble = sections[0] assert "Съдържание" in (preamble.text or "") assert "1. Преглед" in (preamble.text or "") html = preamble.html_text or "" assert "Съдържание" in html assert "1. Преглед" in html assert "2. Инсталация" in html overview = next(s for s in sections if s.title.startswith("1. Преглед")) assert "" in (overview.html_text or "") assert "Какво е NESPERTCAM?" in (overview.html_text or "") def test_docx_external_r_link_image(tmp_path: Path): png = _mini_png() img_path = tmp_path / "ext.png" img_path.write_bytes(png) try: from io import BytesIO from PIL import Image assert Image.open(BytesIO(png)).size[0] >= 50 except Exception: pytest.skip("PIL unavailable for size check") doc = Document() doc.add_heading("1. С картинка", level=1) doc.add_paragraph("Преди фигурата.") try: _add_external_blip_paragraph(doc, img_path) except Exception as exc: pytest.skip(f"cannot build r:link drawing: {exc}") doc.add_paragraph("След фигурата.") out = tmp_path / "ext_img.docx" doc.save(out) sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) body = sections[0] assert body.images, "r:link external image must be loaded" assert "[IMG:" in (body.text or "") assert "[IMG:" in (body.html_text or "") # Успешен прочит → опаковане до _media/ packed = tmp_path / "ext_img_media" / "ext.png" assert packed.is_file() assert packed.stat().st_size == len(png) def test_docx_external_r_link_sidecar_when_unc_missing(tmp_path: Path): """Ако r:link сочи към липсващ UNC/път, четем _media/ до .docx.""" png = _mini_png() try: from io import BytesIO from PIL import Image assert Image.open(BytesIO(png)).size[0] >= 50 except Exception: pytest.skip("PIL unavailable for size check") missing = tmp_path / "nowhere" / "missing_shot.png" # Не създаваме missing — само sidecar media = tmp_path / "linked_media" media.mkdir() (media / "missing_shot.png").write_bytes(png) doc = Document() doc.add_heading("1. Sidecar", level=1) try: _add_external_blip_paragraph(doc, missing) except Exception as exc: pytest.skip(f"cannot build r:link drawing: {exc}") # Поправи target към несъществуващ път (get_or_add може да е създал file URI) out = tmp_path / "linked.docx" doc.save(out) # Гарантираме sidecar до записания docx side = tmp_path / "linked_media" side.mkdir(exist_ok=True) (side / "missing_shot.png").write_bytes(png) sections = merge_preamble_sections(merge_short_sections(parse_docx(out))) body = sections[0] assert body.images, "sidecar next to docx must satisfy r:link" assert "[IMG:" in (body.html_text or "") @pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted") def test_atra_manual_toc_bold_and_image_in_html(): """Регресия върху реалния NESPERTCAM Launcher manual.""" sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL))) assert len(sections) >= 3 first = sections[0] assert "NESPERTCAM" in first.title assert "Съдържание" in (first.text or "") assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "") html0 = first.html_text or "" assert "Съдържание" in html0 assert "1. Преглед" in html0 or "Инсталация" in html0 # Style-bold подзаглавия + изричен bold по-нататък assert any("" in (s.html_text or "") for s in sections) overview = next((s for s in sections if "Преглед" in s.title), None) assert overview is not None assert "Фигура 1" in (overview.text or "") # Drawing е ПРЕДИ caption; трябва [IMG:] ако UNC или atra-manual_media/123.png unc = Path(r"\\192.168.88.15\tmp\___Proekti\2025 ATRA96\otchitane\123.png") sidecar = ATRA_MANUAL.parent / "atra-manual_media" / "123.png" can_load = False try: can_load = unc.is_file() or sidecar.is_file() except OSError: can_load = sidecar.is_file() if can_load: assert overview.images, "Фигура 1 image must be extracted" assert "[IMG:" in (overview.text or "") assert "[IMG:" in (overview.html_text or "") # Ред: картинка, после caption (както в Word) t = overview.text or "" assert t.find("[IMG:") < t.find("Фигура 1") else: assert "Фигура 1" in (overview.text or "") def test_html_preserves_bold_and_style_color(tmp_path: Path): html = tmp_path / "rich.html" html.write_text( """

1. Цветове

Обикновен bold и зелен плюс син.

""", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert sections sec = sections[0] assert "bold" in sec.text html_body = sec.html_text or "" assert "" in html_body or "" in html_body assert "color:#00aa00" in html_body or "color: #00aa00" in html_body assert "font-size" not in html_body # декоративното се маха assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body def test_clean_text_strips_nul_keeps_bullets(): assert "\x00" not in clean_text("a\x00b\tc") assert "•" in clean_text("елемент • едно") assert "→" in clean_text("A → B")