"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
from pathlib import Path
import pytest
from docx import Document
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import RGBColor
from help_processor import (
clean_text,
merge_preamble_sections,
merge_short_sections,
parse_docx,
parse_html,
)
FIXTURES = Path(__file__).parent / "fixtures"
ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx")
def _mini_png(w: int = 64, h: int = 64) -> bytes:
"""Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста."""
try:
from io import BytesIO
from PIL import Image
buf = BytesIO()
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
return buf.getvalue()
except Exception:
# 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично
return (
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
b"\x00\x00\x00\x00IEND\xaeB`\x82"
)
def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"):
"""TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == [])."""
p = doc.add_paragraph()
for child in list(p._element):
if child.tag == qn("w:r"):
p._element.remove(child)
hl = OxmlElement("w:hyperlink")
hl.set(qn("w:anchor"), anchor)
r = OxmlElement("w:r")
t = OxmlElement("w:t")
t.set(qn("xml:space"), "preserve")
t.text = text
r.append(t)
hl.append(r)
p._element.append(hl)
assert len(p.runs) == 0
assert text in (p.text or "")
return p
def _add_external_blip_paragraph(doc: Document, image_path: Path):
"""Параграф с a:blip/@r:link към външен файл (като atra-manual.docx)."""
from docx.opc.constants import RELATIONSHIP_TYPE as RT
rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri())
rId = rel if isinstance(rel, str) else rel.rId
p = doc.add_paragraph()
r = OxmlElement("w:r")
drawing = OxmlElement("w:drawing")
inline = OxmlElement("wp:inline")
inline.set("distT", "0")
inline.set("distB", "0")
inline.set("distL", "0")
inline.set("distR", "0")
extent = OxmlElement("wp:extent")
extent.set("cx", "914400")
extent.set("cy", "914400")
inline.append(extent)
docPr = OxmlElement("wp:docPr")
docPr.set("id", "1")
docPr.set("name", "Picture")
inline.append(docPr)
graphic = OxmlElement("a:graphic")
graphicData = OxmlElement("a:graphicData")
graphicData.set(
"uri", "http://schemas.openxmlformats.org/drawingml/2006/picture"
)
pic = OxmlElement("pic:pic")
nvPicPr = OxmlElement("pic:nvPicPr")
cNvPr = OxmlElement("pic:cNvPr")
cNvPr.set("id", "0")
cNvPr.set("name", "pic")
nvPicPr.append(cNvPr)
nvPicPr.append(OxmlElement("pic:cNvPicPr"))
pic.append(nvPicPr)
blipFill = OxmlElement("pic:blipFill")
blip = OxmlElement("a:blip")
blip.set(qn("r:link"), rId)
blipFill.append(blip)
blipFill.append(OxmlElement("a:stretch"))
pic.append(blipFill)
spPr = OxmlElement("pic:spPr")
pic.append(spPr)
graphicData.append(pic)
graphic.append(graphicData)
inline.append(graphic)
drawing.append(inline)
r.append(drawing)
p._element.append(r)
return p
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
doc = Document()
doc.add_heading("1. Форматиране", level=1)
p = doc.add_paragraph()
r1 = p.add_run("Bold ")
r1.bold = True
r2 = p.add_run("и червено")
r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00)
p2 = doc.add_paragraph()
p2.add_run("Стрелка → и булет • в текста")
img_path = tmp_path / "pic.png"
png = _mini_png()
img_path.write_bytes(png)
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
doc.add_picture(str(img_path))
expect_img = True
except Exception:
expect_img = False
out = tmp_path / "rich.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
assert len(sections) >= 1
body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text))
assert "Bold" in body.text
assert "→" in body.text
assert "•" in body.text
assert "" in (body.html_text or "")
assert "color:#C00000" in (body.html_text or "")
# Plain text без HTML тагове — за Claude / keywords
assert "" not in body.text
if expect_img:
assert body.images
assert "[IMG:" in body.text
assert "[IMG:" in (body.html_text or "")
def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path):
"""TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази."""
doc = Document()
doc.add_heading("NESPERTCAM Launcher - тест", level=1)
doc.add_heading("Съдържание", level=3)
_add_hyperlink_paragraph(doc, "1. Преглед на приложението")
_add_hyperlink_paragraph(doc, "2. Инсталация и настройка")
doc.add_heading("1. Преглед на приложението", level=2)
h3 = doc.add_heading("Какво е NESPERTCAM?", level=3)
# Heading style → run.bold is None, style.font.bold True
assert h3.runs and h3.runs[0].bold is not True
p = doc.add_paragraph()
p.add_run("Тяло с достатъчно думи за секцията на ръководството.")
out = tmp_path / "toc_hyperlink.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
assert sections
preamble = sections[0]
assert "Съдържание" in (preamble.text or "")
assert "1. Преглед" in (preamble.text or "")
html = preamble.html_text or ""
assert "Съдържание" in html
assert "1. Преглед" in html
assert "2. Инсталация" in html
overview = next(s for s in sections if s.title.startswith("1. Преглед"))
assert "" in (overview.html_text or "")
assert "Какво е NESPERTCAM?" in (overview.html_text or "")
def test_docx_external_r_link_image(tmp_path: Path):
png = _mini_png()
img_path = tmp_path / "ext.png"
img_path.write_bytes(png)
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
except Exception:
pytest.skip("PIL unavailable for size check")
doc = Document()
doc.add_heading("1. С картинка", level=1)
doc.add_paragraph("Преди фигурата.")
try:
_add_external_blip_paragraph(doc, img_path)
except Exception as exc:
pytest.skip(f"cannot build r:link drawing: {exc}")
doc.add_paragraph("След фигурата.")
out = tmp_path / "ext_img.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
body = sections[0]
assert body.images, "r:link external image must be loaded"
assert "[IMG:" in (body.text or "")
assert "[IMG:" in (body.html_text or "")
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
def test_atra_manual_toc_bold_and_image_in_html():
"""Регресия върху реалния NESPERTCAM Launcher manual."""
sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL)))
assert len(sections) >= 3
first = sections[0]
assert "NESPERTCAM" in first.title
assert "Съдържание" in (first.text or "")
assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "")
html0 = first.html_text or ""
assert "Съдържание" in html0
assert "1. Преглед" in html0 or "Инсталация" in html0
# Style-bold подзаглавия + изричен bold по-нататък
assert any("" in (s.html_text or "") for s in sections)
# Фигура 1 е външен UNC линк — ако share-ът е достъпен, трябва [IMG:]
overview = next((s for s in sections if "Преглед" in s.title), None)
assert overview is not None
if overview.images or "[IMG:" in (overview.text or ""):
assert "[IMG:" in (overview.html_text or "")
else:
# Документът сочи към външен файл; ако UNC липсва, plain caption остава
assert "Фигура 1" in (overview.text or "")
def test_html_preserves_bold_and_style_color(tmp_path: Path):
html = tmp_path / "rich.html"
html.write_text(
"""
1. Цветове
Обикновен bold и зелен плюс син.
""", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert sections sec = sections[0] assert "bold" in sec.text html_body = sec.html_text or "" assert "" in html_body or "" in html_body assert "color:#00aa00" in html_body or "color: #00aa00" in html_body assert "font-size" not in html_body # декоративното се маха assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body def test_clean_text_strips_nul_keeps_bullets(): assert "\x00" not in clean_text("a\x00b\tc") assert "•" in clean_text("елемент • едно") assert "→" in clean_text("A → B")