Files
rip-help-system/tests/test_rich_content.py
2026-09-14 12:01:04 +03:00

326 lines
12 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
from pathlib import Path
import pytest
from docx import Document
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import RGBColor
from help_processor import (
clean_text,
merge_preamble_sections,
merge_short_sections,
parse_docx,
parse_html,
)
FIXTURES = Path(__file__).parent / "fixtures"
ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx")
def _mini_png(w: int = 64, h: int = 64) -> bytes:
"""Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста."""
try:
from io import BytesIO
from PIL import Image
buf = BytesIO()
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
return buf.getvalue()
except Exception:
# 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично
return (
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
b"\x00\x00\x00\x00IEND\xaeB`\x82"
)
def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"):
"""TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == [])."""
p = doc.add_paragraph()
for child in list(p._element):
if child.tag == qn("w:r"):
p._element.remove(child)
hl = OxmlElement("w:hyperlink")
hl.set(qn("w:anchor"), anchor)
r = OxmlElement("w:r")
t = OxmlElement("w:t")
t.set(qn("xml:space"), "preserve")
t.text = text
r.append(t)
hl.append(r)
p._element.append(hl)
assert len(p.runs) == 0
assert text in (p.text or "")
return p
def _add_external_blip_paragraph(doc: Document, image_path: Path):
"""Параграф с a:blip/@r:link към външен файл (като atra-manual.docx)."""
from docx.opc.constants import RELATIONSHIP_TYPE as RT
rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri())
rId = rel if isinstance(rel, str) else rel.rId
p = doc.add_paragraph()
r = OxmlElement("w:r")
drawing = OxmlElement("w:drawing")
inline = OxmlElement("wp:inline")
inline.set("distT", "0")
inline.set("distB", "0")
inline.set("distL", "0")
inline.set("distR", "0")
extent = OxmlElement("wp:extent")
extent.set("cx", "914400")
extent.set("cy", "914400")
inline.append(extent)
docPr = OxmlElement("wp:docPr")
docPr.set("id", "1")
docPr.set("name", "Picture")
inline.append(docPr)
graphic = OxmlElement("a:graphic")
graphicData = OxmlElement("a:graphicData")
graphicData.set(
"uri", "http://schemas.openxmlformats.org/drawingml/2006/picture"
)
pic = OxmlElement("pic:pic")
nvPicPr = OxmlElement("pic:nvPicPr")
cNvPr = OxmlElement("pic:cNvPr")
cNvPr.set("id", "0")
cNvPr.set("name", "pic")
nvPicPr.append(cNvPr)
nvPicPr.append(OxmlElement("pic:cNvPicPr"))
pic.append(nvPicPr)
blipFill = OxmlElement("pic:blipFill")
blip = OxmlElement("a:blip")
blip.set(qn("r:link"), rId)
blipFill.append(blip)
blipFill.append(OxmlElement("a:stretch"))
pic.append(blipFill)
spPr = OxmlElement("pic:spPr")
pic.append(spPr)
graphicData.append(pic)
graphic.append(graphicData)
inline.append(graphic)
drawing.append(inline)
r.append(drawing)
p._element.append(r)
return p
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
doc = Document()
doc.add_heading("1. Форматиране", level=1)
p = doc.add_paragraph()
r1 = p.add_run("Bold ")
r1.bold = True
r2 = p.add_run("и червено")
r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00)
p2 = doc.add_paragraph()
p2.add_run("Стрелка → и булет • в текста")
img_path = tmp_path / "pic.png"
png = _mini_png()
img_path.write_bytes(png)
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
doc.add_picture(str(img_path))
expect_img = True
except Exception:
expect_img = False
out = tmp_path / "rich.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
assert len(sections) >= 1
body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text))
assert "Bold" in body.text
assert "→" in body.text
assert "•" in body.text
assert "<b>" in (body.html_text or "")
assert "color:#C00000" in (body.html_text or "")
# Plain text без HTML тагове — за Claude / keywords
assert "<b>" not in body.text
if expect_img:
assert body.images
assert "[IMG:" in body.text
assert "[IMG:" in (body.html_text or "")
def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path):
"""TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази."""
doc = Document()
doc.add_heading("NESPERTCAM Launcher - тест", level=1)
doc.add_heading("Съдържание", level=3)
_add_hyperlink_paragraph(doc, "1. Преглед на приложението")
_add_hyperlink_paragraph(doc, "2. Инсталация и настройка")
doc.add_heading("1. Преглед на приложението", level=2)
h3 = doc.add_heading("Какво е NESPERTCAM?", level=3)
# Heading style → run.bold is None, style.font.bold True
assert h3.runs and h3.runs[0].bold is not True
p = doc.add_paragraph()
p.add_run("Тяло с достатъчно думи за секцията на ръководството.")
out = tmp_path / "toc_hyperlink.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
assert sections
preamble = sections[0]
assert "Съдържание" in (preamble.text or "")
assert "1. Преглед" in (preamble.text or "")
html = preamble.html_text or ""
assert "Съдържание" in html
assert "1. Преглед" in html
assert "2. Инсталация" in html
overview = next(s for s in sections if s.title.startswith("1. Преглед"))
assert "<b>" in (overview.html_text or "")
assert "Какво е NESPERTCAM?" in (overview.html_text or "")
def test_docx_external_r_link_image(tmp_path: Path):
png = _mini_png()
img_path = tmp_path / "ext.png"
img_path.write_bytes(png)
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
except Exception:
pytest.skip("PIL unavailable for size check")
doc = Document()
doc.add_heading("1. С картинка", level=1)
doc.add_paragraph("Преди фигурата.")
try:
_add_external_blip_paragraph(doc, img_path)
except Exception as exc:
pytest.skip(f"cannot build r:link drawing: {exc}")
doc.add_paragraph("След фигурата.")
out = tmp_path / "ext_img.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
body = sections[0]
assert body.images, "r:link external image must be loaded"
assert "[IMG:" in (body.text or "")
assert "[IMG:" in (body.html_text or "")
# Успешен прочит → опаковане до <stem>_media/
packed = tmp_path / "ext_img_media" / "ext.png"
assert packed.is_file()
assert packed.stat().st_size == len(png)
def test_docx_external_r_link_sidecar_when_unc_missing(tmp_path: Path):
"""Ако r:link сочи към липсващ UNC/път, четем <stem>_media/<name> до .docx."""
png = _mini_png()
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
except Exception:
pytest.skip("PIL unavailable for size check")
missing = tmp_path / "nowhere" / "missing_shot.png"
# Не създаваме missing — само sidecar
media = tmp_path / "linked_media"
media.mkdir()
(media / "missing_shot.png").write_bytes(png)
doc = Document()
doc.add_heading("1. Sidecar", level=1)
try:
_add_external_blip_paragraph(doc, missing)
except Exception as exc:
pytest.skip(f"cannot build r:link drawing: {exc}")
# Поправи target към несъществуващ път (get_or_add може да е създал file URI)
out = tmp_path / "linked.docx"
doc.save(out)
# Гарантираме sidecar до записания docx
side = tmp_path / "linked_media"
side.mkdir(exist_ok=True)
(side / "missing_shot.png").write_bytes(png)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
body = sections[0]
assert body.images, "sidecar next to docx must satisfy r:link"
assert "[IMG:" in (body.html_text or "")
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
def test_atra_manual_toc_bold_and_image_in_html():
"""Регресия върху реалния NESPERTCAM Launcher manual."""
sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL)))
assert len(sections) >= 3
first = sections[0]
assert "NESPERTCAM" in first.title
assert "Съдържание" in (first.text or "")
assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "")
html0 = first.html_text or ""
assert "Съдържание" in html0
assert "1. Преглед" in html0 or "Инсталация" in html0
# Style-bold подзаглавия + изричен bold по-нататък
assert any("<b>" in (s.html_text or "") for s in sections)
overview = next((s for s in sections if "Преглед" in s.title), None)
assert overview is not None
assert "Фигура 1" in (overview.text or "")
# Drawing е ПРЕДИ caption; трябва [IMG:] ако UNC или atra-manual_media/123.png
unc = Path(r"\\192.168.88.15\tmp\___Proekti\2025 ATRA96\otchitane\123.png")
sidecar = ATRA_MANUAL.parent / "atra-manual_media" / "123.png"
can_load = False
try:
can_load = unc.is_file() or sidecar.is_file()
except OSError:
can_load = sidecar.is_file()
if can_load:
assert overview.images, "Фигура 1 image must be extracted"
assert "[IMG:" in (overview.text or "")
assert "[IMG:" in (overview.html_text or "")
# Ред: картинка, после caption (както в Word)
t = overview.text or ""
assert t.find("[IMG:") < t.find("Фигура 1")
else:
assert "Фигура 1" in (overview.text or "")
def test_html_preserves_bold_and_style_color(tmp_path: Path):
html = tmp_path / "rich.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Цветове</h1>
<p>Обикновен <b>bold</b> и
<span style="color:#00aa00; font-size:20px">зелен</span>
плюс <font color="#0000cc">син</font>.</p>
</body></html>""",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert sections
sec = sections[0]
assert "bold" in sec.text
html_body = sec.html_text or ""
assert "<b>" in html_body or "<strong>" in html_body
assert "color:#00aa00" in html_body or "color: #00aa00" in html_body
assert "font-size" not in html_body # декоративното се маха
assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body
def test_clean_text_strips_nul_keeps_bullets():
assert "\x00" not in clean_text("a\x00b\tc")
assert "•" in clean_text("елемент • едно")
assert "→" in clean_text("A → B")