Files
rip-help-system/tests/test_rich_content.py
2026-09-14 11:24:21 +03:00

106 lines
3.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
from pathlib import Path
from docx import Document
from docx.shared import RGBColor
from help_processor import (
clean_text,
merge_preamble_sections,
merge_short_sections,
parse_docx,
parse_html,
)
FIXTURES = Path(__file__).parent / "fixtures"
def _mini_png(w: int = 64, h: int = 64) -> bytes:
"""Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста."""
try:
from io import BytesIO
from PIL import Image
buf = BytesIO()
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
return buf.getvalue()
except Exception:
# 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично
return (
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
b"\x00\x00\x00\x00IEND\xaeB`\x82"
)
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
doc = Document()
doc.add_heading("1. Форматиране", level=1)
p = doc.add_paragraph()
r1 = p.add_run("Bold ")
r1.bold = True
r2 = p.add_run("и червено")
r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00)
p2 = doc.add_paragraph()
p2.add_run("Стрелка → и булет • в текста")
img_path = tmp_path / "pic.png"
png = _mini_png()
img_path.write_bytes(png)
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
doc.add_picture(str(img_path))
expect_img = True
except Exception:
expect_img = False
out = tmp_path / "rich.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
assert len(sections) >= 1
body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text))
assert "Bold" in body.text
assert "→" in body.text
assert "•" in body.text
assert "<b>" in (body.html_text or "")
assert "color:#C00000" in (body.html_text or "")
# Plain text без HTML тагове — за Claude / keywords
assert "<b>" not in body.text
if expect_img:
assert body.images
assert "[IMG:" in body.text
assert "[IMG:" in (body.html_text or "")
def test_html_preserves_bold_and_style_color(tmp_path: Path):
html = tmp_path / "rich.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Цветове</h1>
<p>Обикновен <b>bold</b> и
<span style="color:#00aa00; font-size:20px">зелен</span>
плюс <font color="#0000cc">син</font>.</p>
</body></html>""",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert sections
sec = sections[0]
assert "bold" in sec.text
html_body = sec.html_text or ""
assert "<b>" in html_body or "<strong>" in html_body
assert "color:#00aa00" in html_body or "color: #00aa00" in html_body
assert "font-size" not in html_body # декоративното се маха
assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body
def test_clean_text_strips_nul_keeps_bullets():
assert "\x00" not in clean_text("a\x00b\tc")
assert "•" in clean_text("елемент • едно")
assert "→" in clean_text("A → B")