326 lines
12 KiB
Python
326 lines
12 KiB
Python
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
|
||
from pathlib import Path
|
||
|
||
import pytest
|
||
from docx import Document
|
||
from docx.oxml import OxmlElement
|
||
from docx.oxml.ns import qn
|
||
from docx.shared import RGBColor
|
||
|
||
from help_processor import (
|
||
clean_text,
|
||
merge_preamble_sections,
|
||
merge_short_sections,
|
||
parse_docx,
|
||
parse_html,
|
||
)
|
||
|
||
FIXTURES = Path(__file__).parent / "fixtures"
|
||
ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx")
|
||
|
||
|
||
def _mini_png(w: int = 64, h: int = 64) -> bytes:
|
||
"""Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста."""
|
||
try:
|
||
from io import BytesIO
|
||
from PIL import Image
|
||
|
||
buf = BytesIO()
|
||
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
|
||
return buf.getvalue()
|
||
except Exception:
|
||
# 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично
|
||
return (
|
||
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
|
||
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
|
||
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
|
||
b"\x00\x00\x00\x00IEND\xaeB`\x82"
|
||
)
|
||
|
||
|
||
def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"):
|
||
"""TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == [])."""
|
||
p = doc.add_paragraph()
|
||
for child in list(p._element):
|
||
if child.tag == qn("w:r"):
|
||
p._element.remove(child)
|
||
hl = OxmlElement("w:hyperlink")
|
||
hl.set(qn("w:anchor"), anchor)
|
||
r = OxmlElement("w:r")
|
||
t = OxmlElement("w:t")
|
||
t.set(qn("xml:space"), "preserve")
|
||
t.text = text
|
||
r.append(t)
|
||
hl.append(r)
|
||
p._element.append(hl)
|
||
assert len(p.runs) == 0
|
||
assert text in (p.text or "")
|
||
return p
|
||
|
||
|
||
def _add_external_blip_paragraph(doc: Document, image_path: Path):
|
||
"""Параграф с a:blip/@r:link към външен файл (като atra-manual.docx)."""
|
||
from docx.opc.constants import RELATIONSHIP_TYPE as RT
|
||
|
||
rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri())
|
||
rId = rel if isinstance(rel, str) else rel.rId
|
||
|
||
p = doc.add_paragraph()
|
||
r = OxmlElement("w:r")
|
||
drawing = OxmlElement("w:drawing")
|
||
inline = OxmlElement("wp:inline")
|
||
inline.set("distT", "0")
|
||
inline.set("distB", "0")
|
||
inline.set("distL", "0")
|
||
inline.set("distR", "0")
|
||
extent = OxmlElement("wp:extent")
|
||
extent.set("cx", "914400")
|
||
extent.set("cy", "914400")
|
||
inline.append(extent)
|
||
docPr = OxmlElement("wp:docPr")
|
||
docPr.set("id", "1")
|
||
docPr.set("name", "Picture")
|
||
inline.append(docPr)
|
||
graphic = OxmlElement("a:graphic")
|
||
graphicData = OxmlElement("a:graphicData")
|
||
graphicData.set(
|
||
"uri", "http://schemas.openxmlformats.org/drawingml/2006/picture"
|
||
)
|
||
pic = OxmlElement("pic:pic")
|
||
nvPicPr = OxmlElement("pic:nvPicPr")
|
||
cNvPr = OxmlElement("pic:cNvPr")
|
||
cNvPr.set("id", "0")
|
||
cNvPr.set("name", "pic")
|
||
nvPicPr.append(cNvPr)
|
||
nvPicPr.append(OxmlElement("pic:cNvPicPr"))
|
||
pic.append(nvPicPr)
|
||
blipFill = OxmlElement("pic:blipFill")
|
||
blip = OxmlElement("a:blip")
|
||
blip.set(qn("r:link"), rId)
|
||
blipFill.append(blip)
|
||
blipFill.append(OxmlElement("a:stretch"))
|
||
pic.append(blipFill)
|
||
spPr = OxmlElement("pic:spPr")
|
||
pic.append(spPr)
|
||
graphicData.append(pic)
|
||
graphic.append(graphicData)
|
||
inline.append(graphic)
|
||
drawing.append(inline)
|
||
r.append(drawing)
|
||
p._element.append(r)
|
||
return p
|
||
|
||
|
||
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
|
||
doc = Document()
|
||
doc.add_heading("1. Форматиране", level=1)
|
||
p = doc.add_paragraph()
|
||
r1 = p.add_run("Bold ")
|
||
r1.bold = True
|
||
r2 = p.add_run("и червено")
|
||
r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00)
|
||
p2 = doc.add_paragraph()
|
||
p2.add_run("Стрелка → и булет • в текста")
|
||
|
||
img_path = tmp_path / "pic.png"
|
||
png = _mini_png()
|
||
img_path.write_bytes(png)
|
||
try:
|
||
from io import BytesIO
|
||
from PIL import Image
|
||
|
||
assert Image.open(BytesIO(png)).size[0] >= 50
|
||
doc.add_picture(str(img_path))
|
||
expect_img = True
|
||
except Exception:
|
||
expect_img = False
|
||
|
||
out = tmp_path / "rich.docx"
|
||
doc.save(out)
|
||
|
||
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||
assert len(sections) >= 1
|
||
body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text))
|
||
assert "Bold" in body.text
|
||
assert "→" in body.text
|
||
assert "•" in body.text
|
||
assert "<b>" in (body.html_text or "")
|
||
assert "color:#C00000" in (body.html_text or "")
|
||
# Plain text без HTML тагове — за Claude / keywords
|
||
assert "<b>" not in body.text
|
||
if expect_img:
|
||
assert body.images
|
||
assert "[IMG:" in body.text
|
||
assert "[IMG:" in (body.html_text or "")
|
||
|
||
|
||
def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path):
|
||
"""TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази."""
|
||
doc = Document()
|
||
doc.add_heading("NESPERTCAM Launcher - тест", level=1)
|
||
doc.add_heading("Съдържание", level=3)
|
||
_add_hyperlink_paragraph(doc, "1. Преглед на приложението")
|
||
_add_hyperlink_paragraph(doc, "2. Инсталация и настройка")
|
||
doc.add_heading("1. Преглед на приложението", level=2)
|
||
h3 = doc.add_heading("Какво е NESPERTCAM?", level=3)
|
||
# Heading style → run.bold is None, style.font.bold True
|
||
assert h3.runs and h3.runs[0].bold is not True
|
||
p = doc.add_paragraph()
|
||
p.add_run("Тяло с достатъчно думи за секцията на ръководството.")
|
||
|
||
out = tmp_path / "toc_hyperlink.docx"
|
||
doc.save(out)
|
||
|
||
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||
assert sections
|
||
preamble = sections[0]
|
||
assert "Съдържание" in (preamble.text or "")
|
||
assert "1. Преглед" in (preamble.text or "")
|
||
html = preamble.html_text or ""
|
||
assert "Съдържание" in html
|
||
assert "1. Преглед" in html
|
||
assert "2. Инсталация" in html
|
||
|
||
overview = next(s for s in sections if s.title.startswith("1. Преглед"))
|
||
assert "<b>" in (overview.html_text or "")
|
||
assert "Какво е NESPERTCAM?" in (overview.html_text or "")
|
||
|
||
|
||
def test_docx_external_r_link_image(tmp_path: Path):
|
||
png = _mini_png()
|
||
img_path = tmp_path / "ext.png"
|
||
img_path.write_bytes(png)
|
||
try:
|
||
from io import BytesIO
|
||
from PIL import Image
|
||
|
||
assert Image.open(BytesIO(png)).size[0] >= 50
|
||
except Exception:
|
||
pytest.skip("PIL unavailable for size check")
|
||
|
||
doc = Document()
|
||
doc.add_heading("1. С картинка", level=1)
|
||
doc.add_paragraph("Преди фигурата.")
|
||
try:
|
||
_add_external_blip_paragraph(doc, img_path)
|
||
except Exception as exc:
|
||
pytest.skip(f"cannot build r:link drawing: {exc}")
|
||
doc.add_paragraph("След фигурата.")
|
||
|
||
out = tmp_path / "ext_img.docx"
|
||
doc.save(out)
|
||
|
||
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||
body = sections[0]
|
||
assert body.images, "r:link external image must be loaded"
|
||
assert "[IMG:" in (body.text or "")
|
||
assert "[IMG:" in (body.html_text or "")
|
||
# Успешен прочит → опаковане до <stem>_media/
|
||
packed = tmp_path / "ext_img_media" / "ext.png"
|
||
assert packed.is_file()
|
||
assert packed.stat().st_size == len(png)
|
||
|
||
|
||
def test_docx_external_r_link_sidecar_when_unc_missing(tmp_path: Path):
|
||
"""Ако r:link сочи към липсващ UNC/път, четем <stem>_media/<name> до .docx."""
|
||
png = _mini_png()
|
||
try:
|
||
from io import BytesIO
|
||
from PIL import Image
|
||
|
||
assert Image.open(BytesIO(png)).size[0] >= 50
|
||
except Exception:
|
||
pytest.skip("PIL unavailable for size check")
|
||
|
||
missing = tmp_path / "nowhere" / "missing_shot.png"
|
||
# Не създаваме missing — само sidecar
|
||
media = tmp_path / "linked_media"
|
||
media.mkdir()
|
||
(media / "missing_shot.png").write_bytes(png)
|
||
|
||
doc = Document()
|
||
doc.add_heading("1. Sidecar", level=1)
|
||
try:
|
||
_add_external_blip_paragraph(doc, missing)
|
||
except Exception as exc:
|
||
pytest.skip(f"cannot build r:link drawing: {exc}")
|
||
# Поправи target към несъществуващ път (get_or_add може да е създал file URI)
|
||
out = tmp_path / "linked.docx"
|
||
doc.save(out)
|
||
|
||
# Гарантираме sidecar до записания docx
|
||
side = tmp_path / "linked_media"
|
||
side.mkdir(exist_ok=True)
|
||
(side / "missing_shot.png").write_bytes(png)
|
||
|
||
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||
body = sections[0]
|
||
assert body.images, "sidecar next to docx must satisfy r:link"
|
||
assert "[IMG:" in (body.html_text or "")
|
||
|
||
|
||
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
|
||
def test_atra_manual_toc_bold_and_image_in_html():
|
||
"""Регресия върху реалния NESPERTCAM Launcher manual."""
|
||
sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL)))
|
||
assert len(sections) >= 3
|
||
first = sections[0]
|
||
assert "NESPERTCAM" in first.title
|
||
assert "Съдържание" in (first.text or "")
|
||
assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "")
|
||
html0 = first.html_text or ""
|
||
assert "Съдържание" in html0
|
||
assert "1. Преглед" in html0 or "Инсталация" in html0
|
||
|
||
# Style-bold подзаглавия + изричен bold по-нататък
|
||
assert any("<b>" in (s.html_text or "") for s in sections)
|
||
|
||
overview = next((s for s in sections if "Преглед" in s.title), None)
|
||
assert overview is not None
|
||
assert "Фигура 1" in (overview.text or "")
|
||
# Drawing е ПРЕДИ caption; трябва [IMG:] ако UNC или atra-manual_media/123.png
|
||
unc = Path(r"\\192.168.88.15\tmp\___Proekti\2025 ATRA96\otchitane\123.png")
|
||
sidecar = ATRA_MANUAL.parent / "atra-manual_media" / "123.png"
|
||
can_load = False
|
||
try:
|
||
can_load = unc.is_file() or sidecar.is_file()
|
||
except OSError:
|
||
can_load = sidecar.is_file()
|
||
if can_load:
|
||
assert overview.images, "Фигура 1 image must be extracted"
|
||
assert "[IMG:" in (overview.text or "")
|
||
assert "[IMG:" in (overview.html_text or "")
|
||
# Ред: картинка, после caption (както в Word)
|
||
t = overview.text or ""
|
||
assert t.find("[IMG:") < t.find("Фигура 1")
|
||
else:
|
||
assert "Фигура 1" in (overview.text or "")
|
||
|
||
|
||
def test_html_preserves_bold_and_style_color(tmp_path: Path):
|
||
html = tmp_path / "rich.html"
|
||
html.write_text(
|
||
"""<!DOCTYPE html><html><body>
|
||
<h1>1. Цветове</h1>
|
||
<p>Обикновен <b>bold</b> и
|
||
<span style="color:#00aa00; font-size:20px">зелен</span>
|
||
плюс <font color="#0000cc">син</font>.</p>
|
||
</body></html>""",
|
||
encoding="utf-8",
|
||
)
|
||
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
|
||
assert sections
|
||
sec = sections[0]
|
||
assert "bold" in sec.text
|
||
html_body = sec.html_text or ""
|
||
assert "<b>" in html_body or "<strong>" in html_body
|
||
assert "color:#00aa00" in html_body or "color: #00aa00" in html_body
|
||
assert "font-size" not in html_body # декоративното се маха
|
||
assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body
|
||
|
||
|
||
def test_clean_text_strips_nul_keeps_bullets():
|
||
assert "\x00" not in clean_text("a\x00b\tc")
|
||
assert "•" in clean_text("елемент • едно")
|
||
assert "→" in clean_text("A → B")
|