Files
rip-help-system/tests/test_html_word_export.py
2026-09-14 12:57:10 +03:00

133 lines
5.6 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""HTML от Word/LibreOffice „Save as…“: относителни картинки + multi-section split."""
from pathlib import Path
from help_processor import (
merge_preamble_sections,
merge_short_sections,
parse_html,
)
def _mini_png(w: int = 64, h: int = 64) -> bytes:
try:
from io import BytesIO
from PIL import Image
buf = BytesIO()
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
return buf.getvalue()
except Exception:
return (
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
b"\x00\x00\x00\x00IEND\xaeB`\x82"
)
def test_html_relative_img_url_encoded(tmp_path: Path):
"""../ + %20 в src се резолвират спрямо HTML папката (без UNC)."""
media = tmp_path / "___Proekti" / "2025 ATRA96" / "otchitane"
media.mkdir(parents=True)
png = _mini_png()
(media / "123.png").write_bytes(png)
html_dir = tmp_path / "docs"
html_dir.mkdir()
html = html_dir / "manual.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Преглед</h1>
<p class="MsoNormal" style="background:#fafafa">
<img src="../___Proekti/2025%20ATRA96/otchitane/123.png"
name="Picture 1" align="bottom" width="1045" height="523" border="0"/>
</p>
<p>Фигура 1: екран с достатъчно текст за тяло на секцията тук.</p>
</body></html>""",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert len(sections) >= 1
body = sections[0]
assert body.images, "relative URL-encoded img must load from disk"
assert "[IMG:" in (body.text or "")
assert "[IMG:" in (body.html_text or "")
def test_html_missing_img_logs_warning(tmp_path: Path, caplog):
html = tmp_path / "gone.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Alone</h1>
<p><img src="../missing%20dir/nope.png" name="Picture 1"/></p>
<p>Тяло без картинка, но с достатъчно думи за секция едно две три.</p>
</body></html>""",
encoding="utf-8",
)
import logging
with caplog.at_level(logging.WARNING, logger="help_processor"):
sections = parse_html(html)
assert sections
assert not sections[0].images
assert any("HTML image not found" in r.message for r in caplog.records)
def test_html_word_softbreak_h2_chapters_split(tmp_path: Path):
"""CR/LF вътре в H2 (типично Save-as HTML) не трябва да остави 1 мега-секция."""
html = tmp_path / "softbreak.html"
# Имитира atra-manual.html: H1 корица, Съдържание, H2 с пренос на ред в заглавието
html.write_text(
"<!DOCTYPE html><html><body>\n"
'<h1 class="western">NESPERTCAM Launcher - Пълно\r\n'
"ръководство на потребителя</h1>\n"
'<h3 class="western">Съдържание</h3>\n'
"<ul>\n"
"<li><p>1. Преглед на приложението</p></li>\n"
"<li><p>2. Инсталация и настройка</p></li>\n"
"</ul>\n"
'<h2 class="western">1. Преглед на\r\nприложението</h2>\n'
"<p>Какво е NESPERTCAM Launcher с достатъчно думи в тялото на главата едно.</p>\n"
'<h2 class="western">2. Инсталация\r\nи настройка</h2>\n'
"<p>Системни изисквания Windows 10 и още текст за втората глава тук.</p>\n"
"</body></html>",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert len(sections) >= 3, f"expected preamble+chapters, got {_titles(sections)}"
assert "NESPERTCAM" in sections[0].title
assert "Съдържание" in sections[0].text
assert sections[1].title == "1. Преглед на приложението"
assert sections[2].title == "2. Инсталация и настройка"
assert "\r" not in sections[1].title and "\n" not in sections[1].title
def test_html_mso_normal_numbered_chapters(tmp_path: Path):
"""Word MsoNormal / bold глави 1. / 2. режат секции след TOC."""
html = tmp_path / "mso.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<p class=MsoTitle><b>Ръководство X</b></p>
<p class=MsoNormal><b>Съдържание</b></p>
<p class=MsoNormal>1. Увод</p>
<p class=MsoNormal>2. Край</p>
<p class=MsoNormal><b>1. Увод</b></p>
<p class=MsoNormal>Текст на увода с няколко думи повече от минимум.</p>
<p class=MsoNormal><b>2. Край</b></p>
<p class=MsoNormal>Текст на края с няколко думи повече от минимум.</p>
</body></html>""",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert len(sections) >= 3
assert sections[0].title == "Ръководство X"
assert "Съдържание" in sections[0].text
assert sections[1].title == "1. Увод"
assert sections[2].title == "2. Край"
def _titles(sections):
return [s.title for s in sections]