.
This commit is contained in:
132
tests/test_html_word_export.py
Normal file
132
tests/test_html_word_export.py
Normal file
@@ -0,0 +1,132 @@
|
||||
"""HTML от Word/LibreOffice „Save as…“: относителни картинки + multi-section split."""
|
||||
from pathlib import Path
|
||||
|
||||
from help_processor import (
|
||||
merge_preamble_sections,
|
||||
merge_short_sections,
|
||||
parse_html,
|
||||
)
|
||||
|
||||
|
||||
def _mini_png(w: int = 64, h: int = 64) -> bytes:
|
||||
try:
|
||||
from io import BytesIO
|
||||
from PIL import Image
|
||||
|
||||
buf = BytesIO()
|
||||
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
|
||||
return buf.getvalue()
|
||||
except Exception:
|
||||
return (
|
||||
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
|
||||
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
|
||||
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
|
||||
b"\x00\x00\x00\x00IEND\xaeB`\x82"
|
||||
)
|
||||
|
||||
|
||||
def test_html_relative_img_url_encoded(tmp_path: Path):
|
||||
"""../ + %20 в src се резолвират спрямо HTML папката (без UNC)."""
|
||||
media = tmp_path / "___Proekti" / "2025 ATRA96" / "otchitane"
|
||||
media.mkdir(parents=True)
|
||||
png = _mini_png()
|
||||
(media / "123.png").write_bytes(png)
|
||||
|
||||
html_dir = tmp_path / "docs"
|
||||
html_dir.mkdir()
|
||||
html = html_dir / "manual.html"
|
||||
html.write_text(
|
||||
"""<!DOCTYPE html><html><body>
|
||||
<h1>1. Преглед</h1>
|
||||
<p class="MsoNormal" style="background:#fafafa">
|
||||
<img src="../___Proekti/2025%20ATRA96/otchitane/123.png"
|
||||
name="Picture 1" align="bottom" width="1045" height="523" border="0"/>
|
||||
</p>
|
||||
<p>Фигура 1: екран с достатъчно текст за тяло на секцията тук.</p>
|
||||
</body></html>""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
|
||||
assert len(sections) >= 1
|
||||
body = sections[0]
|
||||
assert body.images, "relative URL-encoded img must load from disk"
|
||||
assert "[IMG:" in (body.text or "")
|
||||
assert "[IMG:" in (body.html_text or "")
|
||||
|
||||
|
||||
def test_html_missing_img_logs_warning(tmp_path: Path, caplog):
|
||||
html = tmp_path / "gone.html"
|
||||
html.write_text(
|
||||
"""<!DOCTYPE html><html><body>
|
||||
<h1>1. Alone</h1>
|
||||
<p><img src="../missing%20dir/nope.png" name="Picture 1"/></p>
|
||||
<p>Тяло без картинка, но с достатъчно думи за секция едно две три.</p>
|
||||
</body></html>""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
import logging
|
||||
|
||||
with caplog.at_level(logging.WARNING, logger="help_processor"):
|
||||
sections = parse_html(html)
|
||||
assert sections
|
||||
assert not sections[0].images
|
||||
assert any("HTML image not found" in r.message for r in caplog.records)
|
||||
|
||||
|
||||
def test_html_word_softbreak_h2_chapters_split(tmp_path: Path):
|
||||
"""CR/LF вътре в H2 (типично Save-as HTML) не трябва да остави 1 мега-секция."""
|
||||
html = tmp_path / "softbreak.html"
|
||||
# Имитира atra-manual.html: H1 корица, Съдържание, H2 с пренос на ред в заглавието
|
||||
html.write_text(
|
||||
"<!DOCTYPE html><html><body>\n"
|
||||
'<h1 class="western">NESPERTCAM Launcher - Пълно\r\n'
|
||||
"ръководство на потребителя</h1>\n"
|
||||
'<h3 class="western">Съдържание</h3>\n'
|
||||
"<ul>\n"
|
||||
"<li><p>1. Преглед на приложението</p></li>\n"
|
||||
"<li><p>2. Инсталация и настройка</p></li>\n"
|
||||
"</ul>\n"
|
||||
'<h2 class="western">1. Преглед на\r\nприложението</h2>\n'
|
||||
"<p>Какво е NESPERTCAM Launcher с достатъчно думи в тялото на главата едно.</p>\n"
|
||||
'<h2 class="western">2. Инсталация\r\nи настройка</h2>\n'
|
||||
"<p>Системни изисквания Windows 10 и още текст за втората глава тук.</p>\n"
|
||||
"</body></html>",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
|
||||
assert len(sections) >= 3, f"expected preamble+chapters, got {_titles(sections)}"
|
||||
assert "NESPERTCAM" in sections[0].title
|
||||
assert "Съдържание" in sections[0].text
|
||||
assert sections[1].title == "1. Преглед на приложението"
|
||||
assert sections[2].title == "2. Инсталация и настройка"
|
||||
assert "\r" not in sections[1].title and "\n" not in sections[1].title
|
||||
|
||||
|
||||
def test_html_mso_normal_numbered_chapters(tmp_path: Path):
|
||||
"""Word MsoNormal / bold глави 1. / 2. режат секции след TOC."""
|
||||
html = tmp_path / "mso.html"
|
||||
html.write_text(
|
||||
"""<!DOCTYPE html><html><body>
|
||||
<p class=MsoTitle><b>Ръководство X</b></p>
|
||||
<p class=MsoNormal><b>Съдържание</b></p>
|
||||
<p class=MsoNormal>1. Увод</p>
|
||||
<p class=MsoNormal>2. Край</p>
|
||||
<p class=MsoNormal><b>1. Увод</b></p>
|
||||
<p class=MsoNormal>Текст на увода с няколко думи повече от минимум.</p>
|
||||
<p class=MsoNormal><b>2. Край</b></p>
|
||||
<p class=MsoNormal>Текст на края с няколко думи повече от минимум.</p>
|
||||
</body></html>""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
|
||||
assert len(sections) >= 3
|
||||
assert sections[0].title == "Ръководство X"
|
||||
assert "Съдържание" in sections[0].text
|
||||
assert sections[1].title == "1. Увод"
|
||||
assert sections[2].title == "2. Край"
|
||||
|
||||
|
||||
def _titles(sections):
|
||||
return [s.title for s in sections]
|
||||
Reference in New Issue
Block a user