"""HTML от Word/LibreOffice „Save as…“: относителни картинки + multi-section split.""" from pathlib import Path from help_processor import ( merge_preamble_sections, merge_short_sections, parse_html, ) def _mini_png(w: int = 64, h: int = 64) -> bytes: try: from io import BytesIO from PIL import Image buf = BytesIO() Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG") return buf.getvalue() except Exception: return ( b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01" b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00" b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N" b"\x00\x00\x00\x00IEND\xaeB`\x82" ) def test_html_relative_img_url_encoded(tmp_path: Path): """../ + %20 в src се резолвират спрямо HTML папката (без UNC).""" media = tmp_path / "___Proekti" / "2025 ATRA96" / "otchitane" media.mkdir(parents=True) png = _mini_png() (media / "123.png").write_bytes(png) html_dir = tmp_path / "docs" html_dir.mkdir() html = html_dir / "manual.html" html.write_text( """
Фигура 1: екран с достатъчно текст за тяло на секцията тук.
""", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert len(sections) >= 1 body = sections[0] assert body.images, "relative URL-encoded img must load from disk" assert "[IMG:" in (body.text or "") assert "[IMG:" in (body.html_text or "") def test_html_missing_img_logs_warning(tmp_path: Path, caplog): html = tmp_path / "gone.html" html.write_text( """
Тяло без картинка, но с достатъчно думи за секция едно две три.
""", encoding="utf-8", ) import logging with caplog.at_level(logging.WARNING, logger="help_processor"): sections = parse_html(html) assert sections assert not sections[0].images assert any("HTML image not found" in r.message for r in caplog.records) def test_html_word_softbreak_h2_chapters_split(tmp_path: Path): """CR/LF вътре в H2 (типично Save-as HTML) не трябва да остави 1 мега-секция.""" html = tmp_path / "softbreak.html" # Имитира atra-manual.html: H1 корица, Съдържание, H2 с пренос на ред в заглавието html.write_text( "\n" '1. Преглед на приложението
2. Инсталация и настройка
Какво е NESPERTCAM Launcher с достатъчно думи в тялото на главата едно.
\n" 'Системни изисквания Windows 10 и още текст за втората глава тук.
\n" "", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert len(sections) >= 3, f"expected preamble+chapters, got {_titles(sections)}" assert "NESPERTCAM" in sections[0].title assert "Съдържание" in sections[0].text assert sections[1].title == "1. Преглед на приложението" assert sections[2].title == "2. Инсталация и настройка" assert "\r" not in sections[1].title and "\n" not in sections[1].title def test_html_mso_normal_numbered_chapters(tmp_path: Path): """Word MsoNormal / bold глави 1. / 2. режат секции след TOC.""" html = tmp_path / "mso.html" html.write_text( """Ръководство X
Съдържание
1. Увод
2. Край
1. Увод
Текст на увода с няколко думи повече от минимум.
2. Край
Текст на края с няколко думи повече от минимум.
""", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert len(sections) >= 3 assert sections[0].title == "Ръководство X" assert "Съдържание" in sections[0].text assert sections[1].title == "1. Увод" assert sections[2].title == "2. Край" def _titles(sections): return [s.title for s in sections]