"""HTML от Word/LibreOffice „Save as…“: относителни картинки + multi-section split.""" from pathlib import Path from help_processor import ( merge_preamble_sections, merge_short_sections, parse_html, ) def _mini_png(w: int = 64, h: int = 64) -> bytes: try: from io import BytesIO from PIL import Image buf = BytesIO() Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG") return buf.getvalue() except Exception: return ( b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01" b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00" b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N" b"\x00\x00\x00\x00IEND\xaeB`\x82" ) def test_html_relative_img_url_encoded(tmp_path: Path): """../ + %20 в src се резолвират спрямо HTML папката (без UNC).""" media = tmp_path / "___Proekti" / "2025 ATRA96" / "otchitane" media.mkdir(parents=True) png = _mini_png() (media / "123.png").write_bytes(png) html_dir = tmp_path / "docs" html_dir.mkdir() html = html_dir / "manual.html" html.write_text( """

1. Преглед

Фигура 1: екран с достатъчно текст за тяло на секцията тук.

""", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert len(sections) >= 1 body = sections[0] assert body.images, "relative URL-encoded img must load from disk" assert "[IMG:" in (body.text or "") assert "[IMG:" in (body.html_text or "") def test_html_missing_img_logs_warning(tmp_path: Path, caplog): html = tmp_path / "gone.html" html.write_text( """

1. Alone

Тяло без картинка, но с достатъчно думи за секция едно две три.

""", encoding="utf-8", ) import logging with caplog.at_level(logging.WARNING, logger="help_processor"): sections = parse_html(html) assert sections assert not sections[0].images assert any("HTML image not found" in r.message for r in caplog.records) def test_html_word_softbreak_h2_chapters_split(tmp_path: Path): """CR/LF вътре в H2 (типично Save-as HTML) не трябва да остави 1 мега-секция.""" html = tmp_path / "softbreak.html" # Имитира atra-manual.html: H1 корица, Съдържание, H2 с пренос на ред в заглавието html.write_text( "\n" '

NESPERTCAM Launcher - Пълно\r\n' "ръководство на потребителя

\n" '

Съдържание

\n' "\n" '

1. Преглед на\r\nприложението

\n' "

Какво е NESPERTCAM Launcher с достатъчно думи в тялото на главата едно.

\n" '

2. Инсталация\r\nи настройка

\n' "

Системни изисквания Windows 10 и още текст за втората глава тук.

\n" "", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert len(sections) >= 3, f"expected preamble+chapters, got {_titles(sections)}" assert "NESPERTCAM" in sections[0].title assert "Съдържание" in sections[0].text assert sections[1].title == "1. Преглед на приложението" assert sections[2].title == "2. Инсталация и настройка" assert "\r" not in sections[1].title and "\n" not in sections[1].title def test_html_mso_normal_numbered_chapters(tmp_path: Path): """Word MsoNormal / bold глави 1. / 2. режат секции след TOC.""" html = tmp_path / "mso.html" html.write_text( """

Ръководство X

Съдържание

1. Увод

2. Край

1. Увод

Текст на увода с няколко думи повече от минимум.

2. Край

Текст на края с няколко думи повече от минимум.

""", encoding="utf-8", ) sections = merge_preamble_sections(merge_short_sections(parse_html(html))) assert len(sections) >= 3 assert sections[0].title == "Ръководство X" assert "Съдържание" in sections[0].text assert sections[1].title == "1. Увод" assert sections[2].title == "2. Край" def _titles(sections): return [s.title for s in sections]