Files
rip-help-system/tests/test_section_split_preamble.py
2026-09-14 11:10:05 +03:00

102 lines
4.4 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Корица + Съдържание + глави: преамбюл + по една секция на номерирана глава."""
from pathlib import Path
from help_processor import (
merge_preamble_sections,
merge_short_sections,
parse_html,
Section,
)
FIXTURES = Path(__file__).parent / "fixtures"
PREAMBLE = FIXTURES / "nespertcam_launcher_preamble.html"
CHAPTERS = FIXTURES / "nespertcam_launcher_chapters.html"
def _titles(sections):
return [s.title for s in sections]
def test_nespertcam_preamble_yields_two_sections():
sections = merge_preamble_sections(merge_short_sections(parse_html(PREAMBLE)))
assert len(sections) == 2
assert sections[0].title == "NESPERTCAM Launcher - Пълно ръководство на потребителя"
assert sections[1].title == "1. Преглед на приложението"
assert "Съдържание" in sections[0].text
assert "1. Преглед на приложението" in sections[0].text
assert "Какво е NESPERTCAM Launcher?" in sections[1].text
assert "NESPERTCAM64.exe" in sections[1].text
def test_nespertcam_h2_chapters_split_not_one_section():
"""Като atra-manual.docx: H1=корица, TOC като Normal, глави като H2 → ≥3 секции."""
sections = merge_preamble_sections(merge_short_sections(parse_html(CHAPTERS)))
assert len(sections) >= 3
assert sections[0].title == "NESPERTCAM Launcher - Пълно ръководство на потребителя"
assert "Съдържание" in sections[0].text
assert "1. Преглед на приложението" in sections[0].text
assert sections[1].title == "1. Преглед на приложението"
assert sections[2].title == "2. Инсталация и настройка"
assert "Какво е NESPERTCAM Launcher?" in sections[1].text
assert "Основни предимства" in sections[1].text
assert "Фигура 1:" in sections[1].text
assert "Системни изисквания" in sections[2].text
assert "Windows 10" in sections[2].text
# H3/bold/фигура не отварят собствени секции
assert "Какво е NESPERTCAM Launcher?" not in _titles(sections)
assert "Основни предимства" not in _titles(sections)
def test_h2_and_bold_do_not_split_inside_chapter():
html = FIXTURES / "_tmp_h2.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>Глава А</h1>
<h2>Подточка</h2>
<p>Текст под подточката с достатъчно думи за тяло.</p>
<p><b>Още едно bold подзаглавие</b></p>
<p>Още текст в същата секция.</p>
<h1>Глава Б</h1>
<p>Тяло на втората глава.</p>
</body></html>""",
encoding="utf-8",
)
try:
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert _titles(sections) == ["Глава А", "Глава Б"]
assert "Подточка" in sections[0].text
assert "Още едно bold подзаглавие" in sections[0].text
finally:
html.unlink(missing_ok=True)
def test_toc_heading_as_h1_merges_into_cover():
"""Ако „Съдържание“ е H1, merge_preamble го залепва към корицата."""
secs = [
Section("Ръководство X", "", 1),
Section("Съдържание", "1. Увод\n2. Край", 1),
Section("1. Увод", "Текст на увода.", 1),
]
merged = merge_preamble_sections(secs)
assert len(merged) == 2
assert merged[0].title == "Ръководство X"
assert "Съдържание" in merged[0].text
assert merged[1].title == "1. Увод"
def test_heading1_chapters_still_split():
html = FIXTURES / "_tmp_chapters.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Първа</h1><p>Аа аа аа.</p>
<h1>2. Втора</h1><p>Бб бб бб.</p>
<h1>3. Трета</h1><p>Вв вв вв.</p>
</body></html>""",
encoding="utf-8",
)
try:
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert _titles(sections) == ["1. Първа", "2. Втора", "3. Трета"]
finally:
html.unlink(missing_ok=True)