From b5e83c9c6988e378b8870cba4b6a7dad4e0646b2 Mon Sep 17 00:00:00 2001 From: sabo Date: Mon, 14 Sep 2026 12:57:10 +0300 Subject: [PATCH] . --- README.md | 4 +- docs/razdeljane-na-sekcii.md | 10 ++- help_processor.py | 109 ++++++++++++++++++--------- tests/test_html_word_export.py | 132 +++++++++++++++++++++++++++++++++ 4 files changed, 216 insertions(+), 39 deletions(-) create mode 100644 tests/test_html_word_export.py diff --git a/README.md b/README.md index 4a3ee0e..0e5a952 100644 --- a/README.md +++ b/README.md @@ -2,6 +2,8 @@ Обработва help-файлове (`.html`, `.htm`, `.docx`, `.doc`, `.pdf`, `.txt`), декомпозира ги на секции, извлича картинки, класифицира секциите с Claude API (заглавие + ключови думи), и записва всичко в SQL Server. После генерира интерактивен HTML viewer. +**Основен вход в момента:** HTML от Word/LibreOffice „Save as… / уеб страница“ (относителни картинки + H1/H2/MsoNormal структура). PDF и DOCX са вторични. + ## Архитектура ``` @@ -106,7 +108,7 @@ UNIQUE constraint: `(prefix, file_path)` Извличат се по време на парсване: - `.docx` — `` в paragraph drawings → `r:embed` (related_parts) или `r:link` (UNC/`file:///` + sidecar `_media/` до .docx) -- `.html` — локални файлове и `data:` URLs; HTTP пропуска +- `.html` — локални файлове (относителен `src` спрямо HTML папката, вкл. `../` и `%20`) и `data:` URLs; HTTP пропуска; warning при липсващ файл - `.pdf` — `pdfplumber.page.crop(bbox).to_image()` като PNG - `.doc` — след LibreOffice/MS Word конверсия до `.docx` diff --git a/docs/razdeljane-na-sekcii.md b/docs/razdeljane-na-sekcii.md index 09825bf..9a27008 100644 --- a/docs/razdeljane-na-sekcii.md +++ b/docs/razdeljane-na-sekcii.md @@ -61,7 +61,9 @@ TXT и PDF запазват своите правила (номерирани р Без блок „Съдържание“ всяка номерирана глава реже директно. -**Пример (`atra-manual.docx` / NESPERTCAM):** единственото Heading 1 е заглавието на документа; главите 1–9 са Heading 2. Очакване: **10 секции** — преамбюл (title + Съдържание + TOC) + по една за `1.` … `9.`. Не една мега-секция. +`
    ` / `
      ` / таблица веднага след TOC заглавието се третират като TOC блок (остават в преамбюла) и **приключват** TOC фазата — целият списък не се гледа като една глава `1. …`. + +**Пример (`atra-manual.docx` / NESPERTCAM / `atra-manual.html`):** единственото Heading 1 / H1 е заглавието на документа; главите 1–9 са Heading 2 / H2 (понякога с soft break в средата на реда). Очакване: **10 секции** — преамбюл (title + Съдържание + TOC) + по една за `1.` … `9.`. Не една мега-секция. --- @@ -120,10 +122,14 @@ TXT и PDF запазват своите правила (номерирани р ## 5. HTML / HTM (вкл. Word „Запиши като уеб“) +**Текущ основен вход:** HTML, експортиран от Word/LibreOffice чрез „Save as… / Запиши като уеб страница“. PDF и DOCX остават поддържани, но HTML е предпочитаният път за сканиране и проверка. + Премахват се `script`, `style`, `nav`, `footer`, `header`, `noscript`. Събират се **top-level** блокове: `h1`–`h6`, `p`, `ul`, `ol`, `table`, `dl`, `pre`, `blockquote`, `figure`, `hr`, самостоятелни `img`, и `div` **само** ако изглежда като заглавие. Вложен блок не се обработва втори път. +Текстът за решения „граница / TOC / номер на глава“ се нормализира (CR/LF/табове → един интервал), защото Save-as HTML често чупи заглавията с soft break вътре в `

      `. + ### 5.1. Заглавие 1. Тагове: `h1`→ граница (ниво 1); `h2`–`h6`→ в тялото **освен** номерирана глава. @@ -136,7 +142,7 @@ TXT и PDF запазват своите правила (номерирани р Списъци и таблици влизат в текущата секция. В plain text редовете им са с нов ред (не сплескани в един ред). Декоративни атрибути (`class`, `id`, `on*`, `data-*` …) се махат; от `style` се **пазят** `color`, `font-weight` (bold) и `font-style` (italic). Тагове ``/``/``/`` и `` остават. -Картинки: локални и `data:` URI; HTTP(S) се пропускат. Дребни иконки под 50×50 px се изхвърлят. +Картинки: локални и `data:` URI; HTTP(S) се пропускат. Относителният `img src` се резолвира **спрямо директорията на HTML файла** (`../`, URL-encoding `%20` → интервал). Липсващ файл се логва като warning и картинката се пропуска. Дребни иконки под 50×50 px се изхвърлят. При сканиране относителният път към PNG трябва да е валиден на диска (същата папкова структура като при експорта). Ако няма нито една секция: цялото `body` като една секция без заглавие. diff --git a/help_processor.py b/help_processor.py index 3812c9a..51a81fa 100644 --- a/help_processor.py +++ b/help_processor.py @@ -494,7 +494,13 @@ def file_hash(path: Path) -> str: def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]: - """Връща (data, ext) или None. Пропуска HTTP/HTTPS.""" + """Връща (data, ext) или None. Пропуска HTTP/HTTPS. + + Относителните ``src`` се резолвират спрямо директорията на HTML файла. + URL-encoding (``%20``) се декодира — типично за Word/LibreOffice „Save as HTML“. + """ + from urllib.parse import unquote + if not src: return None s = src.strip() @@ -511,14 +517,29 @@ def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]: return data, _ext_from_content_type(m.group(1)) if s.startswith(("http://", "https://")): return None # по правило пропускаме мрежови картинки - # локален път, относителен или абсолютен - p = (base_dir / s).resolve() if not Path(s).is_absolute() else Path(s) + if s.lower().startswith("file:"): + path = _file_url_to_path(s) + if path is None: + log.warning(f" HTML image bad file URL: {src}") + return None + loaded = _read_image_file(path) + if not loaded: + log.warning(f" HTML image not found: {src} → {path}") + return loaded + + # локален път: %20 → space, ../ спрямо HTML папката + decoded = unquote(s).replace("\\", "/") try: + p = Path(decoded) + if not p.is_absolute(): + p = (base_dir / decoded).resolve() if p.is_file(): data = p.read_bytes() - ext = p.suffix.lstrip(".").lower() or "png" + ext = p.suffix.lstrip(".").lower() or "png" return data, ext - except Exception: + log.warning(f" HTML image not found: {src} → {p}") + except Exception as e: + log.warning(f" HTML image unreadable: {src}: {e}") return None return None @@ -622,13 +643,18 @@ _FIGURE_CAPTION_RE = re.compile( ) +def _normalize_inline_ws(text: str) -> str: + """Срива CR/LF/табове в един интервал (Word/LibreOffice soft breaks в

      ).""" + return re.sub(r"\s+", " ", (text or "").strip()) + + def _normalize_chapter_key(text: str) -> str: - return re.sub(r"\s+", " ", (text or "").strip().lower()) + return _normalize_inline_ws(text).lower() def _is_numbered_chapter_heading(text: str) -> bool: """True for top-level numbered chapter titles (not TOC prose, figures, tables).""" - t = (text or "").strip() + t = _normalize_inline_ws(text) if not t or len(t) >= 120: return False if "|" in t: @@ -831,24 +857,34 @@ def _strip_attrs(el): t["style"] = kept_style +def _ingest_html_img( + img, base_dir: Path, sec_images: list, img_counter: list +) -> Optional[str]: + """Извлича една картинка; връща placeholder текст или None.""" + from bs4 import NavigableString + + src = img.get("src") or img.get("data-src") or "" + loaded = _load_html_image(src, base_dir) + if not loaded: + img.decompose() + return None + data, ext = loaded + if not _should_keep_image(data): + img.decompose() + return None + img_counter[0] += 1 + ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext) + sec_images.append(ref) + ph = f"[IMG: {ref.placeholder}]" + img.replace_with(NavigableString(ph)) + return ph + + def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None: """Намира всички в подадения елемент, извлича данните и подменя с NavigableString placeholder ([IMG: img_NN]).""" - from bs4 import NavigableString for img in el.find_all("img"): - src = img.get("src") or img.get("data-src") or "" - loaded = _load_html_image(src, base_dir) - if not loaded: - img.decompose() - continue - data, ext = loaded - if not _should_keep_image(data): - img.decompose() - continue - img_counter[0] += 1 - ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext) - sec_images.append(ref) - img.replace_with(NavigableString(f"[IMG: {ref.placeholder}]")) + _ingest_html_img(img, base_dir, sec_images, img_counter) def parse_html(path: Path) -> list[Section]: @@ -909,14 +945,16 @@ def parse_html(path: Path) -> list[Section]: def start_section(title: str, level: int = 1): nonlocal current_title, current_level, sec_text, sec_html, sec_images flush() - current_title = title + current_title = _normalize_inline_ws(title) current_level = level sec_text, sec_html, sec_images = [], [], [] for el in blocks: - txt = el.get_text(" ", strip=True) + # Soft breaks (\r\n в средата на Word/LO

      ) → един ред за split/TOC + txt = _normalize_inline_ws(el.get_text(" ", strip=True)) is_cover = _html_is_cover_title(el) heading_lvl = _html_heading_level(el) + el_name = (el.name or "").lower() # Корица (Word Title / class Title): заглавие на преамбюла, без нова секция if is_cover and txt: @@ -934,6 +972,12 @@ def parse_html(path: Path) -> list[Section]: append_block_as_body(el) continue + #
        /
          /table в TOC: целият блок НЕ е една „1. …“ глава — край на TOC фазата + if toc_state.phase and el_name in _HTML_PLAIN_NL_TAGS: + toc_state.leave_phase() + append_block_as_body(el) + continue + # Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла numbered_action = toc_state.handle_numbered(txt) if txt else "no" if numbered_action == "toc": @@ -957,23 +1001,16 @@ def parse_html(path: Path) -> list[Section]: start_section(txt, heading_lvl) continue - if el.name == "img": + if el_name == "img": if toc_state.phase: toc_state.leave_phase() - # самостоятелен (не вътре в блок) - _swap_imgs_in_block(el.parent if el.parent and el.parent.name else el, - base_dir, sec_images, img_counter) - # ако е заменен с placeholder, добавяме като текст - img_txt = el.get_text(" ", strip=True) if el.name else "" - if img_txt: - sec_text.append(img_txt) - sec_html.append(f"

          {img_txt}

          ") + ph = _ingest_html_img(el, base_dir, sec_images, img_counter) + if ph: + sec_text.append(ph) + sec_html.append(f"

          {ph}

          ") continue - if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS: - #
            /
              TOC списък — остава в преамбюла, приключва TOC фазата - toc_state.leave_phase() - elif toc_state.phase and txt and not _is_numbered_chapter_heading(txt): + if toc_state.phase and txt and not _is_numbered_chapter_heading(txt): toc_state.leave_phase() append_block_as_body(el) diff --git a/tests/test_html_word_export.py b/tests/test_html_word_export.py new file mode 100644 index 0000000..f192a99 --- /dev/null +++ b/tests/test_html_word_export.py @@ -0,0 +1,132 @@ +"""HTML от Word/LibreOffice „Save as…“: относителни картинки + multi-section split.""" +from pathlib import Path + +from help_processor import ( + merge_preamble_sections, + merge_short_sections, + parse_html, +) + + +def _mini_png(w: int = 64, h: int = 64) -> bytes: + try: + from io import BytesIO + from PIL import Image + + buf = BytesIO() + Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG") + return buf.getvalue() + except Exception: + return ( + b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01" + b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00" + b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N" + b"\x00\x00\x00\x00IEND\xaeB`\x82" + ) + + +def test_html_relative_img_url_encoded(tmp_path: Path): + """../ + %20 в src се резолвират спрямо HTML папката (без UNC).""" + media = tmp_path / "___Proekti" / "2025 ATRA96" / "otchitane" + media.mkdir(parents=True) + png = _mini_png() + (media / "123.png").write_bytes(png) + + html_dir = tmp_path / "docs" + html_dir.mkdir() + html = html_dir / "manual.html" + html.write_text( + """ +

              1. Преглед

              +

              + +

              +

              Фигура 1: екран с достатъчно текст за тяло на секцията тук.

              + """, + encoding="utf-8", + ) + + sections = merge_preamble_sections(merge_short_sections(parse_html(html))) + assert len(sections) >= 1 + body = sections[0] + assert body.images, "relative URL-encoded img must load from disk" + assert "[IMG:" in (body.text or "") + assert "[IMG:" in (body.html_text or "") + + +def test_html_missing_img_logs_warning(tmp_path: Path, caplog): + html = tmp_path / "gone.html" + html.write_text( + """ +

              1. Alone

              +

              +

              Тяло без картинка, но с достатъчно думи за секция едно две три.

              + """, + encoding="utf-8", + ) + import logging + + with caplog.at_level(logging.WARNING, logger="help_processor"): + sections = parse_html(html) + assert sections + assert not sections[0].images + assert any("HTML image not found" in r.message for r in caplog.records) + + +def test_html_word_softbreak_h2_chapters_split(tmp_path: Path): + """CR/LF вътре в H2 (типично Save-as HTML) не трябва да остави 1 мега-секция.""" + html = tmp_path / "softbreak.html" + # Имитира atra-manual.html: H1 корица, Съдържание, H2 с пренос на ред в заглавието + html.write_text( + "\n" + '

              NESPERTCAM Launcher - Пълно\r\n' + "ръководство на потребителя

              \n" + '

              Съдържание

              \n' + "
                \n" + "
              • 1. Преглед на приложението

              • \n" + "
              • 2. Инсталация и настройка

              • \n" + "
              \n" + '

              1. Преглед на\r\nприложението

              \n' + "

              Какво е NESPERTCAM Launcher с достатъчно думи в тялото на главата едно.

              \n" + '

              2. Инсталация\r\nи настройка

              \n' + "

              Системни изисквания Windows 10 и още текст за втората глава тук.

              \n" + "", + encoding="utf-8", + ) + + sections = merge_preamble_sections(merge_short_sections(parse_html(html))) + assert len(sections) >= 3, f"expected preamble+chapters, got {_titles(sections)}" + assert "NESPERTCAM" in sections[0].title + assert "Съдържание" in sections[0].text + assert sections[1].title == "1. Преглед на приложението" + assert sections[2].title == "2. Инсталация и настройка" + assert "\r" not in sections[1].title and "\n" not in sections[1].title + + +def test_html_mso_normal_numbered_chapters(tmp_path: Path): + """Word MsoNormal / bold глави 1. / 2. режат секции след TOC.""" + html = tmp_path / "mso.html" + html.write_text( + """ +

              Ръководство X

              +

              Съдържание

              +

              1. Увод

              +

              2. Край

              +

              1. Увод

              +

              Текст на увода с няколко думи повече от минимум.

              +

              2. Край

              +

              Текст на края с няколко думи повече от минимум.

              + """, + encoding="utf-8", + ) + sections = merge_preamble_sections(merge_short_sections(parse_html(html))) + assert len(sections) >= 3 + assert sections[0].title == "Ръководство X" + assert "Съдържание" in sections[0].text + assert sections[1].title == "1. Увод" + assert sections[2].title == "2. Край" + + +def _titles(sections): + return [s.title for s in sections]