This commit is contained in:
2026-09-14 12:57:10 +03:00
parent 64578acf30
commit b5e83c9c69
4 changed files with 216 additions and 39 deletions

View File

@@ -2,6 +2,8 @@
Обработва help-файлове (`.html`, `.htm`, `.docx`, `.doc`, `.pdf`, `.txt`), декомпозира ги на секции, извлича картинки, класифицира секциите с Claude API (заглавие + ключови думи), и записва всичко в SQL Server. После генерира интерактивен HTML viewer.
**Основен вход в момента:** HTML от Word/LibreOffice „Save as… / уеб страница“ (относителни картинки + H1/H2/MsoNormal структура). PDF и DOCX са вторични.
## Архитектура
```
@@ -106,7 +108,7 @@ UNIQUE constraint: `(prefix, file_path)`
Извличат се по време на парсване:
- `.docx` — `<a:blip>` в paragraph drawings → `r:embed` (related_parts) или `r:link` (UNC/`file:///` + sidecar `<stem>_media/` до .docx)
- `.html` — локални файлове и `data:` URLs; HTTP пропуска
- `.html` — локални файлове (относителен `src` спрямо HTML папката, вкл. `../` и `%20`) и `data:` URLs; HTTP пропуска; warning при липсващ файл
- `.pdf` — `pdfplumber.page.crop(bbox).to_image()` като PNG
- `.doc` — след LibreOffice/MS Word конверсия до `.docx`

View File

@@ -61,7 +61,9 @@ TXT и PDF запазват своите правила (номерирани р
Без блок „Съдържание“ всяка номерирана глава реже директно.
**Пример (`atra-manual.docx` / NESPERTCAM):** единственото Heading 1 е заглавието на документа; главите 1–9 са Heading 2. Очакване: **10 секции** — преамбюл (title + Съдържание + TOC) + по една за `1.` … `9.`. Не една мега-секция.
`<ul>` / `<ol>` / таблица веднага след TOC заглавието се третират като TOC блок (остават в преамбюла) и **приключват** TOC фазата — целият списък не се гледа като една глава `1. …`.
**Пример (`atra-manual.docx` / NESPERTCAM / `atra-manual.html`):** единственото Heading 1 / H1 е заглавието на документа; главите 1–9 са Heading 2 / H2 (понякога с soft break в средата на реда). Очакване: **10 секции** — преамбюл (title + Съдържание + TOC) + по една за `1.` … `9.`. Не една мега-секция.
---
@@ -120,10 +122,14 @@ TXT и PDF запазват своите правила (номерирани р
## 5. HTML / HTM (вкл. Word „Запиши като уеб“)
**Текущ основен вход:** HTML, експортиран от Word/LibreOffice чрез „Save as… / Запиши като уеб страница“. PDF и DOCX остават поддържани, но HTML е предпочитаният път за сканиране и проверка.
Премахват се `script`, `style`, `nav`, `footer`, `header`, `noscript`.
Събират се **top-level** блокове: `h1`–`h6`, `p`, `ul`, `ol`, `table`, `dl`, `pre`, `blockquote`, `figure`, `hr`, самостоятелни `img`, и `div` **само** ако изглежда като заглавие. Вложен блок не се обработва втори път.
Текстът за решения „граница / TOC / номер на глава“ се нормализира (CR/LF/табове → един интервал), защото Save-as HTML често чупи заглавията с soft break вътре в `<h2>`.
### 5.1. Заглавие
1. Тагове: `h1`→ граница (ниво 1); `h2`–`h6`→ в тялото **освен** номерирана глава.
@@ -136,7 +142,7 @@ TXT и PDF запазват своите правила (номерирани р
Списъци и таблици влизат в текущата секция. В plain text редовете им са с нов ред (не сплескани в един ред). Декоративни атрибути (`class`, `id`, `on*`, `data-*` …) се махат; от `style` се **пазят** `color`, `font-weight` (bold) и `font-style` (italic). Тагове `<b>`/`<strong>`/`<i>`/`<em>` и `<font color>` остават.
Картинки: локални и `data:` URI; HTTP(S) се пропускат. Дребни иконки под 50×50 px се изхвърлят.
Картинки: локални и `data:` URI; HTTP(S) се пропускат. Относителният `img src` се резолвира **спрямо директорията на HTML файла** (`../`, URL-encoding `%20` → интервал). Липсващ файл се логва като warning и картинката се пропуска. Дребни иконки под 50×50 px се изхвърлят. При сканиране относителният път към PNG трябва да е валиден на диска (същата папкова структура като при експорта).
Ако няма нито една секция: цялото `body` като една секция без заглавие.

View File

@@ -494,7 +494,13 @@ def file_hash(path: Path) -> str:
def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS."""
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS.
Относителните ``src`` се резолвират спрямо директорията на HTML файла.
URL-encoding (``%20``) се декодира — типично за Word/LibreOffice „Save as HTML“.
"""
from urllib.parse import unquote
if not src:
return None
s = src.strip()
@@ -511,14 +517,29 @@ def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
return data, _ext_from_content_type(m.group(1))
if s.startswith(("http://", "https://")):
return None # по правило пропускаме мрежови картинки
# локален път, относителен или абсолютен
p = (base_dir / s).resolve() if not Path(s).is_absolute() else Path(s)
if s.lower().startswith("file:"):
path = _file_url_to_path(s)
if path is None:
log.warning(f" HTML image bad file URL: {src}")
return None
loaded = _read_image_file(path)
if not loaded:
log.warning(f" HTML image not found: {src} → {path}")
return loaded
# локален път: %20 → space, ../ спрямо HTML папката
decoded = unquote(s).replace("\\", "/")
try:
p = Path(decoded)
if not p.is_absolute():
p = (base_dir / decoded).resolve()
if p.is_file():
data = p.read_bytes()
ext = p.suffix.lstrip(".").lower() or "png"
ext = p.suffix.lstrip(".").lower() or "png"
return data, ext
except Exception:
log.warning(f" HTML image not found: {src} → {p}")
except Exception as e:
log.warning(f" HTML image unreadable: {src}: {e}")
return None
return None
@@ -622,13 +643,18 @@ _FIGURE_CAPTION_RE = re.compile(
)
def _normalize_inline_ws(text: str) -> str:
"""Срива CR/LF/табове в един интервал (Word/LibreOffice soft breaks в <h2>)."""
return re.sub(r"\s+", " ", (text or "").strip())
def _normalize_chapter_key(text: str) -> str:
return re.sub(r"\s+", " ", (text or "").strip().lower())
return _normalize_inline_ws(text).lower()
def _is_numbered_chapter_heading(text: str) -> bool:
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
t = (text or "").strip()
t = _normalize_inline_ws(text)
if not t or len(t) >= 120:
return False
if "|" in t:
@@ -831,24 +857,34 @@ def _strip_attrs(el):
t["style"] = kept_style
def _ingest_html_img(
img, base_dir: Path, sec_images: list, img_counter: list
) -> Optional[str]:
"""Извлича една <img> картинка; връща placeholder текст или None."""
from bs4 import NavigableString
src = img.get("src") or img.get("data-src") or ""
loaded = _load_html_image(src, base_dir)
if not loaded:
img.decompose()
return None
data, ext = loaded
if not _should_keep_image(data):
img.decompose()
return None
img_counter[0] += 1
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
sec_images.append(ref)
ph = f"[IMG: {ref.placeholder}]"
img.replace_with(NavigableString(ph))
return ph
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
"""Намира всички <img> в подадения елемент, извлича данните и подменя с
NavigableString placeholder ([IMG: img_NN])."""
from bs4 import NavigableString
for img in el.find_all("img"):
src = img.get("src") or img.get("data-src") or ""
loaded = _load_html_image(src, base_dir)
if not loaded:
img.decompose()
continue
data, ext = loaded
if not _should_keep_image(data):
img.decompose()
continue
img_counter[0] += 1
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
sec_images.append(ref)
img.replace_with(NavigableString(f"[IMG: {ref.placeholder}]"))
_ingest_html_img(img, base_dir, sec_images, img_counter)
def parse_html(path: Path) -> list[Section]:
@@ -909,14 +945,16 @@ def parse_html(path: Path) -> list[Section]:
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, sec_text, sec_html, sec_images
flush()
current_title = title
current_title = _normalize_inline_ws(title)
current_level = level
sec_text, sec_html, sec_images = [], [], []
for el in blocks:
txt = el.get_text(" ", strip=True)
# Soft breaks (\r\n в средата на Word/LO <h2>) → един ред за split/TOC
txt = _normalize_inline_ws(el.get_text(" ", strip=True))
is_cover = _html_is_cover_title(el)
heading_lvl = _html_heading_level(el)
el_name = (el.name or "").lower()
# Корица (Word Title / class Title): заглавие на преамбюла, без нова секция
if is_cover and txt:
@@ -934,6 +972,12 @@ def parse_html(path: Path) -> list[Section]:
append_block_as_body(el)
continue
# <ul>/<ol>/table в TOC: целият блок НЕ е една „1. …“ глава — край на TOC фазата
if toc_state.phase and el_name in _HTML_PLAIN_NL_TAGS:
toc_state.leave_phase()
append_block_as_body(el)
continue
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
if numbered_action == "toc":
@@ -957,23 +1001,16 @@ def parse_html(path: Path) -> list[Section]:
start_section(txt, heading_lvl)
continue
if el.name == "img":
if el_name == "img":
if toc_state.phase:
toc_state.leave_phase()
# самостоятелен <img> (не вътре в блок)
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
base_dir, sec_images, img_counter)
# ако е заменен с placeholder, добавяме като текст
img_txt = el.get_text(" ", strip=True) if el.name else ""
if img_txt:
sec_text.append(img_txt)
sec_html.append(f"<p>{img_txt}</p>")
ph = _ingest_html_img(el, base_dir, sec_images, img_counter)
if ph:
sec_text.append(ph)
sec_html.append(f"<p>{ph}</p>")
continue
if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS:
# <ol>/<ul> TOC списък — остава в преамбюла, приключва TOC фазата
toc_state.leave_phase()
elif toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
if toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
toc_state.leave_phase()
append_block_as_body(el)

View File

@@ -0,0 +1,132 @@
"""HTML от Word/LibreOffice „Save as…“: относителни картинки + multi-section split."""
from pathlib import Path
from help_processor import (
merge_preamble_sections,
merge_short_sections,
parse_html,
)
def _mini_png(w: int = 64, h: int = 64) -> bytes:
try:
from io import BytesIO
from PIL import Image
buf = BytesIO()
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
return buf.getvalue()
except Exception:
return (
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
b"\x00\x00\x00\x00IEND\xaeB`\x82"
)
def test_html_relative_img_url_encoded(tmp_path: Path):
"""../ + %20 в src се резолвират спрямо HTML папката (без UNC)."""
media = tmp_path / "___Proekti" / "2025 ATRA96" / "otchitane"
media.mkdir(parents=True)
png = _mini_png()
(media / "123.png").write_bytes(png)
html_dir = tmp_path / "docs"
html_dir.mkdir()
html = html_dir / "manual.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Преглед</h1>
<p class="MsoNormal" style="background:#fafafa">
<img src="../___Proekti/2025%20ATRA96/otchitane/123.png"
name="Picture 1" align="bottom" width="1045" height="523" border="0"/>
</p>
<p>Фигура 1: екран с достатъчно текст за тяло на секцията тук.</p>
</body></html>""",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert len(sections) >= 1
body = sections[0]
assert body.images, "relative URL-encoded img must load from disk"
assert "[IMG:" in (body.text or "")
assert "[IMG:" in (body.html_text or "")
def test_html_missing_img_logs_warning(tmp_path: Path, caplog):
html = tmp_path / "gone.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<h1>1. Alone</h1>
<p><img src="../missing%20dir/nope.png" name="Picture 1"/></p>
<p>Тяло без картинка, но с достатъчно думи за секция едно две три.</p>
</body></html>""",
encoding="utf-8",
)
import logging
with caplog.at_level(logging.WARNING, logger="help_processor"):
sections = parse_html(html)
assert sections
assert not sections[0].images
assert any("HTML image not found" in r.message for r in caplog.records)
def test_html_word_softbreak_h2_chapters_split(tmp_path: Path):
"""CR/LF вътре в H2 (типично Save-as HTML) не трябва да остави 1 мега-секция."""
html = tmp_path / "softbreak.html"
# Имитира atra-manual.html: H1 корица, Съдържание, H2 с пренос на ред в заглавието
html.write_text(
"<!DOCTYPE html><html><body>\n"
'<h1 class="western">NESPERTCAM Launcher - Пълно\r\n'
"ръководство на потребителя</h1>\n"
'<h3 class="western">Съдържание</h3>\n'
"<ul>\n"
"<li><p>1. Преглед на приложението</p></li>\n"
"<li><p>2. Инсталация и настройка</p></li>\n"
"</ul>\n"
'<h2 class="western">1. Преглед на\r\nприложението</h2>\n'
"<p>Какво е NESPERTCAM Launcher с достатъчно думи в тялото на главата едно.</p>\n"
'<h2 class="western">2. Инсталация\r\nи настройка</h2>\n'
"<p>Системни изисквания Windows 10 и още текст за втората глава тук.</p>\n"
"</body></html>",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert len(sections) >= 3, f"expected preamble+chapters, got {_titles(sections)}"
assert "NESPERTCAM" in sections[0].title
assert "Съдържание" in sections[0].text
assert sections[1].title == "1. Преглед на приложението"
assert sections[2].title == "2. Инсталация и настройка"
assert "\r" not in sections[1].title and "\n" not in sections[1].title
def test_html_mso_normal_numbered_chapters(tmp_path: Path):
"""Word MsoNormal / bold глави 1. / 2. режат секции след TOC."""
html = tmp_path / "mso.html"
html.write_text(
"""<!DOCTYPE html><html><body>
<p class=MsoTitle><b>Ръководство X</b></p>
<p class=MsoNormal><b>Съдържание</b></p>
<p class=MsoNormal>1. Увод</p>
<p class=MsoNormal>2. Край</p>
<p class=MsoNormal><b>1. Увод</b></p>
<p class=MsoNormal>Текст на увода с няколко думи повече от минимум.</p>
<p class=MsoNormal><b>2. Край</b></p>
<p class=MsoNormal>Текст на края с няколко думи повече от минимум.</p>
</body></html>""",
encoding="utf-8",
)
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
assert len(sections) >= 3
assert sections[0].title == "Ръководство X"
assert "Съдържание" in sections[0].text
assert sections[1].title == "1. Увод"
assert sections[2].title == "2. Край"
def _titles(sections):
return [s.title for s in sections]