"""
help_processor.py
=================
Обработва help-файлове (.doc, .docx, .html, .htm, .txt, .pdf),
декомпозира ги на смислови секции, извлича ключови думи чрез Anthropic API
и записва резултатите в SQL Server + изходна директория.
Поддържа инкрементална обработка: файлове, чийто hash не се е променил,
се прескачат при повторно пускане.
Изисквания (pip install):
pip install anthropic pyodbc python-docx beautifulsoup4 lxml
pip install pdfplumber striprtf chardet
pip install pywin32 # за MS Word fallback на Windows
За .doc (стар формат) е необходим един от:
- LibreOffice (soffice в PATH) — кросплатформено
- MS Word — Windows, чрез pywin32 COM (автоматичен fallback)
- antiword — Linux (apt install antiword)
"""
import os
import re
import sys
import json
import hashlib
import logging
import argparse
import subprocess
import tempfile
from pathlib import Path
from datetime import datetime
from dataclasses import dataclass, field
from typing import Optional
import psycopg2
import anthropic
from docx import Document
from bs4 import BeautifulSoup
from help_codes import (
WIPED_HASH,
file_result as _file_result,
make_code,
parse_code,
remove_section_outputs,
source_basename,
source_identity,
)
try:
import pdfplumber
HAS_PDF = True
except ImportError:
HAS_PDF = False
try:
from PIL import Image
HAS_PIL = True
except ImportError:
HAS_PIL = False
# ──────────────────────────────────────────────
# Конфигурация
# ──────────────────────────────────────────────
# На Windows конзолата често е cp1251 → пренастройваме stdout на utf-8
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
except AttributeError:
pass
_log_handlers = [logging.StreamHandler(sys.stdout)]
try:
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
except OSError:
pass
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)-8s %(message)s",
handlers=_log_handlers,
)
log = logging.getLogger(__name__)
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
AI_MODELS = [
AI_MODEL,
"claude-haiku-4-5",
"claude-haiku-4-5-20251001",
"claude-3-5-haiku-latest",
]
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
# ──────────────────────────────────────────────
# Изображения — помощни
# ──────────────────────────────────────────────
@dataclass
class ImageRef:
placeholder: str # вътрешен ID в текста, напр. "img_01"
data: bytes
ext: str # "png", "jpg", "gif"...
def _img_dimensions(data: bytes) -> Optional[tuple[int, int]]:
if not HAS_PIL:
return None
try:
from io import BytesIO
with Image.open(BytesIO(data)) as im:
return im.size
except Exception:
return None
def _should_keep_image(data: bytes) -> bool:
"""Връща False за дребни иконки/булети под MIN_IMAGE_PX × MIN_IMAGE_PX."""
if not data:
return False
dims = _img_dimensions(data)
if dims is None:
# Не можем да преценим — пазим по подразбиране
return True
w, h = dims
return w >= MIN_IMAGE_PX and h >= MIN_IMAGE_PX
def _ext_from_content_type(ct: str) -> str:
ct = (ct or "").lower()
if "png" in ct: return "png"
if "jpeg" in ct or "jpg" in ct: return "jpg"
if "gif" in ct: return "gif"
if "bmp" in ct: return "bmp"
if "svg" in ct: return "svg"
if "webp" in ct: return "webp"
return "png"
_IMG_PLACEHOLDER_RE = re.compile(r"\[IMG:\s*([A-Za-z0-9_./\\-]+)\s*\]")
# ──────────────────────────────────────────────
# Структури
# ──────────────────────────────────────────────
@dataclass
class Section:
title: str
text: str
level: int = 1 # 1=H1, 2=H2, 3=H3, 0=без заглавие
images: list = field(default_factory=list) # list[ImageRef]
html_text: Optional[str] = None # rich HTML с [IMG: ...] placeholders
@dataclass
class ProcessedSection:
code: str # DOC_003_SEC_012
source_file: str
title: str
keywords: str # "кл1, кл2, кл3"
text: str
images_json: str = "[]" # JSON масив с относителни пътища
html_text: str = "" # rich HTML (само за HTML-source файлове)
char_count: int = 0
def __post_init__(self):
self.char_count = len(self.text)
# ──────────────────────────────────────────────
# База данни
# ──────────────────────────────────────────────
class Database:
"""PostgreSQL backend (psycopg2). Connection string е libpq формат:
'host=... port=... dbname=... user=... password=...'
"""
def __init__(self, conn_str: str):
self.conn_str = conn_str
self.conn = psycopg2.connect(conn_str)
self._ensure_schema()
def _ensure_schema(self):
"""Създава таблиците ако не съществуват (Postgres syntax)."""
cur = self.conn.cursor()
cur.execute("""
CREATE TABLE IF NOT EXISTS rip_help_files (
id SERIAL PRIMARY KEY,
prefix VARCHAR(50) NOT NULL DEFAULT 'HLP',
file_path VARCHAR(1000) NOT NULL,
file_hash CHAR(64) NOT NULL,
processed_at TIMESTAMP NOT NULL DEFAULT NOW(),
section_count INTEGER NOT NULL DEFAULT 0,
UNIQUE (prefix, file_path)
)
""")
cur.execute("""
CREATE TABLE IF NOT EXISTS rip_help_sections (
id SERIAL PRIMARY KEY,
prefix VARCHAR(50) NOT NULL DEFAULT 'HLP',
code VARCHAR(80) NOT NULL UNIQUE,
source_file VARCHAR(1000) NOT NULL,
title VARCHAR(500),
keywords VARCHAR(300),
char_count INTEGER,
output_path VARCHAR(1000),
images TEXT,
html_text TEXT,
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
)
""")
cur.execute("""
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_keywords
ON rip_help_sections(keywords)
""")
cur.execute("""
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_prefix
ON rip_help_sections(prefix)
""")
cur.execute("""
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_source
ON rip_help_sections(prefix, source_file)
""")
cur.execute(
"ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER"
)
self.conn.commit()
log.info("Схемата е проверена / създадена.")
def matching_source_paths(self, prefix: str, identity: str) -> list[str]:
"""Всички file_path/source_file за същия help файл (basename, вкл. стари temp пътища)."""
name = source_basename(identity).lower()
if not name:
return []
found: list[str] = []
seen: set[str] = set()
for p in self.all_source_files(prefix):
if source_basename(p).lower() == name and p not in seen:
seen.add(p)
found.append(p)
return found
def get_file_hash(self, prefix: str, file_path: str) -> Optional[str]:
paths = self.matching_source_paths(prefix, file_path) or [file_path]
cur = self.conn.cursor()
cur.execute(
"SELECT file_hash FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, list(paths)),
)
for (h,) in cur.fetchall():
if not h:
continue
val = str(h).strip()
if val and val != WIPED_HASH:
return val
return None
def upsert_file(
self,
prefix: str,
file_path: str,
file_hash: str,
section_count: int,
file_index: Optional[int] = None,
):
canonical = source_basename(file_path) or file_path
stale = [p for p in self.matching_source_paths(prefix, canonical) if p != canonical]
cur = self.conn.cursor()
if stale:
cur.execute(
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, stale),
)
cur.execute("""
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
VALUES (%s, %s, %s, %s, %s)
ON CONFLICT (prefix, file_path) DO UPDATE SET
file_hash = EXCLUDED.file_hash,
section_count= EXCLUDED.section_count,
processed_at = NOW(),
file_index = COALESCE(EXCLUDED.file_index, rip_help_files.file_index)
""", (prefix, canonical, file_hash, section_count, file_index))
self.conn.commit()
def delete_sections_for_file(self, prefix: str, file_path: str):
paths = self.matching_source_paths(prefix, file_path) or [file_path]
cur = self.conn.cursor()
cur.execute(
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
self.conn.commit()
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
"""(code, source_file, output_path) за всички секции на файла."""
paths = self.matching_source_paths(prefix, identity)
if not paths:
return []
cur = self.conn.cursor()
cur.execute(
"SELECT code, source_file, output_path FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code",
(prefix, list(paths)),
)
return [(r[0], r[1], r[2]) for r in cur.fetchall()]
def file_index_for(self, prefix: str, identity: str) -> Optional[int]:
paths = self.matching_source_paths(prefix, identity)
if not paths:
return None
cur = self.conn.cursor()
cur.execute(
"SELECT file_index FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL",
(prefix, list(paths)),
)
from_col = [r[0] for r in cur.fetchall() if r[0]]
if from_col:
return min(from_col)
cur.execute(
"SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
found: list[int] = []
for (code,) in cur.fetchall():
parsed = parse_code(code)
if parsed:
found.append(parsed[1])
if not found:
return None
tally: dict[int, int] = {}
for idx in found:
tally[idx] = tally.get(idx, 0) + 1
return max(tally, key=lambda k: (tally[k], -k))
def max_file_index(self, prefix: str) -> int:
cur = self.conn.cursor()
cur.execute(
"SELECT COALESCE(MAX(file_index), 0) FROM rip_help_files WHERE prefix=%s",
(prefix,),
)
m1 = int(cur.fetchone()[0] or 0)
cur.execute("SELECT code FROM rip_help_sections WHERE prefix=%s", (prefix,))
m2 = 0
for (code,) in cur.fetchall():
parsed = parse_code(code)
if parsed:
m2 = max(m2, parsed[1])
return max(m1, m2)
def file_index_used_by_others(self, prefix: str, identity: str, idx: int) -> bool:
name = source_basename(identity).lower()
cur = self.conn.cursor()
cur.execute(
"SELECT code, source_file FROM rip_help_sections WHERE prefix=%s",
(prefix,),
)
for code, src in cur.fetchall():
parsed = parse_code(code)
if parsed and parsed[1] == idx and source_basename(src).lower() != name:
return True
return False
def allocate_file_index(self, prefix: str, identity: str) -> int:
existing = self.file_index_for(prefix, identity)
if existing and not self.file_index_used_by_others(prefix, identity, existing):
return existing
return self.max_file_index(prefix) + 1
def wipe_extractions_for_file(
self,
prefix: str,
identity: str,
output_dir: Optional[Path] = None,
) -> dict:
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
rows = self.sections_for_file(prefix, identity)
codes = [r[0] for r in rows]
file_index = self.file_index_for(prefix, identity)
if file_index and self.file_index_used_by_others(prefix, identity, file_index):
file_index = None
paths = self.matching_source_paths(prefix, identity)
if output_dir:
remove_section_outputs(output_dir, codes, [r[2] for r in rows if r[2]])
self.delete_sections_for_file(prefix, identity)
canonical = source_basename(identity) or identity
stale = list(paths) if paths else []
cur = self.conn.cursor()
if stale:
cur.execute(
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, stale),
)
if file_index:
cur.execute("""
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
VALUES (%s, %s, %s, 0, %s)
ON CONFLICT (prefix, file_path) DO UPDATE SET
file_hash = EXCLUDED.file_hash,
section_count = 0,
processed_at = NOW(),
file_index = EXCLUDED.file_index
""", (prefix, canonical, WIPED_HASH, file_index))
self.conn.commit()
return {
"file": canonical,
"prefix": prefix,
"deleted": len(codes),
"codes": codes,
"file_index": file_index,
}
def all_source_files(self, prefix: str) -> list[str]:
"""Връща всички source_file пътища за даден префикс."""
cur = self.conn.cursor()
cur.execute("""
SELECT file_path FROM rip_help_files WHERE prefix=%s
UNION
SELECT source_file FROM rip_help_sections WHERE prefix=%s
""", (prefix, prefix))
return [r[0] for r in cur.fetchall()]
def section_output_paths_for(self, prefix: str, source_files: list[str]) -> list[str]:
if not source_files:
return []
cur = self.conn.cursor()
cur.execute(
"SELECT output_path FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(source_files))
)
return [r[0] for r in cur.fetchall() if r[0]]
def purge_sources(self, prefix: str, source_files: list[str]) -> int:
if not source_files:
return 0
cur = self.conn.cursor()
cur.execute(
"DELETE FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(source_files))
)
sec_deleted = cur.rowcount
cur.execute(
"DELETE FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, list(source_files))
)
self.conn.commit()
return sec_deleted
def insert_section(self, prefix: str, ps: ProcessedSection, output_path: str):
cur = self.conn.cursor()
cur.execute("""
INSERT INTO rip_help_sections
(prefix, code, source_file, title, keywords,
char_count, output_path, images, html_text)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (code) DO UPDATE SET
prefix = EXCLUDED.prefix,
source_file = EXCLUDED.source_file,
title = EXCLUDED.title,
keywords = EXCLUDED.keywords,
char_count = EXCLUDED.char_count,
output_path = EXCLUDED.output_path,
images = EXCLUDED.images,
html_text = EXCLUDED.html_text,
updated_at = NOW()
""", (prefix, ps.code, ps.source_file, ps.title, ps.keywords,
ps.char_count, output_path, ps.images_json, ps.html_text))
self.conn.commit()
def close(self):
self.conn.close()
# ──────────────────────────────────────────────
# Парсъри
# ──────────────────────────────────────────────
def file_hash(path: Path) -> str:
h = hashlib.sha256()
with open(path, "rb") as f:
for chunk in iter(lambda: f.read(65536), b""):
h.update(chunk)
return h.hexdigest()
def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS."""
if not src:
return None
s = src.strip()
if s.startswith("data:"):
# data:image/png;base64,XXXX
m = re.match(r"data:([^;]+);base64,(.+)$", s, re.DOTALL)
if not m:
return None
import base64
try:
data = base64.b64decode(m.group(2))
except Exception:
return None
return data, _ext_from_content_type(m.group(1))
if s.startswith(("http://", "https://")):
return None # по правило пропускаме мрежови картинки
# локален път, относителен или абсолютен
p = (base_dir / s).resolve() if not Path(s).is_absolute() else Path(s)
try:
if p.is_file():
data = p.read_bytes()
ext = p.suffix.lstrip(".").lower() or "png"
return data, ext
except Exception:
return None
return None
def _detect_html_encoding(raw: bytes) -> str:
"""Връща име на encoding: BOM → chardet → fallback (utf-8 ако ASCII, иначе windows-1251)."""
# BOM-и
if raw.startswith(b"\xef\xbb\xbf"):
return "utf-8"
if raw.startswith((b"\xff\xfe", b"\xfe\xff")):
return "utf-16"
# chardet
try:
import chardet
det = chardet.detect(raw[:65536]) or {}
enc = (det.get("encoding") or "").lower()
conf = det.get("confidence", 0) or 0
if enc and conf >= 0.6:
# нормализиране на често срещани имена
if enc in ("cp1251", "ms-cyrl", "windows-1251"):
return "windows-1251"
if enc.startswith("utf"):
return enc
return enc
except Exception:
pass
# fallback: ако байтовете изглеждат "над 127" (т.е. има не-ASCII), приемаме CP1251
if any(b > 127 for b in raw[:8192]):
return "windows-1251"
return "utf-8"
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
"p", "ul", "ol", "table", "dl", "pre",
"blockquote", "figure", "hr"]
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
"valign", "width", "height", "bgcolor", "border")
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
_HEADING_TOKEN_RE = re.compile(
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
r"(\d+)?$",
re.I,
)
# Word „Title“ / MsoTitle ≠ Heading 1 — корица, не граница на секция.
_COVER_TITLE_TOKENS = frozenset({
"title", "mstotitle", "наименование", "doctitle", "documenttitle",
})
_HTML_HEADING_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:heading|заглавие|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
re.I,
)
_HTML_COVER_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:mso)?title(?:\d+)?(?:$|[\s_-])|"
r"(?:^|[\s_-])наименование(?:\d+)?(?:$|[\s_-])",
re.I,
)
_HTML_SUBTITLE_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:subtitle|подзаглавие)(?:\d+)?(?:$|[\s_-])",
re.I,
)
_TOC_HEADING_RE = re.compile(
r"^(съдържание|съдържанието|contents|table of contents|toc|"
r"inhaltsverzeichnis|оглавление|содержание)\s*:?\s*$",
re.I,
)
_HEADING_LEVEL = {
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
"subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
"подзаглавие": 2,
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
}
def _compact_style_token(s: str) -> str:
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
def _is_cover_title_token(token: str) -> bool:
return _compact_style_token(token) in _COVER_TITLE_TOKENS
def _is_toc_heading(text: str) -> bool:
return bool(_TOC_HEADING_RE.match((text or "").strip()))
# Top-level chapter: "1. Title" / "2) Title" — short line, not body/table/figure.
_NUMBERED_CHAPTER_RE = re.compile(
r"^(\d{1,2})[\.\)]\s+(\S.{0,100})$"
)
_FIGURE_CAPTION_RE = re.compile(
r"^(фигура|figure|abb\.?|рис\.?)\s*\d+",
re.I,
)
def _normalize_chapter_key(text: str) -> str:
return re.sub(r"\s+", " ", (text or "").strip().lower())
def _is_numbered_chapter_heading(text: str) -> bool:
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
t = (text or "").strip()
if not t or len(t) >= 120:
return False
if "|" in t:
return False
if _is_toc_heading(t):
return False
if _FIGURE_CAPTION_RE.match(t):
return False
m = _NUMBERED_CHAPTER_RE.match(t)
if not m:
return False
rest = m.group(2).strip()
words = rest.split()
if not words or len(words) > 12:
return False
# Body-like numbered sentences: "1. Copy the files into the folder."
if rest.endswith((".", "!", "?")) and len(words) > 4:
return False
return True
class _TocSplitState:
"""Tracks Съдържание / TOC so first '1. Title' stays in preamble, second starts a section."""
__slots__ = ("phase", "had_toc", "titles")
def __init__(self) -> None:
self.phase = False
self.had_toc = False
self.titles: set[str] = set()
def note_toc_heading(self) -> None:
self.phase = True
self.had_toc = True
def leave_phase(self) -> None:
self.phase = False
def handle_numbered(self, text: str) -> str:
"""Return 'toc' (keep in body), 'split' (new section), or 'no' (not numbered chapter)."""
if not _is_numbered_chapter_heading(text):
return "no"
key = _normalize_chapter_key(text)
if self.phase:
if key in self.titles:
self.phase = False
return "split"
self.titles.add(key)
return "toc"
# After TOC list/block, or docs without TOC: numbered chapter opens a section.
return "split"
def _heading_level_from_token(token: str) -> Optional[int]:
t = _compact_style_token(token)
if not t:
return None
if _is_cover_title_token(t):
return None
if t in _HEADING_LEVEL:
return _HEADING_LEVEL[t]
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
if not m:
return None
n = m.group(2)
if n and n.isdigit():
return min(int(n), 3)
kind = (m.group(1) or "").lower()
if kind in ("subtitle", "подзаглавие"):
return 2
if kind in ("title", "наименование"):
return None
return 1
def _docx_style_tokens(para) -> list[str]:
style = getattr(para, "style", None)
tokens: list[str] = []
seen: set[int] = set()
cur = style
while cur is not None and id(cur) not in seen:
seen.add(id(cur))
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
if token:
tokens.append(str(token))
cur = getattr(cur, "base_style", None)
return tokens
def _docx_is_cover_title(para) -> bool:
return any(_is_cover_title_token(t) for t in _docx_style_tokens(para))
def _docx_heading_level(para) -> Optional[int]:
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word.
Word Title / Наименование не са граница (виж _docx_is_cover_title)."""
if _docx_is_cover_title(para):
return None
for token in _docx_style_tokens(para):
lvl = _heading_level_from_token(token)
if lvl:
return lvl
try:
pPr = para._element.pPr
if pPr is not None and pPr.outlineLvl is not None:
val = int(pPr.outlineLvl.val)
text = (para.text or "").strip()
if 0 <= val <= 2 and text and len(text) < 120:
return min(val + 1, 3)
except Exception:
pass
return None
def _is_bold_heading_text(text: str, runs) -> bool:
if not text or len(text) > 120:
return False
useful = [r for r in (runs or []) if (r.text or "").strip()]
return bool(useful) and all(bool(r.bold) for r in useful)
def _html_is_cover_title(el) -> bool:
raw = el.get("class") if hasattr(el, "get") else None
classes: list[str]
if isinstance(raw, str):
classes = raw.split()
else:
classes = list(raw or [])
for c in classes:
if _is_cover_title_token(c):
return True
return bool(_HTML_COVER_CLASS_RE.search(" ".join(classes)))
def _html_heading_level(el) -> Optional[int]:
name = (getattr(el, "name", None) or "").lower()
if name in _HTML_HEADING_MAP:
return _HTML_HEADING_MAP[name]
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
if _HTML_COVER_CLASS_RE.search(cls):
return None
if _HTML_SUBTITLE_CLASS_RE.search(cls):
return 2
m = _HTML_HEADING_CLASS_RE.search(cls)
if m:
n = m.group(1)
return min(int(n), 3) if n and n.isdigit() else 1
if name in ("p", "div"):
txt = el.get_text(" ", strip=True)
if txt and len(txt) < 120:
inner = "".join(el.stripped_strings)
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
if strong and strong_txt and strong_txt == inner:
return 2
return None
def _html_block_plain_text(el) -> str:
"""Plain ingest text: newlines inside lists/tables, spaces for inline runs."""
sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " "
return el.get_text(sep, strip=True)
def _strip_attrs(el):
"""Премахва decorative атрибути (class, style, on*, data-*)."""
for t in el.find_all(True):
for a in list(t.attrs):
if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"):
del t[a]
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
"""Намира всички в подадения елемент, извлича данните и подменя с
NavigableString placeholder ([IMG: img_NN])."""
from bs4 import NavigableString
for img in el.find_all("img"):
src = img.get("src") or img.get("data-src") or ""
loaded = _load_html_image(src, base_dir)
if not loaded:
img.decompose()
continue
data, ext = loaded
if not _should_keep_image(data):
img.decompose()
continue
img_counter[0] += 1
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
sec_images.append(ref)
img.replace_with(NavigableString(f"[IMG: {ref.placeholder}]"))
def parse_html(path: Path) -> list[Section]:
raw = path.read_bytes()
enc = _detect_html_encoding(raw)
log.debug(f" {path.name} encoding: {enc}")
try:
soup = BeautifulSoup(raw, "lxml", from_encoding=enc)
except Exception:
soup = BeautifulSoup(raw, "lxml")
# Премахваме скриптове и стилове
for tag in soup(["script", "style", "nav", "footer", "header", "noscript"]):
tag.decompose()
base_dir = path.parent
body = soup.body or soup
# Събираме top-level блокови елементи (без да включваме вложените в тях)
consumed = set()
blocks = []
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
name = (el.name or "").lower()
if name == "div" and not _html_heading_level(el) and not _html_is_cover_title(el):
continue
if any(id(par) in consumed for par in el.parents):
continue
consumed.add(id(el))
blocks.append(el)
sections: list[Section] = []
current_title = ""
current_level = 1
sec_text: list[str] = []
sec_html: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
toc_state = _TocSplitState()
def flush():
if current_title or sec_text or sec_html or sec_images:
sec = Section(current_title, "\n".join(sec_text), current_level)
sec.images = list(sec_images)
sec.html_text = "\n".join(sec_html) if sec_html else None
sections.append(sec)
def append_block_as_body(el):
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
_strip_attrs(el)
txt = _html_block_plain_text(el)
if txt:
sec_text.append(txt)
try:
sec_html.append(str(el))
except Exception:
pass
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, sec_text, sec_html, sec_images
flush()
current_title = title
current_level = level
sec_text, sec_html, sec_images = [], [], []
for el in blocks:
txt = el.get_text(" ", strip=True)
is_cover = _html_is_cover_title(el)
heading_lvl = _html_heading_level(el)
# Корица (Word Title / class Title): заглавие на преамбюла, без нова секция
if is_cover and txt:
if not current_title and not sec_text and not sec_html and not sec_images:
current_title = txt
current_level = 1
else:
if toc_state.phase:
toc_state.leave_phase()
append_block_as_body(el)
continue
if txt and _is_toc_heading(txt):
toc_state.note_toc_heading()
append_block_as_body(el)
continue
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
if numbered_action == "toc":
append_block_as_body(el)
continue
if numbered_action == "split":
start_section(txt, heading_lvl or 1)
continue
if heading_lvl:
if not txt:
continue
# H2+ без номерация остават в тялото; H1 / Заглавие 1 режат
if heading_lvl >= 2:
if toc_state.phase:
toc_state.leave_phase()
append_block_as_body(el)
continue
if toc_state.phase:
toc_state.leave_phase()
start_section(txt, heading_lvl)
continue
if el.name == "img":
if toc_state.phase:
toc_state.leave_phase()
# самостоятелен
(не вътре в блок)
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
base_dir, sec_images, img_counter)
# ако е заменен с placeholder, добавяме като текст
img_txt = el.get_text(" ", strip=True) if el.name else ""
if img_txt:
sec_text.append(img_txt)
sec_html.append(f"
{img_txt}
") continue if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS: #