Files
rip-help-system/help_processor.py
2026-09-14 12:57:10 +03:00

2396 lines
86 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""
help_processor.py
=================
Обработва help-файлове (.doc, .docx, .html, .htm, .txt, .pdf),
декомпозира ги на смислови секции, извлича ключови думи чрез Anthropic API
и записва резултатите в SQL Server + изходна директория.
Поддържа инкрементална обработка: файлове, чийто hash не се е променил,
се прескачат при повторно пускане.
Изисквания (pip install):
pip install anthropic pyodbc python-docx beautifulsoup4 lxml
pip install pdfplumber striprtf chardet
pip install pywin32 # за MS Word fallback на Windows
За .doc (стар формат) е необходим един от:
- LibreOffice (soffice в PATH) — кросплатформено
- MS Word — Windows, чрез pywin32 COM (автоматичен fallback)
- antiword — Linux (apt install antiword)
"""
import os
import re
import sys
import html as html_lib
import json
import hashlib
import logging
import argparse
import subprocess
import tempfile
from pathlib import Path
from datetime import datetime
from dataclasses import dataclass, field
from typing import Optional
import psycopg2
import anthropic
from docx import Document
from bs4 import BeautifulSoup
from help_codes import (
WIPED_HASH,
file_result as _file_result,
make_code,
parse_code,
remove_section_outputs,
source_basename,
source_identity,
)
try:
import pdfplumber
HAS_PDF = True
except ImportError:
HAS_PDF = False
try:
from PIL import Image
HAS_PIL = True
except ImportError:
HAS_PIL = False
# ──────────────────────────────────────────────
# Конфигурация
# ──────────────────────────────────────────────
# На Windows конзолата често е cp1251 → пренастройваме stdout на utf-8
try:
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
except AttributeError:
pass
_log_handlers = [logging.StreamHandler(sys.stdout)]
try:
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
except OSError:
pass
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)-8s %(message)s",
handlers=_log_handlers,
)
log = logging.getLogger(__name__)
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
AI_MODELS = [
AI_MODEL,
"claude-haiku-4-5",
"claude-haiku-4-5-20251001",
"claude-3-5-haiku-latest",
]
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
# ──────────────────────────────────────────────
# Изображения — помощни
# ──────────────────────────────────────────────
@dataclass
class ImageRef:
placeholder: str # вътрешен ID в текста, напр. "img_01"
data: bytes
ext: str # "png", "jpg", "gif"...
def _img_dimensions(data: bytes) -> Optional[tuple[int, int]]:
if not HAS_PIL:
return None
try:
from io import BytesIO
with Image.open(BytesIO(data)) as im:
return im.size
except Exception:
return None
def _should_keep_image(data: bytes) -> bool:
"""Връща False за дребни иконки/булети под MIN_IMAGE_PX × MIN_IMAGE_PX."""
if not data:
return False
dims = _img_dimensions(data)
if dims is None:
# Не можем да преценим — пазим по подразбиране
return True
w, h = dims
return w >= MIN_IMAGE_PX and h >= MIN_IMAGE_PX
def _ext_from_content_type(ct: str) -> str:
ct = (ct or "").lower()
if "png" in ct: return "png"
if "jpeg" in ct or "jpg" in ct: return "jpg"
if "gif" in ct: return "gif"
if "bmp" in ct: return "bmp"
if "svg" in ct: return "svg"
if "webp" in ct: return "webp"
return "png"
_IMG_PLACEHOLDER_RE = re.compile(r"\[IMG:\s*([A-Za-z0-9_./\\-]+)\s*\]")
# ──────────────────────────────────────────────
# Структури
# ──────────────────────────────────────────────
@dataclass
class Section:
title: str
text: str
level: int = 1 # 1=H1, 2=H2, 3=H3, 0=без заглавие
images: list = field(default_factory=list) # list[ImageRef]
html_text: Optional[str] = None # rich HTML с [IMG: ...] placeholders
@dataclass
class ProcessedSection:
code: str # DOC_003_SEC_012
source_file: str
title: str
keywords: str # "кл1, кл2, кл3"
text: str
images_json: str = "[]" # JSON масив с относителни пътища
html_text: str = "" # rich HTML (HTML + DOCX; bold/цветове/картинки)
char_count: int = 0
def __post_init__(self):
self.char_count = len(self.text)
# ──────────────────────────────────────────────
# База данни
# ──────────────────────────────────────────────
class Database:
"""PostgreSQL backend (psycopg2). Connection string е libpq формат:
'host=... port=... dbname=... user=... password=...'
"""
def __init__(self, conn_str: str):
self.conn_str = conn_str
self.conn = psycopg2.connect(conn_str)
self._ensure_schema()
def _ensure_schema(self):
"""Създава таблиците ако не съществуват (Postgres syntax)."""
cur = self.conn.cursor()
cur.execute("""
CREATE TABLE IF NOT EXISTS rip_help_files (
id SERIAL PRIMARY KEY,
prefix VARCHAR(50) NOT NULL DEFAULT 'HLP',
file_path VARCHAR(1000) NOT NULL,
file_hash CHAR(64) NOT NULL,
processed_at TIMESTAMP NOT NULL DEFAULT NOW(),
section_count INTEGER NOT NULL DEFAULT 0,
UNIQUE (prefix, file_path)
)
""")
cur.execute("""
CREATE TABLE IF NOT EXISTS rip_help_sections (
id SERIAL PRIMARY KEY,
prefix VARCHAR(50) NOT NULL DEFAULT 'HLP',
code VARCHAR(80) NOT NULL UNIQUE,
source_file VARCHAR(1000) NOT NULL,
title VARCHAR(500),
keywords VARCHAR(300),
char_count INTEGER,
output_path VARCHAR(1000),
images TEXT,
html_text TEXT,
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
)
""")
cur.execute("""
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_keywords
ON rip_help_sections(keywords)
""")
cur.execute("""
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_prefix
ON rip_help_sections(prefix)
""")
cur.execute("""
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_source
ON rip_help_sections(prefix, source_file)
""")
cur.execute(
"ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER"
)
self.conn.commit()
log.info("Схемата е проверена / създадена.")
def matching_source_paths(self, prefix: str, identity: str) -> list[str]:
"""Всички file_path/source_file за същия help файл (basename, вкл. стари temp пътища)."""
name = source_basename(identity).lower()
if not name:
return []
found: list[str] = []
seen: set[str] = set()
for p in self.all_source_files(prefix):
if source_basename(p).lower() == name and p not in seen:
seen.add(p)
found.append(p)
return found
def get_file_hash(self, prefix: str, file_path: str) -> Optional[str]:
paths = self.matching_source_paths(prefix, file_path) or [file_path]
cur = self.conn.cursor()
cur.execute(
"SELECT file_hash FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, list(paths)),
)
for (h,) in cur.fetchall():
if not h:
continue
val = str(h).strip()
if val and val != WIPED_HASH:
return val
return None
def upsert_file(
self,
prefix: str,
file_path: str,
file_hash: str,
section_count: int,
file_index: Optional[int] = None,
):
canonical = source_basename(file_path) or file_path
stale = [p for p in self.matching_source_paths(prefix, canonical) if p != canonical]
cur = self.conn.cursor()
if stale:
cur.execute(
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, stale),
)
cur.execute("""
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
VALUES (%s, %s, %s, %s, %s)
ON CONFLICT (prefix, file_path) DO UPDATE SET
file_hash = EXCLUDED.file_hash,
section_count= EXCLUDED.section_count,
processed_at = NOW(),
file_index = COALESCE(EXCLUDED.file_index, rip_help_files.file_index)
""", (prefix, canonical, file_hash, section_count, file_index))
self.conn.commit()
def delete_sections_for_file(self, prefix: str, file_path: str):
paths = self.matching_source_paths(prefix, file_path) or [file_path]
cur = self.conn.cursor()
cur.execute(
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
self.conn.commit()
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
"""(code, source_file, output_path) за всички секции на файла."""
paths = self.matching_source_paths(prefix, identity)
if not paths:
return []
cur = self.conn.cursor()
cur.execute(
"SELECT code, source_file, output_path FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code",
(prefix, list(paths)),
)
return [(r[0], r[1], r[2]) for r in cur.fetchall()]
def file_index_for(self, prefix: str, identity: str) -> Optional[int]:
paths = self.matching_source_paths(prefix, identity)
if not paths:
return None
cur = self.conn.cursor()
cur.execute(
"SELECT file_index FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL",
(prefix, list(paths)),
)
from_col = [r[0] for r in cur.fetchall() if r[0]]
if from_col:
return min(from_col)
cur.execute(
"SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
found: list[int] = []
for (code,) in cur.fetchall():
parsed = parse_code(code)
if parsed:
found.append(parsed[1])
if not found:
return None
tally: dict[int, int] = {}
for idx in found:
tally[idx] = tally.get(idx, 0) + 1
return max(tally, key=lambda k: (tally[k], -k))
def max_file_index(self, prefix: str) -> int:
cur = self.conn.cursor()
cur.execute(
"SELECT COALESCE(MAX(file_index), 0) FROM rip_help_files WHERE prefix=%s",
(prefix,),
)
m1 = int(cur.fetchone()[0] or 0)
cur.execute("SELECT code FROM rip_help_sections WHERE prefix=%s", (prefix,))
m2 = 0
for (code,) in cur.fetchall():
parsed = parse_code(code)
if parsed:
m2 = max(m2, parsed[1])
return max(m1, m2)
def file_index_used_by_others(self, prefix: str, identity: str, idx: int) -> bool:
name = source_basename(identity).lower()
cur = self.conn.cursor()
cur.execute(
"SELECT code, source_file FROM rip_help_sections WHERE prefix=%s",
(prefix,),
)
for code, src in cur.fetchall():
parsed = parse_code(code)
if parsed and parsed[1] == idx and source_basename(src).lower() != name:
return True
return False
def allocate_file_index(self, prefix: str, identity: str) -> int:
existing = self.file_index_for(prefix, identity)
if existing and not self.file_index_used_by_others(prefix, identity, existing):
return existing
return self.max_file_index(prefix) + 1
def wipe_extractions_for_file(
self,
prefix: str,
identity: str,
output_dir: Optional[Path] = None,
) -> dict:
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
rows = self.sections_for_file(prefix, identity)
codes = [r[0] for r in rows]
file_index = self.file_index_for(prefix, identity)
if file_index and self.file_index_used_by_others(prefix, identity, file_index):
file_index = None
paths = self.matching_source_paths(prefix, identity)
if output_dir:
remove_section_outputs(output_dir, codes, [r[2] for r in rows if r[2]])
self.delete_sections_for_file(prefix, identity)
canonical = source_basename(identity) or identity
stale = list(paths) if paths else []
cur = self.conn.cursor()
if stale:
cur.execute(
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, stale),
)
if file_index:
cur.execute("""
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
VALUES (%s, %s, %s, 0, %s)
ON CONFLICT (prefix, file_path) DO UPDATE SET
file_hash = EXCLUDED.file_hash,
section_count = 0,
processed_at = NOW(),
file_index = EXCLUDED.file_index
""", (prefix, canonical, WIPED_HASH, file_index))
self.conn.commit()
return {
"file": canonical,
"prefix": prefix,
"deleted": len(codes),
"codes": codes,
"file_index": file_index,
}
def all_source_files(self, prefix: str) -> list[str]:
"""Връща всички source_file пътища за даден префикс."""
cur = self.conn.cursor()
cur.execute("""
SELECT file_path FROM rip_help_files WHERE prefix=%s
UNION
SELECT source_file FROM rip_help_sections WHERE prefix=%s
""", (prefix, prefix))
return [r[0] for r in cur.fetchall()]
def section_output_paths_for(self, prefix: str, source_files: list[str]) -> list[str]:
if not source_files:
return []
cur = self.conn.cursor()
cur.execute(
"SELECT output_path FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(source_files))
)
return [r[0] for r in cur.fetchall() if r[0]]
def purge_sources(self, prefix: str, source_files: list[str]) -> int:
if not source_files:
return 0
cur = self.conn.cursor()
cur.execute(
"DELETE FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(source_files))
)
sec_deleted = cur.rowcount
cur.execute(
"DELETE FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, list(source_files))
)
self.conn.commit()
return sec_deleted
def insert_section(self, prefix: str, ps: ProcessedSection, output_path: str):
cur = self.conn.cursor()
cur.execute("""
INSERT INTO rip_help_sections
(prefix, code, source_file, title, keywords,
char_count, output_path, images, html_text)
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s)
ON CONFLICT (code) DO UPDATE SET
prefix = EXCLUDED.prefix,
source_file = EXCLUDED.source_file,
title = EXCLUDED.title,
keywords = EXCLUDED.keywords,
char_count = EXCLUDED.char_count,
output_path = EXCLUDED.output_path,
images = EXCLUDED.images,
html_text = EXCLUDED.html_text,
updated_at = NOW()
""", (prefix, ps.code, ps.source_file, ps.title, ps.keywords,
ps.char_count, output_path, ps.images_json, ps.html_text))
self.conn.commit()
def close(self):
self.conn.close()
# ──────────────────────────────────────────────
# Парсъри
# ──────────────────────────────────────────────
def file_hash(path: Path) -> str:
h = hashlib.sha256()
with open(path, "rb") as f:
for chunk in iter(lambda: f.read(65536), b""):
h.update(chunk)
return h.hexdigest()
def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS.
Относителните ``src`` се резолвират спрямо директорията на HTML файла.
URL-encoding (``%20``) се декодира — типично за Word/LibreOffice „Save as HTML“.
"""
from urllib.parse import unquote
if not src:
return None
s = src.strip()
if s.startswith("data:"):
# data:image/png;base64,XXXX
m = re.match(r"data:([^;]+);base64,(.+)$", s, re.DOTALL)
if not m:
return None
import base64
try:
data = base64.b64decode(m.group(2))
except Exception:
return None
return data, _ext_from_content_type(m.group(1))
if s.startswith(("http://", "https://")):
return None # по правило пропускаме мрежови картинки
if s.lower().startswith("file:"):
path = _file_url_to_path(s)
if path is None:
log.warning(f" HTML image bad file URL: {src}")
return None
loaded = _read_image_file(path)
if not loaded:
log.warning(f" HTML image not found: {src} → {path}")
return loaded
# локален път: %20 → space, ../ спрямо HTML папката
decoded = unquote(s).replace("\\", "/")
try:
p = Path(decoded)
if not p.is_absolute():
p = (base_dir / decoded).resolve()
if p.is_file():
data = p.read_bytes()
ext = p.suffix.lstrip(".").lower() or "png"
return data, ext
log.warning(f" HTML image not found: {src} → {p}")
except Exception as e:
log.warning(f" HTML image unreadable: {src}: {e}")
return None
return None
def _detect_html_encoding(raw: bytes) -> str:
"""Връща име на encoding: BOM → chardet → fallback (utf-8 ако ASCII, иначе windows-1251)."""
# BOM-и
if raw.startswith(b"\xef\xbb\xbf"):
return "utf-8"
if raw.startswith((b"\xff\xfe", b"\xfe\xff")):
return "utf-16"
# chardet
try:
import chardet
det = chardet.detect(raw[:65536]) or {}
enc = (det.get("encoding") or "").lower()
conf = det.get("confidence", 0) or 0
if enc and conf >= 0.6:
# нормализиране на често срещани имена
if enc in ("cp1251", "ms-cyrl", "windows-1251"):
return "windows-1251"
if enc.startswith("utf"):
return enc
return enc
except Exception:
pass
# fallback: ако байтовете изглеждат "над 127" (т.е. има не-ASCII), приемаме CP1251
if any(b > 127 for b in raw[:8192]):
return "windows-1251"
return "utf-8"
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
"p", "ul", "ol", "table", "dl", "pre",
"blockquote", "figure", "hr"]
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
"valign", "width", "height", "bgcolor", "border")
# От style пазим само inline форматиране, нужно за viewer (bold/italic/color).
_HTML_STYLE_KEEP_RES = (
("color", re.compile(r"(?:^|;)\s*color\s*:\s*([^;]+)", re.I)),
("font-weight", re.compile(r"(?:^|;)\s*font-weight\s*:\s*([^;]+)", re.I)),
("font-style", re.compile(r"(?:^|;)\s*font-style\s*:\s*([^;]+)", re.I)),
)
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
_HEADING_TOKEN_RE = re.compile(
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
r"(\d+)?$",
re.I,
)
# Word „Title“ / MsoTitle ≠ Heading 1 — корица, не граница на секция.
_COVER_TITLE_TOKENS = frozenset({
"title", "mstotitle", "наименование", "doctitle", "documenttitle",
})
_HTML_HEADING_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:heading|заглавие|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
re.I,
)
_HTML_COVER_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:mso)?title(?:\d+)?(?:$|[\s_-])|"
r"(?:^|[\s_-])наименование(?:\d+)?(?:$|[\s_-])",
re.I,
)
_HTML_SUBTITLE_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:subtitle|подзаглавие)(?:\d+)?(?:$|[\s_-])",
re.I,
)
_TOC_HEADING_RE = re.compile(
r"^(съдържание|съдържанието|contents|table of contents|toc|"
r"inhaltsverzeichnis|оглавление|содержание)\s*:?\s*$",
re.I,
)
_HEADING_LEVEL = {
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
"subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
"подзаглавие": 2,
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
}
def _compact_style_token(s: str) -> str:
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
def _is_cover_title_token(token: str) -> bool:
return _compact_style_token(token) in _COVER_TITLE_TOKENS
def _is_toc_heading(text: str) -> bool:
return bool(_TOC_HEADING_RE.match((text or "").strip()))
# Top-level chapter: "1. Title" / "2) Title" — short line, not body/table/figure.
_NUMBERED_CHAPTER_RE = re.compile(
r"^(\d{1,2})[\.\)]\s+(\S.{0,100})$"
)
_FIGURE_CAPTION_RE = re.compile(
r"^(фигура|figure|abb\.?|рис\.?)\s*\d+",
re.I,
)
def _normalize_inline_ws(text: str) -> str:
"""Срива CR/LF/табове в един интервал (Word/LibreOffice soft breaks в <h2>)."""
return re.sub(r"\s+", " ", (text or "").strip())
def _normalize_chapter_key(text: str) -> str:
return _normalize_inline_ws(text).lower()
def _is_numbered_chapter_heading(text: str) -> bool:
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
t = _normalize_inline_ws(text)
if not t or len(t) >= 120:
return False
if "|" in t:
return False
if _is_toc_heading(t):
return False
if _FIGURE_CAPTION_RE.match(t):
return False
m = _NUMBERED_CHAPTER_RE.match(t)
if not m:
return False
rest = m.group(2).strip()
words = rest.split()
if not words or len(words) > 12:
return False
# Body-like numbered sentences: "1. Copy the files into the folder."
if rest.endswith((".", "!", "?")) and len(words) > 4:
return False
return True
class _TocSplitState:
"""Tracks Съдържание / TOC so first '1. Title' stays in preamble, second starts a section."""
__slots__ = ("phase", "had_toc", "titles")
def __init__(self) -> None:
self.phase = False
self.had_toc = False
self.titles: set[str] = set()
def note_toc_heading(self) -> None:
self.phase = True
self.had_toc = True
def leave_phase(self) -> None:
self.phase = False
def handle_numbered(self, text: str) -> str:
"""Return 'toc' (keep in body), 'split' (new section), or 'no' (not numbered chapter)."""
if not _is_numbered_chapter_heading(text):
return "no"
key = _normalize_chapter_key(text)
if self.phase:
if key in self.titles:
self.phase = False
return "split"
self.titles.add(key)
return "toc"
# After TOC list/block, or docs without TOC: numbered chapter opens a section.
return "split"
def _heading_level_from_token(token: str) -> Optional[int]:
t = _compact_style_token(token)
if not t:
return None
if _is_cover_title_token(t):
return None
if t in _HEADING_LEVEL:
return _HEADING_LEVEL[t]
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
if not m:
return None
n = m.group(2)
if n and n.isdigit():
return min(int(n), 3)
kind = (m.group(1) or "").lower()
if kind in ("subtitle", "подзаглавие"):
return 2
if kind in ("title", "наименование"):
return None
return 1
def _docx_style_tokens(para) -> list[str]:
style = getattr(para, "style", None)
tokens: list[str] = []
seen: set[int] = set()
cur = style
while cur is not None and id(cur) not in seen:
seen.add(id(cur))
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
if token:
tokens.append(str(token))
cur = getattr(cur, "base_style", None)
return tokens
def _docx_is_cover_title(para) -> bool:
return any(_is_cover_title_token(t) for t in _docx_style_tokens(para))
def _docx_heading_level(para) -> Optional[int]:
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word.
Word Title / Наименование не са граница (виж _docx_is_cover_title)."""
if _docx_is_cover_title(para):
return None
for token in _docx_style_tokens(para):
lvl = _heading_level_from_token(token)
if lvl:
return lvl
try:
pPr = para._element.pPr
if pPr is not None and pPr.outlineLvl is not None:
val = int(pPr.outlineLvl.val)
text = (para.text or "").strip()
if 0 <= val <= 2 and text and len(text) < 120:
return min(val + 1, 3)
except Exception:
pass
return None
def _is_bold_heading_text(text: str, runs) -> bool:
if not text or len(text) > 120:
return False
useful = [r for r in (runs or []) if (r.text or "").strip()]
return bool(useful) and all(bool(r.bold) for r in useful)
def _html_is_cover_title(el) -> bool:
raw = el.get("class") if hasattr(el, "get") else None
classes: list[str]
if isinstance(raw, str):
classes = raw.split()
else:
classes = list(raw or [])
for c in classes:
if _is_cover_title_token(c):
return True
return bool(_HTML_COVER_CLASS_RE.search(" ".join(classes)))
def _html_heading_level(el) -> Optional[int]:
name = (getattr(el, "name", None) or "").lower()
if name in _HTML_HEADING_MAP:
return _HTML_HEADING_MAP[name]
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
if _HTML_COVER_CLASS_RE.search(cls):
return None
if _HTML_SUBTITLE_CLASS_RE.search(cls):
return 2
m = _HTML_HEADING_CLASS_RE.search(cls)
if m:
n = m.group(1)
return min(int(n), 3) if n and n.isdigit() else 1
if name in ("p", "div"):
txt = el.get_text(" ", strip=True)
if txt and len(txt) < 120:
inner = "".join(el.stripped_strings)
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
if strong and strong_txt and strong_txt == inner:
return 2
return None
def _html_block_plain_text(el) -> str:
"""Plain ingest text: newlines inside lists/tables, spaces for inline runs."""
sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " "
return el.get_text(sep, strip=True)
def _kept_inline_style(style: str) -> str:
"""Извлича color / bold / italic от CSS style; останалото се маха."""
if not style:
return ""
parts: list[str] = []
for prop, rx in _HTML_STYLE_KEEP_RES:
m = rx.search(style)
if not m:
continue
val = m.group(1).strip()
if not val:
continue
low = val.lower()
if prop == "font-weight" and low not in (
"bold", "bolder", "600", "700", "800", "900"
):
continue
if prop == "font-style" and "italic" not in low and "oblique" not in low:
continue
parts.append(f"{prop}:{val}")
return ";".join(parts)
def _strip_attrs(el):
"""Премахва decorative атрибути; пази color/bold/italic от style и font color=."""
for t in el.find_all(True):
kept_style = ""
raw_style = t.attrs.get("style") if t.attrs else None
if isinstance(raw_style, str):
kept_style = _kept_inline_style(raw_style)
for a in list(t.attrs):
if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"):
del t[a]
# <font color="..."> остава (color не е в DROP)
if kept_style:
t["style"] = kept_style
def _ingest_html_img(
img, base_dir: Path, sec_images: list, img_counter: list
) -> Optional[str]:
"""Извлича една <img> картинка; връща placeholder текст или None."""
from bs4 import NavigableString
src = img.get("src") or img.get("data-src") or ""
loaded = _load_html_image(src, base_dir)
if not loaded:
img.decompose()
return None
data, ext = loaded
if not _should_keep_image(data):
img.decompose()
return None
img_counter[0] += 1
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
sec_images.append(ref)
ph = f"[IMG: {ref.placeholder}]"
img.replace_with(NavigableString(ph))
return ph
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
"""Намира всички <img> в подадения елемент, извлича данните и подменя с
NavigableString placeholder ([IMG: img_NN])."""
for img in el.find_all("img"):
_ingest_html_img(img, base_dir, sec_images, img_counter)
def parse_html(path: Path) -> list[Section]:
raw = path.read_bytes()
enc = _detect_html_encoding(raw)
log.debug(f" {path.name} encoding: {enc}")
try:
soup = BeautifulSoup(raw, "lxml", from_encoding=enc)
except Exception:
soup = BeautifulSoup(raw, "lxml")
# Премахваме скриптове и стилове
for tag in soup(["script", "style", "nav", "footer", "header", "noscript"]):
tag.decompose()
base_dir = path.parent
body = soup.body or soup
# Събираме top-level блокови елементи (без да включваме вложените в тях)
consumed = set()
blocks = []
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
name = (el.name or "").lower()
if name == "div" and not _html_heading_level(el) and not _html_is_cover_title(el):
continue
if any(id(par) in consumed for par in el.parents):
continue
consumed.add(id(el))
blocks.append(el)
sections: list[Section] = []
current_title = ""
current_level = 1
sec_text: list[str] = []
sec_html: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
toc_state = _TocSplitState()
def flush():
if current_title or sec_text or sec_html or sec_images:
sec = Section(current_title, "\n".join(sec_text), current_level)
sec.images = list(sec_images)
sec.html_text = "\n".join(sec_html) if sec_html else None
sections.append(sec)
def append_block_as_body(el):
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
_strip_attrs(el)
txt = _html_block_plain_text(el)
if txt:
sec_text.append(txt)
try:
sec_html.append(str(el))
except Exception:
pass
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, sec_text, sec_html, sec_images
flush()
current_title = _normalize_inline_ws(title)
current_level = level
sec_text, sec_html, sec_images = [], [], []
for el in blocks:
# Soft breaks (\r\n в средата на Word/LO <h2>) → един ред за split/TOC
txt = _normalize_inline_ws(el.get_text(" ", strip=True))
is_cover = _html_is_cover_title(el)
heading_lvl = _html_heading_level(el)
el_name = (el.name or "").lower()
# Корица (Word Title / class Title): заглавие на преамбюла, без нова секция
if is_cover and txt:
if not current_title and not sec_text and not sec_html and not sec_images:
current_title = txt
current_level = 1
else:
if toc_state.phase:
toc_state.leave_phase()
append_block_as_body(el)
continue
if txt and _is_toc_heading(txt):
toc_state.note_toc_heading()
append_block_as_body(el)
continue
# <ul>/<ol>/table в TOC: целият блок НЕ е една „1. …“ глава — край на TOC фазата
if toc_state.phase and el_name in _HTML_PLAIN_NL_TAGS:
toc_state.leave_phase()
append_block_as_body(el)
continue
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
if numbered_action == "toc":
append_block_as_body(el)
continue
if numbered_action == "split":
start_section(txt, heading_lvl or 1)
continue
if heading_lvl:
if not txt:
continue
# H2+ без номерация остават в тялото; H1 / Заглавие 1 режат
if heading_lvl >= 2:
if toc_state.phase:
toc_state.leave_phase()
append_block_as_body(el)
continue
if toc_state.phase:
toc_state.leave_phase()
start_section(txt, heading_lvl)
continue
if el_name == "img":
if toc_state.phase:
toc_state.leave_phase()
ph = _ingest_html_img(el, base_dir, sec_images, img_counter)
if ph:
sec_text.append(ph)
sec_html.append(f"<p>{ph}</p>")
continue
if toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
toc_state.leave_phase()
append_block_as_body(el)
flush()
if not sections:
plain = body.get_text(" ", strip=True)
return [Section("", plain, 0)]
return sections
def _file_url_to_path(url: str) -> Optional[Path]:
"""file:///… или обикновен път → Path (вкл. UNC от Word r:link)."""
from urllib.parse import unquote
s = unquote((url or "").strip())
if not s:
return None
if s.lower().startswith("file:"):
s = s[5:]
while s.startswith("/"):
s = s[1:]
if not s:
return None
return Path(s)
def _image_ext_from_path(path: Path) -> str:
ext = (path.suffix or "").lstrip(".").lower() or "png"
return "jpg" if ext == "jpeg" else ext
def _read_image_file(path: Path) -> Optional[tuple[bytes, str]]:
"""Чете локален/UNC файл като (data, ext); None при липса/грешка."""
try:
if not path.is_file():
return None
data = path.read_bytes()
except OSError as e:
log.warning(f" Cannot read image file {path}: {e}")
return None
if not data:
return None
return data, _image_ext_from_path(path)
def _docx_linked_media_dir(docx_path: Path) -> Path:
"""Папка до .docx за опаковани r:link картинки: <stem>_media/."""
return docx_path.parent / f"{docx_path.stem}_media"
def _sidecar_image_candidates(
docx_path: Optional[Path], linked_path: Optional[Path], url: str = ""
) -> list[Path]:
"""Кандидати до .docx, ако UNC/file линкът е недостъпен (Coolify, офлайн)."""
if docx_path is None:
return []
name = ""
if linked_path is not None:
name = linked_path.name
if not name:
from urllib.parse import unquote
tail = unquote((url or "").replace("\\", "/").rstrip("/").split("/")[-1])
name = tail.split("?")[0] if tail else ""
if not name:
return []
parent = docx_path.parent
stem = docx_path.stem
return [
parent / name,
_docx_linked_media_dir(docx_path) / name,
parent / "media" / name,
parent / "images" / name,
]
def _pack_linked_image(docx_path: Path, filename: str, data: bytes) -> Optional[Path]:
"""Копира успешно прочетена r:link картинка до <stem>_media/ за следващи сканирания."""
if not filename or not data:
return None
dest_dir = _docx_linked_media_dir(docx_path)
dest = dest_dir / filename
try:
if dest.is_file() and dest.stat().st_size == len(data):
return dest
dest_dir.mkdir(parents=True, exist_ok=True)
dest.write_bytes(data)
log.info(f" Packed linked image → {dest}")
return dest
except OSError as e:
log.warning(f" Cannot pack linked image {filename}: {e}")
return None
def _load_external_image_bytes(
url: str, docx_path: Optional[Path] = None
) -> Optional[tuple[bytes, str]]:
"""Чете външна картинка (r:link); UNC/file, после sidecar до .docx; лог при провал."""
path = _file_url_to_path(url)
tried: list[Path] = []
if path is not None:
tried.append(path)
loaded = _read_image_file(path)
if loaded:
if docx_path is not None:
_pack_linked_image(docx_path, path.name, loaded[0])
return loaded
for cand in _sidecar_image_candidates(docx_path, path, url):
if any(cand == t for t in tried):
continue
tried.append(cand)
loaded = _read_image_file(cand)
if loaded:
log.info(f" External image from sidecar {cand}")
return loaded
targets = ", ".join(str(p) for p in tried) if tried else (url or "(empty)")
log.warning(f" Cannot read linked image (r:link): {targets}")
return None
def _resolve_docx_image_rid(
doc, rId: str, docx_path: Optional[Path] = None
) -> Optional[tuple[bytes, str]]:
"""r:embed → related_parts; r:link → външен файл / sidecar. Връща (data, ext)."""
if not rId:
return None
try:
part = doc.part.related_parts[rId]
data = part.blob
ct = getattr(part, "content_type", "") or ""
if data:
return data, _ext_from_content_type(ct)
except Exception:
pass
try:
rel = doc.part.rels[rId]
except Exception:
return None
if not getattr(rel, "is_external", False):
return None
target = getattr(rel, "target_ref", None) or ""
return _load_external_image_bytes(target, docx_path=docx_path)
def _extract_docx_paragraph_images(
para, doc, docx_path: Optional[Path] = None
) -> list[ImageRef]:
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC + sidecar до .docx).
"""
from docx.oxml.ns import qn
imgs: list[ImageRef] = []
try:
blips = para._element.findall(".//" + qn("a:blip"))
except Exception:
return imgs
embed_attr = qn("r:embed")
link_attr = qn("r:link")
for blip in blips:
# Предпочитаме вградено копие, ако Word е запазил и двете
rId = blip.get(embed_attr) or blip.get(link_attr)
if not rId:
continue
resolved = _resolve_docx_image_rid(doc, rId, docx_path=docx_path)
if not resolved:
# Ако има само link и се провали — опитай обратното атрибутче
other = blip.get(link_attr) if blip.get(embed_attr) else blip.get(embed_attr)
if other and other != rId:
resolved = _resolve_docx_image_rid(doc, other, docx_path=docx_path)
if not resolved:
continue
data, ext = resolved
if not _should_keep_image(data):
continue
imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext))
return imgs
def _docx_run_color_css(run) -> Optional[str]:
"""RGB от run.font.color → '#RRGGBB'; theme/auto цветове се пропускат."""
try:
color = run.font.color
if color is None or color.rgb is None:
return None
return f"#{color.rgb}"
except Exception:
return None
def _docx_style_font_flag(style, attr: str) -> Optional[bool]:
"""True/False от style.font.<attr>; None ако стилът/атрибутът липсва."""
if style is None:
return None
try:
font = style.font
if font is None:
return None
val = getattr(font, attr, None)
if val is None:
return None
return bool(val)
except Exception:
return None
def _docx_run_effective_flag(run, para, attr: str) -> bool:
"""Ефективен bold/italic/underline: изричен run → char style → paragraph style."""
try:
direct = getattr(run, attr, None)
except Exception:
direct = None
if direct is True:
return True
if direct is False:
return False
# None = наследяване
try:
char_style = run.style
except Exception:
char_style = None
flag = _docx_style_font_flag(char_style, attr)
if flag is not None:
return flag
try:
p_style = para.style if para is not None else None
except Exception:
p_style = None
flag = _docx_style_font_flag(p_style, attr)
return bool(flag)
def _iter_docx_para_runs(para):
"""Runs в параграф, вкл. вътре в w:hyperlink (python-docx.para.runs ги пропуска)."""
from docx.oxml.ns import qn
from docx.text.run import Run
for child in para._element.iterchildren():
if child.tag == qn("w:r"):
yield Run(child, para)
elif child.tag == qn("w:hyperlink"):
for r_elem in child.findall(qn("w:r")):
yield Run(r_elem, para)
def _docx_run_to_html(run, para=None) -> str:
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
text = run.text or ""
if not text:
return ""
s = html_lib.escape(text, quote=False).replace("\n", "<br>")
if _docx_run_effective_flag(run, para, "bold"):
s = f"<b>{s}</b>"
if _docx_run_effective_flag(run, para, "italic"):
s = f"<i>{s}</i>"
if _docx_run_effective_flag(run, para, "underline"):
s = f"<u>{s}</u>"
css_color = _docx_run_color_css(run)
if css_color:
s = f'<span style="color:{css_color}">{s}</span>'
return s
def _docx_para_to_html(para) -> str:
"""Параграф → <p>…</p> с inline форматиране от runs (вкл. hyperlink TOC)."""
parts = [_docx_run_to_html(r, para) for r in _iter_docx_para_runs(para)]
inner = "".join(parts)
if not inner.strip():
# Fallback: para.text вижда hyperlink текст, но без runs в .runs
plain = (para.text or "").strip()
if not plain:
return ""
inner = html_lib.escape(plain, quote=False).replace("\n", "<br>")
return f"<p>{inner}</p>"
def _table_to_html(table) -> str:
"""Таблица → прост HTML <table> (plain клетки, без вложен rich text)."""
rows_html: list[str] = []
try:
rows = table.rows
except Exception:
return ""
for row in rows:
cells_html: list[str] = []
try:
cells = row.cells
except Exception:
continue
for cell in cells:
cell_text = " ".join((cell.text or "").split())
if not cell_text:
continue
cells_html.append(f"<td>{html_lib.escape(cell_text, quote=False)}</td>")
if cells_html:
rows_html.append("<tr>" + "".join(cells_html) + "</tr>")
if not rows_html:
return ""
return "<table>" + "".join(rows_html) + "</table>"
def _iter_docx_blocks(doc):
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
from docx.oxml.ns import qn
from docx.table import Table
from docx.text.paragraph import Paragraph
body = doc.element.body
for child in body.iterchildren():
if child.tag == qn("w:p"):
yield "p", Paragraph(child, doc)
elif child.tag == qn("w:tbl"):
yield "tbl", Table(child, doc)
def _table_lines(table) -> list[str]:
lines: list[str] = []
try:
rows = table.rows
except Exception:
return lines
for row in rows:
try:
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
except Exception:
continue
cells = [c for c in cells if c]
if cells:
lines.append(" | ".join(cells))
return lines
def parse_docx(path: Path) -> list[Section]:
docx_path = Path(path)
doc = Document(docx_path)
sections: list[Section] = []
current_title, current_level = "", 1
buf: list[str] = []
buf_html: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
toc_state = _TocSplitState()
def flush():
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), current_level)
sec.images = list(sec_images)
sec.html_text = "\n".join(buf_html) if buf_html else None
sections.append(sec)
def append_para(text: str, para_imgs: list[ImageRef], html_frag: str = ""):
if text:
buf.append(text)
if html_frag:
buf_html.append(html_frag)
for im in para_imgs:
img_counter[0] += 1
im.placeholder = f"img_{img_counter[0]:02d}"
sec_images.append(im)
ph = f"[IMG: {im.placeholder}]"
buf.append(ph)
buf_html.append(f"<p>{ph}</p>")
def append_from_para(para, text: str, para_imgs: list[ImageRef]):
html_frag = _docx_para_to_html(para) if text else ""
append_para(text, para_imgs, html_frag)
def start_section(title: str, level: int = 1):
nonlocal current_title, current_level, buf, buf_html, sec_images
flush()
buf, buf_html, sec_images = [], [], []
current_title = title
current_level = level
for kind, block in _iter_docx_blocks(doc):
if kind == "tbl":
if toc_state.phase:
toc_state.leave_phase()
for line in _table_lines(block):
buf.append(line)
th = _table_to_html(block)
if th:
buf_html.append(th)
continue
para = block
style_name = para.style.name.lower() if para.style else ""
text = para.text.strip()
para_imgs = _extract_docx_paragraph_images(para, doc, docx_path=docx_path)
if not text and not para_imgs:
continue
is_cover = _docx_is_cover_title(para)
level = _docx_heading_level(para)
is_bold_heading = (
not level
and not is_cover
and _is_bold_heading_text(text, para.runs)
and not style_name.startswith("list")
and not para_imgs
)
# Корица (Word Title): заглавие на преамбюла, без нова секция
if is_cover and text:
if not current_title and not buf and not sec_images:
current_title = text
current_level = 1
else:
if toc_state.phase:
toc_state.leave_phase()
append_from_para(para, text, para_imgs)
continue
if text and _is_toc_heading(text):
toc_state.note_toc_heading()
append_from_para(para, text, para_imgs)
continue
numbered_action = toc_state.handle_numbered(text) if text else "no"
if numbered_action == "toc":
append_from_para(para, text, para_imgs)
continue
if numbered_action == "split":
start_section(text, level or 1)
continue
# H2+/bold без номерация на глава — в тялото
if (level or 0) >= 2 or is_bold_heading:
if toc_state.phase:
toc_state.leave_phase()
append_from_para(para, text, para_imgs)
continue
if level == 1:
if toc_state.phase:
toc_state.leave_phase()
start_section(text, 1)
continue
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
toc_state.leave_phase()
append_from_para(para, text, para_imgs)
flush()
if not sections:
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
if not fallback_text:
fallback_text = "\n".join(
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
for line in _table_lines(block)
)
return [Section("", fallback_text, 0)]
return sections
def _convert_doc_with_libreoffice(path: Path, out_dir: Path) -> Optional[Path]:
try:
subprocess.run(
["soffice", "--headless", "--convert-to", "docx",
"--outdir", str(out_dir), str(path)],
check=True, capture_output=True, timeout=60
)
except (subprocess.CalledProcessError, FileNotFoundError, subprocess.TimeoutExpired) as e:
log.debug(f"LibreOffice конверсия неуспешна: {e}")
return None
out = list(out_dir.glob("*.docx"))
return out[0] if out else None
def _convert_doc_with_word(path: Path, out_dir: Path) -> Optional[Path]:
"""Fallback: ползва MS Word през COM на Windows."""
try:
import win32com.client # noqa: F401
import pythoncom
except ImportError:
log.debug("pywin32 не е инсталиран — MS Word fallback недостъпен.")
return None
import win32com.client as wcc
pythoncom.CoInitialize()
word = None
doc = None
try:
word = wcc.DispatchEx("Word.Application")
word.Visible = False
word.DisplayAlerts = False
doc = word.Documents.Open(str(path.resolve()), ReadOnly=True)
out_path = out_dir / (path.stem + ".docx")
# FileFormat=16 → wdFormatXMLDocument (.docx)
doc.SaveAs2(str(out_path.resolve()), FileFormat=16)
return out_path if out_path.exists() else None
except Exception as e:
log.debug(f"MS Word конверсия неуспешна: {e}")
return None
finally:
try:
if doc is not None:
doc.Close(SaveChanges=False)
except Exception:
pass
try:
if word is not None:
word.Quit()
except Exception:
pass
pythoncom.CoUninitialize()
def parse_doc_old(path: Path) -> list[Section]:
"""Конвертира стар .doc до .docx чрез LibreOffice или MS Word, после парси."""
with tempfile.TemporaryDirectory() as tmp:
tmp_dir = Path(tmp)
converted = _convert_doc_with_libreoffice(path, tmp_dir)
engine = "LibreOffice"
if not converted:
converted = _convert_doc_with_word(path, tmp_dir)
engine = "MS Word"
if not converted:
log.warning(
f"Нито LibreOffice, нито MS Word успяха да конвертират {path.name}. "
f"Пробваме като текст."
)
return parse_txt(path)
log.info(f" {path.name} конвертиран чрез {engine}")
return parse_docx(converted)
def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
"""Кропва картинката от PDF страницата и я записва като PNG bytes."""
try:
x0 = float(img_info.get("x0", 0))
x1 = float(img_info.get("x1", 0))
top = float(img_info.get("top", img_info.get("y0", 0)))
bot = float(img_info.get("bottom", img_info.get("y1", 0)))
if x1 <= x0 or bot <= top:
return None
# ограничаваме до страницата (pdfplumber иначе хвърля)
x0 = max(0, x0); top = max(0, top)
x1 = min(page.width, x1); bot = min(page.height, bot)
if x1 - x0 < 1 or bot - top < 1:
return None
cropped = page.crop((x0, top, x1, bot))
pil = cropped.to_image(resolution=resolution).original
from io import BytesIO
buf = BytesIO()
pil.save(buf, format="PNG")
return buf.getvalue()
except Exception as e:
log.debug(f"PDF image render failed: {e}")
return None
_PDF_LINE_Y_TOL = 3.0
# Typical word space in these PDFs is ~0.2–0.3em; only merge tighter (or drop-caps).
_PDF_LETTER_GAP_FRAC = 0.12
# Bare "1" / "2." / "3)" followed by a capital — TOC/list, not "виж 1 и 2".
_PDF_LIST_MARK_RE = re.compile(
r"(?<!\d)(?<![A-Za-zА-Яа-яЁёІіЇїЄєҐґ])(\d{1,2})([.)])?(?=\s+[A-ZА-ЯЁІЇЄҐ])"
)
def _pdf_word_band(w) -> tuple[float, float]:
top = float(w.get("top", 0))
bot = float(w.get("bottom", top + float(w.get("size") or 10)))
if bot <= top:
bot = top + max(float(w.get("size") or 10), 1.0)
return top, bot
def _pdf_cluster_words_by_y(words: list) -> list[dict]:
"""Visual lines by Y (and vertical overlap), not PDF stream order."""
items = [w for w in words if (w.get("text") or "").strip()]
items.sort(key=lambda w: (float(w.get("top", 0)), float(w.get("x0", 0))))
lines: list[dict] = []
for w in items:
top, bot = _pdf_word_band(w)
placed = False
for line in reversed(lines):
overlap = min(bot, line["bot"]) - max(top, line["top"])
band = min(bot - top, line["bot"] - line["top"])
close = abs(top - line["top"]) <= _PDF_LINE_Y_TOL
if close or (band > 0 and overlap >= 0.35 * band):
line["words"].append(w)
line["top"] = min(line["top"], top)
line["bot"] = max(line["bot"], bot)
placed = True
break
# Sorted by top: once this word sits clearly below a line, earlier
# lines are even higher.
if top > line["bot"] + _PDF_LINE_Y_TOL:
break
if not placed:
lines.append({"top": top, "bot": bot, "words": [w]})
lines.sort(key=lambda ln: ln["top"])
return lines
def _pdf_merge_split_letters(ws: list) -> list[dict]:
"""Join drop-cap / styled first letters: 'C' + 'orrelation' → 'Correlation'."""
ordered = sorted(ws, key=lambda w: float(w.get("x0", 0)))
clusters: list[dict] = []
for w in ordered:
text = (w.get("text") or "").strip()
if not text:
continue
x0 = float(w.get("x0", 0))
x1 = float(w.get("x1", x0))
size = float(w.get("size") or 10)
font = str(w.get("fontname") or "")
if not clusters:
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
continue
prev = clusters[-1]
gap = x0 - prev["x1"]
ref = max(size, prev["size"], 1.0)
fonts_differ = bool(prev["font"] and font and prev["font"] != font)
sizes_differ = abs(size - prev["size"]) > 0.6
drop = (
len(prev["text"]) == 1
and prev["text"].isalpha()
and text[0].islower()
)
tiny = gap < _PDF_LETTER_GAP_FRAC * ref or gap < 0.8
if tiny or (drop and gap < 0.55 * ref and (fonts_differ or sizes_differ or gap < 0.2 * ref)):
prev["text"] += text
prev["x1"] = max(prev["x1"], x1)
prev["size"] = max(prev["size"], size)
if font:
prev["font"] = font
else:
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
return clusters
def _pdf_split_list_text(text: str) -> list[str]:
"""Break a flattened TOC/list: '... 1 Foo 2 Bar' → one item per line."""
text = text.strip()
if not text:
return []
matches = list(_PDF_LIST_MARK_RE.finditer(text))
if len(matches) < 2:
return [text]
nums = [int(m.group(1)) for m in matches]
if nums[0] not in (1, 2):
return [text]
if any(nums[i] != nums[0] + i for i in range(len(nums))):
return [text]
parts: list[str] = []
last = 0
for m in matches:
if m.start() > last:
head = text[last:m.start()].strip()
if head:
parts.append(head)
last = m.start()
tail = text[last:].strip()
if tail:
parts.append(tail)
return parts or [text]
def _pdf_words_to_lines(words: list) -> list[dict]:
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
result = []
for line in _pdf_cluster_words_by_y(words):
ws = line["words"]
tokens = _pdf_merge_split_letters(ws)
if not tokens:
continue
joined = " ".join(t["text"] for t in tokens).strip()
if not joined:
continue
size = round(float(ws[0].get("size", 10)), 1)
for piece in _pdf_split_list_text(joined):
result.append({"top": line["top"], "size": size, "text": piece})
return result
def parse_pdf(path: Path) -> list[Section]:
if not HAS_PDF:
log.warning("pdfplumber не е инсталиран. PDF се прескача.")
return []
sections: list[Section] = []
current_title = ""
buf: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0]
prev_size = None
def flush():
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), 2)
sec.images = list(sec_images)
sections.append(sec)
with pdfplumber.open(path) as pdf:
for page in pdf.pages:
# Картинките за страницата (сортирани по y отгоре надолу)
page_images = sorted(
page.images or [],
key=lambda im: float(im.get("top", im.get("y0", 0)))
)
img_queue = []
for im in page_images:
data = _render_pdf_image(page, im)
if not data or not _should_keep_image(data):
continue
img_queue.append((float(im.get("top", 0)), data))
words = page.extract_words(extra_attrs=["size", "fontname"])
visual_lines = _pdf_words_to_lines(words)
line_buf, line_size = [], None
def emit_images_before(y: float):
while img_queue and img_queue[0][0] <= y:
_, data = img_queue.pop(0)
img_counter[0] += 1
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}",
data=data, ext="png")
sec_images.append(ref)
buf.append(f"[IMG: {ref.placeholder}]")
for vl in visual_lines:
sz = vl["size"]
y = vl["top"]
if line_size is None:
line_size = sz
if abs(sz - line_size) > 1:
line_text = "\n".join(line_buf).strip()
if line_text:
if line_size > (prev_size or 10) + 1 and len(line_text) < 150:
flush()
buf, sec_images = [], []
current_title = " ".join(line_text.split())
else:
emit_images_before(y)
buf.append(line_text)
prev_size = line_size
line_buf, line_size = [vl["text"]], sz
else:
line_buf.append(vl["text"])
if line_buf:
emit_images_before(page.height)
buf.append("\n".join(line_buf))
# картинките след всичкия текст на страницата
emit_images_before(page.height + 1)
flush()
return sections or [Section("", "", 0)]
def parse_txt(path: Path) -> list[Section]:
import chardet
raw = path.read_bytes()
enc = chardet.detect(raw)["encoding"] or "utf-8"
text = raw.decode(enc, errors="replace")
return _split_plain_text(text)
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
_PLAIN_NUM_HEADING_RE = re.compile(
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
)
def _split_plain_text(text: str) -> list[Section]:
"""Markdown / номерирани заглавия; иначе една секция."""
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
sections: list[Section] = []
title, level, buf = "", 0, []
def flush():
body = "\n".join(buf).strip()
if title or body:
sections.append(Section(title, body, level))
for line in lines:
raw = line.strip()
md = _PLAIN_MD_HEADING_RE.match(raw)
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
if md or numbered:
flush()
title = md.group(2).strip() if md else raw
level = len(md.group(1)) if md else 2
buf = []
continue
buf.append(line.rstrip())
flush()
return sections or [Section("", text, 0)]
PARSERS = {
".html": parse_html,
".htm": parse_html,
".docx": parse_docx,
".doc": parse_doc_old,
".txt": parse_txt,
".pdf": parse_pdf,
}
# ──────────────────────────────────────────────
# Сегментиране и почистване
# ──────────────────────────────────────────────
def merge_short_sections(sections: list[Section]) -> list[Section]:
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
result: list[Section] = []
for sec in sections:
words = len((sec.text or "").split())
titled = bool((sec.title or "").strip())
if result and not titled and words < MIN_SECTION_TOKENS:
prev = result[-1]
merged = Section(
prev.title,
(prev.text + "\n" + sec.text).strip(),
prev.level,
)
merged.images = (prev.images or []) + (sec.images or [])
html_parts = [h for h in (prev.html_text, sec.html_text) if h]
merged.html_text = "\n".join(html_parts) if html_parts else None
result[-1] = merged
else:
result.append(sec)
return result
def merge_preamble_sections(sections: list[Section]) -> list[Section]:
"""Слива корица (Title) + „Съдържание“/TOC в един преамбюл, ако парсерът ги е разделил."""
if len(sections) < 2:
return sections
first, second = sections[0], sections[1]
first_words = len((first.text or "").split())
if not (first.title or "").strip():
return sections
if first_words >= MIN_SECTION_TOKENS:
return sections
if not _is_toc_heading(second.title or ""):
return sections
body_parts = [p for p in (second.title, second.text) if (p or "").strip()]
if (first.text or "").strip():
body_parts.append(first.text.strip())
merged = Section(first.title, "\n".join(body_parts).strip(), first.level)
merged.images = (first.images or []) + (second.images or [])
html_parts = [h for h in (first.html_text, second.html_text) if h]
merged.html_text = "\n".join(html_parts) if html_parts else None
return [merged] + list(sections[2:])
def clean_text(text: str) -> str:
"""Collapse spaces/tabs but keep newlines (lists, paragraphs).
Маха опасни C0 контроли (NUL и др.); пази Unicode символи (•, →, NBSP→space).
"""
text = text.replace("\r\n", "\n").replace("\r", "\n")
text = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", "", text)
text = text.replace("\u00a0", " ")
text = re.sub(r"[^\S\n]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
text = "\n".join(line.strip() for line in text.split("\n"))
return text.strip()
# ──────────────────────────────────────────────
# AI класификация
# ──────────────────────────────────────────────
def _content_text(msg) -> str:
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
parts: list[str] = []
for block in getattr(msg, "content", None) or []:
btype = getattr(block, "type", None)
if btype in (None, "text"):
t = getattr(block, "text", None)
if t:
parts.append(str(t))
elif isinstance(block, dict) and block.get("text"):
parts.append(str(block["text"]))
return "\n".join(parts).strip()
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
raw = (raw or "").strip()
if not raw:
return None
raw = re.sub(r"^```[a-z]*\n?", "", raw)
raw = re.sub(r"\n?```$", "", raw)
candidates = [raw]
m = re.search(r"\{.*\}", raw, re.S)
if m:
candidates.append(m.group(0))
for cand in candidates:
try:
data = json.loads(cand)
except json.JSONDecodeError:
continue
if not isinstance(data, dict):
continue
t = data.get("title", "")
k = data.get("keywords", "")
if isinstance(k, list):
k = ", ".join(str(x).strip() for x in k if str(x).strip())
return str(t)[:200], str(k)[:300]
return None
_FALLBACK_KW_STOP = {
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
}
def fallback_classify(title: str, text: str) -> tuple[str, str]:
t = (title or "").strip()
if not t:
for line in (text or "").splitlines():
line = line.strip()
if 3 <= len(line) <= 80:
t = line
break
if not t:
t = "Секция"
blob = f"{title} {text}"[:1200]
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
seen: list[str] = []
seen_l: set[str] = set()
for w in words:
wl = w.lower()
if wl in _FALLBACK_KW_STOP or wl in seen_l:
continue
seen.append(w)
seen_l.add(wl)
if len(seen) >= 5:
break
return t[:200], ", ".join(seen)
KEYWORDS_MAX_LEN = 300 # rip_help_sections.keywords VARCHAR(300)
def parse_keyword_list(raw: str) -> list[str]:
"""Разделя ключови думи. Запетая/точка и запетая = фрази; иначе — интервал."""
text = (raw or "").strip()
if not text:
return []
if "," in text or ";" in text:
parts = re.split(r"[,;]+", text)
else:
parts = text.split()
out: list[str] = []
seen: set[str] = set()
for p in parts:
k = " ".join(p.split())
if not k:
continue
key = k.casefold()
if key in seen:
continue
seen.add(key)
out.append(k)
return out
def merge_section_keywords(
manual: str,
generated: str,
max_len: int = KEYWORDS_MAX_LEN,
) -> str:
"""Ръчните думи първи, после генерираните; без дубликати (без значение на регистъра)."""
first = parse_keyword_list(manual)
seen = {k.casefold() for k in first}
rest: list[str] = []
for k in parse_keyword_list(generated):
key = k.casefold()
if key in seen:
continue
seen.add(key)
rest.append(k)
merged = first + rest
if not merged:
return ""
out: list[str] = []
used = 0
for i, k in enumerate(merged):
extra = len(k) + (2 if i else 0) # ", "
if used + extra > max_len:
break
out.append(k)
used += extra
return ", ".join(out)
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
snippet = text[:MAX_AI_CHARS]
prompt = f"""Анализирай следната секция от help-документация и върни JSON обект с два ключа:
- "title": кратко наименование на секцията (до 8 думи, на езика на текста)
- "keywords": списък от до 5 ключови думи/фрази, разделени със запетая (на езика на текста)
Съществуващо заглавие (може да е празно): {title!r}
Текст:
{snippet}
Върни САМО валиден JSON без markdown, без коментари."""
last_err: Optional[Exception] = None
tried: set[str] = set()
for model in AI_MODELS:
if not model or model in tried:
continue
tried.add(model)
try:
msg = client.messages.create(
model=model,
max_tokens=512,
messages=[{"role": "user", "content": prompt}],
)
raw = _content_text(msg)
parsed = _parse_classify_json(raw)
if parsed:
t, k = parsed
return (t or title or "Секция")[:200], k
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
except Exception as e:
last_err = e
log.warning(f"AI classify ({model}) неуспешен: {e}")
continue
if last_err:
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
return fallback_classify(title, text)
# ──────────────────────────────────────────────
# Генериране на кодове — help_codes.py
# ──────────────────────────────────────────────
# ──────────────────────────────────────────────
# Основна обработка
# ──────────────────────────────────────────────
def _db_output_path(local_path: Path, code: str, remote_root: Optional[str]) -> str:
"""Локален staging файл; в БД — сървърен път ако е зададен remote_root."""
if remote_root:
root = remote_root.replace("\\", "/").rstrip("/")
return f"{root}/{code}.txt"
return str(local_path)
def process_file(
path: Path,
file_index: int,
db: Database,
client: anthropic.Anthropic,
output_dir: Path,
prefix: str = "HLP",
force: bool = False,
remote_root: Optional[str] = None,
source_key: Optional[str] = None,
seed_keywords: str = "",
) -> dict:
"""Обработва един файл. Връща статистика (saved, codes, ok)."""
rel = source_key or path.name
fh = file_hash(path)
existing_rows = db.sections_for_file(prefix, rel)
existing_codes = [r[0] for r in existing_rows]
if not force:
stored = db.get_file_hash(prefix, rel)
if stored == fh:
log.info(f" [SKIP] {path.name} (непроменен)")
return _file_result(rel, file_index, existing_codes, saved=0, skipped=True)
log.info(f" [PROC] {path.name}")
ext = path.suffix.lower()
parser = PARSERS.get(ext)
if not parser:
log.warning(f" Неподдържан формат: {ext}")
return _file_result(rel, file_index, existing_codes, saved=0)
try:
sections = parser(path)
except Exception as e:
log.error(f" Грешка при парсване: {e}")
return _file_result(rel, file_index, existing_codes, saved=0)
sections = merge_short_sections(sections)
sections = merge_preamble_sections(sections)
remove_section_outputs(
output_dir,
existing_codes,
[r[2] for r in existing_rows if r[2]],
)
db.delete_sections_for_file(prefix, rel)
images_dir = output_dir / "images"
images_dir.mkdir(parents=True, exist_ok=True)
saved = 0
codes: list[str] = []
ai_errors = 0
for sec in sections:
text = clean_text(sec.text)
html_text = sec.html_text or ""
if not text and not sec.images and not html_text:
if (sec.title or "").strip():
text = sec.title.strip()
else:
continue
sec_index = saved + 1
code = make_code(prefix, file_index, sec_index)
# Записваме картинките на диск и заменяме placeholder-ите в текста + HTML
image_rel_paths: list[str] = []
for ref in sec.images or []:
fname = f"{code}_{ref.placeholder}.{ref.ext}"
disk_path = images_dir / fname
try:
disk_path.write_bytes(ref.data)
except Exception as e:
log.warning(f" Грешка при запис на картинка {fname}: {e}")
continue
rel_path = f"images/{fname}"
image_rel_paths.append(rel_path)
old_ph = f"[IMG: {ref.placeholder}]"
new_ph = f"[IMG: {rel_path}]"
text = text.replace(old_ph, new_ph)
html_text = html_text.replace(old_ph, new_ph)
# Премахваме placeholder-и, останали без файл
text = _IMG_PLACEHOLDER_RE.sub(
lambda m: m.group(0) if "/" in m.group(1) or "\\" in m.group(1) else "",
text
).strip()
html_text = _IMG_PLACEHOLDER_RE.sub(
lambda m: m.group(0) if "/" in m.group(1) or "\\" in m.group(1) else "",
html_text
).strip()
if not text and not image_rel_paths and not html_text:
continue
try:
title, keywords = classify_section(client, sec.title, text)
except Exception as e:
log.warning(f" AI грешка за {code}: {e}")
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
ai_errors += 1
if not (keywords or "").strip():
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
ai_errors += 1
keywords = merge_section_keywords(seed_keywords, keywords)
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
ps = ProcessedSection(
code=code,
source_file=rel,
title=title,
keywords=keywords,
text=text,
images_json=images_json,
html_text=html_text,
)
# Локален staging; в БД — remote път (ако е зададен) за Coolify/API
out_path = output_dir / f"{code}.txt"
out_path.write_text(
f"КОД: {code}\nФАЙЛ: {rel}\nЗАГЛАВИЕ: {title}\nКЛЮЧОВИ ДУМИ: {keywords}\n"
f"КАРТИНКИ: {len(image_rel_paths)}\n"
f"{'─'*60}\n{text}",
encoding="utf-8"
)
db.insert_section(prefix, ps, _db_output_path(out_path, code, remote_root))
saved += 1
codes.append(code)
log.debug(f" {code}: {title[:60]} ({len(image_rel_paths)} img)")
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
log.info(f" → {saved} секции записани")
result = _file_result(rel, file_index, codes, saved=saved)
result["ai_errors"] = ai_errors
result["keywords_ok"] = ai_errors == 0
return result
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
def process_directory(
input_dir: Path,
output_dir: Path,
conn_str: str,
api_key: str,
prefix: str = "HLP",
force: bool = False,
purge_missing: bool = False,
remote_root: Optional[str] = None,
seed_keywords: str = "",
):
if not _PREFIX_RE.match(prefix):
raise ValueError(
f"Невалиден prefix {prefix!r}. Допустими: буква + букви/цифри/подчертавки, до 50 символа."
)
output_dir.mkdir(parents=True, exist_ok=True)
db = Database(conn_str)
client = anthropic.Anthropic(api_key=api_key)
extensions = set(PARSERS.keys())
output_resolved = output_dir.resolve()
def _under_output(p: Path) -> bool:
try:
p.resolve().relative_to(output_resolved)
return True
except ValueError:
return False
files = [
p for p in input_dir.rglob("*")
if p.is_file() and p.suffix.lower() in extensions and not _under_output(p)
]
log.info(f"Prefix={prefix} Намерени {len(files)} файла в {input_dir}")
if remote_root:
log.info(f"DB output_path root: {remote_root.replace(chr(92), '/').rstrip('/')}")
current_ids = {source_identity(p, input_dir) for p in files}
current_names = {source_basename(i).lower() for i in current_ids}
file_results: list[dict] = []
total_sections = 0
try:
for path in sorted(files):
identity = source_identity(path, input_dir)
idx = db.allocate_file_index(prefix, identity)
info = process_file(
path, idx, db, client, output_dir,
prefix=prefix, force=force, remote_root=remote_root,
source_key=identity,
seed_keywords=seed_keywords,
)
file_results.append(info)
total_sections += int(info.get("saved") or 0)
if purge_missing:
existing = set(db.all_source_files(prefix))
orphans = sorted(
e for e in existing
if source_basename(e).lower() not in current_names
)
if not orphans:
log.info(f"Purge: няма orphan записи в БД за prefix={prefix}.")
else:
log.info(f"Purge ({prefix}): намерени {len(orphans)} orphan източника:")
for o in orphans:
log.info(f" - {o}")
disk_paths = db.section_output_paths_for(prefix, orphans)
removed_files = 0
for op in disk_paths:
try:
code = Path(str(op).replace("\\", "/")).stem
local_txt = output_dir / f"{code}.txt"
if local_txt.exists():
local_txt.unlink()
removed_files += 1
opath = Path(op)
if opath.exists():
try:
if opath.resolve() != local_txt.resolve():
opath.unlink()
removed_files += 1
except Exception:
pass
for img in (output_dir / "images").glob(f"{code}_*"):
try:
img.unlink()
removed_files += 1
except Exception:
pass
except Exception as e:
log.debug(f" не успях да изтрия {op}: {e}")
deleted = db.purge_sources(prefix, orphans)
log.info(f"Purge: изтрити {deleted} секции от БД, {removed_files} файла от диска.")
finally:
db.close()
log.info(f"Готово. Prefix={prefix}. Общо нови/обновени секции: {total_sections}")
return {"sections": total_sections, "files": file_results}
# ──────────────────────────────────────────────
# CLI
# ──────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(
description="Help-файл декомпозитор с PostgreSQL + Anthropic"
)
parser.add_argument("input_dir", help="Входна директория с help-файлове")
parser.add_argument("output_dir", help="Изходна директория за текстови секции")
parser.add_argument(
"--conn",
default=os.getenv("HELP_DB_CONN"),
help="Postgres libpq connection string (или HELP_DB_CONN env var)"
)
parser.add_argument(
"--api-key",
default=os.getenv("ANTHROPIC_API_KEY"),
help="Anthropic API ключ (или ANTHROPIC_API_KEY env var)"
)
parser.add_argument(
"--prefix",
default=os.getenv("HELP_PREFIX", "HLP"),
help="Префикс за кодовете/scope в БД (буква + букви/цифри/_, до 50 знака). "
"Default: 'HLP' (или env HELP_PREFIX)."
)
parser.add_argument(
"--force",
action="store_true",
help="Преобработва всички файлове, независимо от hash"
)
parser.add_argument(
"--purge-missing",
action="store_true",
help="След обработката изтрива от БД и диска секциите за източници, "
"които вече не съществуват във входната директория (само в дадения prefix)"
)
parser.add_argument(
"--remote-root",
default=os.getenv("HELP_REMOTE_ROOT"),
help="Сървърен корен за output_path в БД "
"(напр. /mnt/mssql/share/RIP/RIP_Help_Source/Output). "
"Файловете се пишат локално; в БД се записва remote път. "
"Или env HELP_REMOTE_ROOT."
)
parser.add_argument(
"--keywords",
default="",
help="Ръчни ключови думи (запетая). Стоят в началото на всяка секция от това сканиране."
)
args = parser.parse_args()
if not args.api_key:
sys.exit("Грешка: липсва Anthropic API ключ (--api-key или ANTHROPIC_API_KEY).")
if not args.conn:
sys.exit("Грешка: липсва Postgres connection string (--conn или HELP_DB_CONN).")
process_directory(
input_dir=Path(args.input_dir),
output_dir=Path(args.output_dir),
conn_str=args.conn,
api_key=args.api_key,
prefix=args.prefix,
force=args.force,
purge_missing=args.purge_missing,
remote_root=args.remote_root,
seed_keywords=args.keywords,
)
if __name__ == "__main__":
main()