2035 lines
74 KiB
Python
2035 lines
74 KiB
Python
"""
|
||
help_processor.py
|
||
=================
|
||
Обработва help-файлове (.doc, .docx, .html, .htm, .txt, .pdf),
|
||
декомпозира ги на смислови секции, извлича ключови думи чрез Anthropic API
|
||
и записва резултатите в SQL Server + изходна директория.
|
||
|
||
Поддържа инкрементална обработка: файлове, чийто hash не се е променил,
|
||
се прескачат при повторно пускане.
|
||
|
||
Изисквания (pip install):
|
||
pip install anthropic pyodbc python-docx beautifulsoup4 lxml
|
||
pip install pdfplumber striprtf chardet
|
||
pip install pywin32 # за MS Word fallback на Windows
|
||
|
||
За .doc (стар формат) е необходим един от:
|
||
- LibreOffice (soffice в PATH) — кросплатформено
|
||
- MS Word — Windows, чрез pywin32 COM (автоматичен fallback)
|
||
- antiword — Linux (apt install antiword)
|
||
"""
|
||
|
||
import os
|
||
import re
|
||
import sys
|
||
import json
|
||
import hashlib
|
||
import logging
|
||
import argparse
|
||
import subprocess
|
||
import tempfile
|
||
from pathlib import Path
|
||
from datetime import datetime
|
||
from dataclasses import dataclass, field
|
||
from typing import Optional
|
||
|
||
import psycopg2
|
||
import anthropic
|
||
from docx import Document
|
||
from bs4 import BeautifulSoup
|
||
|
||
from help_codes import (
|
||
WIPED_HASH,
|
||
file_result as _file_result,
|
||
make_code,
|
||
parse_code,
|
||
remove_section_outputs,
|
||
source_basename,
|
||
source_identity,
|
||
)
|
||
|
||
try:
|
||
import pdfplumber
|
||
HAS_PDF = True
|
||
except ImportError:
|
||
HAS_PDF = False
|
||
|
||
try:
|
||
from PIL import Image
|
||
HAS_PIL = True
|
||
except ImportError:
|
||
HAS_PIL = False
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Конфигурация
|
||
# ──────────────────────────────────────────────
|
||
|
||
# На Windows конзолата често е cp1251 → пренастройваме stdout на utf-8
|
||
try:
|
||
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
||
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
|
||
except AttributeError:
|
||
pass
|
||
|
||
_log_handlers = [logging.StreamHandler(sys.stdout)]
|
||
try:
|
||
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
|
||
except OSError:
|
||
pass
|
||
logging.basicConfig(
|
||
level=logging.INFO,
|
||
format="%(asctime)s %(levelname)-8s %(message)s",
|
||
handlers=_log_handlers,
|
||
)
|
||
log = logging.getLogger(__name__)
|
||
|
||
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
|
||
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
|
||
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
|
||
AI_MODELS = [
|
||
AI_MODEL,
|
||
"claude-haiku-4-5",
|
||
"claude-haiku-4-5-20251001",
|
||
"claude-3-5-haiku-latest",
|
||
]
|
||
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Изображения — помощни
|
||
# ──────────────────────────────────────────────
|
||
|
||
@dataclass
|
||
class ImageRef:
|
||
placeholder: str # вътрешен ID в текста, напр. "img_01"
|
||
data: bytes
|
||
ext: str # "png", "jpg", "gif"...
|
||
|
||
|
||
def _img_dimensions(data: bytes) -> Optional[tuple[int, int]]:
|
||
if not HAS_PIL:
|
||
return None
|
||
try:
|
||
from io import BytesIO
|
||
with Image.open(BytesIO(data)) as im:
|
||
return im.size
|
||
except Exception:
|
||
return None
|
||
|
||
|
||
def _should_keep_image(data: bytes) -> bool:
|
||
"""Връща False за дребни иконки/булети под MIN_IMAGE_PX × MIN_IMAGE_PX."""
|
||
if not data:
|
||
return False
|
||
dims = _img_dimensions(data)
|
||
if dims is None:
|
||
# Не можем да преценим — пазим по подразбиране
|
||
return True
|
||
w, h = dims
|
||
return w >= MIN_IMAGE_PX and h >= MIN_IMAGE_PX
|
||
|
||
|
||
def _ext_from_content_type(ct: str) -> str:
|
||
ct = (ct or "").lower()
|
||
if "png" in ct: return "png"
|
||
if "jpeg" in ct or "jpg" in ct: return "jpg"
|
||
if "gif" in ct: return "gif"
|
||
if "bmp" in ct: return "bmp"
|
||
if "svg" in ct: return "svg"
|
||
if "webp" in ct: return "webp"
|
||
return "png"
|
||
|
||
|
||
_IMG_PLACEHOLDER_RE = re.compile(r"\[IMG:\s*([A-Za-z0-9_./\\-]+)\s*\]")
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Структури
|
||
# ──────────────────────────────────────────────
|
||
|
||
@dataclass
|
||
class Section:
|
||
title: str
|
||
text: str
|
||
level: int = 1 # 1=H1, 2=H2, 3=H3, 0=без заглавие
|
||
images: list = field(default_factory=list) # list[ImageRef]
|
||
html_text: Optional[str] = None # rich HTML с [IMG: ...] placeholders
|
||
|
||
|
||
@dataclass
|
||
class ProcessedSection:
|
||
code: str # DOC_003_SEC_012
|
||
source_file: str
|
||
title: str
|
||
keywords: str # "кл1, кл2, кл3"
|
||
text: str
|
||
images_json: str = "[]" # JSON масив с относителни пътища
|
||
html_text: str = "" # rich HTML (само за HTML-source файлове)
|
||
char_count: int = 0
|
||
|
||
def __post_init__(self):
|
||
self.char_count = len(self.text)
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# База данни
|
||
# ──────────────────────────────────────────────
|
||
|
||
class Database:
|
||
"""PostgreSQL backend (psycopg2). Connection string е libpq формат:
|
||
'host=... port=... dbname=... user=... password=...'
|
||
"""
|
||
def __init__(self, conn_str: str):
|
||
self.conn_str = conn_str
|
||
self.conn = psycopg2.connect(conn_str)
|
||
self._ensure_schema()
|
||
|
||
def _ensure_schema(self):
|
||
"""Създава таблиците ако не съществуват (Postgres syntax)."""
|
||
cur = self.conn.cursor()
|
||
cur.execute("""
|
||
CREATE TABLE IF NOT EXISTS rip_help_files (
|
||
id SERIAL PRIMARY KEY,
|
||
prefix VARCHAR(50) NOT NULL DEFAULT 'HLP',
|
||
file_path VARCHAR(1000) NOT NULL,
|
||
file_hash CHAR(64) NOT NULL,
|
||
processed_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||
section_count INTEGER NOT NULL DEFAULT 0,
|
||
UNIQUE (prefix, file_path)
|
||
)
|
||
""")
|
||
cur.execute("""
|
||
CREATE TABLE IF NOT EXISTS rip_help_sections (
|
||
id SERIAL PRIMARY KEY,
|
||
prefix VARCHAR(50) NOT NULL DEFAULT 'HLP',
|
||
code VARCHAR(80) NOT NULL UNIQUE,
|
||
source_file VARCHAR(1000) NOT NULL,
|
||
title VARCHAR(500),
|
||
keywords VARCHAR(300),
|
||
char_count INTEGER,
|
||
output_path VARCHAR(1000),
|
||
images TEXT,
|
||
html_text TEXT,
|
||
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||
)
|
||
""")
|
||
cur.execute("""
|
||
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_keywords
|
||
ON rip_help_sections(keywords)
|
||
""")
|
||
cur.execute("""
|
||
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_prefix
|
||
ON rip_help_sections(prefix)
|
||
""")
|
||
cur.execute("""
|
||
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_source
|
||
ON rip_help_sections(prefix, source_file)
|
||
""")
|
||
cur.execute(
|
||
"ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER"
|
||
)
|
||
self.conn.commit()
|
||
log.info("Схемата е проверена / създадена.")
|
||
|
||
def matching_source_paths(self, prefix: str, identity: str) -> list[str]:
|
||
"""Всички file_path/source_file за същия help файл (basename, вкл. стари temp пътища)."""
|
||
name = source_basename(identity).lower()
|
||
if not name:
|
||
return []
|
||
found: list[str] = []
|
||
seen: set[str] = set()
|
||
for p in self.all_source_files(prefix):
|
||
if source_basename(p).lower() == name and p not in seen:
|
||
seen.add(p)
|
||
found.append(p)
|
||
return found
|
||
|
||
def get_file_hash(self, prefix: str, file_path: str) -> Optional[str]:
|
||
paths = self.matching_source_paths(prefix, file_path) or [file_path]
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"SELECT file_hash FROM rip_help_files "
|
||
"WHERE prefix=%s AND file_path = ANY(%s)",
|
||
(prefix, list(paths)),
|
||
)
|
||
for (h,) in cur.fetchall():
|
||
if not h:
|
||
continue
|
||
val = str(h).strip()
|
||
if val and val != WIPED_HASH:
|
||
return val
|
||
return None
|
||
|
||
def upsert_file(
|
||
self,
|
||
prefix: str,
|
||
file_path: str,
|
||
file_hash: str,
|
||
section_count: int,
|
||
file_index: Optional[int] = None,
|
||
):
|
||
canonical = source_basename(file_path) or file_path
|
||
stale = [p for p in self.matching_source_paths(prefix, canonical) if p != canonical]
|
||
cur = self.conn.cursor()
|
||
if stale:
|
||
cur.execute(
|
||
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
|
||
(prefix, stale),
|
||
)
|
||
cur.execute("""
|
||
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
|
||
VALUES (%s, %s, %s, %s, %s)
|
||
ON CONFLICT (prefix, file_path) DO UPDATE SET
|
||
file_hash = EXCLUDED.file_hash,
|
||
section_count= EXCLUDED.section_count,
|
||
processed_at = NOW(),
|
||
file_index = COALESCE(EXCLUDED.file_index, rip_help_files.file_index)
|
||
""", (prefix, canonical, file_hash, section_count, file_index))
|
||
self.conn.commit()
|
||
|
||
def delete_sections_for_file(self, prefix: str, file_path: str):
|
||
paths = self.matching_source_paths(prefix, file_path) or [file_path]
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
|
||
(prefix, list(paths)),
|
||
)
|
||
self.conn.commit()
|
||
|
||
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
|
||
"""(code, source_file, output_path) за всички секции на файла."""
|
||
paths = self.matching_source_paths(prefix, identity)
|
||
if not paths:
|
||
return []
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"SELECT code, source_file, output_path FROM rip_help_sections "
|
||
"WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code",
|
||
(prefix, list(paths)),
|
||
)
|
||
return [(r[0], r[1], r[2]) for r in cur.fetchall()]
|
||
|
||
def file_index_for(self, prefix: str, identity: str) -> Optional[int]:
|
||
paths = self.matching_source_paths(prefix, identity)
|
||
if not paths:
|
||
return None
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"SELECT file_index FROM rip_help_files "
|
||
"WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL",
|
||
(prefix, list(paths)),
|
||
)
|
||
from_col = [r[0] for r in cur.fetchall() if r[0]]
|
||
if from_col:
|
||
return min(from_col)
|
||
cur.execute(
|
||
"SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
|
||
(prefix, list(paths)),
|
||
)
|
||
found: list[int] = []
|
||
for (code,) in cur.fetchall():
|
||
parsed = parse_code(code)
|
||
if parsed:
|
||
found.append(parsed[1])
|
||
if not found:
|
||
return None
|
||
tally: dict[int, int] = {}
|
||
for idx in found:
|
||
tally[idx] = tally.get(idx, 0) + 1
|
||
return max(tally, key=lambda k: (tally[k], -k))
|
||
|
||
def max_file_index(self, prefix: str) -> int:
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"SELECT COALESCE(MAX(file_index), 0) FROM rip_help_files WHERE prefix=%s",
|
||
(prefix,),
|
||
)
|
||
m1 = int(cur.fetchone()[0] or 0)
|
||
cur.execute("SELECT code FROM rip_help_sections WHERE prefix=%s", (prefix,))
|
||
m2 = 0
|
||
for (code,) in cur.fetchall():
|
||
parsed = parse_code(code)
|
||
if parsed:
|
||
m2 = max(m2, parsed[1])
|
||
return max(m1, m2)
|
||
|
||
def file_index_used_by_others(self, prefix: str, identity: str, idx: int) -> bool:
|
||
name = source_basename(identity).lower()
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"SELECT code, source_file FROM rip_help_sections WHERE prefix=%s",
|
||
(prefix,),
|
||
)
|
||
for code, src in cur.fetchall():
|
||
parsed = parse_code(code)
|
||
if parsed and parsed[1] == idx and source_basename(src).lower() != name:
|
||
return True
|
||
return False
|
||
|
||
def allocate_file_index(self, prefix: str, identity: str) -> int:
|
||
existing = self.file_index_for(prefix, identity)
|
||
if existing and not self.file_index_used_by_others(prefix, identity, existing):
|
||
return existing
|
||
return self.max_file_index(prefix) + 1
|
||
|
||
def wipe_extractions_for_file(
|
||
self,
|
||
prefix: str,
|
||
identity: str,
|
||
output_dir: Optional[Path] = None,
|
||
) -> dict:
|
||
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||
rows = self.sections_for_file(prefix, identity)
|
||
codes = [r[0] for r in rows]
|
||
file_index = self.file_index_for(prefix, identity)
|
||
if file_index and self.file_index_used_by_others(prefix, identity, file_index):
|
||
file_index = None
|
||
paths = self.matching_source_paths(prefix, identity)
|
||
if output_dir:
|
||
remove_section_outputs(output_dir, codes, [r[2] for r in rows if r[2]])
|
||
self.delete_sections_for_file(prefix, identity)
|
||
canonical = source_basename(identity) or identity
|
||
stale = list(paths) if paths else []
|
||
cur = self.conn.cursor()
|
||
if stale:
|
||
cur.execute(
|
||
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
|
||
(prefix, stale),
|
||
)
|
||
if file_index:
|
||
cur.execute("""
|
||
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
|
||
VALUES (%s, %s, %s, 0, %s)
|
||
ON CONFLICT (prefix, file_path) DO UPDATE SET
|
||
file_hash = EXCLUDED.file_hash,
|
||
section_count = 0,
|
||
processed_at = NOW(),
|
||
file_index = EXCLUDED.file_index
|
||
""", (prefix, canonical, WIPED_HASH, file_index))
|
||
self.conn.commit()
|
||
return {
|
||
"file": canonical,
|
||
"prefix": prefix,
|
||
"deleted": len(codes),
|
||
"codes": codes,
|
||
"file_index": file_index,
|
||
}
|
||
|
||
def all_source_files(self, prefix: str) -> list[str]:
|
||
"""Връща всички source_file пътища за даден префикс."""
|
||
cur = self.conn.cursor()
|
||
cur.execute("""
|
||
SELECT file_path FROM rip_help_files WHERE prefix=%s
|
||
UNION
|
||
SELECT source_file FROM rip_help_sections WHERE prefix=%s
|
||
""", (prefix, prefix))
|
||
return [r[0] for r in cur.fetchall()]
|
||
|
||
def section_output_paths_for(self, prefix: str, source_files: list[str]) -> list[str]:
|
||
if not source_files:
|
||
return []
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"SELECT output_path FROM rip_help_sections "
|
||
"WHERE prefix=%s AND source_file = ANY(%s)",
|
||
(prefix, list(source_files))
|
||
)
|
||
return [r[0] for r in cur.fetchall() if r[0]]
|
||
|
||
def purge_sources(self, prefix: str, source_files: list[str]) -> int:
|
||
if not source_files:
|
||
return 0
|
||
cur = self.conn.cursor()
|
||
cur.execute(
|
||
"DELETE FROM rip_help_sections "
|
||
"WHERE prefix=%s AND source_file = ANY(%s)",
|
||
(prefix, list(source_files))
|
||
)
|
||
sec_deleted = cur.rowcount
|
||
cur.execute(
|
||
"DELETE FROM rip_help_files "
|
||
"WHERE prefix=%s AND file_path = ANY(%s)",
|
||
(prefix, list(source_files))
|
||
)
|
||
self.conn.commit()
|
||
return sec_deleted
|
||
|
||
def insert_section(self, prefix: str, ps: ProcessedSection, output_path: str):
|
||
cur = self.conn.cursor()
|
||
cur.execute("""
|
||
INSERT INTO rip_help_sections
|
||
(prefix, code, source_file, title, keywords,
|
||
char_count, output_path, images, html_text)
|
||
VALUES (%s, %s, %s, %s, %s, %s, %s, %s, %s)
|
||
ON CONFLICT (code) DO UPDATE SET
|
||
prefix = EXCLUDED.prefix,
|
||
source_file = EXCLUDED.source_file,
|
||
title = EXCLUDED.title,
|
||
keywords = EXCLUDED.keywords,
|
||
char_count = EXCLUDED.char_count,
|
||
output_path = EXCLUDED.output_path,
|
||
images = EXCLUDED.images,
|
||
html_text = EXCLUDED.html_text,
|
||
updated_at = NOW()
|
||
""", (prefix, ps.code, ps.source_file, ps.title, ps.keywords,
|
||
ps.char_count, output_path, ps.images_json, ps.html_text))
|
||
self.conn.commit()
|
||
|
||
def close(self):
|
||
self.conn.close()
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Парсъри
|
||
# ──────────────────────────────────────────────
|
||
|
||
def file_hash(path: Path) -> str:
|
||
h = hashlib.sha256()
|
||
with open(path, "rb") as f:
|
||
for chunk in iter(lambda: f.read(65536), b""):
|
||
h.update(chunk)
|
||
return h.hexdigest()
|
||
|
||
|
||
def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
|
||
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS."""
|
||
if not src:
|
||
return None
|
||
s = src.strip()
|
||
if s.startswith("data:"):
|
||
# data:image/png;base64,XXXX
|
||
m = re.match(r"data:([^;]+);base64,(.+)$", s, re.DOTALL)
|
||
if not m:
|
||
return None
|
||
import base64
|
||
try:
|
||
data = base64.b64decode(m.group(2))
|
||
except Exception:
|
||
return None
|
||
return data, _ext_from_content_type(m.group(1))
|
||
if s.startswith(("http://", "https://")):
|
||
return None # по правило пропускаме мрежови картинки
|
||
# локален път, относителен или абсолютен
|
||
p = (base_dir / s).resolve() if not Path(s).is_absolute() else Path(s)
|
||
try:
|
||
if p.is_file():
|
||
data = p.read_bytes()
|
||
ext = p.suffix.lstrip(".").lower() or "png"
|
||
return data, ext
|
||
except Exception:
|
||
return None
|
||
return None
|
||
|
||
|
||
def _detect_html_encoding(raw: bytes) -> str:
|
||
"""Връща име на encoding: BOM → chardet → fallback (utf-8 ако ASCII, иначе windows-1251)."""
|
||
# BOM-и
|
||
if raw.startswith(b"\xef\xbb\xbf"):
|
||
return "utf-8"
|
||
if raw.startswith((b"\xff\xfe", b"\xfe\xff")):
|
||
return "utf-16"
|
||
# chardet
|
||
try:
|
||
import chardet
|
||
det = chardet.detect(raw[:65536]) or {}
|
||
enc = (det.get("encoding") or "").lower()
|
||
conf = det.get("confidence", 0) or 0
|
||
if enc and conf >= 0.6:
|
||
# нормализиране на често срещани имена
|
||
if enc in ("cp1251", "ms-cyrl", "windows-1251"):
|
||
return "windows-1251"
|
||
if enc.startswith("utf"):
|
||
return enc
|
||
return enc
|
||
except Exception:
|
||
pass
|
||
# fallback: ако байтовете изглеждат "над 127" (т.е. има не-ASCII), приемаме CP1251
|
||
if any(b > 127 for b in raw[:8192]):
|
||
return "windows-1251"
|
||
return "utf-8"
|
||
|
||
|
||
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
||
"p", "ul", "ol", "table", "dl", "pre",
|
||
"blockquote", "figure", "hr"]
|
||
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||
"valign", "width", "height", "bgcolor", "border")
|
||
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||
_HEADING_TOKEN_RE = re.compile(
|
||
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
|
||
r"(\d+)?$",
|
||
re.I,
|
||
)
|
||
# Word „Title“ / MsoTitle ≠ Heading 1 — корица, не граница на секция.
|
||
_COVER_TITLE_TOKENS = frozenset({
|
||
"title", "mstotitle", "наименование", "doctitle", "documenttitle",
|
||
})
|
||
_HTML_HEADING_CLASS_RE = re.compile(
|
||
r"(?:^|[\s_-])(?:heading|заглавие|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
|
||
re.I,
|
||
)
|
||
_HTML_COVER_CLASS_RE = re.compile(
|
||
r"(?:^|[\s_-])(?:mso)?title(?:\d+)?(?:$|[\s_-])|"
|
||
r"(?:^|[\s_-])наименование(?:\d+)?(?:$|[\s_-])",
|
||
re.I,
|
||
)
|
||
_HTML_SUBTITLE_CLASS_RE = re.compile(
|
||
r"(?:^|[\s_-])(?:subtitle|подзаглавие)(?:\d+)?(?:$|[\s_-])",
|
||
re.I,
|
||
)
|
||
_TOC_HEADING_RE = re.compile(
|
||
r"^(съдържание|съдържанието|contents|table of contents|toc|"
|
||
r"inhaltsverzeichnis|оглавление|содержание)\s*:?\s*$",
|
||
re.I,
|
||
)
|
||
_HEADING_LEVEL = {
|
||
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
|
||
"subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
|
||
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
|
||
"подзаглавие": 2,
|
||
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
|
||
}
|
||
|
||
|
||
def _compact_style_token(s: str) -> str:
|
||
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
|
||
|
||
|
||
def _is_cover_title_token(token: str) -> bool:
|
||
return _compact_style_token(token) in _COVER_TITLE_TOKENS
|
||
|
||
|
||
def _is_toc_heading(text: str) -> bool:
|
||
return bool(_TOC_HEADING_RE.match((text or "").strip()))
|
||
|
||
|
||
# Top-level chapter: "1. Title" / "2) Title" — short line, not body/table/figure.
|
||
_NUMBERED_CHAPTER_RE = re.compile(
|
||
r"^(\d{1,2})[\.\)]\s+(\S.{0,100})$"
|
||
)
|
||
_FIGURE_CAPTION_RE = re.compile(
|
||
r"^(фигура|figure|abb\.?|рис\.?)\s*\d+",
|
||
re.I,
|
||
)
|
||
|
||
|
||
def _normalize_chapter_key(text: str) -> str:
|
||
return re.sub(r"\s+", " ", (text or "").strip().lower())
|
||
|
||
|
||
def _is_numbered_chapter_heading(text: str) -> bool:
|
||
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
|
||
t = (text or "").strip()
|
||
if not t or len(t) >= 120:
|
||
return False
|
||
if "|" in t:
|
||
return False
|
||
if _is_toc_heading(t):
|
||
return False
|
||
if _FIGURE_CAPTION_RE.match(t):
|
||
return False
|
||
m = _NUMBERED_CHAPTER_RE.match(t)
|
||
if not m:
|
||
return False
|
||
rest = m.group(2).strip()
|
||
words = rest.split()
|
||
if not words or len(words) > 12:
|
||
return False
|
||
# Body-like numbered sentences: "1. Copy the files into the folder."
|
||
if rest.endswith((".", "!", "?")) and len(words) > 4:
|
||
return False
|
||
return True
|
||
|
||
|
||
class _TocSplitState:
|
||
"""Tracks Съдържание / TOC so first '1. Title' stays in preamble, second starts a section."""
|
||
|
||
__slots__ = ("phase", "had_toc", "titles")
|
||
|
||
def __init__(self) -> None:
|
||
self.phase = False
|
||
self.had_toc = False
|
||
self.titles: set[str] = set()
|
||
|
||
def note_toc_heading(self) -> None:
|
||
self.phase = True
|
||
self.had_toc = True
|
||
|
||
def leave_phase(self) -> None:
|
||
self.phase = False
|
||
|
||
def handle_numbered(self, text: str) -> str:
|
||
"""Return 'toc' (keep in body), 'split' (new section), or 'no' (not numbered chapter)."""
|
||
if not _is_numbered_chapter_heading(text):
|
||
return "no"
|
||
key = _normalize_chapter_key(text)
|
||
if self.phase:
|
||
if key in self.titles:
|
||
self.phase = False
|
||
return "split"
|
||
self.titles.add(key)
|
||
return "toc"
|
||
# After TOC list/block, or docs without TOC: numbered chapter opens a section.
|
||
return "split"
|
||
|
||
|
||
def _heading_level_from_token(token: str) -> Optional[int]:
|
||
t = _compact_style_token(token)
|
||
if not t:
|
||
return None
|
||
if _is_cover_title_token(t):
|
||
return None
|
||
if t in _HEADING_LEVEL:
|
||
return _HEADING_LEVEL[t]
|
||
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
|
||
if not m:
|
||
return None
|
||
n = m.group(2)
|
||
if n and n.isdigit():
|
||
return min(int(n), 3)
|
||
kind = (m.group(1) or "").lower()
|
||
if kind in ("subtitle", "подзаглавие"):
|
||
return 2
|
||
if kind in ("title", "наименование"):
|
||
return None
|
||
return 1
|
||
|
||
|
||
def _docx_style_tokens(para) -> list[str]:
|
||
style = getattr(para, "style", None)
|
||
tokens: list[str] = []
|
||
seen: set[int] = set()
|
||
cur = style
|
||
while cur is not None and id(cur) not in seen:
|
||
seen.add(id(cur))
|
||
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
|
||
if token:
|
||
tokens.append(str(token))
|
||
cur = getattr(cur, "base_style", None)
|
||
return tokens
|
||
|
||
|
||
def _docx_is_cover_title(para) -> bool:
|
||
return any(_is_cover_title_token(t) for t in _docx_style_tokens(para))
|
||
|
||
|
||
def _docx_heading_level(para) -> Optional[int]:
|
||
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word.
|
||
Word Title / Наименование не са граница (виж _docx_is_cover_title)."""
|
||
if _docx_is_cover_title(para):
|
||
return None
|
||
for token in _docx_style_tokens(para):
|
||
lvl = _heading_level_from_token(token)
|
||
if lvl:
|
||
return lvl
|
||
try:
|
||
pPr = para._element.pPr
|
||
if pPr is not None and pPr.outlineLvl is not None:
|
||
val = int(pPr.outlineLvl.val)
|
||
text = (para.text or "").strip()
|
||
if 0 <= val <= 2 and text and len(text) < 120:
|
||
return min(val + 1, 3)
|
||
except Exception:
|
||
pass
|
||
return None
|
||
|
||
|
||
def _is_bold_heading_text(text: str, runs) -> bool:
|
||
if not text or len(text) > 120:
|
||
return False
|
||
useful = [r for r in (runs or []) if (r.text or "").strip()]
|
||
return bool(useful) and all(bool(r.bold) for r in useful)
|
||
|
||
|
||
def _html_is_cover_title(el) -> bool:
|
||
raw = el.get("class") if hasattr(el, "get") else None
|
||
classes: list[str]
|
||
if isinstance(raw, str):
|
||
classes = raw.split()
|
||
else:
|
||
classes = list(raw or [])
|
||
for c in classes:
|
||
if _is_cover_title_token(c):
|
||
return True
|
||
return bool(_HTML_COVER_CLASS_RE.search(" ".join(classes)))
|
||
|
||
|
||
def _html_heading_level(el) -> Optional[int]:
|
||
name = (getattr(el, "name", None) or "").lower()
|
||
if name in _HTML_HEADING_MAP:
|
||
return _HTML_HEADING_MAP[name]
|
||
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
|
||
if _HTML_COVER_CLASS_RE.search(cls):
|
||
return None
|
||
if _HTML_SUBTITLE_CLASS_RE.search(cls):
|
||
return 2
|
||
m = _HTML_HEADING_CLASS_RE.search(cls)
|
||
if m:
|
||
n = m.group(1)
|
||
return min(int(n), 3) if n and n.isdigit() else 1
|
||
if name in ("p", "div"):
|
||
txt = el.get_text(" ", strip=True)
|
||
if txt and len(txt) < 120:
|
||
inner = "".join(el.stripped_strings)
|
||
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
|
||
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
|
||
if strong and strong_txt and strong_txt == inner:
|
||
return 2
|
||
return None
|
||
|
||
|
||
def _html_block_plain_text(el) -> str:
|
||
"""Plain ingest text: newlines inside lists/tables, spaces for inline runs."""
|
||
sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " "
|
||
return el.get_text(sep, strip=True)
|
||
|
||
|
||
def _strip_attrs(el):
|
||
"""Премахва decorative атрибути (class, style, on*, data-*)."""
|
||
for t in el.find_all(True):
|
||
for a in list(t.attrs):
|
||
if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"):
|
||
del t[a]
|
||
|
||
|
||
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
|
||
"""Намира всички <img> в подадения елемент, извлича данните и подменя с
|
||
NavigableString placeholder ([IMG: img_NN])."""
|
||
from bs4 import NavigableString
|
||
for img in el.find_all("img"):
|
||
src = img.get("src") or img.get("data-src") or ""
|
||
loaded = _load_html_image(src, base_dir)
|
||
if not loaded:
|
||
img.decompose()
|
||
continue
|
||
data, ext = loaded
|
||
if not _should_keep_image(data):
|
||
img.decompose()
|
||
continue
|
||
img_counter[0] += 1
|
||
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
|
||
sec_images.append(ref)
|
||
img.replace_with(NavigableString(f"[IMG: {ref.placeholder}]"))
|
||
|
||
|
||
def parse_html(path: Path) -> list[Section]:
|
||
raw = path.read_bytes()
|
||
enc = _detect_html_encoding(raw)
|
||
log.debug(f" {path.name} encoding: {enc}")
|
||
try:
|
||
soup = BeautifulSoup(raw, "lxml", from_encoding=enc)
|
||
except Exception:
|
||
soup = BeautifulSoup(raw, "lxml")
|
||
|
||
# Премахваме скриптове и стилове
|
||
for tag in soup(["script", "style", "nav", "footer", "header", "noscript"]):
|
||
tag.decompose()
|
||
|
||
base_dir = path.parent
|
||
body = soup.body or soup
|
||
|
||
# Събираме top-level блокови елементи (без да включваме вложените в тях)
|
||
consumed = set()
|
||
blocks = []
|
||
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
|
||
name = (el.name or "").lower()
|
||
if name == "div" and not _html_heading_level(el) and not _html_is_cover_title(el):
|
||
continue
|
||
if any(id(par) in consumed for par in el.parents):
|
||
continue
|
||
consumed.add(id(el))
|
||
blocks.append(el)
|
||
|
||
sections: list[Section] = []
|
||
current_title = ""
|
||
current_level = 1
|
||
sec_text: list[str] = []
|
||
sec_html: list[str] = []
|
||
sec_images: list[ImageRef] = []
|
||
img_counter = [0]
|
||
toc_state = _TocSplitState()
|
||
|
||
def flush():
|
||
if current_title or sec_text or sec_html or sec_images:
|
||
sec = Section(current_title, "\n".join(sec_text), current_level)
|
||
sec.images = list(sec_images)
|
||
sec.html_text = "\n".join(sec_html) if sec_html else None
|
||
sections.append(sec)
|
||
|
||
def append_block_as_body(el):
|
||
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
|
||
_strip_attrs(el)
|
||
txt = _html_block_plain_text(el)
|
||
if txt:
|
||
sec_text.append(txt)
|
||
try:
|
||
sec_html.append(str(el))
|
||
except Exception:
|
||
pass
|
||
|
||
def start_section(title: str, level: int = 1):
|
||
nonlocal current_title, current_level, sec_text, sec_html, sec_images
|
||
flush()
|
||
current_title = title
|
||
current_level = level
|
||
sec_text, sec_html, sec_images = [], [], []
|
||
|
||
for el in blocks:
|
||
txt = el.get_text(" ", strip=True)
|
||
is_cover = _html_is_cover_title(el)
|
||
heading_lvl = _html_heading_level(el)
|
||
|
||
# Корица (Word Title / class Title): заглавие на преамбюла, без нова секция
|
||
if is_cover and txt:
|
||
if not current_title and not sec_text and not sec_html and not sec_images:
|
||
current_title = txt
|
||
current_level = 1
|
||
else:
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
append_block_as_body(el)
|
||
continue
|
||
|
||
if txt and _is_toc_heading(txt):
|
||
toc_state.note_toc_heading()
|
||
append_block_as_body(el)
|
||
continue
|
||
|
||
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
|
||
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
|
||
if numbered_action == "toc":
|
||
append_block_as_body(el)
|
||
continue
|
||
if numbered_action == "split":
|
||
start_section(txt, heading_lvl or 1)
|
||
continue
|
||
|
||
if heading_lvl:
|
||
if not txt:
|
||
continue
|
||
# H2+ без номерация остават в тялото; H1 / Заглавие 1 режат
|
||
if heading_lvl >= 2:
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
append_block_as_body(el)
|
||
continue
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
start_section(txt, heading_lvl)
|
||
continue
|
||
|
||
if el.name == "img":
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
# самостоятелен <img> (не вътре в блок)
|
||
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
|
||
base_dir, sec_images, img_counter)
|
||
# ако е заменен с placeholder, добавяме като текст
|
||
img_txt = el.get_text(" ", strip=True) if el.name else ""
|
||
if img_txt:
|
||
sec_text.append(img_txt)
|
||
sec_html.append(f"<p>{img_txt}</p>")
|
||
continue
|
||
|
||
if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS:
|
||
# <ol>/<ul> TOC списък — остава в преамбюла, приключва TOC фазата
|
||
toc_state.leave_phase()
|
||
elif toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
|
||
toc_state.leave_phase()
|
||
append_block_as_body(el)
|
||
|
||
flush()
|
||
|
||
if not sections:
|
||
plain = body.get_text(" ", strip=True)
|
||
return [Section("", plain, 0)]
|
||
return sections
|
||
|
||
|
||
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер."""
|
||
from docx.oxml.ns import qn
|
||
imgs: list[ImageRef] = []
|
||
try:
|
||
blips = para._element.findall(".//" + qn("a:blip"))
|
||
except Exception:
|
||
return imgs
|
||
|
||
embed_attr = qn("r:embed")
|
||
for blip in blips:
|
||
rId = blip.get(embed_attr)
|
||
if not rId:
|
||
continue
|
||
try:
|
||
part = doc.part.related_parts[rId]
|
||
data = part.blob
|
||
ct = getattr(part, "content_type", "") or ""
|
||
except Exception:
|
||
continue
|
||
if not _should_keep_image(data):
|
||
continue
|
||
ext = _ext_from_content_type(ct)
|
||
imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext))
|
||
return imgs
|
||
|
||
|
||
def _iter_docx_blocks(doc):
|
||
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
|
||
from docx.oxml.ns import qn
|
||
from docx.table import Table
|
||
from docx.text.paragraph import Paragraph
|
||
|
||
body = doc.element.body
|
||
for child in body.iterchildren():
|
||
if child.tag == qn("w:p"):
|
||
yield "p", Paragraph(child, doc)
|
||
elif child.tag == qn("w:tbl"):
|
||
yield "tbl", Table(child, doc)
|
||
|
||
|
||
def _table_lines(table) -> list[str]:
|
||
lines: list[str] = []
|
||
try:
|
||
rows = table.rows
|
||
except Exception:
|
||
return lines
|
||
for row in rows:
|
||
try:
|
||
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
|
||
except Exception:
|
||
continue
|
||
cells = [c for c in cells if c]
|
||
if cells:
|
||
lines.append(" | ".join(cells))
|
||
return lines
|
||
|
||
|
||
def parse_docx(path: Path) -> list[Section]:
|
||
doc = Document(path)
|
||
sections: list[Section] = []
|
||
current_title, current_level = "", 1
|
||
buf: list[str] = []
|
||
sec_images: list[ImageRef] = []
|
||
img_counter = [0]
|
||
toc_state = _TocSplitState()
|
||
|
||
def flush():
|
||
if current_title or buf or sec_images:
|
||
sec = Section(current_title, "\n".join(buf), current_level)
|
||
sec.images = list(sec_images)
|
||
sections.append(sec)
|
||
|
||
def append_para(text: str, para_imgs: list[ImageRef]):
|
||
if text:
|
||
buf.append(text)
|
||
for im in para_imgs:
|
||
img_counter[0] += 1
|
||
im.placeholder = f"img_{img_counter[0]:02d}"
|
||
sec_images.append(im)
|
||
buf.append(f"[IMG: {im.placeholder}]")
|
||
|
||
def start_section(title: str, level: int = 1):
|
||
nonlocal current_title, current_level, buf, sec_images
|
||
flush()
|
||
buf, sec_images = [], []
|
||
current_title = title
|
||
current_level = level
|
||
|
||
for kind, block in _iter_docx_blocks(doc):
|
||
if kind == "tbl":
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
for line in _table_lines(block):
|
||
buf.append(line)
|
||
continue
|
||
|
||
para = block
|
||
style_name = para.style.name.lower() if para.style else ""
|
||
text = para.text.strip()
|
||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||
|
||
if not text and not para_imgs:
|
||
continue
|
||
|
||
is_cover = _docx_is_cover_title(para)
|
||
level = _docx_heading_level(para)
|
||
is_bold_heading = (
|
||
not level
|
||
and not is_cover
|
||
and _is_bold_heading_text(text, para.runs)
|
||
and not style_name.startswith("list")
|
||
and not para_imgs
|
||
)
|
||
|
||
# Корица (Word Title): заглавие на преамбюла, без нова секция
|
||
if is_cover and text:
|
||
if not current_title and not buf and not sec_images:
|
||
current_title = text
|
||
current_level = 1
|
||
else:
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
append_para(text, para_imgs)
|
||
continue
|
||
|
||
if text and _is_toc_heading(text):
|
||
toc_state.note_toc_heading()
|
||
append_para(text, para_imgs)
|
||
continue
|
||
|
||
numbered_action = toc_state.handle_numbered(text) if text else "no"
|
||
if numbered_action == "toc":
|
||
append_para(text, para_imgs)
|
||
continue
|
||
if numbered_action == "split":
|
||
start_section(text, level or 1)
|
||
continue
|
||
|
||
# H2+/bold без номерация на глава — в тялото
|
||
if (level or 0) >= 2 or is_bold_heading:
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
append_para(text, para_imgs)
|
||
continue
|
||
|
||
if level == 1:
|
||
if toc_state.phase:
|
||
toc_state.leave_phase()
|
||
start_section(text, 1)
|
||
continue
|
||
|
||
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
|
||
toc_state.leave_phase()
|
||
append_para(text, para_imgs)
|
||
|
||
flush()
|
||
|
||
if not sections:
|
||
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
|
||
if not fallback_text:
|
||
fallback_text = "\n".join(
|
||
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
|
||
for line in _table_lines(block)
|
||
)
|
||
return [Section("", fallback_text, 0)]
|
||
return sections
|
||
|
||
|
||
def _convert_doc_with_libreoffice(path: Path, out_dir: Path) -> Optional[Path]:
|
||
try:
|
||
subprocess.run(
|
||
["soffice", "--headless", "--convert-to", "docx",
|
||
"--outdir", str(out_dir), str(path)],
|
||
check=True, capture_output=True, timeout=60
|
||
)
|
||
except (subprocess.CalledProcessError, FileNotFoundError, subprocess.TimeoutExpired) as e:
|
||
log.debug(f"LibreOffice конверсия неуспешна: {e}")
|
||
return None
|
||
out = list(out_dir.glob("*.docx"))
|
||
return out[0] if out else None
|
||
|
||
|
||
def _convert_doc_with_word(path: Path, out_dir: Path) -> Optional[Path]:
|
||
"""Fallback: ползва MS Word през COM на Windows."""
|
||
try:
|
||
import win32com.client # noqa: F401
|
||
import pythoncom
|
||
except ImportError:
|
||
log.debug("pywin32 не е инсталиран — MS Word fallback недостъпен.")
|
||
return None
|
||
|
||
import win32com.client as wcc
|
||
pythoncom.CoInitialize()
|
||
word = None
|
||
doc = None
|
||
try:
|
||
word = wcc.DispatchEx("Word.Application")
|
||
word.Visible = False
|
||
word.DisplayAlerts = False
|
||
doc = word.Documents.Open(str(path.resolve()), ReadOnly=True)
|
||
out_path = out_dir / (path.stem + ".docx")
|
||
# FileFormat=16 → wdFormatXMLDocument (.docx)
|
||
doc.SaveAs2(str(out_path.resolve()), FileFormat=16)
|
||
return out_path if out_path.exists() else None
|
||
except Exception as e:
|
||
log.debug(f"MS Word конверсия неуспешна: {e}")
|
||
return None
|
||
finally:
|
||
try:
|
||
if doc is not None:
|
||
doc.Close(SaveChanges=False)
|
||
except Exception:
|
||
pass
|
||
try:
|
||
if word is not None:
|
||
word.Quit()
|
||
except Exception:
|
||
pass
|
||
pythoncom.CoUninitialize()
|
||
|
||
|
||
def parse_doc_old(path: Path) -> list[Section]:
|
||
"""Конвертира стар .doc до .docx чрез LibreOffice или MS Word, после парси."""
|
||
with tempfile.TemporaryDirectory() as tmp:
|
||
tmp_dir = Path(tmp)
|
||
|
||
converted = _convert_doc_with_libreoffice(path, tmp_dir)
|
||
engine = "LibreOffice"
|
||
|
||
if not converted:
|
||
converted = _convert_doc_with_word(path, tmp_dir)
|
||
engine = "MS Word"
|
||
|
||
if not converted:
|
||
log.warning(
|
||
f"Нито LibreOffice, нито MS Word успяха да конвертират {path.name}. "
|
||
f"Пробваме като текст."
|
||
)
|
||
return parse_txt(path)
|
||
|
||
log.info(f" {path.name} конвертиран чрез {engine}")
|
||
return parse_docx(converted)
|
||
|
||
|
||
def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
|
||
"""Кропва картинката от PDF страницата и я записва като PNG bytes."""
|
||
try:
|
||
x0 = float(img_info.get("x0", 0))
|
||
x1 = float(img_info.get("x1", 0))
|
||
top = float(img_info.get("top", img_info.get("y0", 0)))
|
||
bot = float(img_info.get("bottom", img_info.get("y1", 0)))
|
||
if x1 <= x0 or bot <= top:
|
||
return None
|
||
# ограничаваме до страницата (pdfplumber иначе хвърля)
|
||
x0 = max(0, x0); top = max(0, top)
|
||
x1 = min(page.width, x1); bot = min(page.height, bot)
|
||
if x1 - x0 < 1 or bot - top < 1:
|
||
return None
|
||
cropped = page.crop((x0, top, x1, bot))
|
||
pil = cropped.to_image(resolution=resolution).original
|
||
from io import BytesIO
|
||
buf = BytesIO()
|
||
pil.save(buf, format="PNG")
|
||
return buf.getvalue()
|
||
except Exception as e:
|
||
log.debug(f"PDF image render failed: {e}")
|
||
return None
|
||
|
||
|
||
_PDF_LINE_Y_TOL = 3.0
|
||
# Typical word space in these PDFs is ~0.2–0.3em; only merge tighter (or drop-caps).
|
||
_PDF_LETTER_GAP_FRAC = 0.12
|
||
# Bare "1" / "2." / "3)" followed by a capital — TOC/list, not "виж 1 и 2".
|
||
_PDF_LIST_MARK_RE = re.compile(
|
||
r"(?<!\d)(?<![A-Za-zА-Яа-яЁёІіЇїЄєҐґ])(\d{1,2})([.)])?(?=\s+[A-ZА-ЯЁІЇЄҐ])"
|
||
)
|
||
|
||
|
||
def _pdf_word_band(w) -> tuple[float, float]:
|
||
top = float(w.get("top", 0))
|
||
bot = float(w.get("bottom", top + float(w.get("size") or 10)))
|
||
if bot <= top:
|
||
bot = top + max(float(w.get("size") or 10), 1.0)
|
||
return top, bot
|
||
|
||
|
||
def _pdf_cluster_words_by_y(words: list) -> list[dict]:
|
||
"""Visual lines by Y (and vertical overlap), not PDF stream order."""
|
||
items = [w for w in words if (w.get("text") or "").strip()]
|
||
items.sort(key=lambda w: (float(w.get("top", 0)), float(w.get("x0", 0))))
|
||
lines: list[dict] = []
|
||
for w in items:
|
||
top, bot = _pdf_word_band(w)
|
||
placed = False
|
||
for line in reversed(lines):
|
||
overlap = min(bot, line["bot"]) - max(top, line["top"])
|
||
band = min(bot - top, line["bot"] - line["top"])
|
||
close = abs(top - line["top"]) <= _PDF_LINE_Y_TOL
|
||
if close or (band > 0 and overlap >= 0.35 * band):
|
||
line["words"].append(w)
|
||
line["top"] = min(line["top"], top)
|
||
line["bot"] = max(line["bot"], bot)
|
||
placed = True
|
||
break
|
||
# Sorted by top: once this word sits clearly below a line, earlier
|
||
# lines are even higher.
|
||
if top > line["bot"] + _PDF_LINE_Y_TOL:
|
||
break
|
||
if not placed:
|
||
lines.append({"top": top, "bot": bot, "words": [w]})
|
||
lines.sort(key=lambda ln: ln["top"])
|
||
return lines
|
||
|
||
|
||
def _pdf_merge_split_letters(ws: list) -> list[dict]:
|
||
"""Join drop-cap / styled first letters: 'C' + 'orrelation' → 'Correlation'."""
|
||
ordered = sorted(ws, key=lambda w: float(w.get("x0", 0)))
|
||
clusters: list[dict] = []
|
||
for w in ordered:
|
||
text = (w.get("text") or "").strip()
|
||
if not text:
|
||
continue
|
||
x0 = float(w.get("x0", 0))
|
||
x1 = float(w.get("x1", x0))
|
||
size = float(w.get("size") or 10)
|
||
font = str(w.get("fontname") or "")
|
||
if not clusters:
|
||
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
|
||
continue
|
||
prev = clusters[-1]
|
||
gap = x0 - prev["x1"]
|
||
ref = max(size, prev["size"], 1.0)
|
||
fonts_differ = bool(prev["font"] and font and prev["font"] != font)
|
||
sizes_differ = abs(size - prev["size"]) > 0.6
|
||
drop = (
|
||
len(prev["text"]) == 1
|
||
and prev["text"].isalpha()
|
||
and text[0].islower()
|
||
)
|
||
tiny = gap < _PDF_LETTER_GAP_FRAC * ref or gap < 0.8
|
||
if tiny or (drop and gap < 0.55 * ref and (fonts_differ or sizes_differ or gap < 0.2 * ref)):
|
||
prev["text"] += text
|
||
prev["x1"] = max(prev["x1"], x1)
|
||
prev["size"] = max(prev["size"], size)
|
||
if font:
|
||
prev["font"] = font
|
||
else:
|
||
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
|
||
return clusters
|
||
|
||
|
||
def _pdf_split_list_text(text: str) -> list[str]:
|
||
"""Break a flattened TOC/list: '... 1 Foo 2 Bar' → one item per line."""
|
||
text = text.strip()
|
||
if not text:
|
||
return []
|
||
matches = list(_PDF_LIST_MARK_RE.finditer(text))
|
||
if len(matches) < 2:
|
||
return [text]
|
||
nums = [int(m.group(1)) for m in matches]
|
||
if nums[0] not in (1, 2):
|
||
return [text]
|
||
if any(nums[i] != nums[0] + i for i in range(len(nums))):
|
||
return [text]
|
||
parts: list[str] = []
|
||
last = 0
|
||
for m in matches:
|
||
if m.start() > last:
|
||
head = text[last:m.start()].strip()
|
||
if head:
|
||
parts.append(head)
|
||
last = m.start()
|
||
tail = text[last:].strip()
|
||
if tail:
|
||
parts.append(tail)
|
||
return parts or [text]
|
||
|
||
|
||
def _pdf_words_to_lines(words: list) -> list[dict]:
|
||
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
|
||
result = []
|
||
for line in _pdf_cluster_words_by_y(words):
|
||
ws = line["words"]
|
||
tokens = _pdf_merge_split_letters(ws)
|
||
if not tokens:
|
||
continue
|
||
joined = " ".join(t["text"] for t in tokens).strip()
|
||
if not joined:
|
||
continue
|
||
size = round(float(ws[0].get("size", 10)), 1)
|
||
for piece in _pdf_split_list_text(joined):
|
||
result.append({"top": line["top"], "size": size, "text": piece})
|
||
return result
|
||
|
||
|
||
def parse_pdf(path: Path) -> list[Section]:
|
||
if not HAS_PDF:
|
||
log.warning("pdfplumber не е инсталиран. PDF се прескача.")
|
||
return []
|
||
|
||
sections: list[Section] = []
|
||
current_title = ""
|
||
buf: list[str] = []
|
||
sec_images: list[ImageRef] = []
|
||
img_counter = [0]
|
||
prev_size = None
|
||
|
||
def flush():
|
||
if current_title or buf or sec_images:
|
||
sec = Section(current_title, "\n".join(buf), 2)
|
||
sec.images = list(sec_images)
|
||
sections.append(sec)
|
||
|
||
with pdfplumber.open(path) as pdf:
|
||
for page in pdf.pages:
|
||
# Картинките за страницата (сортирани по y отгоре надолу)
|
||
page_images = sorted(
|
||
page.images or [],
|
||
key=lambda im: float(im.get("top", im.get("y0", 0)))
|
||
)
|
||
img_queue = []
|
||
for im in page_images:
|
||
data = _render_pdf_image(page, im)
|
||
if not data or not _should_keep_image(data):
|
||
continue
|
||
img_queue.append((float(im.get("top", 0)), data))
|
||
|
||
words = page.extract_words(extra_attrs=["size", "fontname"])
|
||
visual_lines = _pdf_words_to_lines(words)
|
||
line_buf, line_size = [], None
|
||
|
||
def emit_images_before(y: float):
|
||
while img_queue and img_queue[0][0] <= y:
|
||
_, data = img_queue.pop(0)
|
||
img_counter[0] += 1
|
||
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}",
|
||
data=data, ext="png")
|
||
sec_images.append(ref)
|
||
buf.append(f"[IMG: {ref.placeholder}]")
|
||
|
||
for vl in visual_lines:
|
||
sz = vl["size"]
|
||
y = vl["top"]
|
||
if line_size is None:
|
||
line_size = sz
|
||
if abs(sz - line_size) > 1:
|
||
line_text = "\n".join(line_buf).strip()
|
||
if line_text:
|
||
if line_size > (prev_size or 10) + 1 and len(line_text) < 150:
|
||
flush()
|
||
buf, sec_images = [], []
|
||
current_title = " ".join(line_text.split())
|
||
else:
|
||
emit_images_before(y)
|
||
buf.append(line_text)
|
||
prev_size = line_size
|
||
line_buf, line_size = [vl["text"]], sz
|
||
else:
|
||
line_buf.append(vl["text"])
|
||
|
||
if line_buf:
|
||
emit_images_before(page.height)
|
||
buf.append("\n".join(line_buf))
|
||
|
||
# картинките след всичкия текст на страницата
|
||
emit_images_before(page.height + 1)
|
||
|
||
flush()
|
||
|
||
return sections or [Section("", "", 0)]
|
||
|
||
|
||
def parse_txt(path: Path) -> list[Section]:
|
||
import chardet
|
||
raw = path.read_bytes()
|
||
enc = chardet.detect(raw)["encoding"] or "utf-8"
|
||
text = raw.decode(enc, errors="replace")
|
||
return _split_plain_text(text)
|
||
|
||
|
||
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
|
||
_PLAIN_NUM_HEADING_RE = re.compile(
|
||
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
|
||
)
|
||
|
||
|
||
def _split_plain_text(text: str) -> list[Section]:
|
||
"""Markdown / номерирани заглавия; иначе една секция."""
|
||
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
||
sections: list[Section] = []
|
||
title, level, buf = "", 0, []
|
||
|
||
def flush():
|
||
body = "\n".join(buf).strip()
|
||
if title or body:
|
||
sections.append(Section(title, body, level))
|
||
|
||
for line in lines:
|
||
raw = line.strip()
|
||
md = _PLAIN_MD_HEADING_RE.match(raw)
|
||
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
|
||
if md or numbered:
|
||
flush()
|
||
title = md.group(2).strip() if md else raw
|
||
level = len(md.group(1)) if md else 2
|
||
buf = []
|
||
continue
|
||
buf.append(line.rstrip())
|
||
flush()
|
||
return sections or [Section("", text, 0)]
|
||
|
||
|
||
PARSERS = {
|
||
".html": parse_html,
|
||
".htm": parse_html,
|
||
".docx": parse_docx,
|
||
".doc": parse_doc_old,
|
||
".txt": parse_txt,
|
||
".pdf": parse_pdf,
|
||
}
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Сегментиране и почистване
|
||
# ──────────────────────────────────────────────
|
||
|
||
def merge_short_sections(sections: list[Section]) -> list[Section]:
|
||
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
|
||
result: list[Section] = []
|
||
for sec in sections:
|
||
words = len((sec.text or "").split())
|
||
titled = bool((sec.title or "").strip())
|
||
if result and not titled and words < MIN_SECTION_TOKENS:
|
||
prev = result[-1]
|
||
merged = Section(
|
||
prev.title,
|
||
(prev.text + "\n" + sec.text).strip(),
|
||
prev.level,
|
||
)
|
||
merged.images = (prev.images or []) + (sec.images or [])
|
||
html_parts = [h for h in (prev.html_text, sec.html_text) if h]
|
||
merged.html_text = "\n".join(html_parts) if html_parts else None
|
||
result[-1] = merged
|
||
else:
|
||
result.append(sec)
|
||
return result
|
||
|
||
|
||
def merge_preamble_sections(sections: list[Section]) -> list[Section]:
|
||
"""Слива корица (Title) + „Съдържание“/TOC в един преамбюл, ако парсерът ги е разделил."""
|
||
if len(sections) < 2:
|
||
return sections
|
||
first, second = sections[0], sections[1]
|
||
first_words = len((first.text or "").split())
|
||
if not (first.title or "").strip():
|
||
return sections
|
||
if first_words >= MIN_SECTION_TOKENS:
|
||
return sections
|
||
if not _is_toc_heading(second.title or ""):
|
||
return sections
|
||
|
||
body_parts = [p for p in (second.title, second.text) if (p or "").strip()]
|
||
if (first.text or "").strip():
|
||
body_parts.append(first.text.strip())
|
||
merged = Section(first.title, "\n".join(body_parts).strip(), first.level)
|
||
merged.images = (first.images or []) + (second.images or [])
|
||
html_parts = [h for h in (first.html_text, second.html_text) if h]
|
||
merged.html_text = "\n".join(html_parts) if html_parts else None
|
||
return [merged] + list(sections[2:])
|
||
|
||
|
||
def clean_text(text: str) -> str:
|
||
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
|
||
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
||
text = re.sub(r"[^\S\n]+", " ", text)
|
||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||
text = "\n".join(line.strip() for line in text.split("\n"))
|
||
return text.strip()
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# AI класификация
|
||
# ──────────────────────────────────────────────
|
||
|
||
def _content_text(msg) -> str:
|
||
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
|
||
parts: list[str] = []
|
||
for block in getattr(msg, "content", None) or []:
|
||
btype = getattr(block, "type", None)
|
||
if btype in (None, "text"):
|
||
t = getattr(block, "text", None)
|
||
if t:
|
||
parts.append(str(t))
|
||
elif isinstance(block, dict) and block.get("text"):
|
||
parts.append(str(block["text"]))
|
||
return "\n".join(parts).strip()
|
||
|
||
|
||
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
|
||
raw = (raw or "").strip()
|
||
if not raw:
|
||
return None
|
||
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
||
raw = re.sub(r"\n?```$", "", raw)
|
||
candidates = [raw]
|
||
m = re.search(r"\{.*\}", raw, re.S)
|
||
if m:
|
||
candidates.append(m.group(0))
|
||
for cand in candidates:
|
||
try:
|
||
data = json.loads(cand)
|
||
except json.JSONDecodeError:
|
||
continue
|
||
if not isinstance(data, dict):
|
||
continue
|
||
t = data.get("title", "")
|
||
k = data.get("keywords", "")
|
||
if isinstance(k, list):
|
||
k = ", ".join(str(x).strip() for x in k if str(x).strip())
|
||
return str(t)[:200], str(k)[:300]
|
||
return None
|
||
|
||
|
||
_FALLBACK_KW_STOP = {
|
||
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
|
||
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
|
||
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
|
||
}
|
||
|
||
|
||
def fallback_classify(title: str, text: str) -> tuple[str, str]:
|
||
t = (title or "").strip()
|
||
if not t:
|
||
for line in (text or "").splitlines():
|
||
line = line.strip()
|
||
if 3 <= len(line) <= 80:
|
||
t = line
|
||
break
|
||
if not t:
|
||
t = "Секция"
|
||
blob = f"{title} {text}"[:1200]
|
||
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
|
||
seen: list[str] = []
|
||
seen_l: set[str] = set()
|
||
for w in words:
|
||
wl = w.lower()
|
||
if wl in _FALLBACK_KW_STOP or wl in seen_l:
|
||
continue
|
||
seen.append(w)
|
||
seen_l.add(wl)
|
||
if len(seen) >= 5:
|
||
break
|
||
return t[:200], ", ".join(seen)
|
||
|
||
|
||
KEYWORDS_MAX_LEN = 300 # rip_help_sections.keywords VARCHAR(300)
|
||
|
||
|
||
def parse_keyword_list(raw: str) -> list[str]:
|
||
"""Разделя ключови думи. Запетая/точка и запетая = фрази; иначе — интервал."""
|
||
text = (raw or "").strip()
|
||
if not text:
|
||
return []
|
||
if "," in text or ";" in text:
|
||
parts = re.split(r"[,;]+", text)
|
||
else:
|
||
parts = text.split()
|
||
out: list[str] = []
|
||
seen: set[str] = set()
|
||
for p in parts:
|
||
k = " ".join(p.split())
|
||
if not k:
|
||
continue
|
||
key = k.casefold()
|
||
if key in seen:
|
||
continue
|
||
seen.add(key)
|
||
out.append(k)
|
||
return out
|
||
|
||
|
||
def merge_section_keywords(
|
||
manual: str,
|
||
generated: str,
|
||
max_len: int = KEYWORDS_MAX_LEN,
|
||
) -> str:
|
||
"""Ръчните думи първи, после генерираните; без дубликати (без значение на регистъра)."""
|
||
first = parse_keyword_list(manual)
|
||
seen = {k.casefold() for k in first}
|
||
rest: list[str] = []
|
||
for k in parse_keyword_list(generated):
|
||
key = k.casefold()
|
||
if key in seen:
|
||
continue
|
||
seen.add(key)
|
||
rest.append(k)
|
||
merged = first + rest
|
||
if not merged:
|
||
return ""
|
||
out: list[str] = []
|
||
used = 0
|
||
for i, k in enumerate(merged):
|
||
extra = len(k) + (2 if i else 0) # ", "
|
||
if used + extra > max_len:
|
||
break
|
||
out.append(k)
|
||
used += extra
|
||
return ", ".join(out)
|
||
|
||
|
||
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
||
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
||
snippet = text[:MAX_AI_CHARS]
|
||
prompt = f"""Анализирай следната секция от help-документация и върни JSON обект с два ключа:
|
||
- "title": кратко наименование на секцията (до 8 думи, на езика на текста)
|
||
- "keywords": списък от до 5 ключови думи/фрази, разделени със запетая (на езика на текста)
|
||
|
||
Съществуващо заглавие (може да е празно): {title!r}
|
||
|
||
Текст:
|
||
{snippet}
|
||
|
||
Върни САМО валиден JSON без markdown, без коментари."""
|
||
|
||
last_err: Optional[Exception] = None
|
||
tried: set[str] = set()
|
||
for model in AI_MODELS:
|
||
if not model or model in tried:
|
||
continue
|
||
tried.add(model)
|
||
try:
|
||
msg = client.messages.create(
|
||
model=model,
|
||
max_tokens=512,
|
||
messages=[{"role": "user", "content": prompt}],
|
||
)
|
||
raw = _content_text(msg)
|
||
parsed = _parse_classify_json(raw)
|
||
if parsed:
|
||
t, k = parsed
|
||
return (t or title or "Секция")[:200], k
|
||
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
|
||
except Exception as e:
|
||
last_err = e
|
||
log.warning(f"AI classify ({model}) неуспешен: {e}")
|
||
continue
|
||
if last_err:
|
||
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
|
||
return fallback_classify(title, text)
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Генериране на кодове — help_codes.py
|
||
# ──────────────────────────────────────────────
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# Основна обработка
|
||
# ──────────────────────────────────────────────
|
||
|
||
def _db_output_path(local_path: Path, code: str, remote_root: Optional[str]) -> str:
|
||
"""Локален staging файл; в БД — сървърен път ако е зададен remote_root."""
|
||
if remote_root:
|
||
root = remote_root.replace("\\", "/").rstrip("/")
|
||
return f"{root}/{code}.txt"
|
||
return str(local_path)
|
||
|
||
|
||
def process_file(
|
||
path: Path,
|
||
file_index: int,
|
||
db: Database,
|
||
client: anthropic.Anthropic,
|
||
output_dir: Path,
|
||
prefix: str = "HLP",
|
||
force: bool = False,
|
||
remote_root: Optional[str] = None,
|
||
source_key: Optional[str] = None,
|
||
seed_keywords: str = "",
|
||
) -> dict:
|
||
"""Обработва един файл. Връща статистика (saved, codes, ok)."""
|
||
rel = source_key or path.name
|
||
fh = file_hash(path)
|
||
|
||
existing_rows = db.sections_for_file(prefix, rel)
|
||
existing_codes = [r[0] for r in existing_rows]
|
||
|
||
if not force:
|
||
stored = db.get_file_hash(prefix, rel)
|
||
if stored == fh:
|
||
log.info(f" [SKIP] {path.name} (непроменен)")
|
||
return _file_result(rel, file_index, existing_codes, saved=0, skipped=True)
|
||
|
||
log.info(f" [PROC] {path.name}")
|
||
ext = path.suffix.lower()
|
||
parser = PARSERS.get(ext)
|
||
if not parser:
|
||
log.warning(f" Неподдържан формат: {ext}")
|
||
return _file_result(rel, file_index, existing_codes, saved=0)
|
||
|
||
try:
|
||
sections = parser(path)
|
||
except Exception as e:
|
||
log.error(f" Грешка при парсване: {e}")
|
||
return _file_result(rel, file_index, existing_codes, saved=0)
|
||
|
||
sections = merge_short_sections(sections)
|
||
sections = merge_preamble_sections(sections)
|
||
|
||
remove_section_outputs(
|
||
output_dir,
|
||
existing_codes,
|
||
[r[2] for r in existing_rows if r[2]],
|
||
)
|
||
db.delete_sections_for_file(prefix, rel)
|
||
|
||
images_dir = output_dir / "images"
|
||
images_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
saved = 0
|
||
codes: list[str] = []
|
||
ai_errors = 0
|
||
for sec in sections:
|
||
text = clean_text(sec.text)
|
||
html_text = sec.html_text or ""
|
||
if not text and not sec.images and not html_text:
|
||
if (sec.title or "").strip():
|
||
text = sec.title.strip()
|
||
else:
|
||
continue
|
||
|
||
sec_index = saved + 1
|
||
code = make_code(prefix, file_index, sec_index)
|
||
|
||
# Записваме картинките на диск и заменяме placeholder-ите в текста + HTML
|
||
image_rel_paths: list[str] = []
|
||
for ref in sec.images or []:
|
||
fname = f"{code}_{ref.placeholder}.{ref.ext}"
|
||
disk_path = images_dir / fname
|
||
try:
|
||
disk_path.write_bytes(ref.data)
|
||
except Exception as e:
|
||
log.warning(f" Грешка при запис на картинка {fname}: {e}")
|
||
continue
|
||
rel_path = f"images/{fname}"
|
||
image_rel_paths.append(rel_path)
|
||
old_ph = f"[IMG: {ref.placeholder}]"
|
||
new_ph = f"[IMG: {rel_path}]"
|
||
text = text.replace(old_ph, new_ph)
|
||
html_text = html_text.replace(old_ph, new_ph)
|
||
|
||
# Премахваме placeholder-и, останали без файл
|
||
text = _IMG_PLACEHOLDER_RE.sub(
|
||
lambda m: m.group(0) if "/" in m.group(1) or "\\" in m.group(1) else "",
|
||
text
|
||
).strip()
|
||
html_text = _IMG_PLACEHOLDER_RE.sub(
|
||
lambda m: m.group(0) if "/" in m.group(1) or "\\" in m.group(1) else "",
|
||
html_text
|
||
).strip()
|
||
if not text and not image_rel_paths and not html_text:
|
||
continue
|
||
|
||
try:
|
||
title, keywords = classify_section(client, sec.title, text)
|
||
except Exception as e:
|
||
log.warning(f" AI грешка за {code}: {e}")
|
||
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
|
||
ai_errors += 1
|
||
if not (keywords or "").strip():
|
||
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
|
||
ai_errors += 1
|
||
keywords = merge_section_keywords(seed_keywords, keywords)
|
||
|
||
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
||
ps = ProcessedSection(
|
||
code=code,
|
||
source_file=rel,
|
||
title=title,
|
||
keywords=keywords,
|
||
text=text,
|
||
images_json=images_json,
|
||
html_text=html_text,
|
||
)
|
||
|
||
# Локален staging; в БД — remote път (ако е зададен) за Coolify/API
|
||
out_path = output_dir / f"{code}.txt"
|
||
out_path.write_text(
|
||
f"КОД: {code}\nФАЙЛ: {rel}\nЗАГЛАВИЕ: {title}\nКЛЮЧОВИ ДУМИ: {keywords}\n"
|
||
f"КАРТИНКИ: {len(image_rel_paths)}\n"
|
||
f"{'─'*60}\n{text}",
|
||
encoding="utf-8"
|
||
)
|
||
|
||
db.insert_section(prefix, ps, _db_output_path(out_path, code, remote_root))
|
||
saved += 1
|
||
codes.append(code)
|
||
log.debug(f" {code}: {title[:60]} ({len(image_rel_paths)} img)")
|
||
|
||
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
|
||
log.info(f" → {saved} секции записани")
|
||
result = _file_result(rel, file_index, codes, saved=saved)
|
||
result["ai_errors"] = ai_errors
|
||
result["keywords_ok"] = ai_errors == 0
|
||
return result
|
||
|
||
|
||
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
|
||
|
||
|
||
def process_directory(
|
||
input_dir: Path,
|
||
output_dir: Path,
|
||
conn_str: str,
|
||
api_key: str,
|
||
prefix: str = "HLP",
|
||
force: bool = False,
|
||
purge_missing: bool = False,
|
||
remote_root: Optional[str] = None,
|
||
seed_keywords: str = "",
|
||
):
|
||
if not _PREFIX_RE.match(prefix):
|
||
raise ValueError(
|
||
f"Невалиден prefix {prefix!r}. Допустими: буква + букви/цифри/подчертавки, до 50 символа."
|
||
)
|
||
|
||
output_dir.mkdir(parents=True, exist_ok=True)
|
||
db = Database(conn_str)
|
||
client = anthropic.Anthropic(api_key=api_key)
|
||
|
||
extensions = set(PARSERS.keys())
|
||
output_resolved = output_dir.resolve()
|
||
|
||
def _under_output(p: Path) -> bool:
|
||
try:
|
||
p.resolve().relative_to(output_resolved)
|
||
return True
|
||
except ValueError:
|
||
return False
|
||
|
||
files = [
|
||
p for p in input_dir.rglob("*")
|
||
if p.is_file() and p.suffix.lower() in extensions and not _under_output(p)
|
||
]
|
||
log.info(f"Prefix={prefix} Намерени {len(files)} файла в {input_dir}")
|
||
if remote_root:
|
||
log.info(f"DB output_path root: {remote_root.replace(chr(92), '/').rstrip('/')}")
|
||
|
||
current_ids = {source_identity(p, input_dir) for p in files}
|
||
current_names = {source_basename(i).lower() for i in current_ids}
|
||
file_results: list[dict] = []
|
||
total_sections = 0
|
||
try:
|
||
for path in sorted(files):
|
||
identity = source_identity(path, input_dir)
|
||
idx = db.allocate_file_index(prefix, identity)
|
||
info = process_file(
|
||
path, idx, db, client, output_dir,
|
||
prefix=prefix, force=force, remote_root=remote_root,
|
||
source_key=identity,
|
||
seed_keywords=seed_keywords,
|
||
)
|
||
file_results.append(info)
|
||
total_sections += int(info.get("saved") or 0)
|
||
|
||
if purge_missing:
|
||
existing = set(db.all_source_files(prefix))
|
||
orphans = sorted(
|
||
e for e in existing
|
||
if source_basename(e).lower() not in current_names
|
||
)
|
||
if not orphans:
|
||
log.info(f"Purge: няма orphan записи в БД за prefix={prefix}.")
|
||
else:
|
||
log.info(f"Purge ({prefix}): намерени {len(orphans)} orphan източника:")
|
||
for o in orphans:
|
||
log.info(f" - {o}")
|
||
disk_paths = db.section_output_paths_for(prefix, orphans)
|
||
removed_files = 0
|
||
for op in disk_paths:
|
||
try:
|
||
code = Path(str(op).replace("\\", "/")).stem
|
||
local_txt = output_dir / f"{code}.txt"
|
||
if local_txt.exists():
|
||
local_txt.unlink()
|
||
removed_files += 1
|
||
opath = Path(op)
|
||
if opath.exists():
|
||
try:
|
||
if opath.resolve() != local_txt.resolve():
|
||
opath.unlink()
|
||
removed_files += 1
|
||
except Exception:
|
||
pass
|
||
for img in (output_dir / "images").glob(f"{code}_*"):
|
||
try:
|
||
img.unlink()
|
||
removed_files += 1
|
||
except Exception:
|
||
pass
|
||
except Exception as e:
|
||
log.debug(f" не успях да изтрия {op}: {e}")
|
||
deleted = db.purge_sources(prefix, orphans)
|
||
log.info(f"Purge: изтрити {deleted} секции от БД, {removed_files} файла от диска.")
|
||
finally:
|
||
db.close()
|
||
|
||
log.info(f"Готово. Prefix={prefix}. Общо нови/обновени секции: {total_sections}")
|
||
return {"sections": total_sections, "files": file_results}
|
||
|
||
|
||
# ──────────────────────────────────────────────
|
||
# CLI
|
||
# ──────────────────────────────────────────────
|
||
|
||
def main():
|
||
parser = argparse.ArgumentParser(
|
||
description="Help-файл декомпозитор с PostgreSQL + Anthropic"
|
||
)
|
||
parser.add_argument("input_dir", help="Входна директория с help-файлове")
|
||
parser.add_argument("output_dir", help="Изходна директория за текстови секции")
|
||
parser.add_argument(
|
||
"--conn",
|
||
default=os.getenv("HELP_DB_CONN"),
|
||
help="Postgres libpq connection string (или HELP_DB_CONN env var)"
|
||
)
|
||
parser.add_argument(
|
||
"--api-key",
|
||
default=os.getenv("ANTHROPIC_API_KEY"),
|
||
help="Anthropic API ключ (или ANTHROPIC_API_KEY env var)"
|
||
)
|
||
parser.add_argument(
|
||
"--prefix",
|
||
default=os.getenv("HELP_PREFIX", "HLP"),
|
||
help="Префикс за кодовете/scope в БД (буква + букви/цифри/_, до 50 знака). "
|
||
"Default: 'HLP' (или env HELP_PREFIX)."
|
||
)
|
||
parser.add_argument(
|
||
"--force",
|
||
action="store_true",
|
||
help="Преобработва всички файлове, независимо от hash"
|
||
)
|
||
parser.add_argument(
|
||
"--purge-missing",
|
||
action="store_true",
|
||
help="След обработката изтрива от БД и диска секциите за източници, "
|
||
"които вече не съществуват във входната директория (само в дадения prefix)"
|
||
)
|
||
parser.add_argument(
|
||
"--remote-root",
|
||
default=os.getenv("HELP_REMOTE_ROOT"),
|
||
help="Сървърен корен за output_path в БД "
|
||
"(напр. /mnt/mssql/share/RIP/RIP_Help_Source/Output). "
|
||
"Файловете се пишат локално; в БД се записва remote път. "
|
||
"Или env HELP_REMOTE_ROOT."
|
||
)
|
||
parser.add_argument(
|
||
"--keywords",
|
||
default="",
|
||
help="Ръчни ключови думи (запетая). Стоят в началото на всяка секция от това сканиране."
|
||
)
|
||
args = parser.parse_args()
|
||
|
||
if not args.api_key:
|
||
sys.exit("Грешка: липсва Anthropic API ключ (--api-key или ANTHROPIC_API_KEY).")
|
||
if not args.conn:
|
||
sys.exit("Грешка: липсва Postgres connection string (--conn или HELP_DB_CONN).")
|
||
|
||
process_directory(
|
||
input_dir=Path(args.input_dir),
|
||
output_dir=Path(args.output_dir),
|
||
conn_str=args.conn,
|
||
api_key=args.api_key,
|
||
prefix=args.prefix,
|
||
force=args.force,
|
||
purge_missing=args.purge_missing,
|
||
remote_root=args.remote_root,
|
||
seed_keywords=args.keywords,
|
||
)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
main()
|