This commit is contained in:
2026-09-11 13:58:06 +03:00
parent dfbc646c51
commit d4dd7be4ef
4 changed files with 484 additions and 126 deletions

View File

@@ -71,19 +71,27 @@ try:
except AttributeError:
pass
_log_handlers = [logging.StreamHandler(sys.stdout)]
try:
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
except OSError:
pass
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)-8s %(message)s",
handlers=[
logging.StreamHandler(sys.stdout),
logging.FileHandler("help_processor.log", encoding="utf-8"),
],
handlers=_log_handlers,
)
log = logging.getLogger(__name__)
MIN_SECTION_TOKENS = 60 # секции под тази граница се сливат с предишната
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
AI_MODEL = "claude-haiku-4-5"
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
AI_MODELS = [
AI_MODEL,
"claude-haiku-4-5",
"claude-haiku-4-5-20251001",
"claude-3-5-haiku-latest",
]
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
@@ -290,7 +298,7 @@ class Database:
self.conn.commit()
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
"""(code, source_file, output_path) за всички извлечения на файла."""
"""(code, source_file, output_path) за всички секции на файла."""
paths = self.matching_source_paths(prefix, identity)
if not paths:
return []
@@ -371,7 +379,7 @@ class Database:
identity: str,
output_dir: Optional[Path] = None,
) -> dict:
"""Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001."""
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
rows = self.sections_for_file(prefix, identity)
codes = [r[0] for r in rows]
file_index = self.file_index_for(prefix, identity)
@@ -548,6 +556,96 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
"valign", "width", "height", "bgcolor", "border")
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
_HEADING_TOKEN_RE = re.compile(
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
r"(\d+)?$",
re.I,
)
_HTML_HEADING_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
re.I,
)
_HEADING_LEVEL = {
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
"title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
"подзаглавие": 2, "наименование": 1,
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
}
def _compact_style_token(s: str) -> str:
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
def _heading_level_from_token(token: str) -> Optional[int]:
t = _compact_style_token(token)
if not t:
return None
if t in _HEADING_LEVEL:
return _HEADING_LEVEL[t]
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
if not m:
return None
n = m.group(2)
if n and n.isdigit():
return min(int(n), 3)
kind = (m.group(1) or "").lower()
if kind in ("subtitle", "подзаглавие"):
return 2
return 1
def _docx_heading_level(para) -> Optional[int]:
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word."""
style = getattr(para, "style", None)
seen: set[int] = set()
cur = style
while cur is not None and id(cur) not in seen:
seen.add(id(cur))
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
lvl = _heading_level_from_token(str(token or ""))
if lvl:
return lvl
cur = getattr(cur, "base_style", None)
try:
pPr = para._element.pPr
if pPr is not None and pPr.outlineLvl is not None:
val = int(pPr.outlineLvl.val)
text = (para.text or "").strip()
if 0 <= val <= 2 and text and len(text) < 120:
return min(val + 1, 3)
except Exception:
pass
return None
def _is_bold_heading_text(text: str, runs) -> bool:
if not text or len(text) > 120:
return False
useful = [r for r in (runs or []) if (r.text or "").strip()]
return bool(useful) and all(bool(r.bold) for r in useful)
def _html_heading_level(el) -> Optional[int]:
name = (getattr(el, "name", None) or "").lower()
if name in _HTML_HEADING_MAP:
return _HTML_HEADING_MAP[name]
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
m = _HTML_HEADING_CLASS_RE.search(cls)
if m:
n = m.group(1)
return min(int(n), 3) if n and n.isdigit() else 1
if name in ("p", "div"):
txt = el.get_text(" ", strip=True)
if txt and len(txt) < 120:
inner = "".join(el.stripped_strings)
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
if strong and strong_txt and strong_txt == inner:
return 2
return None
def _html_block_plain_text(el) -> str:
@@ -600,12 +698,13 @@ def parse_html(path: Path) -> list[Section]:
base_dir = path.parent
body = soup.body or soup
heading_map = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
# Събираме top-level блокови елементи (без да включваме вложените в тях)
consumed = set()
blocks = []
for el in body.find_all(_HTML_BLOCK_TAGS + ["img"]):
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
name = (el.name or "").lower()
if name == "div" and not _html_heading_level(el):
continue
if any(id(par) in consumed for par in el.parents):
continue
consumed.add(id(el))
@@ -620,20 +719,21 @@ def parse_html(path: Path) -> list[Section]:
img_counter = [0]
def flush():
if sec_text or sec_html or sec_images:
if current_title or sec_text or sec_html or sec_images:
sec = Section(current_title, "\n".join(sec_text), current_level)
sec.images = list(sec_images)
sec.html_text = "\n".join(sec_html) if sec_html else None
sections.append(sec)
for el in blocks:
if el.name in heading_map:
heading_lvl = _html_heading_level(el)
if heading_lvl:
txt = el.get_text(" ", strip=True)
if not txt:
continue
flush()
current_title = txt
current_level = heading_map[el.name]
current_level = heading_lvl
sec_text, sec_html, sec_images = [], [], []
continue
@@ -693,37 +793,74 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
return imgs
def _iter_docx_blocks(doc):
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
from docx.oxml.ns import qn
from docx.table import Table
from docx.text.paragraph import Paragraph
body = doc.element.body
for child in body.iterchildren():
if child.tag == qn("w:p"):
yield "p", Paragraph(child, doc)
elif child.tag == qn("w:tbl"):
yield "tbl", Table(child, doc)
def _table_lines(table) -> list[str]:
lines: list[str] = []
try:
rows = table.rows
except Exception:
return lines
for row in rows:
try:
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
except Exception:
continue
cells = [c for c in cells if c]
if cells:
lines.append(" | ".join(cells))
return lines
def parse_docx(path: Path) -> list[Section]:
doc = Document(path)
sections: list[Section] = []
current_title, current_level = "", 1
buf: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0] # списък за nonlocal-стил мутация
HEADING_STYLES = {"heading 1": 1, "heading 2": 2, "heading 3": 3,
"title": 1, "subtitle": 2}
img_counter = [0]
def flush():
if buf or sec_images:
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), current_level)
sec.images = list(sec_images)
sections.append(sec)
for para in doc.paragraphs:
for kind, block in _iter_docx_blocks(doc):
if kind == "tbl":
for line in _table_lines(block):
buf.append(line)
continue
para = block
style_name = para.style.name.lower() if para.style else ""
text = para.text.strip()
para_imgs = _extract_docx_paragraph_images(para, doc)
text = para.text.strip()
para_imgs = _extract_docx_paragraph_images(para, doc)
if not text and not para_imgs:
continue
level = HEADING_STYLES.get(style_name)
is_bold_heading = bool(text and len(text) < 120 and not style_name.startswith("list")
and para.runs
and all(run.bold for run in para.runs if run.text.strip()))
level = _docx_heading_level(para)
is_bold_heading = (
not level
and _is_bold_heading_text(text, para.runs)
and not style_name.startswith("list")
and not para_imgs
)
if level or (is_bold_heading and not para_imgs):
if level or is_bold_heading:
flush()
buf, sec_images = [], []
current_title = text
@@ -742,6 +879,11 @@ def parse_docx(path: Path) -> list[Section]:
if not sections:
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
if not fallback_text:
fallback_text = "\n".join(
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
for line in _table_lines(block)
)
return [Section("", fallback_text, 0)]
return sections
@@ -986,7 +1128,7 @@ def parse_pdf(path: Path) -> list[Section]:
prev_size = None
def flush():
if buf or sec_images:
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), 2)
sec.images = list(sec_images)
sections.append(sec)
@@ -1055,7 +1197,39 @@ def parse_txt(path: Path) -> list[Section]:
raw = path.read_bytes()
enc = chardet.detect(raw)["encoding"] or "utf-8"
text = raw.decode(enc, errors="replace")
return [Section("", text, 0)]
return _split_plain_text(text)
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
_PLAIN_NUM_HEADING_RE = re.compile(
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
)
def _split_plain_text(text: str) -> list[Section]:
"""Markdown / номерирани заглавия; иначе една секция."""
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
sections: list[Section] = []
title, level, buf = "", 0, []
def flush():
body = "\n".join(buf).strip()
if title or body:
sections.append(Section(title, body, level))
for line in lines:
raw = line.strip()
md = _PLAIN_MD_HEADING_RE.match(raw)
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
if md or numbered:
flush()
title = md.group(2).strip() if md else raw
level = len(md.group(1)) if md else 2
buf = []
continue
buf.append(line.rstrip())
flush()
return sections or [Section("", text, 0)]
PARSERS = {
@@ -1073,15 +1247,16 @@ PARSERS = {
# ──────────────────────────────────────────────
def merge_short_sections(sections: list[Section]) -> list[Section]:
"""Слива секции, по-кратки от MIN_SECTION_TOKENS думи, с предишната."""
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
result: list[Section] = []
for sec in sections:
words = len(sec.text.split())
if result and words < MIN_SECTION_TOKENS:
words = len((sec.text or "").split())
titled = bool((sec.title or "").strip())
if result and not titled and words < MIN_SECTION_TOKENS:
prev = result[-1]
merged = Section(
prev.title,
prev.text + "\n" + sec.text,
(prev.text + "\n" + sec.text).strip(),
prev.level,
)
merged.images = (prev.images or []) + (sec.images or [])
@@ -1106,6 +1281,77 @@ def clean_text(text: str) -> str:
# AI класификация
# ──────────────────────────────────────────────
def _content_text(msg) -> str:
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
parts: list[str] = []
for block in getattr(msg, "content", None) or []:
btype = getattr(block, "type", None)
if btype in (None, "text"):
t = getattr(block, "text", None)
if t:
parts.append(str(t))
elif isinstance(block, dict) and block.get("text"):
parts.append(str(block["text"]))
return "\n".join(parts).strip()
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
raw = (raw or "").strip()
if not raw:
return None
raw = re.sub(r"^```[a-z]*\n?", "", raw)
raw = re.sub(r"\n?```$", "", raw)
candidates = [raw]
m = re.search(r"\{.*\}", raw, re.S)
if m:
candidates.append(m.group(0))
for cand in candidates:
try:
data = json.loads(cand)
except json.JSONDecodeError:
continue
if not isinstance(data, dict):
continue
t = data.get("title", "")
k = data.get("keywords", "")
if isinstance(k, list):
k = ", ".join(str(x).strip() for x in k if str(x).strip())
return str(t)[:200], str(k)[:300]
return None
_FALLBACK_KW_STOP = {
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
}
def fallback_classify(title: str, text: str) -> tuple[str, str]:
t = (title or "").strip()
if not t:
for line in (text or "").splitlines():
line = line.strip()
if 3 <= len(line) <= 80:
t = line
break
if not t:
t = "Секция"
blob = f"{title} {text}"[:1200]
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
seen: list[str] = []
seen_l: set[str] = set()
for w in words:
wl = w.lower()
if wl in _FALLBACK_KW_STOP or wl in seen_l:
continue
seen.append(w)
seen_l.add(wl)
if len(seen) >= 5:
break
return t[:200], ", ".join(seen)
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
snippet = text[:MAX_AI_CHARS]
@@ -1120,23 +1366,31 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl
Върни САМО валиден JSON без markdown, без коментари."""
msg = client.messages.create(
model=AI_MODEL,
max_tokens=200,
messages=[{"role": "user", "content": prompt}]
)
raw = msg.content[0].text.strip()
raw = re.sub(r"^```[a-z]*\n?", "", raw)
raw = re.sub(r"\n?```$", "", raw)
try:
data = json.loads(raw)
t = str(data.get("title", title or "Секция"))[:200]
k = str(data.get("keywords", ""))[:300]
return t, k
except json.JSONDecodeError:
log.warning(f"AI върна невалиден JSON: {raw[:120]}")
return title or "Секция", ""
last_err: Optional[Exception] = None
tried: set[str] = set()
for model in AI_MODELS:
if not model or model in tried:
continue
tried.add(model)
try:
msg = client.messages.create(
model=model,
max_tokens=512,
messages=[{"role": "user", "content": prompt}],
)
raw = _content_text(msg)
parsed = _parse_classify_json(raw)
if parsed:
t, k = parsed
return (t or title or "Секция")[:200], k
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
except Exception as e:
last_err = e
log.warning(f"AI classify ({model}) неуспешен: {e}")
continue
if last_err:
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
return fallback_classify(title, text)
# ──────────────────────────────────────────────
@@ -1207,11 +1461,15 @@ def process_file(
saved = 0
codes: list[str] = []
ai_errors = 0
for sec in sections:
text = clean_text(sec.text)
html_text = sec.html_text or ""
if not text and not sec.images and not html_text:
continue
if (sec.title or "").strip():
text = sec.title.strip()
else:
continue
sec_index = saved + 1
code = make_code(prefix, file_index, sec_index)
@@ -1249,7 +1507,11 @@ def process_file(
title, keywords = classify_section(client, sec.title, text)
except Exception as e:
log.warning(f" AI грешка за {code}: {e}")
title, keywords = sec.title or f"Секция {sec_index}", ""
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
ai_errors += 1
if not (keywords or "").strip():
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
ai_errors += 1
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
ps = ProcessedSection(
@@ -1278,7 +1540,10 @@ def process_file(
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
log.info(f" → {saved} секции записани")
return _file_result(rel, file_index, codes, saved=saved)
result = _file_result(rel, file_index, codes, saved=saved)
result["ai_errors"] = ai_errors
result["keywords_ok"] = ai_errors == 0
return result
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")