.
This commit is contained in:
@@ -71,19 +71,27 @@ try:
|
||||
except AttributeError:
|
||||
pass
|
||||
|
||||
_log_handlers = [logging.StreamHandler(sys.stdout)]
|
||||
try:
|
||||
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s %(levelname)-8s %(message)s",
|
||||
handlers=[
|
||||
logging.StreamHandler(sys.stdout),
|
||||
logging.FileHandler("help_processor.log", encoding="utf-8"),
|
||||
],
|
||||
handlers=_log_handlers,
|
||||
)
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
MIN_SECTION_TOKENS = 60 # секции под тази граница се сливат с предишната
|
||||
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
|
||||
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
|
||||
AI_MODEL = "claude-haiku-4-5"
|
||||
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
|
||||
AI_MODELS = [
|
||||
AI_MODEL,
|
||||
"claude-haiku-4-5",
|
||||
"claude-haiku-4-5-20251001",
|
||||
"claude-3-5-haiku-latest",
|
||||
]
|
||||
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
|
||||
|
||||
|
||||
@@ -290,7 +298,7 @@ class Database:
|
||||
self.conn.commit()
|
||||
|
||||
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
|
||||
"""(code, source_file, output_path) за всички извлечения на файла."""
|
||||
"""(code, source_file, output_path) за всички секции на файла."""
|
||||
paths = self.matching_source_paths(prefix, identity)
|
||||
if not paths:
|
||||
return []
|
||||
@@ -371,7 +379,7 @@ class Database:
|
||||
identity: str,
|
||||
output_dir: Optional[Path] = None,
|
||||
) -> dict:
|
||||
"""Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||||
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||||
rows = self.sections_for_file(prefix, identity)
|
||||
codes = [r[0] for r in rows]
|
||||
file_index = self.file_index_for(prefix, identity)
|
||||
@@ -548,6 +556,96 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
||||
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||||
"valign", "width", "height", "bgcolor", "border")
|
||||
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||||
_HEADING_TOKEN_RE = re.compile(
|
||||
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
|
||||
r"(\d+)?$",
|
||||
re.I,
|
||||
)
|
||||
_HTML_HEADING_CLASS_RE = re.compile(
|
||||
r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
|
||||
re.I,
|
||||
)
|
||||
_HEADING_LEVEL = {
|
||||
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
|
||||
"title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
|
||||
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
|
||||
"подзаглавие": 2, "наименование": 1,
|
||||
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
|
||||
}
|
||||
|
||||
|
||||
def _compact_style_token(s: str) -> str:
|
||||
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
|
||||
|
||||
|
||||
def _heading_level_from_token(token: str) -> Optional[int]:
|
||||
t = _compact_style_token(token)
|
||||
if not t:
|
||||
return None
|
||||
if t in _HEADING_LEVEL:
|
||||
return _HEADING_LEVEL[t]
|
||||
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
|
||||
if not m:
|
||||
return None
|
||||
n = m.group(2)
|
||||
if n and n.isdigit():
|
||||
return min(int(n), 3)
|
||||
kind = (m.group(1) or "").lower()
|
||||
if kind in ("subtitle", "подзаглавие"):
|
||||
return 2
|
||||
return 1
|
||||
|
||||
|
||||
def _docx_heading_level(para) -> Optional[int]:
|
||||
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word."""
|
||||
style = getattr(para, "style", None)
|
||||
seen: set[int] = set()
|
||||
cur = style
|
||||
while cur is not None and id(cur) not in seen:
|
||||
seen.add(id(cur))
|
||||
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
|
||||
lvl = _heading_level_from_token(str(token or ""))
|
||||
if lvl:
|
||||
return lvl
|
||||
cur = getattr(cur, "base_style", None)
|
||||
try:
|
||||
pPr = para._element.pPr
|
||||
if pPr is not None and pPr.outlineLvl is not None:
|
||||
val = int(pPr.outlineLvl.val)
|
||||
text = (para.text or "").strip()
|
||||
if 0 <= val <= 2 and text and len(text) < 120:
|
||||
return min(val + 1, 3)
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _is_bold_heading_text(text: str, runs) -> bool:
|
||||
if not text or len(text) > 120:
|
||||
return False
|
||||
useful = [r for r in (runs or []) if (r.text or "").strip()]
|
||||
return bool(useful) and all(bool(r.bold) for r in useful)
|
||||
|
||||
|
||||
def _html_heading_level(el) -> Optional[int]:
|
||||
name = (getattr(el, "name", None) or "").lower()
|
||||
if name in _HTML_HEADING_MAP:
|
||||
return _HTML_HEADING_MAP[name]
|
||||
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
|
||||
m = _HTML_HEADING_CLASS_RE.search(cls)
|
||||
if m:
|
||||
n = m.group(1)
|
||||
return min(int(n), 3) if n and n.isdigit() else 1
|
||||
if name in ("p", "div"):
|
||||
txt = el.get_text(" ", strip=True)
|
||||
if txt and len(txt) < 120:
|
||||
inner = "".join(el.stripped_strings)
|
||||
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
|
||||
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
|
||||
if strong and strong_txt and strong_txt == inner:
|
||||
return 2
|
||||
return None
|
||||
|
||||
|
||||
def _html_block_plain_text(el) -> str:
|
||||
@@ -600,12 +698,13 @@ def parse_html(path: Path) -> list[Section]:
|
||||
base_dir = path.parent
|
||||
body = soup.body or soup
|
||||
|
||||
heading_map = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||||
|
||||
# Събираме top-level блокови елементи (без да включваме вложените в тях)
|
||||
consumed = set()
|
||||
blocks = []
|
||||
for el in body.find_all(_HTML_BLOCK_TAGS + ["img"]):
|
||||
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
|
||||
name = (el.name or "").lower()
|
||||
if name == "div" and not _html_heading_level(el):
|
||||
continue
|
||||
if any(id(par) in consumed for par in el.parents):
|
||||
continue
|
||||
consumed.add(id(el))
|
||||
@@ -620,20 +719,21 @@ def parse_html(path: Path) -> list[Section]:
|
||||
img_counter = [0]
|
||||
|
||||
def flush():
|
||||
if sec_text or sec_html or sec_images:
|
||||
if current_title or sec_text or sec_html or sec_images:
|
||||
sec = Section(current_title, "\n".join(sec_text), current_level)
|
||||
sec.images = list(sec_images)
|
||||
sec.html_text = "\n".join(sec_html) if sec_html else None
|
||||
sections.append(sec)
|
||||
|
||||
for el in blocks:
|
||||
if el.name in heading_map:
|
||||
heading_lvl = _html_heading_level(el)
|
||||
if heading_lvl:
|
||||
txt = el.get_text(" ", strip=True)
|
||||
if not txt:
|
||||
continue
|
||||
flush()
|
||||
current_title = txt
|
||||
current_level = heading_map[el.name]
|
||||
current_level = heading_lvl
|
||||
sec_text, sec_html, sec_images = [], [], []
|
||||
continue
|
||||
|
||||
@@ -693,37 +793,74 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||
return imgs
|
||||
|
||||
|
||||
def _iter_docx_blocks(doc):
|
||||
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
|
||||
from docx.oxml.ns import qn
|
||||
from docx.table import Table
|
||||
from docx.text.paragraph import Paragraph
|
||||
|
||||
body = doc.element.body
|
||||
for child in body.iterchildren():
|
||||
if child.tag == qn("w:p"):
|
||||
yield "p", Paragraph(child, doc)
|
||||
elif child.tag == qn("w:tbl"):
|
||||
yield "tbl", Table(child, doc)
|
||||
|
||||
|
||||
def _table_lines(table) -> list[str]:
|
||||
lines: list[str] = []
|
||||
try:
|
||||
rows = table.rows
|
||||
except Exception:
|
||||
return lines
|
||||
for row in rows:
|
||||
try:
|
||||
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
|
||||
except Exception:
|
||||
continue
|
||||
cells = [c for c in cells if c]
|
||||
if cells:
|
||||
lines.append(" | ".join(cells))
|
||||
return lines
|
||||
|
||||
|
||||
def parse_docx(path: Path) -> list[Section]:
|
||||
doc = Document(path)
|
||||
sections: list[Section] = []
|
||||
current_title, current_level = "", 1
|
||||
buf: list[str] = []
|
||||
sec_images: list[ImageRef] = []
|
||||
img_counter = [0] # списък за nonlocal-стил мутация
|
||||
|
||||
HEADING_STYLES = {"heading 1": 1, "heading 2": 2, "heading 3": 3,
|
||||
"title": 1, "subtitle": 2}
|
||||
img_counter = [0]
|
||||
|
||||
def flush():
|
||||
if buf or sec_images:
|
||||
if current_title or buf or sec_images:
|
||||
sec = Section(current_title, "\n".join(buf), current_level)
|
||||
sec.images = list(sec_images)
|
||||
sections.append(sec)
|
||||
|
||||
for para in doc.paragraphs:
|
||||
for kind, block in _iter_docx_blocks(doc):
|
||||
if kind == "tbl":
|
||||
for line in _table_lines(block):
|
||||
buf.append(line)
|
||||
continue
|
||||
|
||||
para = block
|
||||
style_name = para.style.name.lower() if para.style else ""
|
||||
text = para.text.strip()
|
||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||||
text = para.text.strip()
|
||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||||
|
||||
if not text and not para_imgs:
|
||||
continue
|
||||
|
||||
level = HEADING_STYLES.get(style_name)
|
||||
is_bold_heading = bool(text and len(text) < 120 and not style_name.startswith("list")
|
||||
and para.runs
|
||||
and all(run.bold for run in para.runs if run.text.strip()))
|
||||
level = _docx_heading_level(para)
|
||||
is_bold_heading = (
|
||||
not level
|
||||
and _is_bold_heading_text(text, para.runs)
|
||||
and not style_name.startswith("list")
|
||||
and not para_imgs
|
||||
)
|
||||
|
||||
if level or (is_bold_heading and not para_imgs):
|
||||
if level or is_bold_heading:
|
||||
flush()
|
||||
buf, sec_images = [], []
|
||||
current_title = text
|
||||
@@ -742,6 +879,11 @@ def parse_docx(path: Path) -> list[Section]:
|
||||
|
||||
if not sections:
|
||||
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
|
||||
if not fallback_text:
|
||||
fallback_text = "\n".join(
|
||||
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
|
||||
for line in _table_lines(block)
|
||||
)
|
||||
return [Section("", fallback_text, 0)]
|
||||
return sections
|
||||
|
||||
@@ -986,7 +1128,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
||||
prev_size = None
|
||||
|
||||
def flush():
|
||||
if buf or sec_images:
|
||||
if current_title or buf or sec_images:
|
||||
sec = Section(current_title, "\n".join(buf), 2)
|
||||
sec.images = list(sec_images)
|
||||
sections.append(sec)
|
||||
@@ -1055,7 +1197,39 @@ def parse_txt(path: Path) -> list[Section]:
|
||||
raw = path.read_bytes()
|
||||
enc = chardet.detect(raw)["encoding"] or "utf-8"
|
||||
text = raw.decode(enc, errors="replace")
|
||||
return [Section("", text, 0)]
|
||||
return _split_plain_text(text)
|
||||
|
||||
|
||||
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
|
||||
_PLAIN_NUM_HEADING_RE = re.compile(
|
||||
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
|
||||
)
|
||||
|
||||
|
||||
def _split_plain_text(text: str) -> list[Section]:
|
||||
"""Markdown / номерирани заглавия; иначе една секция."""
|
||||
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
||||
sections: list[Section] = []
|
||||
title, level, buf = "", 0, []
|
||||
|
||||
def flush():
|
||||
body = "\n".join(buf).strip()
|
||||
if title or body:
|
||||
sections.append(Section(title, body, level))
|
||||
|
||||
for line in lines:
|
||||
raw = line.strip()
|
||||
md = _PLAIN_MD_HEADING_RE.match(raw)
|
||||
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
|
||||
if md or numbered:
|
||||
flush()
|
||||
title = md.group(2).strip() if md else raw
|
||||
level = len(md.group(1)) if md else 2
|
||||
buf = []
|
||||
continue
|
||||
buf.append(line.rstrip())
|
||||
flush()
|
||||
return sections or [Section("", text, 0)]
|
||||
|
||||
|
||||
PARSERS = {
|
||||
@@ -1073,15 +1247,16 @@ PARSERS = {
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def merge_short_sections(sections: list[Section]) -> list[Section]:
|
||||
"""Слива секции, по-кратки от MIN_SECTION_TOKENS думи, с предишната."""
|
||||
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
|
||||
result: list[Section] = []
|
||||
for sec in sections:
|
||||
words = len(sec.text.split())
|
||||
if result and words < MIN_SECTION_TOKENS:
|
||||
words = len((sec.text or "").split())
|
||||
titled = bool((sec.title or "").strip())
|
||||
if result and not titled and words < MIN_SECTION_TOKENS:
|
||||
prev = result[-1]
|
||||
merged = Section(
|
||||
prev.title,
|
||||
prev.text + "\n" + sec.text,
|
||||
(prev.text + "\n" + sec.text).strip(),
|
||||
prev.level,
|
||||
)
|
||||
merged.images = (prev.images or []) + (sec.images or [])
|
||||
@@ -1106,6 +1281,77 @@ def clean_text(text: str) -> str:
|
||||
# AI класификация
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def _content_text(msg) -> str:
|
||||
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
|
||||
parts: list[str] = []
|
||||
for block in getattr(msg, "content", None) or []:
|
||||
btype = getattr(block, "type", None)
|
||||
if btype in (None, "text"):
|
||||
t = getattr(block, "text", None)
|
||||
if t:
|
||||
parts.append(str(t))
|
||||
elif isinstance(block, dict) and block.get("text"):
|
||||
parts.append(str(block["text"]))
|
||||
return "\n".join(parts).strip()
|
||||
|
||||
|
||||
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
|
||||
raw = (raw or "").strip()
|
||||
if not raw:
|
||||
return None
|
||||
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
||||
raw = re.sub(r"\n?```$", "", raw)
|
||||
candidates = [raw]
|
||||
m = re.search(r"\{.*\}", raw, re.S)
|
||||
if m:
|
||||
candidates.append(m.group(0))
|
||||
for cand in candidates:
|
||||
try:
|
||||
data = json.loads(cand)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if not isinstance(data, dict):
|
||||
continue
|
||||
t = data.get("title", "")
|
||||
k = data.get("keywords", "")
|
||||
if isinstance(k, list):
|
||||
k = ", ".join(str(x).strip() for x in k if str(x).strip())
|
||||
return str(t)[:200], str(k)[:300]
|
||||
return None
|
||||
|
||||
|
||||
_FALLBACK_KW_STOP = {
|
||||
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
|
||||
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
|
||||
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
|
||||
}
|
||||
|
||||
|
||||
def fallback_classify(title: str, text: str) -> tuple[str, str]:
|
||||
t = (title or "").strip()
|
||||
if not t:
|
||||
for line in (text or "").splitlines():
|
||||
line = line.strip()
|
||||
if 3 <= len(line) <= 80:
|
||||
t = line
|
||||
break
|
||||
if not t:
|
||||
t = "Секция"
|
||||
blob = f"{title} {text}"[:1200]
|
||||
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
|
||||
seen: list[str] = []
|
||||
seen_l: set[str] = set()
|
||||
for w in words:
|
||||
wl = w.lower()
|
||||
if wl in _FALLBACK_KW_STOP or wl in seen_l:
|
||||
continue
|
||||
seen.append(w)
|
||||
seen_l.add(wl)
|
||||
if len(seen) >= 5:
|
||||
break
|
||||
return t[:200], ", ".join(seen)
|
||||
|
||||
|
||||
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
||||
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
||||
snippet = text[:MAX_AI_CHARS]
|
||||
@@ -1120,23 +1366,31 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl
|
||||
|
||||
Върни САМО валиден JSON без markdown, без коментари."""
|
||||
|
||||
msg = client.messages.create(
|
||||
model=AI_MODEL,
|
||||
max_tokens=200,
|
||||
messages=[{"role": "user", "content": prompt}]
|
||||
)
|
||||
raw = msg.content[0].text.strip()
|
||||
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
||||
raw = re.sub(r"\n?```$", "", raw)
|
||||
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
t = str(data.get("title", title or "Секция"))[:200]
|
||||
k = str(data.get("keywords", ""))[:300]
|
||||
return t, k
|
||||
except json.JSONDecodeError:
|
||||
log.warning(f"AI върна невалиден JSON: {raw[:120]}")
|
||||
return title or "Секция", ""
|
||||
last_err: Optional[Exception] = None
|
||||
tried: set[str] = set()
|
||||
for model in AI_MODELS:
|
||||
if not model or model in tried:
|
||||
continue
|
||||
tried.add(model)
|
||||
try:
|
||||
msg = client.messages.create(
|
||||
model=model,
|
||||
max_tokens=512,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
)
|
||||
raw = _content_text(msg)
|
||||
parsed = _parse_classify_json(raw)
|
||||
if parsed:
|
||||
t, k = parsed
|
||||
return (t or title or "Секция")[:200], k
|
||||
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
|
||||
except Exception as e:
|
||||
last_err = e
|
||||
log.warning(f"AI classify ({model}) неуспешен: {e}")
|
||||
continue
|
||||
if last_err:
|
||||
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
|
||||
return fallback_classify(title, text)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
@@ -1207,11 +1461,15 @@ def process_file(
|
||||
|
||||
saved = 0
|
||||
codes: list[str] = []
|
||||
ai_errors = 0
|
||||
for sec in sections:
|
||||
text = clean_text(sec.text)
|
||||
html_text = sec.html_text or ""
|
||||
if not text and not sec.images and not html_text:
|
||||
continue
|
||||
if (sec.title or "").strip():
|
||||
text = sec.title.strip()
|
||||
else:
|
||||
continue
|
||||
|
||||
sec_index = saved + 1
|
||||
code = make_code(prefix, file_index, sec_index)
|
||||
@@ -1249,7 +1507,11 @@ def process_file(
|
||||
title, keywords = classify_section(client, sec.title, text)
|
||||
except Exception as e:
|
||||
log.warning(f" AI грешка за {code}: {e}")
|
||||
title, keywords = sec.title or f"Секция {sec_index}", ""
|
||||
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
|
||||
ai_errors += 1
|
||||
if not (keywords or "").strip():
|
||||
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
|
||||
ai_errors += 1
|
||||
|
||||
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
||||
ps = ProcessedSection(
|
||||
@@ -1278,7 +1540,10 @@ def process_file(
|
||||
|
||||
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
|
||||
log.info(f" → {saved} секции записани")
|
||||
return _file_result(rel, file_index, codes, saved=saved)
|
||||
result = _file_result(rel, file_index, codes, saved=saved)
|
||||
result["ai_errors"] = ai_errors
|
||||
result["keywords_ok"] = ai_errors == 0
|
||||
return result
|
||||
|
||||
|
||||
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
|
||||
|
||||
Reference in New Issue
Block a user