diff --git a/help_codes.py b/help_codes.py index 8ea2ae1..0fa395c 100644 --- a/help_codes.py +++ b/help_codes.py @@ -1,4 +1,4 @@ -"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията.""" +"""Кодове на секции: PREFIX_0001_SEC_0001 и проверка на номерацията.""" from __future__ import annotations import re diff --git a/help_processor.py b/help_processor.py index f12dd90..8c6799a 100644 --- a/help_processor.py +++ b/help_processor.py @@ -71,19 +71,27 @@ try: except AttributeError: pass +_log_handlers = [logging.StreamHandler(sys.stdout)] +try: + _log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8")) +except OSError: + pass logging.basicConfig( level=logging.INFO, format="%(asctime)s %(levelname)-8s %(message)s", - handlers=[ - logging.StreamHandler(sys.stdout), - logging.FileHandler("help_processor.log", encoding="utf-8"), - ], + handlers=_log_handlers, ) log = logging.getLogger(__name__) -MIN_SECTION_TOKENS = 60 # секции под тази граница се сливат с предишната +MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация -AI_MODEL = "claude-haiku-4-5" +AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5") +AI_MODELS = [ + AI_MODEL, + "claude-haiku-4-5", + "claude-haiku-4-5-20251001", + "claude-3-5-haiku-latest", +] MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети) @@ -290,7 +298,7 @@ class Database: self.conn.commit() def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]: - """(code, source_file, output_path) за всички извлечения на файла.""" + """(code, source_file, output_path) за всички секции на файла.""" paths = self.matching_source_paths(prefix, identity) if not paths: return [] @@ -371,7 +379,7 @@ class Database: identity: str, output_dir: Optional[Path] = None, ) -> dict: - """Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001.""" + """Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001.""" rows = self.sections_for_file(prefix, identity) codes = [r[0] for r in rows] file_index = self.file_index_for(prefix, identity) @@ -548,6 +556,96 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6", _HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"}) _HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align", "valign", "width", "height", "bgcolor", "border") +_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3} +_HEADING_TOKEN_RE = re.compile( + r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)" + r"(\d+)?$", + re.I, +) +_HTML_HEADING_CLASS_RE = re.compile( + r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])", + re.I, +) +_HEADING_LEVEL = { + "heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3, + "title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3, + "заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3, + "подзаглавие": 2, "наименование": 1, + "überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3, +} + + +def _compact_style_token(s: str) -> str: + return re.sub(r"[\s_\-]+", "", (s or "").strip().lower()) + + +def _heading_level_from_token(token: str) -> Optional[int]: + t = _compact_style_token(token) + if not t: + return None + if t in _HEADING_LEVEL: + return _HEADING_LEVEL[t] + m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip()) + if not m: + return None + n = m.group(2) + if n and n.isdigit(): + return min(int(n), 3) + kind = (m.group(1) or "").lower() + if kind in ("subtitle", "подзаглавие"): + return 2 + return 1 + + +def _docx_heading_level(para) -> Optional[int]: + """Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word.""" + style = getattr(para, "style", None) + seen: set[int] = set() + cur = style + while cur is not None and id(cur) not in seen: + seen.add(id(cur)) + for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)): + lvl = _heading_level_from_token(str(token or "")) + if lvl: + return lvl + cur = getattr(cur, "base_style", None) + try: + pPr = para._element.pPr + if pPr is not None and pPr.outlineLvl is not None: + val = int(pPr.outlineLvl.val) + text = (para.text or "").strip() + if 0 <= val <= 2 and text and len(text) < 120: + return min(val + 1, 3) + except Exception: + pass + return None + + +def _is_bold_heading_text(text: str, runs) -> bool: + if not text or len(text) > 120: + return False + useful = [r for r in (runs or []) if (r.text or "").strip()] + return bool(useful) and all(bool(r.bold) for r in useful) + + +def _html_heading_level(el) -> Optional[int]: + name = (getattr(el, "name", None) or "").lower() + if name in _HTML_HEADING_MAP: + return _HTML_HEADING_MAP[name] + cls = " ".join(el.get("class") or []) if hasattr(el, "get") else "" + m = _HTML_HEADING_CLASS_RE.search(cls) + if m: + n = m.group(1) + return min(int(n), 3) if n and n.isdigit() else 1 + if name in ("p", "div"): + txt = el.get_text(" ", strip=True) + if txt and len(txt) < 120: + inner = "".join(el.stripped_strings) + strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else [] + strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip() + if strong and strong_txt and strong_txt == inner: + return 2 + return None def _html_block_plain_text(el) -> str: @@ -600,12 +698,13 @@ def parse_html(path: Path) -> list[Section]: base_dir = path.parent body = soup.body or soup - heading_map = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3} - # Събираме top-level блокови елементи (без да включваме вложените в тях) consumed = set() blocks = [] - for el in body.find_all(_HTML_BLOCK_TAGS + ["img"]): + for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]): + name = (el.name or "").lower() + if name == "div" and not _html_heading_level(el): + continue if any(id(par) in consumed for par in el.parents): continue consumed.add(id(el)) @@ -620,20 +719,21 @@ def parse_html(path: Path) -> list[Section]: img_counter = [0] def flush(): - if sec_text or sec_html or sec_images: + if current_title or sec_text or sec_html or sec_images: sec = Section(current_title, "\n".join(sec_text), current_level) sec.images = list(sec_images) sec.html_text = "\n".join(sec_html) if sec_html else None sections.append(sec) for el in blocks: - if el.name in heading_map: + heading_lvl = _html_heading_level(el) + if heading_lvl: txt = el.get_text(" ", strip=True) if not txt: continue flush() current_title = txt - current_level = heading_map[el.name] + current_level = heading_lvl sec_text, sec_html, sec_images = [], [], [] continue @@ -693,37 +793,74 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]: return imgs +def _iter_docx_blocks(doc): + """Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците).""" + from docx.oxml.ns import qn + from docx.table import Table + from docx.text.paragraph import Paragraph + + body = doc.element.body + for child in body.iterchildren(): + if child.tag == qn("w:p"): + yield "p", Paragraph(child, doc) + elif child.tag == qn("w:tbl"): + yield "tbl", Table(child, doc) + + +def _table_lines(table) -> list[str]: + lines: list[str] = [] + try: + rows = table.rows + except Exception: + return lines + for row in rows: + try: + cells = [" ".join((cell.text or "").split()) for cell in row.cells] + except Exception: + continue + cells = [c for c in cells if c] + if cells: + lines.append(" | ".join(cells)) + return lines + + def parse_docx(path: Path) -> list[Section]: doc = Document(path) sections: list[Section] = [] current_title, current_level = "", 1 buf: list[str] = [] sec_images: list[ImageRef] = [] - img_counter = [0] # списък за nonlocal-стил мутация - - HEADING_STYLES = {"heading 1": 1, "heading 2": 2, "heading 3": 3, - "title": 1, "subtitle": 2} + img_counter = [0] def flush(): - if buf or sec_images: + if current_title or buf or sec_images: sec = Section(current_title, "\n".join(buf), current_level) sec.images = list(sec_images) sections.append(sec) - for para in doc.paragraphs: + for kind, block in _iter_docx_blocks(doc): + if kind == "tbl": + for line in _table_lines(block): + buf.append(line) + continue + + para = block style_name = para.style.name.lower() if para.style else "" - text = para.text.strip() - para_imgs = _extract_docx_paragraph_images(para, doc) + text = para.text.strip() + para_imgs = _extract_docx_paragraph_images(para, doc) if not text and not para_imgs: continue - level = HEADING_STYLES.get(style_name) - is_bold_heading = bool(text and len(text) < 120 and not style_name.startswith("list") - and para.runs - and all(run.bold for run in para.runs if run.text.strip())) + level = _docx_heading_level(para) + is_bold_heading = ( + not level + and _is_bold_heading_text(text, para.runs) + and not style_name.startswith("list") + and not para_imgs + ) - if level or (is_bold_heading and not para_imgs): + if level or is_bold_heading: flush() buf, sec_images = [], [] current_title = text @@ -742,6 +879,11 @@ def parse_docx(path: Path) -> list[Section]: if not sections: fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip()) + if not fallback_text: + fallback_text = "\n".join( + line for kind, block in _iter_docx_blocks(doc) if kind == "tbl" + for line in _table_lines(block) + ) return [Section("", fallback_text, 0)] return sections @@ -986,7 +1128,7 @@ def parse_pdf(path: Path) -> list[Section]: prev_size = None def flush(): - if buf or sec_images: + if current_title or buf or sec_images: sec = Section(current_title, "\n".join(buf), 2) sec.images = list(sec_images) sections.append(sec) @@ -1055,7 +1197,39 @@ def parse_txt(path: Path) -> list[Section]: raw = path.read_bytes() enc = chardet.detect(raw)["encoding"] or "utf-8" text = raw.decode(enc, errors="replace") - return [Section("", text, 0)] + return _split_plain_text(text) + + +_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$") +_PLAIN_NUM_HEADING_RE = re.compile( + r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$" +) + + +def _split_plain_text(text: str) -> list[Section]: + """Markdown / номерирани заглавия; иначе една секция.""" + lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n") + sections: list[Section] = [] + title, level, buf = "", 0, [] + + def flush(): + body = "\n".join(buf).strip() + if title or body: + sections.append(Section(title, body, level)) + + for line in lines: + raw = line.strip() + md = _PLAIN_MD_HEADING_RE.match(raw) + numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120 + if md or numbered: + flush() + title = md.group(2).strip() if md else raw + level = len(md.group(1)) if md else 2 + buf = [] + continue + buf.append(line.rstrip()) + flush() + return sections or [Section("", text, 0)] PARSERS = { @@ -1073,15 +1247,16 @@ PARSERS = { # ────────────────────────────────────────────── def merge_short_sections(sections: list[Section]) -> list[Section]: - """Слива секции, по-кратки от MIN_SECTION_TOKENS думи, с предишната.""" + """Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция.""" result: list[Section] = [] for sec in sections: - words = len(sec.text.split()) - if result and words < MIN_SECTION_TOKENS: + words = len((sec.text or "").split()) + titled = bool((sec.title or "").strip()) + if result and not titled and words < MIN_SECTION_TOKENS: prev = result[-1] merged = Section( prev.title, - prev.text + "\n" + sec.text, + (prev.text + "\n" + sec.text).strip(), prev.level, ) merged.images = (prev.images or []) + (sec.images or []) @@ -1106,6 +1281,77 @@ def clean_text(text: str) -> str: # AI класификация # ────────────────────────────────────────────── +def _content_text(msg) -> str: + """Събира text блокове; Haiku 4.5 може да върне thinking като content[0].""" + parts: list[str] = [] + for block in getattr(msg, "content", None) or []: + btype = getattr(block, "type", None) + if btype in (None, "text"): + t = getattr(block, "text", None) + if t: + parts.append(str(t)) + elif isinstance(block, dict) and block.get("text"): + parts.append(str(block["text"])) + return "\n".join(parts).strip() + + +def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]: + raw = (raw or "").strip() + if not raw: + return None + raw = re.sub(r"^```[a-z]*\n?", "", raw) + raw = re.sub(r"\n?```$", "", raw) + candidates = [raw] + m = re.search(r"\{.*\}", raw, re.S) + if m: + candidates.append(m.group(0)) + for cand in candidates: + try: + data = json.loads(cand) + except json.JSONDecodeError: + continue + if not isinstance(data, dict): + continue + t = data.get("title", "") + k = data.get("keywords", "") + if isinstance(k, list): + k = ", ".join(str(x).strip() for x in k if str(x).strip()) + return str(t)[:200], str(k)[:300] + return None + + +_FALLBACK_KW_STOP = { + "и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до", + "при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to", + "of", "a", "an", "in", "on", "with", "this", "that", "секция", +} + + +def fallback_classify(title: str, text: str) -> tuple[str, str]: + t = (title or "").strip() + if not t: + for line in (text or "").splitlines(): + line = line.strip() + if 3 <= len(line) <= 80: + t = line + break + if not t: + t = "Секция" + blob = f"{title} {text}"[:1200] + words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob) + seen: list[str] = [] + seen_l: set[str] = set() + for w in words: + wl = w.lower() + if wl in _FALLBACK_KW_STOP or wl in seen_l: + continue + seen.append(w) + seen_l.add(wl) + if len(seen) >= 5: + break + return t[:200], ", ".join(seen) + + def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]: """Връща (наименование, 'кл1, кл2, кл3') чрез Claude.""" snippet = text[:MAX_AI_CHARS] @@ -1120,23 +1366,31 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl Върни САМО валиден JSON без markdown, без коментари.""" - msg = client.messages.create( - model=AI_MODEL, - max_tokens=200, - messages=[{"role": "user", "content": prompt}] - ) - raw = msg.content[0].text.strip() - raw = re.sub(r"^```[a-z]*\n?", "", raw) - raw = re.sub(r"\n?```$", "", raw) - - try: - data = json.loads(raw) - t = str(data.get("title", title or "Секция"))[:200] - k = str(data.get("keywords", ""))[:300] - return t, k - except json.JSONDecodeError: - log.warning(f"AI върна невалиден JSON: {raw[:120]}") - return title or "Секция", "" + last_err: Optional[Exception] = None + tried: set[str] = set() + for model in AI_MODELS: + if not model or model in tried: + continue + tried.add(model) + try: + msg = client.messages.create( + model=model, + max_tokens=512, + messages=[{"role": "user", "content": prompt}], + ) + raw = _content_text(msg) + parsed = _parse_classify_json(raw) + if parsed: + t, k = parsed + return (t or title or "Секция")[:200], k + last_err = ValueError(f"no JSON in model output: {raw[:120]!r}") + except Exception as e: + last_err = e + log.warning(f"AI classify ({model}) неуспешен: {e}") + continue + if last_err: + log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}") + return fallback_classify(title, text) # ────────────────────────────────────────────── @@ -1207,11 +1461,15 @@ def process_file( saved = 0 codes: list[str] = [] + ai_errors = 0 for sec in sections: text = clean_text(sec.text) html_text = sec.html_text or "" if not text and not sec.images and not html_text: - continue + if (sec.title or "").strip(): + text = sec.title.strip() + else: + continue sec_index = saved + 1 code = make_code(prefix, file_index, sec_index) @@ -1249,7 +1507,11 @@ def process_file( title, keywords = classify_section(client, sec.title, text) except Exception as e: log.warning(f" AI грешка за {code}: {e}") - title, keywords = sec.title or f"Секция {sec_index}", "" + title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text) + ai_errors += 1 + if not (keywords or "").strip(): + title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text) + ai_errors += 1 images_json = json.dumps(image_rel_paths, ensure_ascii=False) ps = ProcessedSection( @@ -1278,7 +1540,10 @@ def process_file( db.upsert_file(prefix, rel, fh, saved, file_index=file_index) log.info(f" → {saved} секции записани") - return _file_result(rel, file_index, codes, saved=saved) + result = _file_result(rel, file_index, codes, saved=saved) + result["ai_errors"] = ai_errors + result["keywords_ok"] = ai_errors == 0 + return result _PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$") diff --git a/webapp/main.py b/webapp/main.py index d88c7ca..677d444 100644 --- a/webapp/main.py +++ b/webapp/main.py @@ -392,7 +392,7 @@ def api_file_extractions( file: str = Query(..., min_length=1), prefix: str = Query("RIP"), ): - """Брой и номерация на извлеченията за даден help файл.""" + """Брой и номерация на секциите за даден help файл.""" name = _safe_help_filename(file) conn = db_conn() try: @@ -430,7 +430,7 @@ def api_delete_file_extractions( x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"), token: Optional[str] = Query(None), ): - """Изтрива всички извлечения за файла. Самият help файл не се пипа.""" + """Изтрива всички секции за файла. Самият help файл не се пипа.""" rescan_mod._check_token(x_rescan_token or token) name = _safe_help_filename(file) conn = db_conn() diff --git a/webapp/templates/viewer.html b/webapp/templates/viewer.html index cc84d39..b8bf349 100644 --- a/webapp/templates/viewer.html +++ b/webapp/templates/viewer.html @@ -274,6 +274,10 @@ } .rescan-filemeta:empty { display: none; } .rescan-filemeta.warn { color: var(--danger); } + .wipe-box { margin-top: 16px; } + .wipe-box input[type=text] { width: 100%; font-family: var(--mono); } + .src-file { cursor: pointer; } + .src-file:hover { color: var(--accent); text-decoration: underline; } .api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; } @media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } } @@ -360,11 +364,26 @@
+ Копирай името на файла от таб „Ключови думи“ и го постави тук. + Изтриват се всички секции за този файл (не самият help файл). +
+/api/rescan/{job_id}
queued / processing / done / error)./api/file-extractions?file=…&prefix=RIP
- ok ако SEC номерацията е уникална и последователна.ok ако SEC номерацията е уникална и последователна./api/file-extractions?file=…&prefix=RIP
- X-Rescan-Token.X-Rescan-Token.OpenAPI / Swagger: /docs · ReDoc: /redoc
Типичен ERP поток: GET /api/search → избор на code → GET /api/section/{code} за съдържание.