.
This commit is contained in:
@@ -1,4 +1,4 @@
|
||||
"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията."""
|
||||
"""Кодове на секции: PREFIX_0001_SEC_0001 и проверка на номерацията."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
@@ -71,19 +71,27 @@ try:
|
||||
except AttributeError:
|
||||
pass
|
||||
|
||||
_log_handlers = [logging.StreamHandler(sys.stdout)]
|
||||
try:
|
||||
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
|
||||
except OSError:
|
||||
pass
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="%(asctime)s %(levelname)-8s %(message)s",
|
||||
handlers=[
|
||||
logging.StreamHandler(sys.stdout),
|
||||
logging.FileHandler("help_processor.log", encoding="utf-8"),
|
||||
],
|
||||
handlers=_log_handlers,
|
||||
)
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
MIN_SECTION_TOKENS = 60 # секции под тази граница се сливат с предишната
|
||||
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
|
||||
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
|
||||
AI_MODEL = "claude-haiku-4-5"
|
||||
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
|
||||
AI_MODELS = [
|
||||
AI_MODEL,
|
||||
"claude-haiku-4-5",
|
||||
"claude-haiku-4-5-20251001",
|
||||
"claude-3-5-haiku-latest",
|
||||
]
|
||||
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
|
||||
|
||||
|
||||
@@ -290,7 +298,7 @@ class Database:
|
||||
self.conn.commit()
|
||||
|
||||
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
|
||||
"""(code, source_file, output_path) за всички извлечения на файла."""
|
||||
"""(code, source_file, output_path) за всички секции на файла."""
|
||||
paths = self.matching_source_paths(prefix, identity)
|
||||
if not paths:
|
||||
return []
|
||||
@@ -371,7 +379,7 @@ class Database:
|
||||
identity: str,
|
||||
output_dir: Optional[Path] = None,
|
||||
) -> dict:
|
||||
"""Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||||
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||||
rows = self.sections_for_file(prefix, identity)
|
||||
codes = [r[0] for r in rows]
|
||||
file_index = self.file_index_for(prefix, identity)
|
||||
@@ -548,6 +556,96 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
||||
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||||
"valign", "width", "height", "bgcolor", "border")
|
||||
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||||
_HEADING_TOKEN_RE = re.compile(
|
||||
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
|
||||
r"(\d+)?$",
|
||||
re.I,
|
||||
)
|
||||
_HTML_HEADING_CLASS_RE = re.compile(
|
||||
r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
|
||||
re.I,
|
||||
)
|
||||
_HEADING_LEVEL = {
|
||||
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
|
||||
"title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
|
||||
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
|
||||
"подзаглавие": 2, "наименование": 1,
|
||||
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
|
||||
}
|
||||
|
||||
|
||||
def _compact_style_token(s: str) -> str:
|
||||
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
|
||||
|
||||
|
||||
def _heading_level_from_token(token: str) -> Optional[int]:
|
||||
t = _compact_style_token(token)
|
||||
if not t:
|
||||
return None
|
||||
if t in _HEADING_LEVEL:
|
||||
return _HEADING_LEVEL[t]
|
||||
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
|
||||
if not m:
|
||||
return None
|
||||
n = m.group(2)
|
||||
if n and n.isdigit():
|
||||
return min(int(n), 3)
|
||||
kind = (m.group(1) or "").lower()
|
||||
if kind in ("subtitle", "подзаглавие"):
|
||||
return 2
|
||||
return 1
|
||||
|
||||
|
||||
def _docx_heading_level(para) -> Optional[int]:
|
||||
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word."""
|
||||
style = getattr(para, "style", None)
|
||||
seen: set[int] = set()
|
||||
cur = style
|
||||
while cur is not None and id(cur) not in seen:
|
||||
seen.add(id(cur))
|
||||
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
|
||||
lvl = _heading_level_from_token(str(token or ""))
|
||||
if lvl:
|
||||
return lvl
|
||||
cur = getattr(cur, "base_style", None)
|
||||
try:
|
||||
pPr = para._element.pPr
|
||||
if pPr is not None and pPr.outlineLvl is not None:
|
||||
val = int(pPr.outlineLvl.val)
|
||||
text = (para.text or "").strip()
|
||||
if 0 <= val <= 2 and text and len(text) < 120:
|
||||
return min(val + 1, 3)
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def _is_bold_heading_text(text: str, runs) -> bool:
|
||||
if not text or len(text) > 120:
|
||||
return False
|
||||
useful = [r for r in (runs or []) if (r.text or "").strip()]
|
||||
return bool(useful) and all(bool(r.bold) for r in useful)
|
||||
|
||||
|
||||
def _html_heading_level(el) -> Optional[int]:
|
||||
name = (getattr(el, "name", None) or "").lower()
|
||||
if name in _HTML_HEADING_MAP:
|
||||
return _HTML_HEADING_MAP[name]
|
||||
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
|
||||
m = _HTML_HEADING_CLASS_RE.search(cls)
|
||||
if m:
|
||||
n = m.group(1)
|
||||
return min(int(n), 3) if n and n.isdigit() else 1
|
||||
if name in ("p", "div"):
|
||||
txt = el.get_text(" ", strip=True)
|
||||
if txt and len(txt) < 120:
|
||||
inner = "".join(el.stripped_strings)
|
||||
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
|
||||
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
|
||||
if strong and strong_txt and strong_txt == inner:
|
||||
return 2
|
||||
return None
|
||||
|
||||
|
||||
def _html_block_plain_text(el) -> str:
|
||||
@@ -600,12 +698,13 @@ def parse_html(path: Path) -> list[Section]:
|
||||
base_dir = path.parent
|
||||
body = soup.body or soup
|
||||
|
||||
heading_map = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||||
|
||||
# Събираме top-level блокови елементи (без да включваме вложените в тях)
|
||||
consumed = set()
|
||||
blocks = []
|
||||
for el in body.find_all(_HTML_BLOCK_TAGS + ["img"]):
|
||||
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
|
||||
name = (el.name or "").lower()
|
||||
if name == "div" and not _html_heading_level(el):
|
||||
continue
|
||||
if any(id(par) in consumed for par in el.parents):
|
||||
continue
|
||||
consumed.add(id(el))
|
||||
@@ -620,20 +719,21 @@ def parse_html(path: Path) -> list[Section]:
|
||||
img_counter = [0]
|
||||
|
||||
def flush():
|
||||
if sec_text or sec_html or sec_images:
|
||||
if current_title or sec_text or sec_html or sec_images:
|
||||
sec = Section(current_title, "\n".join(sec_text), current_level)
|
||||
sec.images = list(sec_images)
|
||||
sec.html_text = "\n".join(sec_html) if sec_html else None
|
||||
sections.append(sec)
|
||||
|
||||
for el in blocks:
|
||||
if el.name in heading_map:
|
||||
heading_lvl = _html_heading_level(el)
|
||||
if heading_lvl:
|
||||
txt = el.get_text(" ", strip=True)
|
||||
if not txt:
|
||||
continue
|
||||
flush()
|
||||
current_title = txt
|
||||
current_level = heading_map[el.name]
|
||||
current_level = heading_lvl
|
||||
sec_text, sec_html, sec_images = [], [], []
|
||||
continue
|
||||
|
||||
@@ -693,37 +793,74 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||
return imgs
|
||||
|
||||
|
||||
def _iter_docx_blocks(doc):
|
||||
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
|
||||
from docx.oxml.ns import qn
|
||||
from docx.table import Table
|
||||
from docx.text.paragraph import Paragraph
|
||||
|
||||
body = doc.element.body
|
||||
for child in body.iterchildren():
|
||||
if child.tag == qn("w:p"):
|
||||
yield "p", Paragraph(child, doc)
|
||||
elif child.tag == qn("w:tbl"):
|
||||
yield "tbl", Table(child, doc)
|
||||
|
||||
|
||||
def _table_lines(table) -> list[str]:
|
||||
lines: list[str] = []
|
||||
try:
|
||||
rows = table.rows
|
||||
except Exception:
|
||||
return lines
|
||||
for row in rows:
|
||||
try:
|
||||
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
|
||||
except Exception:
|
||||
continue
|
||||
cells = [c for c in cells if c]
|
||||
if cells:
|
||||
lines.append(" | ".join(cells))
|
||||
return lines
|
||||
|
||||
|
||||
def parse_docx(path: Path) -> list[Section]:
|
||||
doc = Document(path)
|
||||
sections: list[Section] = []
|
||||
current_title, current_level = "", 1
|
||||
buf: list[str] = []
|
||||
sec_images: list[ImageRef] = []
|
||||
img_counter = [0] # списък за nonlocal-стил мутация
|
||||
|
||||
HEADING_STYLES = {"heading 1": 1, "heading 2": 2, "heading 3": 3,
|
||||
"title": 1, "subtitle": 2}
|
||||
img_counter = [0]
|
||||
|
||||
def flush():
|
||||
if buf or sec_images:
|
||||
if current_title or buf or sec_images:
|
||||
sec = Section(current_title, "\n".join(buf), current_level)
|
||||
sec.images = list(sec_images)
|
||||
sections.append(sec)
|
||||
|
||||
for para in doc.paragraphs:
|
||||
for kind, block in _iter_docx_blocks(doc):
|
||||
if kind == "tbl":
|
||||
for line in _table_lines(block):
|
||||
buf.append(line)
|
||||
continue
|
||||
|
||||
para = block
|
||||
style_name = para.style.name.lower() if para.style else ""
|
||||
text = para.text.strip()
|
||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||||
text = para.text.strip()
|
||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||||
|
||||
if not text and not para_imgs:
|
||||
continue
|
||||
|
||||
level = HEADING_STYLES.get(style_name)
|
||||
is_bold_heading = bool(text and len(text) < 120 and not style_name.startswith("list")
|
||||
and para.runs
|
||||
and all(run.bold for run in para.runs if run.text.strip()))
|
||||
level = _docx_heading_level(para)
|
||||
is_bold_heading = (
|
||||
not level
|
||||
and _is_bold_heading_text(text, para.runs)
|
||||
and not style_name.startswith("list")
|
||||
and not para_imgs
|
||||
)
|
||||
|
||||
if level or (is_bold_heading and not para_imgs):
|
||||
if level or is_bold_heading:
|
||||
flush()
|
||||
buf, sec_images = [], []
|
||||
current_title = text
|
||||
@@ -742,6 +879,11 @@ def parse_docx(path: Path) -> list[Section]:
|
||||
|
||||
if not sections:
|
||||
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
|
||||
if not fallback_text:
|
||||
fallback_text = "\n".join(
|
||||
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
|
||||
for line in _table_lines(block)
|
||||
)
|
||||
return [Section("", fallback_text, 0)]
|
||||
return sections
|
||||
|
||||
@@ -986,7 +1128,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
||||
prev_size = None
|
||||
|
||||
def flush():
|
||||
if buf or sec_images:
|
||||
if current_title or buf or sec_images:
|
||||
sec = Section(current_title, "\n".join(buf), 2)
|
||||
sec.images = list(sec_images)
|
||||
sections.append(sec)
|
||||
@@ -1055,7 +1197,39 @@ def parse_txt(path: Path) -> list[Section]:
|
||||
raw = path.read_bytes()
|
||||
enc = chardet.detect(raw)["encoding"] or "utf-8"
|
||||
text = raw.decode(enc, errors="replace")
|
||||
return [Section("", text, 0)]
|
||||
return _split_plain_text(text)
|
||||
|
||||
|
||||
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
|
||||
_PLAIN_NUM_HEADING_RE = re.compile(
|
||||
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
|
||||
)
|
||||
|
||||
|
||||
def _split_plain_text(text: str) -> list[Section]:
|
||||
"""Markdown / номерирани заглавия; иначе една секция."""
|
||||
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
||||
sections: list[Section] = []
|
||||
title, level, buf = "", 0, []
|
||||
|
||||
def flush():
|
||||
body = "\n".join(buf).strip()
|
||||
if title or body:
|
||||
sections.append(Section(title, body, level))
|
||||
|
||||
for line in lines:
|
||||
raw = line.strip()
|
||||
md = _PLAIN_MD_HEADING_RE.match(raw)
|
||||
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
|
||||
if md or numbered:
|
||||
flush()
|
||||
title = md.group(2).strip() if md else raw
|
||||
level = len(md.group(1)) if md else 2
|
||||
buf = []
|
||||
continue
|
||||
buf.append(line.rstrip())
|
||||
flush()
|
||||
return sections or [Section("", text, 0)]
|
||||
|
||||
|
||||
PARSERS = {
|
||||
@@ -1073,15 +1247,16 @@ PARSERS = {
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def merge_short_sections(sections: list[Section]) -> list[Section]:
|
||||
"""Слива секции, по-кратки от MIN_SECTION_TOKENS думи, с предишната."""
|
||||
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
|
||||
result: list[Section] = []
|
||||
for sec in sections:
|
||||
words = len(sec.text.split())
|
||||
if result and words < MIN_SECTION_TOKENS:
|
||||
words = len((sec.text or "").split())
|
||||
titled = bool((sec.title or "").strip())
|
||||
if result and not titled and words < MIN_SECTION_TOKENS:
|
||||
prev = result[-1]
|
||||
merged = Section(
|
||||
prev.title,
|
||||
prev.text + "\n" + sec.text,
|
||||
(prev.text + "\n" + sec.text).strip(),
|
||||
prev.level,
|
||||
)
|
||||
merged.images = (prev.images or []) + (sec.images or [])
|
||||
@@ -1106,6 +1281,77 @@ def clean_text(text: str) -> str:
|
||||
# AI класификация
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def _content_text(msg) -> str:
|
||||
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
|
||||
parts: list[str] = []
|
||||
for block in getattr(msg, "content", None) or []:
|
||||
btype = getattr(block, "type", None)
|
||||
if btype in (None, "text"):
|
||||
t = getattr(block, "text", None)
|
||||
if t:
|
||||
parts.append(str(t))
|
||||
elif isinstance(block, dict) and block.get("text"):
|
||||
parts.append(str(block["text"]))
|
||||
return "\n".join(parts).strip()
|
||||
|
||||
|
||||
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
|
||||
raw = (raw or "").strip()
|
||||
if not raw:
|
||||
return None
|
||||
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
||||
raw = re.sub(r"\n?```$", "", raw)
|
||||
candidates = [raw]
|
||||
m = re.search(r"\{.*\}", raw, re.S)
|
||||
if m:
|
||||
candidates.append(m.group(0))
|
||||
for cand in candidates:
|
||||
try:
|
||||
data = json.loads(cand)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
if not isinstance(data, dict):
|
||||
continue
|
||||
t = data.get("title", "")
|
||||
k = data.get("keywords", "")
|
||||
if isinstance(k, list):
|
||||
k = ", ".join(str(x).strip() for x in k if str(x).strip())
|
||||
return str(t)[:200], str(k)[:300]
|
||||
return None
|
||||
|
||||
|
||||
_FALLBACK_KW_STOP = {
|
||||
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
|
||||
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
|
||||
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
|
||||
}
|
||||
|
||||
|
||||
def fallback_classify(title: str, text: str) -> tuple[str, str]:
|
||||
t = (title or "").strip()
|
||||
if not t:
|
||||
for line in (text or "").splitlines():
|
||||
line = line.strip()
|
||||
if 3 <= len(line) <= 80:
|
||||
t = line
|
||||
break
|
||||
if not t:
|
||||
t = "Секция"
|
||||
blob = f"{title} {text}"[:1200]
|
||||
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
|
||||
seen: list[str] = []
|
||||
seen_l: set[str] = set()
|
||||
for w in words:
|
||||
wl = w.lower()
|
||||
if wl in _FALLBACK_KW_STOP or wl in seen_l:
|
||||
continue
|
||||
seen.append(w)
|
||||
seen_l.add(wl)
|
||||
if len(seen) >= 5:
|
||||
break
|
||||
return t[:200], ", ".join(seen)
|
||||
|
||||
|
||||
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
||||
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
||||
snippet = text[:MAX_AI_CHARS]
|
||||
@@ -1120,23 +1366,31 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl
|
||||
|
||||
Върни САМО валиден JSON без markdown, без коментари."""
|
||||
|
||||
msg = client.messages.create(
|
||||
model=AI_MODEL,
|
||||
max_tokens=200,
|
||||
messages=[{"role": "user", "content": prompt}]
|
||||
)
|
||||
raw = msg.content[0].text.strip()
|
||||
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
||||
raw = re.sub(r"\n?```$", "", raw)
|
||||
|
||||
try:
|
||||
data = json.loads(raw)
|
||||
t = str(data.get("title", title or "Секция"))[:200]
|
||||
k = str(data.get("keywords", ""))[:300]
|
||||
return t, k
|
||||
except json.JSONDecodeError:
|
||||
log.warning(f"AI върна невалиден JSON: {raw[:120]}")
|
||||
return title or "Секция", ""
|
||||
last_err: Optional[Exception] = None
|
||||
tried: set[str] = set()
|
||||
for model in AI_MODELS:
|
||||
if not model or model in tried:
|
||||
continue
|
||||
tried.add(model)
|
||||
try:
|
||||
msg = client.messages.create(
|
||||
model=model,
|
||||
max_tokens=512,
|
||||
messages=[{"role": "user", "content": prompt}],
|
||||
)
|
||||
raw = _content_text(msg)
|
||||
parsed = _parse_classify_json(raw)
|
||||
if parsed:
|
||||
t, k = parsed
|
||||
return (t or title or "Секция")[:200], k
|
||||
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
|
||||
except Exception as e:
|
||||
last_err = e
|
||||
log.warning(f"AI classify ({model}) неуспешен: {e}")
|
||||
continue
|
||||
if last_err:
|
||||
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
|
||||
return fallback_classify(title, text)
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
@@ -1207,11 +1461,15 @@ def process_file(
|
||||
|
||||
saved = 0
|
||||
codes: list[str] = []
|
||||
ai_errors = 0
|
||||
for sec in sections:
|
||||
text = clean_text(sec.text)
|
||||
html_text = sec.html_text or ""
|
||||
if not text and not sec.images and not html_text:
|
||||
continue
|
||||
if (sec.title or "").strip():
|
||||
text = sec.title.strip()
|
||||
else:
|
||||
continue
|
||||
|
||||
sec_index = saved + 1
|
||||
code = make_code(prefix, file_index, sec_index)
|
||||
@@ -1249,7 +1507,11 @@ def process_file(
|
||||
title, keywords = classify_section(client, sec.title, text)
|
||||
except Exception as e:
|
||||
log.warning(f" AI грешка за {code}: {e}")
|
||||
title, keywords = sec.title or f"Секция {sec_index}", ""
|
||||
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
|
||||
ai_errors += 1
|
||||
if not (keywords or "").strip():
|
||||
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
|
||||
ai_errors += 1
|
||||
|
||||
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
||||
ps = ProcessedSection(
|
||||
@@ -1278,7 +1540,10 @@ def process_file(
|
||||
|
||||
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
|
||||
log.info(f" → {saved} секции записани")
|
||||
return _file_result(rel, file_index, codes, saved=saved)
|
||||
result = _file_result(rel, file_index, codes, saved=saved)
|
||||
result["ai_errors"] = ai_errors
|
||||
result["keywords_ok"] = ai_errors == 0
|
||||
return result
|
||||
|
||||
|
||||
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
|
||||
|
||||
@@ -392,7 +392,7 @@ def api_file_extractions(
|
||||
file: str = Query(..., min_length=1),
|
||||
prefix: str = Query("RIP"),
|
||||
):
|
||||
"""Брой и номерация на извлеченията за даден help файл."""
|
||||
"""Брой и номерация на секциите за даден help файл."""
|
||||
name = _safe_help_filename(file)
|
||||
conn = db_conn()
|
||||
try:
|
||||
@@ -430,7 +430,7 @@ def api_delete_file_extractions(
|
||||
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
|
||||
token: Optional[str] = Query(None),
|
||||
):
|
||||
"""Изтрива всички извлечения за файла. Самият help файл не се пипа."""
|
||||
"""Изтрива всички секции за файла. Самият help файл не се пипа."""
|
||||
rescan_mod._check_token(x_rescan_token or token)
|
||||
name = _safe_help_filename(file)
|
||||
conn = db_conn()
|
||||
|
||||
@@ -274,6 +274,10 @@
|
||||
}
|
||||
.rescan-filemeta:empty { display: none; }
|
||||
.rescan-filemeta.warn { color: var(--danger); }
|
||||
.wipe-box { margin-top: 16px; }
|
||||
.wipe-box input[type=text] { width: 100%; font-family: var(--mono); }
|
||||
.src-file { cursor: pointer; }
|
||||
.src-file:hover { color: var(--accent); text-decoration: underline; }
|
||||
|
||||
.api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; }
|
||||
@media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } }
|
||||
@@ -360,11 +364,26 @@
|
||||
<div class="rescan-row">
|
||||
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
|
||||
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
|
||||
<button type="button" class="btn-danger" id="rescan-wipe-btn" hidden
|
||||
onclick="confirmWipeFile()">Изтрий извлеченията</button>
|
||||
</div>
|
||||
<div class="rescan-status" id="rescan-status">Избери файл или папка.</div>
|
||||
</div>
|
||||
|
||||
<div class="rescan-box wipe-box">
|
||||
<h3>Изтриване на секции</h3>
|
||||
<p class="stats" style="margin-bottom:10px;line-height:1.45">
|
||||
Копирай името на файла от таб „Ключови думи“ и го постави тук.
|
||||
Изтриват се всички секции за този файл (не самият help файл).
|
||||
</p>
|
||||
<div class="rescan-row">
|
||||
<input type="text" id="wipe-filename" placeholder="напр. atra-manual.docx"
|
||||
oninput="onWipeNameInput()" onkeydown="if(event.key==='Enter'){confirmWipeFile();event.preventDefault();}">
|
||||
</div>
|
||||
<div class="rescan-filemeta" id="wipe-filemeta"></div>
|
||||
<div class="rescan-row">
|
||||
<button type="button" class="btn-danger" id="wipe-btn" onclick="confirmWipeFile()" disabled>Изтрий всички секции</button>
|
||||
</div>
|
||||
<div class="rescan-status" id="wipe-status"></div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="tab-editor" class="panel">
|
||||
@@ -397,14 +416,14 @@
|
||||
<div class="search-mode">
|
||||
<label><input type="radio" name="search-mode" value="all" checked onchange="doSearch()"> всичко</label>
|
||||
<label><input type="radio" name="search-mode" value="keywords" onchange="doSearch()"> само ключови</label>
|
||||
<label><input type="radio" name="search-mode" value="text" onchange="doSearch()"> само извлечения</label>
|
||||
<label><input type="radio" name="search-mode" value="text" onchange="doSearch()"> само секции</label>
|
||||
</div>
|
||||
<span class="stats" id="search-stats"></span>
|
||||
<button class="btn-primary" onclick="addSelectedToGenerator()">Добави избраните → Генератор</button>
|
||||
</div>
|
||||
<div class="card-section-label" style="margin-top:0">Ключови думи — клик за филтър</div>
|
||||
<div class="kw-cloud" id="kw-cloud"></div>
|
||||
<div class="card-section-label">Извлечения (резултати)</div>
|
||||
<div class="card-section-label">Секции (резултати)</div>
|
||||
<div class="results-grid" id="search-results"></div>
|
||||
</div>
|
||||
|
||||
@@ -444,7 +463,7 @@
|
||||
</div>
|
||||
<div class="card-section-label">Ключови думи</div>
|
||||
<div class="card-tags" id="modal-tags"></div>
|
||||
<div class="card-section-label">Извлечение</div>
|
||||
<div class="card-section-label">Секция</div>
|
||||
<div class="modal-body" id="modal-body">Зареждане…</div>
|
||||
</div>
|
||||
</div>
|
||||
@@ -471,9 +490,9 @@
|
||||
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
|
||||
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
|
||||
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||
<div class="desc">Брой и кодове на извлеченията за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
|
||||
<div class="desc">Брой и кодове на секциите за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
|
||||
<div class="api-ep"><span class="method post">DELETE</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||
<div class="desc">Изтрива всички извлечения за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
||||
<div class="desc">Изтрива всички секции за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
||||
|
||||
<p class="api-note">OpenAPI / Swagger: <a href="/docs" target="_blank" rel="noopener">/docs</a> · ReDoc: <a href="/redoc" target="_blank" rel="noopener">/redoc</a></p>
|
||||
<p class="api-note">Типичен ERP поток: <code>GET /api/search</code> → избор на <code>code</code> → <code>GET /api/section/{code}</code> за съдържание.</p>
|
||||
@@ -507,7 +526,7 @@
|
||||
<div class="modal-backdrop" id="wipe-modal" onclick="if(event.target===this)closeWipeModal()">
|
||||
<div class="modal" role="dialog" aria-modal="true" style="width:min(460px,100%)">
|
||||
<div class="modal-head">
|
||||
<h2>Изтриване на извлечения</h2>
|
||||
<h2>Изтриване на секции</h2>
|
||||
<button type="button" class="btn-ghost" onclick="closeWipeModal()">✕</button>
|
||||
</div>
|
||||
<div class="modal-body" id="wipe-modal-body"></div>
|
||||
@@ -566,7 +585,9 @@ const PAGE_PREFIX = {{ (prefix or 'RIP') | tojson }};
|
||||
let scanMode = null;
|
||||
let scanFiles = [];
|
||||
let fileStats = null;
|
||||
let wipeStats = null;
|
||||
let wipeTarget = null;
|
||||
let wipeLookupTimer = null;
|
||||
|
||||
function scanExt(name) {
|
||||
const n = String(name || '');
|
||||
@@ -578,12 +599,24 @@ function fileBase(p) {
|
||||
const parts = String(p).replace(/\\/g, '/').split('/');
|
||||
return parts[parts.length - 1] || '';
|
||||
}
|
||||
function sectionCountLabel(n) {
|
||||
const x = Number(n) || 0;
|
||||
if (x === 1) return '1 секция';
|
||||
return x + ' секции';
|
||||
}
|
||||
function setScanStatus(text, kind) {
|
||||
const st = document.getElementById('rescan-status');
|
||||
st.textContent = text;
|
||||
st.classList.remove('busy', 'ok', 'error');
|
||||
if (kind) st.classList.add(kind);
|
||||
}
|
||||
function setWipeStatus(text, kind) {
|
||||
const st = document.getElementById('wipe-status');
|
||||
if (!st) return;
|
||||
st.textContent = text || '';
|
||||
st.classList.remove('busy', 'ok', 'error');
|
||||
if (kind) st.classList.add(kind);
|
||||
}
|
||||
function fileCountBg(n) {
|
||||
return n === 1 ? '1 файл' : n + ' файла';
|
||||
}
|
||||
@@ -592,39 +625,75 @@ function folderNameFrom(files) {
|
||||
const top = p.replace(/\\/g, '/').split('/')[0];
|
||||
return top || (files[0] && files[0].name) || 'папка';
|
||||
}
|
||||
function statsLine(stats) {
|
||||
if (!stats) return '';
|
||||
const n = stats.count || 0;
|
||||
let line = 'Секции: ' + n;
|
||||
if (n && stats.first && stats.last) {
|
||||
line += stats.first === stats.last
|
||||
? ' · ' + stats.first
|
||||
: ' · ' + stats.first + ' – ' + stats.last;
|
||||
}
|
||||
if (stats.ok === false && stats.issues && stats.issues.length) {
|
||||
line += ' · ' + stats.issues.join('; ');
|
||||
}
|
||||
return line;
|
||||
}
|
||||
function renderFileStats() {
|
||||
const meta = document.getElementById('rescan-filemeta');
|
||||
const wipeBtn = document.getElementById('rescan-wipe-btn');
|
||||
const editorExtra = document.getElementById('editor-file-stats');
|
||||
const n = fileStats ? (fileStats.count || 0) : 0;
|
||||
const fileSelected = scanMode === 'file' && scanFiles[0];
|
||||
if (!fileStats || !fileSelected) {
|
||||
if (meta) meta.textContent = '';
|
||||
if (wipeBtn) wipeBtn.hidden = true;
|
||||
if (editorExtra) editorExtra.textContent = '';
|
||||
if (meta) { meta.textContent = ''; meta.classList.remove('warn'); }
|
||||
return;
|
||||
}
|
||||
let line = 'Извлечения: ' + n;
|
||||
if (n && fileStats.first && fileStats.last) {
|
||||
line += fileStats.first === fileStats.last
|
||||
? ' · ' + fileStats.first
|
||||
: ' · ' + fileStats.first + ' – ' + fileStats.last;
|
||||
}
|
||||
if (meta) {
|
||||
meta.textContent = line;
|
||||
if (fileStats.ok === false && fileStats.issues && fileStats.issues.length) {
|
||||
meta.textContent = line + ' · ' + fileStats.issues.join('; ');
|
||||
meta.classList.add('warn');
|
||||
} else {
|
||||
meta.classList.remove('warn');
|
||||
meta.textContent = statsLine(fileStats);
|
||||
meta.classList.toggle('warn', fileStats.ok === false);
|
||||
}
|
||||
}
|
||||
function renderWipeStats() {
|
||||
const meta = document.getElementById('wipe-filemeta');
|
||||
const btn = document.getElementById('wipe-btn');
|
||||
const name = (document.getElementById('wipe-filename') || {}).value;
|
||||
const base = fileBase((name || '').trim());
|
||||
if (!base) {
|
||||
if (meta) { meta.textContent = ''; meta.classList.remove('warn'); }
|
||||
if (btn) btn.disabled = true;
|
||||
return;
|
||||
}
|
||||
const n = wipeStats ? (wipeStats.count || 0) : 0;
|
||||
if (meta) {
|
||||
meta.textContent = wipeStats
|
||||
? (statsLine(wipeStats) + (wipeStats.file ? ' · ' + wipeStats.file : ''))
|
||||
: '';
|
||||
meta.classList.toggle('warn', !!(wipeStats && wipeStats.ok === false));
|
||||
}
|
||||
if (btn) btn.disabled = false;
|
||||
}
|
||||
async function fetchFileStats(name) {
|
||||
const base = fileBase(name);
|
||||
if (!base) return null;
|
||||
try {
|
||||
const url = '/api/file-extractions?file=' + encodeURIComponent(base)
|
||||
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
|
||||
const res = await fetch(url);
|
||||
const data = await res.json().catch(() => ({}));
|
||||
if (!res.ok) {
|
||||
const detail = data.detail;
|
||||
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
|
||||
}
|
||||
}
|
||||
if (wipeBtn) {
|
||||
wipeBtn.hidden = n === 0;
|
||||
wipeBtn.disabled = false;
|
||||
}
|
||||
if (editorExtra) {
|
||||
editorExtra.textContent = fileStats.file ? (' · ' + fileStats.file + ': ' + n) : '';
|
||||
return data;
|
||||
} catch (e) {
|
||||
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === base.toLowerCase());
|
||||
const codes = rows.map(r => r.code).filter(Boolean).sort();
|
||||
return {
|
||||
file: base,
|
||||
count: rows.length,
|
||||
first: codes[0] || null,
|
||||
last: codes[codes.length - 1] || null,
|
||||
ok: true,
|
||||
issues: [],
|
||||
};
|
||||
}
|
||||
}
|
||||
async function loadFileStats(name) {
|
||||
@@ -633,30 +702,26 @@ async function loadFileStats(name) {
|
||||
renderFileStats();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const url = '/api/file-extractions?file=' + encodeURIComponent(name)
|
||||
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
|
||||
const res = await fetch(url);
|
||||
const data = await res.json().catch(() => ({}));
|
||||
if (!res.ok) {
|
||||
const detail = data.detail;
|
||||
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
|
||||
}
|
||||
fileStats = data;
|
||||
} catch (e) {
|
||||
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === String(name).toLowerCase());
|
||||
const codes = rows.map(r => r.code).filter(Boolean).sort();
|
||||
fileStats = {
|
||||
file: name,
|
||||
count: rows.length,
|
||||
first: codes[0] || null,
|
||||
last: codes[codes.length - 1] || null,
|
||||
ok: true,
|
||||
issues: [],
|
||||
};
|
||||
}
|
||||
fileStats = await fetchFileStats(name);
|
||||
renderFileStats();
|
||||
}
|
||||
function onWipeNameInput() {
|
||||
if (wipeLookupTimer) clearTimeout(wipeLookupTimer);
|
||||
const raw = (document.getElementById('wipe-filename').value || '').trim();
|
||||
const base = fileBase(raw);
|
||||
if (!base) {
|
||||
wipeStats = null;
|
||||
renderWipeStats();
|
||||
setWipeStatus('');
|
||||
return;
|
||||
}
|
||||
wipeLookupTimer = setTimeout(async () => {
|
||||
wipeStats = await fetchFileStats(base);
|
||||
renderWipeStats();
|
||||
const n = wipeStats ? (wipeStats.count || 0) : 0;
|
||||
setWipeStatus(n ? ('Намерени ' + sectionCountLabel(n) + '.') : 'Няма секции за този файл.');
|
||||
}, 250);
|
||||
}
|
||||
async function refreshSections() {
|
||||
const p = {{ (prefix or '') | tojson }};
|
||||
const url = p ? '/api/sections?prefix=' + encodeURIComponent(p) : '/api/sections';
|
||||
@@ -739,12 +804,16 @@ function onScanFolderPicked() {
|
||||
}
|
||||
|
||||
function confirmWipeFile() {
|
||||
const name = scanFiles[0] && scanFiles[0].name;
|
||||
if (!name) return;
|
||||
const n = (fileStats && fileStats.count) || 0;
|
||||
const raw = (document.getElementById('wipe-filename').value || '').trim();
|
||||
const name = fileBase(raw);
|
||||
if (!name) {
|
||||
setWipeStatus('Постави име на файл.', 'error');
|
||||
return;
|
||||
}
|
||||
const n = (wipeStats && wipeStats.count) || 0;
|
||||
wipeTarget = name;
|
||||
document.getElementById('wipe-modal-body').textContent =
|
||||
'Ще се изтрият всички извлечения за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
|
||||
'Ще се изтрият всички секции за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
|
||||
document.getElementById('wipe-modal').classList.add('open');
|
||||
}
|
||||
function closeWipeModal() {
|
||||
@@ -773,12 +842,13 @@ async function doWipeFile() {
|
||||
document.getElementById('wipe-modal').classList.remove('open');
|
||||
wipeTarget = null;
|
||||
await refreshSections();
|
||||
await loadFileStats(name);
|
||||
setScanStatus('Изтрити извлечения: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
|
||||
wipeStats = await fetchFileStats(name);
|
||||
renderWipeStats();
|
||||
setWipeStatus('Изтрити секции: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
|
||||
toast('Изтрити: ' + (data.deleted || 0));
|
||||
} catch (e) {
|
||||
toast(e.message, true);
|
||||
setScanStatus('Грешка: ' + e.message, 'error');
|
||||
setWipeStatus('Грешка: ' + e.message, 'error');
|
||||
} finally {
|
||||
btn.disabled = false;
|
||||
}
|
||||
@@ -788,7 +858,6 @@ async function startRescan() {
|
||||
const btn = document.getElementById('rescan-btn');
|
||||
const fileBtn = document.getElementById('rescan-pick-file');
|
||||
const folderBtn = document.getElementById('rescan-pick-folder');
|
||||
const wipeBtn = document.getElementById('rescan-wipe-btn');
|
||||
if (!scanFiles.length) {
|
||||
setScanStatus(scanMode === 'folder' ? 'Няма подходящи файлове в папката.' : 'Избери файл или папка.', 'error');
|
||||
return;
|
||||
@@ -803,7 +872,6 @@ async function startRescan() {
|
||||
btn.disabled = true;
|
||||
fileBtn.disabled = true;
|
||||
folderBtn.disabled = true;
|
||||
if (wipeBtn) wipeBtn.disabled = true;
|
||||
setScanStatus('Качване...', 'busy');
|
||||
try {
|
||||
const headers = {};
|
||||
@@ -823,7 +891,6 @@ async function startRescan() {
|
||||
btn.disabled = false;
|
||||
fileBtn.disabled = false;
|
||||
folderBtn.disabled = false;
|
||||
if (wipeBtn) wipeBtn.disabled = false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -840,16 +907,24 @@ async function pollRescan(jobId) {
|
||||
if (typeof data.sections === 'number') msg += ' Записани: ' + data.sections;
|
||||
if (files.length === 1) {
|
||||
const f = files[0];
|
||||
msg = 'Готово. Извлечения: ' + (f.count || 0);
|
||||
msg = 'Готово. Секции: ' + (f.count || 0);
|
||||
if (f.first && f.last) {
|
||||
msg += f.first === f.last ? ' · ' + f.first : ' · ' + f.first + ' – ' + f.last;
|
||||
}
|
||||
if (f.ok === false && f.issues && f.issues.length) msg += ' · ' + f.issues.join('; ');
|
||||
if (f.keywords_ok === false || (f.ai_errors || 0) > 0) {
|
||||
msg += ' · ключовите думи са с локален fallback';
|
||||
}
|
||||
}
|
||||
setScanStatus(msg, bad ? 'error' : 'ok');
|
||||
try {
|
||||
await refreshSections();
|
||||
if (scanMode === 'file' && scanFiles[0]) await loadFileStats(scanFiles[0].name);
|
||||
const wipeName = fileBase((document.getElementById('wipe-filename').value || '').trim());
|
||||
if (wipeName) {
|
||||
wipeStats = await fetchFileStats(wipeName);
|
||||
renderWipeStats();
|
||||
}
|
||||
} catch (e) {
|
||||
setScanStatus((msg || 'Готово.') + ' Презареди страницата.', bad ? 'error' : 'ok');
|
||||
}
|
||||
@@ -886,7 +961,9 @@ function renderEditor(rows) {
|
||||
</div>
|
||||
</div>
|
||||
</td>
|
||||
<td><span class="src-file" title="${esc(r.source_file)}">${esc(shortPath(r.source_file))}</span></td>
|
||||
<td><span class="src-file" title="Кликни за копиране: ${esc(fileBase(r.source_file))}"
|
||||
data-file="${esc(fileBase(r.source_file))}"
|
||||
onclick="copySourceFile(this.dataset.file, event)">${esc(shortPath(r.source_file))}</span></td>
|
||||
<td style="font-size:11px;color:var(--muted);font-family:var(--mono);white-space:nowrap">${r.updated_at}</td>
|
||||
</tr>
|
||||
`).join('');
|
||||
@@ -904,6 +981,21 @@ function filterEditor() {
|
||||
renderEditor(filtered);
|
||||
}
|
||||
|
||||
function copySourceFile(name, ev) {
|
||||
if (ev) ev.stopPropagation();
|
||||
const base = fileBase(name);
|
||||
if (!base) return;
|
||||
const inp = document.getElementById('wipe-filename');
|
||||
if (inp) inp.value = base;
|
||||
onWipeNameInput();
|
||||
const done = () => toast('Копирано: ' + base);
|
||||
if (navigator.clipboard && navigator.clipboard.writeText) {
|
||||
navigator.clipboard.writeText(base).then(done, done);
|
||||
} else {
|
||||
done();
|
||||
}
|
||||
}
|
||||
|
||||
function onKwChange(inp) {
|
||||
const code = inp.dataset.code;
|
||||
inp.classList.add('changed');
|
||||
@@ -1098,7 +1190,7 @@ function doSearch() {
|
||||
if (raw && !tokens.length) note = ' · само паразитни думи — показани всички';
|
||||
|
||||
document.getElementById('search-stats').textContent =
|
||||
results.length + ' извлечения' + note;
|
||||
results.length + ' секции' + note;
|
||||
document.getElementById('search-results').innerHTML = results.map(r => `
|
||||
<div class="card ${selected.has(r.code)?'selected':''}" onclick="toggleSelect('${r.code}', this)">
|
||||
<div class="card-header">
|
||||
@@ -1112,10 +1204,10 @@ function doSearch() {
|
||||
<div class="card-tags">
|
||||
${(r.keywords||'').split(',').filter(k=>k.trim()).map(k=>`<span class="tag">${esc(k.trim())}</span>`).join('') || '<span class="stats">—</span>'}
|
||||
</div>
|
||||
<div class="card-section-label">Извлечение</div>
|
||||
<div class="card-section-label">Секция</div>
|
||||
<div class="card-text">${r.text_html || esc(r.text||'(няма текст — провери OUTPUT_DIR)')}</div>
|
||||
<div class="card-actions">
|
||||
<button type="button" class="btn-ghost" onclick="event.stopPropagation(); openSection('${r.code}')">Цяло извлечение</button>
|
||||
<button type="button" class="btn-ghost" onclick="event.stopPropagation(); openSection('${r.code}')">Цяла секция</button>
|
||||
</div>
|
||||
<div class="card-footer">${esc(shortPath(r.source_file))} · ${r.char_count||0} знака</div>
|
||||
</div>
|
||||
@@ -1374,6 +1466,7 @@ try {
|
||||
doSearch();
|
||||
renderGenerator();
|
||||
initApiTab();
|
||||
renderWipeStats();
|
||||
} catch (e) {
|
||||
console.error('init failed', e);
|
||||
toast('Грешка при зареждане на UI: ' + e.message, true);
|
||||
|
||||
Reference in New Issue
Block a user