.
This commit is contained in:
@@ -1,4 +1,4 @@
|
|||||||
"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията."""
|
"""Кодове на секции: PREFIX_0001_SEC_0001 и проверка на номерацията."""
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import re
|
import re
|
||||||
|
|||||||
@@ -71,19 +71,27 @@ try:
|
|||||||
except AttributeError:
|
except AttributeError:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
|
_log_handlers = [logging.StreamHandler(sys.stdout)]
|
||||||
|
try:
|
||||||
|
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
|
||||||
|
except OSError:
|
||||||
|
pass
|
||||||
logging.basicConfig(
|
logging.basicConfig(
|
||||||
level=logging.INFO,
|
level=logging.INFO,
|
||||||
format="%(asctime)s %(levelname)-8s %(message)s",
|
format="%(asctime)s %(levelname)-8s %(message)s",
|
||||||
handlers=[
|
handlers=_log_handlers,
|
||||||
logging.StreamHandler(sys.stdout),
|
|
||||||
logging.FileHandler("help_processor.log", encoding="utf-8"),
|
|
||||||
],
|
|
||||||
)
|
)
|
||||||
log = logging.getLogger(__name__)
|
log = logging.getLogger(__name__)
|
||||||
|
|
||||||
MIN_SECTION_TOKENS = 60 # секции под тази граница се сливат с предишната
|
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
|
||||||
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
|
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
|
||||||
AI_MODEL = "claude-haiku-4-5"
|
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
|
||||||
|
AI_MODELS = [
|
||||||
|
AI_MODEL,
|
||||||
|
"claude-haiku-4-5",
|
||||||
|
"claude-haiku-4-5-20251001",
|
||||||
|
"claude-3-5-haiku-latest",
|
||||||
|
]
|
||||||
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
|
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
|
||||||
|
|
||||||
|
|
||||||
@@ -290,7 +298,7 @@ class Database:
|
|||||||
self.conn.commit()
|
self.conn.commit()
|
||||||
|
|
||||||
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
|
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
|
||||||
"""(code, source_file, output_path) за всички извлечения на файла."""
|
"""(code, source_file, output_path) за всички секции на файла."""
|
||||||
paths = self.matching_source_paths(prefix, identity)
|
paths = self.matching_source_paths(prefix, identity)
|
||||||
if not paths:
|
if not paths:
|
||||||
return []
|
return []
|
||||||
@@ -371,7 +379,7 @@ class Database:
|
|||||||
identity: str,
|
identity: str,
|
||||||
output_dir: Optional[Path] = None,
|
output_dir: Optional[Path] = None,
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||||||
rows = self.sections_for_file(prefix, identity)
|
rows = self.sections_for_file(prefix, identity)
|
||||||
codes = [r[0] for r in rows]
|
codes = [r[0] for r in rows]
|
||||||
file_index = self.file_index_for(prefix, identity)
|
file_index = self.file_index_for(prefix, identity)
|
||||||
@@ -548,6 +556,96 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
|||||||
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||||||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||||||
"valign", "width", "height", "bgcolor", "border")
|
"valign", "width", "height", "bgcolor", "border")
|
||||||
|
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||||||
|
_HEADING_TOKEN_RE = re.compile(
|
||||||
|
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
|
||||||
|
r"(\d+)?$",
|
||||||
|
re.I,
|
||||||
|
)
|
||||||
|
_HTML_HEADING_CLASS_RE = re.compile(
|
||||||
|
r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
|
||||||
|
re.I,
|
||||||
|
)
|
||||||
|
_HEADING_LEVEL = {
|
||||||
|
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
|
||||||
|
"title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
|
||||||
|
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
|
||||||
|
"подзаглавие": 2, "наименование": 1,
|
||||||
|
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _compact_style_token(s: str) -> str:
|
||||||
|
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
|
||||||
|
|
||||||
|
|
||||||
|
def _heading_level_from_token(token: str) -> Optional[int]:
|
||||||
|
t = _compact_style_token(token)
|
||||||
|
if not t:
|
||||||
|
return None
|
||||||
|
if t in _HEADING_LEVEL:
|
||||||
|
return _HEADING_LEVEL[t]
|
||||||
|
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
|
||||||
|
if not m:
|
||||||
|
return None
|
||||||
|
n = m.group(2)
|
||||||
|
if n and n.isdigit():
|
||||||
|
return min(int(n), 3)
|
||||||
|
kind = (m.group(1) or "").lower()
|
||||||
|
if kind in ("subtitle", "подзаглавие"):
|
||||||
|
return 2
|
||||||
|
return 1
|
||||||
|
|
||||||
|
|
||||||
|
def _docx_heading_level(para) -> Optional[int]:
|
||||||
|
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word."""
|
||||||
|
style = getattr(para, "style", None)
|
||||||
|
seen: set[int] = set()
|
||||||
|
cur = style
|
||||||
|
while cur is not None and id(cur) not in seen:
|
||||||
|
seen.add(id(cur))
|
||||||
|
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
|
||||||
|
lvl = _heading_level_from_token(str(token or ""))
|
||||||
|
if lvl:
|
||||||
|
return lvl
|
||||||
|
cur = getattr(cur, "base_style", None)
|
||||||
|
try:
|
||||||
|
pPr = para._element.pPr
|
||||||
|
if pPr is not None and pPr.outlineLvl is not None:
|
||||||
|
val = int(pPr.outlineLvl.val)
|
||||||
|
text = (para.text or "").strip()
|
||||||
|
if 0 <= val <= 2 and text and len(text) < 120:
|
||||||
|
return min(val + 1, 3)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _is_bold_heading_text(text: str, runs) -> bool:
|
||||||
|
if not text or len(text) > 120:
|
||||||
|
return False
|
||||||
|
useful = [r for r in (runs or []) if (r.text or "").strip()]
|
||||||
|
return bool(useful) and all(bool(r.bold) for r in useful)
|
||||||
|
|
||||||
|
|
||||||
|
def _html_heading_level(el) -> Optional[int]:
|
||||||
|
name = (getattr(el, "name", None) or "").lower()
|
||||||
|
if name in _HTML_HEADING_MAP:
|
||||||
|
return _HTML_HEADING_MAP[name]
|
||||||
|
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
|
||||||
|
m = _HTML_HEADING_CLASS_RE.search(cls)
|
||||||
|
if m:
|
||||||
|
n = m.group(1)
|
||||||
|
return min(int(n), 3) if n and n.isdigit() else 1
|
||||||
|
if name in ("p", "div"):
|
||||||
|
txt = el.get_text(" ", strip=True)
|
||||||
|
if txt and len(txt) < 120:
|
||||||
|
inner = "".join(el.stripped_strings)
|
||||||
|
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
|
||||||
|
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
|
||||||
|
if strong and strong_txt and strong_txt == inner:
|
||||||
|
return 2
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _html_block_plain_text(el) -> str:
|
def _html_block_plain_text(el) -> str:
|
||||||
@@ -600,12 +698,13 @@ def parse_html(path: Path) -> list[Section]:
|
|||||||
base_dir = path.parent
|
base_dir = path.parent
|
||||||
body = soup.body or soup
|
body = soup.body or soup
|
||||||
|
|
||||||
heading_map = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
|
||||||
|
|
||||||
# Събираме top-level блокови елементи (без да включваме вложените в тях)
|
# Събираме top-level блокови елементи (без да включваме вложените в тях)
|
||||||
consumed = set()
|
consumed = set()
|
||||||
blocks = []
|
blocks = []
|
||||||
for el in body.find_all(_HTML_BLOCK_TAGS + ["img"]):
|
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
|
||||||
|
name = (el.name or "").lower()
|
||||||
|
if name == "div" and not _html_heading_level(el):
|
||||||
|
continue
|
||||||
if any(id(par) in consumed for par in el.parents):
|
if any(id(par) in consumed for par in el.parents):
|
||||||
continue
|
continue
|
||||||
consumed.add(id(el))
|
consumed.add(id(el))
|
||||||
@@ -620,20 +719,21 @@ def parse_html(path: Path) -> list[Section]:
|
|||||||
img_counter = [0]
|
img_counter = [0]
|
||||||
|
|
||||||
def flush():
|
def flush():
|
||||||
if sec_text or sec_html or sec_images:
|
if current_title or sec_text or sec_html or sec_images:
|
||||||
sec = Section(current_title, "\n".join(sec_text), current_level)
|
sec = Section(current_title, "\n".join(sec_text), current_level)
|
||||||
sec.images = list(sec_images)
|
sec.images = list(sec_images)
|
||||||
sec.html_text = "\n".join(sec_html) if sec_html else None
|
sec.html_text = "\n".join(sec_html) if sec_html else None
|
||||||
sections.append(sec)
|
sections.append(sec)
|
||||||
|
|
||||||
for el in blocks:
|
for el in blocks:
|
||||||
if el.name in heading_map:
|
heading_lvl = _html_heading_level(el)
|
||||||
|
if heading_lvl:
|
||||||
txt = el.get_text(" ", strip=True)
|
txt = el.get_text(" ", strip=True)
|
||||||
if not txt:
|
if not txt:
|
||||||
continue
|
continue
|
||||||
flush()
|
flush()
|
||||||
current_title = txt
|
current_title = txt
|
||||||
current_level = heading_map[el.name]
|
current_level = heading_lvl
|
||||||
sec_text, sec_html, sec_images = [], [], []
|
sec_text, sec_html, sec_images = [], [], []
|
||||||
continue
|
continue
|
||||||
|
|
||||||
@@ -693,24 +793,58 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
|||||||
return imgs
|
return imgs
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_docx_blocks(doc):
|
||||||
|
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
|
||||||
|
from docx.oxml.ns import qn
|
||||||
|
from docx.table import Table
|
||||||
|
from docx.text.paragraph import Paragraph
|
||||||
|
|
||||||
|
body = doc.element.body
|
||||||
|
for child in body.iterchildren():
|
||||||
|
if child.tag == qn("w:p"):
|
||||||
|
yield "p", Paragraph(child, doc)
|
||||||
|
elif child.tag == qn("w:tbl"):
|
||||||
|
yield "tbl", Table(child, doc)
|
||||||
|
|
||||||
|
|
||||||
|
def _table_lines(table) -> list[str]:
|
||||||
|
lines: list[str] = []
|
||||||
|
try:
|
||||||
|
rows = table.rows
|
||||||
|
except Exception:
|
||||||
|
return lines
|
||||||
|
for row in rows:
|
||||||
|
try:
|
||||||
|
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
cells = [c for c in cells if c]
|
||||||
|
if cells:
|
||||||
|
lines.append(" | ".join(cells))
|
||||||
|
return lines
|
||||||
|
|
||||||
|
|
||||||
def parse_docx(path: Path) -> list[Section]:
|
def parse_docx(path: Path) -> list[Section]:
|
||||||
doc = Document(path)
|
doc = Document(path)
|
||||||
sections: list[Section] = []
|
sections: list[Section] = []
|
||||||
current_title, current_level = "", 1
|
current_title, current_level = "", 1
|
||||||
buf: list[str] = []
|
buf: list[str] = []
|
||||||
sec_images: list[ImageRef] = []
|
sec_images: list[ImageRef] = []
|
||||||
img_counter = [0] # списък за nonlocal-стил мутация
|
img_counter = [0]
|
||||||
|
|
||||||
HEADING_STYLES = {"heading 1": 1, "heading 2": 2, "heading 3": 3,
|
|
||||||
"title": 1, "subtitle": 2}
|
|
||||||
|
|
||||||
def flush():
|
def flush():
|
||||||
if buf or sec_images:
|
if current_title or buf or sec_images:
|
||||||
sec = Section(current_title, "\n".join(buf), current_level)
|
sec = Section(current_title, "\n".join(buf), current_level)
|
||||||
sec.images = list(sec_images)
|
sec.images = list(sec_images)
|
||||||
sections.append(sec)
|
sections.append(sec)
|
||||||
|
|
||||||
for para in doc.paragraphs:
|
for kind, block in _iter_docx_blocks(doc):
|
||||||
|
if kind == "tbl":
|
||||||
|
for line in _table_lines(block):
|
||||||
|
buf.append(line)
|
||||||
|
continue
|
||||||
|
|
||||||
|
para = block
|
||||||
style_name = para.style.name.lower() if para.style else ""
|
style_name = para.style.name.lower() if para.style else ""
|
||||||
text = para.text.strip()
|
text = para.text.strip()
|
||||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||||||
@@ -718,12 +852,15 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
if not text and not para_imgs:
|
if not text and not para_imgs:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
level = HEADING_STYLES.get(style_name)
|
level = _docx_heading_level(para)
|
||||||
is_bold_heading = bool(text and len(text) < 120 and not style_name.startswith("list")
|
is_bold_heading = (
|
||||||
and para.runs
|
not level
|
||||||
and all(run.bold for run in para.runs if run.text.strip()))
|
and _is_bold_heading_text(text, para.runs)
|
||||||
|
and not style_name.startswith("list")
|
||||||
|
and not para_imgs
|
||||||
|
)
|
||||||
|
|
||||||
if level or (is_bold_heading and not para_imgs):
|
if level or is_bold_heading:
|
||||||
flush()
|
flush()
|
||||||
buf, sec_images = [], []
|
buf, sec_images = [], []
|
||||||
current_title = text
|
current_title = text
|
||||||
@@ -742,6 +879,11 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
|
|
||||||
if not sections:
|
if not sections:
|
||||||
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
|
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
|
||||||
|
if not fallback_text:
|
||||||
|
fallback_text = "\n".join(
|
||||||
|
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
|
||||||
|
for line in _table_lines(block)
|
||||||
|
)
|
||||||
return [Section("", fallback_text, 0)]
|
return [Section("", fallback_text, 0)]
|
||||||
return sections
|
return sections
|
||||||
|
|
||||||
@@ -986,7 +1128,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
|||||||
prev_size = None
|
prev_size = None
|
||||||
|
|
||||||
def flush():
|
def flush():
|
||||||
if buf or sec_images:
|
if current_title or buf or sec_images:
|
||||||
sec = Section(current_title, "\n".join(buf), 2)
|
sec = Section(current_title, "\n".join(buf), 2)
|
||||||
sec.images = list(sec_images)
|
sec.images = list(sec_images)
|
||||||
sections.append(sec)
|
sections.append(sec)
|
||||||
@@ -1055,7 +1197,39 @@ def parse_txt(path: Path) -> list[Section]:
|
|||||||
raw = path.read_bytes()
|
raw = path.read_bytes()
|
||||||
enc = chardet.detect(raw)["encoding"] or "utf-8"
|
enc = chardet.detect(raw)["encoding"] or "utf-8"
|
||||||
text = raw.decode(enc, errors="replace")
|
text = raw.decode(enc, errors="replace")
|
||||||
return [Section("", text, 0)]
|
return _split_plain_text(text)
|
||||||
|
|
||||||
|
|
||||||
|
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
|
||||||
|
_PLAIN_NUM_HEADING_RE = re.compile(
|
||||||
|
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _split_plain_text(text: str) -> list[Section]:
|
||||||
|
"""Markdown / номерирани заглавия; иначе една секция."""
|
||||||
|
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
|
||||||
|
sections: list[Section] = []
|
||||||
|
title, level, buf = "", 0, []
|
||||||
|
|
||||||
|
def flush():
|
||||||
|
body = "\n".join(buf).strip()
|
||||||
|
if title or body:
|
||||||
|
sections.append(Section(title, body, level))
|
||||||
|
|
||||||
|
for line in lines:
|
||||||
|
raw = line.strip()
|
||||||
|
md = _PLAIN_MD_HEADING_RE.match(raw)
|
||||||
|
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
|
||||||
|
if md or numbered:
|
||||||
|
flush()
|
||||||
|
title = md.group(2).strip() if md else raw
|
||||||
|
level = len(md.group(1)) if md else 2
|
||||||
|
buf = []
|
||||||
|
continue
|
||||||
|
buf.append(line.rstrip())
|
||||||
|
flush()
|
||||||
|
return sections or [Section("", text, 0)]
|
||||||
|
|
||||||
|
|
||||||
PARSERS = {
|
PARSERS = {
|
||||||
@@ -1073,15 +1247,16 @@ PARSERS = {
|
|||||||
# ──────────────────────────────────────────────
|
# ──────────────────────────────────────────────
|
||||||
|
|
||||||
def merge_short_sections(sections: list[Section]) -> list[Section]:
|
def merge_short_sections(sections: list[Section]) -> list[Section]:
|
||||||
"""Слива секции, по-кратки от MIN_SECTION_TOKENS думи, с предишната."""
|
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
|
||||||
result: list[Section] = []
|
result: list[Section] = []
|
||||||
for sec in sections:
|
for sec in sections:
|
||||||
words = len(sec.text.split())
|
words = len((sec.text or "").split())
|
||||||
if result and words < MIN_SECTION_TOKENS:
|
titled = bool((sec.title or "").strip())
|
||||||
|
if result and not titled and words < MIN_SECTION_TOKENS:
|
||||||
prev = result[-1]
|
prev = result[-1]
|
||||||
merged = Section(
|
merged = Section(
|
||||||
prev.title,
|
prev.title,
|
||||||
prev.text + "\n" + sec.text,
|
(prev.text + "\n" + sec.text).strip(),
|
||||||
prev.level,
|
prev.level,
|
||||||
)
|
)
|
||||||
merged.images = (prev.images or []) + (sec.images or [])
|
merged.images = (prev.images or []) + (sec.images or [])
|
||||||
@@ -1106,6 +1281,77 @@ def clean_text(text: str) -> str:
|
|||||||
# AI класификация
|
# AI класификация
|
||||||
# ──────────────────────────────────────────────
|
# ──────────────────────────────────────────────
|
||||||
|
|
||||||
|
def _content_text(msg) -> str:
|
||||||
|
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
|
||||||
|
parts: list[str] = []
|
||||||
|
for block in getattr(msg, "content", None) or []:
|
||||||
|
btype = getattr(block, "type", None)
|
||||||
|
if btype in (None, "text"):
|
||||||
|
t = getattr(block, "text", None)
|
||||||
|
if t:
|
||||||
|
parts.append(str(t))
|
||||||
|
elif isinstance(block, dict) and block.get("text"):
|
||||||
|
parts.append(str(block["text"]))
|
||||||
|
return "\n".join(parts).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
|
||||||
|
raw = (raw or "").strip()
|
||||||
|
if not raw:
|
||||||
|
return None
|
||||||
|
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
||||||
|
raw = re.sub(r"\n?```$", "", raw)
|
||||||
|
candidates = [raw]
|
||||||
|
m = re.search(r"\{.*\}", raw, re.S)
|
||||||
|
if m:
|
||||||
|
candidates.append(m.group(0))
|
||||||
|
for cand in candidates:
|
||||||
|
try:
|
||||||
|
data = json.loads(cand)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
continue
|
||||||
|
if not isinstance(data, dict):
|
||||||
|
continue
|
||||||
|
t = data.get("title", "")
|
||||||
|
k = data.get("keywords", "")
|
||||||
|
if isinstance(k, list):
|
||||||
|
k = ", ".join(str(x).strip() for x in k if str(x).strip())
|
||||||
|
return str(t)[:200], str(k)[:300]
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
_FALLBACK_KW_STOP = {
|
||||||
|
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
|
||||||
|
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
|
||||||
|
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def fallback_classify(title: str, text: str) -> tuple[str, str]:
|
||||||
|
t = (title or "").strip()
|
||||||
|
if not t:
|
||||||
|
for line in (text or "").splitlines():
|
||||||
|
line = line.strip()
|
||||||
|
if 3 <= len(line) <= 80:
|
||||||
|
t = line
|
||||||
|
break
|
||||||
|
if not t:
|
||||||
|
t = "Секция"
|
||||||
|
blob = f"{title} {text}"[:1200]
|
||||||
|
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
|
||||||
|
seen: list[str] = []
|
||||||
|
seen_l: set[str] = set()
|
||||||
|
for w in words:
|
||||||
|
wl = w.lower()
|
||||||
|
if wl in _FALLBACK_KW_STOP or wl in seen_l:
|
||||||
|
continue
|
||||||
|
seen.append(w)
|
||||||
|
seen_l.add(wl)
|
||||||
|
if len(seen) >= 5:
|
||||||
|
break
|
||||||
|
return t[:200], ", ".join(seen)
|
||||||
|
|
||||||
|
|
||||||
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
||||||
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
||||||
snippet = text[:MAX_AI_CHARS]
|
snippet = text[:MAX_AI_CHARS]
|
||||||
@@ -1120,23 +1366,31 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl
|
|||||||
|
|
||||||
Върни САМО валиден JSON без markdown, без коментари."""
|
Върни САМО валиден JSON без markdown, без коментари."""
|
||||||
|
|
||||||
msg = client.messages.create(
|
last_err: Optional[Exception] = None
|
||||||
model=AI_MODEL,
|
tried: set[str] = set()
|
||||||
max_tokens=200,
|
for model in AI_MODELS:
|
||||||
messages=[{"role": "user", "content": prompt}]
|
if not model or model in tried:
|
||||||
)
|
continue
|
||||||
raw = msg.content[0].text.strip()
|
tried.add(model)
|
||||||
raw = re.sub(r"^```[a-z]*\n?", "", raw)
|
|
||||||
raw = re.sub(r"\n?```$", "", raw)
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
data = json.loads(raw)
|
msg = client.messages.create(
|
||||||
t = str(data.get("title", title or "Секция"))[:200]
|
model=model,
|
||||||
k = str(data.get("keywords", ""))[:300]
|
max_tokens=512,
|
||||||
return t, k
|
messages=[{"role": "user", "content": prompt}],
|
||||||
except json.JSONDecodeError:
|
)
|
||||||
log.warning(f"AI върна невалиден JSON: {raw[:120]}")
|
raw = _content_text(msg)
|
||||||
return title or "Секция", ""
|
parsed = _parse_classify_json(raw)
|
||||||
|
if parsed:
|
||||||
|
t, k = parsed
|
||||||
|
return (t or title or "Секция")[:200], k
|
||||||
|
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
|
||||||
|
except Exception as e:
|
||||||
|
last_err = e
|
||||||
|
log.warning(f"AI classify ({model}) неуспешен: {e}")
|
||||||
|
continue
|
||||||
|
if last_err:
|
||||||
|
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
|
||||||
|
return fallback_classify(title, text)
|
||||||
|
|
||||||
|
|
||||||
# ──────────────────────────────────────────────
|
# ──────────────────────────────────────────────
|
||||||
@@ -1207,10 +1461,14 @@ def process_file(
|
|||||||
|
|
||||||
saved = 0
|
saved = 0
|
||||||
codes: list[str] = []
|
codes: list[str] = []
|
||||||
|
ai_errors = 0
|
||||||
for sec in sections:
|
for sec in sections:
|
||||||
text = clean_text(sec.text)
|
text = clean_text(sec.text)
|
||||||
html_text = sec.html_text or ""
|
html_text = sec.html_text or ""
|
||||||
if not text and not sec.images and not html_text:
|
if not text and not sec.images and not html_text:
|
||||||
|
if (sec.title or "").strip():
|
||||||
|
text = sec.title.strip()
|
||||||
|
else:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
sec_index = saved + 1
|
sec_index = saved + 1
|
||||||
@@ -1249,7 +1507,11 @@ def process_file(
|
|||||||
title, keywords = classify_section(client, sec.title, text)
|
title, keywords = classify_section(client, sec.title, text)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
log.warning(f" AI грешка за {code}: {e}")
|
log.warning(f" AI грешка за {code}: {e}")
|
||||||
title, keywords = sec.title or f"Секция {sec_index}", ""
|
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
|
||||||
|
ai_errors += 1
|
||||||
|
if not (keywords or "").strip():
|
||||||
|
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
|
||||||
|
ai_errors += 1
|
||||||
|
|
||||||
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
||||||
ps = ProcessedSection(
|
ps = ProcessedSection(
|
||||||
@@ -1278,7 +1540,10 @@ def process_file(
|
|||||||
|
|
||||||
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
|
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
|
||||||
log.info(f" → {saved} секции записани")
|
log.info(f" → {saved} секции записани")
|
||||||
return _file_result(rel, file_index, codes, saved=saved)
|
result = _file_result(rel, file_index, codes, saved=saved)
|
||||||
|
result["ai_errors"] = ai_errors
|
||||||
|
result["keywords_ok"] = ai_errors == 0
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
|
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
|
||||||
|
|||||||
@@ -392,7 +392,7 @@ def api_file_extractions(
|
|||||||
file: str = Query(..., min_length=1),
|
file: str = Query(..., min_length=1),
|
||||||
prefix: str = Query("RIP"),
|
prefix: str = Query("RIP"),
|
||||||
):
|
):
|
||||||
"""Брой и номерация на извлеченията за даден help файл."""
|
"""Брой и номерация на секциите за даден help файл."""
|
||||||
name = _safe_help_filename(file)
|
name = _safe_help_filename(file)
|
||||||
conn = db_conn()
|
conn = db_conn()
|
||||||
try:
|
try:
|
||||||
@@ -430,7 +430,7 @@ def api_delete_file_extractions(
|
|||||||
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
|
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
|
||||||
token: Optional[str] = Query(None),
|
token: Optional[str] = Query(None),
|
||||||
):
|
):
|
||||||
"""Изтрива всички извлечения за файла. Самият help файл не се пипа."""
|
"""Изтрива всички секции за файла. Самият help файл не се пипа."""
|
||||||
rescan_mod._check_token(x_rescan_token or token)
|
rescan_mod._check_token(x_rescan_token or token)
|
||||||
name = _safe_help_filename(file)
|
name = _safe_help_filename(file)
|
||||||
conn = db_conn()
|
conn = db_conn()
|
||||||
|
|||||||
@@ -274,6 +274,10 @@
|
|||||||
}
|
}
|
||||||
.rescan-filemeta:empty { display: none; }
|
.rescan-filemeta:empty { display: none; }
|
||||||
.rescan-filemeta.warn { color: var(--danger); }
|
.rescan-filemeta.warn { color: var(--danger); }
|
||||||
|
.wipe-box { margin-top: 16px; }
|
||||||
|
.wipe-box input[type=text] { width: 100%; font-family: var(--mono); }
|
||||||
|
.src-file { cursor: pointer; }
|
||||||
|
.src-file:hover { color: var(--accent); text-decoration: underline; }
|
||||||
|
|
||||||
.api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; }
|
.api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; }
|
||||||
@media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } }
|
@media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } }
|
||||||
@@ -360,11 +364,26 @@
|
|||||||
<div class="rescan-row">
|
<div class="rescan-row">
|
||||||
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
|
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
|
||||||
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
|
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
|
||||||
<button type="button" class="btn-danger" id="rescan-wipe-btn" hidden
|
|
||||||
onclick="confirmWipeFile()">Изтрий извлеченията</button>
|
|
||||||
</div>
|
</div>
|
||||||
<div class="rescan-status" id="rescan-status">Избери файл или папка.</div>
|
<div class="rescan-status" id="rescan-status">Избери файл или папка.</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
|
<div class="rescan-box wipe-box">
|
||||||
|
<h3>Изтриване на секции</h3>
|
||||||
|
<p class="stats" style="margin-bottom:10px;line-height:1.45">
|
||||||
|
Копирай името на файла от таб „Ключови думи“ и го постави тук.
|
||||||
|
Изтриват се всички секции за този файл (не самият help файл).
|
||||||
|
</p>
|
||||||
|
<div class="rescan-row">
|
||||||
|
<input type="text" id="wipe-filename" placeholder="напр. atra-manual.docx"
|
||||||
|
oninput="onWipeNameInput()" onkeydown="if(event.key==='Enter'){confirmWipeFile();event.preventDefault();}">
|
||||||
|
</div>
|
||||||
|
<div class="rescan-filemeta" id="wipe-filemeta"></div>
|
||||||
|
<div class="rescan-row">
|
||||||
|
<button type="button" class="btn-danger" id="wipe-btn" onclick="confirmWipeFile()" disabled>Изтрий всички секции</button>
|
||||||
|
</div>
|
||||||
|
<div class="rescan-status" id="wipe-status"></div>
|
||||||
|
</div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
<div id="tab-editor" class="panel">
|
<div id="tab-editor" class="panel">
|
||||||
@@ -397,14 +416,14 @@
|
|||||||
<div class="search-mode">
|
<div class="search-mode">
|
||||||
<label><input type="radio" name="search-mode" value="all" checked onchange="doSearch()"> всичко</label>
|
<label><input type="radio" name="search-mode" value="all" checked onchange="doSearch()"> всичко</label>
|
||||||
<label><input type="radio" name="search-mode" value="keywords" onchange="doSearch()"> само ключови</label>
|
<label><input type="radio" name="search-mode" value="keywords" onchange="doSearch()"> само ключови</label>
|
||||||
<label><input type="radio" name="search-mode" value="text" onchange="doSearch()"> само извлечения</label>
|
<label><input type="radio" name="search-mode" value="text" onchange="doSearch()"> само секции</label>
|
||||||
</div>
|
</div>
|
||||||
<span class="stats" id="search-stats"></span>
|
<span class="stats" id="search-stats"></span>
|
||||||
<button class="btn-primary" onclick="addSelectedToGenerator()">Добави избраните → Генератор</button>
|
<button class="btn-primary" onclick="addSelectedToGenerator()">Добави избраните → Генератор</button>
|
||||||
</div>
|
</div>
|
||||||
<div class="card-section-label" style="margin-top:0">Ключови думи — клик за филтър</div>
|
<div class="card-section-label" style="margin-top:0">Ключови думи — клик за филтър</div>
|
||||||
<div class="kw-cloud" id="kw-cloud"></div>
|
<div class="kw-cloud" id="kw-cloud"></div>
|
||||||
<div class="card-section-label">Извлечения (резултати)</div>
|
<div class="card-section-label">Секции (резултати)</div>
|
||||||
<div class="results-grid" id="search-results"></div>
|
<div class="results-grid" id="search-results"></div>
|
||||||
</div>
|
</div>
|
||||||
|
|
||||||
@@ -444,7 +463,7 @@
|
|||||||
</div>
|
</div>
|
||||||
<div class="card-section-label">Ключови думи</div>
|
<div class="card-section-label">Ключови думи</div>
|
||||||
<div class="card-tags" id="modal-tags"></div>
|
<div class="card-tags" id="modal-tags"></div>
|
||||||
<div class="card-section-label">Извлечение</div>
|
<div class="card-section-label">Секция</div>
|
||||||
<div class="modal-body" id="modal-body">Зареждане…</div>
|
<div class="modal-body" id="modal-body">Зареждане…</div>
|
||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
@@ -471,9 +490,9 @@
|
|||||||
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
|
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
|
||||||
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
|
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
|
||||||
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||||
<div class="desc">Брой и кодове на извлеченията за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
|
<div class="desc">Брой и кодове на секциите за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
|
||||||
<div class="api-ep"><span class="method post">DELETE</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
<div class="api-ep"><span class="method post">DELETE</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||||
<div class="desc">Изтрива всички извлечения за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
<div class="desc">Изтрива всички секции за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
||||||
|
|
||||||
<p class="api-note">OpenAPI / Swagger: <a href="/docs" target="_blank" rel="noopener">/docs</a> · ReDoc: <a href="/redoc" target="_blank" rel="noopener">/redoc</a></p>
|
<p class="api-note">OpenAPI / Swagger: <a href="/docs" target="_blank" rel="noopener">/docs</a> · ReDoc: <a href="/redoc" target="_blank" rel="noopener">/redoc</a></p>
|
||||||
<p class="api-note">Типичен ERP поток: <code>GET /api/search</code> → избор на <code>code</code> → <code>GET /api/section/{code}</code> за съдържание.</p>
|
<p class="api-note">Типичен ERP поток: <code>GET /api/search</code> → избор на <code>code</code> → <code>GET /api/section/{code}</code> за съдържание.</p>
|
||||||
@@ -507,7 +526,7 @@
|
|||||||
<div class="modal-backdrop" id="wipe-modal" onclick="if(event.target===this)closeWipeModal()">
|
<div class="modal-backdrop" id="wipe-modal" onclick="if(event.target===this)closeWipeModal()">
|
||||||
<div class="modal" role="dialog" aria-modal="true" style="width:min(460px,100%)">
|
<div class="modal" role="dialog" aria-modal="true" style="width:min(460px,100%)">
|
||||||
<div class="modal-head">
|
<div class="modal-head">
|
||||||
<h2>Изтриване на извлечения</h2>
|
<h2>Изтриване на секции</h2>
|
||||||
<button type="button" class="btn-ghost" onclick="closeWipeModal()">✕</button>
|
<button type="button" class="btn-ghost" onclick="closeWipeModal()">✕</button>
|
||||||
</div>
|
</div>
|
||||||
<div class="modal-body" id="wipe-modal-body"></div>
|
<div class="modal-body" id="wipe-modal-body"></div>
|
||||||
@@ -566,7 +585,9 @@ const PAGE_PREFIX = {{ (prefix or 'RIP') | tojson }};
|
|||||||
let scanMode = null;
|
let scanMode = null;
|
||||||
let scanFiles = [];
|
let scanFiles = [];
|
||||||
let fileStats = null;
|
let fileStats = null;
|
||||||
|
let wipeStats = null;
|
||||||
let wipeTarget = null;
|
let wipeTarget = null;
|
||||||
|
let wipeLookupTimer = null;
|
||||||
|
|
||||||
function scanExt(name) {
|
function scanExt(name) {
|
||||||
const n = String(name || '');
|
const n = String(name || '');
|
||||||
@@ -578,12 +599,24 @@ function fileBase(p) {
|
|||||||
const parts = String(p).replace(/\\/g, '/').split('/');
|
const parts = String(p).replace(/\\/g, '/').split('/');
|
||||||
return parts[parts.length - 1] || '';
|
return parts[parts.length - 1] || '';
|
||||||
}
|
}
|
||||||
|
function sectionCountLabel(n) {
|
||||||
|
const x = Number(n) || 0;
|
||||||
|
if (x === 1) return '1 секция';
|
||||||
|
return x + ' секции';
|
||||||
|
}
|
||||||
function setScanStatus(text, kind) {
|
function setScanStatus(text, kind) {
|
||||||
const st = document.getElementById('rescan-status');
|
const st = document.getElementById('rescan-status');
|
||||||
st.textContent = text;
|
st.textContent = text;
|
||||||
st.classList.remove('busy', 'ok', 'error');
|
st.classList.remove('busy', 'ok', 'error');
|
||||||
if (kind) st.classList.add(kind);
|
if (kind) st.classList.add(kind);
|
||||||
}
|
}
|
||||||
|
function setWipeStatus(text, kind) {
|
||||||
|
const st = document.getElementById('wipe-status');
|
||||||
|
if (!st) return;
|
||||||
|
st.textContent = text || '';
|
||||||
|
st.classList.remove('busy', 'ok', 'error');
|
||||||
|
if (kind) st.classList.add(kind);
|
||||||
|
}
|
||||||
function fileCountBg(n) {
|
function fileCountBg(n) {
|
||||||
return n === 1 ? '1 файл' : n + ' файла';
|
return n === 1 ? '1 файл' : n + ' файла';
|
||||||
}
|
}
|
||||||
@@ -592,39 +625,75 @@ function folderNameFrom(files) {
|
|||||||
const top = p.replace(/\\/g, '/').split('/')[0];
|
const top = p.replace(/\\/g, '/').split('/')[0];
|
||||||
return top || (files[0] && files[0].name) || 'папка';
|
return top || (files[0] && files[0].name) || 'папка';
|
||||||
}
|
}
|
||||||
|
function statsLine(stats) {
|
||||||
|
if (!stats) return '';
|
||||||
|
const n = stats.count || 0;
|
||||||
|
let line = 'Секции: ' + n;
|
||||||
|
if (n && stats.first && stats.last) {
|
||||||
|
line += stats.first === stats.last
|
||||||
|
? ' · ' + stats.first
|
||||||
|
: ' · ' + stats.first + ' – ' + stats.last;
|
||||||
|
}
|
||||||
|
if (stats.ok === false && stats.issues && stats.issues.length) {
|
||||||
|
line += ' · ' + stats.issues.join('; ');
|
||||||
|
}
|
||||||
|
return line;
|
||||||
|
}
|
||||||
function renderFileStats() {
|
function renderFileStats() {
|
||||||
const meta = document.getElementById('rescan-filemeta');
|
const meta = document.getElementById('rescan-filemeta');
|
||||||
const wipeBtn = document.getElementById('rescan-wipe-btn');
|
|
||||||
const editorExtra = document.getElementById('editor-file-stats');
|
|
||||||
const n = fileStats ? (fileStats.count || 0) : 0;
|
|
||||||
const fileSelected = scanMode === 'file' && scanFiles[0];
|
const fileSelected = scanMode === 'file' && scanFiles[0];
|
||||||
if (!fileStats || !fileSelected) {
|
if (!fileStats || !fileSelected) {
|
||||||
if (meta) meta.textContent = '';
|
if (meta) { meta.textContent = ''; meta.classList.remove('warn'); }
|
||||||
if (wipeBtn) wipeBtn.hidden = true;
|
|
||||||
if (editorExtra) editorExtra.textContent = '';
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
let line = 'Извлечения: ' + n;
|
|
||||||
if (n && fileStats.first && fileStats.last) {
|
|
||||||
line += fileStats.first === fileStats.last
|
|
||||||
? ' · ' + fileStats.first
|
|
||||||
: ' · ' + fileStats.first + ' – ' + fileStats.last;
|
|
||||||
}
|
|
||||||
if (meta) {
|
if (meta) {
|
||||||
meta.textContent = line;
|
meta.textContent = statsLine(fileStats);
|
||||||
if (fileStats.ok === false && fileStats.issues && fileStats.issues.length) {
|
meta.classList.toggle('warn', fileStats.ok === false);
|
||||||
meta.textContent = line + ' · ' + fileStats.issues.join('; ');
|
|
||||||
meta.classList.add('warn');
|
|
||||||
} else {
|
|
||||||
meta.classList.remove('warn');
|
|
||||||
}
|
}
|
||||||
|
}
|
||||||
|
function renderWipeStats() {
|
||||||
|
const meta = document.getElementById('wipe-filemeta');
|
||||||
|
const btn = document.getElementById('wipe-btn');
|
||||||
|
const name = (document.getElementById('wipe-filename') || {}).value;
|
||||||
|
const base = fileBase((name || '').trim());
|
||||||
|
if (!base) {
|
||||||
|
if (meta) { meta.textContent = ''; meta.classList.remove('warn'); }
|
||||||
|
if (btn) btn.disabled = true;
|
||||||
|
return;
|
||||||
}
|
}
|
||||||
if (wipeBtn) {
|
const n = wipeStats ? (wipeStats.count || 0) : 0;
|
||||||
wipeBtn.hidden = n === 0;
|
if (meta) {
|
||||||
wipeBtn.disabled = false;
|
meta.textContent = wipeStats
|
||||||
|
? (statsLine(wipeStats) + (wipeStats.file ? ' · ' + wipeStats.file : ''))
|
||||||
|
: '';
|
||||||
|
meta.classList.toggle('warn', !!(wipeStats && wipeStats.ok === false));
|
||||||
}
|
}
|
||||||
if (editorExtra) {
|
if (btn) btn.disabled = false;
|
||||||
editorExtra.textContent = fileStats.file ? (' · ' + fileStats.file + ': ' + n) : '';
|
}
|
||||||
|
async function fetchFileStats(name) {
|
||||||
|
const base = fileBase(name);
|
||||||
|
if (!base) return null;
|
||||||
|
try {
|
||||||
|
const url = '/api/file-extractions?file=' + encodeURIComponent(base)
|
||||||
|
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
|
||||||
|
const res = await fetch(url);
|
||||||
|
const data = await res.json().catch(() => ({}));
|
||||||
|
if (!res.ok) {
|
||||||
|
const detail = data.detail;
|
||||||
|
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
|
||||||
|
}
|
||||||
|
return data;
|
||||||
|
} catch (e) {
|
||||||
|
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === base.toLowerCase());
|
||||||
|
const codes = rows.map(r => r.code).filter(Boolean).sort();
|
||||||
|
return {
|
||||||
|
file: base,
|
||||||
|
count: rows.length,
|
||||||
|
first: codes[0] || null,
|
||||||
|
last: codes[codes.length - 1] || null,
|
||||||
|
ok: true,
|
||||||
|
issues: [],
|
||||||
|
};
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
async function loadFileStats(name) {
|
async function loadFileStats(name) {
|
||||||
@@ -633,30 +702,26 @@ async function loadFileStats(name) {
|
|||||||
renderFileStats();
|
renderFileStats();
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
try {
|
fileStats = await fetchFileStats(name);
|
||||||
const url = '/api/file-extractions?file=' + encodeURIComponent(name)
|
|
||||||
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
|
|
||||||
const res = await fetch(url);
|
|
||||||
const data = await res.json().catch(() => ({}));
|
|
||||||
if (!res.ok) {
|
|
||||||
const detail = data.detail;
|
|
||||||
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
|
|
||||||
}
|
|
||||||
fileStats = data;
|
|
||||||
} catch (e) {
|
|
||||||
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === String(name).toLowerCase());
|
|
||||||
const codes = rows.map(r => r.code).filter(Boolean).sort();
|
|
||||||
fileStats = {
|
|
||||||
file: name,
|
|
||||||
count: rows.length,
|
|
||||||
first: codes[0] || null,
|
|
||||||
last: codes[codes.length - 1] || null,
|
|
||||||
ok: true,
|
|
||||||
issues: [],
|
|
||||||
};
|
|
||||||
}
|
|
||||||
renderFileStats();
|
renderFileStats();
|
||||||
}
|
}
|
||||||
|
function onWipeNameInput() {
|
||||||
|
if (wipeLookupTimer) clearTimeout(wipeLookupTimer);
|
||||||
|
const raw = (document.getElementById('wipe-filename').value || '').trim();
|
||||||
|
const base = fileBase(raw);
|
||||||
|
if (!base) {
|
||||||
|
wipeStats = null;
|
||||||
|
renderWipeStats();
|
||||||
|
setWipeStatus('');
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
wipeLookupTimer = setTimeout(async () => {
|
||||||
|
wipeStats = await fetchFileStats(base);
|
||||||
|
renderWipeStats();
|
||||||
|
const n = wipeStats ? (wipeStats.count || 0) : 0;
|
||||||
|
setWipeStatus(n ? ('Намерени ' + sectionCountLabel(n) + '.') : 'Няма секции за този файл.');
|
||||||
|
}, 250);
|
||||||
|
}
|
||||||
async function refreshSections() {
|
async function refreshSections() {
|
||||||
const p = {{ (prefix or '') | tojson }};
|
const p = {{ (prefix or '') | tojson }};
|
||||||
const url = p ? '/api/sections?prefix=' + encodeURIComponent(p) : '/api/sections';
|
const url = p ? '/api/sections?prefix=' + encodeURIComponent(p) : '/api/sections';
|
||||||
@@ -739,12 +804,16 @@ function onScanFolderPicked() {
|
|||||||
}
|
}
|
||||||
|
|
||||||
function confirmWipeFile() {
|
function confirmWipeFile() {
|
||||||
const name = scanFiles[0] && scanFiles[0].name;
|
const raw = (document.getElementById('wipe-filename').value || '').trim();
|
||||||
if (!name) return;
|
const name = fileBase(raw);
|
||||||
const n = (fileStats && fileStats.count) || 0;
|
if (!name) {
|
||||||
|
setWipeStatus('Постави име на файл.', 'error');
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
const n = (wipeStats && wipeStats.count) || 0;
|
||||||
wipeTarget = name;
|
wipeTarget = name;
|
||||||
document.getElementById('wipe-modal-body').textContent =
|
document.getElementById('wipe-modal-body').textContent =
|
||||||
'Ще се изтрият всички извлечения за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
|
'Ще се изтрият всички секции за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
|
||||||
document.getElementById('wipe-modal').classList.add('open');
|
document.getElementById('wipe-modal').classList.add('open');
|
||||||
}
|
}
|
||||||
function closeWipeModal() {
|
function closeWipeModal() {
|
||||||
@@ -773,12 +842,13 @@ async function doWipeFile() {
|
|||||||
document.getElementById('wipe-modal').classList.remove('open');
|
document.getElementById('wipe-modal').classList.remove('open');
|
||||||
wipeTarget = null;
|
wipeTarget = null;
|
||||||
await refreshSections();
|
await refreshSections();
|
||||||
await loadFileStats(name);
|
wipeStats = await fetchFileStats(name);
|
||||||
setScanStatus('Изтрити извлечения: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
|
renderWipeStats();
|
||||||
|
setWipeStatus('Изтрити секции: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
|
||||||
toast('Изтрити: ' + (data.deleted || 0));
|
toast('Изтрити: ' + (data.deleted || 0));
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
toast(e.message, true);
|
toast(e.message, true);
|
||||||
setScanStatus('Грешка: ' + e.message, 'error');
|
setWipeStatus('Грешка: ' + e.message, 'error');
|
||||||
} finally {
|
} finally {
|
||||||
btn.disabled = false;
|
btn.disabled = false;
|
||||||
}
|
}
|
||||||
@@ -788,7 +858,6 @@ async function startRescan() {
|
|||||||
const btn = document.getElementById('rescan-btn');
|
const btn = document.getElementById('rescan-btn');
|
||||||
const fileBtn = document.getElementById('rescan-pick-file');
|
const fileBtn = document.getElementById('rescan-pick-file');
|
||||||
const folderBtn = document.getElementById('rescan-pick-folder');
|
const folderBtn = document.getElementById('rescan-pick-folder');
|
||||||
const wipeBtn = document.getElementById('rescan-wipe-btn');
|
|
||||||
if (!scanFiles.length) {
|
if (!scanFiles.length) {
|
||||||
setScanStatus(scanMode === 'folder' ? 'Няма подходящи файлове в папката.' : 'Избери файл или папка.', 'error');
|
setScanStatus(scanMode === 'folder' ? 'Няма подходящи файлове в папката.' : 'Избери файл или папка.', 'error');
|
||||||
return;
|
return;
|
||||||
@@ -803,7 +872,6 @@ async function startRescan() {
|
|||||||
btn.disabled = true;
|
btn.disabled = true;
|
||||||
fileBtn.disabled = true;
|
fileBtn.disabled = true;
|
||||||
folderBtn.disabled = true;
|
folderBtn.disabled = true;
|
||||||
if (wipeBtn) wipeBtn.disabled = true;
|
|
||||||
setScanStatus('Качване...', 'busy');
|
setScanStatus('Качване...', 'busy');
|
||||||
try {
|
try {
|
||||||
const headers = {};
|
const headers = {};
|
||||||
@@ -823,7 +891,6 @@ async function startRescan() {
|
|||||||
btn.disabled = false;
|
btn.disabled = false;
|
||||||
fileBtn.disabled = false;
|
fileBtn.disabled = false;
|
||||||
folderBtn.disabled = false;
|
folderBtn.disabled = false;
|
||||||
if (wipeBtn) wipeBtn.disabled = false;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -840,16 +907,24 @@ async function pollRescan(jobId) {
|
|||||||
if (typeof data.sections === 'number') msg += ' Записани: ' + data.sections;
|
if (typeof data.sections === 'number') msg += ' Записани: ' + data.sections;
|
||||||
if (files.length === 1) {
|
if (files.length === 1) {
|
||||||
const f = files[0];
|
const f = files[0];
|
||||||
msg = 'Готово. Извлечения: ' + (f.count || 0);
|
msg = 'Готово. Секции: ' + (f.count || 0);
|
||||||
if (f.first && f.last) {
|
if (f.first && f.last) {
|
||||||
msg += f.first === f.last ? ' · ' + f.first : ' · ' + f.first + ' – ' + f.last;
|
msg += f.first === f.last ? ' · ' + f.first : ' · ' + f.first + ' – ' + f.last;
|
||||||
}
|
}
|
||||||
if (f.ok === false && f.issues && f.issues.length) msg += ' · ' + f.issues.join('; ');
|
if (f.ok === false && f.issues && f.issues.length) msg += ' · ' + f.issues.join('; ');
|
||||||
|
if (f.keywords_ok === false || (f.ai_errors || 0) > 0) {
|
||||||
|
msg += ' · ключовите думи са с локален fallback';
|
||||||
|
}
|
||||||
}
|
}
|
||||||
setScanStatus(msg, bad ? 'error' : 'ok');
|
setScanStatus(msg, bad ? 'error' : 'ok');
|
||||||
try {
|
try {
|
||||||
await refreshSections();
|
await refreshSections();
|
||||||
if (scanMode === 'file' && scanFiles[0]) await loadFileStats(scanFiles[0].name);
|
if (scanMode === 'file' && scanFiles[0]) await loadFileStats(scanFiles[0].name);
|
||||||
|
const wipeName = fileBase((document.getElementById('wipe-filename').value || '').trim());
|
||||||
|
if (wipeName) {
|
||||||
|
wipeStats = await fetchFileStats(wipeName);
|
||||||
|
renderWipeStats();
|
||||||
|
}
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
setScanStatus((msg || 'Готово.') + ' Презареди страницата.', bad ? 'error' : 'ok');
|
setScanStatus((msg || 'Готово.') + ' Презареди страницата.', bad ? 'error' : 'ok');
|
||||||
}
|
}
|
||||||
@@ -886,7 +961,9 @@ function renderEditor(rows) {
|
|||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
</td>
|
</td>
|
||||||
<td><span class="src-file" title="${esc(r.source_file)}">${esc(shortPath(r.source_file))}</span></td>
|
<td><span class="src-file" title="Кликни за копиране: ${esc(fileBase(r.source_file))}"
|
||||||
|
data-file="${esc(fileBase(r.source_file))}"
|
||||||
|
onclick="copySourceFile(this.dataset.file, event)">${esc(shortPath(r.source_file))}</span></td>
|
||||||
<td style="font-size:11px;color:var(--muted);font-family:var(--mono);white-space:nowrap">${r.updated_at}</td>
|
<td style="font-size:11px;color:var(--muted);font-family:var(--mono);white-space:nowrap">${r.updated_at}</td>
|
||||||
</tr>
|
</tr>
|
||||||
`).join('');
|
`).join('');
|
||||||
@@ -904,6 +981,21 @@ function filterEditor() {
|
|||||||
renderEditor(filtered);
|
renderEditor(filtered);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function copySourceFile(name, ev) {
|
||||||
|
if (ev) ev.stopPropagation();
|
||||||
|
const base = fileBase(name);
|
||||||
|
if (!base) return;
|
||||||
|
const inp = document.getElementById('wipe-filename');
|
||||||
|
if (inp) inp.value = base;
|
||||||
|
onWipeNameInput();
|
||||||
|
const done = () => toast('Копирано: ' + base);
|
||||||
|
if (navigator.clipboard && navigator.clipboard.writeText) {
|
||||||
|
navigator.clipboard.writeText(base).then(done, done);
|
||||||
|
} else {
|
||||||
|
done();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
function onKwChange(inp) {
|
function onKwChange(inp) {
|
||||||
const code = inp.dataset.code;
|
const code = inp.dataset.code;
|
||||||
inp.classList.add('changed');
|
inp.classList.add('changed');
|
||||||
@@ -1098,7 +1190,7 @@ function doSearch() {
|
|||||||
if (raw && !tokens.length) note = ' · само паразитни думи — показани всички';
|
if (raw && !tokens.length) note = ' · само паразитни думи — показани всички';
|
||||||
|
|
||||||
document.getElementById('search-stats').textContent =
|
document.getElementById('search-stats').textContent =
|
||||||
results.length + ' извлечения' + note;
|
results.length + ' секции' + note;
|
||||||
document.getElementById('search-results').innerHTML = results.map(r => `
|
document.getElementById('search-results').innerHTML = results.map(r => `
|
||||||
<div class="card ${selected.has(r.code)?'selected':''}" onclick="toggleSelect('${r.code}', this)">
|
<div class="card ${selected.has(r.code)?'selected':''}" onclick="toggleSelect('${r.code}', this)">
|
||||||
<div class="card-header">
|
<div class="card-header">
|
||||||
@@ -1112,10 +1204,10 @@ function doSearch() {
|
|||||||
<div class="card-tags">
|
<div class="card-tags">
|
||||||
${(r.keywords||'').split(',').filter(k=>k.trim()).map(k=>`<span class="tag">${esc(k.trim())}</span>`).join('') || '<span class="stats">—</span>'}
|
${(r.keywords||'').split(',').filter(k=>k.trim()).map(k=>`<span class="tag">${esc(k.trim())}</span>`).join('') || '<span class="stats">—</span>'}
|
||||||
</div>
|
</div>
|
||||||
<div class="card-section-label">Извлечение</div>
|
<div class="card-section-label">Секция</div>
|
||||||
<div class="card-text">${r.text_html || esc(r.text||'(няма текст — провери OUTPUT_DIR)')}</div>
|
<div class="card-text">${r.text_html || esc(r.text||'(няма текст — провери OUTPUT_DIR)')}</div>
|
||||||
<div class="card-actions">
|
<div class="card-actions">
|
||||||
<button type="button" class="btn-ghost" onclick="event.stopPropagation(); openSection('${r.code}')">Цяло извлечение</button>
|
<button type="button" class="btn-ghost" onclick="event.stopPropagation(); openSection('${r.code}')">Цяла секция</button>
|
||||||
</div>
|
</div>
|
||||||
<div class="card-footer">${esc(shortPath(r.source_file))} · ${r.char_count||0} знака</div>
|
<div class="card-footer">${esc(shortPath(r.source_file))} · ${r.char_count||0} знака</div>
|
||||||
</div>
|
</div>
|
||||||
@@ -1374,6 +1466,7 @@ try {
|
|||||||
doSearch();
|
doSearch();
|
||||||
renderGenerator();
|
renderGenerator();
|
||||||
initApiTab();
|
initApiTab();
|
||||||
|
renderWipeStats();
|
||||||
} catch (e) {
|
} catch (e) {
|
||||||
console.error('init failed', e);
|
console.error('init failed', e);
|
||||||
toast('Грешка при зареждане на UI: ' + e.message, true);
|
toast('Грешка при зареждане на UI: ' + e.message, true);
|
||||||
|
|||||||
Reference in New Issue
Block a user