This commit is contained in:
2026-09-11 13:58:06 +03:00
parent dfbc646c51
commit d4dd7be4ef
4 changed files with 484 additions and 126 deletions

View File

@@ -1,4 +1,4 @@
"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията."""
"""Кодове на секции: PREFIX_0001_SEC_0001 и проверка на номерацията."""
from __future__ import annotations
import re

View File

@@ -71,19 +71,27 @@ try:
except AttributeError:
pass
_log_handlers = [logging.StreamHandler(sys.stdout)]
try:
_log_handlers.append(logging.FileHandler("help_processor.log", encoding="utf-8"))
except OSError:
pass
logging.basicConfig(
level=logging.INFO,
format="%(asctime)s %(levelname)-8s %(message)s",
handlers=[
logging.StreamHandler(sys.stdout),
logging.FileHandler("help_processor.log", encoding="utf-8"),
],
handlers=_log_handlers,
)
log = logging.getLogger(__name__)
MIN_SECTION_TOKENS = 60 # секции под тази граница се сливат с предишната
MIN_SECTION_TOKENS = 60 # кратки секции без заглавие се сливат с предишната
MAX_AI_CHARS = 4000 # максимален текст, изпращан към Claude за класификация
AI_MODEL = "claude-haiku-4-5"
AI_MODEL = os.getenv("ANTHROPIC_MODEL", "claude-haiku-4-5")
AI_MODELS = [
AI_MODEL,
"claude-haiku-4-5",
"claude-haiku-4-5-20251001",
"claude-3-5-haiku-latest",
]
MIN_IMAGE_PX = 50 # картинки под NxN px се пропускат (иконки/булети)
@@ -290,7 +298,7 @@ class Database:
self.conn.commit()
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
"""(code, source_file, output_path) за всички извлечения на файла."""
"""(code, source_file, output_path) за всички секции на файла."""
paths = self.matching_source_paths(prefix, identity)
if not paths:
return []
@@ -371,7 +379,7 @@ class Database:
identity: str,
output_dir: Optional[Path] = None,
) -> dict:
"""Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001."""
"""Изтрива всички секции за файла. Запазва file_index; следващият scan почва от SEC_0001."""
rows = self.sections_for_file(prefix, identity)
codes = [r[0] for r in rows]
file_index = self.file_index_for(prefix, identity)
@@ -548,6 +556,96 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
"valign", "width", "height", "bgcolor", "border")
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
_HEADING_TOKEN_RE = re.compile(
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
r"(\d+)?$",
re.I,
)
_HTML_HEADING_CLASS_RE = re.compile(
r"(?:^|[\s_-])(?:heading|заглавие|title|subtitle|msoheading|überschrift)\s*(\d+)?(?:$|[\s_-])",
re.I,
)
_HEADING_LEVEL = {
"heading1": 1, "heading2": 2, "heading3": 3, "heading4": 3, "heading5": 3, "heading6": 3,
"title": 1, "subtitle": 2, "msoheading1": 1, "msoheading2": 2, "msoheading3": 3,
"заглавие": 1, "заглавие1": 1, "заглавие2": 2, "заглавие3": 3,
"подзаглавие": 2, "наименование": 1,
"überschrift": 1, "überschrift1": 1, "überschrift2": 2, "überschrift3": 3,
}
def _compact_style_token(s: str) -> str:
return re.sub(r"[\s_\-]+", "", (s or "").strip().lower())
def _heading_level_from_token(token: str) -> Optional[int]:
t = _compact_style_token(token)
if not t:
return None
if t in _HEADING_LEVEL:
return _HEADING_LEVEL[t]
m = _HEADING_TOKEN_RE.match(t) or _HEADING_TOKEN_RE.match((token or "").strip())
if not m:
return None
n = m.group(2)
if n and n.isdigit():
return min(int(n), 3)
kind = (m.group(1) or "").lower()
if kind in ("subtitle", "подзаглавие"):
return 2
return 1
def _docx_heading_level(para) -> Optional[int]:
"""Heading 1 / Заглавие 1 / style_id / outlineLvl — включително локализиран Word."""
style = getattr(para, "style", None)
seen: set[int] = set()
cur = style
while cur is not None and id(cur) not in seen:
seen.add(id(cur))
for token in (getattr(cur, "style_id", None), getattr(cur, "name", None)):
lvl = _heading_level_from_token(str(token or ""))
if lvl:
return lvl
cur = getattr(cur, "base_style", None)
try:
pPr = para._element.pPr
if pPr is not None and pPr.outlineLvl is not None:
val = int(pPr.outlineLvl.val)
text = (para.text or "").strip()
if 0 <= val <= 2 and text and len(text) < 120:
return min(val + 1, 3)
except Exception:
pass
return None
def _is_bold_heading_text(text: str, runs) -> bool:
if not text or len(text) > 120:
return False
useful = [r for r in (runs or []) if (r.text or "").strip()]
return bool(useful) and all(bool(r.bold) for r in useful)
def _html_heading_level(el) -> Optional[int]:
name = (getattr(el, "name", None) or "").lower()
if name in _HTML_HEADING_MAP:
return _HTML_HEADING_MAP[name]
cls = " ".join(el.get("class") or []) if hasattr(el, "get") else ""
m = _HTML_HEADING_CLASS_RE.search(cls)
if m:
n = m.group(1)
return min(int(n), 3) if n and n.isdigit() else 1
if name in ("p", "div"):
txt = el.get_text(" ", strip=True)
if txt and len(txt) < 120:
inner = "".join(el.stripped_strings)
strong = el.find_all(["b", "strong"]) if hasattr(el, "find_all") else []
strong_txt = " ".join(s.get_text(" ", strip=True) for s in strong).strip()
if strong and strong_txt and strong_txt == inner:
return 2
return None
def _html_block_plain_text(el) -> str:
@@ -600,12 +698,13 @@ def parse_html(path: Path) -> list[Section]:
base_dir = path.parent
body = soup.body or soup
heading_map = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
# Събираме top-level блокови елементи (без да включваме вложените в тях)
consumed = set()
blocks = []
for el in body.find_all(_HTML_BLOCK_TAGS + ["img"]):
for el in body.find_all(_HTML_BLOCK_TAGS + ["img", "div"]):
name = (el.name or "").lower()
if name == "div" and not _html_heading_level(el):
continue
if any(id(par) in consumed for par in el.parents):
continue
consumed.add(id(el))
@@ -620,20 +719,21 @@ def parse_html(path: Path) -> list[Section]:
img_counter = [0]
def flush():
if sec_text or sec_html or sec_images:
if current_title or sec_text or sec_html or sec_images:
sec = Section(current_title, "\n".join(sec_text), current_level)
sec.images = list(sec_images)
sec.html_text = "\n".join(sec_html) if sec_html else None
sections.append(sec)
for el in blocks:
if el.name in heading_map:
heading_lvl = _html_heading_level(el)
if heading_lvl:
txt = el.get_text(" ", strip=True)
if not txt:
continue
flush()
current_title = txt
current_level = heading_map[el.name]
current_level = heading_lvl
sec_text, sec_html, sec_images = [], [], []
continue
@@ -693,24 +793,58 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
return imgs
def _iter_docx_blocks(doc):
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
from docx.oxml.ns import qn
from docx.table import Table
from docx.text.paragraph import Paragraph
body = doc.element.body
for child in body.iterchildren():
if child.tag == qn("w:p"):
yield "p", Paragraph(child, doc)
elif child.tag == qn("w:tbl"):
yield "tbl", Table(child, doc)
def _table_lines(table) -> list[str]:
lines: list[str] = []
try:
rows = table.rows
except Exception:
return lines
for row in rows:
try:
cells = [" ".join((cell.text or "").split()) for cell in row.cells]
except Exception:
continue
cells = [c for c in cells if c]
if cells:
lines.append(" | ".join(cells))
return lines
def parse_docx(path: Path) -> list[Section]:
doc = Document(path)
sections: list[Section] = []
current_title, current_level = "", 1
buf: list[str] = []
sec_images: list[ImageRef] = []
img_counter = [0] # списък за nonlocal-стил мутация
HEADING_STYLES = {"heading 1": 1, "heading 2": 2, "heading 3": 3,
"title": 1, "subtitle": 2}
img_counter = [0]
def flush():
if buf or sec_images:
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), current_level)
sec.images = list(sec_images)
sections.append(sec)
for para in doc.paragraphs:
for kind, block in _iter_docx_blocks(doc):
if kind == "tbl":
for line in _table_lines(block):
buf.append(line)
continue
para = block
style_name = para.style.name.lower() if para.style else ""
text = para.text.strip()
para_imgs = _extract_docx_paragraph_images(para, doc)
@@ -718,12 +852,15 @@ def parse_docx(path: Path) -> list[Section]:
if not text and not para_imgs:
continue
level = HEADING_STYLES.get(style_name)
is_bold_heading = bool(text and len(text) < 120 and not style_name.startswith("list")
and para.runs
and all(run.bold for run in para.runs if run.text.strip()))
level = _docx_heading_level(para)
is_bold_heading = (
not level
and _is_bold_heading_text(text, para.runs)
and not style_name.startswith("list")
and not para_imgs
)
if level or (is_bold_heading and not para_imgs):
if level or is_bold_heading:
flush()
buf, sec_images = [], []
current_title = text
@@ -742,6 +879,11 @@ def parse_docx(path: Path) -> list[Section]:
if not sections:
fallback_text = "\n".join(p.text for p in doc.paragraphs if p.text.strip())
if not fallback_text:
fallback_text = "\n".join(
line for kind, block in _iter_docx_blocks(doc) if kind == "tbl"
for line in _table_lines(block)
)
return [Section("", fallback_text, 0)]
return sections
@@ -986,7 +1128,7 @@ def parse_pdf(path: Path) -> list[Section]:
prev_size = None
def flush():
if buf or sec_images:
if current_title or buf or sec_images:
sec = Section(current_title, "\n".join(buf), 2)
sec.images = list(sec_images)
sections.append(sec)
@@ -1055,7 +1197,39 @@ def parse_txt(path: Path) -> list[Section]:
raw = path.read_bytes()
enc = chardet.detect(raw)["encoding"] or "utf-8"
text = raw.decode(enc, errors="replace")
return [Section("", text, 0)]
return _split_plain_text(text)
_PLAIN_MD_HEADING_RE = re.compile(r"^(#{1,3})\s+(.+)$")
_PLAIN_NUM_HEADING_RE = re.compile(
r"^(?:(?:\d{1,2}|[IVXLC]{1,6}|[А-ЯA-Z])[\.\)])\s+.{2,80}$"
)
def _split_plain_text(text: str) -> list[Section]:
"""Markdown / номерирани заглавия; иначе една секция."""
lines = (text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n")
sections: list[Section] = []
title, level, buf = "", 0, []
def flush():
body = "\n".join(buf).strip()
if title or body:
sections.append(Section(title, body, level))
for line in lines:
raw = line.strip()
md = _PLAIN_MD_HEADING_RE.match(raw)
numbered = bool(_PLAIN_NUM_HEADING_RE.match(raw)) and len(raw) < 120
if md or numbered:
flush()
title = md.group(2).strip() if md else raw
level = len(md.group(1)) if md else 2
buf = []
continue
buf.append(line.rstrip())
flush()
return sections or [Section("", text, 0)]
PARSERS = {
@@ -1073,15 +1247,16 @@ PARSERS = {
# ──────────────────────────────────────────────
def merge_short_sections(sections: list[Section]) -> list[Section]:
"""Слива секции, по-кратки от MIN_SECTION_TOKENS думи, с предишната."""
"""Слива само кратки секции БЕЗ заглавие с предишната. Заглавие = отделна секция."""
result: list[Section] = []
for sec in sections:
words = len(sec.text.split())
if result and words < MIN_SECTION_TOKENS:
words = len((sec.text or "").split())
titled = bool((sec.title or "").strip())
if result and not titled and words < MIN_SECTION_TOKENS:
prev = result[-1]
merged = Section(
prev.title,
prev.text + "\n" + sec.text,
(prev.text + "\n" + sec.text).strip(),
prev.level,
)
merged.images = (prev.images or []) + (sec.images or [])
@@ -1106,6 +1281,77 @@ def clean_text(text: str) -> str:
# AI класификация
# ──────────────────────────────────────────────
def _content_text(msg) -> str:
"""Събира text блокове; Haiku 4.5 може да върне thinking като content[0]."""
parts: list[str] = []
for block in getattr(msg, "content", None) or []:
btype = getattr(block, "type", None)
if btype in (None, "text"):
t = getattr(block, "text", None)
if t:
parts.append(str(t))
elif isinstance(block, dict) and block.get("text"):
parts.append(str(block["text"]))
return "\n".join(parts).strip()
def _parse_classify_json(raw: str) -> Optional[tuple[str, str]]:
raw = (raw or "").strip()
if not raw:
return None
raw = re.sub(r"^```[a-z]*\n?", "", raw)
raw = re.sub(r"\n?```$", "", raw)
candidates = [raw]
m = re.search(r"\{.*\}", raw, re.S)
if m:
candidates.append(m.group(0))
for cand in candidates:
try:
data = json.loads(cand)
except json.JSONDecodeError:
continue
if not isinstance(data, dict):
continue
t = data.get("title", "")
k = data.get("keywords", "")
if isinstance(k, list):
k = ", ".join(str(x).strip() for x in k if str(x).strip())
return str(t)[:200], str(k)[:300]
return None
_FALLBACK_KW_STOP = {
"и", "или", "но", "за", "от", "на", "в", "във", "с", "със", "по", "към", "до",
"при", "след", "преди", "без", "над", "под", "the", "and", "or", "for", "to",
"of", "a", "an", "in", "on", "with", "this", "that", "секция",
}
def fallback_classify(title: str, text: str) -> tuple[str, str]:
t = (title or "").strip()
if not t:
for line in (text or "").splitlines():
line = line.strip()
if 3 <= len(line) <= 80:
t = line
break
if not t:
t = "Секция"
blob = f"{title} {text}"[:1200]
words = re.findall(r"[A-Za-zА-Яа-яЁёІіЇїЄєҐґ0-9\-]{3,}", blob)
seen: list[str] = []
seen_l: set[str] = set()
for w in words:
wl = w.lower()
if wl in _FALLBACK_KW_STOP or wl in seen_l:
continue
seen.append(w)
seen_l.add(wl)
if len(seen) >= 5:
break
return t[:200], ", ".join(seen)
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
snippet = text[:MAX_AI_CHARS]
@@ -1120,23 +1366,31 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl
Върни САМО валиден JSON без markdown, без коментари."""
msg = client.messages.create(
model=AI_MODEL,
max_tokens=200,
messages=[{"role": "user", "content": prompt}]
)
raw = msg.content[0].text.strip()
raw = re.sub(r"^```[a-z]*\n?", "", raw)
raw = re.sub(r"\n?```$", "", raw)
last_err: Optional[Exception] = None
tried: set[str] = set()
for model in AI_MODELS:
if not model or model in tried:
continue
tried.add(model)
try:
data = json.loads(raw)
t = str(data.get("title", title or "Секция"))[:200]
k = str(data.get("keywords", ""))[:300]
return t, k
except json.JSONDecodeError:
log.warning(f"AI върна невалиден JSON: {raw[:120]}")
return title or "Секция", ""
msg = client.messages.create(
model=model,
max_tokens=512,
messages=[{"role": "user", "content": prompt}],
)
raw = _content_text(msg)
parsed = _parse_classify_json(raw)
if parsed:
t, k = parsed
return (t or title or "Секция")[:200], k
last_err = ValueError(f"no JSON in model output: {raw[:120]!r}")
except Exception as e:
last_err = e
log.warning(f"AI classify ({model}) неуспешен: {e}")
continue
if last_err:
log.warning(f"AI върна невалиден резултат, ползваме локален fallback: {last_err}")
return fallback_classify(title, text)
# ──────────────────────────────────────────────
@@ -1207,10 +1461,14 @@ def process_file(
saved = 0
codes: list[str] = []
ai_errors = 0
for sec in sections:
text = clean_text(sec.text)
html_text = sec.html_text or ""
if not text and not sec.images and not html_text:
if (sec.title or "").strip():
text = sec.title.strip()
else:
continue
sec_index = saved + 1
@@ -1249,7 +1507,11 @@ def process_file(
title, keywords = classify_section(client, sec.title, text)
except Exception as e:
log.warning(f" AI грешка за {code}: {e}")
title, keywords = sec.title or f"Секция {sec_index}", ""
title, keywords = fallback_classify(sec.title or f"Секция {sec_index}", text)
ai_errors += 1
if not (keywords or "").strip():
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
ai_errors += 1
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
ps = ProcessedSection(
@@ -1278,7 +1540,10 @@ def process_file(
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
log.info(f" → {saved} секции записани")
return _file_result(rel, file_index, codes, saved=saved)
result = _file_result(rel, file_index, codes, saved=saved)
result["ai_errors"] = ai_errors
result["keywords_ok"] = ai_errors == 0
return result
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")

View File

@@ -392,7 +392,7 @@ def api_file_extractions(
file: str = Query(..., min_length=1),
prefix: str = Query("RIP"),
):
"""Брой и номерация на извлеченията за даден help файл."""
"""Брой и номерация на секциите за даден help файл."""
name = _safe_help_filename(file)
conn = db_conn()
try:
@@ -430,7 +430,7 @@ def api_delete_file_extractions(
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
token: Optional[str] = Query(None),
):
"""Изтрива всички извлечения за файла. Самият help файл не се пипа."""
"""Изтрива всички секции за файла. Самият help файл не се пипа."""
rescan_mod._check_token(x_rescan_token or token)
name = _safe_help_filename(file)
conn = db_conn()

View File

@@ -274,6 +274,10 @@
}
.rescan-filemeta:empty { display: none; }
.rescan-filemeta.warn { color: var(--danger); }
.wipe-box { margin-top: 16px; }
.wipe-box input[type=text] { width: 100%; font-family: var(--mono); }
.src-file { cursor: pointer; }
.src-file:hover { color: var(--accent); text-decoration: underline; }
.api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; }
@media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } }
@@ -360,11 +364,26 @@
<div class="rescan-row">
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
<button type="button" class="btn-danger" id="rescan-wipe-btn" hidden
onclick="confirmWipeFile()">Изтрий извлеченията</button>
</div>
<div class="rescan-status" id="rescan-status">Избери файл или папка.</div>
</div>
<div class="rescan-box wipe-box">
<h3>Изтриване на секции</h3>
<p class="stats" style="margin-bottom:10px;line-height:1.45">
Копирай името на файла от таб „Ключови думи“ и го постави тук.
Изтриват се всички секции за този файл (не самият help файл).
</p>
<div class="rescan-row">
<input type="text" id="wipe-filename" placeholder="напр. atra-manual.docx"
oninput="onWipeNameInput()" onkeydown="if(event.key==='Enter'){confirmWipeFile();event.preventDefault();}">
</div>
<div class="rescan-filemeta" id="wipe-filemeta"></div>
<div class="rescan-row">
<button type="button" class="btn-danger" id="wipe-btn" onclick="confirmWipeFile()" disabled>Изтрий всички секции</button>
</div>
<div class="rescan-status" id="wipe-status"></div>
</div>
</div>
<div id="tab-editor" class="panel">
@@ -397,14 +416,14 @@
<div class="search-mode">
<label><input type="radio" name="search-mode" value="all" checked onchange="doSearch()"> всичко</label>
<label><input type="radio" name="search-mode" value="keywords" onchange="doSearch()"> само ключови</label>
<label><input type="radio" name="search-mode" value="text" onchange="doSearch()"> само извлечения</label>
<label><input type="radio" name="search-mode" value="text" onchange="doSearch()"> само секции</label>
</div>
<span class="stats" id="search-stats"></span>
<button class="btn-primary" onclick="addSelectedToGenerator()">Добави избраните → Генератор</button>
</div>
<div class="card-section-label" style="margin-top:0">Ключови думи — клик за филтър</div>
<div class="kw-cloud" id="kw-cloud"></div>
<div class="card-section-label">Извлечения (резултати)</div>
<div class="card-section-label">Секции (резултати)</div>
<div class="results-grid" id="search-results"></div>
</div>
@@ -444,7 +463,7 @@
</div>
<div class="card-section-label">Ключови думи</div>
<div class="card-tags" id="modal-tags"></div>
<div class="card-section-label">Извлечение</div>
<div class="card-section-label">Секция</div>
<div class="modal-body" id="modal-body">Зареждане…</div>
</div>
</div>
@@ -471,9 +490,9 @@
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&amp;prefix=RIP</code>
<div class="desc">Брой и кодове на извлеченията за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
<div class="desc">Брой и кодове на секциите за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
<div class="api-ep"><span class="method post">DELETE</span><code>/api/file-extractions?file=…&amp;prefix=RIP</code>
<div class="desc">Изтрива всички извлечения за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
<div class="desc">Изтрива всички секции за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
<p class="api-note">OpenAPI / Swagger: <a href="/docs" target="_blank" rel="noopener">/docs</a> · ReDoc: <a href="/redoc" target="_blank" rel="noopener">/redoc</a></p>
<p class="api-note">Типичен ERP поток: <code>GET /api/search</code> → избор на <code>code</code> → <code>GET /api/section/{code}</code> за съдържание.</p>
@@ -507,7 +526,7 @@
<div class="modal-backdrop" id="wipe-modal" onclick="if(event.target===this)closeWipeModal()">
<div class="modal" role="dialog" aria-modal="true" style="width:min(460px,100%)">
<div class="modal-head">
<h2>Изтриване на извлечения</h2>
<h2>Изтриване на секции</h2>
<button type="button" class="btn-ghost" onclick="closeWipeModal()">✕</button>
</div>
<div class="modal-body" id="wipe-modal-body"></div>
@@ -566,7 +585,9 @@ const PAGE_PREFIX = {{ (prefix or 'RIP') | tojson }};
let scanMode = null;
let scanFiles = [];
let fileStats = null;
let wipeStats = null;
let wipeTarget = null;
let wipeLookupTimer = null;
function scanExt(name) {
const n = String(name || '');
@@ -578,12 +599,24 @@ function fileBase(p) {
const parts = String(p).replace(/\\/g, '/').split('/');
return parts[parts.length - 1] || '';
}
function sectionCountLabel(n) {
const x = Number(n) || 0;
if (x === 1) return '1 секция';
return x + ' секции';
}
function setScanStatus(text, kind) {
const st = document.getElementById('rescan-status');
st.textContent = text;
st.classList.remove('busy', 'ok', 'error');
if (kind) st.classList.add(kind);
}
function setWipeStatus(text, kind) {
const st = document.getElementById('wipe-status');
if (!st) return;
st.textContent = text || '';
st.classList.remove('busy', 'ok', 'error');
if (kind) st.classList.add(kind);
}
function fileCountBg(n) {
return n === 1 ? '1 файл' : n + ' файла';
}
@@ -592,39 +625,75 @@ function folderNameFrom(files) {
const top = p.replace(/\\/g, '/').split('/')[0];
return top || (files[0] && files[0].name) || 'папка';
}
function statsLine(stats) {
if (!stats) return '';
const n = stats.count || 0;
let line = 'Секции: ' + n;
if (n && stats.first && stats.last) {
line += stats.first === stats.last
? ' · ' + stats.first
: ' · ' + stats.first + ' – ' + stats.last;
}
if (stats.ok === false && stats.issues && stats.issues.length) {
line += ' · ' + stats.issues.join('; ');
}
return line;
}
function renderFileStats() {
const meta = document.getElementById('rescan-filemeta');
const wipeBtn = document.getElementById('rescan-wipe-btn');
const editorExtra = document.getElementById('editor-file-stats');
const n = fileStats ? (fileStats.count || 0) : 0;
const fileSelected = scanMode === 'file' && scanFiles[0];
if (!fileStats || !fileSelected) {
if (meta) meta.textContent = '';
if (wipeBtn) wipeBtn.hidden = true;
if (editorExtra) editorExtra.textContent = '';
if (meta) { meta.textContent = ''; meta.classList.remove('warn'); }
return;
}
let line = 'Извлечения: ' + n;
if (n && fileStats.first && fileStats.last) {
line += fileStats.first === fileStats.last
? ' · ' + fileStats.first
: ' · ' + fileStats.first + ' – ' + fileStats.last;
}
if (meta) {
meta.textContent = line;
if (fileStats.ok === false && fileStats.issues && fileStats.issues.length) {
meta.textContent = line + ' · ' + fileStats.issues.join('; ');
meta.classList.add('warn');
} else {
meta.classList.remove('warn');
meta.textContent = statsLine(fileStats);
meta.classList.toggle('warn', fileStats.ok === false);
}
}
function renderWipeStats() {
const meta = document.getElementById('wipe-filemeta');
const btn = document.getElementById('wipe-btn');
const name = (document.getElementById('wipe-filename') || {}).value;
const base = fileBase((name || '').trim());
if (!base) {
if (meta) { meta.textContent = ''; meta.classList.remove('warn'); }
if (btn) btn.disabled = true;
return;
}
if (wipeBtn) {
wipeBtn.hidden = n === 0;
wipeBtn.disabled = false;
const n = wipeStats ? (wipeStats.count || 0) : 0;
if (meta) {
meta.textContent = wipeStats
? (statsLine(wipeStats) + (wipeStats.file ? ' · ' + wipeStats.file : ''))
: '';
meta.classList.toggle('warn', !!(wipeStats && wipeStats.ok === false));
}
if (editorExtra) {
editorExtra.textContent = fileStats.file ? (' · ' + fileStats.file + ': ' + n) : '';
if (btn) btn.disabled = false;
}
async function fetchFileStats(name) {
const base = fileBase(name);
if (!base) return null;
try {
const url = '/api/file-extractions?file=' + encodeURIComponent(base)
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
const res = await fetch(url);
const data = await res.json().catch(() => ({}));
if (!res.ok) {
const detail = data.detail;
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
}
return data;
} catch (e) {
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === base.toLowerCase());
const codes = rows.map(r => r.code).filter(Boolean).sort();
return {
file: base,
count: rows.length,
first: codes[0] || null,
last: codes[codes.length - 1] || null,
ok: true,
issues: [],
};
}
}
async function loadFileStats(name) {
@@ -633,30 +702,26 @@ async function loadFileStats(name) {
renderFileStats();
return;
}
try {
const url = '/api/file-extractions?file=' + encodeURIComponent(name)
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
const res = await fetch(url);
const data = await res.json().catch(() => ({}));
if (!res.ok) {
const detail = data.detail;
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
}
fileStats = data;
} catch (e) {
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === String(name).toLowerCase());
const codes = rows.map(r => r.code).filter(Boolean).sort();
fileStats = {
file: name,
count: rows.length,
first: codes[0] || null,
last: codes[codes.length - 1] || null,
ok: true,
issues: [],
};
}
fileStats = await fetchFileStats(name);
renderFileStats();
}
function onWipeNameInput() {
if (wipeLookupTimer) clearTimeout(wipeLookupTimer);
const raw = (document.getElementById('wipe-filename').value || '').trim();
const base = fileBase(raw);
if (!base) {
wipeStats = null;
renderWipeStats();
setWipeStatus('');
return;
}
wipeLookupTimer = setTimeout(async () => {
wipeStats = await fetchFileStats(base);
renderWipeStats();
const n = wipeStats ? (wipeStats.count || 0) : 0;
setWipeStatus(n ? ('Намерени ' + sectionCountLabel(n) + '.') : 'Няма секции за този файл.');
}, 250);
}
async function refreshSections() {
const p = {{ (prefix or '') | tojson }};
const url = p ? '/api/sections?prefix=' + encodeURIComponent(p) : '/api/sections';
@@ -739,12 +804,16 @@ function onScanFolderPicked() {
}
function confirmWipeFile() {
const name = scanFiles[0] && scanFiles[0].name;
if (!name) return;
const n = (fileStats && fileStats.count) || 0;
const raw = (document.getElementById('wipe-filename').value || '').trim();
const name = fileBase(raw);
if (!name) {
setWipeStatus('Постави име на файл.', 'error');
return;
}
const n = (wipeStats && wipeStats.count) || 0;
wipeTarget = name;
document.getElementById('wipe-modal-body').textContent =
'Ще се изтрият всички извлечения за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
'Ще се изтрият всички секции за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
document.getElementById('wipe-modal').classList.add('open');
}
function closeWipeModal() {
@@ -773,12 +842,13 @@ async function doWipeFile() {
document.getElementById('wipe-modal').classList.remove('open');
wipeTarget = null;
await refreshSections();
await loadFileStats(name);
setScanStatus('Изтрити извлечения: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
wipeStats = await fetchFileStats(name);
renderWipeStats();
setWipeStatus('Изтрити секции: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
toast('Изтрити: ' + (data.deleted || 0));
} catch (e) {
toast(e.message, true);
setScanStatus('Грешка: ' + e.message, 'error');
setWipeStatus('Грешка: ' + e.message, 'error');
} finally {
btn.disabled = false;
}
@@ -788,7 +858,6 @@ async function startRescan() {
const btn = document.getElementById('rescan-btn');
const fileBtn = document.getElementById('rescan-pick-file');
const folderBtn = document.getElementById('rescan-pick-folder');
const wipeBtn = document.getElementById('rescan-wipe-btn');
if (!scanFiles.length) {
setScanStatus(scanMode === 'folder' ? 'Няма подходящи файлове в папката.' : 'Избери файл или папка.', 'error');
return;
@@ -803,7 +872,6 @@ async function startRescan() {
btn.disabled = true;
fileBtn.disabled = true;
folderBtn.disabled = true;
if (wipeBtn) wipeBtn.disabled = true;
setScanStatus('Качване...', 'busy');
try {
const headers = {};
@@ -823,7 +891,6 @@ async function startRescan() {
btn.disabled = false;
fileBtn.disabled = false;
folderBtn.disabled = false;
if (wipeBtn) wipeBtn.disabled = false;
}
}
@@ -840,16 +907,24 @@ async function pollRescan(jobId) {
if (typeof data.sections === 'number') msg += ' Записани: ' + data.sections;
if (files.length === 1) {
const f = files[0];
msg = 'Готово. Извлечения: ' + (f.count || 0);
msg = 'Готово. Секции: ' + (f.count || 0);
if (f.first && f.last) {
msg += f.first === f.last ? ' · ' + f.first : ' · ' + f.first + ' – ' + f.last;
}
if (f.ok === false && f.issues && f.issues.length) msg += ' · ' + f.issues.join('; ');
if (f.keywords_ok === false || (f.ai_errors || 0) > 0) {
msg += ' · ключовите думи са с локален fallback';
}
}
setScanStatus(msg, bad ? 'error' : 'ok');
try {
await refreshSections();
if (scanMode === 'file' && scanFiles[0]) await loadFileStats(scanFiles[0].name);
const wipeName = fileBase((document.getElementById('wipe-filename').value || '').trim());
if (wipeName) {
wipeStats = await fetchFileStats(wipeName);
renderWipeStats();
}
} catch (e) {
setScanStatus((msg || 'Готово.') + ' Презареди страницата.', bad ? 'error' : 'ok');
}
@@ -886,7 +961,9 @@ function renderEditor(rows) {
</div>
</div>
</td>
<td><span class="src-file" title="${esc(r.source_file)}">${esc(shortPath(r.source_file))}</span></td>
<td><span class="src-file" title="Кликни за копиране: ${esc(fileBase(r.source_file))}"
data-file="${esc(fileBase(r.source_file))}"
onclick="copySourceFile(this.dataset.file, event)">${esc(shortPath(r.source_file))}</span></td>
<td style="font-size:11px;color:var(--muted);font-family:var(--mono);white-space:nowrap">${r.updated_at}</td>
</tr>
`).join('');
@@ -904,6 +981,21 @@ function filterEditor() {
renderEditor(filtered);
}
function copySourceFile(name, ev) {
if (ev) ev.stopPropagation();
const base = fileBase(name);
if (!base) return;
const inp = document.getElementById('wipe-filename');
if (inp) inp.value = base;
onWipeNameInput();
const done = () => toast('Копирано: ' + base);
if (navigator.clipboard && navigator.clipboard.writeText) {
navigator.clipboard.writeText(base).then(done, done);
} else {
done();
}
}
function onKwChange(inp) {
const code = inp.dataset.code;
inp.classList.add('changed');
@@ -1098,7 +1190,7 @@ function doSearch() {
if (raw && !tokens.length) note = ' · само паразитни думи — показани всички';
document.getElementById('search-stats').textContent =
results.length + ' извлечения' + note;
results.length + ' секции' + note;
document.getElementById('search-results').innerHTML = results.map(r => `
<div class="card ${selected.has(r.code)?'selected':''}" onclick="toggleSelect('${r.code}', this)">
<div class="card-header">
@@ -1112,10 +1204,10 @@ function doSearch() {
<div class="card-tags">
${(r.keywords||'').split(',').filter(k=>k.trim()).map(k=>`<span class="tag">${esc(k.trim())}</span>`).join('') || '<span class="stats">—</span>'}
</div>
<div class="card-section-label">Извлечение</div>
<div class="card-section-label">Секция</div>
<div class="card-text">${r.text_html || esc(r.text||'(няма текст — провери OUTPUT_DIR)')}</div>
<div class="card-actions">
<button type="button" class="btn-ghost" onclick="event.stopPropagation(); openSection('${r.code}')">Цяло извлечение</button>
<button type="button" class="btn-ghost" onclick="event.stopPropagation(); openSection('${r.code}')">Цяла секция</button>
</div>
<div class="card-footer">${esc(shortPath(r.source_file))} &nbsp;·&nbsp; ${r.char_count||0} знака</div>
</div>
@@ -1374,6 +1466,7 @@ try {
doSearch();
renderGenerator();
initApiTab();
renderWipeStats();
} catch (e) {
console.error('init failed', e);
toast('Грешка при зареждане на UI: ' + e.message, true);