diff --git a/help_codes.py b/help_codes.py new file mode 100644 index 0000000..8ea2ae1 --- /dev/null +++ b/help_codes.py @@ -0,0 +1,149 @@ +"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията.""" +from __future__ import annotations + +import re +from pathlib import Path +from typing import Optional + +WIPED_HASH = "0" * 64 +_CODE_RE = re.compile(r"^(.+)_(\d{4})_SEC_(\d{4})$") + + +def make_code(prefix: str, file_index: int, sec_index: int) -> str: + return f"{prefix}_{file_index:04d}_SEC_{sec_index:04d}" + + +def parse_code(code: str) -> Optional[tuple[str, int, int]]: + m = _CODE_RE.match(str(code or "").strip()) + if not m: + return None + return m.group(1), int(m.group(2)), int(m.group(3)) + + +def source_basename(path_str: str) -> str: + return Path(str(path_str or "").replace("\\", "/")).name + + +def source_identity(path: Path, input_dir: Optional[Path] = None) -> str: + """Стабилен ключ за файла: относителен път в input_dir, иначе basename.""" + if input_dir: + try: + return str(path.resolve().relative_to(input_dir.resolve())).replace("\\", "/") + except ValueError: + pass + return path.name + + +def numbering_report(codes: list[str]) -> dict: + """Проверява уникална последователна SEC номерация за един файл.""" + parsed: list[tuple[str, str, int, int]] = [] + issues: list[str] = [] + for c in codes: + p = parse_code(c) + if not p: + issues.append(f"невалиден код: {c}") + continue + parsed.append((c, p[0], p[1], p[2])) + if not codes: + return { + "count": 0, + "ok": True, + "file_index": None, + "first": None, + "last": None, + "codes": [], + "issues": [], + } + indexes = {p[2] for p in parsed} + if len(indexes) > 1: + issues.append( + "различни file_index в кодовете: " + ", ".join(str(i) for i in sorted(indexes)) + ) + file_index = min(indexes) if indexes else None + secs = [p[3] for p in parsed] + tally: dict[int, int] = {} + for s in secs: + tally[s] = tally.get(s, 0) + 1 + dups = sorted(s for s, n in tally.items() if n > 1) + if dups: + issues.append("дублирани SEC: " + ", ".join(f"{s:04d}" for s in dups[:8])) + n = len(parsed) + expected = list(range(1, n + 1)) + actual = sorted(secs) + if actual != expected: + missing = [x for x in expected if x not in tally] + if missing: + issues.append("липсващи SEC: " + ", ".join(f"{s:04d}" for s in missing[:8])) + elif actual: + issues.append(f"номерацията не е 1…{n:04d}") + codes_sorted = [p[0] for p in sorted(parsed, key=lambda x: (x[2], x[3], x[0]))] + return { + "count": len(codes), + "ok": not issues, + "file_index": file_index, + "first": codes_sorted[0] if codes_sorted else None, + "last": codes_sorted[-1] if codes_sorted else None, + "codes": codes_sorted, + "issues": issues, + } + + +def remove_section_outputs( + output_dir: Path, + codes: list[str], + output_paths: Optional[list[str]] = None, +) -> int: + removed = 0 + images_dir = output_dir / "images" + for code in codes: + if not code: + continue + local_txt = output_dir / f"{code}.txt" + try: + if local_txt.exists(): + local_txt.unlink() + removed += 1 + except Exception: + pass + if images_dir.is_dir(): + for img in images_dir.glob(f"{code}_*"): + try: + img.unlink() + removed += 1 + except Exception: + pass + for op in output_paths or []: + if not op: + continue + try: + opath = Path(op) + stem = Path(str(op).replace("\\", "/")).stem + local_txt = output_dir / f"{stem}.txt" + if opath.exists(): + try: + if not local_txt.exists() or opath.resolve() != local_txt.resolve(): + opath.unlink() + removed += 1 + except Exception: + pass + except Exception: + pass + return removed + + +def file_result( + rel: str, + file_index: Optional[int], + codes: list[str], + saved: int = 0, + skipped: bool = False, +) -> dict: + report = numbering_report(codes) + if file_index is not None: + report["file_index"] = file_index + return { + "file": rel, + "saved": saved, + "skipped": skipped, + **report, + } diff --git a/help_processor.py b/help_processor.py index 57bf277..f12dd90 100644 --- a/help_processor.py +++ b/help_processor.py @@ -38,6 +38,16 @@ import anthropic from docx import Document from bs4 import BeautifulSoup +from help_codes import ( + WIPED_HASH, + file_result as _file_result, + make_code, + parse_code, + remove_section_outputs, + source_basename, + source_identity, +) + try: import pdfplumber HAS_PDF = True @@ -204,38 +214,200 @@ class Database: CREATE INDEX IF NOT EXISTS ix_rip_help_sections_prefix ON rip_help_sections(prefix) """) + cur.execute(""" + CREATE INDEX IF NOT EXISTS ix_rip_help_sections_source + ON rip_help_sections(prefix, source_file) + """) + cur.execute( + "ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER" + ) self.conn.commit() log.info("Схемата е проверена / създадена.") + def matching_source_paths(self, prefix: str, identity: str) -> list[str]: + """Всички file_path/source_file за същия help файл (basename, вкл. стари temp пътища).""" + name = source_basename(identity).lower() + if not name: + return [] + found: list[str] = [] + seen: set[str] = set() + for p in self.all_source_files(prefix): + if source_basename(p).lower() == name and p not in seen: + seen.add(p) + found.append(p) + return found + def get_file_hash(self, prefix: str, file_path: str) -> Optional[str]: + paths = self.matching_source_paths(prefix, file_path) or [file_path] cur = self.conn.cursor() cur.execute( - "SELECT file_hash FROM rip_help_files WHERE prefix=%s AND file_path=%s", - (prefix, file_path) + "SELECT file_hash FROM rip_help_files " + "WHERE prefix=%s AND file_path = ANY(%s)", + (prefix, list(paths)), ) - row = cur.fetchone() - return row[0] if row else None + for (h,) in cur.fetchall(): + if not h: + continue + val = str(h).strip() + if val and val != WIPED_HASH: + return val + return None - def upsert_file(self, prefix: str, file_path: str, file_hash: str, section_count: int): + def upsert_file( + self, + prefix: str, + file_path: str, + file_hash: str, + section_count: int, + file_index: Optional[int] = None, + ): + canonical = source_basename(file_path) or file_path + stale = [p for p in self.matching_source_paths(prefix, canonical) if p != canonical] cur = self.conn.cursor() + if stale: + cur.execute( + "DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)", + (prefix, stale), + ) cur.execute(""" - INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count) - VALUES (%s, %s, %s, %s) + INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index) + VALUES (%s, %s, %s, %s, %s) ON CONFLICT (prefix, file_path) DO UPDATE SET file_hash = EXCLUDED.file_hash, section_count= EXCLUDED.section_count, - processed_at = NOW() - """, (prefix, file_path, file_hash, section_count)) + processed_at = NOW(), + file_index = COALESCE(EXCLUDED.file_index, rip_help_files.file_index) + """, (prefix, canonical, file_hash, section_count, file_index)) self.conn.commit() def delete_sections_for_file(self, prefix: str, file_path: str): + paths = self.matching_source_paths(prefix, file_path) or [file_path] cur = self.conn.cursor() cur.execute( - "DELETE FROM rip_help_sections WHERE prefix=%s AND source_file=%s", - (prefix, file_path) + "DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)", + (prefix, list(paths)), ) self.conn.commit() + def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]: + """(code, source_file, output_path) за всички извлечения на файла.""" + paths = self.matching_source_paths(prefix, identity) + if not paths: + return [] + cur = self.conn.cursor() + cur.execute( + "SELECT code, source_file, output_path FROM rip_help_sections " + "WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code", + (prefix, list(paths)), + ) + return [(r[0], r[1], r[2]) for r in cur.fetchall()] + + def file_index_for(self, prefix: str, identity: str) -> Optional[int]: + paths = self.matching_source_paths(prefix, identity) + if not paths: + return None + cur = self.conn.cursor() + cur.execute( + "SELECT file_index FROM rip_help_files " + "WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL", + (prefix, list(paths)), + ) + from_col = [r[0] for r in cur.fetchall() if r[0]] + if from_col: + return min(from_col) + cur.execute( + "SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)", + (prefix, list(paths)), + ) + found: list[int] = [] + for (code,) in cur.fetchall(): + parsed = parse_code(code) + if parsed: + found.append(parsed[1]) + if not found: + return None + tally: dict[int, int] = {} + for idx in found: + tally[idx] = tally.get(idx, 0) + 1 + return max(tally, key=lambda k: (tally[k], -k)) + + def max_file_index(self, prefix: str) -> int: + cur = self.conn.cursor() + cur.execute( + "SELECT COALESCE(MAX(file_index), 0) FROM rip_help_files WHERE prefix=%s", + (prefix,), + ) + m1 = int(cur.fetchone()[0] or 0) + cur.execute("SELECT code FROM rip_help_sections WHERE prefix=%s", (prefix,)) + m2 = 0 + for (code,) in cur.fetchall(): + parsed = parse_code(code) + if parsed: + m2 = max(m2, parsed[1]) + return max(m1, m2) + + def file_index_used_by_others(self, prefix: str, identity: str, idx: int) -> bool: + name = source_basename(identity).lower() + cur = self.conn.cursor() + cur.execute( + "SELECT code, source_file FROM rip_help_sections WHERE prefix=%s", + (prefix,), + ) + for code, src in cur.fetchall(): + parsed = parse_code(code) + if parsed and parsed[1] == idx and source_basename(src).lower() != name: + return True + return False + + def allocate_file_index(self, prefix: str, identity: str) -> int: + existing = self.file_index_for(prefix, identity) + if existing and not self.file_index_used_by_others(prefix, identity, existing): + return existing + return self.max_file_index(prefix) + 1 + + def wipe_extractions_for_file( + self, + prefix: str, + identity: str, + output_dir: Optional[Path] = None, + ) -> dict: + """Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001.""" + rows = self.sections_for_file(prefix, identity) + codes = [r[0] for r in rows] + file_index = self.file_index_for(prefix, identity) + if file_index and self.file_index_used_by_others(prefix, identity, file_index): + file_index = None + paths = self.matching_source_paths(prefix, identity) + if output_dir: + remove_section_outputs(output_dir, codes, [r[2] for r in rows if r[2]]) + self.delete_sections_for_file(prefix, identity) + canonical = source_basename(identity) or identity + stale = list(paths) if paths else [] + cur = self.conn.cursor() + if stale: + cur.execute( + "DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)", + (prefix, stale), + ) + if file_index: + cur.execute(""" + INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index) + VALUES (%s, %s, %s, 0, %s) + ON CONFLICT (prefix, file_path) DO UPDATE SET + file_hash = EXCLUDED.file_hash, + section_count = 0, + processed_at = NOW(), + file_index = EXCLUDED.file_index + """, (prefix, canonical, WIPED_HASH, file_index)) + self.conn.commit() + return { + "file": canonical, + "prefix": prefix, + "deleted": len(codes), + "codes": codes, + "file_index": file_index, + } + def all_source_files(self, prefix: str) -> list[str]: """Връща всички source_file пътища за даден префикс.""" cur = self.conn.cursor() @@ -968,12 +1140,9 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl # ────────────────────────────────────────────── -# Генериране на кодове +# Генериране на кодове — help_codes.py # ────────────────────────────────────────────── -def make_code(prefix: str, file_index: int, sec_index: int) -> str: - return f"{prefix}_{file_index:04d}_SEC_{sec_index:04d}" - # ────────────────────────────────────────────── # Основна обработка @@ -996,46 +1165,56 @@ def process_file( prefix: str = "HLP", force: bool = False, remote_root: Optional[str] = None, -) -> int: - """Обработва един файл. Връща броя записани секции (0 = пропуснат).""" - rel = str(path) + source_key: Optional[str] = None, +) -> dict: + """Обработва един файл. Връща статистика (saved, codes, ok).""" + rel = source_key or path.name fh = file_hash(path) + existing_rows = db.sections_for_file(prefix, rel) + existing_codes = [r[0] for r in existing_rows] + if not force: stored = db.get_file_hash(prefix, rel) if stored == fh: log.info(f" [SKIP] {path.name} (непроменен)") - return 0 + return _file_result(rel, file_index, existing_codes, saved=0, skipped=True) log.info(f" [PROC] {path.name}") ext = path.suffix.lower() parser = PARSERS.get(ext) if not parser: log.warning(f" Неподдържан формат: {ext}") - return 0 + return _file_result(rel, file_index, existing_codes, saved=0) try: sections = parser(path) except Exception as e: log.error(f" Грешка при парсване: {e}") - return 0 + return _file_result(rel, file_index, existing_codes, saved=0) sections = merge_short_sections(sections) - # Изтриваме старите секции за файла при повторна обработка + remove_section_outputs( + output_dir, + existing_codes, + [r[2] for r in existing_rows if r[2]], + ) db.delete_sections_for_file(prefix, rel) images_dir = output_dir / "images" images_dir.mkdir(parents=True, exist_ok=True) saved = 0 - for i, sec in enumerate(sections, 1): + codes: list[str] = [] + for sec in sections: text = clean_text(sec.text) html_text = sec.html_text or "" if not text and not sec.images and not html_text: continue - code = make_code(prefix, file_index, i) + sec_index = saved + 1 + code = make_code(prefix, file_index, sec_index) # Записваме картинките на диск и заменяме placeholder-ите в текста + HTML image_rel_paths: list[str] = [] @@ -1070,7 +1249,7 @@ def process_file( title, keywords = classify_section(client, sec.title, text) except Exception as e: log.warning(f" AI грешка за {code}: {e}") - title, keywords = sec.title or f"Секция {i}", "" + title, keywords = sec.title or f"Секция {sec_index}", "" images_json = json.dumps(image_rel_paths, ensure_ascii=False) ps = ProcessedSection( @@ -1094,11 +1273,12 @@ def process_file( db.insert_section(prefix, ps, _db_output_path(out_path, code, remote_root)) saved += 1 + codes.append(code) log.debug(f" {code}: {title[:60]} ({len(image_rel_paths)} img)") - db.upsert_file(prefix, rel, fh, saved) + db.upsert_file(prefix, rel, fh, saved, file_index=file_index) log.info(f" → {saved} секции записани") - return saved + return _file_result(rel, file_index, codes, saved=saved) _PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$") @@ -1141,19 +1321,28 @@ def process_directory( if remote_root: log.info(f"DB output_path root: {remote_root.replace(chr(92), '/').rstrip('/')}") - current_paths = {str(p) for p in files} + current_ids = {source_identity(p, input_dir) for p in files} + current_names = {source_basename(i).lower() for i in current_ids} + file_results: list[dict] = [] total_sections = 0 try: - for idx, path in enumerate(sorted(files), 1): - n = process_file( + for path in sorted(files): + identity = source_identity(path, input_dir) + idx = db.allocate_file_index(prefix, identity) + info = process_file( path, idx, db, client, output_dir, prefix=prefix, force=force, remote_root=remote_root, + source_key=identity, ) - total_sections += n + file_results.append(info) + total_sections += int(info.get("saved") or 0) if purge_missing: existing = set(db.all_source_files(prefix)) - orphans = sorted(existing - current_paths) + orphans = sorted( + e for e in existing + if source_basename(e).lower() not in current_names + ) if not orphans: log.info(f"Purge: няма orphan записи в БД за prefix={prefix}.") else: @@ -1191,6 +1380,7 @@ def process_directory( db.close() log.info(f"Готово. Prefix={prefix}. Общо нови/обновени секции: {total_sections}") + return {"sections": total_sections, "files": file_results} # ────────────────────────────────────────────── diff --git a/webapp/main.py b/webapp/main.py index 33e74aa..d88c7ca 100644 --- a/webapp/main.py +++ b/webapp/main.py @@ -27,11 +27,19 @@ from pathlib import Path from typing import Optional import psycopg2 -from fastapi import FastAPI, HTTPException, Query, Request +from fastapi import FastAPI, Header, HTTPException, Query, Request from fastapi.responses import FileResponse, HTMLResponse, JSONResponse from fastapi.templating import Jinja2Templates from pydantic import BaseModel +from help_codes import ( + WIPED_HASH, + numbering_report, + parse_code, + remove_section_outputs, + source_basename, +) + # ────────────────────────────────────────────── # Configuration # ────────────────────────────────────────────── @@ -317,6 +325,183 @@ def update_keywords(code: str, body: KeywordsUpdate): return {"ok": True, "code": code} +def _safe_help_filename(file: str) -> str: + name = Path(str(file or "").replace("\\", "/")).name + if not name or name in (".", ".."): + raise HTTPException(400, "file is required") + return name + + +def _ensure_file_index_col(cur): + cur.execute( + "ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER" + ) + + +def _matching_source_paths(cur, prefix: str, identity: str) -> list[str]: + name = source_basename(identity).lower() + if not name: + return [] + cur.execute( + """ + SELECT file_path FROM rip_help_files WHERE prefix=%s + UNION + SELECT source_file FROM rip_help_sections WHERE prefix=%s + """, + (prefix, prefix), + ) + found: list[str] = [] + seen: set[str] = set() + for (p,) in cur.fetchall(): + if p and source_basename(p).lower() == name and p not in seen: + seen.add(p) + found.append(p) + return found + + +def _file_index_for(cur, prefix: str, paths: list[str]) -> Optional[int]: + if not paths: + return None + cur.execute( + "SELECT file_index FROM rip_help_files " + "WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL", + (prefix, list(paths)), + ) + from_col = [r[0] for r in cur.fetchall() if r[0]] + if from_col: + return min(from_col) + cur.execute( + "SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)", + (prefix, list(paths)), + ) + found: list[int] = [] + for (code,) in cur.fetchall(): + parsed = parse_code(code) + if parsed: + found.append(parsed[1]) + if not found: + return None + tally: dict[int, int] = {} + for idx in found: + tally[idx] = tally.get(idx, 0) + 1 + return max(tally, key=lambda k: (tally[k], -k)) + + +@app.get("/api/file-extractions") +def api_file_extractions( + file: str = Query(..., min_length=1), + prefix: str = Query("RIP"), +): + """Брой и номерация на извлеченията за даден help файл.""" + name = _safe_help_filename(file) + conn = db_conn() + try: + cur = conn.cursor() + _ensure_file_index_col(cur) + conn.commit() + paths = _matching_source_paths(cur, prefix, name) + rows = [] + if paths: + cur.execute( + "SELECT code, source_file FROM rip_help_sections " + "WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code", + (prefix, list(paths)), + ) + rows = cur.fetchall() + codes = [r[0] for r in rows] + report = numbering_report(codes) + fi = _file_index_for(cur, prefix, paths) + if fi is not None: + report["file_index"] = fi + return { + "file": name, + "prefix": prefix, + "sources": sorted({source_basename(r[1]) for r in rows if r[1]}), + **report, + } + finally: + conn.close() + + +@app.delete("/api/file-extractions") +def api_delete_file_extractions( + file: str = Query(..., min_length=1), + prefix: str = Query("RIP"), + x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"), + token: Optional[str] = Query(None), +): + """Изтрива всички извлечения за файла. Самият help файл не се пипа.""" + rescan_mod._check_token(x_rescan_token or token) + name = _safe_help_filename(file) + conn = db_conn() + try: + cur = conn.cursor() + _ensure_file_index_col(cur) + paths = _matching_source_paths(cur, prefix, name) + rows = [] + if paths: + cur.execute( + "SELECT code, output_path FROM rip_help_sections " + "WHERE prefix=%s AND source_file = ANY(%s)", + (prefix, list(paths)), + ) + rows = cur.fetchall() + codes = [r[0] for r in rows] + file_index = _file_index_for(cur, prefix, paths) + remove_section_outputs(OUTPUT_DIR, codes, [r[1] for r in rows if r[1]]) + if file_index: + cur.execute( + "SELECT code, source_file FROM rip_help_sections WHERE prefix=%s", + (prefix,), + ) + name_l = name.lower() + shared = False + for code, src in cur.fetchall(): + parsed = parse_code(code) + if ( + parsed + and parsed[1] == file_index + and source_basename(src).lower() != name_l + ): + shared = True + break + if shared: + file_index = None + if paths: + cur.execute( + "DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)", + (prefix, list(paths)), + ) + cur.execute( + "DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)", + (prefix, list(paths)), + ) + if file_index: + cur.execute( + """ + INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index) + VALUES (%s, %s, %s, 0, %s) + ON CONFLICT (prefix, file_path) DO UPDATE SET + file_hash = EXCLUDED.file_hash, + section_count = 0, + processed_at = NOW(), + file_index = EXCLUDED.file_index + """, + (prefix, name, WIPED_HASH, file_index), + ) + conn.commit() + return { + "ok": True, + "file": name, + "prefix": prefix, + "deleted": len(codes), + "codes": codes, + "file_index": file_index, + } + finally: + conn.close() + + @app.get("/healthz", summary="Статус на услугата и БД; output_dir_ok, txt_count.") def healthz(): db_status = "ok" diff --git a/webapp/rescan.py b/webapp/rescan.py index 24f6f0f..fd4a731 100644 --- a/webapp/rescan.py +++ b/webapp/rescan.py @@ -102,8 +102,11 @@ def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool): # Import here so webapp boots even if processor deps fail later from help_processor import process_directory - process_directory( - input_dir=staging, + input_dir = staging / "input" + if not input_dir.is_dir(): + input_dir = staging + summary = process_directory( + input_dir=input_dir, output_dir=out, conn_str=cfg["conn"], api_key=cfg["api_key"], @@ -112,12 +115,20 @@ def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool): purge_missing=False, remote_root=cfg["remote_root"], ) + files = [] + if isinstance(summary, dict): + files = summary.get("files") or [] + sections = summary.get("sections") + else: + sections = summary _set_job( job_id, status="done", message="ok", output_dir=str(out), prefix=prefix, + files=files, + sections=sections, ) except Exception as e: _set_job(job_id, status="error", message=str(e)) diff --git a/webapp/templates/viewer.html b/webapp/templates/viewer.html index b61f6ff..cc84d39 100644 --- a/webapp/templates/viewer.html +++ b/webapp/templates/viewer.html @@ -268,6 +268,12 @@ .rescan-status.busy { color: var(--accent); } .rescan-status.ok { color: var(--accent2); } .rescan-status.error { color: var(--danger); } + .rescan-filemeta { + font-family: var(--mono); font-size: 12px; color: var(--muted); + margin-bottom: 12px; line-height: 1.45; + } + .rescan-filemeta:empty { display: none; } + .rescan-filemeta.warn { color: var(--danger); } .api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; } @media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } } @@ -350,9 +356,12 @@ webkitdirectory directory multiple onchange="onScanFolderPicked()">
+
+
Избери файл или папка.
@@ -362,6 +371,7 @@
+
@@ -460,6 +470,10 @@
multipart: file (един или повече: zip/docx/…), prefix, force. Header опционално: X-Rescan-Token.
GET/api/rescan/{job_id}
Статус на rescan задача (queued / processing / done / error).
+
GET/api/file-extractions?file=…&prefix=RIP +
Брой и кодове на извлеченията за даден файл; ok ако SEC номерацията е уникална и последователна.
+
DELETE/api/file-extractions?file=…&prefix=RIP +
Изтрива всички извлечения за файла (не самия help файл). Header опционално: X-Rescan-Token.

OpenAPI / Swagger: /docs · ReDoc: /redoc

Типичен ERP поток: GET /api/search → избор на code → GET /api/section/{code} за съдържание.

@@ -490,6 +504,20 @@ + +
@@ -534,14 +562,22 @@ function switchTab(name) { } const SCAN_EXT = ['.zip', '.docx', '.doc', '.html', '.htm', '.pdf', '.txt']; +const PAGE_PREFIX = {{ (prefix or 'RIP') | tojson }}; let scanMode = null; let scanFiles = []; +let fileStats = null; +let wipeTarget = null; function scanExt(name) { const n = String(name || ''); const i = n.lastIndexOf('.'); return i >= 0 ? n.slice(i).toLowerCase() : ''; } +function fileBase(p) { + if (!p) return ''; + const parts = String(p).replace(/\\/g, '/').split('/'); + return parts[parts.length - 1] || ''; +} function setScanStatus(text, kind) { const st = document.getElementById('rescan-status'); st.textContent = text; @@ -556,6 +592,84 @@ function folderNameFrom(files) { const top = p.replace(/\\/g, '/').split('/')[0]; return top || (files[0] && files[0].name) || 'папка'; } +function renderFileStats() { + const meta = document.getElementById('rescan-filemeta'); + const wipeBtn = document.getElementById('rescan-wipe-btn'); + const editorExtra = document.getElementById('editor-file-stats'); + const n = fileStats ? (fileStats.count || 0) : 0; + const fileSelected = scanMode === 'file' && scanFiles[0]; + if (!fileStats || !fileSelected) { + if (meta) meta.textContent = ''; + if (wipeBtn) wipeBtn.hidden = true; + if (editorExtra) editorExtra.textContent = ''; + return; + } + let line = 'Извлечения: ' + n; + if (n && fileStats.first && fileStats.last) { + line += fileStats.first === fileStats.last + ? ' · ' + fileStats.first + : ' · ' + fileStats.first + ' – ' + fileStats.last; + } + if (meta) { + meta.textContent = line; + if (fileStats.ok === false && fileStats.issues && fileStats.issues.length) { + meta.textContent = line + ' · ' + fileStats.issues.join('; '); + meta.classList.add('warn'); + } else { + meta.classList.remove('warn'); + } + } + if (wipeBtn) { + wipeBtn.hidden = n === 0; + wipeBtn.disabled = false; + } + if (editorExtra) { + editorExtra.textContent = fileStats.file ? (' · ' + fileStats.file + ': ' + n) : ''; + } +} +async function loadFileStats(name) { + if (!name) { + fileStats = null; + renderFileStats(); + return; + } + try { + const url = '/api/file-extractions?file=' + encodeURIComponent(name) + + '&prefix=' + encodeURIComponent(PAGE_PREFIX); + const res = await fetch(url); + const data = await res.json().catch(() => ({})); + if (!res.ok) { + const detail = data.detail; + throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status)); + } + fileStats = data; + } catch (e) { + const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === String(name).toLowerCase()); + const codes = rows.map(r => r.code).filter(Boolean).sort(); + fileStats = { + file: name, + count: rows.length, + first: codes[0] || null, + last: codes[codes.length - 1] || null, + ok: true, + issues: [], + }; + } + renderFileStats(); +} +async function refreshSections() { + const p = {{ (prefix or '') | tojson }}; + const url = p ? '/api/sections?prefix=' + encodeURIComponent(p) : '/api/sections'; + const res = await fetch(url); + if (!res.ok) throw new Error('HTTP ' + res.status); + ALL = await res.json(); + document.getElementById('total-count').textContent = ALL.length + ' секции'; + const q = document.getElementById('editor-search'); + if (q && q.value) filterEditor(); + else renderEditor(ALL); + renderKwCloud(); + doSearch(); +} function setScanSelection(mode, files) { scanMode = mode; scanFiles = files; @@ -569,18 +683,22 @@ function setScanSelection(mode, files) { if (!mode) { sel.textContent = ''; setScanStatus('Избери файл или папка.'); + loadFileStats(''); return; } if (mode === 'file') { if (!files[0]) { sel.textContent = ''; setScanStatus('Избери файл.', 'error'); + loadFileStats(''); return; } sel.textContent = 'Файл: ' + files[0].name; setScanStatus('Готово за старт.'); + loadFileStats(files[0].name); return; } + loadFileStats(''); if (!files.length) { sel.textContent = 'Папка: няма подходящи файлове'; setScanStatus('Няма .zip, .docx, .html, .pdf или .txt в папката.', 'error'); @@ -620,17 +738,64 @@ function onScanFolderPicked() { setScanSelection('folder', files); } +function confirmWipeFile() { + const name = scanFiles[0] && scanFiles[0].name; + if (!name) return; + const n = (fileStats && fileStats.count) || 0; + wipeTarget = name; + document.getElementById('wipe-modal-body').textContent = + 'Ще се изтрият всички извлечения за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.'; + document.getElementById('wipe-modal').classList.add('open'); +} +function closeWipeModal() { + const modal = document.getElementById('wipe-modal'); + if (modal) modal.classList.remove('open'); + wipeTarget = null; +} +async function doWipeFile() { + const name = wipeTarget; + if (!name) return; + const btn = document.getElementById('wipe-confirm-btn'); + btn.disabled = true; + try { + const token = localStorage.getItem('RESCAN_TOKEN') || ''; + const headers = {}; + if (token) headers['X-Rescan-Token'] = token; + let url = '/api/file-extractions?file=' + encodeURIComponent(name) + + '&prefix=' + encodeURIComponent(PAGE_PREFIX); + if (token) url += '&token=' + encodeURIComponent(token); + const res = await fetch(url, { method: 'DELETE', headers }); + const data = await res.json().catch(() => ({})); + if (!res.ok) { + const detail = data.detail; + throw new Error(typeof detail === 'string' ? detail : (data.message || ('HTTP ' + res.status))); + } + document.getElementById('wipe-modal').classList.remove('open'); + wipeTarget = null; + await refreshSections(); + await loadFileStats(name); + setScanStatus('Изтрити извлечения: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok'); + toast('Изтрити: ' + (data.deleted || 0)); + } catch (e) { + toast(e.message, true); + setScanStatus('Грешка: ' + e.message, 'error'); + } finally { + btn.disabled = false; + } +} + async function startRescan() { const btn = document.getElementById('rescan-btn'); const fileBtn = document.getElementById('rescan-pick-file'); const folderBtn = document.getElementById('rescan-pick-folder'); + const wipeBtn = document.getElementById('rescan-wipe-btn'); if (!scanFiles.length) { setScanStatus(scanMode === 'folder' ? 'Няма подходящи файлове в папката.' : 'Избери файл или папка.', 'error'); return; } const fd = new FormData(); scanFiles.forEach(f => fd.append('file', f)); - fd.append('prefix', {{ (prefix or 'RIP') | tojson }}); + fd.append('prefix', PAGE_PREFIX); fd.append('force', document.getElementById('rescan-force').checked ? 'true' : 'false'); const token = localStorage.getItem('RESCAN_TOKEN') || ''; if (token) fd.append('token', token); @@ -638,6 +803,7 @@ async function startRescan() { btn.disabled = true; fileBtn.disabled = true; folderBtn.disabled = true; + if (wipeBtn) wipeBtn.disabled = true; setScanStatus('Качване...', 'busy'); try { const headers = {}; @@ -657,6 +823,7 @@ async function startRescan() { btn.disabled = false; fileBtn.disabled = false; folderBtn.disabled = false; + if (wipeBtn) wipeBtn.disabled = false; } } @@ -667,8 +834,25 @@ async function pollRescan(jobId) { const data = await res.json(); const line = data.status + (data.message ? ' — ' + data.message : ''); if (data.status === 'done') { - setScanStatus('Готово. Презареждане...', 'ok'); - location.reload(); + const files = data.files || []; + const bad = files.some(f => f && f.ok === false); + let msg = 'Готово.'; + if (typeof data.sections === 'number') msg += ' Записани: ' + data.sections; + if (files.length === 1) { + const f = files[0]; + msg = 'Готово. Извлечения: ' + (f.count || 0); + if (f.first && f.last) { + msg += f.first === f.last ? ' · ' + f.first : ' · ' + f.first + ' – ' + f.last; + } + if (f.ok === false && f.issues && f.issues.length) msg += ' · ' + f.issues.join('; '); + } + setScanStatus(msg, bad ? 'error' : 'ok'); + try { + await refreshSections(); + if (scanMode === 'file' && scanFiles[0]) await loadFileStats(scanFiles[0].name); + } catch (e) { + setScanStatus((msg || 'Готово.') + ' Презареди страницата.', bad ? 'error' : 'ok'); + } return; } if (data.status === 'error') { @@ -964,7 +1148,10 @@ function closeSection() { document.getElementById('section-modal').classList.remove('open'); } document.addEventListener('keydown', (e) => { - if (e.key === 'Escape') closeSection(); + if (e.key === 'Escape') { + closeWipeModal(); + closeSection(); + } }); function toggleSelect(code, el) {