Give each help file unique sequential extraction codes, show the count, and allow wiping all extractions for a file while testing.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
149
help_codes.py
Normal file
149
help_codes.py
Normal file
@@ -0,0 +1,149 @@
|
||||
"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията."""
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
WIPED_HASH = "0" * 64
|
||||
_CODE_RE = re.compile(r"^(.+)_(\d{4})_SEC_(\d{4})$")
|
||||
|
||||
|
||||
def make_code(prefix: str, file_index: int, sec_index: int) -> str:
|
||||
return f"{prefix}_{file_index:04d}_SEC_{sec_index:04d}"
|
||||
|
||||
|
||||
def parse_code(code: str) -> Optional[tuple[str, int, int]]:
|
||||
m = _CODE_RE.match(str(code or "").strip())
|
||||
if not m:
|
||||
return None
|
||||
return m.group(1), int(m.group(2)), int(m.group(3))
|
||||
|
||||
|
||||
def source_basename(path_str: str) -> str:
|
||||
return Path(str(path_str or "").replace("\\", "/")).name
|
||||
|
||||
|
||||
def source_identity(path: Path, input_dir: Optional[Path] = None) -> str:
|
||||
"""Стабилен ключ за файла: относителен път в input_dir, иначе basename."""
|
||||
if input_dir:
|
||||
try:
|
||||
return str(path.resolve().relative_to(input_dir.resolve())).replace("\\", "/")
|
||||
except ValueError:
|
||||
pass
|
||||
return path.name
|
||||
|
||||
|
||||
def numbering_report(codes: list[str]) -> dict:
|
||||
"""Проверява уникална последователна SEC номерация за един файл."""
|
||||
parsed: list[tuple[str, str, int, int]] = []
|
||||
issues: list[str] = []
|
||||
for c in codes:
|
||||
p = parse_code(c)
|
||||
if not p:
|
||||
issues.append(f"невалиден код: {c}")
|
||||
continue
|
||||
parsed.append((c, p[0], p[1], p[2]))
|
||||
if not codes:
|
||||
return {
|
||||
"count": 0,
|
||||
"ok": True,
|
||||
"file_index": None,
|
||||
"first": None,
|
||||
"last": None,
|
||||
"codes": [],
|
||||
"issues": [],
|
||||
}
|
||||
indexes = {p[2] for p in parsed}
|
||||
if len(indexes) > 1:
|
||||
issues.append(
|
||||
"различни file_index в кодовете: " + ", ".join(str(i) for i in sorted(indexes))
|
||||
)
|
||||
file_index = min(indexes) if indexes else None
|
||||
secs = [p[3] for p in parsed]
|
||||
tally: dict[int, int] = {}
|
||||
for s in secs:
|
||||
tally[s] = tally.get(s, 0) + 1
|
||||
dups = sorted(s for s, n in tally.items() if n > 1)
|
||||
if dups:
|
||||
issues.append("дублирани SEC: " + ", ".join(f"{s:04d}" for s in dups[:8]))
|
||||
n = len(parsed)
|
||||
expected = list(range(1, n + 1))
|
||||
actual = sorted(secs)
|
||||
if actual != expected:
|
||||
missing = [x for x in expected if x not in tally]
|
||||
if missing:
|
||||
issues.append("липсващи SEC: " + ", ".join(f"{s:04d}" for s in missing[:8]))
|
||||
elif actual:
|
||||
issues.append(f"номерацията не е 1…{n:04d}")
|
||||
codes_sorted = [p[0] for p in sorted(parsed, key=lambda x: (x[2], x[3], x[0]))]
|
||||
return {
|
||||
"count": len(codes),
|
||||
"ok": not issues,
|
||||
"file_index": file_index,
|
||||
"first": codes_sorted[0] if codes_sorted else None,
|
||||
"last": codes_sorted[-1] if codes_sorted else None,
|
||||
"codes": codes_sorted,
|
||||
"issues": issues,
|
||||
}
|
||||
|
||||
|
||||
def remove_section_outputs(
|
||||
output_dir: Path,
|
||||
codes: list[str],
|
||||
output_paths: Optional[list[str]] = None,
|
||||
) -> int:
|
||||
removed = 0
|
||||
images_dir = output_dir / "images"
|
||||
for code in codes:
|
||||
if not code:
|
||||
continue
|
||||
local_txt = output_dir / f"{code}.txt"
|
||||
try:
|
||||
if local_txt.exists():
|
||||
local_txt.unlink()
|
||||
removed += 1
|
||||
except Exception:
|
||||
pass
|
||||
if images_dir.is_dir():
|
||||
for img in images_dir.glob(f"{code}_*"):
|
||||
try:
|
||||
img.unlink()
|
||||
removed += 1
|
||||
except Exception:
|
||||
pass
|
||||
for op in output_paths or []:
|
||||
if not op:
|
||||
continue
|
||||
try:
|
||||
opath = Path(op)
|
||||
stem = Path(str(op).replace("\\", "/")).stem
|
||||
local_txt = output_dir / f"{stem}.txt"
|
||||
if opath.exists():
|
||||
try:
|
||||
if not local_txt.exists() or opath.resolve() != local_txt.resolve():
|
||||
opath.unlink()
|
||||
removed += 1
|
||||
except Exception:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
return removed
|
||||
|
||||
|
||||
def file_result(
|
||||
rel: str,
|
||||
file_index: Optional[int],
|
||||
codes: list[str],
|
||||
saved: int = 0,
|
||||
skipped: bool = False,
|
||||
) -> dict:
|
||||
report = numbering_report(codes)
|
||||
if file_index is not None:
|
||||
report["file_index"] = file_index
|
||||
return {
|
||||
"file": rel,
|
||||
"saved": saved,
|
||||
"skipped": skipped,
|
||||
**report,
|
||||
}
|
||||
@@ -38,6 +38,16 @@ import anthropic
|
||||
from docx import Document
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
from help_codes import (
|
||||
WIPED_HASH,
|
||||
file_result as _file_result,
|
||||
make_code,
|
||||
parse_code,
|
||||
remove_section_outputs,
|
||||
source_basename,
|
||||
source_identity,
|
||||
)
|
||||
|
||||
try:
|
||||
import pdfplumber
|
||||
HAS_PDF = True
|
||||
@@ -204,38 +214,200 @@ class Database:
|
||||
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_prefix
|
||||
ON rip_help_sections(prefix)
|
||||
""")
|
||||
cur.execute("""
|
||||
CREATE INDEX IF NOT EXISTS ix_rip_help_sections_source
|
||||
ON rip_help_sections(prefix, source_file)
|
||||
""")
|
||||
cur.execute(
|
||||
"ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER"
|
||||
)
|
||||
self.conn.commit()
|
||||
log.info("Схемата е проверена / създадена.")
|
||||
|
||||
def matching_source_paths(self, prefix: str, identity: str) -> list[str]:
|
||||
"""Всички file_path/source_file за същия help файл (basename, вкл. стари temp пътища)."""
|
||||
name = source_basename(identity).lower()
|
||||
if not name:
|
||||
return []
|
||||
found: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for p in self.all_source_files(prefix):
|
||||
if source_basename(p).lower() == name and p not in seen:
|
||||
seen.add(p)
|
||||
found.append(p)
|
||||
return found
|
||||
|
||||
def get_file_hash(self, prefix: str, file_path: str) -> Optional[str]:
|
||||
paths = self.matching_source_paths(prefix, file_path) or [file_path]
|
||||
cur = self.conn.cursor()
|
||||
cur.execute(
|
||||
"SELECT file_hash FROM rip_help_files WHERE prefix=%s AND file_path=%s",
|
||||
(prefix, file_path)
|
||||
"SELECT file_hash FROM rip_help_files "
|
||||
"WHERE prefix=%s AND file_path = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
row = cur.fetchone()
|
||||
return row[0] if row else None
|
||||
for (h,) in cur.fetchall():
|
||||
if not h:
|
||||
continue
|
||||
val = str(h).strip()
|
||||
if val and val != WIPED_HASH:
|
||||
return val
|
||||
return None
|
||||
|
||||
def upsert_file(self, prefix: str, file_path: str, file_hash: str, section_count: int):
|
||||
def upsert_file(
|
||||
self,
|
||||
prefix: str,
|
||||
file_path: str,
|
||||
file_hash: str,
|
||||
section_count: int,
|
||||
file_index: Optional[int] = None,
|
||||
):
|
||||
canonical = source_basename(file_path) or file_path
|
||||
stale = [p for p in self.matching_source_paths(prefix, canonical) if p != canonical]
|
||||
cur = self.conn.cursor()
|
||||
if stale:
|
||||
cur.execute(
|
||||
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
|
||||
(prefix, stale),
|
||||
)
|
||||
cur.execute("""
|
||||
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count)
|
||||
VALUES (%s, %s, %s, %s)
|
||||
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
|
||||
VALUES (%s, %s, %s, %s, %s)
|
||||
ON CONFLICT (prefix, file_path) DO UPDATE SET
|
||||
file_hash = EXCLUDED.file_hash,
|
||||
section_count= EXCLUDED.section_count,
|
||||
processed_at = NOW()
|
||||
""", (prefix, file_path, file_hash, section_count))
|
||||
processed_at = NOW(),
|
||||
file_index = COALESCE(EXCLUDED.file_index, rip_help_files.file_index)
|
||||
""", (prefix, canonical, file_hash, section_count, file_index))
|
||||
self.conn.commit()
|
||||
|
||||
def delete_sections_for_file(self, prefix: str, file_path: str):
|
||||
paths = self.matching_source_paths(prefix, file_path) or [file_path]
|
||||
cur = self.conn.cursor()
|
||||
cur.execute(
|
||||
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file=%s",
|
||||
(prefix, file_path)
|
||||
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
self.conn.commit()
|
||||
|
||||
def sections_for_file(self, prefix: str, identity: str) -> list[tuple[str, str, Optional[str]]]:
|
||||
"""(code, source_file, output_path) за всички извлечения на файла."""
|
||||
paths = self.matching_source_paths(prefix, identity)
|
||||
if not paths:
|
||||
return []
|
||||
cur = self.conn.cursor()
|
||||
cur.execute(
|
||||
"SELECT code, source_file, output_path FROM rip_help_sections "
|
||||
"WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
return [(r[0], r[1], r[2]) for r in cur.fetchall()]
|
||||
|
||||
def file_index_for(self, prefix: str, identity: str) -> Optional[int]:
|
||||
paths = self.matching_source_paths(prefix, identity)
|
||||
if not paths:
|
||||
return None
|
||||
cur = self.conn.cursor()
|
||||
cur.execute(
|
||||
"SELECT file_index FROM rip_help_files "
|
||||
"WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
from_col = [r[0] for r in cur.fetchall() if r[0]]
|
||||
if from_col:
|
||||
return min(from_col)
|
||||
cur.execute(
|
||||
"SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
found: list[int] = []
|
||||
for (code,) in cur.fetchall():
|
||||
parsed = parse_code(code)
|
||||
if parsed:
|
||||
found.append(parsed[1])
|
||||
if not found:
|
||||
return None
|
||||
tally: dict[int, int] = {}
|
||||
for idx in found:
|
||||
tally[idx] = tally.get(idx, 0) + 1
|
||||
return max(tally, key=lambda k: (tally[k], -k))
|
||||
|
||||
def max_file_index(self, prefix: str) -> int:
|
||||
cur = self.conn.cursor()
|
||||
cur.execute(
|
||||
"SELECT COALESCE(MAX(file_index), 0) FROM rip_help_files WHERE prefix=%s",
|
||||
(prefix,),
|
||||
)
|
||||
m1 = int(cur.fetchone()[0] or 0)
|
||||
cur.execute("SELECT code FROM rip_help_sections WHERE prefix=%s", (prefix,))
|
||||
m2 = 0
|
||||
for (code,) in cur.fetchall():
|
||||
parsed = parse_code(code)
|
||||
if parsed:
|
||||
m2 = max(m2, parsed[1])
|
||||
return max(m1, m2)
|
||||
|
||||
def file_index_used_by_others(self, prefix: str, identity: str, idx: int) -> bool:
|
||||
name = source_basename(identity).lower()
|
||||
cur = self.conn.cursor()
|
||||
cur.execute(
|
||||
"SELECT code, source_file FROM rip_help_sections WHERE prefix=%s",
|
||||
(prefix,),
|
||||
)
|
||||
for code, src in cur.fetchall():
|
||||
parsed = parse_code(code)
|
||||
if parsed and parsed[1] == idx and source_basename(src).lower() != name:
|
||||
return True
|
||||
return False
|
||||
|
||||
def allocate_file_index(self, prefix: str, identity: str) -> int:
|
||||
existing = self.file_index_for(prefix, identity)
|
||||
if existing and not self.file_index_used_by_others(prefix, identity, existing):
|
||||
return existing
|
||||
return self.max_file_index(prefix) + 1
|
||||
|
||||
def wipe_extractions_for_file(
|
||||
self,
|
||||
prefix: str,
|
||||
identity: str,
|
||||
output_dir: Optional[Path] = None,
|
||||
) -> dict:
|
||||
"""Изтрива всички извлечения за файла. Запазва file_index; следващият scan почва от SEC_0001."""
|
||||
rows = self.sections_for_file(prefix, identity)
|
||||
codes = [r[0] for r in rows]
|
||||
file_index = self.file_index_for(prefix, identity)
|
||||
if file_index and self.file_index_used_by_others(prefix, identity, file_index):
|
||||
file_index = None
|
||||
paths = self.matching_source_paths(prefix, identity)
|
||||
if output_dir:
|
||||
remove_section_outputs(output_dir, codes, [r[2] for r in rows if r[2]])
|
||||
self.delete_sections_for_file(prefix, identity)
|
||||
canonical = source_basename(identity) or identity
|
||||
stale = list(paths) if paths else []
|
||||
cur = self.conn.cursor()
|
||||
if stale:
|
||||
cur.execute(
|
||||
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
|
||||
(prefix, stale),
|
||||
)
|
||||
if file_index:
|
||||
cur.execute("""
|
||||
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
|
||||
VALUES (%s, %s, %s, 0, %s)
|
||||
ON CONFLICT (prefix, file_path) DO UPDATE SET
|
||||
file_hash = EXCLUDED.file_hash,
|
||||
section_count = 0,
|
||||
processed_at = NOW(),
|
||||
file_index = EXCLUDED.file_index
|
||||
""", (prefix, canonical, WIPED_HASH, file_index))
|
||||
self.conn.commit()
|
||||
return {
|
||||
"file": canonical,
|
||||
"prefix": prefix,
|
||||
"deleted": len(codes),
|
||||
"codes": codes,
|
||||
"file_index": file_index,
|
||||
}
|
||||
|
||||
def all_source_files(self, prefix: str) -> list[str]:
|
||||
"""Връща всички source_file пътища за даден префикс."""
|
||||
cur = self.conn.cursor()
|
||||
@@ -968,12 +1140,9 @@ def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tupl
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Генериране на кодове
|
||||
# Генериране на кодове — help_codes.py
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
def make_code(prefix: str, file_index: int, sec_index: int) -> str:
|
||||
return f"{prefix}_{file_index:04d}_SEC_{sec_index:04d}"
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Основна обработка
|
||||
@@ -996,46 +1165,56 @@ def process_file(
|
||||
prefix: str = "HLP",
|
||||
force: bool = False,
|
||||
remote_root: Optional[str] = None,
|
||||
) -> int:
|
||||
"""Обработва един файл. Връща броя записани секции (0 = пропуснат)."""
|
||||
rel = str(path)
|
||||
source_key: Optional[str] = None,
|
||||
) -> dict:
|
||||
"""Обработва един файл. Връща статистика (saved, codes, ok)."""
|
||||
rel = source_key or path.name
|
||||
fh = file_hash(path)
|
||||
|
||||
existing_rows = db.sections_for_file(prefix, rel)
|
||||
existing_codes = [r[0] for r in existing_rows]
|
||||
|
||||
if not force:
|
||||
stored = db.get_file_hash(prefix, rel)
|
||||
if stored == fh:
|
||||
log.info(f" [SKIP] {path.name} (непроменен)")
|
||||
return 0
|
||||
return _file_result(rel, file_index, existing_codes, saved=0, skipped=True)
|
||||
|
||||
log.info(f" [PROC] {path.name}")
|
||||
ext = path.suffix.lower()
|
||||
parser = PARSERS.get(ext)
|
||||
if not parser:
|
||||
log.warning(f" Неподдържан формат: {ext}")
|
||||
return 0
|
||||
return _file_result(rel, file_index, existing_codes, saved=0)
|
||||
|
||||
try:
|
||||
sections = parser(path)
|
||||
except Exception as e:
|
||||
log.error(f" Грешка при парсване: {e}")
|
||||
return 0
|
||||
return _file_result(rel, file_index, existing_codes, saved=0)
|
||||
|
||||
sections = merge_short_sections(sections)
|
||||
|
||||
# Изтриваме старите секции за файла при повторна обработка
|
||||
remove_section_outputs(
|
||||
output_dir,
|
||||
existing_codes,
|
||||
[r[2] for r in existing_rows if r[2]],
|
||||
)
|
||||
db.delete_sections_for_file(prefix, rel)
|
||||
|
||||
images_dir = output_dir / "images"
|
||||
images_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
saved = 0
|
||||
for i, sec in enumerate(sections, 1):
|
||||
codes: list[str] = []
|
||||
for sec in sections:
|
||||
text = clean_text(sec.text)
|
||||
html_text = sec.html_text or ""
|
||||
if not text and not sec.images and not html_text:
|
||||
continue
|
||||
|
||||
code = make_code(prefix, file_index, i)
|
||||
sec_index = saved + 1
|
||||
code = make_code(prefix, file_index, sec_index)
|
||||
|
||||
# Записваме картинките на диск и заменяме placeholder-ите в текста + HTML
|
||||
image_rel_paths: list[str] = []
|
||||
@@ -1070,7 +1249,7 @@ def process_file(
|
||||
title, keywords = classify_section(client, sec.title, text)
|
||||
except Exception as e:
|
||||
log.warning(f" AI грешка за {code}: {e}")
|
||||
title, keywords = sec.title or f"Секция {i}", ""
|
||||
title, keywords = sec.title or f"Секция {sec_index}", ""
|
||||
|
||||
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
||||
ps = ProcessedSection(
|
||||
@@ -1094,11 +1273,12 @@ def process_file(
|
||||
|
||||
db.insert_section(prefix, ps, _db_output_path(out_path, code, remote_root))
|
||||
saved += 1
|
||||
codes.append(code)
|
||||
log.debug(f" {code}: {title[:60]} ({len(image_rel_paths)} img)")
|
||||
|
||||
db.upsert_file(prefix, rel, fh, saved)
|
||||
db.upsert_file(prefix, rel, fh, saved, file_index=file_index)
|
||||
log.info(f" → {saved} секции записани")
|
||||
return saved
|
||||
return _file_result(rel, file_index, codes, saved=saved)
|
||||
|
||||
|
||||
_PREFIX_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,49}$")
|
||||
@@ -1141,19 +1321,28 @@ def process_directory(
|
||||
if remote_root:
|
||||
log.info(f"DB output_path root: {remote_root.replace(chr(92), '/').rstrip('/')}")
|
||||
|
||||
current_paths = {str(p) for p in files}
|
||||
current_ids = {source_identity(p, input_dir) for p in files}
|
||||
current_names = {source_basename(i).lower() for i in current_ids}
|
||||
file_results: list[dict] = []
|
||||
total_sections = 0
|
||||
try:
|
||||
for idx, path in enumerate(sorted(files), 1):
|
||||
n = process_file(
|
||||
for path in sorted(files):
|
||||
identity = source_identity(path, input_dir)
|
||||
idx = db.allocate_file_index(prefix, identity)
|
||||
info = process_file(
|
||||
path, idx, db, client, output_dir,
|
||||
prefix=prefix, force=force, remote_root=remote_root,
|
||||
source_key=identity,
|
||||
)
|
||||
total_sections += n
|
||||
file_results.append(info)
|
||||
total_sections += int(info.get("saved") or 0)
|
||||
|
||||
if purge_missing:
|
||||
existing = set(db.all_source_files(prefix))
|
||||
orphans = sorted(existing - current_paths)
|
||||
orphans = sorted(
|
||||
e for e in existing
|
||||
if source_basename(e).lower() not in current_names
|
||||
)
|
||||
if not orphans:
|
||||
log.info(f"Purge: няма orphan записи в БД за prefix={prefix}.")
|
||||
else:
|
||||
@@ -1191,6 +1380,7 @@ def process_directory(
|
||||
db.close()
|
||||
|
||||
log.info(f"Готово. Prefix={prefix}. Общо нови/обновени секции: {total_sections}")
|
||||
return {"sections": total_sections, "files": file_results}
|
||||
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
|
||||
187
webapp/main.py
187
webapp/main.py
@@ -27,11 +27,19 @@ from pathlib import Path
|
||||
from typing import Optional
|
||||
|
||||
import psycopg2
|
||||
from fastapi import FastAPI, HTTPException, Query, Request
|
||||
from fastapi import FastAPI, Header, HTTPException, Query, Request
|
||||
from fastapi.responses import FileResponse, HTMLResponse, JSONResponse
|
||||
from fastapi.templating import Jinja2Templates
|
||||
from pydantic import BaseModel
|
||||
|
||||
from help_codes import (
|
||||
WIPED_HASH,
|
||||
numbering_report,
|
||||
parse_code,
|
||||
remove_section_outputs,
|
||||
source_basename,
|
||||
)
|
||||
|
||||
# ──────────────────────────────────────────────
|
||||
# Configuration
|
||||
# ──────────────────────────────────────────────
|
||||
@@ -317,6 +325,183 @@ def update_keywords(code: str, body: KeywordsUpdate):
|
||||
return {"ok": True, "code": code}
|
||||
|
||||
|
||||
def _safe_help_filename(file: str) -> str:
|
||||
name = Path(str(file or "").replace("\\", "/")).name
|
||||
if not name or name in (".", ".."):
|
||||
raise HTTPException(400, "file is required")
|
||||
return name
|
||||
|
||||
|
||||
def _ensure_file_index_col(cur):
|
||||
cur.execute(
|
||||
"ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER"
|
||||
)
|
||||
|
||||
|
||||
def _matching_source_paths(cur, prefix: str, identity: str) -> list[str]:
|
||||
name = source_basename(identity).lower()
|
||||
if not name:
|
||||
return []
|
||||
cur.execute(
|
||||
"""
|
||||
SELECT file_path FROM rip_help_files WHERE prefix=%s
|
||||
UNION
|
||||
SELECT source_file FROM rip_help_sections WHERE prefix=%s
|
||||
""",
|
||||
(prefix, prefix),
|
||||
)
|
||||
found: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for (p,) in cur.fetchall():
|
||||
if p and source_basename(p).lower() == name and p not in seen:
|
||||
seen.add(p)
|
||||
found.append(p)
|
||||
return found
|
||||
|
||||
|
||||
def _file_index_for(cur, prefix: str, paths: list[str]) -> Optional[int]:
|
||||
if not paths:
|
||||
return None
|
||||
cur.execute(
|
||||
"SELECT file_index FROM rip_help_files "
|
||||
"WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
from_col = [r[0] for r in cur.fetchall() if r[0]]
|
||||
if from_col:
|
||||
return min(from_col)
|
||||
cur.execute(
|
||||
"SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
found: list[int] = []
|
||||
for (code,) in cur.fetchall():
|
||||
parsed = parse_code(code)
|
||||
if parsed:
|
||||
found.append(parsed[1])
|
||||
if not found:
|
||||
return None
|
||||
tally: dict[int, int] = {}
|
||||
for idx in found:
|
||||
tally[idx] = tally.get(idx, 0) + 1
|
||||
return max(tally, key=lambda k: (tally[k], -k))
|
||||
|
||||
|
||||
@app.get("/api/file-extractions")
|
||||
def api_file_extractions(
|
||||
file: str = Query(..., min_length=1),
|
||||
prefix: str = Query("RIP"),
|
||||
):
|
||||
"""Брой и номерация на извлеченията за даден help файл."""
|
||||
name = _safe_help_filename(file)
|
||||
conn = db_conn()
|
||||
try:
|
||||
cur = conn.cursor()
|
||||
_ensure_file_index_col(cur)
|
||||
conn.commit()
|
||||
paths = _matching_source_paths(cur, prefix, name)
|
||||
rows = []
|
||||
if paths:
|
||||
cur.execute(
|
||||
"SELECT code, source_file FROM rip_help_sections "
|
||||
"WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
codes = [r[0] for r in rows]
|
||||
report = numbering_report(codes)
|
||||
fi = _file_index_for(cur, prefix, paths)
|
||||
if fi is not None:
|
||||
report["file_index"] = fi
|
||||
return {
|
||||
"file": name,
|
||||
"prefix": prefix,
|
||||
"sources": sorted({source_basename(r[1]) for r in rows if r[1]}),
|
||||
**report,
|
||||
}
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
@app.delete("/api/file-extractions")
|
||||
def api_delete_file_extractions(
|
||||
file: str = Query(..., min_length=1),
|
||||
prefix: str = Query("RIP"),
|
||||
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
|
||||
token: Optional[str] = Query(None),
|
||||
):
|
||||
"""Изтрива всички извлечения за файла. Самият help файл не се пипа."""
|
||||
rescan_mod._check_token(x_rescan_token or token)
|
||||
name = _safe_help_filename(file)
|
||||
conn = db_conn()
|
||||
try:
|
||||
cur = conn.cursor()
|
||||
_ensure_file_index_col(cur)
|
||||
paths = _matching_source_paths(cur, prefix, name)
|
||||
rows = []
|
||||
if paths:
|
||||
cur.execute(
|
||||
"SELECT code, output_path FROM rip_help_sections "
|
||||
"WHERE prefix=%s AND source_file = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
rows = cur.fetchall()
|
||||
codes = [r[0] for r in rows]
|
||||
file_index = _file_index_for(cur, prefix, paths)
|
||||
remove_section_outputs(OUTPUT_DIR, codes, [r[1] for r in rows if r[1]])
|
||||
if file_index:
|
||||
cur.execute(
|
||||
"SELECT code, source_file FROM rip_help_sections WHERE prefix=%s",
|
||||
(prefix,),
|
||||
)
|
||||
name_l = name.lower()
|
||||
shared = False
|
||||
for code, src in cur.fetchall():
|
||||
parsed = parse_code(code)
|
||||
if (
|
||||
parsed
|
||||
and parsed[1] == file_index
|
||||
and source_basename(src).lower() != name_l
|
||||
):
|
||||
shared = True
|
||||
break
|
||||
if shared:
|
||||
file_index = None
|
||||
if paths:
|
||||
cur.execute(
|
||||
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
cur.execute(
|
||||
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
|
||||
(prefix, list(paths)),
|
||||
)
|
||||
if file_index:
|
||||
cur.execute(
|
||||
"""
|
||||
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
|
||||
VALUES (%s, %s, %s, 0, %s)
|
||||
ON CONFLICT (prefix, file_path) DO UPDATE SET
|
||||
file_hash = EXCLUDED.file_hash,
|
||||
section_count = 0,
|
||||
processed_at = NOW(),
|
||||
file_index = EXCLUDED.file_index
|
||||
""",
|
||||
(prefix, name, WIPED_HASH, file_index),
|
||||
)
|
||||
conn.commit()
|
||||
return {
|
||||
"ok": True,
|
||||
"file": name,
|
||||
"prefix": prefix,
|
||||
"deleted": len(codes),
|
||||
"codes": codes,
|
||||
"file_index": file_index,
|
||||
}
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
@app.get("/healthz", summary="Статус на услугата и БД; output_dir_ok, txt_count.")
|
||||
def healthz():
|
||||
db_status = "ok"
|
||||
|
||||
@@ -102,8 +102,11 @@ def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool):
|
||||
# Import here so webapp boots even if processor deps fail later
|
||||
from help_processor import process_directory
|
||||
|
||||
process_directory(
|
||||
input_dir=staging,
|
||||
input_dir = staging / "input"
|
||||
if not input_dir.is_dir():
|
||||
input_dir = staging
|
||||
summary = process_directory(
|
||||
input_dir=input_dir,
|
||||
output_dir=out,
|
||||
conn_str=cfg["conn"],
|
||||
api_key=cfg["api_key"],
|
||||
@@ -112,12 +115,20 @@ def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool):
|
||||
purge_missing=False,
|
||||
remote_root=cfg["remote_root"],
|
||||
)
|
||||
files = []
|
||||
if isinstance(summary, dict):
|
||||
files = summary.get("files") or []
|
||||
sections = summary.get("sections")
|
||||
else:
|
||||
sections = summary
|
||||
_set_job(
|
||||
job_id,
|
||||
status="done",
|
||||
message="ok",
|
||||
output_dir=str(out),
|
||||
prefix=prefix,
|
||||
files=files,
|
||||
sections=sections,
|
||||
)
|
||||
except Exception as e:
|
||||
_set_job(job_id, status="error", message=str(e))
|
||||
|
||||
@@ -268,6 +268,12 @@
|
||||
.rescan-status.busy { color: var(--accent); }
|
||||
.rescan-status.ok { color: var(--accent2); }
|
||||
.rescan-status.error { color: var(--danger); }
|
||||
.rescan-filemeta {
|
||||
font-family: var(--mono); font-size: 12px; color: var(--muted);
|
||||
margin-bottom: 12px; line-height: 1.45;
|
||||
}
|
||||
.rescan-filemeta:empty { display: none; }
|
||||
.rescan-filemeta.warn { color: var(--danger); }
|
||||
|
||||
.api-layout { display: grid; grid-template-columns: minmax(280px, 1fr) minmax(320px, 1.1fr); gap: 20px; }
|
||||
@media (max-width: 900px) { .api-layout { grid-template-columns: 1fr; } }
|
||||
@@ -350,9 +356,12 @@
|
||||
webkitdirectory directory multiple
|
||||
onchange="onScanFolderPicked()">
|
||||
<div class="rescan-selection" id="rescan-selection"></div>
|
||||
<div class="rescan-filemeta" id="rescan-filemeta"></div>
|
||||
<div class="rescan-row">
|
||||
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
|
||||
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
|
||||
<button type="button" class="btn-danger" id="rescan-wipe-btn" hidden
|
||||
onclick="confirmWipeFile()">Изтрий извлеченията</button>
|
||||
</div>
|
||||
<div class="rescan-status" id="rescan-status">Избери файл или папка.</div>
|
||||
</div>
|
||||
@@ -362,6 +371,7 @@
|
||||
<div class="toolbar">
|
||||
<input type="text" class="search-box" id="editor-search" placeholder="Филтрирай по код, заглавие, ключова дума..." oninput="filterEditor()">
|
||||
<span class="stats" id="editor-stats"></span>
|
||||
<span class="stats" id="editor-file-stats"></span>
|
||||
</div>
|
||||
<div class="tbl-wrap">
|
||||
<table id="editor-table">
|
||||
@@ -460,6 +470,10 @@
|
||||
<div class="desc">multipart: <code>file</code> (един или повече: zip/docx/…), <code>prefix</code>, <code>force</code>. Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
||||
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
|
||||
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
|
||||
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||
<div class="desc">Брой и кодове на извлеченията за даден файл; <code>ok</code> ако SEC номерацията е уникална и последователна.</div></div>
|
||||
<div class="api-ep"><span class="method post">DELETE</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||
<div class="desc">Изтрива всички извлечения за файла (не самия help файл). Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
||||
|
||||
<p class="api-note">OpenAPI / Swagger: <a href="/docs" target="_blank" rel="noopener">/docs</a> · ReDoc: <a href="/redoc" target="_blank" rel="noopener">/redoc</a></p>
|
||||
<p class="api-note">Типичен ERP поток: <code>GET /api/search</code> → избор на <code>code</code> → <code>GET /api/section/{code}</code> за съдържание.</p>
|
||||
@@ -490,6 +504,20 @@
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="modal-backdrop" id="wipe-modal" onclick="if(event.target===this)closeWipeModal()">
|
||||
<div class="modal" role="dialog" aria-modal="true" style="width:min(460px,100%)">
|
||||
<div class="modal-head">
|
||||
<h2>Изтриване на извлечения</h2>
|
||||
<button type="button" class="btn-ghost" onclick="closeWipeModal()">✕</button>
|
||||
</div>
|
||||
<div class="modal-body" id="wipe-modal-body"></div>
|
||||
<div class="rescan-row" style="margin:16px 0 0">
|
||||
<button type="button" class="btn-danger" id="wipe-confirm-btn" onclick="doWipeFile()">Изтрий</button>
|
||||
<button type="button" class="btn-ghost" onclick="closeWipeModal()">Отказ</button>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div id="toast"></div>
|
||||
|
||||
<script type="application/json" id="sections-data">{{ sections_json | safe }}</script>
|
||||
@@ -534,14 +562,22 @@ function switchTab(name) {
|
||||
}
|
||||
|
||||
const SCAN_EXT = ['.zip', '.docx', '.doc', '.html', '.htm', '.pdf', '.txt'];
|
||||
const PAGE_PREFIX = {{ (prefix or 'RIP') | tojson }};
|
||||
let scanMode = null;
|
||||
let scanFiles = [];
|
||||
let fileStats = null;
|
||||
let wipeTarget = null;
|
||||
|
||||
function scanExt(name) {
|
||||
const n = String(name || '');
|
||||
const i = n.lastIndexOf('.');
|
||||
return i >= 0 ? n.slice(i).toLowerCase() : '';
|
||||
}
|
||||
function fileBase(p) {
|
||||
if (!p) return '';
|
||||
const parts = String(p).replace(/\\/g, '/').split('/');
|
||||
return parts[parts.length - 1] || '';
|
||||
}
|
||||
function setScanStatus(text, kind) {
|
||||
const st = document.getElementById('rescan-status');
|
||||
st.textContent = text;
|
||||
@@ -556,6 +592,84 @@ function folderNameFrom(files) {
|
||||
const top = p.replace(/\\/g, '/').split('/')[0];
|
||||
return top || (files[0] && files[0].name) || 'папка';
|
||||
}
|
||||
function renderFileStats() {
|
||||
const meta = document.getElementById('rescan-filemeta');
|
||||
const wipeBtn = document.getElementById('rescan-wipe-btn');
|
||||
const editorExtra = document.getElementById('editor-file-stats');
|
||||
const n = fileStats ? (fileStats.count || 0) : 0;
|
||||
const fileSelected = scanMode === 'file' && scanFiles[0];
|
||||
if (!fileStats || !fileSelected) {
|
||||
if (meta) meta.textContent = '';
|
||||
if (wipeBtn) wipeBtn.hidden = true;
|
||||
if (editorExtra) editorExtra.textContent = '';
|
||||
return;
|
||||
}
|
||||
let line = 'Извлечения: ' + n;
|
||||
if (n && fileStats.first && fileStats.last) {
|
||||
line += fileStats.first === fileStats.last
|
||||
? ' · ' + fileStats.first
|
||||
: ' · ' + fileStats.first + ' – ' + fileStats.last;
|
||||
}
|
||||
if (meta) {
|
||||
meta.textContent = line;
|
||||
if (fileStats.ok === false && fileStats.issues && fileStats.issues.length) {
|
||||
meta.textContent = line + ' · ' + fileStats.issues.join('; ');
|
||||
meta.classList.add('warn');
|
||||
} else {
|
||||
meta.classList.remove('warn');
|
||||
}
|
||||
}
|
||||
if (wipeBtn) {
|
||||
wipeBtn.hidden = n === 0;
|
||||
wipeBtn.disabled = false;
|
||||
}
|
||||
if (editorExtra) {
|
||||
editorExtra.textContent = fileStats.file ? (' · ' + fileStats.file + ': ' + n) : '';
|
||||
}
|
||||
}
|
||||
async function loadFileStats(name) {
|
||||
if (!name) {
|
||||
fileStats = null;
|
||||
renderFileStats();
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const url = '/api/file-extractions?file=' + encodeURIComponent(name)
|
||||
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
|
||||
const res = await fetch(url);
|
||||
const data = await res.json().catch(() => ({}));
|
||||
if (!res.ok) {
|
||||
const detail = data.detail;
|
||||
throw new Error(typeof detail === 'string' ? detail : ('HTTP ' + res.status));
|
||||
}
|
||||
fileStats = data;
|
||||
} catch (e) {
|
||||
const rows = ALL.filter(r => fileBase(r.source_file).toLowerCase() === String(name).toLowerCase());
|
||||
const codes = rows.map(r => r.code).filter(Boolean).sort();
|
||||
fileStats = {
|
||||
file: name,
|
||||
count: rows.length,
|
||||
first: codes[0] || null,
|
||||
last: codes[codes.length - 1] || null,
|
||||
ok: true,
|
||||
issues: [],
|
||||
};
|
||||
}
|
||||
renderFileStats();
|
||||
}
|
||||
async function refreshSections() {
|
||||
const p = {{ (prefix or '') | tojson }};
|
||||
const url = p ? '/api/sections?prefix=' + encodeURIComponent(p) : '/api/sections';
|
||||
const res = await fetch(url);
|
||||
if (!res.ok) throw new Error('HTTP ' + res.status);
|
||||
ALL = await res.json();
|
||||
document.getElementById('total-count').textContent = ALL.length + ' секции';
|
||||
const q = document.getElementById('editor-search');
|
||||
if (q && q.value) filterEditor();
|
||||
else renderEditor(ALL);
|
||||
renderKwCloud();
|
||||
doSearch();
|
||||
}
|
||||
function setScanSelection(mode, files) {
|
||||
scanMode = mode;
|
||||
scanFiles = files;
|
||||
@@ -569,18 +683,22 @@ function setScanSelection(mode, files) {
|
||||
if (!mode) {
|
||||
sel.textContent = '';
|
||||
setScanStatus('Избери файл или папка.');
|
||||
loadFileStats('');
|
||||
return;
|
||||
}
|
||||
if (mode === 'file') {
|
||||
if (!files[0]) {
|
||||
sel.textContent = '';
|
||||
setScanStatus('Избери файл.', 'error');
|
||||
loadFileStats('');
|
||||
return;
|
||||
}
|
||||
sel.textContent = 'Файл: ' + files[0].name;
|
||||
setScanStatus('Готово за старт.');
|
||||
loadFileStats(files[0].name);
|
||||
return;
|
||||
}
|
||||
loadFileStats('');
|
||||
if (!files.length) {
|
||||
sel.textContent = 'Папка: няма подходящи файлове';
|
||||
setScanStatus('Няма .zip, .docx, .html, .pdf или .txt в папката.', 'error');
|
||||
@@ -620,17 +738,64 @@ function onScanFolderPicked() {
|
||||
setScanSelection('folder', files);
|
||||
}
|
||||
|
||||
function confirmWipeFile() {
|
||||
const name = scanFiles[0] && scanFiles[0].name;
|
||||
if (!name) return;
|
||||
const n = (fileStats && fileStats.count) || 0;
|
||||
wipeTarget = name;
|
||||
document.getElementById('wipe-modal-body').textContent =
|
||||
'Ще се изтрият всички извлечения за «' + name + '» (' + n + ' бр.). Самият help файл не се трие.';
|
||||
document.getElementById('wipe-modal').classList.add('open');
|
||||
}
|
||||
function closeWipeModal() {
|
||||
const modal = document.getElementById('wipe-modal');
|
||||
if (modal) modal.classList.remove('open');
|
||||
wipeTarget = null;
|
||||
}
|
||||
async function doWipeFile() {
|
||||
const name = wipeTarget;
|
||||
if (!name) return;
|
||||
const btn = document.getElementById('wipe-confirm-btn');
|
||||
btn.disabled = true;
|
||||
try {
|
||||
const token = localStorage.getItem('RESCAN_TOKEN') || '';
|
||||
const headers = {};
|
||||
if (token) headers['X-Rescan-Token'] = token;
|
||||
let url = '/api/file-extractions?file=' + encodeURIComponent(name)
|
||||
+ '&prefix=' + encodeURIComponent(PAGE_PREFIX);
|
||||
if (token) url += '&token=' + encodeURIComponent(token);
|
||||
const res = await fetch(url, { method: 'DELETE', headers });
|
||||
const data = await res.json().catch(() => ({}));
|
||||
if (!res.ok) {
|
||||
const detail = data.detail;
|
||||
throw new Error(typeof detail === 'string' ? detail : (data.message || ('HTTP ' + res.status)));
|
||||
}
|
||||
document.getElementById('wipe-modal').classList.remove('open');
|
||||
wipeTarget = null;
|
||||
await refreshSections();
|
||||
await loadFileStats(name);
|
||||
setScanStatus('Изтрити извлечения: ' + (data.deleted || 0) + ' за ' + name + '.', 'ok');
|
||||
toast('Изтрити: ' + (data.deleted || 0));
|
||||
} catch (e) {
|
||||
toast(e.message, true);
|
||||
setScanStatus('Грешка: ' + e.message, 'error');
|
||||
} finally {
|
||||
btn.disabled = false;
|
||||
}
|
||||
}
|
||||
|
||||
async function startRescan() {
|
||||
const btn = document.getElementById('rescan-btn');
|
||||
const fileBtn = document.getElementById('rescan-pick-file');
|
||||
const folderBtn = document.getElementById('rescan-pick-folder');
|
||||
const wipeBtn = document.getElementById('rescan-wipe-btn');
|
||||
if (!scanFiles.length) {
|
||||
setScanStatus(scanMode === 'folder' ? 'Няма подходящи файлове в папката.' : 'Избери файл или папка.', 'error');
|
||||
return;
|
||||
}
|
||||
const fd = new FormData();
|
||||
scanFiles.forEach(f => fd.append('file', f));
|
||||
fd.append('prefix', {{ (prefix or 'RIP') | tojson }});
|
||||
fd.append('prefix', PAGE_PREFIX);
|
||||
fd.append('force', document.getElementById('rescan-force').checked ? 'true' : 'false');
|
||||
const token = localStorage.getItem('RESCAN_TOKEN') || '';
|
||||
if (token) fd.append('token', token);
|
||||
@@ -638,6 +803,7 @@ async function startRescan() {
|
||||
btn.disabled = true;
|
||||
fileBtn.disabled = true;
|
||||
folderBtn.disabled = true;
|
||||
if (wipeBtn) wipeBtn.disabled = true;
|
||||
setScanStatus('Качване...', 'busy');
|
||||
try {
|
||||
const headers = {};
|
||||
@@ -657,6 +823,7 @@ async function startRescan() {
|
||||
btn.disabled = false;
|
||||
fileBtn.disabled = false;
|
||||
folderBtn.disabled = false;
|
||||
if (wipeBtn) wipeBtn.disabled = false;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -667,8 +834,25 @@ async function pollRescan(jobId) {
|
||||
const data = await res.json();
|
||||
const line = data.status + (data.message ? ' — ' + data.message : '');
|
||||
if (data.status === 'done') {
|
||||
setScanStatus('Готово. Презареждане...', 'ok');
|
||||
location.reload();
|
||||
const files = data.files || [];
|
||||
const bad = files.some(f => f && f.ok === false);
|
||||
let msg = 'Готово.';
|
||||
if (typeof data.sections === 'number') msg += ' Записани: ' + data.sections;
|
||||
if (files.length === 1) {
|
||||
const f = files[0];
|
||||
msg = 'Готово. Извлечения: ' + (f.count || 0);
|
||||
if (f.first && f.last) {
|
||||
msg += f.first === f.last ? ' · ' + f.first : ' · ' + f.first + ' – ' + f.last;
|
||||
}
|
||||
if (f.ok === false && f.issues && f.issues.length) msg += ' · ' + f.issues.join('; ');
|
||||
}
|
||||
setScanStatus(msg, bad ? 'error' : 'ok');
|
||||
try {
|
||||
await refreshSections();
|
||||
if (scanMode === 'file' && scanFiles[0]) await loadFileStats(scanFiles[0].name);
|
||||
} catch (e) {
|
||||
setScanStatus((msg || 'Готово.') + ' Презареди страницата.', bad ? 'error' : 'ok');
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (data.status === 'error') {
|
||||
@@ -964,7 +1148,10 @@ function closeSection() {
|
||||
document.getElementById('section-modal').classList.remove('open');
|
||||
}
|
||||
document.addEventListener('keydown', (e) => {
|
||||
if (e.key === 'Escape') closeSection();
|
||||
if (e.key === 'Escape') {
|
||||
closeWipeModal();
|
||||
closeSection();
|
||||
}
|
||||
});
|
||||
|
||||
function toggleSelect(code, el) {
|
||||
|
||||
Reference in New Issue
Block a user