Give each help file unique sequential extraction codes, show the count, and allow wiping all extractions for a file while testing.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-09-10 13:26:21 +03:00
parent 170c551d6e
commit 3a30959ae5
5 changed files with 761 additions and 39 deletions

View File

@@ -27,11 +27,19 @@ from pathlib import Path
from typing import Optional
import psycopg2
from fastapi import FastAPI, HTTPException, Query, Request
from fastapi import FastAPI, Header, HTTPException, Query, Request
from fastapi.responses import FileResponse, HTMLResponse, JSONResponse
from fastapi.templating import Jinja2Templates
from pydantic import BaseModel
from help_codes import (
WIPED_HASH,
numbering_report,
parse_code,
remove_section_outputs,
source_basename,
)
# ──────────────────────────────────────────────
# Configuration
# ──────────────────────────────────────────────
@@ -317,6 +325,183 @@ def update_keywords(code: str, body: KeywordsUpdate):
return {"ok": True, "code": code}
def _safe_help_filename(file: str) -> str:
name = Path(str(file or "").replace("\\", "/")).name
if not name or name in (".", ".."):
raise HTTPException(400, "file is required")
return name
def _ensure_file_index_col(cur):
cur.execute(
"ALTER TABLE rip_help_files ADD COLUMN IF NOT EXISTS file_index INTEGER"
)
def _matching_source_paths(cur, prefix: str, identity: str) -> list[str]:
name = source_basename(identity).lower()
if not name:
return []
cur.execute(
"""
SELECT file_path FROM rip_help_files WHERE prefix=%s
UNION
SELECT source_file FROM rip_help_sections WHERE prefix=%s
""",
(prefix, prefix),
)
found: list[str] = []
seen: set[str] = set()
for (p,) in cur.fetchall():
if p and source_basename(p).lower() == name and p not in seen:
seen.add(p)
found.append(p)
return found
def _file_index_for(cur, prefix: str, paths: list[str]) -> Optional[int]:
if not paths:
return None
cur.execute(
"SELECT file_index FROM rip_help_files "
"WHERE prefix=%s AND file_path = ANY(%s) AND file_index IS NOT NULL",
(prefix, list(paths)),
)
from_col = [r[0] for r in cur.fetchall() if r[0]]
if from_col:
return min(from_col)
cur.execute(
"SELECT code FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
found: list[int] = []
for (code,) in cur.fetchall():
parsed = parse_code(code)
if parsed:
found.append(parsed[1])
if not found:
return None
tally: dict[int, int] = {}
for idx in found:
tally[idx] = tally.get(idx, 0) + 1
return max(tally, key=lambda k: (tally[k], -k))
@app.get("/api/file-extractions")
def api_file_extractions(
file: str = Query(..., min_length=1),
prefix: str = Query("RIP"),
):
"""Брой и номерация на извлеченията за даден help файл."""
name = _safe_help_filename(file)
conn = db_conn()
try:
cur = conn.cursor()
_ensure_file_index_col(cur)
conn.commit()
paths = _matching_source_paths(cur, prefix, name)
rows = []
if paths:
cur.execute(
"SELECT code, source_file FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s) ORDER BY code",
(prefix, list(paths)),
)
rows = cur.fetchall()
codes = [r[0] for r in rows]
report = numbering_report(codes)
fi = _file_index_for(cur, prefix, paths)
if fi is not None:
report["file_index"] = fi
return {
"file": name,
"prefix": prefix,
"sources": sorted({source_basename(r[1]) for r in rows if r[1]}),
**report,
}
finally:
conn.close()
@app.delete("/api/file-extractions")
def api_delete_file_extractions(
file: str = Query(..., min_length=1),
prefix: str = Query("RIP"),
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
token: Optional[str] = Query(None),
):
"""Изтрива всички извлечения за файла. Самият help файл не се пипа."""
rescan_mod._check_token(x_rescan_token or token)
name = _safe_help_filename(file)
conn = db_conn()
try:
cur = conn.cursor()
_ensure_file_index_col(cur)
paths = _matching_source_paths(cur, prefix, name)
rows = []
if paths:
cur.execute(
"SELECT code, output_path FROM rip_help_sections "
"WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
rows = cur.fetchall()
codes = [r[0] for r in rows]
file_index = _file_index_for(cur, prefix, paths)
remove_section_outputs(OUTPUT_DIR, codes, [r[1] for r in rows if r[1]])
if file_index:
cur.execute(
"SELECT code, source_file FROM rip_help_sections WHERE prefix=%s",
(prefix,),
)
name_l = name.lower()
shared = False
for code, src in cur.fetchall():
parsed = parse_code(code)
if (
parsed
and parsed[1] == file_index
and source_basename(src).lower() != name_l
):
shared = True
break
if shared:
file_index = None
if paths:
cur.execute(
"DELETE FROM rip_help_sections WHERE prefix=%s AND source_file = ANY(%s)",
(prefix, list(paths)),
)
cur.execute(
"DELETE FROM rip_help_files WHERE prefix=%s AND file_path = ANY(%s)",
(prefix, list(paths)),
)
if file_index:
cur.execute(
"""
INSERT INTO rip_help_files (prefix, file_path, file_hash, section_count, file_index)
VALUES (%s, %s, %s, 0, %s)
ON CONFLICT (prefix, file_path) DO UPDATE SET
file_hash = EXCLUDED.file_hash,
section_count = 0,
processed_at = NOW(),
file_index = EXCLUDED.file_index
""",
(prefix, name, WIPED_HASH, file_index),
)
conn.commit()
return {
"ok": True,
"file": name,
"prefix": prefix,
"deleted": len(codes),
"codes": codes,
"file_index": file_index,
}
finally:
conn.close()
@app.get("/healthz", summary="Статус на услугата и БД; output_dir_ok, txt_count.")
def healthz():
db_status = "ok"