Give each help file unique sequential extraction codes, show the count, and allow wiping all extractions for a file while testing.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-09-10 13:26:21 +03:00
parent 170c551d6e
commit 3a30959ae5
5 changed files with 761 additions and 39 deletions

149
help_codes.py Normal file
View File

@@ -0,0 +1,149 @@
"""Кодове на извлечения: PREFIX_0001_SEC_0001 и проверка на номерацията."""
from __future__ import annotations
import re
from pathlib import Path
from typing import Optional
WIPED_HASH = "0" * 64
_CODE_RE = re.compile(r"^(.+)_(\d{4})_SEC_(\d{4})$")
def make_code(prefix: str, file_index: int, sec_index: int) -> str:
return f"{prefix}_{file_index:04d}_SEC_{sec_index:04d}"
def parse_code(code: str) -> Optional[tuple[str, int, int]]:
m = _CODE_RE.match(str(code or "").strip())
if not m:
return None
return m.group(1), int(m.group(2)), int(m.group(3))
def source_basename(path_str: str) -> str:
return Path(str(path_str or "").replace("\\", "/")).name
def source_identity(path: Path, input_dir: Optional[Path] = None) -> str:
"""Стабилен ключ за файла: относителен път в input_dir, иначе basename."""
if input_dir:
try:
return str(path.resolve().relative_to(input_dir.resolve())).replace("\\", "/")
except ValueError:
pass
return path.name
def numbering_report(codes: list[str]) -> dict:
"""Проверява уникална последователна SEC номерация за един файл."""
parsed: list[tuple[str, str, int, int]] = []
issues: list[str] = []
for c in codes:
p = parse_code(c)
if not p:
issues.append(f"невалиден код: {c}")
continue
parsed.append((c, p[0], p[1], p[2]))
if not codes:
return {
"count": 0,
"ok": True,
"file_index": None,
"first": None,
"last": None,
"codes": [],
"issues": [],
}
indexes = {p[2] for p in parsed}
if len(indexes) > 1:
issues.append(
"различни file_index в кодовете: " + ", ".join(str(i) for i in sorted(indexes))
)
file_index = min(indexes) if indexes else None
secs = [p[3] for p in parsed]
tally: dict[int, int] = {}
for s in secs:
tally[s] = tally.get(s, 0) + 1
dups = sorted(s for s, n in tally.items() if n > 1)
if dups:
issues.append("дублирани SEC: " + ", ".join(f"{s:04d}" for s in dups[:8]))
n = len(parsed)
expected = list(range(1, n + 1))
actual = sorted(secs)
if actual != expected:
missing = [x for x in expected if x not in tally]
if missing:
issues.append("липсващи SEC: " + ", ".join(f"{s:04d}" for s in missing[:8]))
elif actual:
issues.append(f"номерацията не е 1…{n:04d}")
codes_sorted = [p[0] for p in sorted(parsed, key=lambda x: (x[2], x[3], x[0]))]
return {
"count": len(codes),
"ok": not issues,
"file_index": file_index,
"first": codes_sorted[0] if codes_sorted else None,
"last": codes_sorted[-1] if codes_sorted else None,
"codes": codes_sorted,
"issues": issues,
}
def remove_section_outputs(
output_dir: Path,
codes: list[str],
output_paths: Optional[list[str]] = None,
) -> int:
removed = 0
images_dir = output_dir / "images"
for code in codes:
if not code:
continue
local_txt = output_dir / f"{code}.txt"
try:
if local_txt.exists():
local_txt.unlink()
removed += 1
except Exception:
pass
if images_dir.is_dir():
for img in images_dir.glob(f"{code}_*"):
try:
img.unlink()
removed += 1
except Exception:
pass
for op in output_paths or []:
if not op:
continue
try:
opath = Path(op)
stem = Path(str(op).replace("\\", "/")).stem
local_txt = output_dir / f"{stem}.txt"
if opath.exists():
try:
if not local_txt.exists() or opath.resolve() != local_txt.resolve():
opath.unlink()
removed += 1
except Exception:
pass
except Exception:
pass
return removed
def file_result(
rel: str,
file_index: Optional[int],
codes: list[str],
saved: int = 0,
skipped: bool = False,
) -> dict:
report = numbering_report(codes)
if file_index is not None:
report["file_index"] = file_index
return {
"file": rel,
"saved": saved,
"skipped": skipped,
**report,
}