.
This commit is contained in:
@@ -1352,6 +1352,61 @@ def fallback_classify(title: str, text: str) -> tuple[str, str]:
|
|||||||
return t[:200], ", ".join(seen)
|
return t[:200], ", ".join(seen)
|
||||||
|
|
||||||
|
|
||||||
|
KEYWORDS_MAX_LEN = 300 # rip_help_sections.keywords VARCHAR(300)
|
||||||
|
|
||||||
|
|
||||||
|
def parse_keyword_list(raw: str) -> list[str]:
|
||||||
|
"""Разделя ключови думи. Запетая/точка и запетая = фрази; иначе — интервал."""
|
||||||
|
text = (raw or "").strip()
|
||||||
|
if not text:
|
||||||
|
return []
|
||||||
|
if "," in text or ";" in text:
|
||||||
|
parts = re.split(r"[,;]+", text)
|
||||||
|
else:
|
||||||
|
parts = text.split()
|
||||||
|
out: list[str] = []
|
||||||
|
seen: set[str] = set()
|
||||||
|
for p in parts:
|
||||||
|
k = " ".join(p.split())
|
||||||
|
if not k:
|
||||||
|
continue
|
||||||
|
key = k.casefold()
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
out.append(k)
|
||||||
|
return out
|
||||||
|
|
||||||
|
|
||||||
|
def merge_section_keywords(
|
||||||
|
manual: str,
|
||||||
|
generated: str,
|
||||||
|
max_len: int = KEYWORDS_MAX_LEN,
|
||||||
|
) -> str:
|
||||||
|
"""Ръчните думи първи, после генерираните; без дубликати (без значение на регистъра)."""
|
||||||
|
first = parse_keyword_list(manual)
|
||||||
|
seen = {k.casefold() for k in first}
|
||||||
|
rest: list[str] = []
|
||||||
|
for k in parse_keyword_list(generated):
|
||||||
|
key = k.casefold()
|
||||||
|
if key in seen:
|
||||||
|
continue
|
||||||
|
seen.add(key)
|
||||||
|
rest.append(k)
|
||||||
|
merged = first + rest
|
||||||
|
if not merged:
|
||||||
|
return ""
|
||||||
|
out: list[str] = []
|
||||||
|
used = 0
|
||||||
|
for i, k in enumerate(merged):
|
||||||
|
extra = len(k) + (2 if i else 0) # ", "
|
||||||
|
if used + extra > max_len:
|
||||||
|
break
|
||||||
|
out.append(k)
|
||||||
|
used += extra
|
||||||
|
return ", ".join(out)
|
||||||
|
|
||||||
|
|
||||||
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
def classify_section(client: anthropic.Anthropic, title: str, text: str) -> tuple[str, str]:
|
||||||
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
"""Връща (наименование, 'кл1, кл2, кл3') чрез Claude."""
|
||||||
snippet = text[:MAX_AI_CHARS]
|
snippet = text[:MAX_AI_CHARS]
|
||||||
@@ -1420,6 +1475,7 @@ def process_file(
|
|||||||
force: bool = False,
|
force: bool = False,
|
||||||
remote_root: Optional[str] = None,
|
remote_root: Optional[str] = None,
|
||||||
source_key: Optional[str] = None,
|
source_key: Optional[str] = None,
|
||||||
|
seed_keywords: str = "",
|
||||||
) -> dict:
|
) -> dict:
|
||||||
"""Обработва един файл. Връща статистика (saved, codes, ok)."""
|
"""Обработва един файл. Връща статистика (saved, codes, ok)."""
|
||||||
rel = source_key or path.name
|
rel = source_key or path.name
|
||||||
@@ -1512,6 +1568,7 @@ def process_file(
|
|||||||
if not (keywords or "").strip():
|
if not (keywords or "").strip():
|
||||||
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
|
title, keywords = fallback_classify(title or sec.title or f"Секция {sec_index}", text)
|
||||||
ai_errors += 1
|
ai_errors += 1
|
||||||
|
keywords = merge_section_keywords(seed_keywords, keywords)
|
||||||
|
|
||||||
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
images_json = json.dumps(image_rel_paths, ensure_ascii=False)
|
||||||
ps = ProcessedSection(
|
ps = ProcessedSection(
|
||||||
@@ -1558,6 +1615,7 @@ def process_directory(
|
|||||||
force: bool = False,
|
force: bool = False,
|
||||||
purge_missing: bool = False,
|
purge_missing: bool = False,
|
||||||
remote_root: Optional[str] = None,
|
remote_root: Optional[str] = None,
|
||||||
|
seed_keywords: str = "",
|
||||||
):
|
):
|
||||||
if not _PREFIX_RE.match(prefix):
|
if not _PREFIX_RE.match(prefix):
|
||||||
raise ValueError(
|
raise ValueError(
|
||||||
@@ -1598,6 +1656,7 @@ def process_directory(
|
|||||||
path, idx, db, client, output_dir,
|
path, idx, db, client, output_dir,
|
||||||
prefix=prefix, force=force, remote_root=remote_root,
|
prefix=prefix, force=force, remote_root=remote_root,
|
||||||
source_key=identity,
|
source_key=identity,
|
||||||
|
seed_keywords=seed_keywords,
|
||||||
)
|
)
|
||||||
file_results.append(info)
|
file_results.append(info)
|
||||||
total_sections += int(info.get("saved") or 0)
|
total_sections += int(info.get("saved") or 0)
|
||||||
@@ -1693,6 +1752,11 @@ def main():
|
|||||||
"Файловете се пишат локално; в БД се записва remote път. "
|
"Файловете се пишат локално; в БД се записва remote път. "
|
||||||
"Или env HELP_REMOTE_ROOT."
|
"Или env HELP_REMOTE_ROOT."
|
||||||
)
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--keywords",
|
||||||
|
default="",
|
||||||
|
help="Ръчни ключови думи (запетая). Стоят в началото на всяка секция от това сканиране."
|
||||||
|
)
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
if not args.api_key:
|
if not args.api_key:
|
||||||
@@ -1709,6 +1773,7 @@ def main():
|
|||||||
force=args.force,
|
force=args.force,
|
||||||
purge_missing=args.purge_missing,
|
purge_missing=args.purge_missing,
|
||||||
remote_root=args.remote_root,
|
remote_root=args.remote_root,
|
||||||
|
seed_keywords=args.keywords,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
47
tests/test_keywords_merge.py
Normal file
47
tests/test_keywords_merge.py
Normal file
@@ -0,0 +1,47 @@
|
|||||||
|
"""Ръчните ключови думи стоят първи във всяка секция, без дубликати."""
|
||||||
|
from help_processor import merge_section_keywords, parse_keyword_list
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_comma_phrases():
|
||||||
|
assert parse_keyword_list("лимитни карти, справка, ATR") == [
|
||||||
|
"лимитни карти",
|
||||||
|
"справка",
|
||||||
|
"ATR",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_parse_space_separated_words():
|
||||||
|
assert parse_keyword_list("лимит справка ATR") == ["лимит", "справка", "ATR"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_manual_keeps_generated_only():
|
||||||
|
assert merge_section_keywords("", "а, б, в") == "а, б, в"
|
||||||
|
assert merge_section_keywords(" ", "а, б") == "а, б"
|
||||||
|
assert merge_section_keywords("", "") == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_manual_keywords_prepended_on_every_section():
|
||||||
|
generated = "справка, печат, експорт"
|
||||||
|
merged = merge_section_keywords("лимитни карти, ATR", generated)
|
||||||
|
assert merged == "лимитни карти, ATR, справка, печат, експорт"
|
||||||
|
|
||||||
|
|
||||||
|
def test_dedup_keeps_manual_first():
|
||||||
|
# моделът повтаря ръчна дума — остава само първата (ръчната)
|
||||||
|
merged = merge_section_keywords("ATR, справка", "справка, ATR, печат")
|
||||||
|
assert merged == "ATR, справка, печат"
|
||||||
|
|
||||||
|
|
||||||
|
def test_dedup_is_case_insensitive():
|
||||||
|
merged = merge_section_keywords("atr", "ATR, печат")
|
||||||
|
assert merged == "atr, печат"
|
||||||
|
|
||||||
|
|
||||||
|
def test_same_manual_list_for_two_sections():
|
||||||
|
seed = "лимитни карти, ATR"
|
||||||
|
a = merge_section_keywords(seed, "въвеждане, карта")
|
||||||
|
b = merge_section_keywords(seed, "печат, справка")
|
||||||
|
assert a.startswith("лимитни карти, ATR,")
|
||||||
|
assert b.startswith("лимитни карти, ATR,")
|
||||||
|
assert a == "лимитни карти, ATR, въвеждане, карта"
|
||||||
|
assert b == "лимитни карти, ATR, печат, справка"
|
||||||
@@ -81,7 +81,7 @@ def _safe_extract_zip(data: bytes, dest: Path) -> int:
|
|||||||
return count
|
return count
|
||||||
|
|
||||||
|
|
||||||
def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool):
|
def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool, seed_keywords: str = ""):
|
||||||
cfg = _cfg()
|
cfg = _cfg()
|
||||||
try:
|
try:
|
||||||
if not cfg["conn"]:
|
if not cfg["conn"]:
|
||||||
@@ -114,6 +114,7 @@ def _run_rescan(job_id: str, staging: Path, prefix: str, force: bool):
|
|||||||
force=force,
|
force=force,
|
||||||
purge_missing=False,
|
purge_missing=False,
|
||||||
remote_root=cfg["remote_root"],
|
remote_root=cfg["remote_root"],
|
||||||
|
seed_keywords=seed_keywords,
|
||||||
)
|
)
|
||||||
files = []
|
files = []
|
||||||
if isinstance(summary, dict):
|
if isinstance(summary, dict):
|
||||||
@@ -208,12 +209,14 @@ async def rescan(
|
|||||||
file: list[UploadFile] = File(..., description="ZIP, one help file, or several files from a folder"),
|
file: list[UploadFile] = File(..., description="ZIP, one help file, or several files from a folder"),
|
||||||
prefix: str = Form("RIP"),
|
prefix: str = Form("RIP"),
|
||||||
force: str = Form("false"),
|
force: str = Form("false"),
|
||||||
|
keywords: str = Form(""),
|
||||||
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
|
x_rescan_token: Optional[str] = Header(None, alias="X-Rescan-Token"),
|
||||||
token: Optional[str] = Form(None),
|
token: Optional[str] = Form(None),
|
||||||
):
|
):
|
||||||
"""
|
"""
|
||||||
Upload a .zip, one .docx/.html/.pdf/.txt/.doc, or several such files
|
Upload a .zip, one .docx/.html/.pdf/.txt/.doc, or several such files
|
||||||
(folder pick). Process on server, write sections to OUTPUT_DIR and Postgres.
|
(folder pick). Process on server, write sections to OUTPUT_DIR and Postgres.
|
||||||
|
Optional form field ``keywords``: comma-separated terms prepended to every section.
|
||||||
"""
|
"""
|
||||||
_check_token(x_rescan_token or token)
|
_check_token(x_rescan_token or token)
|
||||||
|
|
||||||
@@ -232,6 +235,7 @@ async def rescan(
|
|||||||
|
|
||||||
job_id = uuid.uuid4().hex[:12]
|
job_id = uuid.uuid4().hex[:12]
|
||||||
force_flag = str(force).lower() in ("1", "true", "yes", "on")
|
force_flag = str(force).lower() in ("1", "true", "yes", "on")
|
||||||
|
seed_keywords = (keywords or "").strip()
|
||||||
_set_job(
|
_set_job(
|
||||||
job_id,
|
job_id,
|
||||||
status="queued",
|
status="queued",
|
||||||
@@ -241,7 +245,7 @@ async def rescan(
|
|||||||
force=force_flag,
|
force=force_flag,
|
||||||
created_at=datetime.now(timezone.utc).isoformat(),
|
created_at=datetime.now(timezone.utc).isoformat(),
|
||||||
)
|
)
|
||||||
background_tasks.add_task(_run_rescan, job_id, staging, prefix, force_flag)
|
background_tasks.add_task(_run_rescan, job_id, staging, prefix, force_flag, seed_keywords)
|
||||||
return JSONResponse(
|
return JSONResponse(
|
||||||
{
|
{
|
||||||
"ok": True,
|
"ok": True,
|
||||||
|
|||||||
@@ -276,6 +276,8 @@
|
|||||||
.rescan-filemeta.warn { color: var(--danger); }
|
.rescan-filemeta.warn { color: var(--danger); }
|
||||||
.wipe-box { margin-top: 16px; }
|
.wipe-box { margin-top: 16px; }
|
||||||
.wipe-box input[type=text] { width: 100%; font-family: var(--mono); }
|
.wipe-box input[type=text] { width: 100%; font-family: var(--mono); }
|
||||||
|
.rescan-kw-input { width: 100%; }
|
||||||
|
.rescan-hint { margin-bottom: 10px; line-height: 1.45; }
|
||||||
.src-file { cursor: pointer; }
|
.src-file { cursor: pointer; }
|
||||||
.src-file:hover { color: var(--accent); text-decoration: underline; }
|
.src-file:hover { color: var(--accent); text-decoration: underline; }
|
||||||
|
|
||||||
@@ -361,6 +363,15 @@
|
|||||||
onchange="onScanFolderPicked()">
|
onchange="onScanFolderPicked()">
|
||||||
<div class="rescan-selection" id="rescan-selection"></div>
|
<div class="rescan-selection" id="rescan-selection"></div>
|
||||||
<div class="rescan-filemeta" id="rescan-filemeta"></div>
|
<div class="rescan-filemeta" id="rescan-filemeta"></div>
|
||||||
|
<p class="stats rescan-hint">
|
||||||
|
Ръчни ключови думи (раздели със запетая) — записват се <strong>първи</strong> във всяка секция
|
||||||
|
от това сканиране. Ако файлът вече е сканиран, включи force.
|
||||||
|
</p>
|
||||||
|
<div class="rescan-row">
|
||||||
|
<input type="text" id="rescan-keywords" class="rescan-kw-input"
|
||||||
|
placeholder="напр. лимитни карти, справка, ATR"
|
||||||
|
autocomplete="off">
|
||||||
|
</div>
|
||||||
<div class="rescan-row">
|
<div class="rescan-row">
|
||||||
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
|
<label class="stats"><input type="checkbox" id="rescan-force"> force</label>
|
||||||
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
|
<button type="button" class="btn-primary" id="rescan-btn" onclick="startRescan()">Старт</button>
|
||||||
@@ -486,7 +497,7 @@
|
|||||||
<div class="api-ep"><span class="method post">POST</span><code>/api/keywords/{code}</code>
|
<div class="api-ep"><span class="method post">POST</span><code>/api/keywords/{code}</code>
|
||||||
<div class="desc">Body JSON: <code>{"keywords":"а, б, в"}</code> — обновява ключови думи.</div></div>
|
<div class="desc">Body JSON: <code>{"keywords":"а, б, в"}</code> — обновява ключови думи.</div></div>
|
||||||
<div class="api-ep"><span class="method post">POST</span><code>/api/rescan</code>
|
<div class="api-ep"><span class="method post">POST</span><code>/api/rescan</code>
|
||||||
<div class="desc">multipart: <code>file</code> (един или повече: zip/docx/…), <code>prefix</code>, <code>force</code>. Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
<div class="desc">multipart: <code>file</code> (един или повече: zip/docx/…), <code>prefix</code>, <code>force</code>, по избор <code>keywords</code> (запетая; първи във всяка секция). Header опционално: <code>X-Rescan-Token</code>.</div></div>
|
||||||
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
|
<div class="api-ep"><span class="method">GET</span><code>/api/rescan/{job_id}</code>
|
||||||
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
|
<div class="desc">Статус на rescan задача (<code>queued</code> / <code>processing</code> / <code>done</code> / <code>error</code>).</div></div>
|
||||||
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
<div class="api-ep"><span class="method">GET</span><code>/api/file-extractions?file=…&prefix=RIP</code>
|
||||||
@@ -866,6 +877,7 @@ async function startRescan() {
|
|||||||
scanFiles.forEach(f => fd.append('file', f));
|
scanFiles.forEach(f => fd.append('file', f));
|
||||||
fd.append('prefix', PAGE_PREFIX);
|
fd.append('prefix', PAGE_PREFIX);
|
||||||
fd.append('force', document.getElementById('rescan-force').checked ? 'true' : 'false');
|
fd.append('force', document.getElementById('rescan-force').checked ? 'true' : 'false');
|
||||||
|
fd.append('keywords', (document.getElementById('rescan-keywords').value || '').trim());
|
||||||
const token = localStorage.getItem('RESCAN_TOKEN') || '';
|
const token = localStorage.getItem('RESCAN_TOKEN') || '';
|
||||||
if (token) fd.append('token', token);
|
if (token) fd.append('token', token);
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user