diff --git a/help_processor.py b/help_processor.py index 1f5d7dd..57bf277 100644 --- a/help_processor.py +++ b/help_processor.py @@ -676,25 +676,128 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]: _PDF_LINE_Y_TOL = 3.0 +# Typical word space in these PDFs is ~0.2–0.3em; only merge tighter (or drop-caps). +_PDF_LETTER_GAP_FRAC = 0.12 +# Bare "1" / "2." / "3)" followed by a capital — TOC/list, not "виж 1 и 2". +_PDF_LIST_MARK_RE = re.compile( + r"(? tuple[float, float]: + top = float(w.get("top", 0)) + bot = float(w.get("bottom", top + float(w.get("size") or 10))) + if bot <= top: + bot = top + max(float(w.get("size") or 10), 1.0) + return top, bot + + +def _pdf_cluster_words_by_y(words: list) -> list[dict]: + """Visual lines by Y (and vertical overlap), not PDF stream order.""" + items = [w for w in words if (w.get("text") or "").strip()] + items.sort(key=lambda w: (float(w.get("top", 0)), float(w.get("x0", 0)))) + lines: list[dict] = [] + for w in items: + top, bot = _pdf_word_band(w) + placed = False + for line in reversed(lines): + overlap = min(bot, line["bot"]) - max(top, line["top"]) + band = min(bot - top, line["bot"] - line["top"]) + close = abs(top - line["top"]) <= _PDF_LINE_Y_TOL + if close or (band > 0 and overlap >= 0.35 * band): + line["words"].append(w) + line["top"] = min(line["top"], top) + line["bot"] = max(line["bot"], bot) + placed = True + break + # Sorted by top: once this word sits clearly below a line, earlier + # lines are even higher. + if top > line["bot"] + _PDF_LINE_Y_TOL: + break + if not placed: + lines.append({"top": top, "bot": bot, "words": [w]}) + lines.sort(key=lambda ln: ln["top"]) + return lines + + +def _pdf_merge_split_letters(ws: list) -> list[dict]: + """Join drop-cap / styled first letters: 'C' + 'orrelation' → 'Correlation'.""" + ordered = sorted(ws, key=lambda w: float(w.get("x0", 0))) + clusters: list[dict] = [] + for w in ordered: + text = (w.get("text") or "").strip() + if not text: + continue + x0 = float(w.get("x0", 0)) + x1 = float(w.get("x1", x0)) + size = float(w.get("size") or 10) + font = str(w.get("fontname") or "") + if not clusters: + clusters.append({"text": text, "x1": x1, "size": size, "font": font}) + continue + prev = clusters[-1] + gap = x0 - prev["x1"] + ref = max(size, prev["size"], 1.0) + fonts_differ = bool(prev["font"] and font and prev["font"] != font) + sizes_differ = abs(size - prev["size"]) > 0.6 + drop = ( + len(prev["text"]) == 1 + and prev["text"].isalpha() + and text[0].islower() + ) + tiny = gap < _PDF_LETTER_GAP_FRAC * ref or gap < 0.8 + if tiny or (drop and gap < 0.55 * ref and (fonts_differ or sizes_differ or gap < 0.2 * ref)): + prev["text"] += text + prev["x1"] = max(prev["x1"], x1) + prev["size"] = max(prev["size"], size) + if font: + prev["font"] = font + else: + clusters.append({"text": text, "x1": x1, "size": size, "font": font}) + return clusters + + +def _pdf_split_list_text(text: str) -> list[str]: + """Break a flattened TOC/list: '... 1 Foo 2 Bar' → one item per line.""" + text = text.strip() + if not text: + return [] + matches = list(_PDF_LIST_MARK_RE.finditer(text)) + if len(matches) < 2: + return [text] + nums = [int(m.group(1)) for m in matches] + if nums[0] not in (1, 2): + return [text] + if any(nums[i] != nums[0] + i for i in range(len(nums))): + return [text] + parts: list[str] = [] + last = 0 + for m in matches: + if m.start() > last: + head = text[last:m.start()].strip() + if head: + parts.append(head) + last = m.start() + tail = text[last:].strip() + if tail: + parts.append(tail) + return parts or [text] def _pdf_words_to_lines(words: list) -> list[dict]: """Group pdfplumber words into visual lines by Y, then left-to-right.""" - lines: list[dict] = [] - for w in words: - y = float(w.get("top", 0)) - if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL: - lines[-1]["words"].append(w) - else: - lines.append({"top": y, "words": [w]}) result = [] - for line in lines: - ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0))) - text = " ".join(ww["text"] for ww in ws).strip() - if not text: + for line in _pdf_cluster_words_by_y(words): + ws = line["words"] + tokens = _pdf_merge_split_letters(ws) + if not tokens: + continue + joined = " ".join(t["text"] for t in tokens).strip() + if not joined: continue size = round(float(ws[0].get("size", 10)), 1) - result.append({"top": line["top"], "size": size, "text": text}) + for piece in _pdf_split_list_text(joined): + result.append({"top": line["top"], "size": size, "text": piece}) return result @@ -730,7 +833,7 @@ def parse_pdf(path: Path) -> list[Section]: continue img_queue.append((float(im.get("top", 0)), data)) - words = page.extract_words(extra_attrs=["size"]) + words = page.extract_words(extra_attrs=["size", "fontname"]) visual_lines = _pdf_words_to_lines(words) line_buf, line_size = [], None