Fix PDF extracts: split TOC list items and merge drop-cap letters.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-09-09 14:57:24 +03:00
parent 1db58adafd
commit f060cf700d

View File

@@ -676,25 +676,128 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
_PDF_LINE_Y_TOL = 3.0 _PDF_LINE_Y_TOL = 3.0
# Typical word space in these PDFs is ~0.2–0.3em; only merge tighter (or drop-caps).
_PDF_LETTER_GAP_FRAC = 0.12
# Bare "1" / "2." / "3)" followed by a capital — TOC/list, not "виж 1 и 2".
_PDF_LIST_MARK_RE = re.compile(
r"(?<!\d)(?<![A-Za-zА-Яа-яЁёІіЇїЄєҐґ])(\d{1,2})([.)])?(?=\s+[A-ZА-ЯЁІЇЄҐ])"
)
def _pdf_word_band(w) -> tuple[float, float]:
top = float(w.get("top", 0))
bot = float(w.get("bottom", top + float(w.get("size") or 10)))
if bot <= top:
bot = top + max(float(w.get("size") or 10), 1.0)
return top, bot
def _pdf_cluster_words_by_y(words: list) -> list[dict]:
"""Visual lines by Y (and vertical overlap), not PDF stream order."""
items = [w for w in words if (w.get("text") or "").strip()]
items.sort(key=lambda w: (float(w.get("top", 0)), float(w.get("x0", 0))))
lines: list[dict] = []
for w in items:
top, bot = _pdf_word_band(w)
placed = False
for line in reversed(lines):
overlap = min(bot, line["bot"]) - max(top, line["top"])
band = min(bot - top, line["bot"] - line["top"])
close = abs(top - line["top"]) <= _PDF_LINE_Y_TOL
if close or (band > 0 and overlap >= 0.35 * band):
line["words"].append(w)
line["top"] = min(line["top"], top)
line["bot"] = max(line["bot"], bot)
placed = True
break
# Sorted by top: once this word sits clearly below a line, earlier
# lines are even higher.
if top > line["bot"] + _PDF_LINE_Y_TOL:
break
if not placed:
lines.append({"top": top, "bot": bot, "words": [w]})
lines.sort(key=lambda ln: ln["top"])
return lines
def _pdf_merge_split_letters(ws: list) -> list[dict]:
"""Join drop-cap / styled first letters: 'C' + 'orrelation' → 'Correlation'."""
ordered = sorted(ws, key=lambda w: float(w.get("x0", 0)))
clusters: list[dict] = []
for w in ordered:
text = (w.get("text") or "").strip()
if not text:
continue
x0 = float(w.get("x0", 0))
x1 = float(w.get("x1", x0))
size = float(w.get("size") or 10)
font = str(w.get("fontname") or "")
if not clusters:
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
continue
prev = clusters[-1]
gap = x0 - prev["x1"]
ref = max(size, prev["size"], 1.0)
fonts_differ = bool(prev["font"] and font and prev["font"] != font)
sizes_differ = abs(size - prev["size"]) > 0.6
drop = (
len(prev["text"]) == 1
and prev["text"].isalpha()
and text[0].islower()
)
tiny = gap < _PDF_LETTER_GAP_FRAC * ref or gap < 0.8
if tiny or (drop and gap < 0.55 * ref and (fonts_differ or sizes_differ or gap < 0.2 * ref)):
prev["text"] += text
prev["x1"] = max(prev["x1"], x1)
prev["size"] = max(prev["size"], size)
if font:
prev["font"] = font
else:
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
return clusters
def _pdf_split_list_text(text: str) -> list[str]:
"""Break a flattened TOC/list: '... 1 Foo 2 Bar' → one item per line."""
text = text.strip()
if not text:
return []
matches = list(_PDF_LIST_MARK_RE.finditer(text))
if len(matches) < 2:
return [text]
nums = [int(m.group(1)) for m in matches]
if nums[0] not in (1, 2):
return [text]
if any(nums[i] != nums[0] + i for i in range(len(nums))):
return [text]
parts: list[str] = []
last = 0
for m in matches:
if m.start() > last:
head = text[last:m.start()].strip()
if head:
parts.append(head)
last = m.start()
tail = text[last:].strip()
if tail:
parts.append(tail)
return parts or [text]
def _pdf_words_to_lines(words: list) -> list[dict]: def _pdf_words_to_lines(words: list) -> list[dict]:
"""Group pdfplumber words into visual lines by Y, then left-to-right.""" """Group pdfplumber words into visual lines by Y, then left-to-right."""
lines: list[dict] = []
for w in words:
y = float(w.get("top", 0))
if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL:
lines[-1]["words"].append(w)
else:
lines.append({"top": y, "words": [w]})
result = [] result = []
for line in lines: for line in _pdf_cluster_words_by_y(words):
ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0))) ws = line["words"]
text = " ".join(ww["text"] for ww in ws).strip() tokens = _pdf_merge_split_letters(ws)
if not text: if not tokens:
continue
joined = " ".join(t["text"] for t in tokens).strip()
if not joined:
continue continue
size = round(float(ws[0].get("size", 10)), 1) size = round(float(ws[0].get("size", 10)), 1)
result.append({"top": line["top"], "size": size, "text": text}) for piece in _pdf_split_list_text(joined):
result.append({"top": line["top"], "size": size, "text": piece})
return result return result
@@ -730,7 +833,7 @@ def parse_pdf(path: Path) -> list[Section]:
continue continue
img_queue.append((float(im.get("top", 0)), data)) img_queue.append((float(im.get("top", 0)), data))
words = page.extract_words(extra_attrs=["size"]) words = page.extract_words(extra_attrs=["size", "fontname"])
visual_lines = _pdf_words_to_lines(words) visual_lines = _pdf_words_to_lines(words)
line_buf, line_size = [], None line_buf, line_size = [], None