Fix PDF extracts: split TOC list items and merge drop-cap letters.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -676,25 +676,128 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
|
||||
|
||||
|
||||
_PDF_LINE_Y_TOL = 3.0
|
||||
# Typical word space in these PDFs is ~0.2–0.3em; only merge tighter (or drop-caps).
|
||||
_PDF_LETTER_GAP_FRAC = 0.12
|
||||
# Bare "1" / "2." / "3)" followed by a capital — TOC/list, not "виж 1 и 2".
|
||||
_PDF_LIST_MARK_RE = re.compile(
|
||||
r"(?<!\d)(?<![A-Za-zА-Яа-яЁёІіЇїЄєҐґ])(\d{1,2})([.)])?(?=\s+[A-ZА-ЯЁІЇЄҐ])"
|
||||
)
|
||||
|
||||
|
||||
def _pdf_word_band(w) -> tuple[float, float]:
|
||||
top = float(w.get("top", 0))
|
||||
bot = float(w.get("bottom", top + float(w.get("size") or 10)))
|
||||
if bot <= top:
|
||||
bot = top + max(float(w.get("size") or 10), 1.0)
|
||||
return top, bot
|
||||
|
||||
|
||||
def _pdf_cluster_words_by_y(words: list) -> list[dict]:
|
||||
"""Visual lines by Y (and vertical overlap), not PDF stream order."""
|
||||
items = [w for w in words if (w.get("text") or "").strip()]
|
||||
items.sort(key=lambda w: (float(w.get("top", 0)), float(w.get("x0", 0))))
|
||||
lines: list[dict] = []
|
||||
for w in items:
|
||||
top, bot = _pdf_word_band(w)
|
||||
placed = False
|
||||
for line in reversed(lines):
|
||||
overlap = min(bot, line["bot"]) - max(top, line["top"])
|
||||
band = min(bot - top, line["bot"] - line["top"])
|
||||
close = abs(top - line["top"]) <= _PDF_LINE_Y_TOL
|
||||
if close or (band > 0 and overlap >= 0.35 * band):
|
||||
line["words"].append(w)
|
||||
line["top"] = min(line["top"], top)
|
||||
line["bot"] = max(line["bot"], bot)
|
||||
placed = True
|
||||
break
|
||||
# Sorted by top: once this word sits clearly below a line, earlier
|
||||
# lines are even higher.
|
||||
if top > line["bot"] + _PDF_LINE_Y_TOL:
|
||||
break
|
||||
if not placed:
|
||||
lines.append({"top": top, "bot": bot, "words": [w]})
|
||||
lines.sort(key=lambda ln: ln["top"])
|
||||
return lines
|
||||
|
||||
|
||||
def _pdf_merge_split_letters(ws: list) -> list[dict]:
|
||||
"""Join drop-cap / styled first letters: 'C' + 'orrelation' → 'Correlation'."""
|
||||
ordered = sorted(ws, key=lambda w: float(w.get("x0", 0)))
|
||||
clusters: list[dict] = []
|
||||
for w in ordered:
|
||||
text = (w.get("text") or "").strip()
|
||||
if not text:
|
||||
continue
|
||||
x0 = float(w.get("x0", 0))
|
||||
x1 = float(w.get("x1", x0))
|
||||
size = float(w.get("size") or 10)
|
||||
font = str(w.get("fontname") or "")
|
||||
if not clusters:
|
||||
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
|
||||
continue
|
||||
prev = clusters[-1]
|
||||
gap = x0 - prev["x1"]
|
||||
ref = max(size, prev["size"], 1.0)
|
||||
fonts_differ = bool(prev["font"] and font and prev["font"] != font)
|
||||
sizes_differ = abs(size - prev["size"]) > 0.6
|
||||
drop = (
|
||||
len(prev["text"]) == 1
|
||||
and prev["text"].isalpha()
|
||||
and text[0].islower()
|
||||
)
|
||||
tiny = gap < _PDF_LETTER_GAP_FRAC * ref or gap < 0.8
|
||||
if tiny or (drop and gap < 0.55 * ref and (fonts_differ or sizes_differ or gap < 0.2 * ref)):
|
||||
prev["text"] += text
|
||||
prev["x1"] = max(prev["x1"], x1)
|
||||
prev["size"] = max(prev["size"], size)
|
||||
if font:
|
||||
prev["font"] = font
|
||||
else:
|
||||
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
|
||||
return clusters
|
||||
|
||||
|
||||
def _pdf_split_list_text(text: str) -> list[str]:
|
||||
"""Break a flattened TOC/list: '... 1 Foo 2 Bar' → one item per line."""
|
||||
text = text.strip()
|
||||
if not text:
|
||||
return []
|
||||
matches = list(_PDF_LIST_MARK_RE.finditer(text))
|
||||
if len(matches) < 2:
|
||||
return [text]
|
||||
nums = [int(m.group(1)) for m in matches]
|
||||
if nums[0] not in (1, 2):
|
||||
return [text]
|
||||
if any(nums[i] != nums[0] + i for i in range(len(nums))):
|
||||
return [text]
|
||||
parts: list[str] = []
|
||||
last = 0
|
||||
for m in matches:
|
||||
if m.start() > last:
|
||||
head = text[last:m.start()].strip()
|
||||
if head:
|
||||
parts.append(head)
|
||||
last = m.start()
|
||||
tail = text[last:].strip()
|
||||
if tail:
|
||||
parts.append(tail)
|
||||
return parts or [text]
|
||||
|
||||
|
||||
def _pdf_words_to_lines(words: list) -> list[dict]:
|
||||
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
|
||||
lines: list[dict] = []
|
||||
for w in words:
|
||||
y = float(w.get("top", 0))
|
||||
if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL:
|
||||
lines[-1]["words"].append(w)
|
||||
else:
|
||||
lines.append({"top": y, "words": [w]})
|
||||
result = []
|
||||
for line in lines:
|
||||
ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0)))
|
||||
text = " ".join(ww["text"] for ww in ws).strip()
|
||||
if not text:
|
||||
for line in _pdf_cluster_words_by_y(words):
|
||||
ws = line["words"]
|
||||
tokens = _pdf_merge_split_letters(ws)
|
||||
if not tokens:
|
||||
continue
|
||||
joined = " ".join(t["text"] for t in tokens).strip()
|
||||
if not joined:
|
||||
continue
|
||||
size = round(float(ws[0].get("size", 10)), 1)
|
||||
result.append({"top": line["top"], "size": size, "text": text})
|
||||
for piece in _pdf_split_list_text(joined):
|
||||
result.append({"top": line["top"], "size": size, "text": piece})
|
||||
return result
|
||||
|
||||
|
||||
@@ -730,7 +833,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
||||
continue
|
||||
img_queue.append((float(im.get("top", 0)), data))
|
||||
|
||||
words = page.extract_words(extra_attrs=["size"])
|
||||
words = page.extract_words(extra_attrs=["size", "fontname"])
|
||||
visual_lines = _pdf_words_to_lines(words)
|
||||
line_buf, line_size = [], None
|
||||
|
||||
|
||||
Reference in New Issue
Block a user