Fix PDF extracts: split TOC list items and merge drop-cap letters.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -676,25 +676,128 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
|
|||||||
|
|
||||||
|
|
||||||
_PDF_LINE_Y_TOL = 3.0
|
_PDF_LINE_Y_TOL = 3.0
|
||||||
|
# Typical word space in these PDFs is ~0.2–0.3em; only merge tighter (or drop-caps).
|
||||||
|
_PDF_LETTER_GAP_FRAC = 0.12
|
||||||
|
# Bare "1" / "2." / "3)" followed by a capital — TOC/list, not "виж 1 и 2".
|
||||||
|
_PDF_LIST_MARK_RE = re.compile(
|
||||||
|
r"(?<!\d)(?<![A-Za-zА-Яа-яЁёІіЇїЄєҐґ])(\d{1,2})([.)])?(?=\s+[A-ZА-ЯЁІЇЄҐ])"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_word_band(w) -> tuple[float, float]:
|
||||||
|
top = float(w.get("top", 0))
|
||||||
|
bot = float(w.get("bottom", top + float(w.get("size") or 10)))
|
||||||
|
if bot <= top:
|
||||||
|
bot = top + max(float(w.get("size") or 10), 1.0)
|
||||||
|
return top, bot
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_cluster_words_by_y(words: list) -> list[dict]:
|
||||||
|
"""Visual lines by Y (and vertical overlap), not PDF stream order."""
|
||||||
|
items = [w for w in words if (w.get("text") or "").strip()]
|
||||||
|
items.sort(key=lambda w: (float(w.get("top", 0)), float(w.get("x0", 0))))
|
||||||
|
lines: list[dict] = []
|
||||||
|
for w in items:
|
||||||
|
top, bot = _pdf_word_band(w)
|
||||||
|
placed = False
|
||||||
|
for line in reversed(lines):
|
||||||
|
overlap = min(bot, line["bot"]) - max(top, line["top"])
|
||||||
|
band = min(bot - top, line["bot"] - line["top"])
|
||||||
|
close = abs(top - line["top"]) <= _PDF_LINE_Y_TOL
|
||||||
|
if close or (band > 0 and overlap >= 0.35 * band):
|
||||||
|
line["words"].append(w)
|
||||||
|
line["top"] = min(line["top"], top)
|
||||||
|
line["bot"] = max(line["bot"], bot)
|
||||||
|
placed = True
|
||||||
|
break
|
||||||
|
# Sorted by top: once this word sits clearly below a line, earlier
|
||||||
|
# lines are even higher.
|
||||||
|
if top > line["bot"] + _PDF_LINE_Y_TOL:
|
||||||
|
break
|
||||||
|
if not placed:
|
||||||
|
lines.append({"top": top, "bot": bot, "words": [w]})
|
||||||
|
lines.sort(key=lambda ln: ln["top"])
|
||||||
|
return lines
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_merge_split_letters(ws: list) -> list[dict]:
|
||||||
|
"""Join drop-cap / styled first letters: 'C' + 'orrelation' → 'Correlation'."""
|
||||||
|
ordered = sorted(ws, key=lambda w: float(w.get("x0", 0)))
|
||||||
|
clusters: list[dict] = []
|
||||||
|
for w in ordered:
|
||||||
|
text = (w.get("text") or "").strip()
|
||||||
|
if not text:
|
||||||
|
continue
|
||||||
|
x0 = float(w.get("x0", 0))
|
||||||
|
x1 = float(w.get("x1", x0))
|
||||||
|
size = float(w.get("size") or 10)
|
||||||
|
font = str(w.get("fontname") or "")
|
||||||
|
if not clusters:
|
||||||
|
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
|
||||||
|
continue
|
||||||
|
prev = clusters[-1]
|
||||||
|
gap = x0 - prev["x1"]
|
||||||
|
ref = max(size, prev["size"], 1.0)
|
||||||
|
fonts_differ = bool(prev["font"] and font and prev["font"] != font)
|
||||||
|
sizes_differ = abs(size - prev["size"]) > 0.6
|
||||||
|
drop = (
|
||||||
|
len(prev["text"]) == 1
|
||||||
|
and prev["text"].isalpha()
|
||||||
|
and text[0].islower()
|
||||||
|
)
|
||||||
|
tiny = gap < _PDF_LETTER_GAP_FRAC * ref or gap < 0.8
|
||||||
|
if tiny or (drop and gap < 0.55 * ref and (fonts_differ or sizes_differ or gap < 0.2 * ref)):
|
||||||
|
prev["text"] += text
|
||||||
|
prev["x1"] = max(prev["x1"], x1)
|
||||||
|
prev["size"] = max(prev["size"], size)
|
||||||
|
if font:
|
||||||
|
prev["font"] = font
|
||||||
|
else:
|
||||||
|
clusters.append({"text": text, "x1": x1, "size": size, "font": font})
|
||||||
|
return clusters
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_split_list_text(text: str) -> list[str]:
|
||||||
|
"""Break a flattened TOC/list: '... 1 Foo 2 Bar' → one item per line."""
|
||||||
|
text = text.strip()
|
||||||
|
if not text:
|
||||||
|
return []
|
||||||
|
matches = list(_PDF_LIST_MARK_RE.finditer(text))
|
||||||
|
if len(matches) < 2:
|
||||||
|
return [text]
|
||||||
|
nums = [int(m.group(1)) for m in matches]
|
||||||
|
if nums[0] not in (1, 2):
|
||||||
|
return [text]
|
||||||
|
if any(nums[i] != nums[0] + i for i in range(len(nums))):
|
||||||
|
return [text]
|
||||||
|
parts: list[str] = []
|
||||||
|
last = 0
|
||||||
|
for m in matches:
|
||||||
|
if m.start() > last:
|
||||||
|
head = text[last:m.start()].strip()
|
||||||
|
if head:
|
||||||
|
parts.append(head)
|
||||||
|
last = m.start()
|
||||||
|
tail = text[last:].strip()
|
||||||
|
if tail:
|
||||||
|
parts.append(tail)
|
||||||
|
return parts or [text]
|
||||||
|
|
||||||
|
|
||||||
def _pdf_words_to_lines(words: list) -> list[dict]:
|
def _pdf_words_to_lines(words: list) -> list[dict]:
|
||||||
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
|
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
|
||||||
lines: list[dict] = []
|
|
||||||
for w in words:
|
|
||||||
y = float(w.get("top", 0))
|
|
||||||
if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL:
|
|
||||||
lines[-1]["words"].append(w)
|
|
||||||
else:
|
|
||||||
lines.append({"top": y, "words": [w]})
|
|
||||||
result = []
|
result = []
|
||||||
for line in lines:
|
for line in _pdf_cluster_words_by_y(words):
|
||||||
ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0)))
|
ws = line["words"]
|
||||||
text = " ".join(ww["text"] for ww in ws).strip()
|
tokens = _pdf_merge_split_letters(ws)
|
||||||
if not text:
|
if not tokens:
|
||||||
|
continue
|
||||||
|
joined = " ".join(t["text"] for t in tokens).strip()
|
||||||
|
if not joined:
|
||||||
continue
|
continue
|
||||||
size = round(float(ws[0].get("size", 10)), 1)
|
size = round(float(ws[0].get("size", 10)), 1)
|
||||||
result.append({"top": line["top"], "size": size, "text": text})
|
for piece in _pdf_split_list_text(joined):
|
||||||
|
result.append({"top": line["top"], "size": size, "text": piece})
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
@@ -730,7 +833,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
|||||||
continue
|
continue
|
||||||
img_queue.append((float(im.get("top", 0)), data))
|
img_queue.append((float(im.get("top", 0)), data))
|
||||||
|
|
||||||
words = page.extract_words(extra_attrs=["size"])
|
words = page.extract_words(extra_attrs=["size", "fontname"])
|
||||||
visual_lines = _pdf_words_to_lines(words)
|
visual_lines = _pdf_words_to_lines(words)
|
||||||
line_buf, line_size = [], None
|
line_buf, line_size = [], None
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user