Fix extracts: preserve list and line structure instead of collapsing whitespace.

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
2026-09-09 14:17:06 +03:00
parent d323bee712
commit 1db58adafd

View File

@@ -373,10 +373,17 @@ def _detect_html_encoding(raw: bytes) -> str:
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
"p", "ul", "ol", "table", "dl", "pre",
"blockquote", "figure", "hr"]
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
"valign", "width", "height", "bgcolor", "border")
def _html_block_plain_text(el) -> str:
"""Plain ingest text: newlines inside lists/tables, spaces for inline runs."""
sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " "
return el.get_text(sep, strip=True)
def _strip_attrs(el):
"""Премахва decorative атрибути (class, style, on*, data-*)."""
for t in el.find_all(True):
@@ -471,7 +478,7 @@ def parse_html(path: Path) -> list[Section]:
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
_strip_attrs(el)
txt = el.get_text(" ", strip=True)
txt = _html_block_plain_text(el)
if txt:
sec_text.append(txt)
try:
@@ -668,6 +675,29 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
return None
_PDF_LINE_Y_TOL = 3.0
def _pdf_words_to_lines(words: list) -> list[dict]:
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
lines: list[dict] = []
for w in words:
y = float(w.get("top", 0))
if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL:
lines[-1]["words"].append(w)
else:
lines.append({"top": y, "words": [w]})
result = []
for line in lines:
ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0)))
text = " ".join(ww["text"] for ww in ws).strip()
if not text:
continue
size = round(float(ws[0].get("size", 10)), 1)
result.append({"top": line["top"], "size": size, "text": text})
return result
def parse_pdf(path: Path) -> list[Section]:
if not HAS_PDF:
log.warning("pdfplumber не е инсталиран. PDF се прескача.")
@@ -701,6 +731,7 @@ def parse_pdf(path: Path) -> list[Section]:
img_queue.append((float(im.get("top", 0)), data))
words = page.extract_words(extra_attrs=["size"])
visual_lines = _pdf_words_to_lines(words)
line_buf, line_size = [], None
def emit_images_before(y: float):
@@ -712,29 +743,29 @@ def parse_pdf(path: Path) -> list[Section]:
sec_images.append(ref)
buf.append(f"[IMG: {ref.placeholder}]")
for w in words:
sz = round(float(w.get("size", 10)), 1)
y = float(w.get("top", 0))
for vl in visual_lines:
sz = vl["size"]
y = vl["top"]
if line_size is None:
line_size = sz
if abs(sz - line_size) > 1:
line_text = " ".join(line_buf).strip()
line_text = "\n".join(line_buf).strip()
if line_text:
if line_size > (prev_size or 10) + 1 and len(line_text) < 150:
flush()
buf, sec_images = [], []
current_title = line_text
current_title = " ".join(line_text.split())
else:
emit_images_before(y)
buf.append(line_text)
prev_size = line_size
line_buf, line_size = [w["text"]], sz
line_buf, line_size = [vl["text"]], sz
else:
line_buf.append(w["text"])
line_buf.append(vl["text"])
if line_buf:
emit_images_before(page.height)
buf.append(" ".join(line_buf))
buf.append("\n".join(line_buf))
# картинките след всичкия текст на страницата
emit_images_before(page.height + 1)
@@ -788,8 +819,11 @@ def merge_short_sections(sections: list[Section]) -> list[Section]:
def clean_text(text: str) -> str:
text = re.sub(r"\s+", " ", text)
text = re.sub(r" {2,}", " ", text)
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
text = text.replace("\r\n", "\n").replace("\r", "\n")
text = re.sub(r"[^\S\n]+", " ", text)
text = re.sub(r"\n{3,}", "\n\n", text)
text = "\n".join(line.strip() for line in text.split("\n"))
return text.strip()