Fix extracts: preserve list and line structure instead of collapsing whitespace.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -373,10 +373,17 @@ def _detect_html_encoding(raw: bytes) -> str:
|
||||
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
||||
"p", "ul", "ol", "table", "dl", "pre",
|
||||
"blockquote", "figure", "hr"]
|
||||
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||||
"valign", "width", "height", "bgcolor", "border")
|
||||
|
||||
|
||||
def _html_block_plain_text(el) -> str:
|
||||
"""Plain ingest text: newlines inside lists/tables, spaces for inline runs."""
|
||||
sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " "
|
||||
return el.get_text(sep, strip=True)
|
||||
|
||||
|
||||
def _strip_attrs(el):
|
||||
"""Премахва decorative атрибути (class, style, on*, data-*)."""
|
||||
for t in el.find_all(True):
|
||||
@@ -471,7 +478,7 @@ def parse_html(path: Path) -> list[Section]:
|
||||
|
||||
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
|
||||
_strip_attrs(el)
|
||||
txt = el.get_text(" ", strip=True)
|
||||
txt = _html_block_plain_text(el)
|
||||
if txt:
|
||||
sec_text.append(txt)
|
||||
try:
|
||||
@@ -668,6 +675,29 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
|
||||
return None
|
||||
|
||||
|
||||
_PDF_LINE_Y_TOL = 3.0
|
||||
|
||||
|
||||
def _pdf_words_to_lines(words: list) -> list[dict]:
|
||||
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
|
||||
lines: list[dict] = []
|
||||
for w in words:
|
||||
y = float(w.get("top", 0))
|
||||
if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL:
|
||||
lines[-1]["words"].append(w)
|
||||
else:
|
||||
lines.append({"top": y, "words": [w]})
|
||||
result = []
|
||||
for line in lines:
|
||||
ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0)))
|
||||
text = " ".join(ww["text"] for ww in ws).strip()
|
||||
if not text:
|
||||
continue
|
||||
size = round(float(ws[0].get("size", 10)), 1)
|
||||
result.append({"top": line["top"], "size": size, "text": text})
|
||||
return result
|
||||
|
||||
|
||||
def parse_pdf(path: Path) -> list[Section]:
|
||||
if not HAS_PDF:
|
||||
log.warning("pdfplumber не е инсталиран. PDF се прескача.")
|
||||
@@ -701,6 +731,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
||||
img_queue.append((float(im.get("top", 0)), data))
|
||||
|
||||
words = page.extract_words(extra_attrs=["size"])
|
||||
visual_lines = _pdf_words_to_lines(words)
|
||||
line_buf, line_size = [], None
|
||||
|
||||
def emit_images_before(y: float):
|
||||
@@ -712,29 +743,29 @@ def parse_pdf(path: Path) -> list[Section]:
|
||||
sec_images.append(ref)
|
||||
buf.append(f"[IMG: {ref.placeholder}]")
|
||||
|
||||
for w in words:
|
||||
sz = round(float(w.get("size", 10)), 1)
|
||||
y = float(w.get("top", 0))
|
||||
for vl in visual_lines:
|
||||
sz = vl["size"]
|
||||
y = vl["top"]
|
||||
if line_size is None:
|
||||
line_size = sz
|
||||
if abs(sz - line_size) > 1:
|
||||
line_text = " ".join(line_buf).strip()
|
||||
line_text = "\n".join(line_buf).strip()
|
||||
if line_text:
|
||||
if line_size > (prev_size or 10) + 1 and len(line_text) < 150:
|
||||
flush()
|
||||
buf, sec_images = [], []
|
||||
current_title = line_text
|
||||
current_title = " ".join(line_text.split())
|
||||
else:
|
||||
emit_images_before(y)
|
||||
buf.append(line_text)
|
||||
prev_size = line_size
|
||||
line_buf, line_size = [w["text"]], sz
|
||||
line_buf, line_size = [vl["text"]], sz
|
||||
else:
|
||||
line_buf.append(w["text"])
|
||||
line_buf.append(vl["text"])
|
||||
|
||||
if line_buf:
|
||||
emit_images_before(page.height)
|
||||
buf.append(" ".join(line_buf))
|
||||
buf.append("\n".join(line_buf))
|
||||
|
||||
# картинките след всичкия текст на страницата
|
||||
emit_images_before(page.height + 1)
|
||||
@@ -788,8 +819,11 @@ def merge_short_sections(sections: list[Section]) -> list[Section]:
|
||||
|
||||
|
||||
def clean_text(text: str) -> str:
|
||||
text = re.sub(r"\s+", " ", text)
|
||||
text = re.sub(r" {2,}", " ", text)
|
||||
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
|
||||
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
||||
text = re.sub(r"[^\S\n]+", " ", text)
|
||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||
text = "\n".join(line.strip() for line in text.split("\n"))
|
||||
return text.strip()
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user