Fix extracts: preserve list and line structure instead of collapsing whitespace.
Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
@@ -373,10 +373,17 @@ def _detect_html_encoding(raw: bytes) -> str:
|
|||||||
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
_HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
||||||
"p", "ul", "ol", "table", "dl", "pre",
|
"p", "ul", "ol", "table", "dl", "pre",
|
||||||
"blockquote", "figure", "hr"]
|
"blockquote", "figure", "hr"]
|
||||||
|
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||||||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||||||
"valign", "width", "height", "bgcolor", "border")
|
"valign", "width", "height", "bgcolor", "border")
|
||||||
|
|
||||||
|
|
||||||
|
def _html_block_plain_text(el) -> str:
|
||||||
|
"""Plain ingest text: newlines inside lists/tables, spaces for inline runs."""
|
||||||
|
sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " "
|
||||||
|
return el.get_text(sep, strip=True)
|
||||||
|
|
||||||
|
|
||||||
def _strip_attrs(el):
|
def _strip_attrs(el):
|
||||||
"""Премахва decorative атрибути (class, style, on*, data-*)."""
|
"""Премахва decorative атрибути (class, style, on*, data-*)."""
|
||||||
for t in el.find_all(True):
|
for t in el.find_all(True):
|
||||||
@@ -471,7 +478,7 @@ def parse_html(path: Path) -> list[Section]:
|
|||||||
|
|
||||||
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
|
_swap_imgs_in_block(el, base_dir, sec_images, img_counter)
|
||||||
_strip_attrs(el)
|
_strip_attrs(el)
|
||||||
txt = el.get_text(" ", strip=True)
|
txt = _html_block_plain_text(el)
|
||||||
if txt:
|
if txt:
|
||||||
sec_text.append(txt)
|
sec_text.append(txt)
|
||||||
try:
|
try:
|
||||||
@@ -668,6 +675,29 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
_PDF_LINE_Y_TOL = 3.0
|
||||||
|
|
||||||
|
|
||||||
|
def _pdf_words_to_lines(words: list) -> list[dict]:
|
||||||
|
"""Group pdfplumber words into visual lines by Y, then left-to-right."""
|
||||||
|
lines: list[dict] = []
|
||||||
|
for w in words:
|
||||||
|
y = float(w.get("top", 0))
|
||||||
|
if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL:
|
||||||
|
lines[-1]["words"].append(w)
|
||||||
|
else:
|
||||||
|
lines.append({"top": y, "words": [w]})
|
||||||
|
result = []
|
||||||
|
for line in lines:
|
||||||
|
ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0)))
|
||||||
|
text = " ".join(ww["text"] for ww in ws).strip()
|
||||||
|
if not text:
|
||||||
|
continue
|
||||||
|
size = round(float(ws[0].get("size", 10)), 1)
|
||||||
|
result.append({"top": line["top"], "size": size, "text": text})
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
def parse_pdf(path: Path) -> list[Section]:
|
def parse_pdf(path: Path) -> list[Section]:
|
||||||
if not HAS_PDF:
|
if not HAS_PDF:
|
||||||
log.warning("pdfplumber не е инсталиран. PDF се прескача.")
|
log.warning("pdfplumber не е инсталиран. PDF се прескача.")
|
||||||
@@ -701,6 +731,7 @@ def parse_pdf(path: Path) -> list[Section]:
|
|||||||
img_queue.append((float(im.get("top", 0)), data))
|
img_queue.append((float(im.get("top", 0)), data))
|
||||||
|
|
||||||
words = page.extract_words(extra_attrs=["size"])
|
words = page.extract_words(extra_attrs=["size"])
|
||||||
|
visual_lines = _pdf_words_to_lines(words)
|
||||||
line_buf, line_size = [], None
|
line_buf, line_size = [], None
|
||||||
|
|
||||||
def emit_images_before(y: float):
|
def emit_images_before(y: float):
|
||||||
@@ -712,29 +743,29 @@ def parse_pdf(path: Path) -> list[Section]:
|
|||||||
sec_images.append(ref)
|
sec_images.append(ref)
|
||||||
buf.append(f"[IMG: {ref.placeholder}]")
|
buf.append(f"[IMG: {ref.placeholder}]")
|
||||||
|
|
||||||
for w in words:
|
for vl in visual_lines:
|
||||||
sz = round(float(w.get("size", 10)), 1)
|
sz = vl["size"]
|
||||||
y = float(w.get("top", 0))
|
y = vl["top"]
|
||||||
if line_size is None:
|
if line_size is None:
|
||||||
line_size = sz
|
line_size = sz
|
||||||
if abs(sz - line_size) > 1:
|
if abs(sz - line_size) > 1:
|
||||||
line_text = " ".join(line_buf).strip()
|
line_text = "\n".join(line_buf).strip()
|
||||||
if line_text:
|
if line_text:
|
||||||
if line_size > (prev_size or 10) + 1 and len(line_text) < 150:
|
if line_size > (prev_size or 10) + 1 and len(line_text) < 150:
|
||||||
flush()
|
flush()
|
||||||
buf, sec_images = [], []
|
buf, sec_images = [], []
|
||||||
current_title = line_text
|
current_title = " ".join(line_text.split())
|
||||||
else:
|
else:
|
||||||
emit_images_before(y)
|
emit_images_before(y)
|
||||||
buf.append(line_text)
|
buf.append(line_text)
|
||||||
prev_size = line_size
|
prev_size = line_size
|
||||||
line_buf, line_size = [w["text"]], sz
|
line_buf, line_size = [vl["text"]], sz
|
||||||
else:
|
else:
|
||||||
line_buf.append(w["text"])
|
line_buf.append(vl["text"])
|
||||||
|
|
||||||
if line_buf:
|
if line_buf:
|
||||||
emit_images_before(page.height)
|
emit_images_before(page.height)
|
||||||
buf.append(" ".join(line_buf))
|
buf.append("\n".join(line_buf))
|
||||||
|
|
||||||
# картинките след всичкия текст на страницата
|
# картинките след всичкия текст на страницата
|
||||||
emit_images_before(page.height + 1)
|
emit_images_before(page.height + 1)
|
||||||
@@ -788,8 +819,11 @@ def merge_short_sections(sections: list[Section]) -> list[Section]:
|
|||||||
|
|
||||||
|
|
||||||
def clean_text(text: str) -> str:
|
def clean_text(text: str) -> str:
|
||||||
text = re.sub(r"\s+", " ", text)
|
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
|
||||||
text = re.sub(r" {2,}", " ", text)
|
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
||||||
|
text = re.sub(r"[^\S\n]+", " ", text)
|
||||||
|
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||||
|
text = "\n".join(line.strip() for line in text.split("\n"))
|
||||||
return text.strip()
|
return text.strip()
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user