diff --git a/help_processor.py b/help_processor.py index 4013205..1f5d7dd 100644 --- a/help_processor.py +++ b/help_processor.py @@ -373,10 +373,17 @@ def _detect_html_encoding(raw: bytes) -> str: _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6", "p", "ul", "ol", "table", "dl", "pre", "blockquote", "figure", "hr"] +_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"}) _HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align", "valign", "width", "height", "bgcolor", "border") +def _html_block_plain_text(el) -> str: + """Plain ingest text: newlines inside lists/tables, spaces for inline runs.""" + sep = "\n" if (el.name or "").lower() in _HTML_PLAIN_NL_TAGS else " " + return el.get_text(sep, strip=True) + + def _strip_attrs(el): """Премахва decorative атрибути (class, style, on*, data-*).""" for t in el.find_all(True): @@ -471,7 +478,7 @@ def parse_html(path: Path) -> list[Section]: _swap_imgs_in_block(el, base_dir, sec_images, img_counter) _strip_attrs(el) - txt = el.get_text(" ", strip=True) + txt = _html_block_plain_text(el) if txt: sec_text.append(txt) try: @@ -668,6 +675,29 @@ def _render_pdf_image(page, img_info, resolution: int = 150) -> Optional[bytes]: return None +_PDF_LINE_Y_TOL = 3.0 + + +def _pdf_words_to_lines(words: list) -> list[dict]: + """Group pdfplumber words into visual lines by Y, then left-to-right.""" + lines: list[dict] = [] + for w in words: + y = float(w.get("top", 0)) + if lines and abs(y - lines[-1]["top"]) <= _PDF_LINE_Y_TOL: + lines[-1]["words"].append(w) + else: + lines.append({"top": y, "words": [w]}) + result = [] + for line in lines: + ws = sorted(line["words"], key=lambda ww: float(ww.get("x0", 0))) + text = " ".join(ww["text"] for ww in ws).strip() + if not text: + continue + size = round(float(ws[0].get("size", 10)), 1) + result.append({"top": line["top"], "size": size, "text": text}) + return result + + def parse_pdf(path: Path) -> list[Section]: if not HAS_PDF: log.warning("pdfplumber не е инсталиран. PDF се прескача.") @@ -701,6 +731,7 @@ def parse_pdf(path: Path) -> list[Section]: img_queue.append((float(im.get("top", 0)), data)) words = page.extract_words(extra_attrs=["size"]) + visual_lines = _pdf_words_to_lines(words) line_buf, line_size = [], None def emit_images_before(y: float): @@ -712,29 +743,29 @@ def parse_pdf(path: Path) -> list[Section]: sec_images.append(ref) buf.append(f"[IMG: {ref.placeholder}]") - for w in words: - sz = round(float(w.get("size", 10)), 1) - y = float(w.get("top", 0)) + for vl in visual_lines: + sz = vl["size"] + y = vl["top"] if line_size is None: line_size = sz if abs(sz - line_size) > 1: - line_text = " ".join(line_buf).strip() + line_text = "\n".join(line_buf).strip() if line_text: if line_size > (prev_size or 10) + 1 and len(line_text) < 150: flush() buf, sec_images = [], [] - current_title = line_text + current_title = " ".join(line_text.split()) else: emit_images_before(y) buf.append(line_text) prev_size = line_size - line_buf, line_size = [w["text"]], sz + line_buf, line_size = [vl["text"]], sz else: - line_buf.append(w["text"]) + line_buf.append(vl["text"]) if line_buf: emit_images_before(page.height) - buf.append(" ".join(line_buf)) + buf.append("\n".join(line_buf)) # картинките след всичкия текст на страницата emit_images_before(page.height + 1) @@ -788,8 +819,11 @@ def merge_short_sections(sections: list[Section]) -> list[Section]: def clean_text(text: str) -> str: - text = re.sub(r"\s+", " ", text) - text = re.sub(r" {2,}", " ", text) + """Collapse spaces/tabs but keep newlines (lists, paragraphs).""" + text = text.replace("\r\n", "\n").replace("\r", "\n") + text = re.sub(r"[^\S\n]+", " ", text) + text = re.sub(r"\n{3,}", "\n\n", text) + text = "\n".join(line.strip() for line in text.split("\n")) return text.strip()