.
This commit is contained in:
@@ -494,7 +494,13 @@ def file_hash(path: Path) -> str:
|
||||
|
||||
|
||||
def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
|
||||
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS."""
|
||||
"""Връща (data, ext) или None. Пропуска HTTP/HTTPS.
|
||||
|
||||
Относителните ``src`` се резолвират спрямо директорията на HTML файла.
|
||||
URL-encoding (``%20``) се декодира — типично за Word/LibreOffice „Save as HTML“.
|
||||
"""
|
||||
from urllib.parse import unquote
|
||||
|
||||
if not src:
|
||||
return None
|
||||
s = src.strip()
|
||||
@@ -511,14 +517,29 @@ def _load_html_image(src: str, base_dir: Path) -> Optional[tuple[bytes, str]]:
|
||||
return data, _ext_from_content_type(m.group(1))
|
||||
if s.startswith(("http://", "https://")):
|
||||
return None # по правило пропускаме мрежови картинки
|
||||
# локален път, относителен или абсолютен
|
||||
p = (base_dir / s).resolve() if not Path(s).is_absolute() else Path(s)
|
||||
if s.lower().startswith("file:"):
|
||||
path = _file_url_to_path(s)
|
||||
if path is None:
|
||||
log.warning(f" HTML image bad file URL: {src}")
|
||||
return None
|
||||
loaded = _read_image_file(path)
|
||||
if not loaded:
|
||||
log.warning(f" HTML image not found: {src} → {path}")
|
||||
return loaded
|
||||
|
||||
# локален път: %20 → space, ../ спрямо HTML папката
|
||||
decoded = unquote(s).replace("\\", "/")
|
||||
try:
|
||||
p = Path(decoded)
|
||||
if not p.is_absolute():
|
||||
p = (base_dir / decoded).resolve()
|
||||
if p.is_file():
|
||||
data = p.read_bytes()
|
||||
ext = p.suffix.lstrip(".").lower() or "png"
|
||||
ext = p.suffix.lstrip(".").lower() or "png"
|
||||
return data, ext
|
||||
except Exception:
|
||||
log.warning(f" HTML image not found: {src} → {p}")
|
||||
except Exception as e:
|
||||
log.warning(f" HTML image unreadable: {src}: {e}")
|
||||
return None
|
||||
return None
|
||||
|
||||
@@ -622,13 +643,18 @@ _FIGURE_CAPTION_RE = re.compile(
|
||||
)
|
||||
|
||||
|
||||
def _normalize_inline_ws(text: str) -> str:
|
||||
"""Срива CR/LF/табове в един интервал (Word/LibreOffice soft breaks в <h2>)."""
|
||||
return re.sub(r"\s+", " ", (text or "").strip())
|
||||
|
||||
|
||||
def _normalize_chapter_key(text: str) -> str:
|
||||
return re.sub(r"\s+", " ", (text or "").strip().lower())
|
||||
return _normalize_inline_ws(text).lower()
|
||||
|
||||
|
||||
def _is_numbered_chapter_heading(text: str) -> bool:
|
||||
"""True for top-level numbered chapter titles (not TOC prose, figures, tables)."""
|
||||
t = (text or "").strip()
|
||||
t = _normalize_inline_ws(text)
|
||||
if not t or len(t) >= 120:
|
||||
return False
|
||||
if "|" in t:
|
||||
@@ -831,24 +857,34 @@ def _strip_attrs(el):
|
||||
t["style"] = kept_style
|
||||
|
||||
|
||||
def _ingest_html_img(
|
||||
img, base_dir: Path, sec_images: list, img_counter: list
|
||||
) -> Optional[str]:
|
||||
"""Извлича една <img> картинка; връща placeholder текст или None."""
|
||||
from bs4 import NavigableString
|
||||
|
||||
src = img.get("src") or img.get("data-src") or ""
|
||||
loaded = _load_html_image(src, base_dir)
|
||||
if not loaded:
|
||||
img.decompose()
|
||||
return None
|
||||
data, ext = loaded
|
||||
if not _should_keep_image(data):
|
||||
img.decompose()
|
||||
return None
|
||||
img_counter[0] += 1
|
||||
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
|
||||
sec_images.append(ref)
|
||||
ph = f"[IMG: {ref.placeholder}]"
|
||||
img.replace_with(NavigableString(ph))
|
||||
return ph
|
||||
|
||||
|
||||
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
|
||||
"""Намира всички <img> в подадения елемент, извлича данните и подменя с
|
||||
NavigableString placeholder ([IMG: img_NN])."""
|
||||
from bs4 import NavigableString
|
||||
for img in el.find_all("img"):
|
||||
src = img.get("src") or img.get("data-src") or ""
|
||||
loaded = _load_html_image(src, base_dir)
|
||||
if not loaded:
|
||||
img.decompose()
|
||||
continue
|
||||
data, ext = loaded
|
||||
if not _should_keep_image(data):
|
||||
img.decompose()
|
||||
continue
|
||||
img_counter[0] += 1
|
||||
ref = ImageRef(placeholder=f"img_{img_counter[0]:02d}", data=data, ext=ext)
|
||||
sec_images.append(ref)
|
||||
img.replace_with(NavigableString(f"[IMG: {ref.placeholder}]"))
|
||||
_ingest_html_img(img, base_dir, sec_images, img_counter)
|
||||
|
||||
|
||||
def parse_html(path: Path) -> list[Section]:
|
||||
@@ -909,14 +945,16 @@ def parse_html(path: Path) -> list[Section]:
|
||||
def start_section(title: str, level: int = 1):
|
||||
nonlocal current_title, current_level, sec_text, sec_html, sec_images
|
||||
flush()
|
||||
current_title = title
|
||||
current_title = _normalize_inline_ws(title)
|
||||
current_level = level
|
||||
sec_text, sec_html, sec_images = [], [], []
|
||||
|
||||
for el in blocks:
|
||||
txt = el.get_text(" ", strip=True)
|
||||
# Soft breaks (\r\n в средата на Word/LO <h2>) → един ред за split/TOC
|
||||
txt = _normalize_inline_ws(el.get_text(" ", strip=True))
|
||||
is_cover = _html_is_cover_title(el)
|
||||
heading_lvl = _html_heading_level(el)
|
||||
el_name = (el.name or "").lower()
|
||||
|
||||
# Корица (Word Title / class Title): заглавие на преамбюла, без нова секция
|
||||
if is_cover and txt:
|
||||
@@ -934,6 +972,12 @@ def parse_html(path: Path) -> list[Section]:
|
||||
append_block_as_body(el)
|
||||
continue
|
||||
|
||||
# <ul>/<ol>/table в TOC: целият блок НЕ е една „1. …“ глава — край на TOC фазата
|
||||
if toc_state.phase and el_name in _HTML_PLAIN_NL_TAGS:
|
||||
toc_state.leave_phase()
|
||||
append_block_as_body(el)
|
||||
continue
|
||||
|
||||
# Номерирана глава (H1/H2/bold/plain) — с TOC: първото срещане в съдържанието остава в преамбюла
|
||||
numbered_action = toc_state.handle_numbered(txt) if txt else "no"
|
||||
if numbered_action == "toc":
|
||||
@@ -957,23 +1001,16 @@ def parse_html(path: Path) -> list[Section]:
|
||||
start_section(txt, heading_lvl)
|
||||
continue
|
||||
|
||||
if el.name == "img":
|
||||
if el_name == "img":
|
||||
if toc_state.phase:
|
||||
toc_state.leave_phase()
|
||||
# самостоятелен <img> (не вътре в блок)
|
||||
_swap_imgs_in_block(el.parent if el.parent and el.parent.name else el,
|
||||
base_dir, sec_images, img_counter)
|
||||
# ако е заменен с placeholder, добавяме като текст
|
||||
img_txt = el.get_text(" ", strip=True) if el.name else ""
|
||||
if img_txt:
|
||||
sec_text.append(img_txt)
|
||||
sec_html.append(f"<p>{img_txt}</p>")
|
||||
ph = _ingest_html_img(el, base_dir, sec_images, img_counter)
|
||||
if ph:
|
||||
sec_text.append(ph)
|
||||
sec_html.append(f"<p>{ph}</p>")
|
||||
continue
|
||||
|
||||
if toc_state.phase and (el.name or "").lower() in _HTML_PLAIN_NL_TAGS:
|
||||
# <ol>/<ul> TOC списък — остава в преамбюла, приключва TOC фазата
|
||||
toc_state.leave_phase()
|
||||
elif toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
|
||||
if toc_state.phase and txt and not _is_numbered_chapter_heading(txt):
|
||||
toc_state.leave_phase()
|
||||
append_block_as_body(el)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user