.
This commit is contained in:
@@ -985,8 +985,69 @@ def parse_html(path: Path) -> list[Section]:
|
||||
return sections
|
||||
|
||||
|
||||
def _file_url_to_path(url: str) -> Optional[Path]:
|
||||
"""file:///… или обикновен път → Path (вкл. UNC от Word r:link)."""
|
||||
from urllib.parse import unquote
|
||||
|
||||
s = unquote((url or "").strip())
|
||||
if not s:
|
||||
return None
|
||||
if s.lower().startswith("file:"):
|
||||
s = s[5:]
|
||||
while s.startswith("/"):
|
||||
s = s[1:]
|
||||
if not s:
|
||||
return None
|
||||
return Path(s)
|
||||
|
||||
|
||||
def _load_external_image_bytes(url: str) -> Optional[tuple[bytes, str]]:
|
||||
"""Чете външна картинка (r:link / TargetMode=External); връща (data, ext)."""
|
||||
path = _file_url_to_path(url)
|
||||
if path is None:
|
||||
return None
|
||||
try:
|
||||
if not path.is_file():
|
||||
return None
|
||||
data = path.read_bytes()
|
||||
except OSError:
|
||||
return None
|
||||
if not data:
|
||||
return None
|
||||
ext = (path.suffix or "").lstrip(".").lower() or "png"
|
||||
if ext == "jpeg":
|
||||
ext = "jpg"
|
||||
return data, ext
|
||||
|
||||
|
||||
def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
|
||||
"""r:embed → related_parts; r:link → външен файл. Връща (data, ext)."""
|
||||
if not rId:
|
||||
return None
|
||||
try:
|
||||
part = doc.part.related_parts[rId]
|
||||
data = part.blob
|
||||
ct = getattr(part, "content_type", "") or ""
|
||||
if data:
|
||||
return data, _ext_from_content_type(ct)
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
rel = doc.part.rels[rId]
|
||||
except Exception:
|
||||
return None
|
||||
if not getattr(rel, "is_external", False):
|
||||
return None
|
||||
target = getattr(rel, "target_ref", None) or ""
|
||||
loaded = _load_external_image_bytes(target)
|
||||
return loaded
|
||||
|
||||
|
||||
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер."""
|
||||
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
|
||||
|
||||
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC пътища).
|
||||
"""
|
||||
from docx.oxml.ns import qn
|
||||
imgs: list[ImageRef] = []
|
||||
try:
|
||||
@@ -995,19 +1056,17 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||
return imgs
|
||||
|
||||
embed_attr = qn("r:embed")
|
||||
link_attr = qn("r:link")
|
||||
for blip in blips:
|
||||
rId = blip.get(embed_attr)
|
||||
rId = blip.get(embed_attr) or blip.get(link_attr)
|
||||
if not rId:
|
||||
continue
|
||||
try:
|
||||
part = doc.part.related_parts[rId]
|
||||
data = part.blob
|
||||
ct = getattr(part, "content_type", "") or ""
|
||||
except Exception:
|
||||
resolved = _resolve_docx_image_rid(doc, rId)
|
||||
if not resolved:
|
||||
continue
|
||||
data, ext = resolved
|
||||
if not _should_keep_image(data):
|
||||
continue
|
||||
ext = _ext_from_content_type(ct)
|
||||
imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext))
|
||||
return imgs
|
||||
|
||||
@@ -1023,17 +1082,72 @@ def _docx_run_color_css(run) -> Optional[str]:
|
||||
return None
|
||||
|
||||
|
||||
def _docx_run_to_html(run) -> str:
|
||||
def _docx_style_font_flag(style, attr: str) -> Optional[bool]:
|
||||
"""True/False от style.font.<attr>; None ако стилът/атрибутът липсва."""
|
||||
if style is None:
|
||||
return None
|
||||
try:
|
||||
font = style.font
|
||||
if font is None:
|
||||
return None
|
||||
val = getattr(font, attr, None)
|
||||
if val is None:
|
||||
return None
|
||||
return bool(val)
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _docx_run_effective_flag(run, para, attr: str) -> bool:
|
||||
"""Ефективен bold/italic/underline: изричен run → char style → paragraph style."""
|
||||
try:
|
||||
direct = getattr(run, attr, None)
|
||||
except Exception:
|
||||
direct = None
|
||||
if direct is True:
|
||||
return True
|
||||
if direct is False:
|
||||
return False
|
||||
# None = наследяване
|
||||
try:
|
||||
char_style = run.style
|
||||
except Exception:
|
||||
char_style = None
|
||||
flag = _docx_style_font_flag(char_style, attr)
|
||||
if flag is not None:
|
||||
return flag
|
||||
try:
|
||||
p_style = para.style if para is not None else None
|
||||
except Exception:
|
||||
p_style = None
|
||||
flag = _docx_style_font_flag(p_style, attr)
|
||||
return bool(flag)
|
||||
|
||||
|
||||
def _iter_docx_para_runs(para):
|
||||
"""Runs в параграф, вкл. вътре в w:hyperlink (python-docx.para.runs ги пропуска)."""
|
||||
from docx.oxml.ns import qn
|
||||
from docx.text.run import Run
|
||||
|
||||
for child in para._element.iterchildren():
|
||||
if child.tag == qn("w:r"):
|
||||
yield Run(child, para)
|
||||
elif child.tag == qn("w:hyperlink"):
|
||||
for r_elem in child.findall(qn("w:r")):
|
||||
yield Run(r_elem, para)
|
||||
|
||||
|
||||
def _docx_run_to_html(run, para=None) -> str:
|
||||
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
|
||||
text = run.text or ""
|
||||
if not text:
|
||||
return ""
|
||||
s = html_lib.escape(text, quote=False).replace("\n", "<br>")
|
||||
if run.bold:
|
||||
if _docx_run_effective_flag(run, para, "bold"):
|
||||
s = f"<b>{s}</b>"
|
||||
if run.italic:
|
||||
if _docx_run_effective_flag(run, para, "italic"):
|
||||
s = f"<i>{s}</i>"
|
||||
if run.underline:
|
||||
if _docx_run_effective_flag(run, para, "underline"):
|
||||
s = f"<u>{s}</u>"
|
||||
css_color = _docx_run_color_css(run)
|
||||
if css_color:
|
||||
@@ -1042,11 +1156,15 @@ def _docx_run_to_html(run) -> str:
|
||||
|
||||
|
||||
def _docx_para_to_html(para) -> str:
|
||||
"""Параграф → <p>…</p> с inline форматиране от runs."""
|
||||
parts = [_docx_run_to_html(r) for r in para.runs]
|
||||
"""Параграф → <p>…</p> с inline форматиране от runs (вкл. hyperlink TOC)."""
|
||||
parts = [_docx_run_to_html(r, para) for r in _iter_docx_para_runs(para)]
|
||||
inner = "".join(parts)
|
||||
if not inner.strip():
|
||||
return ""
|
||||
# Fallback: para.text вижда hyperlink текст, но без runs в .runs
|
||||
plain = (para.text or "").strip()
|
||||
if not plain:
|
||||
return ""
|
||||
inner = html_lib.escape(plain, quote=False).replace("\n", "<br>")
|
||||
return f"<p>{inner}</p>"
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user