.
This commit is contained in:
@@ -1001,27 +1001,107 @@ def _file_url_to_path(url: str) -> Optional[Path]:
|
||||
return Path(s)
|
||||
|
||||
|
||||
def _load_external_image_bytes(url: str) -> Optional[tuple[bytes, str]]:
|
||||
"""Чете външна картинка (r:link / TargetMode=External); връща (data, ext)."""
|
||||
path = _file_url_to_path(url)
|
||||
if path is None:
|
||||
return None
|
||||
def _image_ext_from_path(path: Path) -> str:
|
||||
ext = (path.suffix or "").lstrip(".").lower() or "png"
|
||||
return "jpg" if ext == "jpeg" else ext
|
||||
|
||||
|
||||
def _read_image_file(path: Path) -> Optional[tuple[bytes, str]]:
|
||||
"""Чете локален/UNC файл като (data, ext); None при липса/грешка."""
|
||||
try:
|
||||
if not path.is_file():
|
||||
return None
|
||||
data = path.read_bytes()
|
||||
except OSError:
|
||||
except OSError as e:
|
||||
log.warning(f" Cannot read image file {path}: {e}")
|
||||
return None
|
||||
if not data:
|
||||
return None
|
||||
ext = (path.suffix or "").lstrip(".").lower() or "png"
|
||||
if ext == "jpeg":
|
||||
ext = "jpg"
|
||||
return data, ext
|
||||
return data, _image_ext_from_path(path)
|
||||
|
||||
|
||||
def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
|
||||
"""r:embed → related_parts; r:link → външен файл. Връща (data, ext)."""
|
||||
def _docx_linked_media_dir(docx_path: Path) -> Path:
|
||||
"""Папка до .docx за опаковани r:link картинки: <stem>_media/."""
|
||||
return docx_path.parent / f"{docx_path.stem}_media"
|
||||
|
||||
|
||||
def _sidecar_image_candidates(
|
||||
docx_path: Optional[Path], linked_path: Optional[Path], url: str = ""
|
||||
) -> list[Path]:
|
||||
"""Кандидати до .docx, ако UNC/file линкът е недостъпен (Coolify, офлайн)."""
|
||||
if docx_path is None:
|
||||
return []
|
||||
name = ""
|
||||
if linked_path is not None:
|
||||
name = linked_path.name
|
||||
if not name:
|
||||
from urllib.parse import unquote
|
||||
|
||||
tail = unquote((url or "").replace("\\", "/").rstrip("/").split("/")[-1])
|
||||
name = tail.split("?")[0] if tail else ""
|
||||
if not name:
|
||||
return []
|
||||
parent = docx_path.parent
|
||||
stem = docx_path.stem
|
||||
return [
|
||||
parent / name,
|
||||
_docx_linked_media_dir(docx_path) / name,
|
||||
parent / "media" / name,
|
||||
parent / "images" / name,
|
||||
]
|
||||
|
||||
|
||||
def _pack_linked_image(docx_path: Path, filename: str, data: bytes) -> Optional[Path]:
|
||||
"""Копира успешно прочетена r:link картинка до <stem>_media/ за следващи сканирания."""
|
||||
if not filename or not data:
|
||||
return None
|
||||
dest_dir = _docx_linked_media_dir(docx_path)
|
||||
dest = dest_dir / filename
|
||||
try:
|
||||
if dest.is_file() and dest.stat().st_size == len(data):
|
||||
return dest
|
||||
dest_dir.mkdir(parents=True, exist_ok=True)
|
||||
dest.write_bytes(data)
|
||||
log.info(f" Packed linked image → {dest}")
|
||||
return dest
|
||||
except OSError as e:
|
||||
log.warning(f" Cannot pack linked image {filename}: {e}")
|
||||
return None
|
||||
|
||||
|
||||
def _load_external_image_bytes(
|
||||
url: str, docx_path: Optional[Path] = None
|
||||
) -> Optional[tuple[bytes, str]]:
|
||||
"""Чете външна картинка (r:link); UNC/file, после sidecar до .docx; лог при провал."""
|
||||
path = _file_url_to_path(url)
|
||||
tried: list[Path] = []
|
||||
|
||||
if path is not None:
|
||||
tried.append(path)
|
||||
loaded = _read_image_file(path)
|
||||
if loaded:
|
||||
if docx_path is not None:
|
||||
_pack_linked_image(docx_path, path.name, loaded[0])
|
||||
return loaded
|
||||
|
||||
for cand in _sidecar_image_candidates(docx_path, path, url):
|
||||
if any(cand == t for t in tried):
|
||||
continue
|
||||
tried.append(cand)
|
||||
loaded = _read_image_file(cand)
|
||||
if loaded:
|
||||
log.info(f" External image from sidecar {cand}")
|
||||
return loaded
|
||||
|
||||
targets = ", ".join(str(p) for p in tried) if tried else (url or "(empty)")
|
||||
log.warning(f" Cannot read linked image (r:link): {targets}")
|
||||
return None
|
||||
|
||||
|
||||
def _resolve_docx_image_rid(
|
||||
doc, rId: str, docx_path: Optional[Path] = None
|
||||
) -> Optional[tuple[bytes, str]]:
|
||||
"""r:embed → related_parts; r:link → външен файл / sidecar. Връща (data, ext)."""
|
||||
if not rId:
|
||||
return None
|
||||
try:
|
||||
@@ -1039,14 +1119,15 @@ def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
|
||||
if not getattr(rel, "is_external", False):
|
||||
return None
|
||||
target = getattr(rel, "target_ref", None) or ""
|
||||
loaded = _load_external_image_bytes(target)
|
||||
return loaded
|
||||
return _load_external_image_bytes(target, docx_path=docx_path)
|
||||
|
||||
|
||||
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||
def _extract_docx_paragraph_images(
|
||||
para, doc, docx_path: Optional[Path] = None
|
||||
) -> list[ImageRef]:
|
||||
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
|
||||
|
||||
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC пътища).
|
||||
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC + sidecar до .docx).
|
||||
"""
|
||||
from docx.oxml.ns import qn
|
||||
imgs: list[ImageRef] = []
|
||||
@@ -1058,10 +1139,16 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||
embed_attr = qn("r:embed")
|
||||
link_attr = qn("r:link")
|
||||
for blip in blips:
|
||||
# Предпочитаме вградено копие, ако Word е запазил и двете
|
||||
rId = blip.get(embed_attr) or blip.get(link_attr)
|
||||
if not rId:
|
||||
continue
|
||||
resolved = _resolve_docx_image_rid(doc, rId)
|
||||
resolved = _resolve_docx_image_rid(doc, rId, docx_path=docx_path)
|
||||
if not resolved:
|
||||
# Ако има само link и се провали — опитай обратното атрибутче
|
||||
other = blip.get(link_attr) if blip.get(embed_attr) else blip.get(embed_attr)
|
||||
if other and other != rId:
|
||||
resolved = _resolve_docx_image_rid(doc, other, docx_path=docx_path)
|
||||
if not resolved:
|
||||
continue
|
||||
data, ext = resolved
|
||||
@@ -1225,7 +1312,8 @@ def _table_lines(table) -> list[str]:
|
||||
|
||||
|
||||
def parse_docx(path: Path) -> list[Section]:
|
||||
doc = Document(path)
|
||||
docx_path = Path(path)
|
||||
doc = Document(docx_path)
|
||||
sections: list[Section] = []
|
||||
current_title, current_level = "", 1
|
||||
buf: list[str] = []
|
||||
@@ -1279,7 +1367,7 @@ def parse_docx(path: Path) -> list[Section]:
|
||||
para = block
|
||||
style_name = para.style.name.lower() if para.style else ""
|
||||
text = para.text.strip()
|
||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
||||
para_imgs = _extract_docx_paragraph_images(para, doc, docx_path=docx_path)
|
||||
|
||||
if not text and not para_imgs:
|
||||
continue
|
||||
|
||||
Reference in New Issue
Block a user