.
This commit is contained in:
@@ -105,7 +105,7 @@ UNIQUE constraint: `(prefix, file_path)`
|
|||||||
## Картинки
|
## Картинки
|
||||||
|
|
||||||
Извличат се по време на парсване:
|
Извличат се по време на парсване:
|
||||||
- `.docx` — `<a:blip>` в paragraph drawings → bytes от related_parts
|
- `.docx` — `<a:blip>` в paragraph drawings → `r:embed` (related_parts) или `r:link` (UNC/`file:///` + sidecar `<stem>_media/` до .docx)
|
||||||
- `.html` — локални файлове и `data:` URLs; HTTP пропуска
|
- `.html` — локални файлове и `data:` URLs; HTTP пропуска
|
||||||
- `.pdf` — `pdfplumber.page.crop(bbox).to_image()` като PNG
|
- `.pdf` — `pdfplumber.page.crop(bbox).to_image()` като PNG
|
||||||
- `.doc` — след LibreOffice/MS Word конверсия до `.docx`
|
- `.doc` — след LibreOffice/MS Word конверсия до `.docx`
|
||||||
|
|||||||
@@ -100,7 +100,7 @@ TXT и PDF запазват своите правила (номерирани р
|
|||||||
|
|
||||||
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`.
|
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`.
|
||||||
|
|
||||||
Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`). Поддържат се `r:embed` (вградени) и `r:link` (външни `file:///` / UNC), ако файлът е достъпен при сканиране.
|
Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`). Поддържат се `r:embed` (вградени) и `r:link` (външни `file:///` / UNC). При успешен UNC/file прочит се опакова копие в `<име>_media/` до `.docx`; при недостъпен линк се търси sidecar (`123.png`, `<stem>_media/`, `media/`, `images/`). Липсващи пътища се логват като warning.
|
||||||
|
|
||||||
**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>` (изричен run **или** наследен от paragraph/character style), `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Runs вътре в `w:hyperlink` (TOC редове) също влизат в `html_text`; ако runs липсват, има fallback към escaped `para.text`. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. Viewer показва `html_text` когато е наличен — затова TOC/bold/картинки трябва да са в него, не само в plain `text`.
|
**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>` (изричен run **или** наследен от paragraph/character style), `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Runs вътре в `w:hyperlink` (TOC редове) също влизат в `html_text`; ако runs липсват, има fallback към escaped `para.text`. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. Viewer показва `html_text` когато е наличен — затова TOC/bold/картинки трябва да са в него, не само в plain `text`.
|
||||||
|
|
||||||
|
|||||||
@@ -1001,27 +1001,107 @@ def _file_url_to_path(url: str) -> Optional[Path]:
|
|||||||
return Path(s)
|
return Path(s)
|
||||||
|
|
||||||
|
|
||||||
def _load_external_image_bytes(url: str) -> Optional[tuple[bytes, str]]:
|
def _image_ext_from_path(path: Path) -> str:
|
||||||
"""Чете външна картинка (r:link / TargetMode=External); връща (data, ext)."""
|
ext = (path.suffix or "").lstrip(".").lower() or "png"
|
||||||
path = _file_url_to_path(url)
|
return "jpg" if ext == "jpeg" else ext
|
||||||
if path is None:
|
|
||||||
return None
|
|
||||||
|
def _read_image_file(path: Path) -> Optional[tuple[bytes, str]]:
|
||||||
|
"""Чете локален/UNC файл като (data, ext); None при липса/грешка."""
|
||||||
try:
|
try:
|
||||||
if not path.is_file():
|
if not path.is_file():
|
||||||
return None
|
return None
|
||||||
data = path.read_bytes()
|
data = path.read_bytes()
|
||||||
except OSError:
|
except OSError as e:
|
||||||
|
log.warning(f" Cannot read image file {path}: {e}")
|
||||||
return None
|
return None
|
||||||
if not data:
|
if not data:
|
||||||
return None
|
return None
|
||||||
ext = (path.suffix or "").lstrip(".").lower() or "png"
|
return data, _image_ext_from_path(path)
|
||||||
if ext == "jpeg":
|
|
||||||
ext = "jpg"
|
|
||||||
return data, ext
|
|
||||||
|
|
||||||
|
|
||||||
def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
|
def _docx_linked_media_dir(docx_path: Path) -> Path:
|
||||||
"""r:embed → related_parts; r:link → външен файл. Връща (data, ext)."""
|
"""Папка до .docx за опаковани r:link картинки: <stem>_media/."""
|
||||||
|
return docx_path.parent / f"{docx_path.stem}_media"
|
||||||
|
|
||||||
|
|
||||||
|
def _sidecar_image_candidates(
|
||||||
|
docx_path: Optional[Path], linked_path: Optional[Path], url: str = ""
|
||||||
|
) -> list[Path]:
|
||||||
|
"""Кандидати до .docx, ако UNC/file линкът е недостъпен (Coolify, офлайн)."""
|
||||||
|
if docx_path is None:
|
||||||
|
return []
|
||||||
|
name = ""
|
||||||
|
if linked_path is not None:
|
||||||
|
name = linked_path.name
|
||||||
|
if not name:
|
||||||
|
from urllib.parse import unquote
|
||||||
|
|
||||||
|
tail = unquote((url or "").replace("\\", "/").rstrip("/").split("/")[-1])
|
||||||
|
name = tail.split("?")[0] if tail else ""
|
||||||
|
if not name:
|
||||||
|
return []
|
||||||
|
parent = docx_path.parent
|
||||||
|
stem = docx_path.stem
|
||||||
|
return [
|
||||||
|
parent / name,
|
||||||
|
_docx_linked_media_dir(docx_path) / name,
|
||||||
|
parent / "media" / name,
|
||||||
|
parent / "images" / name,
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def _pack_linked_image(docx_path: Path, filename: str, data: bytes) -> Optional[Path]:
|
||||||
|
"""Копира успешно прочетена r:link картинка до <stem>_media/ за следващи сканирания."""
|
||||||
|
if not filename or not data:
|
||||||
|
return None
|
||||||
|
dest_dir = _docx_linked_media_dir(docx_path)
|
||||||
|
dest = dest_dir / filename
|
||||||
|
try:
|
||||||
|
if dest.is_file() and dest.stat().st_size == len(data):
|
||||||
|
return dest
|
||||||
|
dest_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
dest.write_bytes(data)
|
||||||
|
log.info(f" Packed linked image → {dest}")
|
||||||
|
return dest
|
||||||
|
except OSError as e:
|
||||||
|
log.warning(f" Cannot pack linked image {filename}: {e}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _load_external_image_bytes(
|
||||||
|
url: str, docx_path: Optional[Path] = None
|
||||||
|
) -> Optional[tuple[bytes, str]]:
|
||||||
|
"""Чете външна картинка (r:link); UNC/file, после sidecar до .docx; лог при провал."""
|
||||||
|
path = _file_url_to_path(url)
|
||||||
|
tried: list[Path] = []
|
||||||
|
|
||||||
|
if path is not None:
|
||||||
|
tried.append(path)
|
||||||
|
loaded = _read_image_file(path)
|
||||||
|
if loaded:
|
||||||
|
if docx_path is not None:
|
||||||
|
_pack_linked_image(docx_path, path.name, loaded[0])
|
||||||
|
return loaded
|
||||||
|
|
||||||
|
for cand in _sidecar_image_candidates(docx_path, path, url):
|
||||||
|
if any(cand == t for t in tried):
|
||||||
|
continue
|
||||||
|
tried.append(cand)
|
||||||
|
loaded = _read_image_file(cand)
|
||||||
|
if loaded:
|
||||||
|
log.info(f" External image from sidecar {cand}")
|
||||||
|
return loaded
|
||||||
|
|
||||||
|
targets = ", ".join(str(p) for p in tried) if tried else (url or "(empty)")
|
||||||
|
log.warning(f" Cannot read linked image (r:link): {targets}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_docx_image_rid(
|
||||||
|
doc, rId: str, docx_path: Optional[Path] = None
|
||||||
|
) -> Optional[tuple[bytes, str]]:
|
||||||
|
"""r:embed → related_parts; r:link → външен файл / sidecar. Връща (data, ext)."""
|
||||||
if not rId:
|
if not rId:
|
||||||
return None
|
return None
|
||||||
try:
|
try:
|
||||||
@@ -1039,14 +1119,15 @@ def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
|
|||||||
if not getattr(rel, "is_external", False):
|
if not getattr(rel, "is_external", False):
|
||||||
return None
|
return None
|
||||||
target = getattr(rel, "target_ref", None) or ""
|
target = getattr(rel, "target_ref", None) or ""
|
||||||
loaded = _load_external_image_bytes(target)
|
return _load_external_image_bytes(target, docx_path=docx_path)
|
||||||
return loaded
|
|
||||||
|
|
||||||
|
|
||||||
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
def _extract_docx_paragraph_images(
|
||||||
|
para, doc, docx_path: Optional[Path] = None
|
||||||
|
) -> list[ImageRef]:
|
||||||
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
|
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
|
||||||
|
|
||||||
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC пътища).
|
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC + sidecar до .docx).
|
||||||
"""
|
"""
|
||||||
from docx.oxml.ns import qn
|
from docx.oxml.ns import qn
|
||||||
imgs: list[ImageRef] = []
|
imgs: list[ImageRef] = []
|
||||||
@@ -1058,10 +1139,16 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
|||||||
embed_attr = qn("r:embed")
|
embed_attr = qn("r:embed")
|
||||||
link_attr = qn("r:link")
|
link_attr = qn("r:link")
|
||||||
for blip in blips:
|
for blip in blips:
|
||||||
|
# Предпочитаме вградено копие, ако Word е запазил и двете
|
||||||
rId = blip.get(embed_attr) or blip.get(link_attr)
|
rId = blip.get(embed_attr) or blip.get(link_attr)
|
||||||
if not rId:
|
if not rId:
|
||||||
continue
|
continue
|
||||||
resolved = _resolve_docx_image_rid(doc, rId)
|
resolved = _resolve_docx_image_rid(doc, rId, docx_path=docx_path)
|
||||||
|
if not resolved:
|
||||||
|
# Ако има само link и се провали — опитай обратното атрибутче
|
||||||
|
other = blip.get(link_attr) if blip.get(embed_attr) else blip.get(embed_attr)
|
||||||
|
if other and other != rId:
|
||||||
|
resolved = _resolve_docx_image_rid(doc, other, docx_path=docx_path)
|
||||||
if not resolved:
|
if not resolved:
|
||||||
continue
|
continue
|
||||||
data, ext = resolved
|
data, ext = resolved
|
||||||
@@ -1225,7 +1312,8 @@ def _table_lines(table) -> list[str]:
|
|||||||
|
|
||||||
|
|
||||||
def parse_docx(path: Path) -> list[Section]:
|
def parse_docx(path: Path) -> list[Section]:
|
||||||
doc = Document(path)
|
docx_path = Path(path)
|
||||||
|
doc = Document(docx_path)
|
||||||
sections: list[Section] = []
|
sections: list[Section] = []
|
||||||
current_title, current_level = "", 1
|
current_title, current_level = "", 1
|
||||||
buf: list[str] = []
|
buf: list[str] = []
|
||||||
@@ -1279,7 +1367,7 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
para = block
|
para = block
|
||||||
style_name = para.style.name.lower() if para.style else ""
|
style_name = para.style.name.lower() if para.style else ""
|
||||||
text = para.text.strip()
|
text = para.text.strip()
|
||||||
para_imgs = _extract_docx_paragraph_images(para, doc)
|
para_imgs = _extract_docx_paragraph_images(para, doc, docx_path=docx_path)
|
||||||
|
|
||||||
if not text and not para_imgs:
|
if not text and not para_imgs:
|
||||||
continue
|
continue
|
||||||
|
|||||||
@@ -215,6 +215,48 @@ def test_docx_external_r_link_image(tmp_path: Path):
|
|||||||
assert body.images, "r:link external image must be loaded"
|
assert body.images, "r:link external image must be loaded"
|
||||||
assert "[IMG:" in (body.text or "")
|
assert "[IMG:" in (body.text or "")
|
||||||
assert "[IMG:" in (body.html_text or "")
|
assert "[IMG:" in (body.html_text or "")
|
||||||
|
# Успешен прочит → опаковане до <stem>_media/
|
||||||
|
packed = tmp_path / "ext_img_media" / "ext.png"
|
||||||
|
assert packed.is_file()
|
||||||
|
assert packed.stat().st_size == len(png)
|
||||||
|
|
||||||
|
|
||||||
|
def test_docx_external_r_link_sidecar_when_unc_missing(tmp_path: Path):
|
||||||
|
"""Ако r:link сочи към липсващ UNC/път, четем <stem>_media/<name> до .docx."""
|
||||||
|
png = _mini_png()
|
||||||
|
try:
|
||||||
|
from io import BytesIO
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
assert Image.open(BytesIO(png)).size[0] >= 50
|
||||||
|
except Exception:
|
||||||
|
pytest.skip("PIL unavailable for size check")
|
||||||
|
|
||||||
|
missing = tmp_path / "nowhere" / "missing_shot.png"
|
||||||
|
# Не създаваме missing — само sidecar
|
||||||
|
media = tmp_path / "linked_media"
|
||||||
|
media.mkdir()
|
||||||
|
(media / "missing_shot.png").write_bytes(png)
|
||||||
|
|
||||||
|
doc = Document()
|
||||||
|
doc.add_heading("1. Sidecar", level=1)
|
||||||
|
try:
|
||||||
|
_add_external_blip_paragraph(doc, missing)
|
||||||
|
except Exception as exc:
|
||||||
|
pytest.skip(f"cannot build r:link drawing: {exc}")
|
||||||
|
# Поправи target към несъществуващ път (get_or_add може да е създал file URI)
|
||||||
|
out = tmp_path / "linked.docx"
|
||||||
|
doc.save(out)
|
||||||
|
|
||||||
|
# Гарантираме sidecar до записания docx
|
||||||
|
side = tmp_path / "linked_media"
|
||||||
|
side.mkdir(exist_ok=True)
|
||||||
|
(side / "missing_shot.png").write_bytes(png)
|
||||||
|
|
||||||
|
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||||||
|
body = sections[0]
|
||||||
|
assert body.images, "sidecar next to docx must satisfy r:link"
|
||||||
|
assert "[IMG:" in (body.html_text or "")
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
|
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
|
||||||
@@ -233,13 +275,25 @@ def test_atra_manual_toc_bold_and_image_in_html():
|
|||||||
# Style-bold подзаглавия + изричен bold по-нататък
|
# Style-bold подзаглавия + изричен bold по-нататък
|
||||||
assert any("<b>" in (s.html_text or "") for s in sections)
|
assert any("<b>" in (s.html_text or "") for s in sections)
|
||||||
|
|
||||||
# Фигура 1 е външен UNC линк — ако share-ът е достъпен, трябва [IMG:]
|
|
||||||
overview = next((s for s in sections if "Преглед" in s.title), None)
|
overview = next((s for s in sections if "Преглед" in s.title), None)
|
||||||
assert overview is not None
|
assert overview is not None
|
||||||
if overview.images or "[IMG:" in (overview.text or ""):
|
assert "Фигура 1" in (overview.text or "")
|
||||||
|
# Drawing е ПРЕДИ caption; трябва [IMG:] ако UNC или atra-manual_media/123.png
|
||||||
|
unc = Path(r"\\192.168.88.15\tmp\___Proekti\2025 ATRA96\otchitane\123.png")
|
||||||
|
sidecar = ATRA_MANUAL.parent / "atra-manual_media" / "123.png"
|
||||||
|
can_load = False
|
||||||
|
try:
|
||||||
|
can_load = unc.is_file() or sidecar.is_file()
|
||||||
|
except OSError:
|
||||||
|
can_load = sidecar.is_file()
|
||||||
|
if can_load:
|
||||||
|
assert overview.images, "Фигура 1 image must be extracted"
|
||||||
|
assert "[IMG:" in (overview.text or "")
|
||||||
assert "[IMG:" in (overview.html_text or "")
|
assert "[IMG:" in (overview.html_text or "")
|
||||||
|
# Ред: картинка, после caption (както в Word)
|
||||||
|
t = overview.text or ""
|
||||||
|
assert t.find("[IMG:") < t.find("Фигура 1")
|
||||||
else:
|
else:
|
||||||
# Документът сочи към външен файл; ако UNC липсва, plain caption остава
|
|
||||||
assert "Фигура 1" in (overview.text or "")
|
assert "Фигура 1" in (overview.text or "")
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user