.
This commit is contained in:
@@ -100,9 +100,9 @@ TXT и PDF запазват своите правила (номерирани р
|
|||||||
|
|
||||||
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`.
|
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`.
|
||||||
|
|
||||||
Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`).
|
Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`). Поддържат се `r:embed` (вградени) и `r:link` (външни `file:///` / UNC), ако файлът е достъпен при сканиране.
|
||||||
|
|
||||||
**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>`, `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват.
|
**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>` (изричен run **или** наследен от paragraph/character style), `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Runs вътре в `w:hyperlink` (TOC редове) също влизат в `html_text`; ако runs липсват, има fallback към escaped `para.text`. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. Viewer показва `html_text` когато е наличен — затова TOC/bold/картинки трябва да са в него, не само в plain `text`.
|
||||||
|
|
||||||
Празен параграф без картинка се пропуска.
|
Празен параграф без картинка се пропуска.
|
||||||
|
|
||||||
|
|||||||
@@ -985,8 +985,69 @@ def parse_html(path: Path) -> list[Section]:
|
|||||||
return sections
|
return sections
|
||||||
|
|
||||||
|
|
||||||
|
def _file_url_to_path(url: str) -> Optional[Path]:
|
||||||
|
"""file:///… или обикновен път → Path (вкл. UNC от Word r:link)."""
|
||||||
|
from urllib.parse import unquote
|
||||||
|
|
||||||
|
s = unquote((url or "").strip())
|
||||||
|
if not s:
|
||||||
|
return None
|
||||||
|
if s.lower().startswith("file:"):
|
||||||
|
s = s[5:]
|
||||||
|
while s.startswith("/"):
|
||||||
|
s = s[1:]
|
||||||
|
if not s:
|
||||||
|
return None
|
||||||
|
return Path(s)
|
||||||
|
|
||||||
|
|
||||||
|
def _load_external_image_bytes(url: str) -> Optional[tuple[bytes, str]]:
|
||||||
|
"""Чете външна картинка (r:link / TargetMode=External); връща (data, ext)."""
|
||||||
|
path = _file_url_to_path(url)
|
||||||
|
if path is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
if not path.is_file():
|
||||||
|
return None
|
||||||
|
data = path.read_bytes()
|
||||||
|
except OSError:
|
||||||
|
return None
|
||||||
|
if not data:
|
||||||
|
return None
|
||||||
|
ext = (path.suffix or "").lstrip(".").lower() or "png"
|
||||||
|
if ext == "jpeg":
|
||||||
|
ext = "jpg"
|
||||||
|
return data, ext
|
||||||
|
|
||||||
|
|
||||||
|
def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
|
||||||
|
"""r:embed → related_parts; r:link → външен файл. Връща (data, ext)."""
|
||||||
|
if not rId:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
part = doc.part.related_parts[rId]
|
||||||
|
data = part.blob
|
||||||
|
ct = getattr(part, "content_type", "") or ""
|
||||||
|
if data:
|
||||||
|
return data, _ext_from_content_type(ct)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
try:
|
||||||
|
rel = doc.part.rels[rId]
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
if not getattr(rel, "is_external", False):
|
||||||
|
return None
|
||||||
|
target = getattr(rel, "target_ref", None) or ""
|
||||||
|
loaded = _load_external_image_bytes(target)
|
||||||
|
return loaded
|
||||||
|
|
||||||
|
|
||||||
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
||||||
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер."""
|
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
|
||||||
|
|
||||||
|
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC пътища).
|
||||||
|
"""
|
||||||
from docx.oxml.ns import qn
|
from docx.oxml.ns import qn
|
||||||
imgs: list[ImageRef] = []
|
imgs: list[ImageRef] = []
|
||||||
try:
|
try:
|
||||||
@@ -995,19 +1056,17 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
|||||||
return imgs
|
return imgs
|
||||||
|
|
||||||
embed_attr = qn("r:embed")
|
embed_attr = qn("r:embed")
|
||||||
|
link_attr = qn("r:link")
|
||||||
for blip in blips:
|
for blip in blips:
|
||||||
rId = blip.get(embed_attr)
|
rId = blip.get(embed_attr) or blip.get(link_attr)
|
||||||
if not rId:
|
if not rId:
|
||||||
continue
|
continue
|
||||||
try:
|
resolved = _resolve_docx_image_rid(doc, rId)
|
||||||
part = doc.part.related_parts[rId]
|
if not resolved:
|
||||||
data = part.blob
|
|
||||||
ct = getattr(part, "content_type", "") or ""
|
|
||||||
except Exception:
|
|
||||||
continue
|
continue
|
||||||
|
data, ext = resolved
|
||||||
if not _should_keep_image(data):
|
if not _should_keep_image(data):
|
||||||
continue
|
continue
|
||||||
ext = _ext_from_content_type(ct)
|
|
||||||
imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext))
|
imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext))
|
||||||
return imgs
|
return imgs
|
||||||
|
|
||||||
@@ -1023,17 +1082,72 @@ def _docx_run_color_css(run) -> Optional[str]:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def _docx_run_to_html(run) -> str:
|
def _docx_style_font_flag(style, attr: str) -> Optional[bool]:
|
||||||
|
"""True/False от style.font.<attr>; None ако стилът/атрибутът липсва."""
|
||||||
|
if style is None:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
font = style.font
|
||||||
|
if font is None:
|
||||||
|
return None
|
||||||
|
val = getattr(font, attr, None)
|
||||||
|
if val is None:
|
||||||
|
return None
|
||||||
|
return bool(val)
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _docx_run_effective_flag(run, para, attr: str) -> bool:
|
||||||
|
"""Ефективен bold/italic/underline: изричен run → char style → paragraph style."""
|
||||||
|
try:
|
||||||
|
direct = getattr(run, attr, None)
|
||||||
|
except Exception:
|
||||||
|
direct = None
|
||||||
|
if direct is True:
|
||||||
|
return True
|
||||||
|
if direct is False:
|
||||||
|
return False
|
||||||
|
# None = наследяване
|
||||||
|
try:
|
||||||
|
char_style = run.style
|
||||||
|
except Exception:
|
||||||
|
char_style = None
|
||||||
|
flag = _docx_style_font_flag(char_style, attr)
|
||||||
|
if flag is not None:
|
||||||
|
return flag
|
||||||
|
try:
|
||||||
|
p_style = para.style if para is not None else None
|
||||||
|
except Exception:
|
||||||
|
p_style = None
|
||||||
|
flag = _docx_style_font_flag(p_style, attr)
|
||||||
|
return bool(flag)
|
||||||
|
|
||||||
|
|
||||||
|
def _iter_docx_para_runs(para):
|
||||||
|
"""Runs в параграф, вкл. вътре в w:hyperlink (python-docx.para.runs ги пропуска)."""
|
||||||
|
from docx.oxml.ns import qn
|
||||||
|
from docx.text.run import Run
|
||||||
|
|
||||||
|
for child in para._element.iterchildren():
|
||||||
|
if child.tag == qn("w:r"):
|
||||||
|
yield Run(child, para)
|
||||||
|
elif child.tag == qn("w:hyperlink"):
|
||||||
|
for r_elem in child.findall(qn("w:r")):
|
||||||
|
yield Run(r_elem, para)
|
||||||
|
|
||||||
|
|
||||||
|
def _docx_run_to_html(run, para=None) -> str:
|
||||||
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
|
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
|
||||||
text = run.text or ""
|
text = run.text or ""
|
||||||
if not text:
|
if not text:
|
||||||
return ""
|
return ""
|
||||||
s = html_lib.escape(text, quote=False).replace("\n", "<br>")
|
s = html_lib.escape(text, quote=False).replace("\n", "<br>")
|
||||||
if run.bold:
|
if _docx_run_effective_flag(run, para, "bold"):
|
||||||
s = f"<b>{s}</b>"
|
s = f"<b>{s}</b>"
|
||||||
if run.italic:
|
if _docx_run_effective_flag(run, para, "italic"):
|
||||||
s = f"<i>{s}</i>"
|
s = f"<i>{s}</i>"
|
||||||
if run.underline:
|
if _docx_run_effective_flag(run, para, "underline"):
|
||||||
s = f"<u>{s}</u>"
|
s = f"<u>{s}</u>"
|
||||||
css_color = _docx_run_color_css(run)
|
css_color = _docx_run_color_css(run)
|
||||||
if css_color:
|
if css_color:
|
||||||
@@ -1042,11 +1156,15 @@ def _docx_run_to_html(run) -> str:
|
|||||||
|
|
||||||
|
|
||||||
def _docx_para_to_html(para) -> str:
|
def _docx_para_to_html(para) -> str:
|
||||||
"""Параграф → <p>…</p> с inline форматиране от runs."""
|
"""Параграф → <p>…</p> с inline форматиране от runs (вкл. hyperlink TOC)."""
|
||||||
parts = [_docx_run_to_html(r) for r in para.runs]
|
parts = [_docx_run_to_html(r, para) for r in _iter_docx_para_runs(para)]
|
||||||
inner = "".join(parts)
|
inner = "".join(parts)
|
||||||
if not inner.strip():
|
if not inner.strip():
|
||||||
|
# Fallback: para.text вижда hyperlink текст, но без runs в .runs
|
||||||
|
plain = (para.text or "").strip()
|
||||||
|
if not plain:
|
||||||
return ""
|
return ""
|
||||||
|
inner = html_lib.escape(plain, quote=False).replace("\n", "<br>")
|
||||||
return f"<p>{inner}</p>"
|
return f"<p>{inner}</p>"
|
||||||
|
|
||||||
|
|
||||||
|
|||||||
@@ -1,7 +1,10 @@
|
|||||||
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
|
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
|
import pytest
|
||||||
from docx import Document
|
from docx import Document
|
||||||
|
from docx.oxml import OxmlElement
|
||||||
|
from docx.oxml.ns import qn
|
||||||
from docx.shared import RGBColor
|
from docx.shared import RGBColor
|
||||||
|
|
||||||
from help_processor import (
|
from help_processor import (
|
||||||
@@ -13,6 +16,7 @@ from help_processor import (
|
|||||||
)
|
)
|
||||||
|
|
||||||
FIXTURES = Path(__file__).parent / "fixtures"
|
FIXTURES = Path(__file__).parent / "fixtures"
|
||||||
|
ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx")
|
||||||
|
|
||||||
|
|
||||||
def _mini_png(w: int = 64, h: int = 64) -> bytes:
|
def _mini_png(w: int = 64, h: int = 64) -> bytes:
|
||||||
@@ -34,6 +38,79 @@ def _mini_png(w: int = 64, h: int = 64) -> bytes:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"):
|
||||||
|
"""TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == [])."""
|
||||||
|
p = doc.add_paragraph()
|
||||||
|
for child in list(p._element):
|
||||||
|
if child.tag == qn("w:r"):
|
||||||
|
p._element.remove(child)
|
||||||
|
hl = OxmlElement("w:hyperlink")
|
||||||
|
hl.set(qn("w:anchor"), anchor)
|
||||||
|
r = OxmlElement("w:r")
|
||||||
|
t = OxmlElement("w:t")
|
||||||
|
t.set(qn("xml:space"), "preserve")
|
||||||
|
t.text = text
|
||||||
|
r.append(t)
|
||||||
|
hl.append(r)
|
||||||
|
p._element.append(hl)
|
||||||
|
assert len(p.runs) == 0
|
||||||
|
assert text in (p.text or "")
|
||||||
|
return p
|
||||||
|
|
||||||
|
|
||||||
|
def _add_external_blip_paragraph(doc: Document, image_path: Path):
|
||||||
|
"""Параграф с a:blip/@r:link към външен файл (като atra-manual.docx)."""
|
||||||
|
from docx.opc.constants import RELATIONSHIP_TYPE as RT
|
||||||
|
|
||||||
|
rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri())
|
||||||
|
rId = rel if isinstance(rel, str) else rel.rId
|
||||||
|
|
||||||
|
p = doc.add_paragraph()
|
||||||
|
r = OxmlElement("w:r")
|
||||||
|
drawing = OxmlElement("w:drawing")
|
||||||
|
inline = OxmlElement("wp:inline")
|
||||||
|
inline.set("distT", "0")
|
||||||
|
inline.set("distB", "0")
|
||||||
|
inline.set("distL", "0")
|
||||||
|
inline.set("distR", "0")
|
||||||
|
extent = OxmlElement("wp:extent")
|
||||||
|
extent.set("cx", "914400")
|
||||||
|
extent.set("cy", "914400")
|
||||||
|
inline.append(extent)
|
||||||
|
docPr = OxmlElement("wp:docPr")
|
||||||
|
docPr.set("id", "1")
|
||||||
|
docPr.set("name", "Picture")
|
||||||
|
inline.append(docPr)
|
||||||
|
graphic = OxmlElement("a:graphic")
|
||||||
|
graphicData = OxmlElement("a:graphicData")
|
||||||
|
graphicData.set(
|
||||||
|
"uri", "http://schemas.openxmlformats.org/drawingml/2006/picture"
|
||||||
|
)
|
||||||
|
pic = OxmlElement("pic:pic")
|
||||||
|
nvPicPr = OxmlElement("pic:nvPicPr")
|
||||||
|
cNvPr = OxmlElement("pic:cNvPr")
|
||||||
|
cNvPr.set("id", "0")
|
||||||
|
cNvPr.set("name", "pic")
|
||||||
|
nvPicPr.append(cNvPr)
|
||||||
|
nvPicPr.append(OxmlElement("pic:cNvPicPr"))
|
||||||
|
pic.append(nvPicPr)
|
||||||
|
blipFill = OxmlElement("pic:blipFill")
|
||||||
|
blip = OxmlElement("a:blip")
|
||||||
|
blip.set(qn("r:link"), rId)
|
||||||
|
blipFill.append(blip)
|
||||||
|
blipFill.append(OxmlElement("a:stretch"))
|
||||||
|
pic.append(blipFill)
|
||||||
|
spPr = OxmlElement("pic:spPr")
|
||||||
|
pic.append(spPr)
|
||||||
|
graphicData.append(pic)
|
||||||
|
graphic.append(graphicData)
|
||||||
|
inline.append(graphic)
|
||||||
|
drawing.append(inline)
|
||||||
|
r.append(drawing)
|
||||||
|
p._element.append(r)
|
||||||
|
return p
|
||||||
|
|
||||||
|
|
||||||
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
|
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
|
||||||
doc = Document()
|
doc = Document()
|
||||||
doc.add_heading("1. Форматиране", level=1)
|
doc.add_heading("1. Форматиране", level=1)
|
||||||
@@ -77,6 +154,95 @@ def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
|
|||||||
assert "[IMG:" in (body.html_text or "")
|
assert "[IMG:" in (body.html_text or "")
|
||||||
|
|
||||||
|
|
||||||
|
def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path):
|
||||||
|
"""TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази."""
|
||||||
|
doc = Document()
|
||||||
|
doc.add_heading("NESPERTCAM Launcher - тест", level=1)
|
||||||
|
doc.add_heading("Съдържание", level=3)
|
||||||
|
_add_hyperlink_paragraph(doc, "1. Преглед на приложението")
|
||||||
|
_add_hyperlink_paragraph(doc, "2. Инсталация и настройка")
|
||||||
|
doc.add_heading("1. Преглед на приложението", level=2)
|
||||||
|
h3 = doc.add_heading("Какво е NESPERTCAM?", level=3)
|
||||||
|
# Heading style → run.bold is None, style.font.bold True
|
||||||
|
assert h3.runs and h3.runs[0].bold is not True
|
||||||
|
p = doc.add_paragraph()
|
||||||
|
p.add_run("Тяло с достатъчно думи за секцията на ръководството.")
|
||||||
|
|
||||||
|
out = tmp_path / "toc_hyperlink.docx"
|
||||||
|
doc.save(out)
|
||||||
|
|
||||||
|
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||||||
|
assert sections
|
||||||
|
preamble = sections[0]
|
||||||
|
assert "Съдържание" in (preamble.text or "")
|
||||||
|
assert "1. Преглед" in (preamble.text or "")
|
||||||
|
html = preamble.html_text or ""
|
||||||
|
assert "Съдържание" in html
|
||||||
|
assert "1. Преглед" in html
|
||||||
|
assert "2. Инсталация" in html
|
||||||
|
|
||||||
|
overview = next(s for s in sections if s.title.startswith("1. Преглед"))
|
||||||
|
assert "<b>" in (overview.html_text or "")
|
||||||
|
assert "Какво е NESPERTCAM?" in (overview.html_text or "")
|
||||||
|
|
||||||
|
|
||||||
|
def test_docx_external_r_link_image(tmp_path: Path):
|
||||||
|
png = _mini_png()
|
||||||
|
img_path = tmp_path / "ext.png"
|
||||||
|
img_path.write_bytes(png)
|
||||||
|
try:
|
||||||
|
from io import BytesIO
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
assert Image.open(BytesIO(png)).size[0] >= 50
|
||||||
|
except Exception:
|
||||||
|
pytest.skip("PIL unavailable for size check")
|
||||||
|
|
||||||
|
doc = Document()
|
||||||
|
doc.add_heading("1. С картинка", level=1)
|
||||||
|
doc.add_paragraph("Преди фигурата.")
|
||||||
|
try:
|
||||||
|
_add_external_blip_paragraph(doc, img_path)
|
||||||
|
except Exception as exc:
|
||||||
|
pytest.skip(f"cannot build r:link drawing: {exc}")
|
||||||
|
doc.add_paragraph("След фигурата.")
|
||||||
|
|
||||||
|
out = tmp_path / "ext_img.docx"
|
||||||
|
doc.save(out)
|
||||||
|
|
||||||
|
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||||||
|
body = sections[0]
|
||||||
|
assert body.images, "r:link external image must be loaded"
|
||||||
|
assert "[IMG:" in (body.text or "")
|
||||||
|
assert "[IMG:" in (body.html_text or "")
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
|
||||||
|
def test_atra_manual_toc_bold_and_image_in_html():
|
||||||
|
"""Регресия върху реалния NESPERTCAM Launcher manual."""
|
||||||
|
sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL)))
|
||||||
|
assert len(sections) >= 3
|
||||||
|
first = sections[0]
|
||||||
|
assert "NESPERTCAM" in first.title
|
||||||
|
assert "Съдържание" in (first.text or "")
|
||||||
|
assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "")
|
||||||
|
html0 = first.html_text or ""
|
||||||
|
assert "Съдържание" in html0
|
||||||
|
assert "1. Преглед" in html0 or "Инсталация" in html0
|
||||||
|
|
||||||
|
# Style-bold подзаглавия + изричен bold по-нататък
|
||||||
|
assert any("<b>" in (s.html_text or "") for s in sections)
|
||||||
|
|
||||||
|
# Фигура 1 е външен UNC линк — ако share-ът е достъпен, трябва [IMG:]
|
||||||
|
overview = next((s for s in sections if "Преглед" in s.title), None)
|
||||||
|
assert overview is not None
|
||||||
|
if overview.images or "[IMG:" in (overview.text or ""):
|
||||||
|
assert "[IMG:" in (overview.html_text or "")
|
||||||
|
else:
|
||||||
|
# Документът сочи към външен файл; ако UNC липсва, plain caption остава
|
||||||
|
assert "Фигура 1" in (overview.text or "")
|
||||||
|
|
||||||
|
|
||||||
def test_html_preserves_bold_and_style_color(tmp_path: Path):
|
def test_html_preserves_bold_and_style_color(tmp_path: Path):
|
||||||
html = tmp_path / "rich.html"
|
html = tmp_path / "rich.html"
|
||||||
html.write_text(
|
html.write_text(
|
||||||
|
|||||||
Reference in New Issue
Block a user