This commit is contained in:
2026-09-14 11:47:29 +03:00
parent 971f60652d
commit 0dad80bb17
3 changed files with 301 additions and 17 deletions

View File

@@ -100,9 +100,9 @@ TXT и PDF запазват своите правила (номерирани р
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`. Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`.
Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`). Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`). Поддържат се `r:embed` (вградени) и `r:link` (външни `file:///` / UNC), ако файлът е достъпен при сканиране.
**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>`, `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. **Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>` (изричен run **или** наследен от paragraph/character style), `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Runs вътре в `w:hyperlink` (TOC редове) също влизат в `html_text`; ако runs липсват, има fallback към escaped `para.text`. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват. Viewer показва `html_text` когато е наличен — затова TOC/bold/картинки трябва да са в него, не само в plain `text`.
Празен параграф без картинка се пропуска. Празен параграф без картинка се пропуска.

View File

@@ -985,8 +985,69 @@ def parse_html(path: Path) -> list[Section]:
return sections return sections
def _file_url_to_path(url: str) -> Optional[Path]:
"""file:///… или обикновен път → Path (вкл. UNC от Word r:link)."""
from urllib.parse import unquote
s = unquote((url or "").strip())
if not s:
return None
if s.lower().startswith("file:"):
s = s[5:]
while s.startswith("/"):
s = s[1:]
if not s:
return None
return Path(s)
def _load_external_image_bytes(url: str) -> Optional[tuple[bytes, str]]:
"""Чете външна картинка (r:link / TargetMode=External); връща (data, ext)."""
path = _file_url_to_path(url)
if path is None:
return None
try:
if not path.is_file():
return None
data = path.read_bytes()
except OSError:
return None
if not data:
return None
ext = (path.suffix or "").lstrip(".").lower() or "png"
if ext == "jpeg":
ext = "jpg"
return data, ext
def _resolve_docx_image_rid(doc, rId: str) -> Optional[tuple[bytes, str]]:
"""r:embed → related_parts; r:link → външен файл. Връща (data, ext)."""
if not rId:
return None
try:
part = doc.part.related_parts[rId]
data = part.blob
ct = getattr(part, "content_type", "") or ""
if data:
return data, _ext_from_content_type(ct)
except Exception:
pass
try:
rel = doc.part.rels[rId]
except Exception:
return None
if not getattr(rel, "is_external", False):
return None
target = getattr(rel, "target_ref", None) or ""
loaded = _load_external_image_bytes(target)
return loaded
def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]: def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
"""Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.""" """Намира drawing-и в параграф; връща ImageRef-и за филтрираните по размер.
Поддържа r:embed (вградени) и r:link (външни file:/// / UNC пътища).
"""
from docx.oxml.ns import qn from docx.oxml.ns import qn
imgs: list[ImageRef] = [] imgs: list[ImageRef] = []
try: try:
@@ -995,19 +1056,17 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
return imgs return imgs
embed_attr = qn("r:embed") embed_attr = qn("r:embed")
link_attr = qn("r:link")
for blip in blips: for blip in blips:
rId = blip.get(embed_attr) rId = blip.get(embed_attr) or blip.get(link_attr)
if not rId: if not rId:
continue continue
try: resolved = _resolve_docx_image_rid(doc, rId)
part = doc.part.related_parts[rId] if not resolved:
data = part.blob
ct = getattr(part, "content_type", "") or ""
except Exception:
continue continue
data, ext = resolved
if not _should_keep_image(data): if not _should_keep_image(data):
continue continue
ext = _ext_from_content_type(ct)
imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext)) imgs.append(ImageRef(placeholder=f"__IMG_{len(imgs)+1}__", data=data, ext=ext))
return imgs return imgs
@@ -1023,17 +1082,72 @@ def _docx_run_color_css(run) -> Optional[str]:
return None return None
def _docx_run_to_html(run) -> str: def _docx_style_font_flag(style, attr: str) -> Optional[bool]:
"""True/False от style.font.<attr>; None ако стилът/атрибутът липсва."""
if style is None:
return None
try:
font = style.font
if font is None:
return None
val = getattr(font, attr, None)
if val is None:
return None
return bool(val)
except Exception:
return None
def _docx_run_effective_flag(run, para, attr: str) -> bool:
"""Ефективен bold/italic/underline: изричен run → char style → paragraph style."""
try:
direct = getattr(run, attr, None)
except Exception:
direct = None
if direct is True:
return True
if direct is False:
return False
# None = наследяване
try:
char_style = run.style
except Exception:
char_style = None
flag = _docx_style_font_flag(char_style, attr)
if flag is not None:
return flag
try:
p_style = para.style if para is not None else None
except Exception:
p_style = None
flag = _docx_style_font_flag(p_style, attr)
return bool(flag)
def _iter_docx_para_runs(para):
"""Runs в параграф, вкл. вътре в w:hyperlink (python-docx.para.runs ги пропуска)."""
from docx.oxml.ns import qn
from docx.text.run import Run
for child in para._element.iterchildren():
if child.tag == qn("w:r"):
yield Run(child, para)
elif child.tag == qn("w:hyperlink"):
for r_elem in child.findall(qn("w:r")):
yield Run(r_elem, para)
def _docx_run_to_html(run, para=None) -> str:
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват.""" """Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
text = run.text or "" text = run.text or ""
if not text: if not text:
return "" return ""
s = html_lib.escape(text, quote=False).replace("\n", "<br>") s = html_lib.escape(text, quote=False).replace("\n", "<br>")
if run.bold: if _docx_run_effective_flag(run, para, "bold"):
s = f"<b>{s}</b>" s = f"<b>{s}</b>"
if run.italic: if _docx_run_effective_flag(run, para, "italic"):
s = f"<i>{s}</i>" s = f"<i>{s}</i>"
if run.underline: if _docx_run_effective_flag(run, para, "underline"):
s = f"<u>{s}</u>" s = f"<u>{s}</u>"
css_color = _docx_run_color_css(run) css_color = _docx_run_color_css(run)
if css_color: if css_color:
@@ -1042,11 +1156,15 @@ def _docx_run_to_html(run) -> str:
def _docx_para_to_html(para) -> str: def _docx_para_to_html(para) -> str:
"""Параграф → <p>…</p> с inline форматиране от runs.""" """Параграф → <p>…</p> с inline форматиране от runs (вкл. hyperlink TOC)."""
parts = [_docx_run_to_html(r) for r in para.runs] parts = [_docx_run_to_html(r, para) for r in _iter_docx_para_runs(para)]
inner = "".join(parts) inner = "".join(parts)
if not inner.strip(): if not inner.strip():
# Fallback: para.text вижда hyperlink текст, но без runs в .runs
plain = (para.text or "").strip()
if not plain:
return "" return ""
inner = html_lib.escape(plain, quote=False).replace("\n", "<br>")
return f"<p>{inner}</p>" return f"<p>{inner}</p>"

View File

@@ -1,7 +1,10 @@
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML).""" """Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
from pathlib import Path from pathlib import Path
import pytest
from docx import Document from docx import Document
from docx.oxml import OxmlElement
from docx.oxml.ns import qn
from docx.shared import RGBColor from docx.shared import RGBColor
from help_processor import ( from help_processor import (
@@ -13,6 +16,7 @@ from help_processor import (
) )
FIXTURES = Path(__file__).parent / "fixtures" FIXTURES = Path(__file__).parent / "fixtures"
ATRA_MANUAL = Path(r"q:\RIP_Help_Source\atra-manual.docx")
def _mini_png(w: int = 64, h: int = 64) -> bytes: def _mini_png(w: int = 64, h: int = 64) -> bytes:
@@ -34,6 +38,79 @@ def _mini_png(w: int = 64, h: int = 64) -> bytes:
) )
def _add_hyperlink_paragraph(doc: Document, text: str, anchor: str = "toc1"):
"""TOC-style параграф: текстът е само в w:hyperlink/w:r (para.runs == [])."""
p = doc.add_paragraph()
for child in list(p._element):
if child.tag == qn("w:r"):
p._element.remove(child)
hl = OxmlElement("w:hyperlink")
hl.set(qn("w:anchor"), anchor)
r = OxmlElement("w:r")
t = OxmlElement("w:t")
t.set(qn("xml:space"), "preserve")
t.text = text
r.append(t)
hl.append(r)
p._element.append(hl)
assert len(p.runs) == 0
assert text in (p.text or "")
return p
def _add_external_blip_paragraph(doc: Document, image_path: Path):
"""Параграф с a:blip/@r:link към външен файл (като atra-manual.docx)."""
from docx.opc.constants import RELATIONSHIP_TYPE as RT
rel = doc.part.rels.get_or_add_ext_rel(RT.IMAGE, image_path.resolve().as_uri())
rId = rel if isinstance(rel, str) else rel.rId
p = doc.add_paragraph()
r = OxmlElement("w:r")
drawing = OxmlElement("w:drawing")
inline = OxmlElement("wp:inline")
inline.set("distT", "0")
inline.set("distB", "0")
inline.set("distL", "0")
inline.set("distR", "0")
extent = OxmlElement("wp:extent")
extent.set("cx", "914400")
extent.set("cy", "914400")
inline.append(extent)
docPr = OxmlElement("wp:docPr")
docPr.set("id", "1")
docPr.set("name", "Picture")
inline.append(docPr)
graphic = OxmlElement("a:graphic")
graphicData = OxmlElement("a:graphicData")
graphicData.set(
"uri", "http://schemas.openxmlformats.org/drawingml/2006/picture"
)
pic = OxmlElement("pic:pic")
nvPicPr = OxmlElement("pic:nvPicPr")
cNvPr = OxmlElement("pic:cNvPr")
cNvPr.set("id", "0")
cNvPr.set("name", "pic")
nvPicPr.append(cNvPr)
nvPicPr.append(OxmlElement("pic:cNvPicPr"))
pic.append(nvPicPr)
blipFill = OxmlElement("pic:blipFill")
blip = OxmlElement("a:blip")
blip.set(qn("r:link"), rId)
blipFill.append(blip)
blipFill.append(OxmlElement("a:stretch"))
pic.append(blipFill)
spPr = OxmlElement("pic:spPr")
pic.append(spPr)
graphicData.append(pic)
graphic.append(graphicData)
inline.append(graphic)
drawing.append(inline)
r.append(drawing)
p._element.append(r)
return p
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path): def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
doc = Document() doc = Document()
doc.add_heading("1. Форматиране", level=1) doc.add_heading("1. Форматиране", level=1)
@@ -77,6 +154,95 @@ def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
assert "[IMG:" in (body.html_text or "") assert "[IMG:" in (body.html_text or "")
def test_docx_hyperlink_toc_and_style_bold_in_html(tmp_path: Path):
"""TOC в w:hyperlink + Heading bold без run.bold → html_text ги пази."""
doc = Document()
doc.add_heading("NESPERTCAM Launcher - тест", level=1)
doc.add_heading("Съдържание", level=3)
_add_hyperlink_paragraph(doc, "1. Преглед на приложението")
_add_hyperlink_paragraph(doc, "2. Инсталация и настройка")
doc.add_heading("1. Преглед на приложението", level=2)
h3 = doc.add_heading("Какво е NESPERTCAM?", level=3)
# Heading style → run.bold is None, style.font.bold True
assert h3.runs and h3.runs[0].bold is not True
p = doc.add_paragraph()
p.add_run("Тяло с достатъчно думи за секцията на ръководството.")
out = tmp_path / "toc_hyperlink.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
assert sections
preamble = sections[0]
assert "Съдържание" in (preamble.text or "")
assert "1. Преглед" in (preamble.text or "")
html = preamble.html_text or ""
assert "Съдържание" in html
assert "1. Преглед" in html
assert "2. Инсталация" in html
overview = next(s for s in sections if s.title.startswith("1. Преглед"))
assert "<b>" in (overview.html_text or "")
assert "Какво е NESPERTCAM?" in (overview.html_text or "")
def test_docx_external_r_link_image(tmp_path: Path):
png = _mini_png()
img_path = tmp_path / "ext.png"
img_path.write_bytes(png)
try:
from io import BytesIO
from PIL import Image
assert Image.open(BytesIO(png)).size[0] >= 50
except Exception:
pytest.skip("PIL unavailable for size check")
doc = Document()
doc.add_heading("1. С картинка", level=1)
doc.add_paragraph("Преди фигурата.")
try:
_add_external_blip_paragraph(doc, img_path)
except Exception as exc:
pytest.skip(f"cannot build r:link drawing: {exc}")
doc.add_paragraph("След фигурата.")
out = tmp_path / "ext_img.docx"
doc.save(out)
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
body = sections[0]
assert body.images, "r:link external image must be loaded"
assert "[IMG:" in (body.text or "")
assert "[IMG:" in (body.html_text or "")
@pytest.mark.skipif(not ATRA_MANUAL.is_file(), reason="atra-manual.docx not mounted")
def test_atra_manual_toc_bold_and_image_in_html():
"""Регресия върху реалния NESPERTCAM Launcher manual."""
sections = merge_preamble_sections(merge_short_sections(parse_docx(ATRA_MANUAL)))
assert len(sections) >= 3
first = sections[0]
assert "NESPERTCAM" in first.title
assert "Съдържание" in (first.text or "")
assert "1. Преглед" in (first.text or "") or "Инсталация" in (first.text or "")
html0 = first.html_text or ""
assert "Съдържание" in html0
assert "1. Преглед" in html0 or "Инсталация" in html0
# Style-bold подзаглавия + изричен bold по-нататък
assert any("<b>" in (s.html_text or "") for s in sections)
# Фигура 1 е външен UNC линк — ако share-ът е достъпен, трябва [IMG:]
overview = next((s for s in sections if "Преглед" in s.title), None)
assert overview is not None
if overview.images or "[IMG:" in (overview.text or ""):
assert "[IMG:" in (overview.html_text or "")
else:
# Документът сочи към външен файл; ако UNC липсва, plain caption остава
assert "Фигура 1" in (overview.text or "")
def test_html_preserves_bold_and_style_color(tmp_path: Path): def test_html_preserves_bold_and_style_color(tmp_path: Path):
html = tmp_path / "rich.html" html = tmp_path / "rich.html"
html.write_text( html.write_text(