.
This commit is contained in:
@@ -92,7 +92,7 @@ UNIQUE constraint: `(prefix, file_path)`
|
|||||||
| char_count | INT | Размер на чистия текст |
|
| char_count | INT | Размер на чистия текст |
|
||||||
| output_path | NVARCHAR(1000) | Път до `.txt` файла |
|
| output_path | NVARCHAR(1000) | Път до `.txt` файла |
|
||||||
| images | NVARCHAR(MAX) | JSON масив с относителни пътища |
|
| images | NVARCHAR(MAX) | JSON масив с относителни пътища |
|
||||||
| html_text | NVARCHAR(MAX) | Rich HTML с форматиране (само за `.html` източници) |
|
| html_text | NVARCHAR(MAX) | Rich HTML (`.html` и `.docx`: bold/цветове/картинки) |
|
||||||
| created_at, updated_at | DATETIME2 | |
|
| created_at, updated_at | DATETIME2 | |
|
||||||
|
|
||||||
## HTML Viewer — 3 / 4 таба
|
## HTML Viewer — 3 / 4 таба
|
||||||
|
|||||||
@@ -98,9 +98,11 @@ TXT и PDF запазват своите правила (номерирани р
|
|||||||
|
|
||||||
### 3.3. Таблици и картинки
|
### 3.3. Таблици и картинки
|
||||||
|
|
||||||
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` и редовете се добавят в тялото на текущата секция.
|
Таблица: всеки ред се сплесква до `клетка | клетка | клетка` в plain text; в `html_text` — прост `<table>`.
|
||||||
|
|
||||||
Картинки в параграф: записват се като `[IMG: img_NN]` в същото тяло.
|
Картинки в параграф: записват се като `[IMG: img_NN]` в plain text и в `html_text` (viewer ги заменя с `<img>`).
|
||||||
|
|
||||||
|
**Rich HTML (DOCX):** за всеки параграф в тялото се строи `html_text` от Word runs — `<b>` / `<i>` / `<u>`, `<span style="color:#RRGGBB">` при изричен RGB цвят, спец. символи (•, →, …) чрез HTML escape. Plain `text` остава без markup (за split / Claude). Theme/auto цветове без RGB не се записват.
|
||||||
|
|
||||||
Празен параграф без картинка се пропуска.
|
Празен параграф без картинка се пропуска.
|
||||||
|
|
||||||
@@ -132,7 +134,7 @@ TXT и PDF запазват своите правила (номерирани р
|
|||||||
|
|
||||||
### 5.2. Съдържание
|
### 5.2. Съдържание
|
||||||
|
|
||||||
Списъци и таблици влизат в текущата секция. В plain text редовете им са с нов ред (не сплескани в един ред). Декоративни атрибути (`class`, `style`, `on*`, `data-*` …) се махат от запазения HTML.
|
Списъци и таблици влизат в текущата секция. В plain text редовете им са с нов ред (не сплескани в един ред). Декоративни атрибути (`class`, `id`, `on*`, `data-*` …) се махат; от `style` се **пазят** `color`, `font-weight` (bold) и `font-style` (italic). Тагове `<b>`/`<strong>`/`<i>`/`<em>` и `<font color>` остават.
|
||||||
|
|
||||||
Картинки: локални и `data:` URI; HTTP(S) се пропускат. Дребни иконки под 50×50 px се изхвърлят.
|
Картинки: локални и `data:` URI; HTTP(S) се пропускат. Дребни иконки под 50×50 px се изхвърлят.
|
||||||
|
|
||||||
@@ -192,7 +194,7 @@ TXT и PDF запазват своите правила (номерирани р
|
|||||||
- Ако няма текст, картинки и HTML, но има заглавие → тялото става самото заглавие. Затова „голо“ заглавие оцелява като секция.
|
- Ако няма текст, картинки и HTML, но има заглавие → тялото става самото заглавие. Затова „голо“ заглавие оцелява като секция.
|
||||||
- Ако няма нито текст, нито заглавие, нито картинки → секцията се пропуска.
|
- Ако няма нито текст, нито заглавие, нито картинки → секцията се пропуска.
|
||||||
|
|
||||||
`clean_text()` срива поредици от интервали/табове, но пази нови редове (списъци и абзаци).
|
`clean_text()` срива поредици от интервали/табове, но пази нови редове (списъци и абзаци). Маха опасни C0 контроли (NUL и др.); Unicode символи (•, →) остават; NBSP става обикновен интервал.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
|||||||
@@ -22,6 +22,7 @@ help_processor.py
|
|||||||
import os
|
import os
|
||||||
import re
|
import re
|
||||||
import sys
|
import sys
|
||||||
|
import html as html_lib
|
||||||
import json
|
import json
|
||||||
import hashlib
|
import hashlib
|
||||||
import logging
|
import logging
|
||||||
@@ -164,7 +165,7 @@ class ProcessedSection:
|
|||||||
keywords: str # "кл1, кл2, кл3"
|
keywords: str # "кл1, кл2, кл3"
|
||||||
text: str
|
text: str
|
||||||
images_json: str = "[]" # JSON масив с относителни пътища
|
images_json: str = "[]" # JSON масив с относителни пътища
|
||||||
html_text: str = "" # rich HTML (само за HTML-source файлове)
|
html_text: str = "" # rich HTML (HTML + DOCX; bold/цветове/картинки)
|
||||||
char_count: int = 0
|
char_count: int = 0
|
||||||
|
|
||||||
def __post_init__(self):
|
def __post_init__(self):
|
||||||
@@ -556,6 +557,12 @@ _HTML_BLOCK_TAGS = ["h1", "h2", "h3", "h4", "h5", "h6",
|
|||||||
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
_HTML_PLAIN_NL_TAGS = frozenset({"ul", "ol", "table", "dl", "pre", "blockquote"})
|
||||||
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
_HTML_DROP_ATTRS = ("class", "style", "id", "lang", "dir", "align",
|
||||||
"valign", "width", "height", "bgcolor", "border")
|
"valign", "width", "height", "bgcolor", "border")
|
||||||
|
# От style пазим само inline форматиране, нужно за viewer (bold/italic/color).
|
||||||
|
_HTML_STYLE_KEEP_RES = (
|
||||||
|
("color", re.compile(r"(?:^|;)\s*color\s*:\s*([^;]+)", re.I)),
|
||||||
|
("font-weight", re.compile(r"(?:^|;)\s*font-weight\s*:\s*([^;]+)", re.I)),
|
||||||
|
("font-style", re.compile(r"(?:^|;)\s*font-style\s*:\s*([^;]+)", re.I)),
|
||||||
|
)
|
||||||
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
_HTML_HEADING_MAP = {"h1": 1, "h2": 2, "h3": 3, "h4": 3, "h5": 3, "h6": 3}
|
||||||
_HEADING_TOKEN_RE = re.compile(
|
_HEADING_TOKEN_RE = re.compile(
|
||||||
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
|
r"^(heading|title|subtitle|заглавие|подзаглавие|наименование|überschrift|msoheading)"
|
||||||
@@ -786,12 +793,42 @@ def _html_block_plain_text(el) -> str:
|
|||||||
return el.get_text(sep, strip=True)
|
return el.get_text(sep, strip=True)
|
||||||
|
|
||||||
|
|
||||||
|
def _kept_inline_style(style: str) -> str:
|
||||||
|
"""Извлича color / bold / italic от CSS style; останалото се маха."""
|
||||||
|
if not style:
|
||||||
|
return ""
|
||||||
|
parts: list[str] = []
|
||||||
|
for prop, rx in _HTML_STYLE_KEEP_RES:
|
||||||
|
m = rx.search(style)
|
||||||
|
if not m:
|
||||||
|
continue
|
||||||
|
val = m.group(1).strip()
|
||||||
|
if not val:
|
||||||
|
continue
|
||||||
|
low = val.lower()
|
||||||
|
if prop == "font-weight" and low not in (
|
||||||
|
"bold", "bolder", "600", "700", "800", "900"
|
||||||
|
):
|
||||||
|
continue
|
||||||
|
if prop == "font-style" and "italic" not in low and "oblique" not in low:
|
||||||
|
continue
|
||||||
|
parts.append(f"{prop}:{val}")
|
||||||
|
return ";".join(parts)
|
||||||
|
|
||||||
|
|
||||||
def _strip_attrs(el):
|
def _strip_attrs(el):
|
||||||
"""Премахва decorative атрибути (class, style, on*, data-*)."""
|
"""Премахва decorative атрибути; пази color/bold/italic от style и font color=."""
|
||||||
for t in el.find_all(True):
|
for t in el.find_all(True):
|
||||||
|
kept_style = ""
|
||||||
|
raw_style = t.attrs.get("style") if t.attrs else None
|
||||||
|
if isinstance(raw_style, str):
|
||||||
|
kept_style = _kept_inline_style(raw_style)
|
||||||
for a in list(t.attrs):
|
for a in list(t.attrs):
|
||||||
if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"):
|
if a in _HTML_DROP_ATTRS or a.startswith("on") or a.startswith("data-"):
|
||||||
del t[a]
|
del t[a]
|
||||||
|
# <font color="..."> остава (color не е в DROP)
|
||||||
|
if kept_style:
|
||||||
|
t["style"] = kept_style
|
||||||
|
|
||||||
|
|
||||||
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
|
def _swap_imgs_in_block(el, base_dir: Path, sec_images: list, img_counter: list) -> None:
|
||||||
@@ -975,6 +1012,69 @@ def _extract_docx_paragraph_images(para, doc) -> list[ImageRef]:
|
|||||||
return imgs
|
return imgs
|
||||||
|
|
||||||
|
|
||||||
|
def _docx_run_color_css(run) -> Optional[str]:
|
||||||
|
"""RGB от run.font.color → '#RRGGBB'; theme/auto цветове се пропускат."""
|
||||||
|
try:
|
||||||
|
color = run.font.color
|
||||||
|
if color is None or color.rgb is None:
|
||||||
|
return None
|
||||||
|
return f"#{color.rgb}"
|
||||||
|
except Exception:
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _docx_run_to_html(run) -> str:
|
||||||
|
"""Един Word run → HTML с <b>/<i>/<u> и color span. Спец. символи се escape-ват."""
|
||||||
|
text = run.text or ""
|
||||||
|
if not text:
|
||||||
|
return ""
|
||||||
|
s = html_lib.escape(text, quote=False).replace("\n", "<br>")
|
||||||
|
if run.bold:
|
||||||
|
s = f"<b>{s}</b>"
|
||||||
|
if run.italic:
|
||||||
|
s = f"<i>{s}</i>"
|
||||||
|
if run.underline:
|
||||||
|
s = f"<u>{s}</u>"
|
||||||
|
css_color = _docx_run_color_css(run)
|
||||||
|
if css_color:
|
||||||
|
s = f'<span style="color:{css_color}">{s}</span>'
|
||||||
|
return s
|
||||||
|
|
||||||
|
|
||||||
|
def _docx_para_to_html(para) -> str:
|
||||||
|
"""Параграф → <p>…</p> с inline форматиране от runs."""
|
||||||
|
parts = [_docx_run_to_html(r) for r in para.runs]
|
||||||
|
inner = "".join(parts)
|
||||||
|
if not inner.strip():
|
||||||
|
return ""
|
||||||
|
return f"<p>{inner}</p>"
|
||||||
|
|
||||||
|
|
||||||
|
def _table_to_html(table) -> str:
|
||||||
|
"""Таблица → прост HTML <table> (plain клетки, без вложен rich text)."""
|
||||||
|
rows_html: list[str] = []
|
||||||
|
try:
|
||||||
|
rows = table.rows
|
||||||
|
except Exception:
|
||||||
|
return ""
|
||||||
|
for row in rows:
|
||||||
|
cells_html: list[str] = []
|
||||||
|
try:
|
||||||
|
cells = row.cells
|
||||||
|
except Exception:
|
||||||
|
continue
|
||||||
|
for cell in cells:
|
||||||
|
cell_text = " ".join((cell.text or "").split())
|
||||||
|
if not cell_text:
|
||||||
|
continue
|
||||||
|
cells_html.append(f"<td>{html_lib.escape(cell_text, quote=False)}</td>")
|
||||||
|
if cells_html:
|
||||||
|
rows_html.append("<tr>" + "".join(cells_html) + "</tr>")
|
||||||
|
if not rows_html:
|
||||||
|
return ""
|
||||||
|
return "<table>" + "".join(rows_html) + "</table>"
|
||||||
|
|
||||||
|
|
||||||
def _iter_docx_blocks(doc):
|
def _iter_docx_blocks(doc):
|
||||||
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
|
"""Параграфи и таблици в document-order (python-docx .paragraphs пропуска таблиците)."""
|
||||||
from docx.oxml.ns import qn
|
from docx.oxml.ns import qn
|
||||||
@@ -1011,6 +1111,7 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
sections: list[Section] = []
|
sections: list[Section] = []
|
||||||
current_title, current_level = "", 1
|
current_title, current_level = "", 1
|
||||||
buf: list[str] = []
|
buf: list[str] = []
|
||||||
|
buf_html: list[str] = []
|
||||||
sec_images: list[ImageRef] = []
|
sec_images: list[ImageRef] = []
|
||||||
img_counter = [0]
|
img_counter = [0]
|
||||||
toc_state = _TocSplitState()
|
toc_state = _TocSplitState()
|
||||||
@@ -1019,21 +1120,30 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
if current_title or buf or sec_images:
|
if current_title or buf or sec_images:
|
||||||
sec = Section(current_title, "\n".join(buf), current_level)
|
sec = Section(current_title, "\n".join(buf), current_level)
|
||||||
sec.images = list(sec_images)
|
sec.images = list(sec_images)
|
||||||
|
sec.html_text = "\n".join(buf_html) if buf_html else None
|
||||||
sections.append(sec)
|
sections.append(sec)
|
||||||
|
|
||||||
def append_para(text: str, para_imgs: list[ImageRef]):
|
def append_para(text: str, para_imgs: list[ImageRef], html_frag: str = ""):
|
||||||
if text:
|
if text:
|
||||||
buf.append(text)
|
buf.append(text)
|
||||||
|
if html_frag:
|
||||||
|
buf_html.append(html_frag)
|
||||||
for im in para_imgs:
|
for im in para_imgs:
|
||||||
img_counter[0] += 1
|
img_counter[0] += 1
|
||||||
im.placeholder = f"img_{img_counter[0]:02d}"
|
im.placeholder = f"img_{img_counter[0]:02d}"
|
||||||
sec_images.append(im)
|
sec_images.append(im)
|
||||||
buf.append(f"[IMG: {im.placeholder}]")
|
ph = f"[IMG: {im.placeholder}]"
|
||||||
|
buf.append(ph)
|
||||||
|
buf_html.append(f"<p>{ph}</p>")
|
||||||
|
|
||||||
|
def append_from_para(para, text: str, para_imgs: list[ImageRef]):
|
||||||
|
html_frag = _docx_para_to_html(para) if text else ""
|
||||||
|
append_para(text, para_imgs, html_frag)
|
||||||
|
|
||||||
def start_section(title: str, level: int = 1):
|
def start_section(title: str, level: int = 1):
|
||||||
nonlocal current_title, current_level, buf, sec_images
|
nonlocal current_title, current_level, buf, buf_html, sec_images
|
||||||
flush()
|
flush()
|
||||||
buf, sec_images = [], []
|
buf, buf_html, sec_images = [], [], []
|
||||||
current_title = title
|
current_title = title
|
||||||
current_level = level
|
current_level = level
|
||||||
|
|
||||||
@@ -1043,6 +1153,9 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
toc_state.leave_phase()
|
toc_state.leave_phase()
|
||||||
for line in _table_lines(block):
|
for line in _table_lines(block):
|
||||||
buf.append(line)
|
buf.append(line)
|
||||||
|
th = _table_to_html(block)
|
||||||
|
if th:
|
||||||
|
buf_html.append(th)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
para = block
|
para = block
|
||||||
@@ -1071,17 +1184,17 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
else:
|
else:
|
||||||
if toc_state.phase:
|
if toc_state.phase:
|
||||||
toc_state.leave_phase()
|
toc_state.leave_phase()
|
||||||
append_para(text, para_imgs)
|
append_from_para(para, text, para_imgs)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if text and _is_toc_heading(text):
|
if text and _is_toc_heading(text):
|
||||||
toc_state.note_toc_heading()
|
toc_state.note_toc_heading()
|
||||||
append_para(text, para_imgs)
|
append_from_para(para, text, para_imgs)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
numbered_action = toc_state.handle_numbered(text) if text else "no"
|
numbered_action = toc_state.handle_numbered(text) if text else "no"
|
||||||
if numbered_action == "toc":
|
if numbered_action == "toc":
|
||||||
append_para(text, para_imgs)
|
append_from_para(para, text, para_imgs)
|
||||||
continue
|
continue
|
||||||
if numbered_action == "split":
|
if numbered_action == "split":
|
||||||
start_section(text, level or 1)
|
start_section(text, level or 1)
|
||||||
@@ -1091,7 +1204,7 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
if (level or 0) >= 2 or is_bold_heading:
|
if (level or 0) >= 2 or is_bold_heading:
|
||||||
if toc_state.phase:
|
if toc_state.phase:
|
||||||
toc_state.leave_phase()
|
toc_state.leave_phase()
|
||||||
append_para(text, para_imgs)
|
append_from_para(para, text, para_imgs)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
if level == 1:
|
if level == 1:
|
||||||
@@ -1102,7 +1215,7 @@ def parse_docx(path: Path) -> list[Section]:
|
|||||||
|
|
||||||
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
|
if toc_state.phase and text and not _is_numbered_chapter_heading(text):
|
||||||
toc_state.leave_phase()
|
toc_state.leave_phase()
|
||||||
append_para(text, para_imgs)
|
append_from_para(para, text, para_imgs)
|
||||||
|
|
||||||
flush()
|
flush()
|
||||||
|
|
||||||
@@ -1521,8 +1634,13 @@ def merge_preamble_sections(sections: list[Section]) -> list[Section]:
|
|||||||
|
|
||||||
|
|
||||||
def clean_text(text: str) -> str:
|
def clean_text(text: str) -> str:
|
||||||
"""Collapse spaces/tabs but keep newlines (lists, paragraphs)."""
|
"""Collapse spaces/tabs but keep newlines (lists, paragraphs).
|
||||||
|
|
||||||
|
Маха опасни C0 контроли (NUL и др.); пази Unicode символи (•, →, NBSP→space).
|
||||||
|
"""
|
||||||
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
text = text.replace("\r\n", "\n").replace("\r", "\n")
|
||||||
|
text = re.sub(r"[\x00-\x08\x0b\x0c\x0e-\x1f\x7f]", "", text)
|
||||||
|
text = text.replace("\u00a0", " ")
|
||||||
text = re.sub(r"[^\S\n]+", " ", text)
|
text = re.sub(r"[^\S\n]+", " ", text)
|
||||||
text = re.sub(r"\n{3,}", "\n\n", text)
|
text = re.sub(r"\n{3,}", "\n\n", text)
|
||||||
text = "\n".join(line.strip() for line in text.split("\n"))
|
text = "\n".join(line.strip() for line in text.split("\n"))
|
||||||
|
|||||||
105
tests/test_rich_content.py
Normal file
105
tests/test_rich_content.py
Normal file
@@ -0,0 +1,105 @@
|
|||||||
|
"""Bold / цветове / картинки / спец. символи в html_text (DOCX + HTML)."""
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
from docx import Document
|
||||||
|
from docx.shared import RGBColor
|
||||||
|
|
||||||
|
from help_processor import (
|
||||||
|
clean_text,
|
||||||
|
merge_preamble_sections,
|
||||||
|
merge_short_sections,
|
||||||
|
parse_docx,
|
||||||
|
parse_html,
|
||||||
|
)
|
||||||
|
|
||||||
|
FIXTURES = Path(__file__).parent / "fixtures"
|
||||||
|
|
||||||
|
|
||||||
|
def _mini_png(w: int = 64, h: int = 64) -> bytes:
|
||||||
|
"""Минимален валиден RGB PNG ≥ MIN_IMAGE_PX без PIL зависимост в теста."""
|
||||||
|
try:
|
||||||
|
from io import BytesIO
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
buf = BytesIO()
|
||||||
|
Image.new("RGB", (w, h), color=(40, 120, 200)).save(buf, format="PNG")
|
||||||
|
return buf.getvalue()
|
||||||
|
except Exception:
|
||||||
|
# 1×1 PNG — ще се филтрира; тестът за картинка се пропуска частично
|
||||||
|
return (
|
||||||
|
b"\x89PNG\r\n\x1a\n\x00\x00\x00\rIHDR\x00\x00\x00\x01"
|
||||||
|
b"\x00\x00\x00\x01\x08\x02\x00\x00\x00\x90wS\xde\x00\x00\x00"
|
||||||
|
b"\x0cIDATx\x9cc\xf8\x0f\x00\x00\x01\x01\x00\x05\x18\xd8N"
|
||||||
|
b"\x00\x00\x00\x00IEND\xaeB`\x82"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def test_docx_preserves_bold_color_specials_and_images(tmp_path: Path):
|
||||||
|
doc = Document()
|
||||||
|
doc.add_heading("1. Форматиране", level=1)
|
||||||
|
p = doc.add_paragraph()
|
||||||
|
r1 = p.add_run("Bold ")
|
||||||
|
r1.bold = True
|
||||||
|
r2 = p.add_run("и червено")
|
||||||
|
r2.font.color.rgb = RGBColor(0xC0, 0x00, 0x00)
|
||||||
|
p2 = doc.add_paragraph()
|
||||||
|
p2.add_run("Стрелка → и булет • в текста")
|
||||||
|
|
||||||
|
img_path = tmp_path / "pic.png"
|
||||||
|
png = _mini_png()
|
||||||
|
img_path.write_bytes(png)
|
||||||
|
try:
|
||||||
|
from io import BytesIO
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
assert Image.open(BytesIO(png)).size[0] >= 50
|
||||||
|
doc.add_picture(str(img_path))
|
||||||
|
expect_img = True
|
||||||
|
except Exception:
|
||||||
|
expect_img = False
|
||||||
|
|
||||||
|
out = tmp_path / "rich.docx"
|
||||||
|
doc.save(out)
|
||||||
|
|
||||||
|
sections = merge_preamble_sections(merge_short_sections(parse_docx(out)))
|
||||||
|
assert len(sections) >= 1
|
||||||
|
body = next(s for s in sections if "Bold" in (s.text or "") or (s.html_text and "Bold" in s.html_text))
|
||||||
|
assert "Bold" in body.text
|
||||||
|
assert "→" in body.text
|
||||||
|
assert "•" in body.text
|
||||||
|
assert "<b>" in (body.html_text or "")
|
||||||
|
assert "color:#C00000" in (body.html_text or "")
|
||||||
|
# Plain text без HTML тагове — за Claude / keywords
|
||||||
|
assert "<b>" not in body.text
|
||||||
|
if expect_img:
|
||||||
|
assert body.images
|
||||||
|
assert "[IMG:" in body.text
|
||||||
|
assert "[IMG:" in (body.html_text or "")
|
||||||
|
|
||||||
|
|
||||||
|
def test_html_preserves_bold_and_style_color(tmp_path: Path):
|
||||||
|
html = tmp_path / "rich.html"
|
||||||
|
html.write_text(
|
||||||
|
"""<!DOCTYPE html><html><body>
|
||||||
|
<h1>1. Цветове</h1>
|
||||||
|
<p>Обикновен <b>bold</b> и
|
||||||
|
<span style="color:#00aa00; font-size:20px">зелен</span>
|
||||||
|
плюс <font color="#0000cc">син</font>.</p>
|
||||||
|
</body></html>""",
|
||||||
|
encoding="utf-8",
|
||||||
|
)
|
||||||
|
sections = merge_preamble_sections(merge_short_sections(parse_html(html)))
|
||||||
|
assert sections
|
||||||
|
sec = sections[0]
|
||||||
|
assert "bold" in sec.text
|
||||||
|
html_body = sec.html_text or ""
|
||||||
|
assert "<b>" in html_body or "<strong>" in html_body
|
||||||
|
assert "color:#00aa00" in html_body or "color: #00aa00" in html_body
|
||||||
|
assert "font-size" not in html_body # декоративното се маха
|
||||||
|
assert 'color="#0000cc"' in html_body or "color:#0000cc" in html_body
|
||||||
|
|
||||||
|
|
||||||
|
def test_clean_text_strips_nul_keeps_bullets():
|
||||||
|
assert "\x00" not in clean_text("a\x00b\tc")
|
||||||
|
assert "•" in clean_text("елемент • едно")
|
||||||
|
assert "→" in clean_text("A → B")
|
||||||
Reference in New Issue
Block a user