Repository navigation
Expand file tree
/
Copy pathxhtml_style_postprocess.py
More file actions
477 lines (416 loc) · 23.2 KB
/
Copy pathxhtml_style_postprocess.py
File metadata and controls
477 lines (416 loc) · 23.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
# WriterAgent - AI Writing Assistant for LibreOffice
# Copyright (c) 2026 KeithCu
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# Pure string/CSS post-processing of LibreOffice "XHTML Writer File" output into the
# agent-facing semantic HTML described in docs/writer/html-style-model-plan.md (read path):
# - named paragraph styles -> data-lo-style="<compact token>" (no spaces: "Heading1")
# - direct character overrides (text-* synthetic classes) -> inline style="..."
# - synthetic paragraph-* autostyle classes (P1, P2, ...) -> name from FODT Pn->parent map
# (when supplied), else CSS fingerprint, else omitted
#
# v1 limitations (see docs/writer/html-style-model-plan.md#v1-limitations-shipped):
# - Whole-paragraph Para* overrides (center, margins, paragraph font, para colour) do not
# round-trip on WRITE; FODT recovers the base style NAME only. Char-level overrides via
# text-* spans survive (including run colour). Read REPORTS geometry, paragraph font, and
# fo:color as read-only data-lo-para, taken from the flat-ODF sidecar (see
# extract_autostyle_overrides_from_fodt): the XHTML CSS is flattened and cannot tell an
# override from an inherited value.
# - Inside <table>: paragraph-* classes stripped; no data-lo-style (cell styles out of scope).
# - Colliding compact tokens (two UNO names -> same token): token omitted on read.
#
# Post-v1: cached UNO ParaStyleName index to drop dual export and improve resolution.
# No UNO / no model access in this module — string pipeline only.
from __future__ import annotations
import html as _html
import re
from html.parser import HTMLParser
from typing import Any
# Block tags that carry a paragraph style (read) and that produce exactly one paragraph on
# import (write, where each consumes one positional data-lo-style slot). NOTE: <div> is
# deliberately EXCLUDED — LibreOffice never emits a paragraph style on a <div>, and on write a
# <div> is a transparent container (not its own paragraph). Treating it as a styleable block
# desynced styles (read/write asymmetry); keeping it out makes both paths agree.
BLOCK_TAGS = {"p", "h1", "h2", "h3", "h4", "h5", "h6", "li", "blockquote", "pre"}
_AUTOSTYLE_RE = re.compile(r"^P\d+$")
_STYLE_BLOCK_RE = re.compile(r"<style[^>]*>(.*?)</style>", re.DOTALL | re.IGNORECASE)
_RULE_RE = re.compile(r"\.([A-Za-z0-9_\-]+)\s*\{([^}]*)\}")
_BODY_RE = re.compile(r"<body[^>]*>(.*)</body>", re.DOTALL | re.IGNORECASE)
_PARA_CLASS_RE = re.compile(r"\bparagraph-[A-Za-z0-9_\-]+")
_TRAILING_EMPTY_P_RE = re.compile(
r"\s*<p\b[^>]*>(?:\s| | | | )*</p>\s*$", re.IGNORECASE
)
from plugin.framework.deal_shim import deal
@deal.post(lambda result: isinstance(result, str))
def decode_lo_css_class_suffix(suffix: str) -> str:
"""Reverse ODF URL-style encoding in a CSS class suffix (``Heading_20_1`` -> ``Heading 1``)."""
# re.sub on unbounded suffix hangs deep check.
# crosshair: off
# Plain str only — CrossHair LazyIntSymbolicStr is isinstance(str) but breaks re.sub.
if type(suffix) is not str:
return ""
return re.sub(r"_([0-9a-fA-F]{2})_", lambda m: chr(int(m.group(1), 16)), suffix)
@deal.post(lambda result: isinstance(result, str) and " " not in result)
def compact_lo_style_name(uno_name: str) -> str:
"""Agent-facing token: drop spaces (``Heading 1`` -> ``Heading1``)."""
# crosshair: off
if type(uno_name) is not str:
return ""
return uno_name.replace(" ", "")
_FODT_STYLE_RE = re.compile(r"<style:style\b([^>]*?)/?>", re.IGNORECASE)
@deal.post(lambda result: isinstance(result, dict))
def extract_autostyle_parents_from_fodt(fodt: str) -> dict[str, str]:
"""Map automatic paragraph style name (``P1``, ``P2``, ...) -> its parent named style, from a
flat ODF (.fodt) export. The XHTML export flattens this parent away, so an autostyle's CSS
often matches no named rule (the common case after a StarWriter HTML import); the flat ODF
keeps ``style:parent-style-name``, and the automatic style name (``Pn``) is identical to the
XHTML ``paragraph-Pn`` class suffix — a reliable, order-independent join. Parent values may
still be ODF-encoded (``Text_20_body``); decode at use via ``decode_lo_css_class_suffix``.
"""
# FODT regex finditer on unbounded XML hangs deep check.
# crosshair: off
out = {}
text = fodt if type(fodt) is str else ""
for m in _FODT_STYLE_RE.finditer(text):
attrs = m.group(1)
fam = re.search(r'style:family="([^"]*)"', attrs)
if not fam or fam.group(1) != "paragraph":
continue
name = re.search(r'style:name="([^"]*)"', attrs)
parent = re.search(r'style:parent-style-name="([^"]*)"', attrs)
if name and parent and _AUTOSTYLE_RE.match(name.group(1)):
# XML attribute values are entity-escaped (a style named "A & B" -> "A & B").
out[name.group(1)] = _html.unescape(parent.group(1))
return out
def _clean_decl_ordered(decl: str) -> str:
"""Cleaned declaration, original order preserved, no trailing ``;`` (for inlining)."""
# crosshair: off
# CSS decl tokenization (cover-all 35546602462: ~18m xhtml module). Doable later with DEAL_MAX_HTML_CHUNK dual-profile.
parts = [p.strip() for p in decl.split(";") if p.strip()]
return "; ".join(parts)
def _normalize_decl(decl: str) -> str:
"""Order-independent fingerprint of a declaration set (for autostyle matching)."""
# crosshair: off
# CSS decl normalize/sort (cover-all 35546602462). Doable later with closed decl alphabet.
parts = [p.strip() for p in decl.split(";") if p.strip()]
return ";".join(sorted(parts))
# Direct paragraph overrides surfaced to the agent as read-only ``data-lo-para``. v1 dropped them
# entirely, so an agent could not tell an indented block quote from body text, nor verify an
# indent it had just set.
#
# Read from the flat-ODF sidecar, NOT from the XHTML CSS: an autostyle's XHTML rule is FLATTENED
# (base style + override merged, with no rule emitted for the base at all), so it cannot say which
# values were hand-set — reporting it would present a style's own indent as a direct override. The
# .fodt automatic style holds exactly the overrides and nothing inherited.
#
# Geometry, paragraph-level font, and paragraph colour are reported: a .docx reused as a model
# carries its font as a whole-paragraph override, which is precisely what makes an applied style
# look like it did nothing. fo:color was missing here, so Issue-1-style center+red never showed
# red in data-lo-para (char spans could still carry colour).
#
# INFORMATIONAL only — the write path ignores ``data-lo-para`` (it still cannot restore Para* when
# applying a named style), which is why it is not emitted as ``style=""``: that would round-trip
# back and be silently swallowed. ODF attribute -> CSS name, in emit order.
_FODT_OVERRIDE_ATTRS = (
("fo:margin-left", "margin-left"),
("fo:margin-right", "margin-right"),
("fo:margin-top", "margin-top"),
("fo:margin-bottom", "margin-bottom"),
("fo:text-indent", "text-indent"),
("fo:text-align", "text-align"),
("fo:line-height", "line-height"),
("style:font-name", "font-family"),
("fo:font-family", "font-family"),
("fo:font-size", "font-size"),
("fo:font-weight", "font-weight"),
("fo:font-style", "font-style"),
("fo:color", "color"),
)
# ODF writes alignment relative to writing direction; the agent reads CSS.
_ODF_TEXT_ALIGN = {"start": "left", "end": "right"}
_FODT_PARA_STYLE_RE = re.compile(
r'<style:style\b([^>]*?)style:family="paragraph"([^>]*?)(?:/>|>(.*?)</style:style>)',
re.IGNORECASE | re.DOTALL)
def _fodt_override_css(style_block: str) -> str:
"""CSS-ish ``prop: value`` list for one flat-ODF automatic paragraph style's own attributes."""
# crosshair: off
# FODT style-block regex/string walk (cover-all 35546602462). Doable later with DEAL_MAX_HTML_CHUNK.
parts = []
seen = set()
for odf_attr, css_name in _FODT_OVERRIDE_ATTRS:
if css_name in seen:
continue
m = re.search(r'\b%s="([^"]*)"' % re.escape(odf_attr), style_block)
if not m:
continue
value = _html.unescape(m.group(1)).strip()
if not value:
continue
if css_name == "text-align":
value = _ODF_TEXT_ALIGN.get(value, value)
parts.append("%s:%s" % (css_name, value))
seen.add(css_name)
return "; ".join(parts)
def extract_autostyle_overrides_from_fodt(fodt: str) -> dict[str, str]:
"""Map automatic paragraph style name (``P1``, ...) -> its DIRECT overrides as CSS text.
Companion to :func:`extract_autostyle_parents_from_fodt`, parsed from the same export. An
automatic style in flat ODF lists only what was set on top of its parent, so what comes back
is exactly the hand-set formatting — the thing the XHTML projection cannot express.
"""
# crosshair: off
# FODT override extraction over unbounded XML (cover-all 35546602462). Doable later with DEAL_MAX_HTML_CHUNK.
out = {}
text = fodt if type(fodt) is str else ""
for m in _FODT_PARA_STYLE_RE.finditer(text):
head = (m.group(1) or "") + (m.group(2) or "")
name = re.search(r'style:name="([^"]*)"', head)
if not name or not _AUTOSTYLE_RE.match(name.group(1)):
continue
css = _fodt_override_css(head + (m.group(3) or ""))
if css:
out[name.group(1)] = css
return out
@deal.post(lambda result: isinstance(result, tuple) and len(result) == 2 and isinstance(result[0], dict) and isinstance(result[1], dict))
def parse_style_block(xhtml: str) -> tuple[dict[str, str], dict[str, str]]:
"""Return ``(raw_map, norm_map)`` for every class rule in the ``<style>`` block(s).
``raw_map[class] = "decl; decl"`` (order preserved) is used to inline char overrides;
``norm_map[class] = "decl;decl"`` (sorted) is used to fingerprint autostyles.
"""
# Style-block regex on unbounded XHTML hangs deep check.
# crosshair: off
raw_map = {}
norm_map = {}
text = xhtml if type(xhtml) is str else ""
for block in _STYLE_BLOCK_RE.findall(text):
for name, decl in _RULE_RE.findall(block):
raw_map[name] = _clean_decl_ordered(decl)
norm_map[name] = _normalize_decl(decl)
return raw_map, norm_map
def _strip_body(xhtml: str) -> str:
"""Return the inner HTML of ``<body>`` (or the input unchanged if there is no body)."""
# crosshair: off
# XHTML body slice (cover-all 35546602462). Doable later with DEAL_MAX_HTML_CHUNK dual-profile.
m = _BODY_RE.search(xhtml or "")
return m.group(1).strip() if m else (xhtml or "")
def _drop_trailing_empty_paragraphs(html: str) -> str:
"""Drop whitespace/`` ``-only ``<p>`` blocks at the very end (LO export ghost paras)."""
# crosshair: off
# HTML trailing-empty scan (cover-all 35546602462: top xhtml sink). Doable later with DEAL_MAX_HTML_CHUNK.
while True:
new = _TRAILING_EMPTY_P_RE.sub("", html)
if new == html:
return html
html = new
def _inject_attr(start_tag: str, attr: str) -> str:
"""Insert *attr* (e.g. ``' data-lo-style="X"'``) just before the closing ``>`` of a tag."""
# crosshair: off
# start-tag attribute inject (cover-all 35546602462). Doable later with closed tag alphabet.
if start_tag.endswith("/>"):
return start_tag[:-2].rstrip() + attr + "/>"
if start_tag.endswith(">"):
return start_tag[:-1].rstrip() + attr + ">"
return start_tag + attr
def _attr_value(attrs: list[tuple[str, str | None]], key: str) -> str:
# crosshair: off
# attr list scan (cover-all 35546602462). Doable later with closed attrs domain.
for k, v in attrs:
if k == key:
return v or ""
return ""
class _SemanticTransformer(HTMLParser):
"""Rewrite LO XHTML body into the agent-facing semantic HTML.
Top-level paragraph blocks get ``data-lo-style="<compact token>"`` (named styles only;
autostyles resolved by CSS fingerprint or omitted). Char overrides (``text-*`` classes)
are inlined as ``style="..."``. Inside ``<table>`` the synthetic ``paragraph-*`` class is
stripped (no leak) but no ``data-lo-style`` is emitted — cell styles are out of scope for
v1 round-trip. Tags are re-emitted verbatim via ``get_starttag_text()``; only the
attributes we change are rewritten with string ops.
"""
_raw: dict[str, str]
_autostyle_parents: dict[str, str]
_autostyle_overrides: dict[str, str]
_colliding_tokens: set[str]
_norm: dict[str, str]
_table_depth: int
def __init__(self, raw_map: dict[str, str], norm_map: dict[str, str], autostyle_parents: dict[str, str] | None = None, autostyle_overrides: dict[str, str] | None = None) -> None:
# crosshair: off
# HTMLParser transformer state (cover-all 35546602462). Doable later with closed style maps.
super().__init__(convert_charrefs=False)
self._raw = raw_map
self._autostyle_parents = autostyle_parents or {}
self._autostyle_overrides = autostyle_overrides or {}
# Map a named paragraph rule's fingerprint -> list of its compact tokens.
self._named_fingerprint: dict[str, list[str]] = {}
# Map a compact token -> set of distinct UNO names that compact to it. When more than
# one named style produces the same token (e.g. "Heading 1" and a literal "Heading1"),
# the token is AMBIGUOUS and must not be emitted — see _colliding_tokens below.
token_names: dict[str, set[str]] = {}
for cls, norm in norm_map.items():
if not cls.startswith("paragraph-"):
continue
suffix = cls[len("paragraph-"):]
if _AUTOSTYLE_RE.match(suffix):
continue
name = decode_lo_css_class_suffix(suffix)
token = compact_lo_style_name(name)
self._named_fingerprint.setdefault(norm, []).append(token)
token_names.setdefault(token, set()).add(name)
# Tokens produced by >1 distinct named style collide: the write path could not tell
# them apart, so we omit (write treats a missing token as the default body style)
# rather than emit a token that would silently resolve to the wrong style.
self._colliding_tokens = {t for t, names in token_names.items() if len(names) > 1}
self._norm = norm_map
self._table_depth = 0
self._out: list[str] = []
def _paragraph_token(self, suffix: str) -> str | None:
"""Compact ``data-lo-style`` token for a ``paragraph-<suffix>`` class, or ``None``.
For autostyles (``paragraph-Pn``) the flat-ODF parent map (when supplied) is
authoritative: the XHTML export flattens the parent away, so the autostyle CSS often
matches no named rule (the common case after a StarWriter HTML import), but the .fodt
export keeps the parent and ``Pn`` is the same name in both exports — so the real style
name is recovered. CSS fingerprint is the fallback when there is no FODT map.
KNOWN LIMITATION (v1): the whole-paragraph DIRECT *overrides* themselves (alignment,
margins, whole-paragraph colour baked into the autostyle CSS) are still dropped — only
the style NAME is recovered, and character-level overrides survive via inline ``text-*``
spans. Whole-paragraph Para* overrides do not round-trip (the write path cannot restore
Para* when applying a named style).
"""
# crosshair: off
# paragraph class rewrite (cover-all 35546602462). Doable later with closed class alphabet.
if _AUTOSTYLE_RE.match(suffix):
# Authoritative: the flat-ODF parent of this autostyle (Pn -> named base).
parent = self._autostyle_parents.get(suffix)
if parent:
token = compact_lo_style_name(decode_lo_css_class_suffix(parent))
if token not in self._colliding_tokens:
return token
# Fallback: resolve by CSS fingerprint against named rules.
norm = self._norm.get("paragraph-" + suffix)
matches = self._named_fingerprint.get(norm) if norm else None
if matches and len(matches) == 1 and matches[0] not in self._colliding_tokens:
return matches[0]
return None # not uniquely resolvable -> omit (write treats absence as body)
token = compact_lo_style_name(decode_lo_css_class_suffix(suffix))
if token in self._colliding_tokens:
return None # ambiguous compact token -> omit (Issue 2)
return token
def _rewrite_block(self, raw: str, attrs: list[tuple[str, str | None]]) -> str:
# crosshair: off
# block tag rewrite (cover-all 35546602462). Doable later with closed tag alphabet.
classes = _attr_value(attrs, "class")
para_classes = _PARA_CLASS_RE.findall(classes)
token = None
para_css = ""
if para_classes and self._table_depth == 0:
suffix = para_classes[0][len("paragraph-"):]
token = self._paragraph_token(suffix)
# Only an autostyle (Pn) carries direct overrides — a plain named-style class
# describes the style itself — and only the flat-ODF sidecar can say which values
# those are, so without it nothing is reported rather than a guess.
if _AUTOSTYLE_RE.match(suffix):
para_css = self._autostyle_overrides.get(suffix, "")
# Strip every paragraph-* class (top-level or cell) so it never leaks to the agent.
raw = _strip_class_names(raw, _PARA_CLASS_RE)
if token:
raw = _inject_attr(raw, ' data-lo-style="%s"' % token)
if para_css:
# ODF values are unescaped before this (a font name can contain "). Without
# quote=True the injected attribute closes early and the tag is not well-formed.
raw = _inject_attr(raw, ' data-lo-para="%s"' % _html.escape(para_css, quote=True))
return raw
def _rewrite_span(self, raw: str, attrs: list[tuple[str, str | None]]) -> str:
# crosshair: off
# span rewrite (cover-all 35546602462). Doable later with closed tag alphabet.
classes = _attr_value(attrs, "class")
names = classes.split()
text_names = [c for c in names if c.startswith("text-")]
if not text_names:
return raw
css_parts = [self._raw[c] for c in text_names if c in self._raw]
remaining = [c for c in names if not c.startswith("text-")]
raw = re.sub(r'\s+class="[^"]*"', "", raw)
raw = re.sub(r"\s+class='[^']*'", "", raw)
ins = ""
if css_parts:
ins += ' style="%s"' % "; ".join(css_parts)
if remaining:
ins += ' class="%s"' % " ".join(remaining)
return _inject_attr(raw, ins) if ins else raw
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
# crosshair: off
# HTMLParser starttag (cover-all 35546602462). Engine-hostile over symbolic HTML.
raw = self.get_starttag_text() or ("<%s>" % tag)
t = tag.lower()
if t == "table":
self._table_depth += 1
self._out.append(raw)
elif t in BLOCK_TAGS or (t == "div" and _PARA_CLASS_RE.search(_attr_value(attrs, "class"))):
# A paragraph that holds a frame (an image) is exported as <div class="paragraph-X">:
# it is one Writer paragraph, so it gets its data-lo-style like a <p>. Without it the
# write side's positional styles skipped that paragraph and the next ones shifted.
self._out.append(self._rewrite_block(raw, attrs))
elif t == "span":
self._out.append(self._rewrite_span(raw, attrs))
else:
self._out.append(raw)
def handle_startendtag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
# crosshair: off
# HTMLParser startendtag (cover-all 35546602462). Engine-hostile over symbolic HTML.
self._out.append(self.get_starttag_text() or ("<%s/>" % tag))
def handle_endtag(self, tag: str) -> None:
# crosshair: off
# HTMLParser endtag (cover-all 35546602462). Engine-hostile over symbolic HTML.
if tag.lower() == "table" and self._table_depth > 0:
self._table_depth -= 1
self._out.append("</%s>" % tag)
def handle_data(self, data: str) -> None:
# crosshair: off
# HTMLParser data (cover-all 35546602462). Engine-hostile over symbolic HTML.
self._out.append(data)
def handle_entityref(self, name: str) -> None:
# crosshair: off
# HTMLParser entityref (cover-all 35546602462). Engine-hostile over symbolic HTML.
self._out.append("&%s;" % name)
def handle_charref(self, name: str) -> None:
# crosshair: off
# HTMLParser charref (cover-all 35546602462). Engine-hostile over symbolic HTML.
self._out.append("&#%s;" % name)
def handle_comment(self, data: str) -> None:
# crosshair: off
# HTMLParser comment (cover-all 35546602462). Engine-hostile over symbolic HTML.
self._out.append("<!--%s-->" % data)
def result(self) -> str:
# crosshair: off
# HTMLParser result join (cover-all 35546602462). Doable later with DEAL_MAX_HTML_CHUNK.
return "".join(self._out)
def _strip_class_names(start_tag: str, name_re: re.Pattern[str]) -> str:
"""Remove class names matching *name_re* from a start tag; drop the attr if it empties."""
# crosshair: off
# class= regex rewrite (cover-all 35546602462: top xhtml sink). Doable later with DEAL_MAX_HTML_CHUNK.
def _sub(m: Any) -> str:
kept = [c for c in m.group(2).split() if not name_re.fullmatch(c)]
return (" class=%s%s%s" % (m.group(1), " ".join(kept), m.group(1))) if kept else ""
return re.sub(r'\s+class=(["\'])(.*?)\1', _sub, start_tag)
def xhtml_to_semantic_html(full_xhtml: str, autostyle_parents: dict[str, str] | None = None, autostyle_overrides: dict[str, str] | None = None) -> str:
"""Convert raw LO ``XHTML Writer File`` output to the agent-facing semantic HTML.
Single entry point: parse the ``<style>`` block, extract the body, inline char overrides,
emit ``data-lo-style`` compact tokens for named paragraph styles, drop trailing ghost
paragraphs. Pure string work — fully unit-testable without LibreOffice.
*autostyle_parents* — ``{Pn: parent_style_name}`` from ``extract_autostyle_parents_from_fodt``
on a paired flat-ODF export. Joined by autostyle name (``paragraph-Pn`` suffix), not block
index. Recovers the base style **name** after StarWriter write→read when CSS fingerprint
fails; does not recover whole-paragraph Para* overrides (v1 limitation — see plan doc).
"""
# crosshair: off
# HTMLParser over unbounded XHTML (cover-all 35546602462: ~18m module). Doable later with UNDER_CROSSHAIR dual-profile + DEAL_MAX_HTML_CHUNK (pytest fixtures exceed 512).
raw_map, norm_map = parse_style_block(full_xhtml)
body = _strip_body(full_xhtml)
tr = _SemanticTransformer(raw_map, norm_map, autostyle_parents, autostyle_overrides)
tr.feed(body)
tr.close()
out = _drop_trailing_empty_paragraphs(tr.result())
return out.strip()