Add a text widget (rich text: bold/italic/underline, per-run color/highlight)
Build and push server image / test (push) Successful in 27s
Build and push server image / build-and-push (push) Successful in 1m59s
Build and push server image / deploy (push) Successful in 1m9s

A new self-contained widget type showing user-authored rich text -- no
live upstream to poll, like the static image widget, just word-wrapped
styled text instead of an uploaded image.

The dialog's contenteditable HTML is never stored or replayed as HTML:
app/text_content.py parses it server-side (on save) into a plain
paragraphs-of-styled-runs structure -- the actual sanitization
boundary, since raw HTML never round-trips back into any browser DOM
(the dialog rebuilds its editor from that same JSON via
createElement/textContent). app/widgets/text.py renders it with a
custom word-wrap/shrink-to-fit layout, using real vendored font weights
(app/fonts/NotoSans-{Regular,Bold,Italic,BoldItalic}.ttf, OFL-licensed
like the emoji fonts already there) rather than every other widget's
single ImageFont.load_default() -- the one widget type where that
distinction matters.
This commit is contained in:
Thomas Faour
2026-07-25 14:21:52 +00:00
parent f1fda9bdee
commit 3735c5bfa7
23 changed files with 1062 additions and 13 deletions
+169
View File
@@ -0,0 +1,169 @@
"""Parses a contenteditable div's serialized innerHTML (widget_dialog_
text.js's POSTed content_html) into a plain, storage-safe run structure
-- list of paragraphs, each a list of {"text", "bold", "italic",
"underline", "color", "bg"} runs -- for the text widget (see
models.TextWidgetConfig, app/widgets/text.py).
This is the sanitization boundary the checklist's stored-XSS note
(CLAUDE.md, another linked user could have set this) is about: raw HTML
never round-trips back into any browser DOM. Only text content and a
small fixed set of style flags survive parsing; every tag, attribute,
and CSS property not explicitly recognized below is simply discarded --
there's no allowlist-of-tags-to-keep-as-HTML step where something could
slip through unescaped, because nothing is ever re-emitted as HTML at
all. The dialog reconstructs its editor from this same run structure via
safe DOM calls (createElement/textContent), never innerHTML."""
from __future__ import annotations
import re
from html.parser import HTMLParser
# Generous ceilings, not exact UX limits -- just stop a direct API call
# (bypassing the dialog's own textarea-ish size) from storing something
# pathologically large. MAX_INPUT_CHARS bounds parse work; MAX_TOTAL_CHARS
# bounds what's actually kept (a widget's on-panel region is a few
# hundred pixels -- there is no legible use for more than a few thousand
# characters of body text there).
MAX_INPUT_CHARS = 200_000
MAX_TOTAL_CHARS = 4_000
_BASE_STYLE = {"bold": False, "italic": False, "underline": False, "color": None, "bg": None}
_BLOCK_TAGS = {"div", "p", "li"}
_VOID_TAGS = {"br"}
_HEX6 = re.compile(r"^#([0-9a-fA-F]{6})$")
_HEX3 = re.compile(r"^#([0-9a-fA-F]{3})$")
_RGB = re.compile(r"^rgba?\(\s*(\d{1,3})\s*,\s*(\d{1,3})\s*,\s*(\d{1,3})\s*(?:,\s*[\d.]+\s*)?\)$")
_STYLE_PROP = re.compile(r"([a-zA-Z-]+)\s*:\s*([^;]+)")
_BOLD_WEIGHTS = {"bold", "bolder", "600", "700", "800", "900"}
def _normalize_color(value: str) -> str | None:
""""#1a2b3c" / "#abc" / "rgb(26, 43, 60)" -> "#1a2b3c". Anything else
(a CSS named color, "transparent", garbage) -> None, i.e. dropped --
this is the one place an arbitrary style-attribute string could try
to smuggle something through, so it's a strict allowlist match, not
a best-effort parse."""
value = value.strip()
m = _HEX6.match(value)
if m:
return "#" + m.group(1).lower()
m = _HEX3.match(value)
if m:
return "#" + "".join(c * 2 for c in m.group(1)).lower()
m = _RGB.match(value)
if m:
r, g, b = (max(0, min(255, int(x))) for x in m.groups())
return f"#{r:02x}{g:02x}{b:02x}"
return None
class _RichTextParser(HTMLParser):
def __init__(self) -> None:
super().__init__(convert_charrefs=True)
self.paragraphs: list[list[dict]] = [[]]
self._style_stack: list[dict] = [_BASE_STYLE]
self._at_line_start = True
def _break(self, tag: str) -> None:
# Coalesces contenteditable's per-line block wrapping (Chrome
# wraps every line in its own <div> even without a deliberate
# blank line) down to one paragraph break per actual line gap,
# while still letting an explicit <br> when already at a fresh
# line start (Chrome's "<div><br></div>" idiom for a blank line,
# or a genuine double Shift+Enter) add a real blank paragraph.
if not self._at_line_start:
self.paragraphs.append([])
self._at_line_start = True
elif tag == "br":
self.paragraphs.append([])
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
if tag in _BLOCK_TAGS or tag in _VOID_TAGS:
self._break(tag)
if tag in _VOID_TAGS:
return
style = dict(self._style_stack[-1])
attrs_dict = {k: v for k, v in attrs if v is not None}
if tag in ("b", "strong"):
style["bold"] = True
elif tag in ("i", "em"):
style["italic"] = True
elif tag == "u":
style["underline"] = True
elif tag == "font":
color = _normalize_color(attrs_dict.get("color", ""))
if color:
style["color"] = color
elif tag == "span":
for prop, val in _STYLE_PROP.findall(attrs_dict.get("style", "")):
prop = prop.strip().lower()
val = val.strip()
if prop == "color":
color = _normalize_color(val)
if color:
style["color"] = color
elif prop == "background-color":
color = _normalize_color(val)
if color:
style["bg"] = color
elif prop == "font-weight" and val.lower() in _BOLD_WEIGHTS:
style["bold"] = True
elif prop == "font-style" and val.lower() == "italic":
style["italic"] = True
elif prop == "text-decoration" and "underline" in val.lower():
style["underline"] = True
# Pushed for every non-void tag, including ones with no
# recognized style effect (script/a/img/...) -- keeps push/pop
# balanced against handle_endtag regardless of tag, without
# needing to track which tags actually pushed something.
self._style_stack.append(style)
def handle_endtag(self, tag: str) -> None:
if tag in _VOID_TAGS:
return
if len(self._style_stack) > 1:
self._style_stack.pop()
def handle_data(self, data: str) -> None:
if not data:
return
style = self._style_stack[-1]
self.paragraphs[-1].append({"text": data, **style})
if data.strip():
self._at_line_start = False
def parse_rich_text(html: str) -> list[list[dict]]:
"""The sanitization entry point -- see module docstring. Always
returns a valid (possibly all-empty) paragraphs structure, never
raises for malformed markup (html.parser tolerates unclosed/
mismatched tags; handle_endtag's length guard tolerates an
over-popped stack)."""
parser = _RichTextParser()
parser.feed(html[:MAX_INPUT_CHARS])
parser.close()
paragraphs = parser.paragraphs
total = 0
truncated: list[list[dict]] = []
for para in paragraphs:
new_para: list[dict] = []
for run in para:
remaining = MAX_TOTAL_CHARS - total
if remaining <= 0:
break
text = run["text"][:remaining]
total += len(text)
new_para.append({**run, "text": text})
truncated.append(new_para)
if total >= MAX_TOTAL_CHARS:
break
return truncated
def has_text(paragraphs: list[list[dict]] | None) -> bool:
if not paragraphs:
return False
return any(run["text"].strip() for para in paragraphs for run in para)