"""Small allow-list sanitizer for HTML rendered inside the BMC Hub UI.""" import html from html.parser import HTMLParser from typing import Optional class _SafeHtmlSanitizer(HTMLParser): _ALLOWED_TAGS = { "a", "b", "strong", "i", "em", "u", "s", "p", "div", "span", "br", "hr", "blockquote", "pre", "code", "ul", "ol", "li", "h1", "h2", "h3", "h4", "h5", "h6", "table", "thead", "tbody", "tfoot", "tr", "th", "td", "caption", } _VOID_TAGS = {"br", "hr"} _DROP_WITH_CONTENT = {"script", "style", "iframe", "object", "embed", "svg", "math", "head"} _ALLOWED_ATTRS = { "a": {"href", "title"}, "th": {"colspan", "rowspan"}, "td": {"colspan", "rowspan"}, } def __init__(self): super().__init__(convert_charrefs=True) self._parts: list[str] = [] self._drop_depth = 0 def handle_starttag(self, tag, attrs): tag = str(tag or "").lower() if tag in self._DROP_WITH_CONTENT: self._drop_depth += 1 return if self._drop_depth or tag not in self._ALLOWED_TAGS: return safe_attrs: list[str] = [] for key, value in attrs or []: key = str(key or "").lower() if key not in self._ALLOWED_ATTRS.get(tag, set()): continue value = str(value or "").strip() if key == "href": normalized = value.lower() if not normalized.startswith(("https://", "http://", "mailto:", "tel:", "/")): continue if key in {"colspan", "rowspan"}: try: number = int(value) except (TypeError, ValueError): continue if number < 1 or number > 100: continue value = str(number) safe_attrs.append(f'{key}="{html.escape(value, quote=True)}"') attrs_html = f" {' '.join(safe_attrs)}" if safe_attrs else "" if tag == "a": attrs_html += ' target="_blank" rel="noopener noreferrer"' self._parts.append(f"<{tag}{attrs_html}>") def handle_startendtag(self, tag, attrs): if str(tag or "").lower() in self._DROP_WITH_CONTENT: return self.handle_starttag(tag, attrs) def handle_endtag(self, tag): tag = str(tag or "").lower() if tag in self._DROP_WITH_CONTENT: self._drop_depth = max(0, self._drop_depth - 1) return if self._drop_depth or tag not in self._ALLOWED_TAGS or tag in self._VOID_TAGS: return self._parts.append(f"") def handle_data(self, data): if not self._drop_depth: self._parts.append(html.escape(data or "")) def handle_entityref(self, name): if not self._drop_depth: self._parts.append(f"&{name};") def handle_charref(self, name): if not self._drop_depth: self._parts.append(f"&#{name};") def get_html(self) -> str: return "".join(self._parts).strip() def sanitize_safe_html(value: Optional[str]) -> str: """Return safe renderable HTML while preserving ordinary plain text.""" raw = str(value or "").strip() if not raw: return "" if "<" not in raw and ">" not in raw: return html.escape(raw) sanitizer = _SafeHtmlSanitizer() try: sanitizer.feed(raw) sanitizer.close() return sanitizer.get_html() except Exception: return html.escape(raw)