Packages
phoenix_kit
1.7.201
1.7.208
1.7.207
1.7.206
1.7.205
1.7.204
1.7.203
1.7.202
1.7.201
1.7.200
1.7.199
1.7.198
1.7.197
1.7.196
1.7.194
1.7.193
1.7.192
1.7.191
1.7.190
1.7.189
1.7.187
1.7.186
1.7.185
1.7.184
1.7.183
1.7.182
1.7.181
1.7.180
1.7.179
1.7.178
1.7.177
1.7.176
1.7.175
1.7.174
1.7.173
1.7.172
1.7.171
1.7.170
1.7.169
1.7.168
1.7.167
1.7.166
1.7.165
1.7.164
1.7.162
1.7.161
1.7.160
1.7.159
1.7.157
1.7.156
1.7.155
1.7.154
1.7.153
1.7.152
1.7.151
1.7.150
1.7.149
1.7.146
1.7.145
1.7.144
1.7.143
1.7.138
1.7.133
1.7.132
1.7.131
1.7.130
1.7.128
1.7.126
1.7.125
1.7.121
1.7.120
1.7.119
1.7.118
1.7.117
1.7.116
1.7.115
1.7.114
1.7.113
1.7.112
1.7.111
1.7.110
1.7.109
1.7.108
1.7.107
1.7.106
1.7.105
1.7.104
1.7.103
1.7.102
1.7.101
1.7.100
1.7.99
1.7.98
1.7.97
1.7.96
1.7.95
1.7.94
1.7.93
1.7.92
1.7.91
1.7.90
1.7.89
1.7.88
1.7.87
1.7.86
1.7.85
1.7.84
1.7.83
1.7.82
1.7.81
1.7.80
1.7.79
1.7.78
1.7.77
1.7.76
1.7.75
1.7.74
1.7.71
1.7.70
1.7.69
1.7.66
1.7.65
1.7.64
1.7.63
1.7.62
1.7.61
1.7.59
1.7.58
1.7.57
1.7.56
1.7.55
1.7.54
1.7.53
1.7.52
1.7.51
1.7.49
1.7.44
1.7.43
1.7.42
1.7.41
1.7.39
1.7.38
1.7.37
1.7.36
1.7.34
1.7.33
1.7.31
1.7.30
1.7.29
1.7.28
1.7.27
1.7.26
1.7.25
1.7.24
1.7.23
1.7.22
1.7.21
1.7.20
1.7.19
1.7.18
1.7.17
1.7.16
1.7.15
1.7.14
1.7.13
1.7.12
1.7.11
1.7.10
1.7.9
1.7.8
1.7.7
1.7.6
1.7.5
1.7.4
1.7.3
1.7.2
1.7.1
1.7.0
1.6.20
1.6.19
1.6.18
1.6.17
1.6.16
1.6.15
1.6.14
1.6.13
1.6.12
1.6.11
1.6.10
1.6.9
1.6.8
1.6.7
1.6.6
1.6.5
1.6.4
1.6.3
1.5.2
1.5.1
1.5.0
1.4.9
1.4.8
1.4.7
1.4.6
1.4.5
1.4.4
1.4.3
1.4.2
1.4.1
1.4.0
1.3.2
1.3.1
1.3.0
1.2.10
1.2.9
1.2.8
1.2.7
1.2.5
1.2.4
1.2.2
1.2.1
1.2.0
1.1.0
1.0.0
A foundation for building Elixir Phoenix apps — SaaS, social networks, ERP systems, marketplaces, and more
Current section
Files
Jump to
Current section
Files
lib/phoenix_kit/utils/html_sanitizer.ex
defmodule PhoenixKit.Utils.HtmlSanitizer do
@moduledoc """
HTML sanitization for rich text content in entities.
This module provides basic HTML sanitization to prevent XSS attacks
while allowing safe HTML tags commonly used in rich text editors.
## Allowed Tags
The following tags are allowed:
- Block elements: p, div, br, hr, h1-h6, blockquote, pre, code
- Inline elements: span, strong, b, em, i, u, s, a, sub, sup, mark
- Lists: ul, ol, li
- Tables: table, thead, tbody, tr, th, td
- Media placeholders: img (with src validation)
## Removed Content
The following are stripped completely:
- script tags and content
- style tags and content
- event handlers (onclick, onerror, etc.)
- javascript: and data: URLs
- iframe, object, embed tags
## Usage
iex> PhoenixKit.Utils.HtmlSanitizer.sanitize("<p>Hello</p><script>alert('xss')</script>")
"<p>Hello</p>"
iex> PhoenixKit.Utils.HtmlSanitizer.sanitize("<a href=\"javascript:alert('xss')\">Click</a>")
"<a>Click</a>"
"""
# Note: These are documented for reference. The current simple implementation
# strips dangerous content rather than whitelisting allowed tags.
# A more complete implementation using a library like HtmlSanitizeEx would use these.
#
# Allowed tags:
# p div br hr h1-h6 blockquote pre code
# span strong b em i u s a sub sup mark
# ul ol li table thead tbody tr th td img
#
# Allowed attributes:
# a: href title target rel
# img: src alt title width height
# td/th: colspan rowspan
# all: class id
@doc """
Sanitizes HTML content by removing dangerous elements and attributes.
Returns sanitized HTML string that is safe to render.
## Parameters
- `html` - The HTML string to sanitize
## Examples
iex> PhoenixKit.Utils.HtmlSanitizer.sanitize("<p onclick=\"alert('xss')\">Hello</p>")
"<p>Hello</p>"
"""
def sanitize(nil), do: nil
def sanitize(""), do: ""
def sanitize(html) when is_binary(html) do
html
|> remove_dangerous_patterns()
|> sanitize_urls()
|> String.trim()
end
def sanitize(other), do: other
@doc """
Sanitizes all rich_text fields in an entity data map.
Takes entity field definitions and data, returns data with all
rich_text fields sanitized.
## Parameters
- `fields_definition` - List of field definition maps
- `data` - Map of field key => value
## Examples
iex> fields = [%{"type" => "rich_text", "key" => "content"}]
iex> data = %{"content" => "<script>alert('xss')</script><p>Hello</p>"}
iex> PhoenixKit.Utils.HtmlSanitizer.sanitize_rich_text_fields(fields, data)
%{"content" => "<p>Hello</p>"}
"""
def sanitize_rich_text_fields(fields_definition, data)
when is_list(fields_definition) and is_map(data) do
rich_text_keys =
fields_definition
|> Enum.filter(fn field -> field["type"] == "rich_text" end)
|> Enum.map(fn field -> field["key"] end)
Enum.reduce(rich_text_keys, data, fn key, acc ->
case Map.get(acc, key) do
nil -> acc
value -> Map.put(acc, key, sanitize(value))
end
end)
end
def sanitize_rich_text_fields(_fields, data), do: data
# Private functions
defp remove_dangerous_patterns(html) do
dangerous_patterns = [
# Script tags with content
~r/<script\b[^>]*>[\s\S]*?<\/script>/i,
# Style tags with content
~r/<style\b[^>]*>[\s\S]*?<\/style>/i,
# Event handlers
~r/\s+on\w+\s*=\s*["'][^"']*["']/i,
~r/\s+on\w+\s*=\s*[^\s>]+/i,
# Dangerous tags
~r/<\s*(iframe|object|embed|form|input|button|meta|link|base)\b[^>]*>/i,
~r/<\/\s*(iframe|object|embed|form|input|button|meta|link|base)\s*>/i
]
Enum.reduce(dangerous_patterns, html, fn pattern, acc ->
Regex.replace(pattern, acc, "")
end)
end
# Schemes a link/media URL may use; anything else (or an unlisted scheme) is
# dropped. Relative/fragment/query URLs (no scheme) are allowed.
@allowed_schemes ~w(http https mailto tel)
@dangerous_scheme ~r/^(?:javascript|vbscript|data|file|blob):/
@scheme ~r/^[a-z][a-z0-9+.\-]*:/i
# A few named entities whose decoded form matters for scheme detection
# (a browser decodes `javascript:alert(1)` before dispatching).
@named_entities %{
"tab" => "\t",
"newline" => "\n",
"colon" => ":",
"sol" => "/",
"lpar" => "(",
"rpar" => ")",
"num" => "#"
}
# Sanitize `href`/`src` URLs with an ALLOWLIST over a normalized value, not a
# scheme blacklist. MDEx renders raw HTML (`unsafe: true`) and the browser
# decodes entities + ignores whitespace/control chars in the scheme, so a
# blacklist is trivially bypassed (`javascript:`, `java	script:`,
# `java\tscript:`). We decode entities and strip those chars BEFORE checking,
# and only ever REMOVE an attribute — never rewrite the visible URL — so the
# transform is fail-safe even if decoding is imperfect.
defp sanitize_urls(html) do
html
|> scrub_url_attr("href")
|> scrub_url_attr("src")
end
defp scrub_url_attr(html, attr) do
regex = ~r/\s#{attr}\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s>]+))/i
Regex.replace(regex, html, fn full, dquoted, squoted, unquoted ->
value = dquoted <> squoted <> unquoted
if safe_url?(value), do: full, else: ""
end)
end
defp safe_url?(value) do
normalized =
value
|> decode_entities()
# Browsers ignore ASCII control chars + whitespace when parsing a scheme.
|> String.replace(~r/[\x00-\x20\x7f]/u, "")
|> String.downcase()
cond do
Regex.match?(@dangerous_scheme, normalized) -> false
not Regex.match?(@scheme, normalized) -> true
Regex.match?(~r/^(?:#{Enum.join(@allowed_schemes, "|")}):/, normalized) -> true
true -> false
end
end
defp decode_entities(str) do
str
|> replace_entities(~r/&#x([0-9a-f]+);?/i, &String.to_integer(&1, 16))
|> replace_entities(~r/&#([0-9]+);?/, &String.to_integer/1)
|> then(
&Regex.replace(~r/&([a-z]+);/i, &1, fn whole, name ->
Map.get(@named_entities, String.downcase(name), whole)
end)
)
end
defp replace_entities(str, regex, to_codepoint) do
Regex.replace(regex, str, fn _whole, digits -> codepoint(to_codepoint.(digits)) end)
end
defp codepoint(n) when is_integer(n) and n in 0..0x10FFFF do
<<n::utf8>>
rescue
# Surrogate/invalid code points can't be encoded — treat as removed.
_ -> ""
end
defp codepoint(_), do: ""
end