update bluemonday to v1.0.5 to fix https://github.com/microcosm-cc/bluemonday/issues/111
This commit is contained in:
+24
-4
@@ -229,7 +229,7 @@ func (p *Policy) sanitize(r io.Reader) *bytes.Buffer {
|
||||
|
||||
case html.StartTagToken:
|
||||
|
||||
mostRecentlyStartedToken = strings.ToLower(token.Data)
|
||||
mostRecentlyStartedToken = normaliseElementName(token.Data)
|
||||
|
||||
aps, ok := p.elsAndAttrs[token.Data]
|
||||
if !ok {
|
||||
@@ -272,7 +272,7 @@ func (p *Policy) sanitize(r io.Reader) *bytes.Buffer {
|
||||
|
||||
case html.EndTagToken:
|
||||
|
||||
if mostRecentlyStartedToken == strings.ToLower(token.Data) {
|
||||
if mostRecentlyStartedToken == normaliseElementName(token.Data) {
|
||||
mostRecentlyStartedToken = ""
|
||||
}
|
||||
|
||||
@@ -350,11 +350,11 @@ func (p *Policy) sanitize(r io.Reader) *bytes.Buffer {
|
||||
|
||||
if !skipElementContent {
|
||||
switch mostRecentlyStartedToken {
|
||||
case "script":
|
||||
case `script`:
|
||||
// not encouraged, but if a policy allows JavaScript we
|
||||
// should not HTML escape it as that would break the output
|
||||
buff.WriteString(token.Data)
|
||||
case "style":
|
||||
case `style`:
|
||||
// not encouraged, but if a policy allows CSS styles we
|
||||
// should not HTML escape it as that would break the output
|
||||
buff.WriteString(token.Data)
|
||||
@@ -887,3 +887,23 @@ func (p *Policy) matchRegex(elementName string) (map[string]attrPolicy, bool) {
|
||||
}
|
||||
return aps, matched
|
||||
}
|
||||
|
||||
|
||||
// normaliseElementName takes a HTML element like <script> which is user input
|
||||
// and returns a lower case version of it that is immune to UTF-8 to ASCII
|
||||
// conversion tricks (like the use of upper case cyrillic i scrİpt which a
|
||||
// strings.ToLower would convert to script). Instead this func will preserve
|
||||
// all non-ASCII as their escaped equivalent, i.e. \u0130 which reveals the
|
||||
// characters when lower cased
|
||||
func normaliseElementName(str string) string {
|
||||
// that useful QuoteToASCII put quote marks at the start and end
|
||||
// so those are trimmed off
|
||||
return strings.TrimSuffix(
|
||||
strings.TrimPrefix(
|
||||
strings.ToLower(
|
||||
strconv.QuoteToASCII(str),
|
||||
),
|
||||
`"`),
|
||||
`"`,
|
||||
)
|
||||
}
|
||||
Reference in New Issue
Block a user