diff --git a/modules/charset/escape_stream.go b/modules/charset/escape_stream.go index 360e77782fb..df31c3b16cf 100644 --- a/modules/charset/escape_stream.go +++ b/modules/charset/escape_stream.go @@ -8,11 +8,13 @@ import ( "fmt" "html" "io" + "strings" "unicode" "unicode/utf8" "gitea.dev/modules/setting" "gitea.dev/modules/translation" + "gitea.dev/modules/util" ) type htmlChunkReader struct { @@ -30,6 +32,10 @@ type escapeStreamer struct { ambiguousTables []*AmbiguousTable allowed map[rune]bool + tagPartial []byte // partial tag content, used to detect if we are in some tags + + inTagMath bool // MathML operators like U+2212 are intended and wrapping them breaks the math layout + out io.Writer } @@ -62,6 +68,7 @@ func escapeStream(locale translation.Locale, in io.Reader, out io.Writer, opts . for i, part := range parts { if partInTag[i] { lastIsTag = true + es.trackHtmlTag(part) if _, err := out.Write(part); err != nil { return nil, err } @@ -75,7 +82,11 @@ func escapeStream(locale translation.Locale, in io.Reader, out io.Writer, opts . return nil, err } } - if err = es.detectAndWriteRunes(part); err != nil { + if es.inTagMath { + if _, err := out.Write(part); err != nil { + return nil, err + } + } else if err = es.detectAndWriteRunes(part); err != nil { return nil, err } } @@ -83,6 +94,34 @@ func escapeStream(locale translation.Locale, in io.Reader, out io.Writer, opts . } } +// trackHtmlTag receives tag parts, a tag might be split into multiple parts +func (e *escapeStreamer) trackHtmlTag(part []byte) { + const maxHeadLen = 100 // only read the first N bytes of the tag for detection purpose + if part[0] == '<' { + // start a new tag + e.tagPartial = e.tagPartial[:0] + } + if len(e.tagPartial) >= maxHeadLen { + return + } + e.tagPartial = append(e.tagPartial, part[:min(len(part), maxHeadLen-len(e.tagPartial))]...) + + isTag := func(prefix string) bool { + if len(e.tagPartial) < len(prefix)+1 { + return false + } + if !util.AsciiEqualFold(e.tagPartial[:len(prefix)], []byte(prefix)) { + return false + } + return strings.IndexByte(" \t\n\r\f>", e.tagPartial[len(prefix)]) != -1 + } + if isTag("๐พ`, status: EscapeStatus{Escaped: true, HasAmbiguous: true}, }, + { + name: "ambiguous in math", + text: "โˆ’b โˆ’", + result: `โˆ’b โˆ’`, + status: EscapeStatus{Escaped: true, HasAmbiguous: true}, + }, } func TestEscapeControlReader(t *testing.T) { @@ -156,6 +162,24 @@ func TestEscapeControlReader(t *testing.T) { } } +func TestTrackHtmlTag(t *testing.T) { + e := &escapeStreamer{} + for _, tt := range []struct { + parts []string + inMath bool + }{ + {[]string{"`}, true}, + {[]string{""}, true}, + {[]string{""}, false}, + {[]string{""}, false}, + } { + for _, part := range tt.parts { + e.trackHtmlTag([]byte(part)) + } + assert.Equal(t, tt.inMath, e.inTagMath, "%v", tt.parts) + } +} + func TestSettingAmbiguousUnicodeDetection(t *testing.T) { defer test.MockVariableValue(&setting.UI.AmbiguousUnicodeDetection, true)() _, out := EscapeControlHTML("aย test", &translation.MockLocale{}) diff --git a/modules/util/string.go b/modules/util/string.go index 0d1532b7d61..e077f4eff2d 100644 --- a/modules/util/string.go +++ b/modules/util/string.go @@ -121,7 +121,7 @@ func asciiLower(b byte) byte { // AsciiEqualFold is from Golang https://cs.opensource.google/go/go/+/refs/tags/go1.24.4:src/net/http/internal/ascii/print.go // ASCII only. In most cases for protocols, we should only use this but not [strings.EqualFold] -func AsciiEqualFold(s, t string) bool { +func AsciiEqualFold[T string | []byte](s, t T) bool { if len(s) != len(t) { return false }