perf(emoji): optimize FindEmojiSubmatchIndex using slice-based Trie (#38573)

This pull request optimizes `FindEmojiSubmatchIndex` in Gitea's emoji
package (`modules/emoji/emoji.go`) by replacing the
`strings.Replacer`-based search with a slice-based trie and a
constant-time starting-byte check (`isStartingByte`).

The new implementation avoids heap allocations during the search and
reduces CPU overhead when rendering Markdown, particularly for plain
text that does not contain emojis.

### Verification

Verified with unit tests:

```sh
go test -count=1 ./modules/emoji/...
```

Benchmarks:

```sh
go test -bench=. -benchmem ./modules/emoji/...
```

### Results

| Benchmark | Before | After |
| ---------- | ------ | ----- |
| `BenchmarkFindEmojiSubmatchIndex` | 168.3 ns/op, 2 allocs/op | 85.78
ns/op, 1 alloc/op |
| `BenchmarkFindEmojiSubmatchIndexNoMatch` | 239.8 ns/op, 1 alloc/op |
105.1 ns/op, 0 allocs/op |
### Benchmark Output

```text
goos: linux
goarch: amd64
pkg: gitea.dev/modules/emoji
cpu: Intel(R) Core(TM) i5-7400 CPU @ 3.00GHz

BenchmarkFindEmojiSubmatchIndex-4              13498539        85.78 ns/op      16 B/op   1 allocs/op
BenchmarkFindEmojiSubmatchIndexNoMatch-4       11220450       105.1 ns/op        0 B/op   0 allocs/op
BenchmarkFindEmojiSubmatchIndexOld-4            6569360       168.3 ns/op      48 B/op   2 allocs/op
BenchmarkFindEmojiSubmatchIndexOldNoMatch-4     5026116       239.8 ns/op      32 B/op   1 allocs/op
```

---------

Signed-off-by: Sudhanshu Singh <sudhanshuwriterblc@gmail.com>
Co-authored-by: silverwind <me@silverwind.io>
Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
This commit is contained in:
Shudhanshu Singh
2026-07-23 13:59:28 +05:30
committed by GitHub
parent acbdc44d00
commit ee10ae168c
4 changed files with 156 additions and 76 deletions
+27 -69
View File
@@ -5,12 +5,12 @@
package emoji package emoji
import ( import (
"io"
"sort" "sort"
"strings" "strings"
"sync/atomic" "sync/atomic"
"gitea.dev/modules/setting" "gitea.dev/modules/setting"
"gitea.dev/modules/util"
) )
// Gemoji is a set of emoji data. // Gemoji is a set of emoji data.
@@ -26,11 +26,12 @@ type Emoji struct {
} }
type globalVarsStruct struct { type globalVarsStruct struct {
codeMap map[string]int // emoji unicode code to its emoji data. codeMap map[string]int // emoji Unicode code to its emoji data.
aliasMap map[string]int // the alias to its emoji data. aliasMap map[string]int // the alias to its emoji data.
emptyReplacer *strings.Replacer // string replacer for emoji codes, used for finding emoji positions. trie *util.TrieNode // trie for finding emoji positions.
codeReplacer *strings.Replacer // string replacer for emoji codes. isStartingByte [256]bool // fast-path skip for starting bytes.
aliasReplacer *strings.Replacer // string replacer for emoji aliases. codeReplacer *strings.Replacer // string replacer for emoji codes.
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
} }
var globalVarsStore atomic.Pointer[globalVarsStruct] var globalVarsStore atomic.Pointer[globalVarsStruct]
@@ -44,10 +45,10 @@ func globalVars() *globalVarsStruct {
vars = &globalVarsStruct{} vars = &globalVarsStruct{}
vars.codeMap = make(map[string]int, len(GemojiData)) vars.codeMap = make(map[string]int, len(GemojiData))
vars.aliasMap = make(map[string]int, len(GemojiData)) vars.aliasMap = make(map[string]int, len(GemojiData))
vars.trie = &util.TrieNode{}
// process emoji codes and aliases // process emoji codes and aliases
codePairs := make([]string, 0) codePairs := make([]string, 0)
emptyPairs := make([]string, 0)
aliasPairs := make([]string, 0) aliasPairs := make([]string, 0)
// sort from largest to small so we match combined emoji first // sort from largest to small so we match combined emoji first
@@ -81,20 +82,20 @@ func globalVars() *globalVarsStruct {
if firstAlias != "" { if firstAlias != "" {
vars.codeMap[emoji.Emoji] = idx vars.codeMap[emoji.Emoji] = idx
codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":") codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":")
emptyPairs = append(emptyPairs, emoji.Emoji, emoji.Emoji) vars.trie.Insert(emoji.Emoji)
vars.isStartingByte[emoji.Emoji[0]] = true
} }
} }
// create replacers // create replacers
vars.emptyReplacer = strings.NewReplacer(emptyPairs...)
vars.codeReplacer = strings.NewReplacer(codePairs...) vars.codeReplacer = strings.NewReplacer(codePairs...)
vars.aliasReplacer = strings.NewReplacer(aliasPairs...) vars.aliasReplacer = strings.NewReplacer(aliasPairs...)
globalVarsStore.Store(vars) globalVarsStore.Store(vars)
return vars return vars
} }
// FromCode retrieves the emoji data based on the provided unicode code (ie, // FromCode retrieves the emoji data based on the provided Unicode code
// "\u2618" will return the Gemoji data for "shamrock"). // e.g.: "\u2618" will return the Gemoji data for "shamrock".
func FromCode(code string) *Emoji { func FromCode(code string) *Emoji {
i, ok := globalVars().codeMap[code] i, ok := globalVars().codeMap[code]
if !ok { if !ok {
@@ -104,9 +105,8 @@ func FromCode(code string) *Emoji {
return &GemojiData[i] return &GemojiData[i]
} }
// FromAlias retrieves the emoji data based on the provided alias in the form // FromAlias retrieves the emoji data based on the provided alias in the form "alias" or ":alias:"
// "alias" or ":alias:" (ie, "shamrock" or ":shamrock:" will return the Gemoji // e.g.: "shamrock" or ":shamrock:" will return the Gemoji data for "shamrock".
// data for "shamrock").
func FromAlias(alias string) *Emoji { func FromAlias(alias string) *Emoji {
if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") { if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") {
alias = alias[1 : len(alias)-1] alias = alias[1 : len(alias)-1]
@@ -120,69 +120,27 @@ func FromAlias(alias string) *Emoji {
return &GemojiData[i] return &GemojiData[i]
} }
// ReplaceCodes replaces all emoji codes with the first corresponding emoji // ReplaceCodes replaces all emoji codes with the first corresponding emoji alias in the form of ":alias:"
// alias (in the form of ":alias:") (ie, "\u2618" will be converted to // e.g.: "\u2618" will be converted to ":shamrock:".
// ":shamrock:").
func ReplaceCodes(s string) string { func ReplaceCodes(s string) string {
return globalVars().codeReplacer.Replace(s) return globalVars().codeReplacer.Replace(s)
} }
// ReplaceAliases replaces all aliases of the form ":alias:" with its // ReplaceAliases replaces all aliases of the form ":alias:" with its corresponding Unicode value.
// corresponding unicode value.
func ReplaceAliases(s string) string { func ReplaceAliases(s string) string {
return globalVars().aliasReplacer.Replace(s) return globalVars().aliasReplacer.Replace(s)
} }
type rememberSecondWriteWriter struct { // FindEmojiSubmatchIndex returns index-pair of the first emoji in a string
pos int
idx int
end int
writecount int
}
func (n *rememberSecondWriteWriter) Write(p []byte) (int, error) {
n.writecount++
if n.writecount == 2 {
n.idx = n.pos
n.end = n.pos + len(p)
n.pos += len(p)
return len(p), io.EOF
}
n.pos += len(p)
return len(p), nil
}
func (n *rememberSecondWriteWriter) WriteString(s string) (int, error) {
n.writecount++
if n.writecount == 2 {
n.idx = n.pos
n.end = n.pos + len(s)
n.pos += len(s)
return len(s), io.EOF
}
n.pos += len(s)
return len(s), nil
}
// FindEmojiSubmatchIndex returns index pair of longest emoji in a string
func FindEmojiSubmatchIndex(s string) []int { func FindEmojiSubmatchIndex(s string) []int {
secondWriteWriter := rememberSecondWriteWriter{} vars := globalVars()
for i := 0; i < len(s); i++ {
// A faster and clean implementation would copy the trie tree formation in strings.NewReplacer but if !vars.isStartingByte[s[i]] {
// we can be lazy here. continue
// }
// The implementation of strings.Replacer.WriteString is such that the first index of the emoji if matchLen := vars.trie.Match(s, i); matchLen > 0 {
// submatch is simply the second thing that is written to WriteString in the writer. return []int{i, i + matchLen}
// }
// Therefore we can simply take the index of the second write as our first emoji
//
// FIXME: just copy the trie implementation from strings.NewReplacer
_, _ = globalVars().emptyReplacer.WriteString(&secondWriteWriter, s)
// if we wrote less than twice then we never "replaced"
if secondWriteWriter.writecount < 2 {
return nil
} }
return nil
return []int{secondWriteWriter.idx, secondWriteWriter.end}
} }
+39 -7
View File
@@ -61,13 +61,18 @@ func TestReplacers(t *testing.T) {
} }
} }
const (
testInputWithEmojis = "This is a test string containing some emojis like \U0001f44d and \U0001f37a and some text in between."
testInputNoEmojis = "This is a test string containing no emojis at all, just plain old ASCII text, which should ideally be scanned very quickly by our trie implementation."
)
func TestFindEmojiSubmatchIndex(t *testing.T) { func TestFindEmojiSubmatchIndex(t *testing.T) {
type testcase struct { type testcase struct {
teststring string input string
expected []int expected []int
} }
testcases := []testcase{ testCases := []testcase{
{ {
"\U0001f44d", "\U0001f44d",
[]int{0, len("\U0001f44d")}, []int{0, len("\U0001f44d")},
@@ -81,13 +86,40 @@ func TestFindEmojiSubmatchIndex(t *testing.T) {
[]int{1, 1 + len("\U0001f44d")}, []int{1, 1 + len("\U0001f44d")},
}, },
{ {
string([]byte{'\u0001'}) + "\U0001f44d", "\u0001\U0001f44d",
[]int{1, 1 + len("\U0001f44d")}, []int{1, 1 + len("\U0001f44d")},
}, },
{
// This package can handle keycap emoji if it is registered in the emoji data.
// However, many other places (e.g.: markup rendering) also might not handle such cases correctly.
// For example: how is "**{U+FE0F}{U+20E3}**" rendered in Markdown/Markup?
"a 8\U0000fe0f\U000020e3 b", // keycap emoji "8\ufe0f\u20e3" in emoji data
[]int{2, 2 + len("8\U0000fe0f\U000020e3")},
},
{
testInputWithEmojis,
[]int{50, 54},
},
{
testInputNoEmojis,
nil,
},
} }
for _, kase := range testcases { for _, tc := range testCases {
actual := FindEmojiSubmatchIndex(kase.teststring) actual := FindEmojiSubmatchIndex(tc.input)
assert.Equal(t, kase.expected, actual) assert.Equal(t, tc.expected, actual)
}
}
func BenchmarkFindEmojiSubmatchIndex(b *testing.B) {
for i := 0; i < b.N; i++ {
_ = FindEmojiSubmatchIndex(testInputWithEmojis)
}
}
func BenchmarkFindEmojiSubmatchIndexNoMatch(b *testing.B) {
for i := 0; i < b.N; i++ {
_ = FindEmojiSubmatchIndex(testInputNoEmojis)
} }
} }
+56
View File
@@ -0,0 +1,56 @@
// Copyright 2026 The Gitea Authors. All rights reserved.
// SPDX-License-Identifier: MIT
package util
type trieEdge struct {
b byte
node *TrieNode
}
// TrieNode represents a node in a slice-based Trie for fast byte-sequence prefix matching.
type TrieNode struct {
children []trieEdge
isEnd bool
}
func (t *TrieNode) child(b byte) *TrieNode {
for _, edge := range t.children {
if edge.b == b {
return edge.node
}
}
return nil
}
// Insert adds a string (byte sequence) to the Trie.
func (t *TrieNode) Insert(val string) {
curr := t
for i := 0; i < len(val); i++ {
next := curr.child(val[i])
if next == nil {
next = &TrieNode{}
curr.children = append(curr.children, trieEdge{b: val[i], node: next})
}
curr = next
}
curr.isEnd = true
}
// Match returns the length of the longest matching prefix starting at index `start` in string `s`.
// It returns -1 if no prefix is matched.
func (t *TrieNode) Match(s string, start int) int {
curr := t
matchLen := -1
for j := start; j < len(s); j++ {
next := curr.child(s[j])
if next == nil {
break
}
curr = next
if curr.isEnd {
matchLen = j - start + 1
}
}
return matchLen
}
+34
View File
@@ -0,0 +1,34 @@
// Copyright 2026 The Gitea Authors. All rights reserved.
// SPDX-License-Identifier: MIT
package util
import (
"testing"
"github.com/stretchr/testify/assert"
)
func TestTrie(t *testing.T) {
trie := &TrieNode{}
trie.Insert("apple")
trie.Insert("apricot")
trie.Insert("banana")
trie.Insert("app")
// Test exact matches
assert.Equal(t, 5, trie.Match("apple", 0))
assert.Equal(t, 7, trie.Match("apricot", 0))
assert.Equal(t, 6, trie.Match("banana", 0))
assert.Equal(t, 3, trie.Match("app", 0))
// Test partial match (longest match priority)
assert.Equal(t, 5, trie.Match("apple-pie", 0))
// Test suffix/nested position match
assert.Equal(t, 5, trie.Match("sweet apple", 6))
// Test no match
assert.Equal(t, -1, trie.Match("orange", 0))
assert.Equal(t, -1, trie.Match("ap", 0)) // prefix exists but is not an end node
}