mirror of
https://github.com/go-gitea/gitea.git
synced 2026-08-11 05:36:25 +00:00
perf(emoji): optimize FindEmojiSubmatchIndex using slice-based Trie (#38573)
This pull request optimizes `FindEmojiSubmatchIndex` in Gitea's emoji package (`modules/emoji/emoji.go`) by replacing the `strings.Replacer`-based search with a slice-based trie and a constant-time starting-byte check (`isStartingByte`). The new implementation avoids heap allocations during the search and reduces CPU overhead when rendering Markdown, particularly for plain text that does not contain emojis. ### Verification Verified with unit tests: ```sh go test -count=1 ./modules/emoji/... ``` Benchmarks: ```sh go test -bench=. -benchmem ./modules/emoji/... ``` ### Results | Benchmark | Before | After | | ---------- | ------ | ----- | | `BenchmarkFindEmojiSubmatchIndex` | 168.3 ns/op, 2 allocs/op | 85.78 ns/op, 1 alloc/op | | `BenchmarkFindEmojiSubmatchIndexNoMatch` | 239.8 ns/op, 1 alloc/op | 105.1 ns/op, 0 allocs/op | ### Benchmark Output ```text goos: linux goarch: amd64 pkg: gitea.dev/modules/emoji cpu: Intel(R) Core(TM) i5-7400 CPU @ 3.00GHz BenchmarkFindEmojiSubmatchIndex-4 13498539 85.78 ns/op 16 B/op 1 allocs/op BenchmarkFindEmojiSubmatchIndexNoMatch-4 11220450 105.1 ns/op 0 B/op 0 allocs/op BenchmarkFindEmojiSubmatchIndexOld-4 6569360 168.3 ns/op 48 B/op 2 allocs/op BenchmarkFindEmojiSubmatchIndexOldNoMatch-4 5026116 239.8 ns/op 32 B/op 1 allocs/op ``` --------- Signed-off-by: Sudhanshu Singh <sudhanshuwriterblc@gmail.com> Co-authored-by: silverwind <me@silverwind.io> Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
This commit is contained in:
+27
-69
@@ -5,12 +5,12 @@
|
|||||||
package emoji
|
package emoji
|
||||||
|
|
||||||
import (
|
import (
|
||||||
"io"
|
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
"sync/atomic"
|
"sync/atomic"
|
||||||
|
|
||||||
"gitea.dev/modules/setting"
|
"gitea.dev/modules/setting"
|
||||||
|
"gitea.dev/modules/util"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Gemoji is a set of emoji data.
|
// Gemoji is a set of emoji data.
|
||||||
@@ -26,11 +26,12 @@ type Emoji struct {
|
|||||||
}
|
}
|
||||||
|
|
||||||
type globalVarsStruct struct {
|
type globalVarsStruct struct {
|
||||||
codeMap map[string]int // emoji unicode code to its emoji data.
|
codeMap map[string]int // emoji Unicode code to its emoji data.
|
||||||
aliasMap map[string]int // the alias to its emoji data.
|
aliasMap map[string]int // the alias to its emoji data.
|
||||||
emptyReplacer *strings.Replacer // string replacer for emoji codes, used for finding emoji positions.
|
trie *util.TrieNode // trie for finding emoji positions.
|
||||||
codeReplacer *strings.Replacer // string replacer for emoji codes.
|
isStartingByte [256]bool // fast-path skip for starting bytes.
|
||||||
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
|
codeReplacer *strings.Replacer // string replacer for emoji codes.
|
||||||
|
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
|
||||||
}
|
}
|
||||||
|
|
||||||
var globalVarsStore atomic.Pointer[globalVarsStruct]
|
var globalVarsStore atomic.Pointer[globalVarsStruct]
|
||||||
@@ -44,10 +45,10 @@ func globalVars() *globalVarsStruct {
|
|||||||
vars = &globalVarsStruct{}
|
vars = &globalVarsStruct{}
|
||||||
vars.codeMap = make(map[string]int, len(GemojiData))
|
vars.codeMap = make(map[string]int, len(GemojiData))
|
||||||
vars.aliasMap = make(map[string]int, len(GemojiData))
|
vars.aliasMap = make(map[string]int, len(GemojiData))
|
||||||
|
vars.trie = &util.TrieNode{}
|
||||||
|
|
||||||
// process emoji codes and aliases
|
// process emoji codes and aliases
|
||||||
codePairs := make([]string, 0)
|
codePairs := make([]string, 0)
|
||||||
emptyPairs := make([]string, 0)
|
|
||||||
aliasPairs := make([]string, 0)
|
aliasPairs := make([]string, 0)
|
||||||
|
|
||||||
// sort from largest to small so we match combined emoji first
|
// sort from largest to small so we match combined emoji first
|
||||||
@@ -81,20 +82,20 @@ func globalVars() *globalVarsStruct {
|
|||||||
if firstAlias != "" {
|
if firstAlias != "" {
|
||||||
vars.codeMap[emoji.Emoji] = idx
|
vars.codeMap[emoji.Emoji] = idx
|
||||||
codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":")
|
codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":")
|
||||||
emptyPairs = append(emptyPairs, emoji.Emoji, emoji.Emoji)
|
vars.trie.Insert(emoji.Emoji)
|
||||||
|
vars.isStartingByte[emoji.Emoji[0]] = true
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// create replacers
|
// create replacers
|
||||||
vars.emptyReplacer = strings.NewReplacer(emptyPairs...)
|
|
||||||
vars.codeReplacer = strings.NewReplacer(codePairs...)
|
vars.codeReplacer = strings.NewReplacer(codePairs...)
|
||||||
vars.aliasReplacer = strings.NewReplacer(aliasPairs...)
|
vars.aliasReplacer = strings.NewReplacer(aliasPairs...)
|
||||||
globalVarsStore.Store(vars)
|
globalVarsStore.Store(vars)
|
||||||
return vars
|
return vars
|
||||||
}
|
}
|
||||||
|
|
||||||
// FromCode retrieves the emoji data based on the provided unicode code (ie,
|
// FromCode retrieves the emoji data based on the provided Unicode code
|
||||||
// "\u2618" will return the Gemoji data for "shamrock").
|
// e.g.: "\u2618" will return the Gemoji data for "shamrock".
|
||||||
func FromCode(code string) *Emoji {
|
func FromCode(code string) *Emoji {
|
||||||
i, ok := globalVars().codeMap[code]
|
i, ok := globalVars().codeMap[code]
|
||||||
if !ok {
|
if !ok {
|
||||||
@@ -104,9 +105,8 @@ func FromCode(code string) *Emoji {
|
|||||||
return &GemojiData[i]
|
return &GemojiData[i]
|
||||||
}
|
}
|
||||||
|
|
||||||
// FromAlias retrieves the emoji data based on the provided alias in the form
|
// FromAlias retrieves the emoji data based on the provided alias in the form "alias" or ":alias:"
|
||||||
// "alias" or ":alias:" (ie, "shamrock" or ":shamrock:" will return the Gemoji
|
// e.g.: "shamrock" or ":shamrock:" will return the Gemoji data for "shamrock".
|
||||||
// data for "shamrock").
|
|
||||||
func FromAlias(alias string) *Emoji {
|
func FromAlias(alias string) *Emoji {
|
||||||
if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") {
|
if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") {
|
||||||
alias = alias[1 : len(alias)-1]
|
alias = alias[1 : len(alias)-1]
|
||||||
@@ -120,69 +120,27 @@ func FromAlias(alias string) *Emoji {
|
|||||||
return &GemojiData[i]
|
return &GemojiData[i]
|
||||||
}
|
}
|
||||||
|
|
||||||
// ReplaceCodes replaces all emoji codes with the first corresponding emoji
|
// ReplaceCodes replaces all emoji codes with the first corresponding emoji alias in the form of ":alias:"
|
||||||
// alias (in the form of ":alias:") (ie, "\u2618" will be converted to
|
// e.g.: "\u2618" will be converted to ":shamrock:".
|
||||||
// ":shamrock:").
|
|
||||||
func ReplaceCodes(s string) string {
|
func ReplaceCodes(s string) string {
|
||||||
return globalVars().codeReplacer.Replace(s)
|
return globalVars().codeReplacer.Replace(s)
|
||||||
}
|
}
|
||||||
|
|
||||||
// ReplaceAliases replaces all aliases of the form ":alias:" with its
|
// ReplaceAliases replaces all aliases of the form ":alias:" with its corresponding Unicode value.
|
||||||
// corresponding unicode value.
|
|
||||||
func ReplaceAliases(s string) string {
|
func ReplaceAliases(s string) string {
|
||||||
return globalVars().aliasReplacer.Replace(s)
|
return globalVars().aliasReplacer.Replace(s)
|
||||||
}
|
}
|
||||||
|
|
||||||
type rememberSecondWriteWriter struct {
|
// FindEmojiSubmatchIndex returns index-pair of the first emoji in a string
|
||||||
pos int
|
|
||||||
idx int
|
|
||||||
end int
|
|
||||||
writecount int
|
|
||||||
}
|
|
||||||
|
|
||||||
func (n *rememberSecondWriteWriter) Write(p []byte) (int, error) {
|
|
||||||
n.writecount++
|
|
||||||
if n.writecount == 2 {
|
|
||||||
n.idx = n.pos
|
|
||||||
n.end = n.pos + len(p)
|
|
||||||
n.pos += len(p)
|
|
||||||
return len(p), io.EOF
|
|
||||||
}
|
|
||||||
n.pos += len(p)
|
|
||||||
return len(p), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
func (n *rememberSecondWriteWriter) WriteString(s string) (int, error) {
|
|
||||||
n.writecount++
|
|
||||||
if n.writecount == 2 {
|
|
||||||
n.idx = n.pos
|
|
||||||
n.end = n.pos + len(s)
|
|
||||||
n.pos += len(s)
|
|
||||||
return len(s), io.EOF
|
|
||||||
}
|
|
||||||
n.pos += len(s)
|
|
||||||
return len(s), nil
|
|
||||||
}
|
|
||||||
|
|
||||||
// FindEmojiSubmatchIndex returns index pair of longest emoji in a string
|
|
||||||
func FindEmojiSubmatchIndex(s string) []int {
|
func FindEmojiSubmatchIndex(s string) []int {
|
||||||
secondWriteWriter := rememberSecondWriteWriter{}
|
vars := globalVars()
|
||||||
|
for i := 0; i < len(s); i++ {
|
||||||
// A faster and clean implementation would copy the trie tree formation in strings.NewReplacer but
|
if !vars.isStartingByte[s[i]] {
|
||||||
// we can be lazy here.
|
continue
|
||||||
//
|
}
|
||||||
// The implementation of strings.Replacer.WriteString is such that the first index of the emoji
|
if matchLen := vars.trie.Match(s, i); matchLen > 0 {
|
||||||
// submatch is simply the second thing that is written to WriteString in the writer.
|
return []int{i, i + matchLen}
|
||||||
//
|
}
|
||||||
// Therefore we can simply take the index of the second write as our first emoji
|
|
||||||
//
|
|
||||||
// FIXME: just copy the trie implementation from strings.NewReplacer
|
|
||||||
_, _ = globalVars().emptyReplacer.WriteString(&secondWriteWriter, s)
|
|
||||||
|
|
||||||
// if we wrote less than twice then we never "replaced"
|
|
||||||
if secondWriteWriter.writecount < 2 {
|
|
||||||
return nil
|
|
||||||
}
|
}
|
||||||
|
return nil
|
||||||
return []int{secondWriteWriter.idx, secondWriteWriter.end}
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -61,13 +61,18 @@ func TestReplacers(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const (
|
||||||
|
testInputWithEmojis = "This is a test string containing some emojis like \U0001f44d and \U0001f37a and some text in between."
|
||||||
|
testInputNoEmojis = "This is a test string containing no emojis at all, just plain old ASCII text, which should ideally be scanned very quickly by our trie implementation."
|
||||||
|
)
|
||||||
|
|
||||||
func TestFindEmojiSubmatchIndex(t *testing.T) {
|
func TestFindEmojiSubmatchIndex(t *testing.T) {
|
||||||
type testcase struct {
|
type testcase struct {
|
||||||
teststring string
|
input string
|
||||||
expected []int
|
expected []int
|
||||||
}
|
}
|
||||||
|
|
||||||
testcases := []testcase{
|
testCases := []testcase{
|
||||||
{
|
{
|
||||||
"\U0001f44d",
|
"\U0001f44d",
|
||||||
[]int{0, len("\U0001f44d")},
|
[]int{0, len("\U0001f44d")},
|
||||||
@@ -81,13 +86,40 @@ func TestFindEmojiSubmatchIndex(t *testing.T) {
|
|||||||
[]int{1, 1 + len("\U0001f44d")},
|
[]int{1, 1 + len("\U0001f44d")},
|
||||||
},
|
},
|
||||||
{
|
{
|
||||||
string([]byte{'\u0001'}) + "\U0001f44d",
|
"\u0001\U0001f44d",
|
||||||
[]int{1, 1 + len("\U0001f44d")},
|
[]int{1, 1 + len("\U0001f44d")},
|
||||||
},
|
},
|
||||||
|
{
|
||||||
|
// This package can handle keycap emoji if it is registered in the emoji data.
|
||||||
|
// However, many other places (e.g.: markup rendering) also might not handle such cases correctly.
|
||||||
|
// For example: how is "**{U+FE0F}{U+20E3}**" rendered in Markdown/Markup?
|
||||||
|
"a 8\U0000fe0f\U000020e3 b", // keycap emoji "8\ufe0f\u20e3" in emoji data
|
||||||
|
[]int{2, 2 + len("8\U0000fe0f\U000020e3")},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
testInputWithEmojis,
|
||||||
|
[]int{50, 54},
|
||||||
|
},
|
||||||
|
{
|
||||||
|
testInputNoEmojis,
|
||||||
|
nil,
|
||||||
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
for _, kase := range testcases {
|
for _, tc := range testCases {
|
||||||
actual := FindEmojiSubmatchIndex(kase.teststring)
|
actual := FindEmojiSubmatchIndex(tc.input)
|
||||||
assert.Equal(t, kase.expected, actual)
|
assert.Equal(t, tc.expected, actual)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func BenchmarkFindEmojiSubmatchIndex(b *testing.B) {
|
||||||
|
for i := 0; i < b.N; i++ {
|
||||||
|
_ = FindEmojiSubmatchIndex(testInputWithEmojis)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
func BenchmarkFindEmojiSubmatchIndexNoMatch(b *testing.B) {
|
||||||
|
for i := 0; i < b.N; i++ {
|
||||||
|
_ = FindEmojiSubmatchIndex(testInputNoEmojis)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,56 @@
|
|||||||
|
// Copyright 2026 The Gitea Authors. All rights reserved.
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
package util
|
||||||
|
|
||||||
|
type trieEdge struct {
|
||||||
|
b byte
|
||||||
|
node *TrieNode
|
||||||
|
}
|
||||||
|
|
||||||
|
// TrieNode represents a node in a slice-based Trie for fast byte-sequence prefix matching.
|
||||||
|
type TrieNode struct {
|
||||||
|
children []trieEdge
|
||||||
|
isEnd bool
|
||||||
|
}
|
||||||
|
|
||||||
|
func (t *TrieNode) child(b byte) *TrieNode {
|
||||||
|
for _, edge := range t.children {
|
||||||
|
if edge.b == b {
|
||||||
|
return edge.node
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
// Insert adds a string (byte sequence) to the Trie.
|
||||||
|
func (t *TrieNode) Insert(val string) {
|
||||||
|
curr := t
|
||||||
|
for i := 0; i < len(val); i++ {
|
||||||
|
next := curr.child(val[i])
|
||||||
|
if next == nil {
|
||||||
|
next = &TrieNode{}
|
||||||
|
curr.children = append(curr.children, trieEdge{b: val[i], node: next})
|
||||||
|
}
|
||||||
|
curr = next
|
||||||
|
}
|
||||||
|
curr.isEnd = true
|
||||||
|
}
|
||||||
|
|
||||||
|
// Match returns the length of the longest matching prefix starting at index `start` in string `s`.
|
||||||
|
// It returns -1 if no prefix is matched.
|
||||||
|
func (t *TrieNode) Match(s string, start int) int {
|
||||||
|
curr := t
|
||||||
|
matchLen := -1
|
||||||
|
for j := start; j < len(s); j++ {
|
||||||
|
next := curr.child(s[j])
|
||||||
|
if next == nil {
|
||||||
|
break
|
||||||
|
}
|
||||||
|
curr = next
|
||||||
|
if curr.isEnd {
|
||||||
|
matchLen = j - start + 1
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return matchLen
|
||||||
|
}
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
// Copyright 2026 The Gitea Authors. All rights reserved.
|
||||||
|
// SPDX-License-Identifier: MIT
|
||||||
|
|
||||||
|
package util
|
||||||
|
|
||||||
|
import (
|
||||||
|
"testing"
|
||||||
|
|
||||||
|
"github.com/stretchr/testify/assert"
|
||||||
|
)
|
||||||
|
|
||||||
|
func TestTrie(t *testing.T) {
|
||||||
|
trie := &TrieNode{}
|
||||||
|
trie.Insert("apple")
|
||||||
|
trie.Insert("apricot")
|
||||||
|
trie.Insert("banana")
|
||||||
|
trie.Insert("app")
|
||||||
|
|
||||||
|
// Test exact matches
|
||||||
|
assert.Equal(t, 5, trie.Match("apple", 0))
|
||||||
|
assert.Equal(t, 7, trie.Match("apricot", 0))
|
||||||
|
assert.Equal(t, 6, trie.Match("banana", 0))
|
||||||
|
assert.Equal(t, 3, trie.Match("app", 0))
|
||||||
|
|
||||||
|
// Test partial match (longest match priority)
|
||||||
|
assert.Equal(t, 5, trie.Match("apple-pie", 0))
|
||||||
|
|
||||||
|
// Test suffix/nested position match
|
||||||
|
assert.Equal(t, 5, trie.Match("sweet apple", 6))
|
||||||
|
|
||||||
|
// Test no match
|
||||||
|
assert.Equal(t, -1, trie.Match("orange", 0))
|
||||||
|
assert.Equal(t, -1, trie.Match("ap", 0)) // prefix exists but is not an end node
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user