mirror of
https://github.com/go-gitea/gitea.git
synced 2026-07-23 09:22:46 +00:00
perf(emoji): optimize FindEmojiSubmatchIndex using slice-based Trie (#38573)
This pull request optimizes `FindEmojiSubmatchIndex` in Gitea's emoji package (`modules/emoji/emoji.go`) by replacing the `strings.Replacer`-based search with a slice-based trie and a constant-time starting-byte check (`isStartingByte`). The new implementation avoids heap allocations during the search and reduces CPU overhead when rendering Markdown, particularly for plain text that does not contain emojis. ### Verification Verified with unit tests: ```sh go test -count=1 ./modules/emoji/... ``` Benchmarks: ```sh go test -bench=. -benchmem ./modules/emoji/... ``` ### Results | Benchmark | Before | After | | ---------- | ------ | ----- | | `BenchmarkFindEmojiSubmatchIndex` | 168.3 ns/op, 2 allocs/op | 85.78 ns/op, 1 alloc/op | | `BenchmarkFindEmojiSubmatchIndexNoMatch` | 239.8 ns/op, 1 alloc/op | 105.1 ns/op, 0 allocs/op | ### Benchmark Output ```text goos: linux goarch: amd64 pkg: gitea.dev/modules/emoji cpu: Intel(R) Core(TM) i5-7400 CPU @ 3.00GHz BenchmarkFindEmojiSubmatchIndex-4 13498539 85.78 ns/op 16 B/op 1 allocs/op BenchmarkFindEmojiSubmatchIndexNoMatch-4 11220450 105.1 ns/op 0 B/op 0 allocs/op BenchmarkFindEmojiSubmatchIndexOld-4 6569360 168.3 ns/op 48 B/op 2 allocs/op BenchmarkFindEmojiSubmatchIndexOldNoMatch-4 5026116 239.8 ns/op 32 B/op 1 allocs/op ``` --------- Signed-off-by: Sudhanshu Singh <sudhanshuwriterblc@gmail.com> Co-authored-by: silverwind <me@silverwind.io> Co-authored-by: wxiaoguang <wxiaoguang@gmail.com>
This commit is contained in:
@@ -5,12 +5,12 @@
|
||||
package emoji
|
||||
|
||||
import (
|
||||
"io"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
|
||||
"gitea.dev/modules/setting"
|
||||
"gitea.dev/modules/util"
|
||||
)
|
||||
|
||||
// Gemoji is a set of emoji data.
|
||||
@@ -26,11 +26,12 @@ type Emoji struct {
|
||||
}
|
||||
|
||||
type globalVarsStruct struct {
|
||||
codeMap map[string]int // emoji unicode code to its emoji data.
|
||||
aliasMap map[string]int // the alias to its emoji data.
|
||||
emptyReplacer *strings.Replacer // string replacer for emoji codes, used for finding emoji positions.
|
||||
codeReplacer *strings.Replacer // string replacer for emoji codes.
|
||||
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
|
||||
codeMap map[string]int // emoji Unicode code to its emoji data.
|
||||
aliasMap map[string]int // the alias to its emoji data.
|
||||
trie *util.TrieNode // trie for finding emoji positions.
|
||||
isStartingByte [256]bool // fast-path skip for starting bytes.
|
||||
codeReplacer *strings.Replacer // string replacer for emoji codes.
|
||||
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
|
||||
}
|
||||
|
||||
var globalVarsStore atomic.Pointer[globalVarsStruct]
|
||||
@@ -44,10 +45,10 @@ func globalVars() *globalVarsStruct {
|
||||
vars = &globalVarsStruct{}
|
||||
vars.codeMap = make(map[string]int, len(GemojiData))
|
||||
vars.aliasMap = make(map[string]int, len(GemojiData))
|
||||
vars.trie = &util.TrieNode{}
|
||||
|
||||
// process emoji codes and aliases
|
||||
codePairs := make([]string, 0)
|
||||
emptyPairs := make([]string, 0)
|
||||
aliasPairs := make([]string, 0)
|
||||
|
||||
// sort from largest to small so we match combined emoji first
|
||||
@@ -81,20 +82,20 @@ func globalVars() *globalVarsStruct {
|
||||
if firstAlias != "" {
|
||||
vars.codeMap[emoji.Emoji] = idx
|
||||
codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":")
|
||||
emptyPairs = append(emptyPairs, emoji.Emoji, emoji.Emoji)
|
||||
vars.trie.Insert(emoji.Emoji)
|
||||
vars.isStartingByte[emoji.Emoji[0]] = true
|
||||
}
|
||||
}
|
||||
|
||||
// create replacers
|
||||
vars.emptyReplacer = strings.NewReplacer(emptyPairs...)
|
||||
vars.codeReplacer = strings.NewReplacer(codePairs...)
|
||||
vars.aliasReplacer = strings.NewReplacer(aliasPairs...)
|
||||
globalVarsStore.Store(vars)
|
||||
return vars
|
||||
}
|
||||
|
||||
// FromCode retrieves the emoji data based on the provided unicode code (ie,
|
||||
// "\u2618" will return the Gemoji data for "shamrock").
|
||||
// FromCode retrieves the emoji data based on the provided Unicode code
|
||||
// e.g.: "\u2618" will return the Gemoji data for "shamrock".
|
||||
func FromCode(code string) *Emoji {
|
||||
i, ok := globalVars().codeMap[code]
|
||||
if !ok {
|
||||
@@ -104,9 +105,8 @@ func FromCode(code string) *Emoji {
|
||||
return &GemojiData[i]
|
||||
}
|
||||
|
||||
// FromAlias retrieves the emoji data based on the provided alias in the form
|
||||
// "alias" or ":alias:" (ie, "shamrock" or ":shamrock:" will return the Gemoji
|
||||
// data for "shamrock").
|
||||
// FromAlias retrieves the emoji data based on the provided alias in the form "alias" or ":alias:"
|
||||
// e.g.: "shamrock" or ":shamrock:" will return the Gemoji data for "shamrock".
|
||||
func FromAlias(alias string) *Emoji {
|
||||
if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") {
|
||||
alias = alias[1 : len(alias)-1]
|
||||
@@ -120,69 +120,27 @@ func FromAlias(alias string) *Emoji {
|
||||
return &GemojiData[i]
|
||||
}
|
||||
|
||||
// ReplaceCodes replaces all emoji codes with the first corresponding emoji
|
||||
// alias (in the form of ":alias:") (ie, "\u2618" will be converted to
|
||||
// ":shamrock:").
|
||||
// ReplaceCodes replaces all emoji codes with the first corresponding emoji alias in the form of ":alias:"
|
||||
// e.g.: "\u2618" will be converted to ":shamrock:".
|
||||
func ReplaceCodes(s string) string {
|
||||
return globalVars().codeReplacer.Replace(s)
|
||||
}
|
||||
|
||||
// ReplaceAliases replaces all aliases of the form ":alias:" with its
|
||||
// corresponding unicode value.
|
||||
// ReplaceAliases replaces all aliases of the form ":alias:" with its corresponding Unicode value.
|
||||
func ReplaceAliases(s string) string {
|
||||
return globalVars().aliasReplacer.Replace(s)
|
||||
}
|
||||
|
||||
type rememberSecondWriteWriter struct {
|
||||
pos int
|
||||
idx int
|
||||
end int
|
||||
writecount int
|
||||
}
|
||||
|
||||
func (n *rememberSecondWriteWriter) Write(p []byte) (int, error) {
|
||||
n.writecount++
|
||||
if n.writecount == 2 {
|
||||
n.idx = n.pos
|
||||
n.end = n.pos + len(p)
|
||||
n.pos += len(p)
|
||||
return len(p), io.EOF
|
||||
}
|
||||
n.pos += len(p)
|
||||
return len(p), nil
|
||||
}
|
||||
|
||||
func (n *rememberSecondWriteWriter) WriteString(s string) (int, error) {
|
||||
n.writecount++
|
||||
if n.writecount == 2 {
|
||||
n.idx = n.pos
|
||||
n.end = n.pos + len(s)
|
||||
n.pos += len(s)
|
||||
return len(s), io.EOF
|
||||
}
|
||||
n.pos += len(s)
|
||||
return len(s), nil
|
||||
}
|
||||
|
||||
// FindEmojiSubmatchIndex returns index pair of longest emoji in a string
|
||||
// FindEmojiSubmatchIndex returns index-pair of the first emoji in a string
|
||||
func FindEmojiSubmatchIndex(s string) []int {
|
||||
secondWriteWriter := rememberSecondWriteWriter{}
|
||||
|
||||
// A faster and clean implementation would copy the trie tree formation in strings.NewReplacer but
|
||||
// we can be lazy here.
|
||||
//
|
||||
// The implementation of strings.Replacer.WriteString is such that the first index of the emoji
|
||||
// submatch is simply the second thing that is written to WriteString in the writer.
|
||||
//
|
||||
// Therefore we can simply take the index of the second write as our first emoji
|
||||
//
|
||||
// FIXME: just copy the trie implementation from strings.NewReplacer
|
||||
_, _ = globalVars().emptyReplacer.WriteString(&secondWriteWriter, s)
|
||||
|
||||
// if we wrote less than twice then we never "replaced"
|
||||
if secondWriteWriter.writecount < 2 {
|
||||
return nil
|
||||
vars := globalVars()
|
||||
for i := 0; i < len(s); i++ {
|
||||
if !vars.isStartingByte[s[i]] {
|
||||
continue
|
||||
}
|
||||
if matchLen := vars.trie.Match(s, i); matchLen > 0 {
|
||||
return []int{i, i + matchLen}
|
||||
}
|
||||
}
|
||||
|
||||
return []int{secondWriteWriter.idx, secondWriteWriter.end}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -61,13 +61,18 @@ func TestReplacers(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
const (
|
||||
testInputWithEmojis = "This is a test string containing some emojis like \U0001f44d and \U0001f37a and some text in between."
|
||||
testInputNoEmojis = "This is a test string containing no emojis at all, just plain old ASCII text, which should ideally be scanned very quickly by our trie implementation."
|
||||
)
|
||||
|
||||
func TestFindEmojiSubmatchIndex(t *testing.T) {
|
||||
type testcase struct {
|
||||
teststring string
|
||||
expected []int
|
||||
input string
|
||||
expected []int
|
||||
}
|
||||
|
||||
testcases := []testcase{
|
||||
testCases := []testcase{
|
||||
{
|
||||
"\U0001f44d",
|
||||
[]int{0, len("\U0001f44d")},
|
||||
@@ -81,13 +86,40 @@ func TestFindEmojiSubmatchIndex(t *testing.T) {
|
||||
[]int{1, 1 + len("\U0001f44d")},
|
||||
},
|
||||
{
|
||||
string([]byte{'\u0001'}) + "\U0001f44d",
|
||||
"\u0001\U0001f44d",
|
||||
[]int{1, 1 + len("\U0001f44d")},
|
||||
},
|
||||
{
|
||||
// This package can handle keycap emoji if it is registered in the emoji data.
|
||||
// However, many other places (e.g.: markup rendering) also might not handle such cases correctly.
|
||||
// For example: how is "**{U+FE0F}{U+20E3}**" rendered in Markdown/Markup?
|
||||
"a 8\U0000fe0f\U000020e3 b", // keycap emoji "8\ufe0f\u20e3" in emoji data
|
||||
[]int{2, 2 + len("8\U0000fe0f\U000020e3")},
|
||||
},
|
||||
{
|
||||
testInputWithEmojis,
|
||||
[]int{50, 54},
|
||||
},
|
||||
{
|
||||
testInputNoEmojis,
|
||||
nil,
|
||||
},
|
||||
}
|
||||
|
||||
for _, kase := range testcases {
|
||||
actual := FindEmojiSubmatchIndex(kase.teststring)
|
||||
assert.Equal(t, kase.expected, actual)
|
||||
for _, tc := range testCases {
|
||||
actual := FindEmojiSubmatchIndex(tc.input)
|
||||
assert.Equal(t, tc.expected, actual)
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkFindEmojiSubmatchIndex(b *testing.B) {
|
||||
for i := 0; i < b.N; i++ {
|
||||
_ = FindEmojiSubmatchIndex(testInputWithEmojis)
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkFindEmojiSubmatchIndexNoMatch(b *testing.B) {
|
||||
for i := 0; i < b.N; i++ {
|
||||
_ = FindEmojiSubmatchIndex(testInputNoEmojis)
|
||||
}
|
||||
}
|
||||
|
||||
56
modules/util/trie.go
Normal file
56
modules/util/trie.go
Normal file
@@ -0,0 +1,56 @@
|
||||
// Copyright 2026 The Gitea Authors. All rights reserved.
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package util
|
||||
|
||||
type trieEdge struct {
|
||||
b byte
|
||||
node *TrieNode
|
||||
}
|
||||
|
||||
// TrieNode represents a node in a slice-based Trie for fast byte-sequence prefix matching.
|
||||
type TrieNode struct {
|
||||
children []trieEdge
|
||||
isEnd bool
|
||||
}
|
||||
|
||||
func (t *TrieNode) child(b byte) *TrieNode {
|
||||
for _, edge := range t.children {
|
||||
if edge.b == b {
|
||||
return edge.node
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Insert adds a string (byte sequence) to the Trie.
|
||||
func (t *TrieNode) Insert(val string) {
|
||||
curr := t
|
||||
for i := 0; i < len(val); i++ {
|
||||
next := curr.child(val[i])
|
||||
if next == nil {
|
||||
next = &TrieNode{}
|
||||
curr.children = append(curr.children, trieEdge{b: val[i], node: next})
|
||||
}
|
||||
curr = next
|
||||
}
|
||||
curr.isEnd = true
|
||||
}
|
||||
|
||||
// Match returns the length of the longest matching prefix starting at index `start` in string `s`.
|
||||
// It returns -1 if no prefix is matched.
|
||||
func (t *TrieNode) Match(s string, start int) int {
|
||||
curr := t
|
||||
matchLen := -1
|
||||
for j := start; j < len(s); j++ {
|
||||
next := curr.child(s[j])
|
||||
if next == nil {
|
||||
break
|
||||
}
|
||||
curr = next
|
||||
if curr.isEnd {
|
||||
matchLen = j - start + 1
|
||||
}
|
||||
}
|
||||
return matchLen
|
||||
}
|
||||
34
modules/util/trie_test.go
Normal file
34
modules/util/trie_test.go
Normal file
@@ -0,0 +1,34 @@
|
||||
// Copyright 2026 The Gitea Authors. All rights reserved.
|
||||
// SPDX-License-Identifier: MIT
|
||||
|
||||
package util
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/stretchr/testify/assert"
|
||||
)
|
||||
|
||||
func TestTrie(t *testing.T) {
|
||||
trie := &TrieNode{}
|
||||
trie.Insert("apple")
|
||||
trie.Insert("apricot")
|
||||
trie.Insert("banana")
|
||||
trie.Insert("app")
|
||||
|
||||
// Test exact matches
|
||||
assert.Equal(t, 5, trie.Match("apple", 0))
|
||||
assert.Equal(t, 7, trie.Match("apricot", 0))
|
||||
assert.Equal(t, 6, trie.Match("banana", 0))
|
||||
assert.Equal(t, 3, trie.Match("app", 0))
|
||||
|
||||
// Test partial match (longest match priority)
|
||||
assert.Equal(t, 5, trie.Match("apple-pie", 0))
|
||||
|
||||
// Test suffix/nested position match
|
||||
assert.Equal(t, 5, trie.Match("sweet apple", 6))
|
||||
|
||||
// Test no match
|
||||
assert.Equal(t, -1, trie.Match("orange", 0))
|
||||
assert.Equal(t, -1, trie.Match("ap", 0)) // prefix exists but is not an end node
|
||||
}
|
||||
Reference in New Issue
Block a user