diff --git a/modules/emoji/emoji.go b/modules/emoji/emoji.go index 26d552a2a2..7953a5929c 100644 --- a/modules/emoji/emoji.go +++ b/modules/emoji/emoji.go @@ -5,12 +5,12 @@ package emoji import ( - "io" "sort" "strings" "sync/atomic" "gitea.dev/modules/setting" + "gitea.dev/modules/util" ) // Gemoji is a set of emoji data. @@ -26,11 +26,12 @@ type Emoji struct { } type globalVarsStruct struct { - codeMap map[string]int // emoji unicode code to its emoji data. - aliasMap map[string]int // the alias to its emoji data. - emptyReplacer *strings.Replacer // string replacer for emoji codes, used for finding emoji positions. - codeReplacer *strings.Replacer // string replacer for emoji codes. - aliasReplacer *strings.Replacer // string replacer for emoji aliases. + codeMap map[string]int // emoji Unicode code to its emoji data. + aliasMap map[string]int // the alias to its emoji data. + trie *util.TrieNode // trie for finding emoji positions. + isStartingByte [256]bool // fast-path skip for starting bytes. + codeReplacer *strings.Replacer // string replacer for emoji codes. + aliasReplacer *strings.Replacer // string replacer for emoji aliases. } var globalVarsStore atomic.Pointer[globalVarsStruct] @@ -44,10 +45,10 @@ func globalVars() *globalVarsStruct { vars = &globalVarsStruct{} vars.codeMap = make(map[string]int, len(GemojiData)) vars.aliasMap = make(map[string]int, len(GemojiData)) + vars.trie = &util.TrieNode{} // process emoji codes and aliases codePairs := make([]string, 0) - emptyPairs := make([]string, 0) aliasPairs := make([]string, 0) // sort from largest to small so we match combined emoji first @@ -81,20 +82,20 @@ func globalVars() *globalVarsStruct { if firstAlias != "" { vars.codeMap[emoji.Emoji] = idx codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":") - emptyPairs = append(emptyPairs, emoji.Emoji, emoji.Emoji) + vars.trie.Insert(emoji.Emoji) + vars.isStartingByte[emoji.Emoji[0]] = true } } // create replacers - vars.emptyReplacer = strings.NewReplacer(emptyPairs...) vars.codeReplacer = strings.NewReplacer(codePairs...) vars.aliasReplacer = strings.NewReplacer(aliasPairs...) globalVarsStore.Store(vars) return vars } -// FromCode retrieves the emoji data based on the provided unicode code (ie, -// "\u2618" will return the Gemoji data for "shamrock"). +// FromCode retrieves the emoji data based on the provided Unicode code +// e.g.: "\u2618" will return the Gemoji data for "shamrock". func FromCode(code string) *Emoji { i, ok := globalVars().codeMap[code] if !ok { @@ -104,9 +105,8 @@ func FromCode(code string) *Emoji { return &GemojiData[i] } -// FromAlias retrieves the emoji data based on the provided alias in the form -// "alias" or ":alias:" (ie, "shamrock" or ":shamrock:" will return the Gemoji -// data for "shamrock"). +// FromAlias retrieves the emoji data based on the provided alias in the form "alias" or ":alias:" +// e.g.: "shamrock" or ":shamrock:" will return the Gemoji data for "shamrock". func FromAlias(alias string) *Emoji { if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") { alias = alias[1 : len(alias)-1] @@ -120,69 +120,27 @@ func FromAlias(alias string) *Emoji { return &GemojiData[i] } -// ReplaceCodes replaces all emoji codes with the first corresponding emoji -// alias (in the form of ":alias:") (ie, "\u2618" will be converted to -// ":shamrock:"). +// ReplaceCodes replaces all emoji codes with the first corresponding emoji alias in the form of ":alias:" +// e.g.: "\u2618" will be converted to ":shamrock:". func ReplaceCodes(s string) string { return globalVars().codeReplacer.Replace(s) } -// ReplaceAliases replaces all aliases of the form ":alias:" with its -// corresponding unicode value. +// ReplaceAliases replaces all aliases of the form ":alias:" with its corresponding Unicode value. func ReplaceAliases(s string) string { return globalVars().aliasReplacer.Replace(s) } -type rememberSecondWriteWriter struct { - pos int - idx int - end int - writecount int -} - -func (n *rememberSecondWriteWriter) Write(p []byte) (int, error) { - n.writecount++ - if n.writecount == 2 { - n.idx = n.pos - n.end = n.pos + len(p) - n.pos += len(p) - return len(p), io.EOF - } - n.pos += len(p) - return len(p), nil -} - -func (n *rememberSecondWriteWriter) WriteString(s string) (int, error) { - n.writecount++ - if n.writecount == 2 { - n.idx = n.pos - n.end = n.pos + len(s) - n.pos += len(s) - return len(s), io.EOF - } - n.pos += len(s) - return len(s), nil -} - -// FindEmojiSubmatchIndex returns index pair of longest emoji in a string +// FindEmojiSubmatchIndex returns index-pair of the first emoji in a string func FindEmojiSubmatchIndex(s string) []int { - secondWriteWriter := rememberSecondWriteWriter{} - - // A faster and clean implementation would copy the trie tree formation in strings.NewReplacer but - // we can be lazy here. - // - // The implementation of strings.Replacer.WriteString is such that the first index of the emoji - // submatch is simply the second thing that is written to WriteString in the writer. - // - // Therefore we can simply take the index of the second write as our first emoji - // - // FIXME: just copy the trie implementation from strings.NewReplacer - _, _ = globalVars().emptyReplacer.WriteString(&secondWriteWriter, s) - - // if we wrote less than twice then we never "replaced" - if secondWriteWriter.writecount < 2 { - return nil + vars := globalVars() + for i := 0; i < len(s); i++ { + if !vars.isStartingByte[s[i]] { + continue + } + if matchLen := vars.trie.Match(s, i); matchLen > 0 { + return []int{i, i + matchLen} + } } - - return []int{secondWriteWriter.idx, secondWriteWriter.end} + return nil } diff --git a/modules/emoji/emoji_test.go b/modules/emoji/emoji_test.go index e33239bf19..f6be671759 100644 --- a/modules/emoji/emoji_test.go +++ b/modules/emoji/emoji_test.go @@ -61,13 +61,18 @@ func TestReplacers(t *testing.T) { } } +const ( + testInputWithEmojis = "This is a test string containing some emojis like \U0001f44d and \U0001f37a and some text in between." + testInputNoEmojis = "This is a test string containing no emojis at all, just plain old ASCII text, which should ideally be scanned very quickly by our trie implementation." +) + func TestFindEmojiSubmatchIndex(t *testing.T) { type testcase struct { - teststring string - expected []int + input string + expected []int } - testcases := []testcase{ + testCases := []testcase{ { "\U0001f44d", []int{0, len("\U0001f44d")}, @@ -81,13 +86,40 @@ func TestFindEmojiSubmatchIndex(t *testing.T) { []int{1, 1 + len("\U0001f44d")}, }, { - string([]byte{'\u0001'}) + "\U0001f44d", + "\u0001\U0001f44d", []int{1, 1 + len("\U0001f44d")}, }, + { + // This package can handle keycap emoji if it is registered in the emoji data. + // However, many other places (e.g.: markup rendering) also might not handle such cases correctly. + // For example: how is "**{U+FE0F}{U+20E3}**" rendered in Markdown/Markup? + "a 8\U0000fe0f\U000020e3 b", // keycap emoji "8\ufe0f\u20e3" in emoji data + []int{2, 2 + len("8\U0000fe0f\U000020e3")}, + }, + { + testInputWithEmojis, + []int{50, 54}, + }, + { + testInputNoEmojis, + nil, + }, } - for _, kase := range testcases { - actual := FindEmojiSubmatchIndex(kase.teststring) - assert.Equal(t, kase.expected, actual) + for _, tc := range testCases { + actual := FindEmojiSubmatchIndex(tc.input) + assert.Equal(t, tc.expected, actual) + } +} + +func BenchmarkFindEmojiSubmatchIndex(b *testing.B) { + for i := 0; i < b.N; i++ { + _ = FindEmojiSubmatchIndex(testInputWithEmojis) + } +} + +func BenchmarkFindEmojiSubmatchIndexNoMatch(b *testing.B) { + for i := 0; i < b.N; i++ { + _ = FindEmojiSubmatchIndex(testInputNoEmojis) } } diff --git a/modules/util/trie.go b/modules/util/trie.go new file mode 100644 index 0000000000..1ae384eee6 --- /dev/null +++ b/modules/util/trie.go @@ -0,0 +1,56 @@ +// Copyright 2026 The Gitea Authors. All rights reserved. +// SPDX-License-Identifier: MIT + +package util + +type trieEdge struct { + b byte + node *TrieNode +} + +// TrieNode represents a node in a slice-based Trie for fast byte-sequence prefix matching. +type TrieNode struct { + children []trieEdge + isEnd bool +} + +func (t *TrieNode) child(b byte) *TrieNode { + for _, edge := range t.children { + if edge.b == b { + return edge.node + } + } + return nil +} + +// Insert adds a string (byte sequence) to the Trie. +func (t *TrieNode) Insert(val string) { + curr := t + for i := 0; i < len(val); i++ { + next := curr.child(val[i]) + if next == nil { + next = &TrieNode{} + curr.children = append(curr.children, trieEdge{b: val[i], node: next}) + } + curr = next + } + curr.isEnd = true +} + +// Match returns the length of the longest matching prefix starting at index `start` in string `s`. +// It returns -1 if no prefix is matched. +func (t *TrieNode) Match(s string, start int) int { + curr := t + matchLen := -1 + for j := start; j < len(s); j++ { + next := curr.child(s[j]) + if next == nil { + break + } + curr = next + if curr.isEnd { + matchLen = j - start + 1 + } + } + return matchLen +} diff --git a/modules/util/trie_test.go b/modules/util/trie_test.go new file mode 100644 index 0000000000..b06b1a6ad7 --- /dev/null +++ b/modules/util/trie_test.go @@ -0,0 +1,34 @@ +// Copyright 2026 The Gitea Authors. All rights reserved. +// SPDX-License-Identifier: MIT + +package util + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestTrie(t *testing.T) { + trie := &TrieNode{} + trie.Insert("apple") + trie.Insert("apricot") + trie.Insert("banana") + trie.Insert("app") + + // Test exact matches + assert.Equal(t, 5, trie.Match("apple", 0)) + assert.Equal(t, 7, trie.Match("apricot", 0)) + assert.Equal(t, 6, trie.Match("banana", 0)) + assert.Equal(t, 3, trie.Match("app", 0)) + + // Test partial match (longest match priority) + assert.Equal(t, 5, trie.Match("apple-pie", 0)) + + // Test suffix/nested position match + assert.Equal(t, 5, trie.Match("sweet apple", 6)) + + // Test no match + assert.Equal(t, -1, trie.Match("orange", 0)) + assert.Equal(t, -1, trie.Match("ap", 0)) // prefix exists but is not an end node +}