Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
96 changes: 27 additions & 69 deletions modules/emoji/emoji.go
Original file line number Diff line number Diff line change
Expand Up @@ -5,12 +5,12 @@
package emoji

import (
"io"
"sort"
"strings"
"sync/atomic"

"gitea.dev/modules/setting"
"gitea.dev/modules/util"
)

// Gemoji is a set of emoji data.
Expand All @@ -26,11 +26,12 @@ type Emoji struct {
}

type globalVarsStruct struct {
codeMap map[string]int // emoji unicode code to its emoji data.
aliasMap map[string]int // the alias to its emoji data.
emptyReplacer *strings.Replacer // string replacer for emoji codes, used for finding emoji positions.
codeReplacer *strings.Replacer // string replacer for emoji codes.
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
codeMap map[string]int // emoji Unicode code to its emoji data.
aliasMap map[string]int // the alias to its emoji data.
trie *util.TrieNode // trie for finding emoji positions.
isStartingByte [256]bool // fast-path skip for starting bytes.
codeReplacer *strings.Replacer // string replacer for emoji codes.
aliasReplacer *strings.Replacer // string replacer for emoji aliases.
}

var globalVarsStore atomic.Pointer[globalVarsStruct]
Expand All @@ -44,10 +45,10 @@ func globalVars() *globalVarsStruct {
vars = &globalVarsStruct{}
vars.codeMap = make(map[string]int, len(GemojiData))
vars.aliasMap = make(map[string]int, len(GemojiData))
vars.trie = &util.TrieNode{}

// process emoji codes and aliases
codePairs := make([]string, 0)
emptyPairs := make([]string, 0)
aliasPairs := make([]string, 0)

// sort from largest to small so we match combined emoji first
Expand Down Expand Up @@ -81,20 +82,20 @@ func globalVars() *globalVarsStruct {
if firstAlias != "" {
vars.codeMap[emoji.Emoji] = idx
codePairs = append(codePairs, emoji.Emoji, ":"+emoji.Aliases[0]+":")
emptyPairs = append(emptyPairs, emoji.Emoji, emoji.Emoji)
vars.trie.Insert(emoji.Emoji)
vars.isStartingByte[emoji.Emoji[0]] = true
}
}

// create replacers
vars.emptyReplacer = strings.NewReplacer(emptyPairs...)
vars.codeReplacer = strings.NewReplacer(codePairs...)
vars.aliasReplacer = strings.NewReplacer(aliasPairs...)
globalVarsStore.Store(vars)
return vars
}

// FromCode retrieves the emoji data based on the provided unicode code (ie,
// "\u2618" will return the Gemoji data for "shamrock").
// FromCode retrieves the emoji data based on the provided Unicode code
// e.g.: "\u2618" will return the Gemoji data for "shamrock".
func FromCode(code string) *Emoji {
i, ok := globalVars().codeMap[code]
if !ok {
Expand All @@ -104,9 +105,8 @@ func FromCode(code string) *Emoji {
return &GemojiData[i]
}

// FromAlias retrieves the emoji data based on the provided alias in the form
// "alias" or ":alias:" (ie, "shamrock" or ":shamrock:" will return the Gemoji
// data for "shamrock").
// FromAlias retrieves the emoji data based on the provided alias in the form "alias" or ":alias:"
// e.g.: "shamrock" or ":shamrock:" will return the Gemoji data for "shamrock".
func FromAlias(alias string) *Emoji {
if strings.HasPrefix(alias, ":") && strings.HasSuffix(alias, ":") {
alias = alias[1 : len(alias)-1]
Expand All @@ -120,69 +120,27 @@ func FromAlias(alias string) *Emoji {
return &GemojiData[i]
}

// ReplaceCodes replaces all emoji codes with the first corresponding emoji
// alias (in the form of ":alias:") (ie, "\u2618" will be converted to
// ":shamrock:").
// ReplaceCodes replaces all emoji codes with the first corresponding emoji alias in the form of ":alias:"
// e.g.: "\u2618" will be converted to ":shamrock:".
func ReplaceCodes(s string) string {
return globalVars().codeReplacer.Replace(s)
}

// ReplaceAliases replaces all aliases of the form ":alias:" with its
// corresponding unicode value.
// ReplaceAliases replaces all aliases of the form ":alias:" with its corresponding Unicode value.
func ReplaceAliases(s string) string {
return globalVars().aliasReplacer.Replace(s)
}

type rememberSecondWriteWriter struct {
pos int
idx int
end int
writecount int
}

func (n *rememberSecondWriteWriter) Write(p []byte) (int, error) {
n.writecount++
if n.writecount == 2 {
n.idx = n.pos
n.end = n.pos + len(p)
n.pos += len(p)
return len(p), io.EOF
}
n.pos += len(p)
return len(p), nil
}

func (n *rememberSecondWriteWriter) WriteString(s string) (int, error) {
n.writecount++
if n.writecount == 2 {
n.idx = n.pos
n.end = n.pos + len(s)
n.pos += len(s)
return len(s), io.EOF
}
n.pos += len(s)
return len(s), nil
}

// FindEmojiSubmatchIndex returns index pair of longest emoji in a string
// FindEmojiSubmatchIndex returns index-pair of the first emoji in a string
func FindEmojiSubmatchIndex(s string) []int {
secondWriteWriter := rememberSecondWriteWriter{}

// A faster and clean implementation would copy the trie tree formation in strings.NewReplacer but
// we can be lazy here.
//
// The implementation of strings.Replacer.WriteString is such that the first index of the emoji
// submatch is simply the second thing that is written to WriteString in the writer.
//
// Therefore we can simply take the index of the second write as our first emoji
//
// FIXME: just copy the trie implementation from strings.NewReplacer
_, _ = globalVars().emptyReplacer.WriteString(&secondWriteWriter, s)

// if we wrote less than twice then we never "replaced"
if secondWriteWriter.writecount < 2 {
return nil
vars := globalVars()
for i := 0; i < len(s); i++ {
if !vars.isStartingByte[s[i]] {
continue
}
if matchLen := vars.trie.Match(s, i); matchLen > 0 {
return []int{i, i + matchLen}
}
}

return []int{secondWriteWriter.idx, secondWriteWriter.end}
return nil
}
46 changes: 39 additions & 7 deletions modules/emoji/emoji_test.go
Original file line number Diff line number Diff line change
Expand Up @@ -61,13 +61,18 @@ func TestReplacers(t *testing.T) {
}
}

const (
testInputWithEmojis = "This is a test string containing some emojis like \U0001f44d and \U0001f37a and some text in between."
testInputNoEmojis = "This is a test string containing no emojis at all, just plain old ASCII text, which should ideally be scanned very quickly by our trie implementation."
)

func TestFindEmojiSubmatchIndex(t *testing.T) {
type testcase struct {
teststring string
expected []int
input string
expected []int
}

testcases := []testcase{
testCases := []testcase{
{
"\U0001f44d",
[]int{0, len("\U0001f44d")},
Expand All @@ -81,13 +86,40 @@ func TestFindEmojiSubmatchIndex(t *testing.T) {
[]int{1, 1 + len("\U0001f44d")},
},
{
string([]byte{'\u0001'}) + "\U0001f44d",
"\u0001\U0001f44d",
[]int{1, 1 + len("\U0001f44d")},
},
{
// This package can handle keycap emoji if it is registered in the emoji data.
// However, many other places (e.g.: markup rendering) also might not handle such cases correctly.
// For example: how is "**{U+FE0F}{U+20E3}**" rendered in Markdown/Markup?
"a 8\U0000fe0f\U000020e3 b", // keycap emoji "8\ufe0f\u20e3" in emoji data
[]int{2, 2 + len("8\U0000fe0f\U000020e3")},
},
{
testInputWithEmojis,
[]int{50, 54},
},
{
testInputNoEmojis,
nil,
},
}

for _, tc := range testCases {
actual := FindEmojiSubmatchIndex(tc.input)
assert.Equal(t, tc.expected, actual)
}
}

func BenchmarkFindEmojiSubmatchIndex(b *testing.B) {
for i := 0; i < b.N; i++ {
_ = FindEmojiSubmatchIndex(testInputWithEmojis)
}
}

for _, kase := range testcases {
actual := FindEmojiSubmatchIndex(kase.teststring)
assert.Equal(t, kase.expected, actual)
func BenchmarkFindEmojiSubmatchIndexNoMatch(b *testing.B) {
for i := 0; i < b.N; i++ {
_ = FindEmojiSubmatchIndex(testInputNoEmojis)
}
}
56 changes: 56 additions & 0 deletions modules/util/trie.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,56 @@
// Copyright 2026 The Gitea Authors. All rights reserved.
// SPDX-License-Identifier: MIT

package util

type trieEdge struct {
b byte
node *TrieNode
}

// TrieNode represents a node in a slice-based Trie for fast byte-sequence prefix matching.
type TrieNode struct {
children []trieEdge
isEnd bool
}

func (t *TrieNode) child(b byte) *TrieNode {
for _, edge := range t.children {
if edge.b == b {
return edge.node
}
}
return nil
}

// Insert adds a string (byte sequence) to the Trie.
func (t *TrieNode) Insert(val string) {
curr := t
for i := 0; i < len(val); i++ {
next := curr.child(val[i])
if next == nil {
next = &TrieNode{}
curr.children = append(curr.children, trieEdge{b: val[i], node: next})
}
curr = next
}
curr.isEnd = true
}

// Match returns the length of the longest matching prefix starting at index `start` in string `s`.
// It returns -1 if no prefix is matched.
func (t *TrieNode) Match(s string, start int) int {
curr := t
matchLen := -1
for j := start; j < len(s); j++ {
next := curr.child(s[j])
if next == nil {
break
}
curr = next
if curr.isEnd {
matchLen = j - start + 1
}
}
return matchLen
}
34 changes: 34 additions & 0 deletions modules/util/trie_test.go
Original file line number Diff line number Diff line change
@@ -0,0 +1,34 @@
// Copyright 2026 The Gitea Authors. All rights reserved.
// SPDX-License-Identifier: MIT

package util

import (
"testing"

"github.com/stretchr/testify/assert"
)

func TestTrie(t *testing.T) {
trie := &TrieNode{}
trie.Insert("apple")
trie.Insert("apricot")
trie.Insert("banana")
trie.Insert("app")

// Test exact matches
assert.Equal(t, 5, trie.Match("apple", 0))
assert.Equal(t, 7, trie.Match("apricot", 0))
assert.Equal(t, 6, trie.Match("banana", 0))
assert.Equal(t, 3, trie.Match("app", 0))

// Test partial match (longest match priority)
assert.Equal(t, 5, trie.Match("apple-pie", 0))

// Test suffix/nested position match
assert.Equal(t, 5, trie.Match("sweet apple", 6))

// Test no match
assert.Equal(t, -1, trie.Match("orange", 0))
assert.Equal(t, -1, trie.Match("ap", 0)) // prefix exists but is not an end node
}
Loading