mirror of
https://github.com/odin-lang/Odin.git
synced 2026-08-14 01:34:35 +00:00
Merge pull request #7274 from karl-zylinski/claude/grapheme-clusters-text-field-e36817
Add `text` field to `utf8.Grapheme`, fix caret translation by grapheme
This commit is contained in:
@@ -311,7 +311,7 @@ translate_position :: proc(s: ^State, t: Translation) -> int {
|
||||
for {
|
||||
_, g = utf8.decode_grapheme_iterate(&it) or_break
|
||||
}
|
||||
pos -= max(g.width, 1)
|
||||
pos -= max(len(g.text), 1)
|
||||
} else {
|
||||
pos -= 1
|
||||
for pos >= 0 && is_continuation_byte(buf[pos]) {
|
||||
@@ -323,7 +323,7 @@ translate_position :: proc(s: ^State, t: Translation) -> int {
|
||||
it := utf8.decode_grapheme_iterator_make(string(buf[pos:]))
|
||||
|
||||
_, g, _ := utf8.decode_grapheme_iterate(&it)
|
||||
pos += max(g.width, 1)
|
||||
pos += max(len(g.text), 1)
|
||||
} else {
|
||||
pos += 1
|
||||
for pos < len(buf) && is_continuation_byte(buf[pos]) {
|
||||
|
||||
@@ -21,6 +21,7 @@ normalized_east_asian_width :: unicode.normalized_east_asian_width
|
||||
|
||||
|
||||
Grapheme :: struct {
|
||||
text: string, // The text of the grapheme, a slice of the string it was decoded from.
|
||||
byte_index: int,
|
||||
rune_index: int,
|
||||
width: int,
|
||||
@@ -152,10 +153,11 @@ decode_grapheme_iterate :: proc(it: ^Grapheme_Iterator) -> (text: string, graphe
|
||||
it.width += normalized_east_asian_width(this_rune)
|
||||
if it.continue_grapheme {
|
||||
grapheme = it.current_grapheme
|
||||
text = it.str[it.current_grapheme.byte_index:byte_index]
|
||||
grapheme.text = it.str[it.current_grapheme.byte_index:byte_index]
|
||||
text = grapheme.text
|
||||
ok = true
|
||||
}
|
||||
it.current_grapheme = Grapheme{byte_index, it.rune_count, it.width - it.last_width}
|
||||
it.current_grapheme = Grapheme{byte_index = byte_index, rune_index = it.rune_count, width = it.width - it.last_width}
|
||||
it.continue_grapheme = true
|
||||
|
||||
|
||||
@@ -392,7 +394,8 @@ decode_grapheme_iterate :: proc(it: ^Grapheme_Iterator) -> (text: string, graphe
|
||||
// a new grapheme is encountered.
|
||||
if !ok && it.continue_grapheme {
|
||||
grapheme = it.current_grapheme
|
||||
text = it.str[it.current_grapheme.byte_index:]
|
||||
grapheme.text = it.str[it.current_grapheme.byte_index:]
|
||||
text = grapheme.text
|
||||
ok = true
|
||||
it.continue_grapheme = false
|
||||
}
|
||||
|
||||
@@ -50,6 +50,7 @@ download_assets :: proc "contextless" () {
|
||||
@(require) import "sys/posix"
|
||||
@(require) import "sys/kqueue"
|
||||
@(require) import "sys/windows"
|
||||
@(require) import "text/edit"
|
||||
@(require) import "text/i18n"
|
||||
@(require) import "text/match"
|
||||
@(require) import "text/regex"
|
||||
|
||||
113
tests/core/text/edit/test_core_text_edit.odin
Normal file
113
tests/core/text/edit/test_core_text_edit.odin
Normal file
@@ -0,0 +1,113 @@
|
||||
package test_core_text_edit
|
||||
|
||||
import "core:slice"
|
||||
import "core:strings"
|
||||
import "core:testing"
|
||||
import "core:text/edit"
|
||||
|
||||
// "hi", a thumbs-up with a skin tone modifier, an "e" with a combining acute, and "!".
|
||||
// The escapes are spelled out so the byte offsets below do not depend on how this
|
||||
// file happens to be normalized.
|
||||
//
|
||||
// byte: 0 1 2 6 10 11 13 14
|
||||
// rune: h i U+1F44D U+1F3FD e U+0301 !
|
||||
GRAPHEME_SAMPLE :: "hi\U0001F44D\U0001F3FDe\u0301!"
|
||||
|
||||
WORD_SAMPLE :: "foo bar baz"
|
||||
|
||||
State :: struct {
|
||||
state: edit.State,
|
||||
builder: strings.Builder,
|
||||
}
|
||||
|
||||
state_init :: proc(s: ^State, str: string, translate_by_grapheme: bool) {
|
||||
s.builder = strings.builder_make()
|
||||
strings.write_string(&s.builder, str)
|
||||
|
||||
edit.init(&s.state, context.allocator, context.allocator)
|
||||
edit.setup_once(&s.state, &s.builder)
|
||||
s.state.translate_by_grapheme = translate_by_grapheme
|
||||
}
|
||||
|
||||
state_destroy :: proc(s: ^State) {
|
||||
edit.destroy(&s.state)
|
||||
strings.builder_destroy(&s.builder)
|
||||
}
|
||||
|
||||
// Walk the caret from `start` in direction `t` until it stops moving, collecting
|
||||
// every position it comes to rest on.
|
||||
walk :: proc(s: ^State, start: int, t: edit.Translation) -> (stops: [dynamic]int) {
|
||||
s.state.selection = {start, start}
|
||||
for {
|
||||
prev := s.state.selection[0]
|
||||
edit.move_to(&s.state, t)
|
||||
if s.state.selection[0] == prev {
|
||||
return
|
||||
}
|
||||
append(&stops, s.state.selection[0])
|
||||
}
|
||||
}
|
||||
|
||||
expect_walk :: proc(t: ^testing.T, s: ^State, start: int, translation: edit.Translation, expected: []int) {
|
||||
stops := walk(s, start, translation)
|
||||
defer delete(stops)
|
||||
|
||||
testing.expectf(t, slice.equal(stops[:], expected),
|
||||
"%v from %v: expected stops %v, got %v", translation, start, expected, stops[:])
|
||||
}
|
||||
|
||||
// Moving by grapheme must stop on grapheme cluster boundaries, never inside the
|
||||
// emoji modifier sequence or between a base rune and its combining marks.
|
||||
@(test)
|
||||
test_translate_by_grapheme :: proc(t: ^testing.T) {
|
||||
s: State
|
||||
state_init(&s, GRAPHEME_SAMPLE, true)
|
||||
defer state_destroy(&s)
|
||||
|
||||
expect_walk(t, &s, 0, .Right, {1, 2, 10, 13, 14})
|
||||
expect_walk(t, &s, len(GRAPHEME_SAMPLE), .Left, {13, 10, 2, 1, 0})
|
||||
}
|
||||
|
||||
// The default translation moves by codepoint, so it steps through the two runes
|
||||
// of the emoji sequence and the two runes of "e" + combining acute separately.
|
||||
@(test)
|
||||
test_translate_by_codepoint :: proc(t: ^testing.T) {
|
||||
s: State
|
||||
state_init(&s, GRAPHEME_SAMPLE, false)
|
||||
defer state_destroy(&s)
|
||||
|
||||
expect_walk(t, &s, 0, .Right, {1, 2, 6, 10, 11, 13, 14})
|
||||
expect_walk(t, &s, len(GRAPHEME_SAMPLE), .Left, {13, 11, 10, 6, 2, 1, 0})
|
||||
}
|
||||
|
||||
@(test)
|
||||
test_translate_by_word :: proc(t: ^testing.T) {
|
||||
s: State
|
||||
state_init(&s, WORD_SAMPLE, false)
|
||||
defer state_destroy(&s)
|
||||
|
||||
expect_walk(t, &s, 0, .Word_Right, {4, 9, 12})
|
||||
expect_walk(t, &s, len(WORD_SAMPLE), .Word_Left, {9, 4, 0})
|
||||
|
||||
// From inside "bar", to the edges of that word.
|
||||
s.state.selection = {5, 5}
|
||||
testing.expect_value(t, edit.translate_position(&s.state, .Word_Start), 4)
|
||||
testing.expect_value(t, edit.translate_position(&s.state, .Word_End), 7)
|
||||
}
|
||||
|
||||
@(test)
|
||||
test_translate_to_bounds :: proc(t: ^testing.T) {
|
||||
s: State
|
||||
state_init(&s, GRAPHEME_SAMPLE, true)
|
||||
defer state_destroy(&s)
|
||||
|
||||
s.state.selection = {5, 5}
|
||||
testing.expect_value(t, edit.translate_position(&s.state, .Start), 0)
|
||||
testing.expect_value(t, edit.translate_position(&s.state, .End), len(GRAPHEME_SAMPLE))
|
||||
|
||||
// Translating past either end must clamp rather than run off the buffer.
|
||||
s.state.selection = {0, 0}
|
||||
testing.expect_value(t, edit.translate_position(&s.state, .Left), 0)
|
||||
s.state.selection = {len(GRAPHEME_SAMPLE), len(GRAPHEME_SAMPLE)}
|
||||
testing.expect_value(t, edit.translate_position(&s.state, .Right), len(GRAPHEME_SAMPLE))
|
||||
}
|
||||
Reference in New Issue
Block a user