diff --git a/core/text/edit/text_edit.odin b/core/text/edit/text_edit.odin index 58e184309..eebe938f4 100644 --- a/core/text/edit/text_edit.odin +++ b/core/text/edit/text_edit.odin @@ -311,7 +311,7 @@ translate_position :: proc(s: ^State, t: Translation) -> int { for { _, g = utf8.decode_grapheme_iterate(&it) or_break } - pos -= max(g.width, 1) + pos -= max(len(g.text), 1) } else { pos -= 1 for pos >= 0 && is_continuation_byte(buf[pos]) { @@ -323,7 +323,7 @@ translate_position :: proc(s: ^State, t: Translation) -> int { it := utf8.decode_grapheme_iterator_make(string(buf[pos:])) _, g, _ := utf8.decode_grapheme_iterate(&it) - pos += max(g.width, 1) + pos += max(len(g.text), 1) } else { pos += 1 for pos < len(buf) && is_continuation_byte(buf[pos]) { diff --git a/core/unicode/utf8/grapheme.odin b/core/unicode/utf8/grapheme.odin index bcd234972..c4ef3767e 100644 --- a/core/unicode/utf8/grapheme.odin +++ b/core/unicode/utf8/grapheme.odin @@ -21,6 +21,7 @@ normalized_east_asian_width :: unicode.normalized_east_asian_width Grapheme :: struct { + text: string, // The text of the grapheme, a slice of the string it was decoded from. byte_index: int, rune_index: int, width: int, @@ -152,10 +153,11 @@ decode_grapheme_iterate :: proc(it: ^Grapheme_Iterator) -> (text: string, graphe it.width += normalized_east_asian_width(this_rune) if it.continue_grapheme { grapheme = it.current_grapheme - text = it.str[it.current_grapheme.byte_index:byte_index] + grapheme.text = it.str[it.current_grapheme.byte_index:byte_index] + text = grapheme.text ok = true } - it.current_grapheme = Grapheme{byte_index, it.rune_count, it.width - it.last_width} + it.current_grapheme = Grapheme{byte_index = byte_index, rune_index = it.rune_count, width = it.width - it.last_width} it.continue_grapheme = true @@ -392,7 +394,8 @@ decode_grapheme_iterate :: proc(it: ^Grapheme_Iterator) -> (text: string, graphe // a new grapheme is encountered. if !ok && it.continue_grapheme { grapheme = it.current_grapheme - text = it.str[it.current_grapheme.byte_index:] + grapheme.text = it.str[it.current_grapheme.byte_index:] + text = grapheme.text ok = true it.continue_grapheme = false } diff --git a/tests/core/normal.odin b/tests/core/normal.odin index 99a30343c..303acd232 100644 --- a/tests/core/normal.odin +++ b/tests/core/normal.odin @@ -50,6 +50,7 @@ download_assets :: proc "contextless" () { @(require) import "sys/posix" @(require) import "sys/kqueue" @(require) import "sys/windows" +@(require) import "text/edit" @(require) import "text/i18n" @(require) import "text/match" @(require) import "text/regex" diff --git a/tests/core/text/edit/test_core_text_edit.odin b/tests/core/text/edit/test_core_text_edit.odin new file mode 100644 index 000000000..ab651b3b9 --- /dev/null +++ b/tests/core/text/edit/test_core_text_edit.odin @@ -0,0 +1,113 @@ +package test_core_text_edit + +import "core:slice" +import "core:strings" +import "core:testing" +import "core:text/edit" + +// "hi", a thumbs-up with a skin tone modifier, an "e" with a combining acute, and "!". +// The escapes are spelled out so the byte offsets below do not depend on how this +// file happens to be normalized. +// +// byte: 0 1 2 6 10 11 13 14 +// rune: h i U+1F44D U+1F3FD e U+0301 ! +GRAPHEME_SAMPLE :: "hi\U0001F44D\U0001F3FDe\u0301!" + +WORD_SAMPLE :: "foo bar baz" + +State :: struct { + state: edit.State, + builder: strings.Builder, +} + +state_init :: proc(s: ^State, str: string, translate_by_grapheme: bool) { + s.builder = strings.builder_make() + strings.write_string(&s.builder, str) + + edit.init(&s.state, context.allocator, context.allocator) + edit.setup_once(&s.state, &s.builder) + s.state.translate_by_grapheme = translate_by_grapheme +} + +state_destroy :: proc(s: ^State) { + edit.destroy(&s.state) + strings.builder_destroy(&s.builder) +} + +// Walk the caret from `start` in direction `t` until it stops moving, collecting +// every position it comes to rest on. +walk :: proc(s: ^State, start: int, t: edit.Translation) -> (stops: [dynamic]int) { + s.state.selection = {start, start} + for { + prev := s.state.selection[0] + edit.move_to(&s.state, t) + if s.state.selection[0] == prev { + return + } + append(&stops, s.state.selection[0]) + } +} + +expect_walk :: proc(t: ^testing.T, s: ^State, start: int, translation: edit.Translation, expected: []int) { + stops := walk(s, start, translation) + defer delete(stops) + + testing.expectf(t, slice.equal(stops[:], expected), + "%v from %v: expected stops %v, got %v", translation, start, expected, stops[:]) +} + +// Moving by grapheme must stop on grapheme cluster boundaries, never inside the +// emoji modifier sequence or between a base rune and its combining marks. +@(test) +test_translate_by_grapheme :: proc(t: ^testing.T) { + s: State + state_init(&s, GRAPHEME_SAMPLE, true) + defer state_destroy(&s) + + expect_walk(t, &s, 0, .Right, {1, 2, 10, 13, 14}) + expect_walk(t, &s, len(GRAPHEME_SAMPLE), .Left, {13, 10, 2, 1, 0}) +} + +// The default translation moves by codepoint, so it steps through the two runes +// of the emoji sequence and the two runes of "e" + combining acute separately. +@(test) +test_translate_by_codepoint :: proc(t: ^testing.T) { + s: State + state_init(&s, GRAPHEME_SAMPLE, false) + defer state_destroy(&s) + + expect_walk(t, &s, 0, .Right, {1, 2, 6, 10, 11, 13, 14}) + expect_walk(t, &s, len(GRAPHEME_SAMPLE), .Left, {13, 11, 10, 6, 2, 1, 0}) +} + +@(test) +test_translate_by_word :: proc(t: ^testing.T) { + s: State + state_init(&s, WORD_SAMPLE, false) + defer state_destroy(&s) + + expect_walk(t, &s, 0, .Word_Right, {4, 9, 12}) + expect_walk(t, &s, len(WORD_SAMPLE), .Word_Left, {9, 4, 0}) + + // From inside "bar", to the edges of that word. + s.state.selection = {5, 5} + testing.expect_value(t, edit.translate_position(&s.state, .Word_Start), 4) + testing.expect_value(t, edit.translate_position(&s.state, .Word_End), 7) +} + +@(test) +test_translate_to_bounds :: proc(t: ^testing.T) { + s: State + state_init(&s, GRAPHEME_SAMPLE, true) + defer state_destroy(&s) + + s.state.selection = {5, 5} + testing.expect_value(t, edit.translate_position(&s.state, .Start), 0) + testing.expect_value(t, edit.translate_position(&s.state, .End), len(GRAPHEME_SAMPLE)) + + // Translating past either end must clamp rather than run off the buffer. + s.state.selection = {0, 0} + testing.expect_value(t, edit.translate_position(&s.state, .Left), 0) + s.state.selection = {len(GRAPHEME_SAMPLE), len(GRAPHEME_SAMPLE)} + testing.expect_value(t, edit.translate_position(&s.state, .Right), len(GRAPHEME_SAMPLE)) +}