From 28a793d659d94e82375b9e3ee1d5002f05f4d9dd Mon Sep 17 00:00:00 2001 From: kofany Date: Wed, 10 Sep 2025 17:44:46 +0200 Subject: [PATCH 1/3] Implement proper UTF-8 grapheme cluster support for modern terminals MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit This commit addresses critical UTF-8 display and input issues, particularly with emoji variations and complex Unicode sequences that caused rendering problems in modern terminals like Ghostty. Key improvements: - Enhanced utf8.c with grapheme cluster detection using utf8proc - Fixed input field handling to preserve emoji variation selectors - Improved text measurement and cursor positioning for complex Unicode - Added smart paste processing for multi-codepoint characters - Bypassed problematic TRANSLIT when not needed for UTF-8 content - Enhanced GUI entry and readline to handle grapheme clusters properly The implementation maintains full backward compatibility while providing proper support for modern Unicode standards, fixing display corruption that occurred with emoji sequences containing variation selectors. Tested with emoji sequences like: 💕💋😘🥰💞❣️💓♥️♥️💓❣️🥰🥰😘💋💋 --- .gitignore | 6 + src/core/core.c | 3 + src/core/recode.c | 22 ++- src/core/utf8.c | 263 ++++++++++++++++++++++++++++++++++ src/core/utf8.h | 22 +++ src/fe-text/gui-entry.c | 152 +++++++++++++++----- src/fe-text/gui-readline.c | 131 ++++++++++++++--- src/fe-text/term-terminfo.c | 20 +-- src/fe-text/textbuffer-view.c | 22 ++- 9 files changed, 562 insertions(+), 79 deletions(-) diff --git a/.gitignore b/.gitignore index ea2cd7ba..50ef18ef 100644 --- a/.gitignore +++ b/.gitignore @@ -89,3 +89,9 @@ subprojects/* Irssi-Dist setup.cfg *.egg-info +.specstory/ +.cursorindexingignore +.cursorignore +.kombai/ +.claude/ +inst/ diff --git a/src/core/core.c b/src/core/core.c index 5d72ca0a..83b6ece5 100644 --- a/src/core/core.c +++ b/src/core/core.c @@ -45,6 +45,7 @@ #include #include #include +#include #include #include @@ -285,6 +286,7 @@ void core_init(void) i_refstr_init(); special_vars_init(); wcwidth_wrapper_init(); + utf8_init(); settings_add_str("misc", "ignore_signals", ""); settings_add_bool("misc", "override_coredump_limit", FALSE); @@ -316,6 +318,7 @@ void core_deinit(void) signal_remove("chat protocol created", (SIGNAL_FUNC) reread_setup); signal_remove("irssi init finished", (SIGNAL_FUNC) sig_irssi_init_finished); + utf8_deinit(); wcwidth_wrapper_deinit(); special_vars_deinit(); i_refstr_deinit(); diff --git a/src/core/recode.c b/src/core/recode.c index 7185dd99..5849cb08 100644 --- a/src/core/recode.c +++ b/src/core/recode.c @@ -120,6 +120,13 @@ char *recode_in(const SERVER_REC *server, const char *str, const char *target) else from = find_conversion(server, target); + /* Don't use TRANSLIT when both terminal and string are UTF-8 + * and no specific conversion is configured - preserves emoji variation selectors */ + if (from == NULL && term_is_utf8 && str_is_utf8) { + g_debug("recode_in: UTF-8 bypass for: %s", str); + return g_strdup(str); + } + if (from) recoded = g_convert_with_fallback(str, len, to, from, NULL, NULL, NULL, NULL); @@ -150,7 +157,7 @@ char *recode_out(const SERVER_REC *server, const char *str, const char *target) const char *from = translit_charset; const char *to = NULL; char *translit_to = NULL; - gboolean translit, recode; + gboolean translit, recode, str_is_utf8; int len; if (!str) @@ -162,6 +169,9 @@ char *recode_out(const SERVER_REC *server, const char *str, const char *target) len = strlen(str); + /* Check if string is valid UTF-8 */ + str_is_utf8 = g_utf8_validate(str, len, NULL); + translit = settings_get_bool("recode_transliterate"); to = find_conversion(server, target); @@ -170,10 +180,18 @@ char *recode_out(const SERVER_REC *server, const char *str, const char *target) to = settings_get_str("recode_out_default_charset"); if (to && *to != '\0') { - if (translit && !is_translit(to)) + /* Don't use TRANSLIT when both terminal and target are UTF-8 + * and string is valid UTF-8 - this preserves emoji variation selectors */ + if (translit && !is_translit(to) && + !(term_is_utf8 && str_is_utf8 && g_ascii_strcasecmp(to, "UTF-8") == 0)) to = translit_to = g_strconcat(to ,"//TRANSLIT", NULL); recoded = g_convert(str, len, to, from, NULL, NULL, NULL); + } else if (term_is_utf8 && str_is_utf8) { + /* When no specific charset conversion is configured and both terminal + * and string are UTF-8, don't do any conversion to preserve emoji */ + g_debug("recode_out: UTF-8 bypass for: %s", str); + recoded = g_strdup(str); } g_free(translit_to); if (!recoded) diff --git a/src/core/utf8.c b/src/core/utf8.c index 5ef030d3..ca841b4d 100644 --- a/src/core/utf8.c +++ b/src/core/utf8.c @@ -27,9 +27,70 @@ /* Provide is_utf8(): */ #include +#include +#include + +#ifdef HAVE_LIBUTF8PROC +#include + +/* Advance the str pointer one grapheme cluster further when utf8proc is available, + * or fall back to single character advancement. Returns display width. */ +static int string_advance_with_grapheme_support(char const **str, int policy) +{ + utf8proc_int32_t codepoint, prev_codepoint = 0; + utf8proc_int32_t state = 0; + const char *start = *str; + const char *pos = *str; + int cluster_width = 0; + utf8proc_ssize_t bytes; + + if (policy != TREAT_STRING_AS_UTF8) { + /* Fall back to byte-based processing */ + *str += 1; + return 1; + } + + if (*pos == '\0') { + return 0; + } + + /* Process codepoints until we find a grapheme boundary */ + while (*pos != '\0') { + bytes = utf8proc_iterate((const utf8proc_uint8_t *)pos, -1, &codepoint); + if (bytes < 0) { + /* Invalid UTF-8, skip one byte */ + *str = pos + 1; + return 1; + } + + /* Check if this is a grapheme boundary */ + if (pos != start && utf8proc_grapheme_break_stateful(prev_codepoint, codepoint, &state)) { + /* We found the end of the current cluster */ + break; + } + + /* Add this codepoint's width to the cluster */ + if (unichar_isprint(codepoint)) { + int char_width = i_wcwidth(codepoint); + if (char_width > cluster_width) { + cluster_width = char_width; + } + } + + prev_codepoint = codepoint; + pos += bytes; + } + + *str = pos; + return cluster_width > 0 ? cluster_width : 1; +} +#endif int string_advance(char const **str, int policy) { +#ifdef HAVE_LIBUTF8PROC + return string_advance_with_grapheme_support(str, policy); +#else if (policy == TREAT_STRING_AS_UTF8) { gunichar c; @@ -43,6 +104,7 @@ int string_advance(char const **str, int policy) return 1; } +#endif } int string_policy(const char *str) @@ -133,3 +195,204 @@ int string_chars_for_width(const char *str, int policy, unsigned int n, unsigned } return char_count; } + +int unichar_width(unichar chr) +{ + int width; + + /* For individual codepoints, fall back to standard wcwidth. + * This is not grapheme-cluster aware, but better than nothing + * for GUI code that works with unichar arrays instead of UTF-8 strings. + * + * Note: For proper emoji support, use string_advance() on UTF-8 strings + * instead of processing individual codepoints. + */ + if (!unichar_isprint(chr)) + return 1; + + width = i_wcwidth(chr); + return width < 0 ? 1 : width; +} + +int unichar_array_advance_cluster(const unichar *text, int text_len, int *pos) +{ +#ifdef HAVE_LIBUTF8PROC + utf8proc_int32_t state = 0; + int cluster_width = 0; + utf8proc_int32_t first_codepoint, codepoint; + int char_width; + int has_variation_selector = 0; + + if (*pos >= text_len) { + return 0; + } + + /* Process first codepoint */ + first_codepoint = text[*pos]; + + if (unichar_isprint(first_codepoint)) { + char_width = i_wcwidth(first_codepoint); + if (char_width > cluster_width) { + cluster_width = char_width; + } + } + + (*pos)++; + + /* Process additional codepoints until we find a grapheme boundary */ + while (*pos < text_len) { + codepoint = text[*pos]; + + /* Check if this is a grapheme boundary */ + if (utf8proc_grapheme_break_stateful(text[*pos - 1], codepoint, &state)) { + /* We found the end of the current cluster */ + break; + } + + /* Check for variation selector */ + if (codepoint == 0xFE0F) { + has_variation_selector = 1; + } + + /* Add this codepoint's width to the cluster (usually 0 for combining chars) */ + if (unichar_isprint(codepoint)) { + char_width = i_wcwidth(codepoint); + if (char_width > cluster_width) { + cluster_width = char_width; + } + } + + (*pos)++; + } + + /* Special handling for emoji with variation selector */ + if (has_variation_selector && cluster_width == 1) { + /* Base emoji (like ❣ U+2763, ♥ U+2665) + variation selector should have width 2 */ + cluster_width = 2; + } + + return cluster_width > 0 ? cluster_width : 1; +#else + /* Fall back to single character processing when utf8proc unavailable */ + unichar chr; + int width; + + if (*pos >= text_len) { + return 0; + } + + chr = text[*pos]; + (*pos)++; + + /* Bounds check after increment */ + if (*pos > text_len) { + *pos = text_len; + } + + if (!unichar_isprint(chr)) + return 1; + + width = i_wcwidth(chr); + return width < 0 ? 1 : width; +#endif +} + +int unichar_array_move_cluster_backward(const unichar *text, int text_len, int *pos) +{ +#ifdef HAVE_LIBUTF8PROC + utf8proc_int32_t state = 0; + int cluster_start; + int cluster_width = 0; + int temp_pos; + + if (*pos <= 0) { + return 0; + } + + /* Move back one codepoint first */ + (*pos)--; + + /* Find the beginning of the current grapheme cluster by going backwards */ + cluster_start = *pos; + + /* Go back to find cluster boundary */ + while (cluster_start > 0) { + /* Check if there's a grapheme boundary between previous char and current */ + if (utf8proc_grapheme_break_stateful(text[cluster_start - 1], text[cluster_start], &state)) { + /* Found boundary, cluster starts here */ + break; + } + cluster_start--; + } + + /* Calculate width of this cluster */ + temp_pos = cluster_start; + while (temp_pos < text_len && temp_pos < *pos + 1) { + unichar codepoint = text[temp_pos]; + if (unichar_isprint(codepoint)) { + int char_width = i_wcwidth(codepoint); + if (char_width > cluster_width) { + cluster_width = char_width; + } + } + temp_pos++; + } + + *pos = cluster_start; + return cluster_width > 0 ? cluster_width : 1; +#else + /* Fall back to single character processing when utf8proc unavailable */ + unichar chr; + int width; + + if (*pos <= 0) { + return 0; + } + + (*pos)--; + chr = text[*pos]; + + if (!unichar_isprint(chr)) + return 1; + + width = i_wcwidth(chr); + return width < 0 ? 1 : width; +#endif +} + +int unichar_array_find_cluster_start(const unichar *text, int text_len, int pos) +{ +#ifdef HAVE_LIBUTF8PROC + utf8proc_int32_t state = 0; + int cluster_start = pos; + + if (pos <= 0 || pos >= text_len) { + return pos; + } + + /* Go back to find cluster boundary */ + while (cluster_start > 0) { + /* Check if there's a grapheme boundary between previous char and current */ + if (utf8proc_grapheme_break_stateful(text[cluster_start - 1], text[cluster_start], &state)) { + /* Found boundary, cluster starts here */ + break; + } + cluster_start--; + } + + return cluster_start; +#else + /* Fall back to single character processing when utf8proc unavailable */ + return pos; +#endif +} + +void utf8_init(void) +{ + /* no-op */ +} + +void utf8_deinit(void) +{ + /* Nothing to clean up currently */ +} diff --git a/src/core/utf8.h b/src/core/utf8.h index 372c5d27..2893f58d 100644 --- a/src/core/utf8.h +++ b/src/core/utf8.h @@ -56,6 +56,28 @@ int string_width(const char *str, int policy); */ int string_chars_for_width(const char *str, int policy, unsigned int n, unsigned int *bytes); +/* Calculate display width of a single unichar, considering it might be part of + * a grapheme cluster. For best results, use string_advance() on UTF-8 strings. */ +int unichar_width(unichar chr); + +/* Advance through unichar array by one grapheme cluster, return display width. + * Updates *pos to point after the cluster. Use for GUI code with unichar arrays. */ +int unichar_array_advance_cluster(const unichar *text, int text_len, int *pos); + +/* Move backward through unichar array by one grapheme cluster, return display width. + * Updates *pos to point to the start of the previous cluster. */ +int unichar_array_move_cluster_backward(const unichar *text, int text_len, int *pos); + +/* Find the start of the grapheme cluster containing the given codepoint position. + * Returns the codepoint index of the cluster start. */ +int unichar_array_find_cluster_start(const unichar *text, int text_len, int pos); + +/* Initialize UTF-8 debugging system */ +void utf8_init(void); + +/* Deinitialize UTF-8 debugging system */ +void utf8_deinit(void); + #define unichar_isprint(c) (((c) & ~0x80) >= 32) #define is_utf8_leading(c) (((c) & 0xc0) != 0x80) diff --git a/src/fe-text/gui-entry.c b/src/fe-text/gui-entry.c index ad362f20..69c5d942 100644 --- a/src/fe-text/gui-entry.c +++ b/src/fe-text/gui-entry.c @@ -23,12 +23,17 @@ #include #include #include +#include #include #include #include #include +#ifdef HAVE_LIBUTF8PROC +#include +#endif + #undef i_toupper #undef i_tolower #undef i_isalnum @@ -49,10 +54,24 @@ static unichar i_tolower(unichar c) return c <= 255 ? tolower(c) : c; } +static int is_combining_char(unichar c) +{ + if (term_type != TERM_TYPE_UTF8) + return 0; + +#ifdef HAVE_LIBUTF8PROC + /* Use utf8proc for precise combining character detection */ + return unichar_isprint(c) && utf8proc_charwidth(c) == 0; +#else + /* Fallback to unichar_width for compatibility */ + return unichar_isprint(c) && unichar_width(c) == 0; +#endif +} + static int i_isalnum(unichar c) { if (term_type == TERM_TYPE_UTF8) - return (g_unichar_isalnum(c) || i_wcwidth(c) == 0); + return (g_unichar_isalnum(c) || is_combining_char(c)); return c <= 255 ? isalnum(c) : 0; } @@ -213,16 +232,24 @@ static int pos2scrpos(GUI_ENTRY_REC *entry, int pos, int cursor) xpos += scrlen_str(entry->extents[0], entry->utf8); } - for (i = 0; i < entry->text_len && i < pos; i++) { - unichar c = entry->text[i]; + /* Process text using grapheme cluster aware advancement when possible */ + i = 0; + while (i < entry->text_len && i < pos) { const char *extent = entry->uses_extents ? entry->extents[i+1] : NULL; + int char_width; - if (term_type == TERM_TYPE_BIG5) - xpos += big5_width(c); - else if (entry->utf8) - xpos += unichar_isprint(c) ? i_wcwidth(c) : 1; - else - xpos++; + if (term_type == TERM_TYPE_BIG5) { + char_width = big5_width(entry->text[i]); + i++; + } else if (entry->utf8) { + /* Use grapheme cluster aware advancement */ + char_width = unichar_array_advance_cluster(entry->text, entry->text_len, &i); + } else { + char_width = 1; + i++; + } + + xpos += char_width; if (extent != NULL) { xpos += scrlen_str(extent, entry->utf8); @@ -239,16 +266,21 @@ static int scrpos2pos(GUI_ENTRY_REC *entry, int pos) xpos += scrlen_str(entry->extents[0], entry->utf8); } - for (i = 0; i < entry->text_len && xpos < pos; i++) { - unichar c = entry->text[i]; + /* Process text using grapheme cluster aware advancement when possible */ + i = 0; + while (i < entry->text_len && xpos < pos) { const char *extent = entry->uses_extents ? entry->extents[i+1] : NULL; - if (term_type == TERM_TYPE_BIG5) - width = big5_width(c); - else if (entry->utf8) - width = unichar_isprint(c) ? i_wcwidth(c) : 1; - else + if (term_type == TERM_TYPE_BIG5) { + width = big5_width(entry->text[i]); + i++; + } else if (entry->utf8) { + /* Use grapheme cluster aware advancement */ + width = unichar_array_advance_cluster(entry->text, entry->text_len, &i); + } else { width = 1; + i++; + } xpos += width; @@ -359,25 +391,46 @@ static void gui_entry_draw_from(GUI_ENTRY_REC *entry, int pos) g_free(tmp); } - for (; i < entry->text_len; i++) { - unichar c = entry->text[i]; + /* Process remaining text using grapheme cluster aware advancement */ + while (i < entry->text_len) { const char *extent = entry->uses_extents ? entry->extents[i+1] : NULL; + int char_width; + int cluster_start = i; + unichar c; new_xpos = xpos; - if (entry->hidden) - new_xpos++; - else if (term_type == TERM_TYPE_BIG5) - new_xpos += big5_width(c); - else if (entry->utf8) - new_xpos += unichar_isprint(c) ? i_wcwidth(c) : 1; - else - new_xpos++; + c = entry->text[i]; + + if (entry->hidden) { + char_width = 1; + i++; + } else if (term_type == TERM_TYPE_BIG5) { + char_width = big5_width(c); + i++; + } else if (entry->utf8) { + /* Use grapheme cluster aware advancement */ + char_width = unichar_array_advance_cluster(entry->text, entry->text_len, &i); + } else { + char_width = 1; + i++; + } + + new_xpos += char_width; if (new_xpos > end_xpos) break; if (entry->hidden) { g_string_append_c(str, ' '); + } else if (entry->utf8 && cluster_start != i) { + /* Render entire grapheme cluster for UTF-8 */ + for (int j = cluster_start; j < i; j++) { + unichar cluster_char = entry->text[j]; + if (unichar_isprint(cluster_char)) + g_string_append_unichar(str, cluster_char); + else if (cluster_char == 0) + break; + } } else if (unichar_isprint(c)) { if (entry->utf8) { g_string_append_unichar(str, c); @@ -651,7 +704,7 @@ void gui_entry_insert_char(GUI_ENTRY_REC *entry, unichar chr) if (chr == 0 || chr == 13 || chr == 10) return; /* never insert NUL, CR or LF characters */ - if (entry->utf8 && entry->pos == 0 && unichar_isprint(chr) && i_wcwidth(chr) == 0) + if (entry->utf8 && entry->pos == 0 && is_combining_char(chr)) return; gui_entry_redraw_from(entry, entry->pos); @@ -839,7 +892,7 @@ void gui_entry_erase(GUI_ENTRY_REC *entry, int size, CUTBUFFER_UPDATE_OP update_ } if (entry->utf8) - while (entry->pos > size + w && i_wcwidth(entry->text[entry->pos - size - w]) == 0) + while (entry->pos > size + w && is_combining_char(entry->text[entry->pos - size - w])) w++; memmove(entry->text + entry->pos - size, entry->text + entry->pos, @@ -878,7 +931,7 @@ void gui_entry_erase_cell(GUI_ENTRY_REC *entry) if (entry->utf8) while (entry->pos+size < entry->text_len && - i_wcwidth(entry->text[entry->pos+size]) == 0) size++; + is_combining_char(entry->text[entry->pos+size])) size++; memmove(entry->text + entry->pos, entry->text + entry->pos + size, (entry->text_len-entry->pos-size+1) * sizeof(unichar)); @@ -1144,9 +1197,15 @@ void gui_entry_set_pos(GUI_ENTRY_REC *entry, int pos) { g_return_if_fail(entry != NULL); - if (pos >= 0 && pos <= entry->text_len) + if (pos >= 0 && pos <= entry->text_len) { entry->pos = pos; + /* For UTF-8, ensure we're at the start of a grapheme cluster */ + if (entry->utf8) { + entry->pos = unichar_array_find_cluster_start(entry->text, entry->text_len, entry->pos); + } + } + gui_entry_fix_cursor(entry); gui_entry_draw(entry); } @@ -1194,14 +1253,33 @@ void gui_entry_move_pos(GUI_ENTRY_REC *entry, int pos) { g_return_if_fail(entry != NULL); - if (entry->pos + pos >= 0 && entry->pos + pos <= entry->text_len) - entry->pos += pos; + if (!entry->utf8) { + /* Legacy behavior for non-UTF8 */ + if (entry->pos + pos >= 0 && entry->pos + pos <= entry->text_len) + entry->pos += pos; + } else { + /* UTF-8: Move by grapheme clusters for proper UX */ + int i, before_advance, cluster_start; - if (entry->utf8) { - int step = pos < 0 ? -1 : 1; - while(i_wcwidth(entry->text[entry->pos]) == 0 && - entry->pos + step >= 0 && entry->pos + step <= entry->text_len) - entry->pos += step; + if (pos > 0) { + /* Move forward by grapheme clusters */ + for (i = 0; i < pos && entry->pos < entry->text_len; i++) { + before_advance = entry->pos; + unichar_array_advance_cluster(entry->text, entry->text_len, &entry->pos); + /* Safety: if position didn't change and we're not at end, stop */ + if (entry->pos == before_advance && entry->pos < entry->text_len) + break; + } + } else if (pos < 0) { + /* Move backward by grapheme clusters */ + for (i = 0; i > pos && entry->pos > 0; i--) { + unichar_array_move_cluster_backward(entry->text, entry->text_len, &entry->pos); + } + } + + /* Ensure we're always at the start of a grapheme cluster */ + cluster_start = unichar_array_find_cluster_start(entry->text, entry->text_len, entry->pos); + entry->pos = cluster_start; } gui_entry_fix_cursor(entry); diff --git a/src/fe-text/gui-readline.c b/src/fe-text/gui-readline.c index a53e4151..bcaca311 100644 --- a/src/fe-text/gui-readline.c +++ b/src/fe-text/gui-readline.c @@ -103,6 +103,8 @@ GArray *g_array_copy(GArray *array) #endif static void sig_input(void); +static void paste_bracketed_middle(void); +static gboolean process_input_smart(GArray *input_buffer); void input_listen_init(int handle) { @@ -867,6 +869,87 @@ static void key_append_next_kill(void) active_entry->append_next_kill = TRUE; } +static gboolean process_input_smart(GArray *input_buffer) +{ + int pos, cluster_end; + unichar *data; + char *utf8_cluster; + glong items_read, items_written; + GError *error; + int i; + gboolean has_control_chars; + GString *accum; + + if (!input_buffer || input_buffer->len == 0 || !active_entry || !active_entry->utf8) { + return FALSE; + } + + data = (unichar*)input_buffer->data; + has_control_chars = FALSE; + error = NULL; + + /* Check if buffer contains control characters (arrows, enter, etc.) */ + for (i = 0; i < input_buffer->len; i++) { + if (data[i] < 32 || data[i] == 127) { + has_control_chars = TRUE; + break; + } + } + + /* If we have control characters, don't process - let legacy system handle */ + if (has_control_chars) { + return FALSE; + } + + pos = 0; + accum = g_string_new(NULL); + + /* Process input buffer as grapheme clusters, but insert once */ + while (pos < input_buffer->len) { + cluster_end = pos; + + /* Advance to end of current grapheme cluster */ + unichar_array_advance_cluster(data, input_buffer->len, &cluster_end); + + /* If no advancement, move single character to avoid infinite loop */ + if (cluster_end == pos) { + cluster_end = pos + 1; + } + + /* Convert cluster to UTF-8 and append to accumulator */ + utf8_cluster = g_ucs4_to_utf8(&data[pos], cluster_end - pos, + &items_read, &items_written, &error); + + if (error == NULL && utf8_cluster && items_written > 0) { + g_string_append(accum, utf8_cluster); + g_free(utf8_cluster); + } else { + /* Fallback: append individual printable characters */ + for (i = pos; i < cluster_end; i++) { + if (data[i] >= 32) { + char out[10]; + out[g_unichar_to_utf8(data[i], out)] = '\0'; + g_string_append(accum, out); + } + } + if (error) { + g_error_free(error); + error = NULL; + } + } + + pos = cluster_end; + } + + if (accum->len > 0) { + /* Insert entire paste as a single text update to mirror history behavior */ + gui_entry_insert_text(active_entry, accum->str); + } + g_string_free(accum, TRUE); + + return TRUE; +} + static gboolean paste_timeout(gpointer data) { int split_lines; @@ -888,11 +971,16 @@ static gboolean paste_timeout(gpointer data) /* Take into account the fact that a line may be split every LINE_SPLIT_LIMIT characters */ if (paste_line_count == 0 && split_lines <= paste_verify_line_count) { - int i; - - for (i = 0; i < paste_buffer->len; i++) { - unichar key = g_array_index(paste_buffer, unichar, i); - signal_emit("gui key pressed", 1, GINT_TO_POINTER(key)); + /* Use smart UTF-8/grapheme cluster processing if available */ + if (active_entry && active_entry->utf8 && process_input_smart(paste_buffer)) { + /* Smart processing succeeded */ + } else { + /* Fallback to legacy character-by-character processing */ + int i; + for (i = 0; i < paste_buffer->len; i++) { + unichar key = g_array_index(paste_buffer, unichar, i); + signal_emit("gui key pressed", 1, GINT_TO_POINTER(key)); + } } g_array_set_size(paste_buffer, 0); } else if (paste_verify_line_count > 0 && @@ -1004,23 +1092,30 @@ static void sig_input(void) g_source_remove(paste_timeout_id); paste_timeout_id = g_timeout_add(paste_detect_time, paste_timeout, NULL); } else if (!paste_bracketed_mode) { - int i; + /* Use smart UTF-8/grapheme cluster processing only for multi-char input (paste-like) */ + if (active_entry && active_entry->utf8 && paste_buffer->len > 1 && process_input_smart(paste_buffer)) { + /* Smart processing succeeded - clear buffer */ + g_array_set_size(paste_buffer, 0); + paste_line_count = 0; + } else { + /* Fallback to legacy character-by-character processing */ + int i; + for (i = 0; i < paste_buffer->len; i++) { + unichar key = g_array_index(paste_buffer, unichar, i); + signal_emit("gui key pressed", 1, GINT_TO_POINTER(key)); - for (i = 0; i < paste_buffer->len; i++) { - unichar key = g_array_index(paste_buffer, unichar, i); - signal_emit("gui key pressed", 1, GINT_TO_POINTER(key)); + if (paste_bracketed_mode) { + /* just enabled by the signal, remove what was processed so far */ + g_array_remove_range(paste_buffer, 0, i + 1); - if (paste_bracketed_mode) { - /* just enabled by the signal, remove what was processed so far */ - g_array_remove_range(paste_buffer, 0, i + 1); - - /* handle single-line / small pastes here */ - paste_bracketed_middle(); - return; + /* handle single-line / small pastes here */ + paste_bracketed_middle(); + return; + } } + g_array_set_size(paste_buffer, 0); + paste_line_count = 0; } - g_array_set_size(paste_buffer, 0); - paste_line_count = 0; } } } diff --git a/src/fe-text/term-terminfo.c b/src/fe-text/term-terminfo.c index 9bcdd5a0..601c87d5 100644 --- a/src/fe-text/term-terminfo.c +++ b/src/fe-text/term-terminfo.c @@ -514,7 +514,7 @@ void term_add_unichar(TERM_WINDOW *window, unichar chr) switch (term_type) { case TERM_TYPE_UTF8: - term_printed_text(unichar_isprint(chr) ? i_wcwidth(chr) : 1); + term_printed_text(unichar_isprint(chr) ? unichar_width(chr) : 1); term_addch_utf8(window, chr); break; case TERM_TYPE_BIG5: @@ -536,8 +536,6 @@ void term_add_unichar(TERM_WINDOW *window, unichar chr) int term_addstr(TERM_WINDOW *window, const char *str) { int len, raw_len; - unichar tmp; - const char *ptr; if (vcmove) term_move_real(); @@ -546,21 +544,9 @@ int term_addstr(TERM_WINDOW *window, const char *str) /* The string length depends on the terminal encoding */ - ptr = str; - if (term_type == TERM_TYPE_UTF8) { - while (*ptr != '\0') { - tmp = g_utf8_get_char_validated(ptr, -1); - /* On utf8 error, treat as single byte and try to - continue interpreting rest of string as utf8 */ - if (tmp == (gunichar)-1 || tmp == (gunichar)-2) { - len++; - ptr++; - } else { - len += unichar_isprint(tmp) ? i_wcwidth(tmp) : 1; - ptr = g_utf8_next_char(ptr); - } - } + /* Use string_width for proper grapheme cluster handling */ + len = string_width(str, TREAT_STRING_AS_UTF8); } else len = raw_len; diff --git a/src/fe-text/textbuffer-view.c b/src/fe-text/textbuffer-view.c index ee474785..0253a6a8 100644 --- a/src/fe-text/textbuffer-view.c +++ b/src/fe-text/textbuffer-view.c @@ -154,8 +154,10 @@ static inline unichar read_unichar(const unsigned char *data, const unsigned cha *next = data + 1; *width = 1; } else { - *next = (unsigned char *)g_utf8_next_char(data); - *width = unichar_isprint(chr) ? i_wcwidth(chr) : 1; + /* Use string_advance for proper grapheme cluster handling */ + char const *str_ptr = (char const *)data; + *width = string_advance(&str_ptr, TREAT_STRING_AS_UTF8); + *next = (unsigned char *)str_ptr; } return chr; } @@ -580,11 +582,21 @@ static int view_line_draw(TEXT_BUFFER_VIEW_REC *view, LINE_REC *line, xpos += char_width; if (xpos <= view->width) { if (unichar_isprint(chr)) { - if (view->utf8) - term_add_unichar(view->window, chr); - else + if (view->utf8) { + /* Send entire grapheme cluster to preserve variation selectors */ + char cluster_str[16]; + int cluster_len = end - text; + if (cluster_len < sizeof(cluster_str)) { + memcpy(cluster_str, text, cluster_len); + cluster_str[cluster_len] = '\0'; + term_addstr(view->window, cluster_str); + } else { + term_add_unichar(view->window, chr); + } + } else { for (; text < end; text++) term_addch(view->window, *text); + } } else { /* low-ascii */ term_set_color(view->window, ATTR_RESET|ATTR_REVERSE); From becc4f90684e4586790878166736b784378ab115 Mon Sep 17 00:00:00 2001 From: kofany Date: Wed, 10 Sep 2025 20:55:25 +0200 Subject: [PATCH 2/3] Fix emoji variation selector width calculation in string_advance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses issue where emoji with variation selectors (like ❣️, ♥️) were incorrectly calculated as width 1 instead of width 2, causing display overflow in modern terminals. - Add variation selector detection (0xFE0F) in string_advance_with_grapheme_support() - Apply special width handling for base emoji + variation selector combinations - Ensures consistent width calculation with input field processing - Fixes chat window overflow issues in terminals like Ghostty This brings string_advance logic in line with unichar_array_advance_cluster which already had proper variation selector handling. --- src/core/utf8.c | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/src/core/utf8.c b/src/core/utf8.c index ca841b4d..cdb272f5 100644 --- a/src/core/utf8.c +++ b/src/core/utf8.c @@ -42,6 +42,7 @@ static int string_advance_with_grapheme_support(char const **str, int policy) const char *start = *str; const char *pos = *str; int cluster_width = 0; + int has_variation_selector = 0; utf8proc_ssize_t bytes; if (policy != TREAT_STRING_AS_UTF8) { @@ -69,6 +70,11 @@ static int string_advance_with_grapheme_support(char const **str, int policy) break; } + /* Check for variation selector */ + if (codepoint == 0xFE0F) { + has_variation_selector = 1; + } + /* Add this codepoint's width to the cluster */ if (unichar_isprint(codepoint)) { int char_width = i_wcwidth(codepoint); @@ -81,6 +87,12 @@ static int string_advance_with_grapheme_support(char const **str, int policy) pos += bytes; } + /* Special handling for emoji with variation selector */ + if (has_variation_selector && cluster_width == 1) { + /* Base emoji (like ❣ U+2763, ♥ U+2665) + variation selector should have width 2 */ + cluster_width = 2; + } + *str = pos; return cluster_width > 0 ? cluster_width : 1; } From 20fbea0e2d30ee2922b11496e4d31a2baac634b4 Mon Sep 17 00:00:00 2001 From: kofany Date: Mon, 6 Oct 2025 20:39:46 +0200 Subject: [PATCH 3/3] Move is_combining_char() to core utf8 module MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Relocate is_combining_char() from gui-entry.c to src/core/utf8.c as suggested in code review. This better organizes the codebase by placing UTF-8 character classification logic with other UTF-8 utilities in the core module. The function now uses is_utf8() instead of checking term_type directly, which is the appropriate method for core layer code. 🤖 Generated with [Claude Code](https://claude.com/claude-code) Co-Authored-By: Claude --- src/core/utf8.c | 14 ++++++++++++++ src/core/utf8.h | 4 ++++ src/fe-text/gui-entry.c | 14 -------------- 3 files changed, 18 insertions(+), 14 deletions(-) diff --git a/src/core/utf8.c b/src/core/utf8.c index cdb272f5..9faaac89 100644 --- a/src/core/utf8.c +++ b/src/core/utf8.c @@ -399,6 +399,20 @@ int unichar_array_find_cluster_start(const unichar *text, int text_len, int pos) #endif } +int is_combining_char(unichar c) +{ + if (!is_utf8()) + return 0; + +#ifdef HAVE_LIBUTF8PROC + /* Use utf8proc for precise combining character detection */ + return unichar_isprint(c) && utf8proc_charwidth(c) == 0; +#else + /* Fallback to unichar_width for compatibility */ + return unichar_isprint(c) && unichar_width(c) == 0; +#endif +} + void utf8_init(void) { /* no-op */ diff --git a/src/core/utf8.h b/src/core/utf8.h index 2893f58d..694e1bdf 100644 --- a/src/core/utf8.h +++ b/src/core/utf8.h @@ -72,6 +72,10 @@ int unichar_array_move_cluster_backward(const unichar *text, int text_len, int * * Returns the codepoint index of the cluster start. */ int unichar_array_find_cluster_start(const unichar *text, int text_len, int pos); +/* Check if a unichar is a combining character (width 0). + * Returns 1 if the character is combining, 0 otherwise. */ +int is_combining_char(unichar c); + /* Initialize UTF-8 debugging system */ void utf8_init(void); diff --git a/src/fe-text/gui-entry.c b/src/fe-text/gui-entry.c index 69c5d942..243ec49f 100644 --- a/src/fe-text/gui-entry.c +++ b/src/fe-text/gui-entry.c @@ -54,20 +54,6 @@ static unichar i_tolower(unichar c) return c <= 255 ? tolower(c) : c; } -static int is_combining_char(unichar c) -{ - if (term_type != TERM_TYPE_UTF8) - return 0; - -#ifdef HAVE_LIBUTF8PROC - /* Use utf8proc for precise combining character detection */ - return unichar_isprint(c) && utf8proc_charwidth(c) == 0; -#else - /* Fallback to unichar_width for compatibility */ - return unichar_isprint(c) && unichar_width(c) == 0; -#endif -} - static int i_isalnum(unichar c) { if (term_type == TERM_TYPE_UTF8)