From 49e067570dfa208c45d76f0b602664fd11a629ef Mon Sep 17 00:00:00 2001 From: Philip Withnall Date: Tue, 31 Mar 2026 16:13:57 +0100 Subject: [PATCH] gregex: Fix case changing substitutions with G_REGEX_RAW In `G_REGEX_RAW` mode, the input string is treated as a byte array (basically ASCII) rather than a unichar array. Accordingly, the case changing code for substitutions needs to operate on bytes with `G_REGEX_RAW`, rather than operating on unichars. This fixes a potential buffer overflow when trying to do a case change on a match of a set of bytes which are a truncated multi-byte UTF-8 encoding at the end of the input buffer. Spotted by linhlhq as #YWH-PGM9867-193. I adapted their reproducer as the unit test, but implemented the fix in `gregex.c` independently. Signed-off-by: Philip Withnall Fixes: #3918 --- glib/gregex.c | 59 ++++++++++++++++++++++++++++++++++------------ glib/tests/regex.c | 53 +++++++++++++++++++++++++++++++++++++++++ 2 files changed, 97 insertions(+), 15 deletions(-) diff --git a/glib/gregex.c b/glib/gregex.c index 52d09f1e45..c261714150 100644 --- a/glib/gregex.c +++ b/glib/gregex.c @@ -3422,19 +3422,25 @@ split_replacement (const gchar *replacement, return g_list_reverse (list); } -/* Change the case of c based on change_case. */ -#define CHANGE_CASE(c, change_case) \ +/* Change the case of c based on change_case. + * g_ascii_to*() will happily pass through non-ASCII bytes unchanged. */ +#define UTF8_CHANGE_CASE(c, change_case) \ (((change_case) & CHANGE_CASE_LOWER_MASK) ? \ g_unichar_tolower (c) : \ g_unichar_toupper (c)) +#define RAW_CHANGE_CASE(c, change_case) \ + (((change_case) & CHANGE_CASE_LOWER_MASK) ? \ + g_ascii_tolower (c) : \ + g_ascii_toupper (c)) +/* If @text_is_raw is set, @text might not be valid UTF-8 (but will be + * nul-terminated). */ static void string_append (GString *string, const gchar *text, + gboolean text_is_raw, ChangeCase *change_case) { - gunichar c; - if (text[0] == '\0') return; @@ -3444,22 +3450,44 @@ string_append (GString *string, } else if (*change_case & CHANGE_CASE_SINGLE_MASK) { - c = g_utf8_get_char (text); - g_string_append_unichar (string, CHANGE_CASE (c, *change_case)); - g_string_append (string, g_utf8_next_char (text)); + if (!text_is_raw) + { + gunichar c = g_utf8_get_char (text); + g_string_append_unichar (string, UTF8_CHANGE_CASE (c, *change_case)); + g_string_append (string, g_utf8_next_char (text)); + } + else + { + g_string_append_c (string, RAW_CHANGE_CASE (text[0], *change_case)); + g_string_append (string, text + 1); + } + *change_case = CHANGE_CASE_NONE; } else { - while (*text != '\0') + if (!text_is_raw) { - c = g_utf8_get_char (text); - g_string_append_unichar (string, CHANGE_CASE (c, *change_case)); - text = g_utf8_next_char (text); + while (*text != '\0') + { + gunichar c = g_utf8_get_char (text); + g_string_append_unichar (string, UTF8_CHANGE_CASE (c, *change_case)); + text = g_utf8_next_char (text); + } + } + else + { + while (*text != '\0') + { + char c = *text; + g_string_append_c (string, RAW_CHANGE_CASE (c, *change_case)); + text++; + } } } } +/* @match_info is (nullable) */ static gboolean interpolate_replacement (const GMatchInfo *match_info, GString *result, @@ -3469,6 +3497,7 @@ interpolate_replacement (const GMatchInfo *match_info, InterpolationData *idata; gchar *match; ChangeCase change_case = CHANGE_CASE_NONE; + gboolean is_raw = (match_info != NULL && (match_info->regex->orig_compile_opts & G_REGEX_RAW)); for (list = data; list; list = list->next) { @@ -3476,10 +3505,10 @@ interpolate_replacement (const GMatchInfo *match_info, switch (idata->type) { case REPL_TYPE_STRING: - string_append (result, idata->text, &change_case); + string_append (result, idata->text, is_raw, &change_case); break; case REPL_TYPE_CHARACTER: - g_string_append_c (result, CHANGE_CASE (idata->c, change_case)); + g_string_append_c (result, UTF8_CHANGE_CASE (idata->c, change_case)); if (change_case & CHANGE_CASE_SINGLE_MASK) change_case = CHANGE_CASE_NONE; break; @@ -3487,7 +3516,7 @@ interpolate_replacement (const GMatchInfo *match_info, match = g_match_info_fetch (match_info, idata->num); if (match) { - string_append (result, match, &change_case); + string_append (result, match, is_raw, &change_case); g_free (match); } break; @@ -3495,7 +3524,7 @@ interpolate_replacement (const GMatchInfo *match_info, match = g_match_info_fetch_named (match_info, idata->text); if (match) { - string_append (result, match, &change_case); + string_append (result, match, is_raw, &change_case); g_free (match); } break; diff --git a/glib/tests/regex.c b/glib/tests/regex.c index a774683f75..ed15851416 100644 --- a/glib/tests/regex.c +++ b/glib/tests/regex.c @@ -2531,6 +2531,58 @@ test_compiled_regex_after_jit_failure (void) g_regex_unref (regex); } +static void +test_replace_raw_change_case (void) +{ + GError *local_error = NULL; + GRegex *regex = NULL; + + g_test_bug ("https://gitlab.gnome.org/GNOME/glib/-/issues/3918"); + g_test_summary ("Test that case changes as part of a replacement are handled correctly in G_REGEX_RAW mode"); + + /* + * Match a multi-byte sequence in RAW mode. The pattern matches + * exactly 2 bytes. The subject contains a 4-byte UTF-8 lead (0xF4) + * followed by only one continuation byte, then NUL. + * + * The matched substring will be "\xf4\x80" (2 bytes, heap-allocated + * as 3-byte buffer with NUL). If the code regresses and tries to handle + * the replacement as UTF-8 then g_utf8_get_char() would see 0xF4 and try + * to read 4 bytes, going 1 byte past the NUL into OOB territory. + */ + regex = g_regex_new ("..", G_REGEX_RAW, 0, &local_error); + g_assert_no_error (local_error); + + /* + * Build a subject string with truncated UTF-8. + * \xF4 = 4-byte UTF-8 lead byte + * \x80 = continuation byte + * No 3rd/4th continuation bytes — the match is only 2 bytes. + * + * \U\0 = uppercase the entire match → triggers string_append() + * with case change on the 2-byte non-UTF-8 match. + */ + char subject[] = "\xf4\x80"; + char *result = g_regex_replace (regex, subject, -1, 0, "\\U\\0", 0, &local_error); + g_assert_no_error (local_error); + + g_clear_pointer (&result, g_free); + g_clear_pointer (®ex, g_regex_unref); + + /* + * Second variant: single-char case change \u with \0 backreference. + */ + regex = g_regex_new (".", G_REGEX_RAW, 0, &local_error); + g_assert_no_error (local_error); + + char subject2[] = "\xe6\xb0"; /* 3-byte UTF-8 lead, only 2 bytes */ + result = g_regex_replace (regex, subject2, -1, 0, "\\u\\0", 0, &local_error); + g_assert_no_error (local_error); + + g_clear_pointer (&result, g_free); + g_clear_pointer (®ex, g_regex_unref); +} + int main (int argc, char *argv[]) { @@ -2552,6 +2604,7 @@ main (int argc, char *argv[]) g_test_add_func ("/regex/jit-unsupported-matching", test_jit_unsupported_matching_options); g_test_add_func ("/regex/unmatched-named-subpattern", test_unmatched_named_subpattern); g_test_add_func ("/regex/compiled-regex-after-jit-failure", test_compiled_regex_after_jit_failure); + g_test_add_func ("/regex/replace-raw-change-case", test_replace_raw_change_case); /* TEST_NEW(pattern, compile_opts, match_opts) */ TEST_NEW("[A-Z]+", G_REGEX_CASELESS | G_REGEX_EXTENDED | G_REGEX_OPTIMIZE, G_REGEX_MATCH_NOTBOL | G_REGEX_MATCH_PARTIAL); -- GitLab