From 6522ee76cb09a3c97ea3ee002b81343a20473ee3 Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Thu, 1 Oct 2026 18:04:18 -1000 Subject: Replace the stray bytes of text that is not UTF-8 A commit whose encoding header names something iconv cannot convert, or whose text is labelled UTF-8 without being it, reached a page declared UTF-8 with its bytes untouched. The atom feed already replaced them, so its UTF-8 reader moves to shared.c and every ident, subject, message and tag text passes through it. --- source/shared.c | 49 +++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 49 insertions(+) (limited to 'source/shared.c') diff --git a/source/shared.c b/source/shared.c index 78dc3f4..b68cbbf 100644 --- a/source/shared.c +++ b/source/shared.c @@ -598,6 +598,55 @@ char *cgit_strdup_first_line(const char *text) return line; } +/* + * The lead-byte ranges fold in the overlong, surrogate and out-of-range + * cases, so 0 is the only error signal a caller needs. + */ +size_t cgit_utf8_seq_len(const unsigned char *p, size_t left) +{ + size_t len, i; + + if (p[0] >= 0xc2 && p[0] <= 0xdf) + len = 2; + else if (p[0] >= 0xe0 && p[0] <= 0xef) + len = 3; + else if (p[0] >= 0xf0 && p[0] <= 0xf4) + len = 4; + else + return 0; + if (left < len) + return 0; + for (i = 1; i < len; i++) + if ((p[i] & 0xc0) != 0x80) + return 0; + if (p[0] == 0xe0 && p[1] < 0xa0) + return 0; + if (p[0] == 0xed && p[1] > 0x9f) + return 0; + if (p[0] == 0xf0 && p[1] < 0x90) + return 0; + if (p[0] == 0xf4 && p[1] > 0x8f) + return 0; + return len; +} + +void cgit_utf8_sanitize(char *text) +{ + unsigned char *p = (unsigned char *)text; + size_t left = text ? strlen(text) : 0; + + while (left) { + size_t seq = *p < 0x80 ? 1 : cgit_utf8_seq_len(p, left); + + if (!seq) { + *p = '?'; + seq = 1; + } + p += seq; + left -= seq; + } +} + char *cgit_expand_macros(const char *text) { static char result[MACRO_EXPANSION_BUFSIZE]; -- cgit v2.8.0