From 6522ee76cb09a3c97ea3ee002b81343a20473ee3 Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Thu, 1 Oct 2026 18:04:18 -1000 Subject: Replace the stray bytes of text that is not UTF-8 A commit whose encoding header names something iconv cannot convert, or whose text is labelled UTF-8 without being it, reached a page declared UTF-8 with its bytes untouched. The atom feed already replaced them, so its UTF-8 reader moves to shared.c and every ident, subject, message and tag text passes through it. --- source/ui-atom.c | 35 +---------------------------------- 1 file changed, 1 insertion(+), 34 deletions(-) (limited to 'source/ui-atom.c') diff --git a/source/ui-atom.c b/source/ui-atom.c index b3265b7..b536437 100644 --- a/source/ui-atom.c +++ b/source/ui-atom.c @@ -27,39 +27,6 @@ static const char *feed_date(timestamp_t when) return show_date(when, 0, date_mode_from_type(DATE_ISO8601_STRICT)); } -/* - * How many bytes the UTF-8 sequence at p holds, or 0 when invalid. The - * lead-byte ranges fold in the overlong, surrogate and out-of-range cases, - * so 0 is the only error signal a caller needs. - */ -static size_t utf8_seq_len(const unsigned char *p, size_t left) -{ - size_t len, i; - - if (p[0] >= 0xc2 && p[0] <= 0xdf) - len = 2; - else if (p[0] >= 0xe0 && p[0] <= 0xef) - len = 3; - else if (p[0] >= 0xf0 && p[0] <= 0xf4) - len = 4; - else - return 0; - if (left < len) - return 0; - for (i = 1; i < len; i++) - if ((p[i] & 0xc0) != 0x80) - return 0; - if (p[0] == 0xe0 && p[1] < 0xa0) - return 0; - if (p[0] == 0xed && p[1] > 0x9f) - return 0; - if (p[0] == 0xf0 && p[1] < 0x90) - return 0; - if (p[0] == 0xf4 && p[1] > 0x8f) - return 0; - return len; -} - /* * The XML counterpart of html_txt. A browser shrugs at a stray control byte * or broken UTF-8, but an XML reader must reject the whole feed, so both @@ -85,7 +52,7 @@ static void xml_txt(const char *txt) strbuf_addch(&sb, c); } else if (c < 0x20) { strbuf_addstr(&sb, XML_REPLACEMENT); - } else if ((seq = utf8_seq_len(p, left))) { + } else if ((seq = cgit_utf8_seq_len(p, left))) { // U+FFFE and U+FFFF are valid UTF-8 but not XML. if (seq == 3 && p[0] == 0xef && p[1] == 0xbf && p[2] >= 0xbe) strbuf_addstr(&sb, XML_REPLACEMENT); -- cgit v2.8.0