From 6522ee76cb09a3c97ea3ee002b81343a20473ee3 Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Thu, 1 Oct 2026 18:04:18 -1000 Subject: Replace the stray bytes of text that is not UTF-8 A commit whose encoding header names something iconv cannot convert, or whose text is labelled UTF-8 without being it, reached a page declared UTF-8 with its bytes untouched. The atom feed already replaced them, so its UTF-8 reader moves to shared.c and every ident, subject, message and tag text passes through it. --- source/shared.h | 12 ++++++++++++ 1 file changed, 12 insertions(+) (limited to 'source/shared.h') diff --git a/source/shared.h b/source/shared.h index d61c1f1..ca2ac6a 100644 --- a/source/shared.h +++ b/source/shared.h @@ -89,6 +89,18 @@ extern void strbuf_ensure_end(struct strbuf *sb, char c); extern int cgit_read_first_line(const char *path, char **buf, size_t *size); extern char *cgit_strdup_first_line(const char *text); +/* + * How many bytes the UTF-8 sequence at p holds, or 0 when it is invalid, + * looking at most left bytes ahead. + */ +extern size_t cgit_utf8_seq_len(const unsigned char *p, size_t left); + +/* + * Replace every byte of text that is not part of a valid UTF-8 sequence with + * a question mark, in place. + */ +extern void cgit_utf8_sanitize(char *text); + /* * Replace every $token in text with the matching environment variable. The * result is a static buffer, so a caller that needs to keep it must copy it. -- cgit v2.8.0