diff options
context:
space:
mode:
authorBryce Kwon <bryce@brycekwon.com>
committerBryce Kwon <bryce@brycekwon.com>
commit
parent
tree
download
Replace the stray bytes of text that is not UTF-8
A commit whose encoding header names something iconv cannot convert, or whose text is labelled UTF-8 without being it, reached a page declared UTF-8 with its bytes untouched. The atom feed already replaced them, so its UTF-8 reader moves to shared.c and every ident, subject, message and tag text passes through it.
Diffstat (limited to '')
-rw-r--r--source/shared.c49
1 file changed, 49 insertions, 0 deletions
diff --git a/source/shared.c b/source/shared.c
index 78dc3f4..b68cbbf 100644
--- a/source/shared.c
+++ b/source/shared.c
@@ -598,6 +598,55 @@ char *cgit_strdup_first_line(const char *text)
return line;
}
+/*
+ * The lead-byte ranges fold in the overlong, surrogate and out-of-range
+ * cases, so 0 is the only error signal a caller needs.
+ */
+size_t cgit_utf8_seq_len(const unsigned char *p, size_t left)
+{
+ size_t len, i;
+
+ if (p[0] >= 0xc2 && p[0] <= 0xdf)
+ len = 2;
+ else if (p[0] >= 0xe0 && p[0] <= 0xef)
+ len = 3;
+ else if (p[0] >= 0xf0 && p[0] <= 0xf4)
+ len = 4;
+ else
+ return 0;
+ if (left < len)
+ return 0;
+ for (i = 1; i < len; i++)
+ if ((p[i] & 0xc0) != 0x80)
+ return 0;
+ if (p[0] == 0xe0 && p[1] < 0xa0)
+ return 0;
+ if (p[0] == 0xed && p[1] > 0x9f)
+ return 0;
+ if (p[0] == 0xf0 && p[1] < 0x90)
+ return 0;
+ if (p[0] == 0xf4 && p[1] > 0x8f)
+ return 0;
+ return len;
+}
+
+void cgit_utf8_sanitize(char *text)
+{
+ unsigned char *p = (unsigned char *)text;
+ size_t left = text ? strlen(text) : 0;
+
+ while (left) {
+ size_t seq = *p < 0x80 ? 1 : cgit_utf8_seq_len(p, left);
+
+ if (!seq) {
+ *p = '?';
+ seq = 1;
+ }
+ p += seq;
+ left -= seq;
+ }
+}
+
char *cgit_expand_macros(const char *text)
{
static char result[MACRO_EXPANSION_BUFSIZE];