diff options
context:
space:
mode:
authorBryce Kwon <bryce@brycekwon.com>
committerBryce Kwon <bryce@brycekwon.com>
commit
parent
tree
download
Replace the stray bytes of text that is not UTF-8
A commit whose encoding header names something iconv cannot convert, or whose text is labelled UTF-8 without being it, reached a page declared UTF-8 with its bytes untouched. The atom feed already replaced them, so its UTF-8 reader moves to shared.c and every ident, subject, message and tag text passes through it.
Diffstat (limited to 'source')
-rw-r--r--source/parsing.c27
-rw-r--r--source/shared.c49
-rw-r--r--source/shared.h12
-rw-r--r--source/ui-atom.c35
4 files changed, 80 insertions, 43 deletions
diff --git a/source/parsing.c b/source/parsing.c
index 655973b..8aa0ee4 100644
--- a/source/parsing.c
+++ b/source/parsing.c
@@ -92,8 +92,11 @@ static int end_of_header(const char *p)
}
/*
- * A git built without iconv still offers reencode_string, where it always
- * fails, so the field is left in whatever encoding it arrived in.
+ * Converts a field into the page encoding, which is UTF-8. A git built
+ * without iconv still offers reencode_string, where it always fails, and a
+ * field can be labelled UTF-8 without being it, so whatever is left that is
+ * not UTF-8 has its stray bytes replaced, and a page declared UTF-8 never
+ * carries any.
*/
static const char *reencode(char **text, const char *from, const char *to)
{
@@ -102,14 +105,14 @@ static const char *reencode(char **text, const char *from, const char *to)
if (!*text || !from || !to)
return *text;
- if (!strcasecmp(from, to))
- return *text;
-
- converted = reencode_string(*text, to, from);
- if (converted) {
- free(*text);
- *text = converted;
+ if (strcasecmp(from, to)) {
+ converted = reencode_string(*text, to, from);
+ if (converted) {
+ free(*text);
+ *text = converted;
+ }
}
+ cgit_utf8_sanitize(*text);
return *text;
}
@@ -256,6 +259,12 @@ struct taginfo *cgit_parse_tag(struct tag *tag)
if (p && *p)
info->msg = xstrdup(p);
+ // A tag carries no encoding header, so its text is taken as UTF-8 and
+ // cleaned the way a commit's is.
+ cgit_utf8_sanitize(info->tagger);
+ cgit_utf8_sanitize(info->tagger_email);
+ cgit_utf8_sanitize(info->msg);
+
cleanup:
free(data);
return info;
diff --git a/source/shared.c b/source/shared.c
index 78dc3f4..b68cbbf 100644
--- a/source/shared.c
+++ b/source/shared.c
@@ -598,6 +598,55 @@ char *cgit_strdup_first_line(const char *text)
return line;
}
+/*
+ * The lead-byte ranges fold in the overlong, surrogate and out-of-range
+ * cases, so 0 is the only error signal a caller needs.
+ */
+size_t cgit_utf8_seq_len(const unsigned char *p, size_t left)
+{
+ size_t len, i;
+
+ if (p[0] >= 0xc2 && p[0] <= 0xdf)
+ len = 2;
+ else if (p[0] >= 0xe0 && p[0] <= 0xef)
+ len = 3;
+ else if (p[0] >= 0xf0 && p[0] <= 0xf4)
+ len = 4;
+ else
+ return 0;
+ if (left < len)
+ return 0;
+ for (i = 1; i < len; i++)
+ if ((p[i] & 0xc0) != 0x80)
+ return 0;
+ if (p[0] == 0xe0 && p[1] < 0xa0)
+ return 0;
+ if (p[0] == 0xed && p[1] > 0x9f)
+ return 0;
+ if (p[0] == 0xf0 && p[1] < 0x90)
+ return 0;
+ if (p[0] == 0xf4 && p[1] > 0x8f)
+ return 0;
+ return len;
+}
+
+void cgit_utf8_sanitize(char *text)
+{
+ unsigned char *p = (unsigned char *)text;
+ size_t left = text ? strlen(text) : 0;
+
+ while (left) {
+ size_t seq = *p < 0x80 ? 1 : cgit_utf8_seq_len(p, left);
+
+ if (!seq) {
+ *p = '?';
+ seq = 1;
+ }
+ p += seq;
+ left -= seq;
+ }
+}
+
char *cgit_expand_macros(const char *text)
{
static char result[MACRO_EXPANSION_BUFSIZE];
diff --git a/source/shared.h b/source/shared.h
index d61c1f1..ca2ac6a 100644
--- a/source/shared.h
+++ b/source/shared.h
@@ -90,6 +90,18 @@ extern int cgit_read_first_line(const char *path, char **buf, size_t *size);
extern char *cgit_strdup_first_line(const char *text);
/*
+ * How many bytes the UTF-8 sequence at p holds, or 0 when it is invalid,
+ * looking at most left bytes ahead.
+ */
+extern size_t cgit_utf8_seq_len(const unsigned char *p, size_t left);
+
+/*
+ * Replace every byte of text that is not part of a valid UTF-8 sequence with
+ * a question mark, in place.
+ */
+extern void cgit_utf8_sanitize(char *text);
+
+/*
* Replace every $token in text with the matching environment variable. The
* result is a static buffer, so a caller that needs to keep it must copy it.
*/
diff --git a/source/ui-atom.c b/source/ui-atom.c
index b3265b7..b536437 100644
--- a/source/ui-atom.c
+++ b/source/ui-atom.c
@@ -28,39 +28,6 @@ static const char *feed_date(timestamp_t when)
}
/*
- * How many bytes the UTF-8 sequence at p holds, or 0 when invalid. The
- * lead-byte ranges fold in the overlong, surrogate and out-of-range cases,
- * so 0 is the only error signal a caller needs.
- */
-static size_t utf8_seq_len(const unsigned char *p, size_t left)
-{
- size_t len, i;
-
- if (p[0] >= 0xc2 && p[0] <= 0xdf)
- len = 2;
- else if (p[0] >= 0xe0 && p[0] <= 0xef)
- len = 3;
- else if (p[0] >= 0xf0 && p[0] <= 0xf4)
- len = 4;
- else
- return 0;
- if (left < len)
- return 0;
- for (i = 1; i < len; i++)
- if ((p[i] & 0xc0) != 0x80)
- return 0;
- if (p[0] == 0xe0 && p[1] < 0xa0)
- return 0;
- if (p[0] == 0xed && p[1] > 0x9f)
- return 0;
- if (p[0] == 0xf0 && p[1] < 0x90)
- return 0;
- if (p[0] == 0xf4 && p[1] > 0x8f)
- return 0;
- return len;
-}
-
-/*
* The XML counterpart of html_txt. A browser shrugs at a stray control byte
* or broken UTF-8, but an XML reader must reject the whole feed, so both
* are replaced instead of passed through.
@@ -85,7 +52,7 @@ static void xml_txt(const char *txt)
strbuf_addch(&sb, c);
} else if (c < 0x20) {
strbuf_addstr(&sb, XML_REPLACEMENT);
- } else if ((seq = utf8_seq_len(p, left))) {
+ } else if ((seq = cgit_utf8_seq_len(p, left))) {
// U+FFFE and U+FFFF are valid UTF-8 but not XML.
if (seq == 3 && p[0] == 0xef && p[1] == 0xbf && p[2] >= 0xbe)
strbuf_addstr(&sb, XML_REPLACEMENT);