diff options
| -rw-r--r-- | source/parsing.c | 27 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
| -rw-r--r-- | source/shared.c | 49 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
| -rw-r--r-- | source/shared.h | 12 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
| -rw-r--r-- | source/ui-atom.c | 35 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
| -rwxr-xr-x | tests/t0106-commit.sh | 19 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
5 files changed, 99 insertions, 43 deletions
diff --git a/source/parsing.c b/source/parsing.c index 655973b..8aa0ee4 100644 --- a/source/parsing.c +++ b/source/parsing.c @@ -92,8 +92,11 @@ static int end_of_header(const char *p) } /* - * A git built without iconv still offers reencode_string, where it always - * fails, so the field is left in whatever encoding it arrived in. + * Converts a field into the page encoding, which is UTF-8. A git built + * without iconv still offers reencode_string, where it always fails, and a + * field can be labelled UTF-8 without being it, so whatever is left that is + * not UTF-8 has its stray bytes replaced, and a page declared UTF-8 never + * carries any. */ static const char *reencode(char **text, const char *from, const char *to) { @@ -102,14 +105,14 @@ static const char *reencode(char **text, const char *from, const char *to) if (!*text || !from || !to) return *text; - if (!strcasecmp(from, to)) - return *text; - - converted = reencode_string(*text, to, from); - if (converted) { - free(*text); - *text = converted; + if (strcasecmp(from, to)) { + converted = reencode_string(*text, to, from); + if (converted) { + free(*text); + *text = converted; + } } + cgit_utf8_sanitize(*text); return *text; } @@ -256,6 +259,12 @@ struct taginfo *cgit_parse_tag(struct tag *tag) if (p && *p) info->msg = xstrdup(p); + // A tag carries no encoding header, so its text is taken as UTF-8 and + // cleaned the way a commit's is. + cgit_utf8_sanitize(info->tagger); + cgit_utf8_sanitize(info->tagger_email); + cgit_utf8_sanitize(info->msg); + cleanup: free(data); return info; diff --git a/source/shared.c b/source/shared.c index 78dc3f4..b68cbbf 100644 --- a/source/shared.c +++ b/source/shared.c @@ -598,6 +598,55 @@ char *cgit_strdup_first_line(const char *text) return line; } +/* + * The lead-byte ranges fold in the overlong, surrogate and out-of-range + * cases, so 0 is the only error signal a caller needs. + */ +size_t cgit_utf8_seq_len(const unsigned char *p, size_t left) +{ + size_t len, i; + + if (p[0] >= 0xc2 && p[0] <= 0xdf) + len = 2; + else if (p[0] >= 0xe0 && p[0] <= 0xef) + len = 3; + else if (p[0] >= 0xf0 && p[0] <= 0xf4) + len = 4; + else + return 0; + if (left < len) + return 0; + for (i = 1; i < len; i++) + if ((p[i] & 0xc0) != 0x80) + return 0; + if (p[0] == 0xe0 && p[1] < 0xa0) + return 0; + if (p[0] == 0xed && p[1] > 0x9f) + return 0; + if (p[0] == 0xf0 && p[1] < 0x90) + return 0; + if (p[0] == 0xf4 && p[1] > 0x8f) + return 0; + return len; +} + +void cgit_utf8_sanitize(char *text) +{ + unsigned char *p = (unsigned char *)text; + size_t left = text ? strlen(text) : 0; + + while (left) { + size_t seq = *p < 0x80 ? 1 : cgit_utf8_seq_len(p, left); + + if (!seq) { + *p = '?'; + seq = 1; + } + p += seq; + left -= seq; + } +} + char *cgit_expand_macros(const char *text) { static char result[MACRO_EXPANSION_BUFSIZE]; diff --git a/source/shared.h b/source/shared.h index d61c1f1..ca2ac6a 100644 --- a/source/shared.h +++ b/source/shared.h @@ -90,6 +90,18 @@ extern int cgit_read_first_line(const char *path, char **buf, size_t *size); extern char *cgit_strdup_first_line(const char *text); /* + * How many bytes the UTF-8 sequence at p holds, or 0 when it is invalid, + * looking at most left bytes ahead. + */ +extern size_t cgit_utf8_seq_len(const unsigned char *p, size_t left); + +/* + * Replace every byte of text that is not part of a valid UTF-8 sequence with + * a question mark, in place. + */ +extern void cgit_utf8_sanitize(char *text); + +/* * Replace every $token in text with the matching environment variable. The * result is a static buffer, so a caller that needs to keep it must copy it. */ diff --git a/source/ui-atom.c b/source/ui-atom.c index b3265b7..b536437 100644 --- a/source/ui-atom.c +++ b/source/ui-atom.c @@ -28,39 +28,6 @@ static const char *feed_date(timestamp_t when) } /* - * How many bytes the UTF-8 sequence at p holds, or 0 when invalid. The - * lead-byte ranges fold in the overlong, surrogate and out-of-range cases, - * so 0 is the only error signal a caller needs. - */ -static size_t utf8_seq_len(const unsigned char *p, size_t left) -{ - size_t len, i; - - if (p[0] >= 0xc2 && p[0] <= 0xdf) - len = 2; - else if (p[0] >= 0xe0 && p[0] <= 0xef) - len = 3; - else if (p[0] >= 0xf0 && p[0] <= 0xf4) - len = 4; - else - return 0; - if (left < len) - return 0; - for (i = 1; i < len; i++) - if ((p[i] & 0xc0) != 0x80) - return 0; - if (p[0] == 0xe0 && p[1] < 0xa0) - return 0; - if (p[0] == 0xed && p[1] > 0x9f) - return 0; - if (p[0] == 0xf0 && p[1] < 0x90) - return 0; - if (p[0] == 0xf4 && p[1] > 0x8f) - return 0; - return len; -} - -/* * The XML counterpart of html_txt. A browser shrugs at a stray control byte * or broken UTF-8, but an XML reader must reject the whole feed, so both * are replaced instead of passed through. @@ -85,7 +52,7 @@ static void xml_txt(const char *txt) strbuf_addch(&sb, c); } else if (c < 0x20) { strbuf_addstr(&sb, XML_REPLACEMENT); - } else if ((seq = utf8_seq_len(p, left))) { + } else if ((seq = cgit_utf8_seq_len(p, left))) { // U+FFFE and U+FFFF are valid UTF-8 but not XML. if (seq == 3 && p[0] == 0xef && p[1] == 0xbf && p[2] >= 0xbe) strbuf_addstr(&sb, XML_REPLACEMENT); diff --git a/tests/t0106-commit.sh b/tests/t0106-commit.sh index 39f1822..2b924a0 100755 --- a/tests/t0106-commit.sh +++ b/tests/t0106-commit.sh @@ -37,4 +37,23 @@ test_expect_success 'root commit contains diff' ' grep "<div class=.add.>+1</div>" tmp ' +# A message that is not UTF-8, because its encoding header names something +# iconv cannot convert or is missing altogether, would otherwise reach a +# page declared UTF-8 as it is, so its stray bytes become question marks. +test_expect_success 'a message that is not UTF-8 has its stray bytes replaced' ' + ( + cd repos/foo && + tree=$(git rev-parse HEAD^{tree}) && + printf "tree %s\nauthor A U Thor <author@example.com> 1735689600 +0000\n" "$tree" >raw && + printf "committer A U Thor <author@example.com> 1735689600 +0000\n" >>raw && + printf "encoding no-such-charset\n\ncaf\351 subject\n\nbody \377 end\n" >>raw && + cid=$(git hash-object --literally -t commit -w raw) && + git update-ref refs/heads/latin "$cid" + ) && + cgit_query "url=foo/commit&h=latin" >tmp && + grep "<div class=.commit-subject.>caf? subject<" tmp && + grep "body ? end" tmp && + ! grep "caf$(printf "\351")" tmp +' + test_done |
