diff options
| author | Bryce Kwon <bryce@brycekwon.com> | |
|---|---|---|
| committer | Bryce Kwon <bryce@brycekwon.com> | |
| commit | ||
| parent | ||
| tree | ||
| download | ||
Bring the atom feed up to the rfc
Diffstat (limited to 'source/ui-atom.c')
| -rw-r--r-- | source/ui-atom.c | 138 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
1 file changed, 117 insertions, 21 deletions
diff --git a/source/ui-atom.c b/source/ui-atom.c index aa9237a..b0adbf1 100644 --- a/source/ui-atom.c +++ b/source/ui-atom.c @@ -24,6 +24,87 @@ static const char *feed_date(timestamp_t when) return show_date(when, 0, date_mode_from_type(DATE_ISO8601_STRICT)); } +// U+FFFD, which stands in for every byte the feed cannot carry. +#define XML_REPLACEMENT "\xef\xbf\xbd" + +/* + * How many bytes the UTF-8 sequence at p holds, or 0 when the bytes there are + * not a valid sequence. The lead-byte ranges fold in the overlong, surrogate + * and out-of-range cases, so a 0 is the only error signal a caller needs. + */ +static size_t utf8_seq_len(const unsigned char *p, size_t left) +{ + size_t len, i; + + if (p[0] >= 0xc2 && p[0] <= 0xdf) + len = 2; + else if (p[0] >= 0xe0 && p[0] <= 0xef) + len = 3; + else if (p[0] >= 0xf0 && p[0] <= 0xf4) + len = 4; + else + return 0; + if (left < len) + return 0; + for (i = 1; i < len; i++) + if ((p[i] & 0xc0) != 0x80) + return 0; + if (p[0] == 0xe0 && p[1] < 0xa0) + return 0; + if (p[0] == 0xed && p[1] > 0x9f) + return 0; + if (p[0] == 0xf0 && p[1] < 0x90) + return 0; + if (p[0] == 0xf4 && p[1] > 0x8f) + return 0; + return len; +} + +/* + * The XML counterpart of html_txt. Commit metadata is arbitrary bytes, and + * where a browser shrugs at a stray control byte or a broken UTF-8 sequence, + * an XML reader must reject the whole feed, so both are replaced instead of + * passed through. + */ +static void xml_txt(const char *txt) +{ + const unsigned char *p = (const unsigned char *)txt; + size_t left = txt ? strlen(txt) : 0; + struct strbuf sb = STRBUF_INIT; + + while (left) { + unsigned char c = *p; + size_t seq; + + if (c == '<') { + strbuf_addstr(&sb, "<"); + } else if (c == '>') { + strbuf_addstr(&sb, ">"); + } else if (c == '&') { + strbuf_addstr(&sb, "&"); + } else if (c == '\t' || c == '\n' || c == '\r' || (c >= 0x20 && c < 0x80)) { + strbuf_addch(&sb, c); + } else if (c < 0x20) { + strbuf_addstr(&sb, XML_REPLACEMENT); + } else if ((seq = utf8_seq_len(p, left))) { + // U+FFFE and U+FFFF are valid UTF-8 but not XML. + if (seq == 3 && p[0] == 0xef && p[1] == 0xbf && + p[2] >= 0xbe) + strbuf_addstr(&sb, XML_REPLACEMENT); + else + strbuf_add(&sb, p, seq); + p += seq - 1; + left -= seq - 1; + } else { + strbuf_addstr(&sb, XML_REPLACEMENT); + } + p++; + left--; + } + html_raw(sb.buf, sb.len); + strbuf_release(&sb); +} + /* * cgit keeps an address with the angle brackets it was written with, while * Atom wants the bare address. @@ -43,7 +124,7 @@ static void print_email(const char *email) *end = '\0'; html("<email>"); - html_txt(start); + xml_txt(start); html("</email>\n"); free(copy); } @@ -57,24 +138,29 @@ static void print_entry(struct commit *commit, const char *host) hex = oid_to_hex(&commit->object.oid); html("<entry>\n"); html("<title>"); - html_txt(info->subject); + xml_txt(info->subject); html("</title>\n"); html("<updated>"); - html_txt(feed_date(info->committer_date)); + xml_txt(feed_date(info->committer_date)); html("</updated>\n"); html("<author>\n"); - if (info->author) { - html("<name>"); - html_txt(info->author); - html("</name>\n"); - } + // A person construct must hold a name, so a nameless commit falls + // back to the address and then to a placeholder. + html("<name>"); + if (info->author) + xml_txt(info->author); + else if (info->author_email) + xml_txt(info->author_email); + else + html("unknown"); + html("</name>\n"); if (info->author_email && !ctx.cfg.noplainemail) print_email(info->author_email); html("</author>\n"); html("<published>"); - html_txt(feed_date(info->author_date)); + xml_txt(feed_date(info->author_date)); html("</published>\n"); - if (host) { + { char *pageurl; char delim = '&'; @@ -95,8 +181,8 @@ static void print_entry(struct commit *commit, const char *host) html("<id>"); html_txtf("urn:%s:%s", the_hash_algo->name, hex); html("</id>\n"); - html("<content type='text'>\n"); - html_txt(info->msg); + html("<content type='text'>"); + xml_txt(info->msg); html("</content>\n"); html("</entry>\n"); cgit_free_commitinfo(info); @@ -132,31 +218,41 @@ void cgit_print_atom(char *tip, const char *path, int max_count) setup_revisions(argc, argv, &rev, NULL); prepare_revision_walk(&rev); + // CGI guarantees a server name, so only a bare test run reaches the + // fallback, and a placeholder there keeps the mandatory feed id and + // the links present rather than dropping them. host = cgit_hosturl(); - ctx.page.mimetype = "text/xml"; + if (!host) + host = xstrdup("localhost"); + ctx.page.mimetype = "application/atom+xml"; ctx.page.charset = "utf-8"; cgit_print_http_headers(); + html("<?xml version='1.0' encoding='utf-8'?>\n"); html("<feed xmlns='http://www.w3.org/2005/Atom'>\n"); html("<title>"); - html_txt(ctx.repo->name); + xml_txt(ctx.repo->name); if (path) { html("/"); - html_txt(path); + xml_txt(path); } if (tip && !ctx.qry.show_all) { html(", branch "); - html_txt(tip); + xml_txt(tip); } html("</title>\n"); html("<subtitle>"); - html_txt(ctx.repo->desc); + xml_txt(ctx.repo->desc); html("</subtitle>\n"); - if (host) { + { char *fullurl = cgit_currentfullurl(); char *repourl = cgit_repourl(ctx.repo->url); + struct strbuf idbuf = STRBUF_INIT; + + strbuf_addf(&idbuf, "%s%s%s", cgit_httpscheme(), host, fullurl); html("<id>"); - html_txtf("%s%s%s", cgit_httpscheme(), host, fullurl); + xml_txt(idbuf.buf); html("</id>\n"); + strbuf_release(&idbuf); html("<link rel='self' href='"); html_attrf("%s%s%s", cgit_httpscheme(), host, fullurl); html("'/>\n"); @@ -169,7 +265,7 @@ void cgit_print_atom(char *tip, const char *path, int max_count) while ((commit = get_revision(&rev)) != NULL) { if (need_updated) { html("<updated>"); - html_txt(feed_date(commit->date)); + xml_txt(feed_date(commit->date)); html("</updated>\n"); need_updated = false; } @@ -183,7 +279,7 @@ void cgit_print_atom(char *tip, const char *path, int max_count) // Atom makes a feed level updated mandatory, and an empty feed // has no commit to take one from. html("<updated>"); - html_txt(feed_date(time(NULL))); + xml_txt(feed_date(time(NULL))); html("</updated>\n"); } html("</feed>\n"); |
