diff options
| author | Bryce Kwon <bryce@brycekwon.com> | |
|---|---|---|
| committer | Bryce Kwon <bryce@brycekwon.com> | |
| commit | ||
| parent | ||
| tree | ||
| download | ||
Restyle the sources and fix the audit's findings
Diffstat (limited to 'source/parsing.c')
| -rw-r--r-- | source/parsing.c | 278 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
1 file changed, 144 insertions, 134 deletions
diff --git a/source/parsing.c b/source/parsing.c index 0d63b51..de2798a 100644 --- a/source/parsing.c +++ b/source/parsing.c @@ -1,127 +1,61 @@ -/* parsing.c: parsing of config files - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The two kinds of raw text cgit reads before it can render a page, the path + * of the incoming request and the body of a commit or a tag object. Reading + * the request path settles which repository and which page the request works + * on. Commit and tag objects arrive as header lines, a blank line, and then + * the message, so the readers here walk the headers, copy out the fields the + * pages need, and re-encode the message into the page encoding. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" +#include "parsing.h" +#include "shared.h" -/* - * url syntax: [repo ['/' cmd [ '/' path]]] - * repo: any valid repo url, may contain '/' - * cmd: log | commit | diff | tree | view | blob | snapshot - * path: any valid path, may contain '/' - * - */ -void cgit_parse_url(const char *url) -{ - char *c, *cmd, *p, *buf; - struct cgit_repo *repo; - - if (!url || url[0] == '\0') - return; - - ctx.qry.page = NULL; - ctx.repo = cgit_get_repoinfo(url); - if (ctx.repo) { - ctx.qry.repo = ctx.repo->url; - return; - } - - buf = xstrdup(url); - cmd = NULL; - c = strchr(buf, '/'); - while (c) { - c[0] = '\0'; - repo = cgit_get_repoinfo(buf); - if (repo) { - ctx.repo = repo; - cmd = c; - } - c[0] = '/'; - c = strchr(c + 1, '/'); - } - - if (ctx.repo) { - ctx.qry.repo = ctx.repo->url; - p = strchr(cmd + 1, '/'); - if (p) { - p[0] = '\0'; - if (p[1]) - ctx.qry.path = cgit_trim_end(p + 1, '/'); - } - if (cmd[1]) - ctx.qry.page = xstrdup(cmd + 1); - } - free(buf); -} - -static char *substr(const char *head, const char *tail) +static char *substr(const char *start, const char *end) { size_t len; char *buf; - if (tail < head) + if (end < start) return xstrdup(""); - // head points into the commit buffer, so strlcpy would measure the - // whole remaining commit just to copy a name or a subject off the front - // of it. - len = tail - head; + // start points into the object buffer, so strlcpy would measure the + // whole rest of the object to copy a name off the front of it. + len = end - start; buf = xmalloc(len + 1); - memcpy(buf, head, len); + memcpy(buf, start, len); buf[len] = '\0'; return buf; } -static void parse_user(const char *t, char **name, char **email, unsigned long *date, int *tz) +static void parse_user(const char *line, char **name, char **email, + timestamp_t *date, int *tz) { struct ident_split ident; - unsigned email_len; + struct strbuf address = STRBUF_INIT; + ptrdiff_t email_len; - if (!split_ident_line(&ident, t, strchrnul(t, '\n') - t)) { + if (!split_ident_line(&ident, line, strchrnul(line, '\n') - line)) { *name = substr(ident.name_begin, ident.name_end); + // Assembled rather than formatted. The length is a pointer + // difference, and the precision of a %.*s conversion has to be + // an int, which cannot hold one on a 64 bit host. email_len = ident.mail_end - ident.mail_begin; - *email = xmalloc(strlen("<") + email_len + strlen(">") + 1); - xsnprintf(*email, email_len + 3, "<%.*s>", email_len, ident.mail_begin); + strbuf_addch(&address, '<'); + if (email_len > 0) + strbuf_add(&address, ident.mail_begin, email_len); + strbuf_addch(&address, '>'); + *email = strbuf_detach(&address, NULL); if (ident.date_begin) - *date = strtoul(ident.date_begin, NULL, 10); + *date = parse_timestamp(ident.date_begin, NULL, 10); if (ident.tz_begin) *tz = atoi(ident.tz_begin); } } -#ifdef NO_ICONV -#define reencode(a, b, c) -#else -static const char *reencode(char **txt, const char *src_enc, const char *dst_enc) -{ - char *tmp; - - if (!txt) - return NULL; - - if (!*txt || !src_enc || !dst_enc) - return *txt; - - /* no encoding needed if src_enc equals dst_enc */ - if (!strcasecmp(src_enc, dst_enc)) - return *txt; - - tmp = reencode_string(*txt, dst_enc, src_enc); - if (tmp) { - free(*txt); - *txt = tmp; - } - return *txt; -} -#endif - static const char *next_header_line(const char *p) { p = strchr(p, '\n'); @@ -135,17 +69,89 @@ static int end_of_header(const char *p) return !p || (*p == '\n'); } +/* + * A git built without iconv still offers reencode_string, where it always + * fails, so the field is left in whatever encoding it arrived in. + */ +static const char *reencode(char **text, const char *from, const char *to) +{ + char *converted; + + if (!text) + return NULL; + + if (!*text || !from || !to) + return *text; + + if (!strcasecmp(from, to)) + return *text; + + converted = reencode_string(*text, to, from); + if (converted) { + free(*text); + *text = converted; + } + return *text; +} + +/* + * A repository url may itself contain slashes, so every slash-separated + * prefix is looked up and the longest one that names a repository wins. + */ +void cgit_parse_url(const char *url) +{ + char *buf, *slash, *repo_end, *page_end; + struct cgit_repo *repo; + + if (!url || url[0] == '\0') + return; + + ctx.qry.page = NULL; + ctx.repo = cgit_get_repoinfo(url); + if (ctx.repo) { + ctx.qry.repo = ctx.repo->url; + return; + } + + buf = xstrdup(url); + repo_end = NULL; + slash = strchr(buf, '/'); + while (slash) { + slash[0] = '\0'; + repo = cgit_get_repoinfo(buf); + if (repo) { + ctx.repo = repo; + repo_end = slash; + } + slash[0] = '/'; + slash = strchr(slash + 1, '/'); + } + + if (ctx.repo) { + ctx.qry.repo = ctx.repo->url; + page_end = strchr(repo_end + 1, '/'); + if (page_end) { + page_end[0] = '\0'; + if (page_end[1]) + ctx.qry.path = cgit_trim_end(page_end + 1, '/'); + } + if (repo_end[1]) + ctx.qry.page = xstrdup(repo_end + 1); + } + free(buf); +} + struct commitinfo *cgit_parse_commit(struct commit *commit) { - struct commitinfo *ret; + struct commitinfo *info; const char *p = repo_get_commit_buffer(the_repository, commit, NULL); - const char *t; + const char *eol; - ret = xcalloc(1, sizeof(struct commitinfo)); - ret->commit = commit; + info = xcalloc(1, sizeof(struct commitinfo)); + info->commit = commit; if (!p) - return ret; + return info; if (!skip_prefix(p, "tree ", &p)) die("Bad commit: %s", oid_to_hex(&commit->object.oid)); @@ -155,27 +161,29 @@ struct commitinfo *cgit_parse_commit(struct commit *commit) p += the_hash_algo->hexsz + 1; if (p && skip_prefix(p, "author ", &p)) { - parse_user(p, &ret->author, &ret->author_email, - &ret->author_date, &ret->author_tz); + parse_user(p, &info->author, &info->author_email, + &info->author_date, &info->author_tz); p = next_header_line(p); } if (p && skip_prefix(p, "committer ", &p)) { - parse_user(p, &ret->committer, &ret->committer_email, - &ret->committer_date, &ret->committer_tz); + parse_user(p, &info->committer, &info->committer_email, + &info->committer_date, &info->committer_tz); p = next_header_line(p); } if (p && skip_prefix(p, "encoding ", &p)) { - t = strchr(p, '\n'); - if (t) { - ret->msg_encoding = substr(p, t + 1); - p = t + 1; + eol = strchr(p, '\n'); + if (eol) { + info->msg_encoding = substr(p, eol + 1); + p = eol + 1; } } - if (!ret->msg_encoding) - ret->msg_encoding = xstrdup("UTF-8"); + // Git only writes the header when the message is in something other + // than UTF-8, so its absence means UTF-8. + if (!info->msg_encoding) + info->msg_encoding = xstrdup("UTF-8"); while (!end_of_header(p)) p = next_header_line(p); @@ -183,27 +191,28 @@ struct commitinfo *cgit_parse_commit(struct commit *commit) p++; if (p) { - t = strchrnul(p, '\n'); - ret->subject = substr(p, t); - while (*t == '\n') - t++; - ret->msg = xstrdup(t); + eol = strchrnul(p, '\n'); + info->subject = substr(p, eol); + while (*eol == '\n') + eol++; + info->msg = xstrdup(eol); } else { - // A crafted commit can end right after its headers with no - // message. Keep subject and msg as empty strings so callers - // can treat them as text unconditionally. - ret->subject = xstrdup(""); - ret->msg = xstrdup(""); + // Reached when an object is truncated mid header, which + // leaves nothing at all after them. Callers render subject + // and msg as text without checking, so they get empty + // strings rather than NULL. + info->subject = xstrdup(""); + info->msg = xstrdup(""); } - reencode(&ret->author, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->author_email, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->committer, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->committer_email, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->subject, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->msg, ret->msg_encoding, PAGE_ENCODING); + reencode(&info->author, info->msg_encoding, PAGE_ENCODING); + reencode(&info->author_email, info->msg_encoding, PAGE_ENCODING); + reencode(&info->committer, info->msg_encoding, PAGE_ENCODING); + reencode(&info->committer_email, info->msg_encoding, PAGE_ENCODING); + reencode(&info->subject, info->msg_encoding, PAGE_ENCODING); + reencode(&info->msg, info->msg_encoding, PAGE_ENCODING); - return ret; + return info; } struct taginfo *cgit_parse_tag(struct tag *tag) @@ -212,18 +221,19 @@ struct taginfo *cgit_parse_tag(struct tag *tag) enum object_type type; unsigned long size; const char *p; - struct taginfo *ret = NULL; + struct taginfo *info = NULL; - data = odb_read_object(the_repository->objects, &tag->object.oid, &type, &size); + data = odb_read_object(the_repository->objects, &tag->object.oid, + &type, &size); if (!data || type != OBJ_TAG) goto cleanup; - ret = xcalloc(1, sizeof(struct taginfo)); + info = xcalloc(1, sizeof(struct taginfo)); for (p = data; !end_of_header(p); p = next_header_line(p)) { if (skip_prefix(p, "tagger ", &p)) { - parse_user(p, &ret->tagger, &ret->tagger_email, - &ret->tagger_date, &ret->tagger_tz); + parse_user(p, &info->tagger, &info->tagger_email, + &info->tagger_date, &info->tagger_tz); } } @@ -231,9 +241,9 @@ struct taginfo *cgit_parse_tag(struct tag *tag) p++; if (p && *p) - ret->msg = xstrdup(p); + info->msg = xstrdup(p); cleanup: free(data); - return ret; + return info; } |
