diff options
context:
space:
mode:
authorBryce Kwon <bryce@brycekwon.com>
committerBryce Kwon <bryce@brycekwon.com>
commit
parent
tree
download
Restyle the sources and fix the audit's findings
Diffstat (limited to 'source/parsing.c')
-rw-r--r--source/parsing.c278
1 file changed, 144 insertions, 134 deletions
diff --git a/source/parsing.c b/source/parsing.c
index 0d63b51..de2798a 100644
--- a/source/parsing.c
+++ b/source/parsing.c
@@ -1,127 +1,61 @@
-/* parsing.c: parsing of config files
- *
- * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com>
- *
- * Licensed under GNU General Public License v2
- * (see LICENSE.txt for full license text)
+/*
+ * The two kinds of raw text cgit reads before it can render a page, the path
+ * of the incoming request and the body of a commit or a tag object. Reading
+ * the request path settles which repository and which page the request works
+ * on. Commit and tag objects arrive as header lines, a blank line, and then
+ * the message, so the readers here walk the headers, copy out the fields the
+ * pages need, and re-encode the message into the page encoding.
*/
#define USE_THE_REPOSITORY_VARIABLE
#include "cgit.h"
+#include "parsing.h"
+#include "shared.h"
-/*
- * url syntax: [repo ['/' cmd [ '/' path]]]
- * repo: any valid repo url, may contain '/'
- * cmd: log | commit | diff | tree | view | blob | snapshot
- * path: any valid path, may contain '/'
- *
- */
-void cgit_parse_url(const char *url)
-{
- char *c, *cmd, *p, *buf;
- struct cgit_repo *repo;
-
- if (!url || url[0] == '\0')
- return;
-
- ctx.qry.page = NULL;
- ctx.repo = cgit_get_repoinfo(url);
- if (ctx.repo) {
- ctx.qry.repo = ctx.repo->url;
- return;
- }
-
- buf = xstrdup(url);
- cmd = NULL;
- c = strchr(buf, '/');
- while (c) {
- c[0] = '\0';
- repo = cgit_get_repoinfo(buf);
- if (repo) {
- ctx.repo = repo;
- cmd = c;
- }
- c[0] = '/';
- c = strchr(c + 1, '/');
- }
-
- if (ctx.repo) {
- ctx.qry.repo = ctx.repo->url;
- p = strchr(cmd + 1, '/');
- if (p) {
- p[0] = '\0';
- if (p[1])
- ctx.qry.path = cgit_trim_end(p + 1, '/');
- }
- if (cmd[1])
- ctx.qry.page = xstrdup(cmd + 1);
- }
- free(buf);
-}
-
-static char *substr(const char *head, const char *tail)
+static char *substr(const char *start, const char *end)
{
size_t len;
char *buf;
- if (tail < head)
+ if (end < start)
return xstrdup("");
- // head points into the commit buffer, so strlcpy would measure the
- // whole remaining commit just to copy a name or a subject off the front
- // of it.
- len = tail - head;
+ // start points into the object buffer, so strlcpy would measure the
+ // whole rest of the object to copy a name off the front of it.
+ len = end - start;
buf = xmalloc(len + 1);
- memcpy(buf, head, len);
+ memcpy(buf, start, len);
buf[len] = '\0';
return buf;
}
-static void parse_user(const char *t, char **name, char **email, unsigned long *date, int *tz)
+static void parse_user(const char *line, char **name, char **email,
+ timestamp_t *date, int *tz)
{
struct ident_split ident;
- unsigned email_len;
+ struct strbuf address = STRBUF_INIT;
+ ptrdiff_t email_len;
- if (!split_ident_line(&ident, t, strchrnul(t, '\n') - t)) {
+ if (!split_ident_line(&ident, line, strchrnul(line, '\n') - line)) {
*name = substr(ident.name_begin, ident.name_end);
+ // Assembled rather than formatted. The length is a pointer
+ // difference, and the precision of a %.*s conversion has to be
+ // an int, which cannot hold one on a 64 bit host.
email_len = ident.mail_end - ident.mail_begin;
- *email = xmalloc(strlen("<") + email_len + strlen(">") + 1);
- xsnprintf(*email, email_len + 3, "<%.*s>", email_len, ident.mail_begin);
+ strbuf_addch(&address, '<');
+ if (email_len > 0)
+ strbuf_add(&address, ident.mail_begin, email_len);
+ strbuf_addch(&address, '>');
+ *email = strbuf_detach(&address, NULL);
if (ident.date_begin)
- *date = strtoul(ident.date_begin, NULL, 10);
+ *date = parse_timestamp(ident.date_begin, NULL, 10);
if (ident.tz_begin)
*tz = atoi(ident.tz_begin);
}
}
-#ifdef NO_ICONV
-#define reencode(a, b, c)
-#else
-static const char *reencode(char **txt, const char *src_enc, const char *dst_enc)
-{
- char *tmp;
-
- if (!txt)
- return NULL;
-
- if (!*txt || !src_enc || !dst_enc)
- return *txt;
-
- /* no encoding needed if src_enc equals dst_enc */
- if (!strcasecmp(src_enc, dst_enc))
- return *txt;
-
- tmp = reencode_string(*txt, dst_enc, src_enc);
- if (tmp) {
- free(*txt);
- *txt = tmp;
- }
- return *txt;
-}
-#endif
-
static const char *next_header_line(const char *p)
{
p = strchr(p, '\n');
@@ -135,17 +69,89 @@ static int end_of_header(const char *p)
return !p || (*p == '\n');
}
+/*
+ * A git built without iconv still offers reencode_string, where it always
+ * fails, so the field is left in whatever encoding it arrived in.
+ */
+static const char *reencode(char **text, const char *from, const char *to)
+{
+ char *converted;
+
+ if (!text)
+ return NULL;
+
+ if (!*text || !from || !to)
+ return *text;
+
+ if (!strcasecmp(from, to))
+ return *text;
+
+ converted = reencode_string(*text, to, from);
+ if (converted) {
+ free(*text);
+ *text = converted;
+ }
+ return *text;
+}
+
+/*
+ * A repository url may itself contain slashes, so every slash-separated
+ * prefix is looked up and the longest one that names a repository wins.
+ */
+void cgit_parse_url(const char *url)
+{
+ char *buf, *slash, *repo_end, *page_end;
+ struct cgit_repo *repo;
+
+ if (!url || url[0] == '\0')
+ return;
+
+ ctx.qry.page = NULL;
+ ctx.repo = cgit_get_repoinfo(url);
+ if (ctx.repo) {
+ ctx.qry.repo = ctx.repo->url;
+ return;
+ }
+
+ buf = xstrdup(url);
+ repo_end = NULL;
+ slash = strchr(buf, '/');
+ while (slash) {
+ slash[0] = '\0';
+ repo = cgit_get_repoinfo(buf);
+ if (repo) {
+ ctx.repo = repo;
+ repo_end = slash;
+ }
+ slash[0] = '/';
+ slash = strchr(slash + 1, '/');
+ }
+
+ if (ctx.repo) {
+ ctx.qry.repo = ctx.repo->url;
+ page_end = strchr(repo_end + 1, '/');
+ if (page_end) {
+ page_end[0] = '\0';
+ if (page_end[1])
+ ctx.qry.path = cgit_trim_end(page_end + 1, '/');
+ }
+ if (repo_end[1])
+ ctx.qry.page = xstrdup(repo_end + 1);
+ }
+ free(buf);
+}
+
struct commitinfo *cgit_parse_commit(struct commit *commit)
{
- struct commitinfo *ret;
+ struct commitinfo *info;
const char *p = repo_get_commit_buffer(the_repository, commit, NULL);
- const char *t;
+ const char *eol;
- ret = xcalloc(1, sizeof(struct commitinfo));
- ret->commit = commit;
+ info = xcalloc(1, sizeof(struct commitinfo));
+ info->commit = commit;
if (!p)
- return ret;
+ return info;
if (!skip_prefix(p, "tree ", &p))
die("Bad commit: %s", oid_to_hex(&commit->object.oid));
@@ -155,27 +161,29 @@ struct commitinfo *cgit_parse_commit(struct commit *commit)
p += the_hash_algo->hexsz + 1;
if (p && skip_prefix(p, "author ", &p)) {
- parse_user(p, &ret->author, &ret->author_email,
- &ret->author_date, &ret->author_tz);
+ parse_user(p, &info->author, &info->author_email,
+ &info->author_date, &info->author_tz);
p = next_header_line(p);
}
if (p && skip_prefix(p, "committer ", &p)) {
- parse_user(p, &ret->committer, &ret->committer_email,
- &ret->committer_date, &ret->committer_tz);
+ parse_user(p, &info->committer, &info->committer_email,
+ &info->committer_date, &info->committer_tz);
p = next_header_line(p);
}
if (p && skip_prefix(p, "encoding ", &p)) {
- t = strchr(p, '\n');
- if (t) {
- ret->msg_encoding = substr(p, t + 1);
- p = t + 1;
+ eol = strchr(p, '\n');
+ if (eol) {
+ info->msg_encoding = substr(p, eol + 1);
+ p = eol + 1;
}
}
- if (!ret->msg_encoding)
- ret->msg_encoding = xstrdup("UTF-8");
+ // Git only writes the header when the message is in something other
+ // than UTF-8, so its absence means UTF-8.
+ if (!info->msg_encoding)
+ info->msg_encoding = xstrdup("UTF-8");
while (!end_of_header(p))
p = next_header_line(p);
@@ -183,27 +191,28 @@ struct commitinfo *cgit_parse_commit(struct commit *commit)
p++;
if (p) {
- t = strchrnul(p, '\n');
- ret->subject = substr(p, t);
- while (*t == '\n')
- t++;
- ret->msg = xstrdup(t);
+ eol = strchrnul(p, '\n');
+ info->subject = substr(p, eol);
+ while (*eol == '\n')
+ eol++;
+ info->msg = xstrdup(eol);
} else {
- // A crafted commit can end right after its headers with no
- // message. Keep subject and msg as empty strings so callers
- // can treat them as text unconditionally.
- ret->subject = xstrdup("");
- ret->msg = xstrdup("");
+ // Reached when an object is truncated mid header, which
+ // leaves nothing at all after them. Callers render subject
+ // and msg as text without checking, so they get empty
+ // strings rather than NULL.
+ info->subject = xstrdup("");
+ info->msg = xstrdup("");
}
- reencode(&ret->author, ret->msg_encoding, PAGE_ENCODING);
- reencode(&ret->author_email, ret->msg_encoding, PAGE_ENCODING);
- reencode(&ret->committer, ret->msg_encoding, PAGE_ENCODING);
- reencode(&ret->committer_email, ret->msg_encoding, PAGE_ENCODING);
- reencode(&ret->subject, ret->msg_encoding, PAGE_ENCODING);
- reencode(&ret->msg, ret->msg_encoding, PAGE_ENCODING);
+ reencode(&info->author, info->msg_encoding, PAGE_ENCODING);
+ reencode(&info->author_email, info->msg_encoding, PAGE_ENCODING);
+ reencode(&info->committer, info->msg_encoding, PAGE_ENCODING);
+ reencode(&info->committer_email, info->msg_encoding, PAGE_ENCODING);
+ reencode(&info->subject, info->msg_encoding, PAGE_ENCODING);
+ reencode(&info->msg, info->msg_encoding, PAGE_ENCODING);
- return ret;
+ return info;
}
struct taginfo *cgit_parse_tag(struct tag *tag)
@@ -212,18 +221,19 @@ struct taginfo *cgit_parse_tag(struct tag *tag)
enum object_type type;
unsigned long size;
const char *p;
- struct taginfo *ret = NULL;
+ struct taginfo *info = NULL;
- data = odb_read_object(the_repository->objects, &tag->object.oid, &type, &size);
+ data = odb_read_object(the_repository->objects, &tag->object.oid,
+ &type, &size);
if (!data || type != OBJ_TAG)
goto cleanup;
- ret = xcalloc(1, sizeof(struct taginfo));
+ info = xcalloc(1, sizeof(struct taginfo));
for (p = data; !end_of_header(p); p = next_header_line(p)) {
if (skip_prefix(p, "tagger ", &p)) {
- parse_user(p, &ret->tagger, &ret->tagger_email,
- &ret->tagger_date, &ret->tagger_tz);
+ parse_user(p, &info->tagger, &info->tagger_email,
+ &info->tagger_date, &info->tagger_tz);
}
}
@@ -231,9 +241,9 @@ struct taginfo *cgit_parse_tag(struct tag *tag)
p++;
if (p && *p)
- ret->msg = xstrdup(p);
+ info->msg = xstrdup(p);
cleanup:
free(data);
- return ret;
+ return info;
}