/* * The two kinds of raw text cgit reads before it can render a page, the path * of the incoming request and the body of a commit or a tag object. Reading * the request path settles which repository and which page the request works * on. Commit and tag objects arrive as header lines, a blank line, and then * the message, so the readers here walk the headers, copy out the fields the * pages need, and re-encode the message into the page encoding. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" #include "parsing.h" #include "shared.h" static char *substr(const char *start, const char *end) { size_t len; char *buf; if (end < start) return xstrdup(""); // start points into the object buffer, so strlcpy would measure the // whole rest of the object to copy a name off the front of it. len = end - start; buf = xmalloc(len + 1); memcpy(buf, start, len); buf[len] = '\0'; return buf; } static void parse_user(const char *line, char **name, char **email, timestamp_t *date, int *tz) { struct ident_split ident; struct strbuf address = STRBUF_INIT; ptrdiff_t email_len; if (!split_ident_line(&ident, line, strchrnul(line, '\n') - line)) { *name = substr(ident.name_begin, ident.name_end); // Assembled rather than formatted. The length is a pointer // difference, and the precision of a %.*s conversion has to be // an int, which cannot hold one on a 64 bit host. email_len = ident.mail_end - ident.mail_begin; strbuf_addch(&address, '<'); if (email_len > 0) strbuf_add(&address, ident.mail_begin, email_len); strbuf_addch(&address, '>'); *email = strbuf_detach(&address, NULL); if (ident.date_begin) *date = parse_timestamp(ident.date_begin, NULL, 10); if (ident.tz_begin) *tz = atoi(ident.tz_begin); } } static const char *next_header_line(const char *p) { p = strchr(p, '\n'); if (!p) return NULL; return p + 1; } static int end_of_header(const char *p) { return !p || (*p == '\n'); } /* * A git built without iconv still offers reencode_string, where it always * fails, so the field is left in whatever encoding it arrived in. */ static const char *reencode(char **text, const char *from, const char *to) { char *converted; if (!text) return NULL; if (!*text || !from || !to) return *text; if (!strcasecmp(from, to)) return *text; converted = reencode_string(*text, to, from); if (converted) { free(*text); *text = converted; } return *text; } /* * A repository url may itself contain slashes, so every slash-separated * prefix is looked up and the longest one that names a repository wins. */ void cgit_parse_url(const char *url) { char *buf, *slash, *repo_end, *page_end; struct cgit_repo *repo; if (!url || url[0] == '\0') return; ctx.qry.page = NULL; ctx.repo = cgit_get_repoinfo(url); if (ctx.repo) { ctx.qry.repo = ctx.repo->url; return; } buf = xstrdup(url); repo_end = NULL; slash = strchr(buf, '/'); while (slash) { slash[0] = '\0'; repo = cgit_get_repoinfo(buf); if (repo) { ctx.repo = repo; repo_end = slash; } slash[0] = '/'; slash = strchr(slash + 1, '/'); } if (ctx.repo) { ctx.qry.repo = ctx.repo->url; page_end = strchr(repo_end + 1, '/'); if (page_end) { page_end[0] = '\0'; if (page_end[1]) ctx.qry.path = cgit_trim_end(page_end + 1, '/'); } if (repo_end[1]) ctx.qry.page = xstrdup(repo_end + 1); } free(buf); } struct commitinfo *cgit_parse_commit(struct commit *commit) { struct commitinfo *info; const char *p = repo_get_commit_buffer(the_repository, commit, NULL); const char *eol; info = xcalloc(1, sizeof(struct commitinfo)); info->commit = commit; if (!p) return info; if (!skip_prefix(p, "tree ", &p)) die("Bad commit: %s", oid_to_hex(&commit->object.oid)); p += the_hash_algo->hexsz + 1; while (skip_prefix(p, "parent ", &p)) p += the_hash_algo->hexsz + 1; if (p && skip_prefix(p, "author ", &p)) { parse_user(p, &info->author, &info->author_email, &info->author_date, &info->author_tz); p = next_header_line(p); } if (p && skip_prefix(p, "committer ", &p)) { parse_user(p, &info->committer, &info->committer_email, &info->committer_date, &info->committer_tz); p = next_header_line(p); } if (p && skip_prefix(p, "encoding ", &p)) { eol = strchr(p, '\n'); if (eol) { info->msg_encoding = substr(p, eol + 1); p = eol + 1; } } // Git only writes the header when the message is in something other // than UTF-8, so its absence means UTF-8. if (!info->msg_encoding) info->msg_encoding = xstrdup("UTF-8"); while (!end_of_header(p)) p = next_header_line(p); while (p && *p == '\n') p++; if (p) { eol = strchrnul(p, '\n'); info->subject = substr(p, eol); while (*eol == '\n') eol++; info->msg = xstrdup(eol); } else { // Reached when an object is truncated mid header, which // leaves nothing at all after them. Callers render subject // and msg as text without checking, so they get empty // strings rather than NULL. info->subject = xstrdup(""); info->msg = xstrdup(""); } reencode(&info->author, info->msg_encoding, PAGE_ENCODING); reencode(&info->author_email, info->msg_encoding, PAGE_ENCODING); reencode(&info->committer, info->msg_encoding, PAGE_ENCODING); reencode(&info->committer_email, info->msg_encoding, PAGE_ENCODING); reencode(&info->subject, info->msg_encoding, PAGE_ENCODING); reencode(&info->msg, info->msg_encoding, PAGE_ENCODING); return info; } struct taginfo *cgit_parse_tag(struct tag *tag) { void *data; enum object_type type; unsigned long size; const char *p; struct taginfo *info = NULL; data = odb_read_object(the_repository->objects, &tag->object.oid, &type, &size); if (!data || type != OBJ_TAG) goto cleanup; info = xcalloc(1, sizeof(struct taginfo)); for (p = data; !end_of_header(p); p = next_header_line(p)) { if (skip_prefix(p, "tagger ", &p)) { parse_user(p, &info->tagger, &info->tagger_email, &info->tagger_date, &info->tagger_tz); } } while (p && *p == '\n') p++; if (p && *p) info->msg = xstrdup(p); cleanup: free(data); return info; }