From 80767bc9732bf6716697198e53ff2cb8d4ae96be Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Wed, 12 Aug 2026 18:23:17 -1000 Subject: Restyle the sources and fix the audit's findings --- source/cache.c | 464 +++++++------ source/cache.h | 41 +- source/cgit.c | 1761 ++++++++++++++++++++++++++----------------------- source/cgit.h | 186 ++---- source/cgit.mk | 61 +- source/cmd.c | 164 ++--- source/cmd.h | 16 +- source/config.c | 113 ++++ source/config.h | 16 + source/configfile.c | 100 --- source/configfile.h | 10 - source/filter.c | 324 +++++---- source/filter.h | 54 ++ source/gen-version.sh | 34 +- source/html.c | 277 ++++---- source/html.h | 70 +- source/parsing.c | 278 ++++---- source/parsing.h | 20 + source/scan-tree.c | 254 +++---- source/scan-tree.h | 14 + source/shared.c | 620 +++++++++-------- source/shared.h | 92 +++ source/ui-atom.c | 107 +-- source/ui-atom.h | 11 +- source/ui-blame.c | 324 +++++---- source/ui-blame.h | 13 +- source/ui-blob.c | 147 +++-- source/ui-blob.h | 22 +- source/ui-clone.c | 107 ++- source/ui-clone.h | 19 +- source/ui-commit.c | 186 +++--- source/ui-commit.h | 12 +- source/ui-diff.c | 629 +++++++++--------- source/ui-diff.h | 20 +- source/ui-empty.c | 15 +- source/ui-empty.h | 13 +- source/ui-log.c | 507 +++++++------- source/ui-log.h | 13 +- source/ui-patch.c | 124 ++-- source/ui-patch.h | 13 +- source/ui-plain.c | 144 ++-- source/ui-plain.h | 12 +- source/ui-refs.c | 252 +++---- source/ui-refs.h | 23 +- source/ui-repolist.c | 363 +++++----- source/ui-repolist.h | 13 +- source/ui-shared.c | 1525 +++++++++++++++++++++--------------------- source/ui-shared.h | 64 +- source/ui-snapshot.c | 298 +++++---- source/ui-snapshot.h | 40 +- source/ui-ssdiff.c | 501 +++++++------- source/ui-ssdiff.h | 19 +- source/ui-stats.c | 574 ++++++++-------- source/ui-stats.h | 36 +- source/ui-summary.c | 143 ++-- source/ui-summary.h | 13 +- source/ui-tag.c | 174 ++--- source/ui-tag.h | 12 +- source/ui-tree.c | 381 ++++++----- source/ui-tree.h | 13 +- 60 files changed, 6449 insertions(+), 5402 deletions(-) create mode 100644 source/config.c create mode 100644 source/config.h delete mode 100644 source/configfile.c delete mode 100644 source/configfile.h create mode 100644 source/filter.h create mode 100644 source/parsing.h create mode 100644 source/shared.h (limited to 'source') diff --git a/source/cache.c b/source/cache.c index c6d0427..d6e450a 100644 --- a/source/cache.c +++ b/source/cache.c @@ -1,33 +1,49 @@ -/* cache.c: cache management - * - * Copyright (C) 2006-2014 cgit Development Team - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) - * - * - * The cache is just a directory structure where each file is a cache slot, - * and each filename is based on the hash of some key (e.g. the cgit url). - * Each file contains the full key followed by the cached content for that - * key. - * +/* + * The cache that lets a repeated request be answered from disk instead of + * being rendered again. A slot is one file named after the hash of the + * request key, holding that key and then the page it rendered to, and a lock + * file beside it is where a replacement page is written before being renamed + * over the slot. Only the process holding that lock rebuilds a slot, so a + * request arriving while a stale slot is being rebuilt is served the stale + * page, and a request with no usable slot at all renders straight to the + * client without caching anything. */ -#include "cgit.h" #include "cache.h" +#include "cgit.h" #include "html.h" +#include "shared.h" #ifdef HAVE_LINUX_SENDFILE #include #endif +// One read of a slot file. The stored key has to be recognised out of a +// single such read, so this also bounds how long a cacheable key can be, see +// key_fits_slot. #define CACHE_BUFSIZE (1024 * 4) -/* Crude implementation of 32-bit FNV-1 hash algorithm, - * see http://www.isthe.com/chongo/tech/comp/fnv/ for details - * about the magic numbers. - */ + +// A slot is named by this many hex digits of the key hash, which is also how +// cache_ls tells slots from the lock files sitting beside them. +#define SLOT_NAME_LEN 8 + +// The 32 bit FNV-1 offset basis and prime. #define FNV_OFFSET 0x811c9dc5 #define FNV_PRIME 0x01000193 +/* + * Cache trouble goes to stderr, which under CGI is the web server's error + * log, so that it cannot land in the middle of the page being written to + * stdout. + */ +__attribute__((format (printf,1,2))) +static void log_error(const char *format, ...) +{ + va_list args; + va_start(args, format); + vfprintf(stderr, format, args); + va_end(args); +} + struct cache_slot { const char *key; size_t keylen; @@ -35,56 +51,56 @@ struct cache_slot { cache_fill_fn fn; int cache_fd; int lock_fd; - int stdout_fd; - const char *cache_name; - const char *lock_name; - int match; - struct stat cache_st; - int bufsize; + int saved_stdout; + const char *path; + const char *lock_path; + int key_matches; + // The slot as it was when it was opened, or the lock file once + // fill_slot has written a page into it. + struct stat st; + // How much of the slot was read into buf, not the size of buf. + int buflen; char buf[CACHE_BUFSIZE]; }; -/* Open an existing cache slot and fill the cache buffer with - * (part of) the content of the cache file. Return 0 on success - * and errno otherwise. - */ static int open_slot(struct cache_slot *slot) { - char *bufz; - ssize_t bufkeylen = -1; + char *nul; + ssize_t keylen = -1; - slot->cache_fd = open(slot->cache_name, O_RDONLY); + slot->cache_fd = open(slot->path, O_RDONLY); if (slot->cache_fd == -1) return errno; - if (fstat(slot->cache_fd, &slot->cache_st)) + if (fstat(slot->cache_fd, &slot->st)) return errno; - slot->bufsize = xread(slot->cache_fd, slot->buf, sizeof(slot->buf)); - if (slot->bufsize < 0) + slot->buflen = xread(slot->cache_fd, slot->buf, sizeof(slot->buf)); + if (slot->buflen < 0) return errno; - bufz = memchr(slot->buf, 0, slot->bufsize); - if (bufz) - bufkeylen = bufz - slot->buf; + nul = memchr(slot->buf, 0, slot->buflen); + if (nul) + keylen = nul - slot->buf; if (slot->key) - slot->match = bufkeylen >= 0 && (size_t)bufkeylen == slot->keylen && - !memcmp(slot->key, slot->buf, bufkeylen + 1); + slot->key_matches = keylen >= 0 && + (size_t)keylen == slot->keylen && + !memcmp(slot->key, slot->buf, keylen + 1); return 0; } -/* A key longer than the buffer above can never be read back, so a slot keyed - * on one would never match and every such request would regenerate its page - * while still writing a slot nothing can use. Those requests skip the cache - * instead. */ +/* + * A key longer than the buffer above can never be read back by open_slot, so + * a slot keyed on one would never match and every such request would + * regenerate its page while still writing a slot nothing can use. + */ static int key_fits_slot(const char *key) { return strlen(key) + 1 <= CACHE_BUFSIZE; } -/* Close the active cache slot */ static int close_slot(struct cache_slot *slot) { int err = 0; @@ -97,7 +113,6 @@ static int close_slot(struct cache_slot *slot) return err; } -/* Print the content of the active cache slot (but skip the key). */ static int print_slot(struct cache_slot *slot) { off_t off; @@ -108,7 +123,7 @@ static int print_slot(struct cache_slot *slot) off = slot->keylen + 1; #ifdef HAVE_LINUX_SENDFILE - size = slot->cache_st.st_size; + size = slot->st.st_size; do { ssize_t ret; @@ -116,7 +131,10 @@ static int print_slot(struct cache_slot *slot) if (ret < 0) { if (errno == EAGAIN || errno == EINTR) continue; - /* Fall back to read/write on EINVAL or ENOSYS */ + // EINVAL and ENOSYS mean this kernel or this pair of + // descriptors cannot do sendfile at all, so fall back + // to the read and write loop rather than fail the + // request. if (errno == EINVAL || errno == ENOSYS) break; return errno; @@ -141,30 +159,41 @@ static int print_slot(struct cache_slot *slot) } while (1); } -/* Check if the slot has expired */ +static int serve_slot(struct cache_slot *slot) +{ + int err; + + err = print_slot(slot); + if (err) + log_error("[cgit] error printing cache %s: %s (%d)\n", + slot->path, + strerror(err), + err); + return err; +} + static int is_expired(struct cache_slot *slot) { if (slot->ttl < 0) return 0; - else - return slot->cache_st.st_mtime + slot->ttl * 60 < time(NULL); + return slot->st.st_mtime + slot->ttl * SECONDS_PER_MINUTE < time(NULL); } -/* Check if the slot has been modified since we opened it. - * NB: If stat() fails, we pretend the file is modified. +/* + * A stat that fails counts as modified, so that the caller leaves alone a file + * it was unable to look at. */ static int is_modified(struct cache_slot *slot) { - struct stat st; + struct stat current; - if (stat(slot->cache_name, &st)) + if (stat(slot->path, ¤t)) return 1; - return (st.st_ino != slot->cache_st.st_ino || - st.st_mtime != slot->cache_st.st_mtime || - st.st_size != slot->cache_st.st_size); + return (current.st_ino != slot->st.st_ino || + current.st_mtime != slot->st.st_mtime || + current.st_size != slot->st.st_size); } -/* Close an open lockfile */ static int close_lock(struct cache_slot *slot) { int err = 0; @@ -177,9 +206,11 @@ static int close_lock(struct cache_slot *slot) return err; } -/* Create a lockfile used to store the generated content for a cache - * slot, and write the slot key + \0 into it. - * Returns 0 on success and errno otherwise. +/* + * The lock file becomes the slot once it is renamed, so it has to open with + * the key the same way a slot does. The lock is taken without blocking, + * because failing to get it is how a second process learns that this slot is + * already being rebuilt, so it returns an errno instead of waiting. */ static int lock_slot(struct cache_slot *slot) { @@ -190,7 +221,7 @@ static int lock_slot(struct cache_slot *slot) .l_len = 0, }; - slot->lock_fd = open(slot->lock_name, O_RDWR | O_CREAT, + slot->lock_fd = open(slot->lock_path, O_RDWR | O_CREAT, S_IRUSR | S_IWUSR); if (slot->lock_fd == -1) return errno; @@ -200,6 +231,8 @@ static int lock_slot(struct cache_slot *slot) slot->lock_fd = -1; return saved_errno; } + // A run that died before its rename leaves the lock file behind, so + // start from empty now that nobody else can be writing it. if (ftruncate(slot->lock_fd, 0) < 0) return errno; if (xwrite(slot->lock_fd, slot->key, slot->keylen + 1) < 0) @@ -207,24 +240,19 @@ static int lock_slot(struct cache_slot *slot) return 0; } -/* Release the current lockfile. If `replace_old_slot` is set the - * lockfile replaces the old cache slot, otherwise the lockfile is - * just deleted. - */ static int unlock_slot(struct cache_slot *slot, int replace_old_slot) { int err; if (replace_old_slot) - err = rename(slot->lock_name, slot->cache_name); + err = rename(slot->lock_path, slot->path); else - err = unlink(slot->lock_name); + err = unlink(slot->lock_path); - /* Restore stdout and close the temporary FD. */ - if (slot->stdout_fd >= 0) { - dup2(slot->stdout_fd, STDOUT_FILENO); - close(slot->stdout_fd); - slot->stdout_fd = -1; + if (slot->saved_stdout >= 0) { + dup2(slot->saved_stdout, STDOUT_FILENO); + close(slot->saved_stdout); + slot->saved_stdout = -1; } if (err) @@ -233,48 +261,84 @@ static int unlock_slot(struct cache_slot *slot, int replace_old_slot) return 0; } -/* Generate the content for the current cache slot by redirecting - * stdout to the lock-fd and invoking the callback function +// Only one slot is ever being filled at a time, so a single pointer is enough +// for cache_abandon_fill to find its way back to the client. +static struct cache_slot *slot_being_filled; + +void cache_abandon_fill(void) +{ + struct cache_slot *slot = slot_being_filled; + + if (!slot) + return; + slot_being_filled = NULL; + + // Emptied while stdout still points at the lock file, so the half + // rendered page goes into the file about to be removed rather than + // reaching the client ahead of whatever is written next. + html_flush(); + + if (slot->saved_stdout >= 0) { + dup2(slot->saved_stdout, STDOUT_FILENO); + close(slot->saved_stdout); + slot->saved_stdout = -1; + } + unlink(slot->lock_path); +} + +/* + * Renders with stdout pointed at the lock file, and on success or failure + * alike it is unlock_slot that gives stdout back. */ static int fill_slot(struct cache_slot *slot) { - /* Preserve stdout */ - slot->stdout_fd = dup(STDOUT_FILENO); - if (slot->stdout_fd == -1) + slot->saved_stdout = dup(STDOUT_FILENO); + if (slot->saved_stdout == -1) return errno; - /* Redirect stdout to lockfile */ if (dup2(slot->lock_fd, STDOUT_FILENO) == -1) return errno; - /* Generate cache content */ + slot_being_filled = slot; slot->fn(); + slot_being_filled = NULL; - /* Make sure any buffered data is flushed to the file */ + // The page is sitting in html.c's buffer and then in stdio's, and all + // of it has to reach the lock file before that file is renamed into + // place. html_flush(); if (fflush(stdout)) return errno; - /* update stat info */ - if (fstat(slot->lock_fd, &slot->cache_st)) + // print_slot takes the length of what it copies from here, and what + // it copies after a fill is the lock file rather than the old slot. + if (fstat(slot->lock_fd, &slot->st)) return errno; return 0; } -unsigned long cache_hash_str(const char *str) +/* + * Giving up is always a valid outcome, because the caller still has the + * expired copy open and can serve that. + */ +static void refresh_slot(struct cache_slot *slot) { - unsigned long h = FNV_OFFSET; - unsigned char *s = (unsigned char *)str; - - if (!s) - return h; - - while (*s) { - h *= FNV_PRIME; - h ^= *s++; + if (lock_slot(slot)) + return; + + // If another process replaced the slot between open_slot and + // lock_slot, the copy already open is served rather than the newer + // one, which would mean opening that file and comparing the key in it, + // not worth a second descriptor and read on every expiry. + if (is_modified(slot) || fill_slot(slot)) { + unlock_slot(slot, 0); + close_lock(slot); + } else { + close_slot(slot); + unlock_slot(slot, 1); + slot->cache_fd = slot->lock_fd; } - return h; } static int process_slot(struct cache_slot *slot) @@ -282,216 +346,180 @@ static int process_slot(struct cache_slot *slot) int err; err = open_slot(slot); - if (!err && slot->match) { - if (is_expired(slot)) { - if (!lock_slot(slot)) { - /* If the cachefile has been replaced between - * `open_slot` and `lock_slot`, we'll just - * serve the stale content from the original - * cachefile. This way we avoid pruning the - * newly generated slot. The same code-path - * is chosen if fill_slot() fails for some - * reason. - * - * TODO? check if the new slot contains the - * same key as the old one, since we would - * prefer to serve the newest content. - * This will require us to open yet another - * file-descriptor and read and compare the - * key from the new file, so for now we're - * lazy and just ignore the new file. - */ - if (is_modified(slot) || fill_slot(slot)) { - unlock_slot(slot, 0); - close_lock(slot); - } else { - close_slot(slot); - unlock_slot(slot, 1); - slot->cache_fd = slot->lock_fd; - } - } - } - if ((err = print_slot(slot)) != 0) { - cache_log("[cgit] error printing cache %s: %s (%d)\n", - slot->cache_name, - strerror(err), - err); - } + if (!err && slot->key_matches) { + if (is_expired(slot)) + refresh_slot(slot); + err = serve_slot(slot); close_slot(slot); return err; } - /* If the cache slot does not exist (or its key doesn't match the - * current key), lets try to create a new cache slot for this - * request. If this fails (for whatever reason), lets just generate - * the content without caching it and fool the caller to believe - * everything worked out (but print a warning on stdout). - */ - + // If any part of creating a slot fails the page is still rendered + // straight to the client and the caller is told the request succeeded, + // because it did. close_slot(slot); if ((err = lock_slot(slot)) != 0) { - cache_log("[cgit] Unable to lock slot %s: %s (%d)\n", - slot->lock_name, strerror(err), err); + log_error("[cgit] Unable to lock slot %s: %s (%d)\n", + slot->lock_path, strerror(err), err); slot->fn(); return 0; } if ((err = fill_slot(slot)) != 0) { - cache_log("[cgit] Unable to fill slot %s: %s (%d)\n", - slot->lock_name, strerror(err), err); + log_error("[cgit] Unable to fill slot %s: %s (%d)\n", + slot->lock_path, strerror(err), err); unlock_slot(slot, 0); close_lock(slot); slot->fn(); return 0; } - // We've got a valid cache slot in the lock file, which - // is about to replace the old cache slot. But if we - // release the lockfile and then try to open the new cache - // slot, we might get a race condition with a concurrent - // writer for the same cache slot (with a different key). - // Lets avoid such a race by just printing the content of - // the lock file. + + // Opening the slot by name after the rename could land on a file a + // concurrent writer put there for a different key, so what gets + // printed is the descriptor still open on the lock file. slot->cache_fd = slot->lock_fd; unlock_slot(slot, 1); - if ((err = print_slot(slot)) != 0) { - cache_log("[cgit] error printing cache %s: %s (%d)\n", - slot->cache_name, - strerror(err), - err); - } + err = serve_slot(slot); close_slot(slot); return err; } -/* Print cached content to stdout, generate the content if necessary. */ +// The result lives in a static buffer and is only good until the next call. +static char *format_time(const char *format, time_t when) +{ + static char buf[64]; + struct tm tm; + + if (!when) + return NULL; + gmtime_r(&when, &tm); + strftime(buf, sizeof(buf) - 1, format, &tm); + return buf; +} + +/* + * The accumulator is an unsigned long rather than a fixed 32 bit type, so on a + * 64 bit host this is not the published FNV-1 value. All that decides is which + * slot a key lands in, and nothing outside a single build has to agree on the + * answer. + */ +unsigned long cache_hash_str(const char *str) +{ + unsigned long h = FNV_OFFSET; + unsigned char *s = (unsigned char *)str; + + if (!s) + return h; + + while (*s) { + h *= FNV_PRIME; + h ^= *s++; + } + return h; +} + int cache_process(int size, const char *path, const char *key, int ttl, cache_fill_fn fn) { unsigned long hash; int i; - struct strbuf filename = STRBUF_INIT; - struct strbuf lockname = STRBUF_INIT; + struct strbuf slot_path = STRBUF_INIT; + struct strbuf lock_path = STRBUF_INIT; struct cache_slot slot; int result; - /* If the cache is disabled, just generate the content */ if (size <= 0 || ttl == 0) { fn(); return 0; } - /* Verify input, calculate filenames */ if (!path) { - cache_log("[cgit] Cache path not specified, caching is disabled\n"); + log_error("[cgit] Cache path not specified, caching is disabled\n"); fn(); return 0; } if (!key) key = ""; if (!key_fits_slot(key)) { - cache_log("[cgit] Cache key too long for a slot, caching is " + log_error("[cgit] Cache key too long for a slot, caching is " "disabled for this request\n"); fn(); return 0; } hash = cache_hash_str(key) % size; - strbuf_addstr(&filename, path); - strbuf_ensure_end(&filename, '/'); - for (i = 0; i < 8; i++) { - strbuf_addf(&filename, "%x", (unsigned char)(hash & 0xf)); + strbuf_addstr(&slot_path, path); + strbuf_ensure_end(&slot_path, '/'); + for (i = 0; i < SLOT_NAME_LEN; i++) { + strbuf_addf(&slot_path, "%x", (unsigned char)(hash & 0xf)); hash >>= 4; } - strbuf_addbuf(&lockname, &filename); - strbuf_addstr(&lockname, ".lock"); + strbuf_addbuf(&lock_path, &slot_path); + strbuf_addstr(&lock_path, ".lock"); slot.fn = fn; slot.ttl = ttl; - slot.stdout_fd = -1; - slot.cache_name = filename.buf; - slot.lock_name = lockname.buf; + slot.saved_stdout = -1; + slot.path = slot_path.buf; + slot.lock_path = lock_path.buf; slot.key = key; slot.keylen = strlen(key); result = process_slot(&slot); - strbuf_release(&filename); - strbuf_release(&lockname); + strbuf_release(&slot_path); + strbuf_release(&lock_path); return result; } -/* Return a strftime formatted date/time - * NB: the result from this function is to shared memory - */ -static char *sprintftime(const char *format, time_t time) -{ - static char buf[64]; - struct tm tm; - - if (!time) - return NULL; - gmtime_r(&time, &tm); - strftime(buf, sizeof(buf)-1, format, &tm); - return buf; -} - int cache_ls(const char *path) { DIR *dir; struct dirent *ent; int err = 0; + // A NULL key leaves open_slot with nothing to compare against, so + // every slot it opens is simply read. struct cache_slot slot = { NULL }; - struct strbuf fullname = STRBUF_INIT; + struct strbuf slot_path = STRBUF_INIT; size_t prefixlen; char *nul; int keylen; if (!path) { - cache_log("[cgit] cache path not specified\n"); + log_error("[cgit] cache path not specified\n"); return -1; } dir = opendir(path); if (!dir) { err = errno; - cache_log("[cgit] unable to open path %s: %s (%d)\n", + log_error("[cgit] unable to open path %s: %s (%d)\n", path, strerror(err), err); return err; } - strbuf_addstr(&fullname, path); - strbuf_ensure_end(&fullname, '/'); - prefixlen = fullname.len; + strbuf_addstr(&slot_path, path); + strbuf_ensure_end(&slot_path, '/'); + prefixlen = slot_path.len; while ((ent = readdir(dir)) != NULL) { - if (strlen(ent->d_name) != 8) + if (strlen(ent->d_name) != SLOT_NAME_LEN) continue; - strbuf_setlen(&fullname, prefixlen); - strbuf_addstr(&fullname, ent->d_name); - slot.cache_name = fullname.buf; + strbuf_setlen(&slot_path, prefixlen); + strbuf_addstr(&slot_path, ent->d_name); + slot.path = slot_path.buf; if ((err = open_slot(&slot)) != 0) { - cache_log("[cgit] unable to open path %s: %s (%d)\n", - fullname.buf, strerror(err), err); + log_error("[cgit] unable to open path %s: %s (%d)\n", + slot_path.buf, strerror(err), err); continue; } - // The stored key is NUL-terminated within the buffer, but a - // truncated or corrupt slot may not be. Bound the print to - // what was read so %s cannot run off the end. - nul = memchr(slot.buf, 0, slot.bufsize); - keylen = nul ? (int)(nul - slot.buf) : slot.bufsize; + // A truncated or corrupt slot may hold no NUL, so the print is + // bounded by what was read and cannot run off the end. + nul = memchr(slot.buf, 0, slot.buflen); + keylen = nul ? (int)(nul - slot.buf) : slot.buflen; htmlf("%s %s %10"PRIuMAX" %.*s\n", - fullname.buf, - sprintftime("%Y-%m-%d %H:%M:%S", - slot.cache_st.st_mtime), - (uintmax_t)slot.cache_st.st_size, + slot_path.buf, + format_time("%Y-%m-%d %H:%M:%S", + slot.st.st_mtime), + (uintmax_t)slot.st.st_size, keylen, slot.buf); close_slot(&slot); } closedir(dir); - strbuf_release(&fullname); + strbuf_release(&slot_path); return 0; } - -/* Print a message to stdout */ -void cache_log(const char *format, ...) -{ - va_list args; - va_start(args, format); - vfprintf(stderr, format, args); - va_end(args); -} - diff --git a/source/cache.h b/source/cache.h index f8018be..db77999 100644 --- a/source/cache.h +++ b/source/cache.h @@ -1,6 +1,7 @@ /* - * Since git has it's own cache.h which we include, - * lets test on CGIT_CACHE_H to avoid confusion + * The front door to cgit's page cache. A caller hands over a key identifying + * the request and a callback that renders the page, and gets back either the + * copy already on disk or a freshly rendered one. */ #ifndef CGIT_CACHE_H @@ -8,30 +9,30 @@ typedef void (*cache_fill_fn)(void); - -/* Print cached content to stdout, generate the content if necessary. - * - * Parameters - * size max number of cache files - * path directory used to store cache files - * key the key used to lookup cache files - * ttl max cache time in seconds for this key - * fn content generator function for this key - * - * Return value - * 0 indicates success, everything else is an error +/* + * Write the page for key to stdout, taking it from the cache when a fresh + * slot for that key is there and rendering it through fn when it is not. size + * is how many slots the cache may use and path is the directory holding them. + * ttl is how many minutes a slot for this key stays fresh, where a negative + * ttl never expires and a ttl of zero skips the cache for this request. + * Returns 0 when the page was written, and an errno value when it was not. */ extern int cache_process(int size, const char *path, const char *key, int ttl, cache_fill_fn fn); - -/* List info about all cache entries on stdout */ +// Write one line per cache slot to stdout, giving its path, modification +// time, size and key. extern int cache_ls(const char *path); -/* Print a message to stdout */ -__attribute__((format (printf,1,2))) -extern void cache_log(const char *format, ...); +/* + * Give up on the slot being filled, discarding what has been rendered into it + * and putting stdout back on the client. Anything that ends a request part way + * through rendering has to call this before it writes what the visitor should + * see, because until then stdout is the cache file and the visitor is on + * course to receive nothing at all. Does nothing when no slot is being filled. + */ +extern void cache_abandon_fill(void); extern unsigned long cache_hash_str(const char *str); -#endif /* CGIT_CACHE_H */ +#endif // CGIT_CACHE_H diff --git a/source/cgit.c b/source/cgit.c index 7cce3d3..5c11a93 100644 --- a/source/cgit.c +++ b/source/cgit.c @@ -1,43 +1,74 @@ -/* cgit.c: cgi for the git scm - * - * Copyright (C) 2006-2014 cgit Development Team - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The entry point every request passes through. It reads cgitrc and the query + * string into the global ctx, works out how long the answer may be cached, + * and then hands over to the cache, which either serves a copy it already has + * or calls back here to render one. Rendering means authenticating the + * request, resolving the repository and the branch it names, and dispatching + * to the page handler cmd.c maps the page name to. Run from a shell rather + * than from a web server the same binary instead prints what it was built + * with, or scans a directory tree and writes out the repositories it found in + * cgitrc form. */ #define USE_THE_REPOSITORY_VARIABLE -#include "cgit.h" #include "cache.h" +#include "cgit.h" #include "cmd.h" -#include "configfile.h" +#include "config.h" +#include "filter.h" #include "html.h" -#include "ui-shared.h" -#include "ui-stats.h" +#include "parsing.h" +#include "scan-tree.h" +#include "shared.h" #include "ui-blob.h" +#include "ui-diff.h" #include "ui-empty.h" -#include "ui-summary.h" -#include "scan-tree.h" +#include "ui-shared.h" +#include "ui-snapshot.h" +#include "ui-stats.h" -/* We intentionally keep this rather small, instead of looping and - * feeding it to the filter a couple bytes at a time. This way, the - * filter itself does not need to handle any denial of service or - * buffer bloat issues. If this winds up being too small, people - * will complain on the mailing list, and we'll increase it as needed. */ +// Deliberately small. The whole body is read at once and handed to the auth +// filter, so a filter never has to defend itself against a huge or a +// dribbled-out POST. #define MAX_AUTHENTICATION_POST_BYTES 4096 + +// The filter bar matches the search against every item a page lists, so this +// bounds both the work one request can ask for and the cache key the query +// lands in. +#define MAX_SEARCH_LEN 512 + +// Furthest into a listing a request may ask to start. A walk has to step over +// every row it skips, so this bounds the work an offset alone can buy. +#define MAX_QUERY_OFFSET 100000 + +// cgit knows fewer than eight archive formats, so this stands in for every bit +// a snapshots mask can carry. +#define ALL_SNAPSHOT_FORMATS 0xFF + +// An Expires header has no way of saying never, so a page whose ttl is +// negative claims ten years. +#define NEVER_EXPIRES_SECONDS (10 * 365 * 24 * 60 * 60) + +/* + * The first branch found is the fallback, so a repository whose default branch + * does not exist still has something to show. + */ +struct refmatch { + char *wanted; + char *first; + int found; +}; + const char *cgit_version = CGIT_VERSION; /* - * Isolate git from the calling user's configuration. Ignore the system and - * global config and attributes, so a snapshot cannot be broken by something - * like a core.excludesfile pointing at a "~" path that git can no longer - * expand once HOME is unset below. - * - * Called at the top of cmd_main rather than from a constructor attribute. - * git-compat-util.h defines __attribute__ away on a compiler that does not - * support it, which would leave this silently never running. Nothing git does - * before cmd_main reads configuration, so an ordinary call is equivalent. + * Isolate git from the calling user's configuration, so a snapshot cannot be + * broken by something like a core.excludesfile pointing at a "~" path that git + * can no longer expand once HOME is unset below. Called from cmd_main rather + * than from a constructor attribute, because git-compat-util.h defines + * __attribute__ away on a compiler that does not support it, which would leave + * this silently never running. */ static void isolate_git_environment(void) { @@ -48,160 +79,444 @@ static void isolate_git_environment(void) unsetenv("XDG_CONFIG_HOME"); } -static void add_mimetype(const char *name, const char *value) +/* + * Turn a die from anywhere inside git into a rendered page, since a CGI that + * produced no output leaves the visitor with whatever the web server makes of + * it. + */ +static NORETURN void die_routine(const char *msg, va_list params) { - struct string_list_item *item; + // A page is rendered with stdout pointed at the cache file, so the + // message would otherwise be written there instead of to the visitor. + cache_abandon_fill(); + cgit_vprint_error_page(400, "Bad request", msg, params); + exit(0); +} - item = string_list_insert(&ctx.cfg.mimetypes, name); - item->util = xstrdup(value); +static void prepare_context(void) +{ + memset(&ctx, 0, sizeof(ctx)); + ctx.cfg.agefile = "info/web/last-modified"; + ctx.cfg.cache_size = 0; + ctx.cfg.cache_root = CGIT_CACHE_ROOT; + ctx.cfg.cache_about_ttl = 15; + ctx.cfg.cache_snapshot_ttl = 5; + ctx.cfg.cache_repo_ttl = 5; + ctx.cfg.cache_root_ttl = 5; + ctx.cfg.cache_scanrc_ttl = 15; + ctx.cfg.cache_dynamic_ttl = 5; + ctx.cfg.cache_static_ttl = -1; + ctx.cfg.case_sensitive_sort = 1; + ctx.cfg.branch_sort = 0; + ctx.cfg.commit_sort = 0; + ctx.cfg.logo = "/cgit.png"; + ctx.cfg.favicon = "/favicon.ico"; + ctx.cfg.local_time = 0; + ctx.cfg.date_mode = date_mode_from_type(DATE_SHORT); + ctx.cfg.enable_relative_dates = 1; + ctx.cfg.enable_http_clone = 1; + ctx.cfg.enable_index_owner = 1; + ctx.cfg.enable_tree_linenumbers = 1; + ctx.cfg.enable_git_config = 0; + ctx.cfg.max_repo_count = 50; + ctx.cfg.max_commit_count = 50; + ctx.cfg.max_patch_count = 50; + ctx.cfg.max_diff_files = 200; + ctx.cfg.max_diff_lines = 1000; + ctx.cfg.max_msg_len = 80; + ctx.cfg.max_ref_count = 200; + ctx.cfg.max_repodesc_len = 80; + // Counted in kilobytes, so ten megabytes, which bounds the memory one + // request can be made to allocate for a blob. + ctx.cfg.max_blob_size = 10 * 1024; + ctx.cfg.max_stats = 0; + ctx.cfg.project_list = NULL; + ctx.cfg.renamelimit = -1; + ctx.cfg.remove_suffix = 0; + ctx.cfg.robots = "index, nofollow"; + ctx.cfg.root_title = "Git repository browser"; + ctx.cfg.root_desc = "a fast webinterface for the git dscm"; + ctx.cfg.scan_hidden_path = 0; + ctx.cfg.script_name = CGIT_SCRIPT_NAME; + ctx.cfg.section = ""; + ctx.cfg.repository_sort = "name"; + ctx.cfg.section_sort = 0; + ctx.cfg.summary_branches = 10; + ctx.cfg.summary_log = 10; + ctx.cfg.summary_tags = 10; + ctx.cfg.max_atom_items = 10; + ctx.cfg.difftype = DIFF_UNIFIED; + ctx.env.cgit_config = getenv("CGIT_CONFIG"); + ctx.env.http_host = getenv("HTTP_HOST"); + ctx.env.https = getenv("HTTPS"); + ctx.env.no_http = getenv("NO_HTTP"); + ctx.env.path_info = getenv("PATH_INFO"); + ctx.env.query_string = getenv("QUERY_STRING"); + ctx.env.request_method = getenv("REQUEST_METHOD"); + ctx.env.script_name = getenv("SCRIPT_NAME"); + ctx.env.server_name = getenv("SERVER_NAME"); + ctx.env.server_port = getenv("SERVER_PORT"); + ctx.env.http_cookie = getenv("HTTP_COOKIE"); + ctx.env.http_referer = getenv("HTTP_REFERER"); + ctx.env.content_length = getenv("CONTENT_LENGTH") ? + strtoul(getenv("CONTENT_LENGTH"), NULL, 10) : 0; + ctx.env.authenticated = 0; + ctx.page.mimetype = "text/html"; + ctx.page.charset = PAGE_ENCODING; + ctx.page.filename = NULL; + ctx.page.size = 0; + ctx.page.modified = time(NULL); + ctx.page.expires = ctx.page.modified; + ctx.page.etag = NULL; + string_list_init_dup(&ctx.cfg.mimetypes); + if (ctx.env.script_name) + ctx.cfg.script_name = xstrdup(ctx.env.script_name); + if (ctx.env.query_string) + ctx.qry.raw = xstrdup(ctx.env.query_string); + if (!ctx.env.cgit_config) + ctx.env.cgit_config = CGIT_CONFIG; } -static void process_cached_repolist(const char *path); +static void print_version(void) +{ + printf("CGit %s | https://github.com/brycekwon/cgit\n\nCompiled in features:\n", CGIT_VERSION); +#ifdef NO_LUA + printf("[-] "); +#else + printf("[+] "); +#endif + printf("Lua scripting\n"); +#ifndef HAVE_LINUX_SENDFILE + printf("[-] "); +#else + printf("[+] "); +#endif + printf("Linux sendfile() usage\n"); +} -void cgit_repo_config(struct cgit_repo *repo, const char *name, const char *value) +static int cmp_repos(const void *a, const void *b) { - const char *path; - struct string_list_item *item; + const struct cgit_repo *repo_a = a, *repo_b = b; + return strcmp(repo_a->url, repo_b->url); +} - if (!strcmp(name, "name")) - repo->name = cgit_strdup_first_line(value); - else if (!strcmp(name, "clone-url")) - repo->clone_url = cgit_strdup_first_line(value); - else if (!strcmp(name, "desc")) - repo->desc = cgit_strdup_first_line(value); - else if (!strcmp(name, "owner")) - repo->owner = cgit_strdup_first_line(value); - else if (!strcmp(name, "homepage")) - repo->homepage = cgit_strdup_first_line(value); - else if (!strcmp(name, "defbranch")) - repo->defbranch = cgit_strdup_first_line(value); - else if (!strcmp(name, "extra-head-content")) - repo->extra_head_content = cgit_strdup_first_line(value); - else if (!strcmp(name, "snapshots")) - repo->snapshots = ctx.cfg.snapshots & cgit_parse_snapshots_mask(value); - else if (!strcmp(name, "enable-blame")) - repo->enable_blame = atoi(value); - else if (!strcmp(name, "enable-commit-graph")) - repo->enable_commit_graph = atoi(value); - else if (!strcmp(name, "enable-follow-links")) - repo->enable_follow_links = atoi(value); - else if (!strcmp(name, "enable-log-filecount")) - repo->enable_log_filecount = atoi(value); - else if (!strcmp(name, "enable-log-linecount")) - repo->enable_log_linecount = atoi(value); - else if (!strcmp(name, "enable-remote-branches")) - repo->enable_remote_branches = atoi(value); - else if (!strcmp(name, "enable-subject-links")) - repo->enable_subject_links = atoi(value); - else if (!strcmp(name, "enable-html-serving")) - repo->enable_html_serving = atoi(value); - else if (!strcmp(name, "enable-stats")) - repo->enable_stats = atoi(value); - else if (!strcmp(name, "branch-sort")) { - if (!strcmp(value, "age")) - repo->branch_sort = 1; - if (!strcmp(value, "name")) - repo->branch_sort = 0; - } else if (!strcmp(name, "commit-sort")) { - if (!strcmp(value, "date")) - repo->commit_sort = 1; - if (!strcmp(value, "topo")) - repo->commit_sort = 2; - } else if (!strcmp(name, "max-stats")) - repo->max_stats = cgit_find_stats_period(value, NULL); - else if (!strcmp(name, "module-link")) - repo->module_link= cgit_strdup_first_line(value); - else if (skip_prefix(name, "module-link.", &path)) { - item = string_list_append(&repo->submodules, cgit_strdup_first_line(path)); - item->util = cgit_strdup_first_line(value); - } else if (!strcmp(name, "section")) - repo->section = cgit_strdup_first_line(value); - else if (!strcmp(name, "snapshot-prefix")) - repo->snapshot_prefix = cgit_strdup_first_line(value); - else if (!strcmp(name, "readme") && value != NULL) { - if (repo->readme.items == ctx.cfg.readme.items) - memset(&repo->readme, 0, sizeof(repo->readme)); - string_list_append(&repo->readme, cgit_strdup_first_line(value)); - } else if (!strcmp(name, "logo") && value != NULL) - repo->logo = cgit_strdup_first_line(value); - else if (!strcmp(name, "logo-link") && value != NULL) - repo->logo_link = cgit_strdup_first_line(value); - else if (!strcmp(name, "hide")) - repo->hide = atoi(value); - else if (!strcmp(name, "ignore")) - repo->ignore = atoi(value); - else if (ctx.cfg.enable_filter_overrides) { - if (!strcmp(name, "about-filter")) - repo->about_filter = cgit_new_filter(value, ABOUT); - else if (!strcmp(name, "commit-filter")) - repo->commit_filter = cgit_new_filter(value, COMMIT); - else if (!strcmp(name, "source-filter")) - repo->source_filter = cgit_new_filter(value, SOURCE); - else if (!strcmp(name, "email-filter")) - repo->email_filter = cgit_new_filter(value, EMAIL); +static char *build_snapshot_setting(int mask) +{ + const struct cgit_snapshot_format *format; + struct strbuf result = STRBUF_INIT; + + for (format = cgit_snapshot_formats; format->suffix; format++) { + if (cgit_snapshot_format_bit(format) & mask) { + if (result.len) + strbuf_addch(&result, ' '); + strbuf_addstr(&result, format->suffix); + } } + return strbuf_detach(&result, NULL); } -static void config_cb(const char *name, const char *value) +static void print_repo(FILE *f, struct cgit_repo *repo) { - const char *arg; + struct string_list_item *item; - if (!strcmp(name, "section")) - ctx.cfg.section = cgit_strdup_first_line(value); - else if (!strcmp(name, "repo.url")) - ctx.repo = cgit_add_repo(value); - else if (ctx.repo && !strcmp(name, "repo.path")) - ctx.repo->path = cgit_trim_end(value, '/'); - else if (ctx.repo && skip_prefix(name, "repo.", &arg)) - cgit_repo_config(ctx.repo, arg, value); - else if (!strcmp(name, "readme")) - string_list_append(&ctx.cfg.readme, cgit_strdup_first_line(value)); - else if (!strcmp(name, "root-title")) - ctx.cfg.root_title = cgit_strdup_first_line(value); - else if (!strcmp(name, "root-desc")) - ctx.cfg.root_desc = cgit_strdup_first_line(value); - else if (!strcmp(name, "root-readme")) - ctx.cfg.root_readme = cgit_strdup_first_line(value); - else if (!strcmp(name, "css")) - string_list_append(&ctx.cfg.css, cgit_strdup_first_line(value)); - else if (!strcmp(name, "js")) - string_list_append(&ctx.cfg.js, cgit_strdup_first_line(value)); - else if (!strcmp(name, "favicon")) - ctx.cfg.favicon = cgit_strdup_first_line(value); - else if (!strcmp(name, "footer")) - ctx.cfg.footer = cgit_strdup_first_line(value); - else if (!strcmp(name, "head-include")) - ctx.cfg.head_include = cgit_strdup_first_line(value); - else if (!strcmp(name, "header")) - ctx.cfg.header = cgit_strdup_first_line(value); - else if (!strcmp(name, "logo")) - ctx.cfg.logo = cgit_strdup_first_line(value); - else if (!strcmp(name, "logo-link")) - ctx.cfg.logo_link = cgit_strdup_first_line(value); - else if (!strcmp(name, "module-link")) - ctx.cfg.module_link = cgit_strdup_first_line(value); - else if (!strcmp(name, "strict-export")) - ctx.cfg.strict_export = cgit_strdup_first_line(value); - else if (!strcmp(name, "virtual-root")) - ctx.cfg.virtual_root = cgit_ensure_end(value, '/'); - else if (!strcmp(name, "noplainemail")) - ctx.cfg.noplainemail = atoi(value); - else if (!strcmp(name, "noheader")) - ctx.cfg.noheader = atoi(value); - else if (!strcmp(name, "snapshots")) - ctx.cfg.snapshots = cgit_parse_snapshots_mask(value); - else if (!strcmp(name, "enable-filter-overrides")) - ctx.cfg.enable_filter_overrides = atoi(value); - else if (!strcmp(name, "enable-follow-links")) - ctx.cfg.enable_follow_links = atoi(value); - else if (!strcmp(name, "enable-stats")) - ctx.cfg.enable_stats = atoi(value); - else if (!strcmp(name, "enable-http-clone")) - ctx.cfg.enable_http_clone = atoi(value); - else if (!strcmp(name, "enable-index-links")) - ctx.cfg.enable_index_links = atoi(value); - else if (!strcmp(name, "enable-index-owner")) - ctx.cfg.enable_index_owner = atoi(value); + fprintf(f, "repo.url=%s\n", repo->url); + fprintf(f, "repo.name=%s\n", repo->name); + fprintf(f, "repo.path=%s\n", repo->path); + if (repo->owner) + fprintf(f, "repo.owner=%s\n", repo->owner); + if (repo->desc) + fprintf(f, "repo.desc=%s\n", repo->desc); + for_each_string_list_item(item, &repo->readme) { + if (item->util) + fprintf(f, "repo.readme=%s:%s\n", (char *)item->util, item->string); + else + fprintf(f, "repo.readme=%s\n", item->string); + } + if (repo->defbranch) + fprintf(f, "repo.defbranch=%s\n", repo->defbranch); + if (repo->extra_head_content) + fprintf(f, "repo.extra-head-content=%s\n", repo->extra_head_content); + if (repo->module_link) + fprintf(f, "repo.module-link=%s\n", repo->module_link); + if (repo->section) + fprintf(f, "repo.section=%s\n", repo->section); + if (repo->homepage) + fprintf(f, "repo.homepage=%s\n", repo->homepage); + if (repo->clone_url) + fprintf(f, "repo.clone-url=%s\n", repo->clone_url); + fprintf(f, "repo.enable-blame=%d\n", repo->enable_blame); + fprintf(f, "repo.enable-commit-graph=%d\n", repo->enable_commit_graph); + fprintf(f, "repo.enable-follow-links=%d\n", repo->enable_follow_links); + fprintf(f, "repo.enable-log-filecount=%d\n", repo->enable_log_filecount); + fprintf(f, "repo.enable-log-linecount=%d\n", repo->enable_log_linecount); + if (repo->about_filter && repo->about_filter != ctx.cfg.about_filter) + cgit_fprintf_filter(repo->about_filter, f, "repo.about-filter="); + if (repo->commit_filter && repo->commit_filter != ctx.cfg.commit_filter) + cgit_fprintf_filter(repo->commit_filter, f, "repo.commit-filter="); + if (repo->source_filter && repo->source_filter != ctx.cfg.source_filter) + cgit_fprintf_filter(repo->source_filter, f, "repo.source-filter="); + if (repo->email_filter && repo->email_filter != ctx.cfg.email_filter) + cgit_fprintf_filter(repo->email_filter, f, "repo.email-filter="); + if (repo->snapshots != ctx.cfg.snapshots) { + char *formats = build_snapshot_setting(repo->snapshots); + fprintf(f, "repo.snapshots=%s\n", formats ? formats : ""); + free(formats); + } + if (repo->snapshot_prefix) + fprintf(f, "repo.snapshot-prefix=%s\n", repo->snapshot_prefix); + if (repo->enable_stats != ctx.cfg.enable_stats) + fprintf(f, "repo.enable-stats=%d\n", repo->enable_stats); + if (repo->max_stats != ctx.cfg.max_stats) + fprintf(f, "repo.max-stats=%s\n", + cgit_find_stats_periodname(repo->max_stats)); + if (repo->logo) + fprintf(f, "repo.logo=%s\n", repo->logo); + if (repo->logo_link) + fprintf(f, "repo.logo-link=%s\n", repo->logo_link); + fprintf(f, "repo.enable-remote-branches=%d\n", repo->enable_remote_branches); + fprintf(f, "repo.enable-subject-links=%d\n", repo->enable_subject_links); + fprintf(f, "repo.enable-html-serving=%d\n", repo->enable_html_serving); + if (repo->branch_sort == 1) + fprintf(f, "repo.branch-sort=age\n"); + if (repo->commit_sort) { + if (repo->commit_sort == 1) + fprintf(f, "repo.commit-sort=date\n"); + else if (repo->commit_sort == 2) + fprintf(f, "repo.commit-sort=topo\n"); + } + fprintf(f, "repo.hide=%d\n", repo->hide); + fprintf(f, "repo.ignore=%d\n", repo->ignore); + fprintf(f, "\n"); +} + +static void print_repolist(FILE *f, struct cgit_repolist *list, int start) +{ + int i; + + for (i = start; i < list->count; i++) + print_repo(f, &list->repos[i]); +} + +static void parse_args(int argc, const char **argv) +{ + int i; + const char *arg; + int scanned = 0; + + for (i = 1; i < argc; i++) { + if (!strcmp(argv[i], "--version")) { + print_version(); + exit(0); + } + if (skip_prefix(argv[i], "--cache=", &arg)) { + ctx.cfg.cache_root = xstrdup(arg); + } else if (!strcmp(argv[i], "--nohttp")) { + ctx.env.no_http = "1"; + } else if (skip_prefix(argv[i], "--query=", &arg)) { + ctx.qry.raw = xstrdup(arg); + } else if (skip_prefix(argv[i], "--repo=", &arg)) { + ctx.qry.repo = xstrdup(arg); + } else if (skip_prefix(argv[i], "--page=", &arg)) { + ctx.qry.page = xstrdup(arg); + } else if (skip_prefix(argv[i], "--head=", &arg)) { + ctx.qry.head = xstrdup(arg); + ctx.qry.has_symref = 1; + } else if (skip_prefix(argv[i], "--oid=", &arg)) { + ctx.qry.oid = xstrdup(arg); + ctx.qry.has_oid = 1; + } else if (skip_prefix(argv[i], "--ofs=", &arg)) { + ctx.qry.ofs = atoi(arg); + } else if (skip_prefix(argv[i], "--scan-tree=", &arg) || + skip_prefix(argv[i], "--scan-path=", &arg)) { + // A repository's own snapshots setting is masked with + // the global one, which normally comes from cgitrc. + // That has not been read yet here, so an empty mask + // would discard whatever the repository asked for. + ctx.cfg.snapshots = ALL_SNAPSHOT_FORMATS; + scanned++; + scan_tree(arg); + } + } + if (scanned) { + qsort(cgit_repolist.repos, cgit_repolist.count, + sizeof(struct cgit_repo), cmp_repos); + print_repolist(stdout, &cgit_repolist, 0); + exit(0); + } +} + +static int generate_cached_repolist(const char *path, const char *cached_rc) +{ + struct strbuf locked_rc = STRBUF_INIT; + int err = 0; + int first; + FILE *f; + + strbuf_addf(&locked_rc, "%s.lock", cached_rc); + f = fopen(locked_rc.buf, "wx"); + if (!f) { + // An existing lock file only means concurrent requests, which + // is not worth a line in the server log. + err = errno; + if (err != EEXIST) + fprintf(stderr, "[cgit] Error opening %s: %s (%d)\n", + locked_rc.buf, strerror(err), err); + goto out; + } + first = cgit_repolist.count; + if (ctx.cfg.project_list) + scan_projects(path, ctx.cfg.project_list); + else + scan_tree(path); + print_repolist(f, &cgit_repolist, first); + // Closed before the rename, because print_repolist writes through stdio + // and a rename over the live file would otherwise publish a repolist + // that stops wherever the buffer happened to end. + if (fclose(f)) { + err = errno; + fprintf(stderr, "[cgit] Error writing %s: %s (%d)\n", + locked_rc.buf, strerror(err), err); + unlink(locked_rc.buf); + goto out; + } + if (rename(locked_rc.buf, cached_rc)) { + err = errno; + fprintf(stderr, "[cgit] Error renaming %s to %s: %s (%d)\n", + locked_rc.buf, cached_rc, strerror(err), err); + unlink(locked_rc.buf); + } +out: + strbuf_release(&locked_rc); + return err; +} + +// A cached repolist is itself a config file, so these two call each other. +static void apply_config(const char *name, const char *value); + +static void process_cached_repolist(const char *path) +{ + struct stat st; + struct strbuf cached_rc = STRBUF_INIT; + time_t age; + unsigned long hash; + int devnull; + + hash = cache_hash_str(path); + if (ctx.cfg.project_list) + hash += cache_hash_str(ctx.cfg.project_list); + strbuf_addf(&cached_rc, "%s/rc-%8lx", ctx.cfg.cache_root, hash); + + if (stat(cached_rc.buf, &st)) { + // Nothing is cached yet, so this request scans in its own + // process, leaving no copy behind when it cannot take the lock. + if (generate_cached_repolist(path, cached_rc.buf)) { + if (ctx.cfg.project_list) + scan_projects(path, ctx.cfg.project_list); + else + scan_tree(path); + } + goto out; + } + + config_file_parse(cached_rc.buf, apply_config); + + age = time(NULL) - st.st_mtime; + if (age <= (ctx.cfg.cache_scanrc_ttl * 60)) + goto out; + + // The list just parsed is stale but usable, so a child rebuilds it + // while this request answers from what it already has. + if (fork()) + goto out; + + // The child inherits the descriptors of the request, and the web server + // reads stdout until every holder of it is gone, so leaving them in + // place would keep the visitor waiting for the whole scan after their + // page was written. Anything the scan prints would land on that + // response as well. + devnull = open("/dev/null", O_RDWR); + if (devnull >= 0) { + dup2(devnull, STDIN_FILENO); + dup2(devnull, STDOUT_FILENO); + dup2(devnull, STDERR_FILENO); + if (devnull > STDERR_FILENO) + close(devnull); + } + // _exit rather than exit, so the handlers the request registered do + // not run a second time in the child. + _exit(generate_cached_repolist(path, cached_rc.buf)); +out: + strbuf_release(&cached_rc); +} + +static void add_mimetype(const char *name, const char *value) +{ + struct string_list_item *item; + + item = string_list_insert(&ctx.cfg.mimetypes, name); + item->util = xstrdup(value); +} + +static void apply_config(const char *name, const char *value) +{ + const char *arg; + + if (!strcmp(name, "section")) + ctx.cfg.section = cgit_strdup_first_line(value); + else if (!strcmp(name, "repo.url")) + ctx.repo = cgit_add_repo(value); + else if (ctx.repo && !strcmp(name, "repo.path")) + ctx.repo->path = cgit_trim_end(value, '/'); + else if (ctx.repo && skip_prefix(name, "repo.", &arg)) + cgit_repo_config(ctx.repo, arg, value); + else if (!strcmp(name, "readme")) + string_list_append(&ctx.cfg.readme, cgit_strdup_first_line(value)); + else if (!strcmp(name, "root-title")) + ctx.cfg.root_title = cgit_strdup_first_line(value); + else if (!strcmp(name, "root-desc")) + ctx.cfg.root_desc = cgit_strdup_first_line(value); + else if (!strcmp(name, "root-readme")) + ctx.cfg.root_readme = cgit_strdup_first_line(value); + else if (!strcmp(name, "css")) + string_list_append(&ctx.cfg.css, cgit_strdup_first_line(value)); + else if (!strcmp(name, "js")) + string_list_append(&ctx.cfg.js, cgit_strdup_first_line(value)); + else if (!strcmp(name, "favicon")) + ctx.cfg.favicon = cgit_strdup_first_line(value); + else if (!strcmp(name, "footer")) + ctx.cfg.footer = cgit_strdup_first_line(value); + else if (!strcmp(name, "head-include")) + ctx.cfg.head_include = cgit_strdup_first_line(value); + else if (!strcmp(name, "header")) + ctx.cfg.header = cgit_strdup_first_line(value); + else if (!strcmp(name, "logo")) + ctx.cfg.logo = cgit_strdup_first_line(value); + else if (!strcmp(name, "logo-link")) + ctx.cfg.logo_link = cgit_strdup_first_line(value); + else if (!strcmp(name, "module-link")) + ctx.cfg.module_link = cgit_strdup_first_line(value); + else if (!strcmp(name, "strict-export")) + ctx.cfg.strict_export = cgit_strdup_first_line(value); + else if (!strcmp(name, "virtual-root")) + ctx.cfg.virtual_root = cgit_ensure_end(value, '/'); + else if (!strcmp(name, "noplainemail")) + ctx.cfg.noplainemail = atoi(value); + else if (!strcmp(name, "noheader")) + ctx.cfg.noheader = atoi(value); + else if (!strcmp(name, "snapshots")) + ctx.cfg.snapshots = cgit_parse_snapshots_mask(value); + else if (!strcmp(name, "enable-filter-overrides")) + ctx.cfg.enable_filter_overrides = atoi(value); + else if (!strcmp(name, "enable-follow-links")) + ctx.cfg.enable_follow_links = atoi(value); + else if (!strcmp(name, "enable-stats")) + ctx.cfg.enable_stats = atoi(value); + else if (!strcmp(name, "enable-http-clone")) + ctx.cfg.enable_http_clone = atoi(value); + else if (!strcmp(name, "enable-index-links")) + ctx.cfg.enable_index_links = atoi(value); + else if (!strcmp(name, "enable-index-owner")) + ctx.cfg.enable_index_owner = atoi(value); else if (!strcmp(name, "enable-blame")) ctx.cfg.enable_blame = atoi(value); else if (!strcmp(name, "enable-commit-graph")) @@ -282,7 +597,7 @@ static void config_cb(const char *name, const char *value) ctx.cfg.max_patch_count = atoi(value); else if (!strcmp(name, "project-list")) ctx.cfg.project_list = cgit_strdup_first_line(cgit_expand_macros(value)); - else if (!strcmp(name, "scan-path")) + else if (!strcmp(name, "scan-path")) { if (ctx.cfg.cache_size) process_cached_repolist(cgit_expand_macros(value)); else if (ctx.cfg.project_list) @@ -290,7 +605,7 @@ static void config_cb(const char *name, const char *value) ctx.cfg.project_list); else scan_tree(cgit_expand_macros(value)); - else if (!strcmp(name, "scan-hidden-path")) + } else if (!strcmp(name, "scan-hidden-path")) ctx.cfg.scan_hidden_path = atoi(value); else if (!strcmp(name, "section-from-path")) ctx.cfg.section_from_path = atoi(value); @@ -339,360 +654,101 @@ static void config_cb(const char *name, const char *value) } else if (skip_prefix(name, "mimetype.", &arg)) add_mimetype(arg, value); else if (!strcmp(name, "include")) - parse_configfile(cgit_expand_macros(value), config_cb); + config_file_parse(cgit_expand_macros(value), apply_config); } -static void querystring_cb(const char *name, const char *value) -{ - if (!value) - value = ""; - - if (!strcmp(name,"r")) { - ctx.qry.repo = xstrdup(value); - ctx.repo = cgit_get_repoinfo(value); - } else if (!strcmp(name, "p")) { - ctx.qry.page = xstrdup(value); - } else if (!strcmp(name, "url")) { - if (*value == '/') - value++; - ctx.qry.url = xstrdup(value); - cgit_parse_url(value); - } else if (!strcmp(name, "qt")) { - ctx.qry.grep = xstrdup(value); - } else if (!strcmp(name, "q")) { - /* A query is matched against every repository, ref or commit - * the page lists, so bound what one request can ask to be - * compared. Nothing legible reaches this length, and the value - * also lands in the cache key. */ - if (strlen(value) > CGIT_MAX_SEARCH_LEN) - ctx.qry.search = xstrndup(value, CGIT_MAX_SEARCH_LEN); - else - ctx.qry.search = xstrdup(value); - } else if (!strcmp(name, "h")) { - ctx.qry.head = xstrdup(value); - ctx.qry.has_symref = 1; - } else if (!strcmp(name, "id")) { - ctx.qry.oid = xstrdup(value); - ctx.qry.has_oid = 1; - } else if (!strcmp(name, "id2")) { - ctx.qry.oid2 = xstrdup(value); - ctx.qry.has_oid = 1; - } else if (!strcmp(name, "ofs")) { - /* Bound the upper end so a crafted value cannot force a walk - * over the whole history (and strtol avoids the atoi overflow). - * Only clamp the upper end: ofs is overloaded, the stats page - * submits -1 for "all authors", and the log skip loop already - * floors negatives at zero. */ - long ofs = strtol(value, NULL, 10); - if (ofs > 100000) - ofs = 100000; - else if (ofs < -1) - ofs = -1; - ctx.qry.ofs = ofs; - } else if (!strcmp(name, "path")) { - ctx.qry.path = cgit_trim_end(value, '/'); - } else if (!strcmp(name, "s")) { - ctx.qry.sort = xstrdup(value); - } else if (!strcmp(name, "showmsg")) { - ctx.qry.showmsg = atoi(value); - } else if (!strcmp(name, "period")) { - ctx.qry.period = xstrdup(value); - } else if (!strcmp(name, "dt")) { - ctx.qry.difftype = atoi(value); - ctx.qry.has_difftype = 1; - } else if (!strcmp(name, "ss")) { - /* No longer generated, but there may be links out there. */ - ctx.qry.difftype = atoi(value) ? DIFF_SSDIFF : DIFF_UNIFIED; - ctx.qry.has_difftype = 1; - } else if (!strcmp(name, "all")) { - ctx.qry.show_all = atoi(value); - } else if (!strcmp(name, "context")) { - ctx.qry.context = atoi(value); - } else if (!strcmp(name, "ignorews")) { - ctx.qry.ignorews = atoi(value); - } else if (!strcmp(name, "follow")) { - ctx.qry.follow = atoi(value); - } -} - -static void prepare_context(void) -{ - memset(&ctx, 0, sizeof(ctx)); - ctx.cfg.agefile = "info/web/last-modified"; - ctx.cfg.cache_size = 0; - ctx.cfg.cache_root = CGIT_CACHE_ROOT; - ctx.cfg.cache_about_ttl = 15; - ctx.cfg.cache_snapshot_ttl = 5; - ctx.cfg.cache_repo_ttl = 5; - ctx.cfg.cache_root_ttl = 5; - ctx.cfg.cache_scanrc_ttl = 15; - ctx.cfg.cache_dynamic_ttl = 5; - ctx.cfg.cache_static_ttl = -1; - ctx.cfg.case_sensitive_sort = 1; - ctx.cfg.branch_sort = 0; - ctx.cfg.commit_sort = 0; - ctx.cfg.logo = "/cgit.png"; - ctx.cfg.favicon = "/favicon.ico"; - ctx.cfg.local_time = 0; - ctx.cfg.date_mode = date_mode_from_type(DATE_SHORT); - ctx.cfg.enable_relative_dates = 1; - ctx.cfg.enable_http_clone = 1; - ctx.cfg.enable_index_owner = 1; - ctx.cfg.enable_tree_linenumbers = 1; - ctx.cfg.enable_git_config = 0; - ctx.cfg.max_repo_count = 50; - ctx.cfg.max_commit_count = 50; - ctx.cfg.max_patch_count = 50; - ctx.cfg.max_diff_files = 200; /* larger commits render stat only */ - ctx.cfg.max_diff_lines = 1000; /* larger file diffs link out */ - ctx.cfg.max_msg_len = 80; - ctx.cfg.max_ref_count = 200; /* refs beyond this paginate */ - ctx.cfg.max_repodesc_len = 80; - ctx.cfg.max_blob_size = 10 * 1024; /* 10 MB; bounds per-request memory */ - ctx.cfg.max_stats = 0; - ctx.cfg.project_list = NULL; - ctx.cfg.renamelimit = -1; - ctx.cfg.remove_suffix = 0; - ctx.cfg.robots = "index, nofollow"; - ctx.cfg.root_title = "Git repository browser"; - ctx.cfg.root_desc = "a fast webinterface for the git dscm"; - ctx.cfg.scan_hidden_path = 0; - ctx.cfg.script_name = CGIT_SCRIPT_NAME; - ctx.cfg.section = ""; - ctx.cfg.repository_sort = "name"; - ctx.cfg.section_sort = 0; - ctx.cfg.summary_branches = 10; - ctx.cfg.summary_log = 10; - ctx.cfg.summary_tags = 10; - ctx.cfg.max_atom_items = 10; - ctx.cfg.difftype = DIFF_UNIFIED; - ctx.env.cgit_config = getenv("CGIT_CONFIG"); - ctx.env.http_host = getenv("HTTP_HOST"); - ctx.env.https = getenv("HTTPS"); - ctx.env.no_http = getenv("NO_HTTP"); - ctx.env.path_info = getenv("PATH_INFO"); - ctx.env.query_string = getenv("QUERY_STRING"); - ctx.env.request_method = getenv("REQUEST_METHOD"); - ctx.env.script_name = getenv("SCRIPT_NAME"); - ctx.env.server_name = getenv("SERVER_NAME"); - ctx.env.server_port = getenv("SERVER_PORT"); - ctx.env.http_cookie = getenv("HTTP_COOKIE"); - ctx.env.http_referer = getenv("HTTP_REFERER"); - ctx.env.content_length = getenv("CONTENT_LENGTH") ? strtoul(getenv("CONTENT_LENGTH"), NULL, 10) : 0; - ctx.env.authenticated = 0; - ctx.page.mimetype = "text/html"; - ctx.page.charset = PAGE_ENCODING; - ctx.page.filename = NULL; - ctx.page.size = 0; - ctx.page.modified = time(NULL); - ctx.page.expires = ctx.page.modified; - ctx.page.etag = NULL; - string_list_init_dup(&ctx.cfg.mimetypes); - if (ctx.env.script_name) - ctx.cfg.script_name = xstrdup(ctx.env.script_name); - if (ctx.env.query_string) - ctx.qry.raw = xstrdup(ctx.env.query_string); - if (!ctx.env.cgit_config) - ctx.env.cgit_config = CGIT_CONFIG; -} - -struct refmatch { - char *req_ref; - char *first_ref; - int match; -}; - -static int find_current_ref(const struct reference *ref, void *cb_data) -{ - struct refmatch *info; - - info = (struct refmatch *)cb_data; - if (!strcmp(ref->name, info->req_ref)) - info->match = 1; - if (!info->first_ref) - info->first_ref = xstrdup(ref->name); - return info->match; -} - -static void free_refmatch_inner(struct refmatch *info) -{ - if (info->first_ref) - free(info->first_ref); -} - -static char *find_default_branch(struct cgit_repo *repo) -{ - struct refmatch info; - char *ref; - - info.req_ref = repo->defbranch; - info.first_ref = NULL; - info.match = 0; - refs_for_each_branch_ref(get_main_ref_store(the_repository), - find_current_ref, &info); - if (info.match) - ref = info.req_ref; - else - ref = info.first_ref; - if (ref) - ref = xstrdup(ref); - free_refmatch_inner(&info); - - return ref; -} - -static char *guess_defbranch(void) -{ - const char *ref, *refname; - struct object_id oid; - - ref = refs_resolve_ref_unsafe(get_main_ref_store(the_repository), - "HEAD", 0, &oid, NULL); - if (!ref || !skip_prefix(ref, "refs/heads/", &refname)) - return "master"; - return xstrdup(refname); -} - -/* The caller must free filename and ref after calling this. */ -static inline void parse_readme(const char *readme, char **filename, char **ref, struct cgit_repo *repo) -{ - const char *colon; - - *filename = NULL; - *ref = NULL; - - if (!readme || !readme[0]) - return; - - /* Check if the readme is tracked in the git repo. */ - colon = strchr(readme, ':'); - if (colon && strlen(colon) > 1) { - /* If it starts with a colon, we want to use head given - * from query or the default branch */ - if (colon == readme && ctx.qry.head) - *ref = xstrdup(ctx.qry.head); - else if (colon == readme && repo->defbranch) - *ref = xstrdup(repo->defbranch); - else - *ref = xstrndup(readme, colon - readme); - readme = colon + 1; - } - - /* Prepend repo path to relative readme path unless tracked. */ - if (!(*ref) && readme[0] != '/') - *filename = cgit_fmtalloc("%s/%s", repo->path, readme); - else - *filename = xstrdup(readme); -} -static void choose_readme(struct cgit_repo *repo) -{ - int found; - char *filename, *ref; - struct string_list_item *entry; - - if (!repo->readme.nr) - return; - - found = 0; - for_each_string_list_item(entry, &repo->readme) { - parse_readme(entry->string, &filename, &ref, repo); - if (!filename) { - free(filename); - free(ref); - continue; - } - if (ref) { - if (cgit_ref_path_exists(filename, ref, 1)) { - found = 1; - break; - } - } - else if (!access(filename, R_OK)) { - found = 1; - break; - } - free(filename); - free(ref); - } - repo->readme.strdup_strings = 1; - string_list_clear(&repo->readme, 0); - repo->readme.strdup_strings = 0; - if (found) - string_list_append(&repo->readme, filename)->util = ref; -} - -static void prepare_repo_env(int *nongit) +/* + * Read a whole number a request supplied, clamped into the range the caller + * accepts. strtol rather than atoi, because atoi has no defined behaviour once + * the digits overflow and every value here arrives straight from the query + * string. + */ +static int query_int(const char *value, int min, int max) { - /* The path to the git repository. */ - setenv("GIT_DIR", ctx.repo->path, 1); + long number = strtol(value, NULL, 10); - /* Setup the git directory and initialize the notes system. Both of these - * load local configuration from the git repository, so we do them both while - * the HOME variables are unset. */ - setup_git_directory_gently(the_repository, nongit); - load_display_notes(NULL); + if (number < min) + return min; + if (number > max) + return max; + return number; } -static int prepare_repo_cmd(int nongit) +static void apply_query_param(const char *name, const char *value) { - struct object_id oid; - int rc; - - if (nongit) { - const char *name = ctx.repo->name; - rc = errno; - ctx.page.title = cgit_fmtalloc("%s - %s", ctx.cfg.root_title, - "config error"); - ctx.repo = NULL; - cgit_print_http_headers(); - cgit_print_docstart(); - cgit_print_pageheader(); - cgit_print_error("Failed to open %s: %s", name, - rc ? strerror(rc) : "Not a valid git repository"); - cgit_print_docend(); - return 1; - } - ctx.page.title = cgit_fmtalloc("%s - %s", ctx.repo->name, ctx.repo->desc); - - if (!ctx.repo->defbranch) - ctx.repo->defbranch = guess_defbranch(); - - if (!ctx.qry.head) { - ctx.qry.nohead = 1; - ctx.qry.head = find_default_branch(ctx.repo); - } - - if (!ctx.qry.head) { - ctx.empty_repo = 1; - // Before the document starts, since the carries - // and those clone urls expand macros such - // as $CGIT_REPO_URL out of this environment. - cgit_prepare_repo_env(ctx.repo); - cgit_print_http_headers(); - cgit_print_docstart(); - cgit_print_pageheader(); - cgit_print_empty_repo(); - cgit_print_docend(); - return 1; - } - - if (repo_get_oid(the_repository, ctx.qry.head, &oid)) { - char *old_head = ctx.qry.head; - ctx.qry.head = xstrdup(ctx.repo->defbranch); - cgit_print_error_page(404, "Not found", - "Invalid branch: %s", old_head); - free(old_head); - return 1; + if (!value) + value = ""; + + if (!strcmp(name,"r")) { + ctx.qry.repo = xstrdup(value); + ctx.repo = cgit_get_repoinfo(value); + } else if (!strcmp(name, "p")) { + ctx.qry.page = xstrdup(value); + } else if (!strcmp(name, "url")) { + // Every leading slash goes, not just one. What is left is + // joined onto the virtual root, so a value like //example.com + // would otherwise survive as /example.com and make that join a + // scheme-relative link to another host. + while (*value == '/') + value++; + ctx.qry.url = xstrdup(value); + cgit_parse_url(value); + } else if (!strcmp(name, "qt")) { + ctx.qry.grep = xstrdup(value); + } else if (!strcmp(name, "q")) { + if (strlen(value) > MAX_SEARCH_LEN) + ctx.qry.search = xstrndup(value, MAX_SEARCH_LEN); + else + ctx.qry.search = xstrdup(value); + } else if (!strcmp(name, "h")) { + ctx.qry.head = xstrdup(value); + ctx.qry.has_symref = 1; + } else if (!strcmp(name, "id")) { + ctx.qry.oid = xstrdup(value); + ctx.qry.has_oid = 1; + } else if (!strcmp(name, "id2")) { + ctx.qry.oid2 = xstrdup(value); + ctx.qry.has_oid = 1; + } else if (!strcmp(name, "ofs")) { + // Bounded above so a crafted value cannot force a walk over the + // whole history. Negatives stop at -1 rather than at zero, + // because the offset is overloaded, the stats page submits -1 + // for all authors, and the log skip loop floors a negative + // itself. + ctx.qry.ofs = query_int(value, -1, MAX_QUERY_OFFSET); + } else if (!strcmp(name, "path")) { + ctx.qry.path = cgit_trim_end(value, '/'); + } else if (!strcmp(name, "s")) { + ctx.qry.sort = xstrdup(value); + } else if (!strcmp(name, "showmsg")) { + ctx.qry.showmsg = query_int(value, INT_MIN, INT_MAX); + } else if (!strcmp(name, "period")) { + ctx.qry.period = xstrdup(value); + } else if (!strcmp(name, "dt")) { + ctx.qry.difftype = query_int(value, INT_MIN, INT_MAX); + ctx.qry.has_difftype = 1; + } else if (!strcmp(name, "ss")) { + // No longer generated, but old links still carry it. + ctx.qry.difftype = query_int(value, INT_MIN, INT_MAX) ? + DIFF_SSDIFF : DIFF_UNIFIED; + ctx.qry.has_difftype = 1; + } else if (!strcmp(name, "all")) { + ctx.qry.show_all = query_int(value, INT_MIN, INT_MAX); + } else if (!strcmp(name, "context")) { + // Context lines are not counted against max-diff-lines, so an + // unbounded width turns a whole blob into context and renders + // it in full however small the change was. + ctx.qry.context = query_int(value, 0, MAX_DIFF_CONTEXT_LINES); + } else if (!strcmp(name, "ignorews")) { + ctx.qry.ignorews = query_int(value, INT_MIN, INT_MAX); + } else if (!strcmp(name, "follow")) { + ctx.qry.follow = query_int(value, INT_MIN, INT_MAX); } - string_list_sort(&ctx.repo->submodules); - cgit_prepare_repo_env(ctx.repo); - choose_readme(ctx.repo); - return 0; } -static inline void open_auth_filter(const char *function) +static void open_auth_filter(const char *action) { - cgit_open_filter(ctx.cfg.auth_filter, function, + cgit_open_filter(ctx.cfg.auth_filter, action, ctx.env.http_cookie ? ctx.env.http_cookie : "", ctx.env.request_method ? ctx.env.request_method : "", ctx.env.query_string ? ctx.env.query_string : "", @@ -706,8 +762,11 @@ static inline void open_auth_filter(const char *function) cgit_loginurl()); } -/* The filter is expected to spit out "Status: " and all headers. */ -static inline void authenticate_post(void) +/* + * The filter answers the login POST itself, writing the status line and every + * header, so nothing here prints any and the process ends before this returns. + */ +static void authenticate_post(void) { char buffer[MAX_AUTHENTICATION_POST_BYTES]; size_t len; @@ -725,369 +784,398 @@ static inline void authenticate_post(void) exit(0); } -static inline void authenticate_cookie(void) +static void authenticate_cookie(void) { - /* If we don't have an auth_filter, consider all cookies valid, and thus return early. */ if (!ctx.cfg.auth_filter) { ctx.env.authenticated = 1; return; } - /* If we're having something POST'd to /login, we're authenticating POST, - * instead of the cookie, so call authenticate_post and bail out early. - * This pattern here should match /?p=login with POST. */ - if (ctx.env.request_method && ctx.qry.page && !ctx.repo && \ - !strcmp(ctx.env.request_method, "POST") && !strcmp(ctx.qry.page, "login")) { + if (ctx.env.request_method && ctx.qry.page && !ctx.repo && + !strcmp(ctx.env.request_method, "POST") && + !strcmp(ctx.qry.page, "login")) { authenticate_post(); return; } - /* If we've made it this far, we're authenticating the cookie for real, so do that. */ open_auth_filter("authenticate-cookie"); ctx.env.authenticated = cgit_close_filter(ctx.cfg.auth_filter); } -static void process_request(void) +// Every cache-*-ttl setting is written in minutes, and so is this. +static int calc_ttl(void) { - struct cgit_cmd *cmd; - int nongit = 0; - - /* If we're not yet authenticated, no matter what page we're on, - * display the authentication body from the auth_filter. This should - * never be cached. */ - if (!ctx.env.authenticated) { - ctx.page.title = "Authentication Required"; - cgit_print_http_headers(); - cgit_print_docstart(); - cgit_print_pageheader(); - open_auth_filter("body"); - cgit_close_filter(ctx.cfg.auth_filter); - cgit_print_docend(); - return; - } - - if (ctx.repo) - prepare_repo_env(&nongit); + if (!ctx.repo) + return ctx.cfg.cache_root_ttl; - cmd = cgit_get_cmd(); - if (!cmd) { - ctx.page.title = "cgit error"; - cgit_print_error_page(404, "Not found", "Invalid request"); - return; - } + if (!ctx.qry.page) + return ctx.cfg.cache_repo_ttl; - if (!ctx.cfg.enable_http_clone && cmd->is_clone) { - ctx.page.title = "cgit error"; - cgit_print_error_page(404, "Not found", "Invalid request"); - return; - } + if (!strcmp(ctx.qry.page, "about")) + return ctx.cfg.cache_about_ttl; - if (cmd->want_repo && !ctx.repo) { - cgit_print_error_page(400, "Bad request", - "No repository selected"); - return; - } + if (!strcmp(ctx.qry.page, "snapshot")) + return ctx.cfg.cache_snapshot_ttl; - /* If cmd->want_vpath is set, assume ctx.qry.path contains a "virtual" - * in-project path limit to be made available at ctx.qry.vpath. - * Otherwise, no path limit is in effect (ctx.qry.vpath = NULL). - */ - ctx.qry.vpath = cmd->want_vpath ? ctx.qry.path : NULL; + if (ctx.qry.has_oid) + return ctx.cfg.cache_static_ttl; - if (ctx.repo && prepare_repo_cmd(nongit)) - return; + if (ctx.qry.has_symref) + return ctx.cfg.cache_dynamic_ttl; - cmd->fn(); + return ctx.cfg.cache_repo_ttl; } -static int cmp_repos(const void *a, const void *b) +/* + * The scheme and the host are folded in because the absolute urls a page + * carries, its clone urls and atom links, are built from them, so a request + * arriving with a spoofed Host must not poison the page served to a visitor + * who came in on the real one. + */ +static void build_cache_key(struct strbuf *key) { - const struct cgit_repo *ra = a, *rb = b; - return strcmp(ra->url, rb->url); + char *hosturl = cgit_hosturl(); + const char *parts[] = { + cgit_httpscheme(), + hosturl ? hosturl : "", + ctx.env.path_info ? ctx.env.path_info : "", + ctx.env.query_string ? ctx.env.query_string : "", + }; + size_t i; + + // Each part is written behind its own length, so nothing a value + // contains can make two different requests spell one key. The path and + // the query come from the environment rather than the query string cgit + // rebuilds, since that rebuild folds the two together and would let the + // PATH_INFO and QUERY_STRING forms of one request share a slot. + for (i = 0; i < ARRAY_SIZE(parts); i++) + strbuf_addf(key, "%zu|%s", strlen(parts[i]), parts[i]); + free(hosturl); } -static char *build_snapshot_setting(int bitmap) +// Returning non-zero ends git's walk, so the search stops at the first hit. +static int find_current_ref(const struct reference *ref, void *data) { - const struct cgit_snapshot_format *f; - struct strbuf result = STRBUF_INIT; + struct refmatch *match = data; - for (f = cgit_snapshot_formats; f->suffix; f++) { - if (cgit_snapshot_format_bit(f) & bitmap) { - if (result.len) - strbuf_addch(&result, ' '); - strbuf_addstr(&result, f->suffix); - } - } - return strbuf_detach(&result, NULL); + if (!strcmp(ref->name, match->wanted)) + match->found = 1; + if (!match->first) + match->first = xstrdup(ref->name); + return match->found; } -static void print_repo(FILE *f, struct cgit_repo *repo) +static char *find_default_branch(struct cgit_repo *repo) { - struct string_list_item *item; - fprintf(f, "repo.url=%s\n", repo->url); - fprintf(f, "repo.name=%s\n", repo->name); - fprintf(f, "repo.path=%s\n", repo->path); - if (repo->owner) - fprintf(f, "repo.owner=%s\n", repo->owner); - if (repo->desc) - fprintf(f, "repo.desc=%s\n", repo->desc); - for_each_string_list_item(item, &repo->readme) { - if (item->util) - fprintf(f, "repo.readme=%s:%s\n", (char *)item->util, item->string); - else - fprintf(f, "repo.readme=%s\n", item->string); - } - if (repo->defbranch) - fprintf(f, "repo.defbranch=%s\n", repo->defbranch); - if (repo->extra_head_content) - fprintf(f, "repo.extra-head-content=%s\n", repo->extra_head_content); - if (repo->module_link) - fprintf(f, "repo.module-link=%s\n", repo->module_link); - if (repo->section) - fprintf(f, "repo.section=%s\n", repo->section); - if (repo->homepage) - fprintf(f, "repo.homepage=%s\n", repo->homepage); - if (repo->clone_url) - fprintf(f, "repo.clone-url=%s\n", repo->clone_url); - fprintf(f, "repo.enable-blame=%d\n", - repo->enable_blame); - fprintf(f, "repo.enable-commit-graph=%d\n", - repo->enable_commit_graph); - fprintf(f, "repo.enable-follow-links=%d\n", - repo->enable_follow_links); - fprintf(f, "repo.enable-log-filecount=%d\n", - repo->enable_log_filecount); - fprintf(f, "repo.enable-log-linecount=%d\n", - repo->enable_log_linecount); - if (repo->about_filter && repo->about_filter != ctx.cfg.about_filter) - cgit_fprintf_filter(repo->about_filter, f, "repo.about-filter="); - if (repo->commit_filter && repo->commit_filter != ctx.cfg.commit_filter) - cgit_fprintf_filter(repo->commit_filter, f, "repo.commit-filter="); - if (repo->source_filter && repo->source_filter != ctx.cfg.source_filter) - cgit_fprintf_filter(repo->source_filter, f, "repo.source-filter="); - if (repo->email_filter && repo->email_filter != ctx.cfg.email_filter) - cgit_fprintf_filter(repo->email_filter, f, "repo.email-filter="); - if (repo->snapshots != ctx.cfg.snapshots) { - char *tmp = build_snapshot_setting(repo->snapshots); - fprintf(f, "repo.snapshots=%s\n", tmp ? tmp : ""); - free(tmp); - } - if (repo->snapshot_prefix) - fprintf(f, "repo.snapshot-prefix=%s\n", repo->snapshot_prefix); - if (repo->enable_stats != ctx.cfg.enable_stats) - fprintf(f, "repo.enable-stats=%d\n", repo->enable_stats); - if (repo->max_stats != ctx.cfg.max_stats) - fprintf(f, "repo.max-stats=%s\n", - cgit_find_stats_periodname(repo->max_stats)); - if (repo->logo) - fprintf(f, "repo.logo=%s\n", repo->logo); - if (repo->logo_link) - fprintf(f, "repo.logo-link=%s\n", repo->logo_link); - fprintf(f, "repo.enable-remote-branches=%d\n", repo->enable_remote_branches); - fprintf(f, "repo.enable-subject-links=%d\n", repo->enable_subject_links); - fprintf(f, "repo.enable-html-serving=%d\n", repo->enable_html_serving); - if (repo->branch_sort == 1) - fprintf(f, "repo.branch-sort=age\n"); - if (repo->commit_sort) { - if (repo->commit_sort == 1) - fprintf(f, "repo.commit-sort=date\n"); - else if (repo->commit_sort == 2) - fprintf(f, "repo.commit-sort=topo\n"); - } - fprintf(f, "repo.hide=%d\n", repo->hide); - fprintf(f, "repo.ignore=%d\n", repo->ignore); - fprintf(f, "\n"); + struct refmatch match; + char *ref; + + match.wanted = repo->defbranch; + match.first = NULL; + match.found = 0; + refs_for_each_branch_ref(get_main_ref_store(the_repository), + find_current_ref, &match); + if (match.found) + ref = match.wanted; + else + ref = match.first; + if (ref) + ref = xstrdup(ref); + free(match.first); + + return ref; } -static void print_repolist(FILE *f, struct cgit_repolist *list, int start) +static char *guess_defbranch(void) { - int i; + const char *ref, *refname; + struct object_id oid; - for (i = start; i < list->count; i++) - print_repo(f, &list->repos[i]); + ref = refs_resolve_ref_unsafe(get_main_ref_store(the_repository), + "HEAD", 0, &oid, NULL); + if (!ref || !skip_prefix(ref, "refs/heads/", &refname)) + return "master"; + return xstrdup(refname); } -/* Scan 'path' for git repositories, save the resulting repolist in 'cached_rc' - * and return 0 on success. +/* + * Split one readme setting into the file it names and the ref that file is + * read from, leaving the ref NULL for a file on disk. The caller frees both. */ -static int generate_cached_repolist(const char *path, const char *cached_rc) +static void parse_readme(const char *readme, char **filename, char **ref, + struct cgit_repo *repo) { - struct strbuf locked_rc = STRBUF_INIT; - int result = 0; - int idx; - FILE *f; + const char *colon; - strbuf_addf(&locked_rc, "%s.lock", cached_rc); - f = fopen(locked_rc.buf, "wx"); - if (!f) { - /* Inform about the error unless the lockfile already existed, - * since that only means we've got concurrent requests. - */ - result = errno; - if (result != EEXIST) - fprintf(stderr, "[cgit] Error opening %s: %s (%d)\n", - locked_rc.buf, strerror(result), result); - goto out; + *filename = NULL; + *ref = NULL; + + if (!readme || !readme[0]) + return; + + // A colon separates a ref from a path, so a setting carrying one names + // a file tracked in the repository rather than one on disk. + colon = strchr(readme, ':'); + if (colon && strlen(colon) > 1) { + if (colon == readme && ctx.qry.head) + *ref = xstrdup(ctx.qry.head); + else if (colon == readme && repo->defbranch) + *ref = xstrdup(repo->defbranch); + else + *ref = xstrndup(readme, colon - readme); + readme = colon + 1; } - idx = cgit_repolist.count; - if (ctx.cfg.project_list) - scan_projects(path, ctx.cfg.project_list); + + if (!(*ref) && readme[0] != '/') + *filename = cgit_fmtalloc("%s/%s", repo->path, readme); else - scan_tree(path); - print_repolist(f, &cgit_repolist, idx); - if (rename(locked_rc.buf, cached_rc)) - fprintf(stderr, "[cgit] Error renaming %s to %s: %s (%d)\n", - locked_rc.buf, cached_rc, strerror(errno), errno); - fclose(f); -out: - strbuf_release(&locked_rc); - return result; + *filename = xstrdup(readme); +} + +static void choose_readme(struct cgit_repo *repo) +{ + int found; + char *filename, *ref; + struct string_list_item *entry; + + if (!repo->readme.nr) + return; + + found = 0; + for_each_string_list_item(entry, &repo->readme) { + parse_readme(entry->string, &filename, &ref, repo); + if (!filename) { + free(filename); + free(ref); + continue; + } + if (ref) { + if (cgit_ref_path_exists(filename, ref, 1)) { + found = 1; + break; + } + } + else if (!access(filename, R_OK)) { + found = 1; + break; + } + free(filename); + free(ref); + } + repo->readme.strdup_strings = 1; + string_list_clear(&repo->readme, 0); + repo->readme.strdup_strings = 0; + if (found) + string_list_append(&repo->readme, filename)->util = ref; } -static void process_cached_repolist(const char *path) +static void prepare_repo_env(int *nongit) { - struct stat st; - struct strbuf cached_rc = STRBUF_INIT; - time_t age; - unsigned long hash; + setenv("GIT_DIR", ctx.repo->path, 1); - hash = cache_hash_str(path); - if (ctx.cfg.project_list) - hash += cache_hash_str(ctx.cfg.project_list); - strbuf_addf(&cached_rc, "%s/rc-%8lx", ctx.cfg.cache_root, hash); + // Both read configuration out of the repository, with the user's own + // git configuration already stripped by isolate_git_environment. + setup_git_directory_gently(the_repository, nongit); + load_display_notes(NULL); +} - if (stat(cached_rc.buf, &st)) { - /* Nothing is cached, we need to scan without forking. And - * if we fail to generate a cached repolist, we need to - * invoke scan_tree manually. - */ - if (generate_cached_repolist(path, cached_rc.buf)) { - if (ctx.cfg.project_list) - scan_projects(path, ctx.cfg.project_list); - else - scan_tree(path); - } - goto out; +/* + * Returns non-zero once it has written a complete response of its own, in + * which case the caller must not render a page over the top of it. + */ +static int prepare_repo_cmd(int nongit) +{ + struct object_id oid; + int err; + + if (nongit) { + const char *name = ctx.repo->name; + err = errno; + ctx.page.title = cgit_fmtalloc("%s - %s", ctx.cfg.root_title, + "config error"); + ctx.repo = NULL; + cgit_print_http_headers(); + cgit_print_docstart(); + cgit_print_pageheader(); + cgit_print_error("Failed to open %s: %s", name, + err ? strerror(err) : "Not a valid git repository"); + cgit_print_docend(); + return 1; } + ctx.page.title = cgit_fmtalloc("%s - %s", ctx.repo->name, ctx.repo->desc); - parse_configfile(cached_rc.buf, config_cb); + if (!ctx.repo->defbranch) + ctx.repo->defbranch = guess_defbranch(); - /* If the cached configfile hasn't expired, lets exit now */ - age = time(NULL) - st.st_mtime; - if (age <= (ctx.cfg.cache_scanrc_ttl * 60)) - goto out; + if (!ctx.qry.head) { + ctx.qry.nohead = 1; + ctx.qry.head = find_default_branch(ctx.repo); + } - /* The cached repolist has been parsed, but it was old. So lets - * rescan the specified path and generate a new cached repolist - * in a child-process to avoid latency for the current request. - */ - if (fork()) - goto out; + if (!ctx.qry.head) { + ctx.empty_repo = 1; + // Before the document starts, since the carries + // and those clone urls expand macros such + // as $CGIT_REPO_URL out of this environment. + cgit_prepare_repo_env(ctx.repo); + cgit_print_http_headers(); + cgit_print_docstart(); + cgit_print_pageheader(); + cgit_print_empty_repo(); + cgit_print_docend(); + return 1; + } - exit(generate_cached_repolist(path, cached_rc.buf)); -out: - strbuf_release(&cached_rc); + if (repo_get_oid(the_repository, ctx.qry.head, &oid)) { + char *old_head = ctx.qry.head; + ctx.qry.head = xstrdup(ctx.repo->defbranch); + cgit_print_error_page(404, "Not found", + "Invalid branch: %s", old_head); + free(old_head); + return 1; + } + string_list_sort(&ctx.repo->submodules); + cgit_prepare_repo_env(ctx.repo); + choose_readme(ctx.repo); + return 0; } -static void cgit_parse_args(int argc, const char **argv) +static void process_request(void) { - int i; - const char *arg; - int scan = 0; - - for (i = 1; i < argc; i++) { - if (!strcmp(argv[i], "--version")) { - printf("CGit %s | https://github.com/brycekwon/cgit\n\nCompiled in features:\n", CGIT_VERSION); -#ifdef NO_LUA - printf("[-] "); -#else - printf("[+] "); -#endif - printf("Lua scripting\n"); -#ifndef HAVE_LINUX_SENDFILE - printf("[-] "); -#else - printf("[+] "); -#endif - printf("Linux sendfile() usage\n"); + const struct cgit_cmd *cmd; + int nongit = 0; - exit(0); - } - if (skip_prefix(argv[i], "--cache=", &arg)) { - ctx.cfg.cache_root = xstrdup(arg); - } else if (!strcmp(argv[i], "--nohttp")) { - ctx.env.no_http = "1"; - } else if (skip_prefix(argv[i], "--query=", &arg)) { - ctx.qry.raw = xstrdup(arg); - } else if (skip_prefix(argv[i], "--repo=", &arg)) { - ctx.qry.repo = xstrdup(arg); - } else if (skip_prefix(argv[i], "--page=", &arg)) { - ctx.qry.page = xstrdup(arg); - } else if (skip_prefix(argv[i], "--head=", &arg)) { - ctx.qry.head = xstrdup(arg); - ctx.qry.has_symref = 1; - } else if (skip_prefix(argv[i], "--oid=", &arg)) { - ctx.qry.oid = xstrdup(arg); - ctx.qry.has_oid = 1; - } else if (skip_prefix(argv[i], "--ofs=", &arg)) { - ctx.qry.ofs = atoi(arg); - } else if (skip_prefix(argv[i], "--scan-tree=", &arg) || - skip_prefix(argv[i], "--scan-path=", &arg)) { - /* - * HACK: The global snapshot bit mask defines the set - * of allowed snapshot formats, but the config file - * hasn't been parsed yet so the mask is currently 0. - * By setting all bits high before scanning we make - * sure that any in-repo cgitrc snapshot setting is - * respected by scan_tree(). - * - * NOTE: We assume that there aren't more than 8 - * different snapshot formats supported by cgit... - */ - ctx.cfg.snapshots = 0xFF; - scan++; - scan_tree(arg); - } - } - if (scan) { - qsort(cgit_repolist.repos, cgit_repolist.count, - sizeof(struct cgit_repo), cmp_repos); - print_repolist(stdout, &cgit_repolist, 0); - exit(0); + // An unauthenticated request is answered with the filter's own body + // whatever page it asked for. + if (!ctx.env.authenticated) { + ctx.page.title = "Authentication Required"; + cgit_print_http_headers(); + cgit_print_docstart(); + cgit_print_pageheader(); + open_auth_filter("body"); + cgit_close_filter(ctx.cfg.auth_filter); + cgit_print_docend(); + return; } -} -static int calc_ttl(void) -{ - if (!ctx.repo) - return ctx.cfg.cache_root_ttl; + if (ctx.repo) + prepare_repo_env(&nongit); - if (!ctx.qry.page) - return ctx.cfg.cache_repo_ttl; + cmd = cgit_get_cmd(); + if (!cmd) { + ctx.page.title = "cgit error"; + cgit_print_error_page(404, "Not found", "Invalid request"); + return; + } - if (!strcmp(ctx.qry.page, "about")) - return ctx.cfg.cache_about_ttl; + if (!ctx.cfg.enable_http_clone && cmd->is_clone) { + ctx.page.title = "cgit error"; + cgit_print_error_page(404, "Not found", "Invalid request"); + return; + } - if (!strcmp(ctx.qry.page, "snapshot")) - return ctx.cfg.cache_snapshot_ttl; + if (cmd->want_repo && !ctx.repo) { + cgit_print_error_page(400, "Bad request", + "No repository selected"); + return; + } - if (ctx.qry.has_oid) - return ctx.cfg.cache_static_ttl; + ctx.qry.vpath = cmd->want_vpath ? ctx.qry.path : NULL; - if (ctx.qry.has_symref) - return ctx.cfg.cache_dynamic_ttl; + if (ctx.repo && prepare_repo_cmd(nongit)) + return; - return ctx.cfg.cache_repo_ttl; + cmd->fn(); } -static NORETURN void cgit_die_routine(const char *msg, va_list params) +void cgit_repo_config(struct cgit_repo *repo, const char *name, const char *value) { - cgit_vprint_error_page(400, "Bad request", msg, params); - exit(0); + const char *path; + struct string_list_item *item; + + if (!strcmp(name, "name")) + repo->name = cgit_strdup_first_line(value); + else if (!strcmp(name, "clone-url")) + repo->clone_url = cgit_strdup_first_line(value); + else if (!strcmp(name, "desc")) + repo->desc = cgit_strdup_first_line(value); + else if (!strcmp(name, "owner")) + repo->owner = cgit_strdup_first_line(value); + else if (!strcmp(name, "homepage")) + repo->homepage = cgit_strdup_first_line(value); + else if (!strcmp(name, "defbranch")) + repo->defbranch = cgit_strdup_first_line(value); + else if (!strcmp(name, "extra-head-content")) + repo->extra_head_content = cgit_strdup_first_line(value); + else if (!strcmp(name, "snapshots")) + repo->snapshots = ctx.cfg.snapshots & cgit_parse_snapshots_mask(value); + else if (!strcmp(name, "enable-blame")) + repo->enable_blame = atoi(value); + else if (!strcmp(name, "enable-commit-graph")) + repo->enable_commit_graph = atoi(value); + else if (!strcmp(name, "enable-follow-links")) + repo->enable_follow_links = atoi(value); + else if (!strcmp(name, "enable-log-filecount")) + repo->enable_log_filecount = atoi(value); + else if (!strcmp(name, "enable-log-linecount")) + repo->enable_log_linecount = atoi(value); + else if (!strcmp(name, "enable-remote-branches")) + repo->enable_remote_branches = atoi(value); + else if (!strcmp(name, "enable-subject-links")) + repo->enable_subject_links = atoi(value); + else if (!strcmp(name, "enable-html-serving")) + repo->enable_html_serving = atoi(value); + else if (!strcmp(name, "enable-stats")) + repo->enable_stats = atoi(value); + else if (!strcmp(name, "branch-sort")) { + if (!strcmp(value, "age")) + repo->branch_sort = 1; + if (!strcmp(value, "name")) + repo->branch_sort = 0; + } else if (!strcmp(name, "commit-sort")) { + if (!strcmp(value, "date")) + repo->commit_sort = 1; + if (!strcmp(value, "topo")) + repo->commit_sort = 2; + } else if (!strcmp(name, "max-stats")) + repo->max_stats = cgit_find_stats_period(value, NULL); + else if (!strcmp(name, "module-link")) + repo->module_link = cgit_strdup_first_line(value); + else if (skip_prefix(name, "module-link.", &path)) { + item = string_list_append(&repo->submodules, + cgit_strdup_first_line(path)); + item->util = cgit_strdup_first_line(value); + } else if (!strcmp(name, "section")) + repo->section = cgit_strdup_first_line(value); + else if (!strcmp(name, "snapshot-prefix")) + repo->snapshot_prefix = cgit_strdup_first_line(value); + else if (!strcmp(name, "readme") && value != NULL) { + if (repo->readme.items == ctx.cfg.readme.items) + memset(&repo->readme, 0, sizeof(repo->readme)); + string_list_append(&repo->readme, cgit_strdup_first_line(value)); + } else if (!strcmp(name, "logo") && value != NULL) + repo->logo = cgit_strdup_first_line(value); + else if (!strcmp(name, "logo-link") && value != NULL) + repo->logo_link = cgit_strdup_first_line(value); + else if (!strcmp(name, "hide")) + repo->hide = atoi(value); + else if (!strcmp(name, "ignore")) + repo->ignore = atoi(value); + else if (ctx.cfg.enable_filter_overrides) { + if (!strcmp(name, "about-filter")) + repo->about_filter = cgit_new_filter(value, ABOUT); + else if (!strcmp(name, "commit-filter")) + repo->commit_filter = cgit_new_filter(value, COMMIT); + else if (!strcmp(name, "source-filter")) + repo->source_filter = cgit_new_filter(value, SOURCE); + else if (!strcmp(name, "email-filter")) + repo->email_filter = cgit_new_filter(value, EMAIL); + } } int cmd_main(int argc, const char **argv) { + struct strbuf cache_key = STRBUF_INIT; const char *path; int err, ttl; @@ -1097,71 +1185,58 @@ int cmd_main(int argc, const char **argv) // Registered second so it runs first, since exit is reached from error // paths and from the HEAD shortcut with a page still buffered. atexit(html_flush); - set_die_routine(cgit_die_routine); + set_die_routine(die_routine); prepare_context(); cgit_repolist.length = 0; cgit_repolist.count = 0; cgit_repolist.repos = NULL; - cgit_parse_args(argc, argv); - parse_configfile(cgit_expand_macros(ctx.env.cgit_config), config_cb); + parse_args(argc, argv); + config_file_parse(cgit_expand_macros(ctx.env.cgit_config), apply_config); ctx.repo = NULL; - http_parse_querystring(ctx.qry.raw, querystring_cb); + http_parse_querystring(ctx.qry.raw, apply_query_param); - /* If virtual-root isn't specified in cgitrc, lets pretend - * that virtual-root equals SCRIPT_NAME, minus any possibly - * trailing slashes. - */ if (!ctx.cfg.virtual_root && ctx.cfg.script_name) ctx.cfg.virtual_root = cgit_ensure_end(ctx.cfg.script_name, '/'); - /* If no url parameter is specified on the querystring, lets - * use PATH_INFO as url. This allows cgit to work with virtual - * urls without the need for rewriterules in the webserver (as - * long as PATH_INFO is included in the cache lookup key). - */ + // Falling back to PATH_INFO lets cgit serve virtual urls without a + // rewrite rule in the web server, and folding it into the raw query + // string keeps it part of the cache key. path = ctx.env.path_info; if (!ctx.qry.url && path) { - if (path[0] == '/') + // Stripped like the url parameter and for the same reason, so + // a request for //example.com cannot turn into a link off site. + while (*path == '/') path++; ctx.qry.url = xstrdup(path); if (ctx.qry.raw) { - char *newqry = cgit_fmtalloc("%s?%s", path, ctx.qry.raw); + char *path_and_query = cgit_fmtalloc("%s?%s", path, ctx.qry.raw); free(ctx.qry.raw); - ctx.qry.raw = newqry; + ctx.qry.raw = path_and_query; } else ctx.qry.raw = xstrdup(ctx.qry.url); cgit_parse_url(ctx.qry.url); } - /* Before we go any further, we set ctx.env.authenticated by checking to see - * if the supplied cookie is valid. All cookies are valid if there is no - * auth_filter. If there is an auth_filter, the filter decides. */ authenticate_cookie(); ttl = calc_ttl(); if (ttl < 0) - ctx.page.expires += 10 * 365 * 24 * 60 * 60; /* 10 years */ + ctx.page.expires += NEVER_EXPIRES_SECONDS; else ctx.page.expires += ttl * 60; - if (!ctx.env.authenticated || (ctx.env.request_method && !strcmp(ctx.env.request_method, "HEAD"))) + // An unauthenticated request gets a body meant for one visitor, and a + // HEAD request stops after the headers. + if (!ctx.env.authenticated || + (ctx.env.request_method && !strcmp(ctx.env.request_method, "HEAD"))) ctx.cfg.cache_size = 0; - /* Fold the scheme and host into the cache key. Absolute URLs in the - * output (clone urls, atom links) are built from these, so a request - * with a spoofed Host must not poison the cached page served to a - * visitor arriving on a different host. */ - { - struct strbuf cache_key = STRBUF_INIT; - char *hosturl = cgit_hosturl(); - strbuf_addf(&cache_key, "%s%s|%s", cgit_httpscheme(), - hosturl ? hosturl : "", - ctx.qry.raw ? ctx.qry.raw : ""); - free(hosturl); - err = cache_process(ctx.cfg.cache_size, ctx.cfg.cache_root, - cache_key.buf, ttl, process_request); - strbuf_release(&cache_key); - } + + build_cache_key(&cache_key); + err = cache_process(ctx.cfg.cache_size, ctx.cfg.cache_root, + cache_key.buf, ttl, process_request); + strbuf_release(&cache_key); + cgit_cleanup_filters(); if (err) cgit_print_error("Error processing page: %s (%d)", diff --git a/source/cgit.h b/source/cgit.h index b05876e..eadb680 100644 --- a/source/cgit.h +++ b/source/cgit.h @@ -1,95 +1,71 @@ +/* + * The vocabulary shared by every part of cgit. It pulls in the git headers + * the renderers are written against, declares the record types they pass + * around, and declares the single global ctx holding the parsed request, the + * effective configuration and the repository being served. Anything that + * belongs to one module is declared in that module's own header instead. + */ + #ifndef CGIT_H #define CGIT_H -#include - +// git-compat-util.h sets the feature test macros that git's own headers and +// the system headers are then compiled against, so it has to come first. #include #include +#include #include +#include #include -#include #include +#include #include #include #include #include #include #include -#include #include +#include #include +#include #include #include #include #include +#include #include #include #include #include +#include #include +#include #include #include #include -/* Add isgraph(x) to Git's sane ctype support (see git-compat-util.h) */ -#undef isgraph -#define isgraph(x) (isprint((x)) && !isspace((x))) - - -/* - * Limits used for relative dates - */ -#define TM_MIN 60 -#define TM_HOUR (TM_MIN * 60) -#define TM_DAY (TM_HOUR * 24) -#define TM_WEEK (TM_DAY * 7) -#define TM_YEAR (TM_DAY * 365) -#define TM_MONTH (TM_YEAR / 12.0) - - -/* - * Default encoding - */ +// Every page cgit renders is UTF-8, so commit text in another encoding is +// converted on the way in. #define PAGE_ENCODING "UTF-8" -#define BIT(x) (1U << (x)) - -/* - * Longest search string a request may supply. The filter bar matches its - * query against every item a page lists, so this bounds the work one request - * can ask for. It is not a configuration knob, in the same way as the ofs - * ceiling in querystring_cb. - */ -#define CGIT_MAX_SEARCH_LEN 512 - -typedef void (*configfn)(const char *name, const char *value); -typedef void (*filepair_fn)(struct diff_filepair *pair); -typedef void (*linediff_fn)(char *line, int len); +// The month is a double because a twelfth of a year is not a whole number of +// seconds. +#define SECONDS_PER_MINUTE 60 +#define SECONDS_PER_HOUR (SECONDS_PER_MINUTE * 60) +#define SECONDS_PER_DAY (SECONDS_PER_HOUR * 24) +#define SECONDS_PER_WEEK (SECONDS_PER_DAY * 7) +#define SECONDS_PER_YEAR (SECONDS_PER_DAY * 365) +#define SECONDS_PER_MONTH (SECONDS_PER_YEAR / 12.0) typedef enum { DIFF_UNIFIED, DIFF_SSDIFF, DIFF_STATONLY } diff_type; -typedef enum { - ABOUT, COMMIT, SOURCE, EMAIL, AUTH -} filter_type; - -struct cgit_filter { - int (*open)(struct cgit_filter *, va_list ap); - int (*close)(struct cgit_filter *); - void (*fprintfp)(struct cgit_filter *, FILE *, const char *prefix); - void (*cleanup)(struct cgit_filter *); - int argument_count; -}; - -struct cgit_exec_filter { - struct cgit_filter base; - char *cmd; - char **argv; - int old_stdout; - int pid; -}; +// Filters are only ever held by pointer here, so their layout stays private +// to filter.h. +struct cgit_filter; struct cgit_repo { char *url; @@ -140,11 +116,13 @@ struct commitinfo { struct commit *commit; char *author; char *author_email; - unsigned long author_date; + // git's own width for a commit timestamp, which is wider than a long + // on a 32 bit host and on Windows. + timestamp_t author_date; int author_tz; char *committer; char *committer_email; - unsigned long committer_date; + timestamp_t committer_date; int committer_tz; char *subject; char *msg; @@ -154,7 +132,7 @@ struct commitinfo { struct taginfo { char *tagger; char *tagger_email; - unsigned long tagger_date; + timestamp_t tagger_date; int tagger_tz; char *msg; }; @@ -224,7 +202,7 @@ struct cgit_config { char *script_name; char *section; char *repository_sort; - char *virtual_root; /* Always ends with '/'. */ + char *virtual_root; // Always ends with a slash. char *strict_export; struct date_mode date_mode; int cache_size; @@ -316,7 +294,9 @@ struct cgit_environment { const char *server_port; const char *http_cookie; const char *http_referer; - unsigned int content_length; + // A byte count, so size_t rather than a fixed width type that would + // wrap at four gigabytes on the way in. + size_t content_length; int authenticated; }; @@ -326,97 +306,17 @@ struct cgit_context { struct cgit_config cfg; struct cgit_repo *repo; struct cgit_page page; - // Set once the repository is known to hold no commits, so the page - // header can drop the tabs that cannot render anything. + // Set when the repository holds no commits, so the page header can drop + // the tabs that cannot render anything. int empty_repo; }; -typedef int (*write_archive_fn_t)(const char *, const char *); - -struct cgit_snapshot_format { - const char *suffix; - const char *mimetype; - write_archive_fn_t write_func; -}; - extern const char *cgit_version; extern struct cgit_repolist cgit_repolist; extern struct cgit_context ctx; -extern const struct cgit_snapshot_format cgit_snapshot_formats[]; -extern char *cgit_default_repo_desc; -extern struct cgit_repo *cgit_add_repo(const char *url); -extern struct cgit_repo *cgit_get_repoinfo(const char *url); extern void cgit_repo_config(struct cgit_repo *repo, const char *name, const char *value); -extern int cgit_die_unless_zero(int result, const char *msg); -extern int cgit_die_unless_positive(int result, const char *msg); -extern int cgit_die_unless_non_negative(int result, const char *msg); - -extern char *cgit_trim_end(const char *str, char c); -extern char *cgit_ensure_end(const char *str, char c); - -extern void strbuf_ensure_end(struct strbuf *sb, char c); - -extern void cgit_add_ref(struct reflist *list, struct refinfo *ref); -extern void cgit_free_reflist_inner(struct reflist *list); -extern int cgit_refs_cb(const struct reference *ref, void *cb_data); - -extern void cgit_free_commitinfo(struct commitinfo *info); -extern void cgit_free_taginfo(struct taginfo *info); - -void cgit_diff_tree_cb(struct diff_queue_struct *q, - struct diff_options *options, void *data); - -extern int cgit_diff_files(const struct object_id *old_oid, - const struct object_id *new_oid, - unsigned long *old_size, unsigned long *new_size, - int *binary, int context, int ignorews, - linediff_fn fn); - -extern void cgit_diff_tree(const struct object_id *old_oid, - const struct object_id *new_oid, - filepair_fn fn, const char *prefix, int ignorews); - -extern void cgit_diff_commit(struct commit *commit, filepair_fn fn, - const char *prefix); - -__attribute__((format (printf,1,2))) -extern char *cgit_fmt(const char *format,...); - -__attribute__((format (printf,1,2))) -extern char *cgit_fmtalloc(const char *format,...); - -extern struct commitinfo *cgit_parse_commit(struct commit *commit); -extern struct taginfo *cgit_parse_tag(struct tag *tag); -extern void cgit_parse_url(const char *url); - -extern const char *cgit_repobasename(const char *reponame); - -extern int cgit_parse_snapshots_mask(const char *str); -extern void cgit_parse_date_format(const char *format, struct date_mode *mode); -extern const struct object_id *cgit_snapshot_get_sig(const char *ref, - const struct cgit_snapshot_format *f); -extern unsigned cgit_snapshot_format_bit(const struct cgit_snapshot_format *f); - -extern int cgit_open_filter(struct cgit_filter *filter, ...); -extern int cgit_close_filter(struct cgit_filter *filter); -extern void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, const char *prefix); -extern void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, char **argv); -extern struct cgit_filter *cgit_new_filter(const char *cmd, filter_type filtertype); -extern void cgit_cleanup_filters(void); -extern void cgit_init_filters(void); - -extern void cgit_prepare_repo_env(struct cgit_repo * repo); - -extern int cgit_read_first_line(const char *path, char **buf, size_t *size); - -extern char *cgit_strdup_first_line(const char *txt); - -extern char *cgit_expand_macros(const char *txt); - -extern char *cgit_get_mimetype_for_filename(const char *filename); - -#endif /* CGIT_H */ +#endif // CGIT_H diff --git a/source/cgit.mk b/source/cgit.mk index e0a3312..1cee1fb 100644 --- a/source/cgit.mk +++ b/source/cgit.mk @@ -1,34 +1,30 @@ -# This Makefile runs inside the bundled Git tree (vendor/git) so that it -# can reuse Git's build variables and platform detection. The top-level -# Makefile invokes it as: +# The rules that compile and link cgit itself. They run with the bundled Git +# tree as the working directory so that Git's build variables and platform +# detection can be reused, which is why every path back into the project leads +# through CGIT_ROOT. The top level Makefile invokes it roughly as # # make -C vendor/git -f ../../source/cgit.mk ../../build/cgit # -# so every path back into the project reaches through CGIT_ROOT ("../.."): -# sources come from ../../source and all build output goes to ../../build. +# so sources are read from ../../source and all output is written to +# ../../build. include Makefile -# Locations relative to vendor/git, where this file is run. SRCDIR and -# BUILDDIR name the root-relative subdirs (matching the top-level Makefile and -# used by the version recipe, which cds to the root); CGIT_SRC and CGIT_BUILD -# are their full paths from here. +# SRCDIR and BUILDDIR are named relative to the project root, matching the top +# level Makefile, because the version recipe changes into the root before using +# them. CGIT_SRC and CGIT_BUILD are the same two directories reached from +# vendor/git, where everything else here runs. CGIT_ROOT = ../.. SRCDIR = source BUILDDIR = build CGIT_SRC = $(CGIT_ROOT)/$(SRCDIR) CGIT_BUILD = $(CGIT_ROOT)/$(BUILDDIR) -# Emit zero-initialised globals as plain definitions instead of common -# symbols, the default everywhere but Apple clang. ld64 otherwise derives -# a 32 KB alignment from the size of git's 64 KB packet_buffer and warns -# about reducing it on every macOS link. Use override so a command-line -# CFLAGS (as tools/release-build.sh passes) still keeps the flag. -override CFLAGS += -fno-common - +# Read again here because a sub make inherits only the variables the top level +# Makefile exports, which leaves out the build options this file reads. -include $(CGIT_ROOT)/cgit.conf -# The CGIT_* variables are inherited from the top-level Makefile. - +# CGIT_VERSION and the other CGIT_ values used below come from the top level +# Makefile, which exports them, rather than being defined in this file. $(CGIT_BUILD)/VERSION: force-version @mkdir -p $(CGIT_BUILD)/ @cd $(CGIT_ROOT) && '$(SHELL_PATH_SQ)' $(SRCDIR)/gen-version.sh "$(CGIT_VERSION)" $(BUILDDIR)/VERSION @@ -37,12 +33,12 @@ $(CGIT_BUILD)/VERSION: force-version # The language the cgit sources are written in. Both GCC and Clang default to # this today, so pinning it changes nothing now and stops the meaning of the -# sources drifting when a compiler moves its default on (GCC 15 defaults to -# gnu23). The GNU dialect rather than plain c17 because git's headers use GNU -# extensions, and because dlsym cannot be used through a conforming cast. -# -# Only the cgit objects are held to this. Git keeps whatever its own build -# decides, which on some platforms is a different standard again. +# sources drifting when a compiler moves its default on, as GCC 15 did by +# defaulting to gnu23. The GNU dialect rather than plain c17 because git's +# headers use GNU extensions, and because dlsym cannot be used through a +# conforming cast. Only the cgit objects are held to this, and Git keeps +# whatever its own build decides, which on some platforms is a different +# standard again. CGIT_STD ?= gnu17 # CGIT_CFLAGS is tracked separately so that changing it does not force a @@ -52,8 +48,8 @@ CGIT_CFLAGS += -DCGIT_CONFIG='"$(CGIT_CONFIG)"' CGIT_CFLAGS += -DCGIT_SCRIPT_NAME='"$(CGIT_SCRIPT_NAME)"' CGIT_CFLAGS += -DCGIT_CACHE_ROOT='"$(CACHE_ROOT)"' -# Reaches only the cgit objects, so a caller can tighten the build (CI passes -# -Werror) without holding git's own sources to the same standard. +# Reaches only the cgit objects, so a caller can tighten the build, the way CI +# passes -Werror, without holding git's own sources to the same standard. CGIT_CFLAGS += $(CGIT_EXTRA_CFLAGS) PKG_CONFIG ?= pkg-config @@ -88,12 +84,15 @@ endif endif -# Add -ldl to linker flags on systems that commonly use GNU libc. +# The filters reach libc's write through dlsym, which lives in a library of its +# own on the systems that use GNU libc. ifneq (,$(filter $(uname_S),Linux GNU GNU/kFreeBSD)) CGIT_LIBS += -ldl endif -# glibc 2.1+ offers sendfile which the most common C library on Linux +# The cache sends a stored page straight to stdout with sendfile where it can. +# Only Linux is assumed to offer it, so everywhere else falls back to reading +# and writing the file by hand. ifeq ($(uname_S),Linux) HAVE_LINUX_SENDFILE = YesPlease endif @@ -105,7 +104,7 @@ endif CGIT_OBJ_NAMES += cgit.o CGIT_OBJ_NAMES += cache.o CGIT_OBJ_NAMES += cmd.o -CGIT_OBJ_NAMES += configfile.o +CGIT_OBJ_NAMES += config.o CGIT_OBJ_NAMES += filter.o CGIT_OBJ_NAMES += html.o CGIT_OBJ_NAMES += parsing.o @@ -140,8 +139,8 @@ $(CGIT_VERSION_OBJS): $(CGIT_BUILD)/VERSION $(CGIT_VERSION_OBJS): EXTRA_CPPFLAGS = \ -DCGIT_VERSION='"$(CGIT_VERSION)"' -# Git handles dependencies using ":=" so dependencies in CGIT_OBJS are not -# handled by that and we must handle them ourselves. +# Git builds its list of dependency files with := before this file adds the +# cgit objects, so those are missing from it and have to be picked up here. cgit_dep_files := $(foreach f,$(CGIT_OBJS),$(dir $f).depend/$(notdir $f).d) cgit_dep_files_present := $(wildcard $(cgit_dep_files)) ifneq ($(cgit_dep_files_present),) diff --git a/source/cmd.c b/source/cmd.c index c17d8d9..456a937 100644 --- a/source/cmd.c +++ b/source/cmd.c @@ -1,15 +1,14 @@ -/* cmd.c: the cgit command dispatcher - * - * Copyright (C) 2006-2017 cgit Development Team - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * One thunk per page, each pulling the parts of the request its renderer + * needs out of the global ctx, plus the table that maps a page name onto its + * thunk. Keeping the thunks here lets every renderer take ordinary arguments + * instead of reaching into the request state itself. */ +#include "cache.h" #include "cgit.h" #include "cmd.h" -#include "cache.h" -#include "ui-shared.h" +#include "html.h" #include "ui-atom.h" #include "ui-blame.h" #include "ui-blob.h" @@ -21,50 +20,56 @@ #include "ui-plain.h" #include "ui-refs.h" #include "ui-repolist.h" +#include "ui-shared.h" #include "ui-snapshot.h" #include "ui-stats.h" #include "ui-summary.h" #include "ui-tag.h" #include "ui-tree.h" -#define def_cmd(name, want_repo, want_vpath, is_clone) \ - {#name, name##_fn, want_repo, want_vpath, is_clone} -static void HEAD_fn(void) +static void head_fn(void) { cgit_clone_head(); } -static void atom_fn(void) +static void about_fn(void) { - cgit_print_atom(ctx.qry.head, ctx.qry.path, ctx.cfg.max_atom_items); + char *currenturl, *redirect; + size_t path_info_len; + + if (!ctx.repo) { + cgit_print_site_readme(); + return; + } + + // The about page resolves relative links against its own URL, so it + // only works with a trailing slash. + path_info_len = ctx.env.path_info ? strlen(ctx.env.path_info) : 0; + if (!ctx.qry.path && + (!ctx.qry.url || !*ctx.qry.url || + ctx.qry.url[strlen(ctx.qry.url) - 1] != '/') && + (!path_info_len || ctx.env.path_info[path_info_len - 1] != '/')) { + currenturl = cgit_currenturl(); + redirect = cgit_fmtalloc("%s/", currenturl); + cgit_redirect(redirect, true); + free(currenturl); + free(redirect); + } else if (ctx.repo->readme.nr) { + cgit_print_repo_readme(ctx.qry.path); + } else if (ctx.repo->homepage) { + cgit_redirect(ctx.repo->homepage, false); + } else { + currenturl = cgit_currenturl(); + redirect = cgit_fmtalloc("%s../", currenturl); + cgit_redirect(redirect, false); + free(currenturl); + free(redirect); + } } -static void about_fn(void) +static void atom_fn(void) { - if (ctx.repo) { - size_t path_info_len = ctx.env.path_info ? strlen(ctx.env.path_info) : 0; - if (!ctx.qry.path && - (!ctx.qry.url || !*ctx.qry.url || - ctx.qry.url[strlen(ctx.qry.url) - 1] != '/') && - (!path_info_len || ctx.env.path_info[path_info_len - 1] != '/')) { - char *currenturl = cgit_currenturl(); - char *redirect = cgit_fmtalloc("%s/", currenturl); - cgit_redirect(redirect, true); - free(currenturl); - free(redirect); - } else if (ctx.repo->readme.nr) - cgit_print_repo_readme(ctx.qry.path); - else if (ctx.repo->homepage) - cgit_redirect(ctx.repo->homepage, false); - else { - char *currenturl = cgit_currenturl(); - char *redirect = cgit_fmtalloc("%s../", currenturl); - cgit_redirect(redirect, false); - free(currenturl); - free(redirect); - } - } else - cgit_print_site_readme(); + cgit_print_atom(ctx.qry.head, ctx.qry.path, ctx.cfg.max_atom_items); } static void blame_fn(void) @@ -90,11 +95,6 @@ static void diff_fn(void) cgit_print_diff(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path, 1, 0); } -static void rawdiff_fn(void) -{ - cgit_print_diff(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path, 1, 1); -} - static void info_fn(void) { cgit_clone_info(); @@ -110,8 +110,8 @@ static void log_fn(void) static void ls_cache_fn(void) { - /* The listing exposes the cache path and the URLs other visitors - * requested, so it stays off unless an admin opts in. */ + // The listing names the cache path and the URLs other visitors asked + // for, so it stays off until an administrator opts in. if (!ctx.cfg.enable_cache_list) { cgit_print_error_page(404, "Not found", "Not found"); return; @@ -127,11 +127,6 @@ static void objects_fn(void) cgit_clone_objects(); } -static void repolist_fn(void) -{ - cgit_print_repolist(); -} - static void patch_fn(void) { cgit_print_patch(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path); @@ -142,11 +137,21 @@ static void plain_fn(void) cgit_print_plain(); } +static void rawdiff_fn(void) +{ + cgit_print_diff(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path, 1, 1); +} + static void refs_fn(void) { cgit_print_refs(); } +static void repolist_fn(void) +{ + cgit_print_repolist(); +} + static void snapshot_fn(void) { cgit_print_snapshot(ctx.qry.head, ctx.qry.oid, ctx.qry.path, @@ -176,31 +181,34 @@ static void tree_fn(void) cgit_print_tree(ctx.qry.oid, ctx.qry.path); } -struct cgit_cmd *cgit_get_cmd(void) -{ - static struct cgit_cmd cmds[] = { - def_cmd(HEAD, 1, 0, 1), - def_cmd(atom, 1, 0, 0), - def_cmd(about, 0, 0, 0), - def_cmd(blame, 1, 1, 0), - def_cmd(blob, 1, 0, 0), - def_cmd(commit, 1, 1, 0), - def_cmd(diff, 1, 1, 0), - def_cmd(info, 1, 0, 1), - def_cmd(log, 1, 1, 0), - def_cmd(ls_cache, 0, 0, 0), - def_cmd(objects, 1, 0, 1), - def_cmd(patch, 1, 1, 0), - def_cmd(plain, 1, 0, 0), - def_cmd(rawdiff, 1, 1, 0), - def_cmd(refs, 1, 0, 0), - def_cmd(repolist, 0, 0, 0), - def_cmd(snapshot, 1, 0, 0), - def_cmd(stats, 1, 1, 0), - def_cmd(summary, 1, 0, 0), - def_cmd(tag, 1, 0, 0), - def_cmd(tree, 1, 1, 0), - }; +// The name is what a request spells as p=, so these strings are part of the +// URL space and cannot be renamed. +static const struct cgit_cmd commands[] = { + { .name = "HEAD", .fn = head_fn, .want_repo = 1, .is_clone = 1 }, + { .name = "about", .fn = about_fn }, + { .name = "atom", .fn = atom_fn, .want_repo = 1 }, + { .name = "blame", .fn = blame_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "blob", .fn = blob_fn, .want_repo = 1 }, + { .name = "commit", .fn = commit_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "diff", .fn = diff_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "info", .fn = info_fn, .want_repo = 1, .is_clone = 1 }, + { .name = "log", .fn = log_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "ls_cache", .fn = ls_cache_fn }, + { .name = "objects", .fn = objects_fn, .want_repo = 1, .is_clone = 1 }, + { .name = "patch", .fn = patch_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "plain", .fn = plain_fn, .want_repo = 1 }, + { .name = "rawdiff", .fn = rawdiff_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "refs", .fn = refs_fn, .want_repo = 1 }, + { .name = "repolist", .fn = repolist_fn }, + { .name = "snapshot", .fn = snapshot_fn, .want_repo = 1 }, + { .name = "stats", .fn = stats_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "summary", .fn = summary_fn, .want_repo = 1 }, + { .name = "tag", .fn = tag_fn, .want_repo = 1 }, + { .name = "tree", .fn = tree_fn, .want_repo = 1, .want_vpath = 1 }, +}; + +const struct cgit_cmd *cgit_get_cmd(void) +{ size_t i; if (ctx.qry.page == NULL) { @@ -210,8 +218,8 @@ struct cgit_cmd *cgit_get_cmd(void) ctx.qry.page = "repolist"; } - for (i = 0; i < ARRAY_SIZE(cmds); i++) - if (!strcmp(ctx.qry.page, cmds[i].name)) - return &cmds[i]; + for (i = 0; i < ARRAY_SIZE(commands); i++) + if (!strcmp(ctx.qry.page, commands[i].name)) + return &commands[i]; return NULL; } diff --git a/source/cmd.h b/source/cmd.h index 6249b1d..7299028 100644 --- a/source/cmd.h +++ b/source/cmd.h @@ -1,5 +1,12 @@ -#ifndef CMD_H -#define CMD_H +/* + * The dispatch table that maps the page name in a request to the function + * rendering it. Each entry also records what the request needs resolved + * first, so the caller can look up the repository or the path within the tree + * and reject a bad request once, rather than every page repeating the check. + */ + +#ifndef CGIT_CMD_H +#define CGIT_CMD_H typedef void (*cgit_cmd_fn)(void); @@ -11,6 +18,7 @@ struct cgit_cmd { is_clone:1; }; -extern struct cgit_cmd *cgit_get_cmd(void); +// The entry naming ctx.qry.page, or NULL when no page goes by that name. +extern const struct cgit_cmd *cgit_get_cmd(void); -#endif /* CMD_H */ +#endif // CGIT_CMD_H diff --git a/source/config.c b/source/config.c new file mode 100644 index 0000000..6a5d136 --- /dev/null +++ b/source/config.c @@ -0,0 +1,113 @@ +/* + * The reader for cgit's config files, which are the main cgitrc, any file it + * pulls in with an include line, the cgitrc that sits beside a scanned + * repository, and the cached repolist. A file is a sequence of name=value + * lines, and each pair is handed to a callback that decides what it means, so + * nothing here knows a single key by name. A value runs to the end of its line + * and keeps whatever spacing it has, there is no quoting. A line whose first + * non blank character is a hash or a semicolon is a comment, and blank lines + * are ignored. + */ + +#include "cgit.h" +#include "config.h" + +#define MAX_INCLUDE_NESTING 8 + +/* + * Folds away a carriage return that sits right before a newline so a file + * saved with DOS line endings parses the same as one saved with Unix endings. + */ +static int next_char(FILE *f) +{ + int c = fgetc(f); + + if (c == '\r') { + c = fgetc(f); + if (c != '\n') { + ungetc(c, f); + c = '\r'; + } + } + return c; +} + +static void skip_line(FILE *f) +{ + int c; + + while ((c = next_char(f)) && c != '\n' && c != EOF) + ; +} + +static int next_content_char(FILE *f) +{ + int c = next_char(f); + + for (;;) { + if (c == EOF) + return EOF; + if (c == '#' || c == ';') + skip_line(f); + else if (!isspace(c)) + return c; + c = next_char(f); + } +} + +static int read_entry(FILE *f, struct strbuf *name, struct strbuf *value) +{ + int c; + + strbuf_reset(value); + + for (;;) { + c = next_content_char(f); + if (c == EOF) + return 0; + + strbuf_reset(name); + while (c != '=' && c != '\n' && c != EOF) { + strbuf_addch(name, c); + c = next_char(f); + } + if (c == EOF) + return 0; + // Dropping just the line with no equals sign keeps one typo + // from hiding the rest of the file. + if (c != '=') + continue; + + c = next_char(f); + while (c != '\n' && c != EOF) { + strbuf_addch(value, c); + c = next_char(f); + } + + return 1; + } +} + +int config_file_parse(const char *filename, config_file_value_fn fn) +{ + static int nesting; + struct strbuf name = STRBUF_INIT; + struct strbuf value = STRBUF_INIT; + FILE *f; + + // An include line calls back into here, so a file that includes itself, + // directly or round a longer loop, would recurse until the stack gave + // out. + if (nesting > MAX_INCLUDE_NESTING) + return -1; + if (!(f = fopen(filename, "r"))) + return -1; + nesting++; + while (read_entry(f, &name, &value)) + fn(name.buf, value.buf); + nesting--; + fclose(f); + strbuf_release(&name); + strbuf_release(&value); + return 0; +} diff --git a/source/config.h b/source/config.h new file mode 100644 index 0000000..ab42e2f --- /dev/null +++ b/source/config.h @@ -0,0 +1,16 @@ +/* + * The interface to cgit's config file reader. A caller names a file and passes + * a callback that receives every name and value in it, in the order they + * appear. + */ + +#ifndef CGIT_CONFIG_H +#define CGIT_CONFIG_H + +#include "cgit.h" + +typedef void (*config_file_value_fn)(const char *name, const char *value); + +extern int config_file_parse(const char *filename, config_file_value_fn fn); + +#endif // CGIT_CONFIG_H diff --git a/source/configfile.c b/source/configfile.c deleted file mode 100644 index 9bf2da1..0000000 --- a/source/configfile.c +++ /dev/null @@ -1,100 +0,0 @@ -/* configfile.c: parsing of config files - * - * Copyright (C) 2006-2014 cgit Development Team - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) - */ - -#include -#include "configfile.h" - -static int next_char(FILE *f) -{ - int c = fgetc(f); - if (c == '\r') { - c = fgetc(f); - if (c != '\n') { - ungetc(c, f); - c = '\r'; - } - } - return c; -} - -static void skip_line(FILE *f) -{ - int c; - - while ((c = next_char(f)) && c != '\n' && c != EOF) - ; -} - -static int read_config_line(FILE *f, struct strbuf *name, struct strbuf *value) -{ - int c; - - strbuf_reset(value); - - for (;;) { - c = next_char(f); - - // Skip comments and preceding spaces. - for (;;) { - if (c == EOF) - return 0; - else if (c == '#' || c == ';') - skip_line(f); - else if (!isspace(c)) - break; - c = next_char(f); - } - - // Read variable name. - strbuf_reset(name); - while (c != '=') { - if (c == EOF) - return 0; - // A line without '=' is malformed. Skip it and keep - // reading so a typo does not drop the rest of the file. - if (c == '\n') - break; - strbuf_addch(name, c); - c = next_char(f); - } - if (c != '=') - continue; - - // Read variable value. - c = next_char(f); - while (c != '\n' && c != EOF) { - strbuf_addch(value, c); - c = next_char(f); - } - - return 1; - } -} - -int parse_configfile(const char *filename, configfile_value_fn fn) -{ - static int nesting; - struct strbuf name = STRBUF_INIT; - struct strbuf value = STRBUF_INIT; - FILE *f; - - /* cancel deeply nested include-commands */ - if (nesting > 8) - return -1; - if (!(f = fopen(filename, "r"))) - return -1; - nesting++; - while (read_config_line(f, &name, &value)) - fn(name.buf, value.buf); - nesting--; - fclose(f); - strbuf_release(&name); - strbuf_release(&value); - return 0; -} - diff --git a/source/configfile.h b/source/configfile.h deleted file mode 100644 index af7ca19..0000000 --- a/source/configfile.h +++ /dev/null @@ -1,10 +0,0 @@ -#ifndef CONFIGFILE_H -#define CONFIGFILE_H - -#include "cgit.h" - -typedef void (*configfile_value_fn)(const char *name, const char *value); - -extern int parse_configfile(const char *filename, configfile_value_fn fn); - -#endif /* CONFIGFILE_H */ diff --git a/source/filter.c b/source/filter.c index f195534..3c0c7e4 100644 --- a/source/filter.c +++ b/source/filter.c @@ -1,13 +1,16 @@ -/* filter.c: filter framework functions - * - * Copyright (C) 2006-2014 cgit Development Team - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The two backends behind a cgit filter and the plumbing that hands stdout to + * whichever one is open. An exec filter forks a program and feeds it the page + * through a pipe, while a Lua filter runs a script inside cgit and feeds it + * the page through a write that has been interposed in front of libc's. A + * cgitrc command picks its backend with an exec or lua prefix, and a command + * with no prefix is taken as a program to run. */ #include "cgit.h" +#include "filter.h" #include "html.h" +#include "shared.h" #ifndef NO_LUA #include #include @@ -15,32 +18,10 @@ #include #endif -static inline void reap_filter(struct cgit_filter *filter) -{ - if (filter && filter->cleanup) - filter->cleanup(filter); -} - -void cgit_cleanup_filters(void) -{ - int i; - reap_filter(ctx.cfg.about_filter); - reap_filter(ctx.cfg.commit_filter); - reap_filter(ctx.cfg.source_filter); - reap_filter(ctx.cfg.email_filter); - reap_filter(ctx.cfg.auth_filter); - for (i = 0; i < cgit_repolist.count; ++i) { - reap_filter(cgit_repolist.repos[i].about_filter); - reap_filter(cgit_repolist.repos[i].commit_filter); - reap_filter(cgit_repolist.repos[i].source_filter); - reap_filter(cgit_repolist.repos[i].email_filter); - } -} - static int open_exec_filter(struct cgit_filter *base, va_list ap) { struct cgit_exec_filter *filter = (struct cgit_exec_filter *)base; - int pipe_fh[2]; + int pipefd[2]; int i; for (i = 0; i < filter->base.argument_count; i++) @@ -48,19 +29,21 @@ static int open_exec_filter(struct cgit_filter *base, va_list ap) filter->old_stdout = cgit_die_unless_positive(dup(STDOUT_FILENO), "Unable to duplicate STDOUT"); - cgit_die_unless_zero(pipe(pipe_fh), "Unable to create pipe to subprocess"); - filter->pid = cgit_die_unless_non_negative(fork(), "Unable to create subprocess"); + cgit_die_unless_zero(pipe(pipefd), + "Unable to create pipe to subprocess"); + filter->pid = cgit_die_unless_non_negative(fork(), + "Unable to create subprocess"); if (filter->pid == 0) { - close(pipe_fh[1]); - cgit_die_unless_non_negative(dup2(pipe_fh[0], STDIN_FILENO), + close(pipefd[1]); + cgit_die_unless_non_negative(dup2(pipefd[0], STDIN_FILENO), "Unable to use pipe as STDIN"); execvp(filter->cmd, filter->argv); die_errno("Unable to exec subprocess %s", filter->cmd); } - close(pipe_fh[0]); - cgit_die_unless_non_negative(dup2(pipe_fh[1], STDOUT_FILENO), + close(pipefd[0]); + cgit_die_unless_non_negative(dup2(pipefd[1], STDOUT_FILENO), "Unable to use pipe as STDOUT"); - close(pipe_fh[1]); + close(pipefd[1]); return 0; } @@ -83,10 +66,10 @@ done: for (i = 0; i < filter->base.argument_count; i++) filter->argv[i + 1] = NULL; return WEXITSTATUS(exit_status); - } -static void fprintf_exec_filter(struct cgit_filter *base, FILE *f, const char *prefix) +static void fprintf_exec_filter(struct cgit_filter *base, FILE *f, + const char *prefix) { struct cgit_exec_filter *filter = (struct cgit_exec_filter *)base; fprintf(f, "%sexec:%s\n", prefix, filter->cmd); @@ -107,21 +90,23 @@ static void cleanup_exec_filter(struct cgit_filter *base) static struct cgit_filter *new_exec_filter(const char *cmd, int argument_count) { - struct cgit_exec_filter *f; - int args_size = 0; + struct cgit_exec_filter *filter; + int argv_size; - f = xmalloc(sizeof(*f)); - /* We leave argv for now and assign it below. */ - cgit_exec_filter_init(f, cgit_strdup_first_line(cmd), NULL); - f->base.argument_count = argument_count; - args_size = (2 + argument_count) * sizeof(char *); - f->argv = xmalloc(args_size); - memset(f->argv, 0, args_size); - f->argv[0] = f->cmd; - return &f->base; + filter = xmalloc(sizeof(*filter)); + cgit_exec_filter_init(filter, cgit_strdup_first_line(cmd), NULL); + filter->base.argument_count = argument_count; + // argv is the command, then a slot per argument, then the NULL that + // execvp needs to find the end. + argv_size = (2 + argument_count) * sizeof(char *); + filter->argv = xmalloc(argv_size); + memset(filter->argv, 0, argv_size); + filter->argv[0] = filter->cmd; + return &filter->base; } -void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, char **argv) +void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, + char **argv) { memset(filter, 0, sizeof(*filter)); filter->base.open = open_exec_filter; @@ -130,7 +115,6 @@ void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, char **ar filter->base.cleanup = cleanup_exec_filter; filter->cmd = cmd; filter->argv = argv; - /* The argument count for open_filter is zero by default, unless called from new_filter, above. */ filter->base.argument_count = 0; } @@ -141,8 +125,17 @@ void cgit_init_filters(void) #endif #ifndef NO_LUA +struct lua_filter { + struct cgit_filter base; + char *script_file; + lua_State *lua_state; +}; + +typedef ssize_t (*filter_write_fn)(struct cgit_filter *base, const void *buf, + size_t count); + static ssize_t (*libc_write)(int fd, const void *buf, size_t count); -static ssize_t (*filter_write)(struct cgit_filter *base, const void *buf, size_t count) = NULL; +static filter_write_fn filter_write = NULL; static struct cgit_filter *current_write_filter = NULL; void cgit_init_filters(void) @@ -152,6 +145,10 @@ void cgit_init_filters(void) die("Could not locate libc's write function"); } +/* + * Interposing on write is what lets a Lua filter see output produced by code + * that has no idea a filter is running. + */ ssize_t write(int fd, const void *buf, size_t count) { if (fd != STDOUT_FILENO || !filter_write) @@ -159,13 +156,15 @@ ssize_t write(int fd, const void *buf, size_t count) return filter_write(current_write_filter, buf, count); } -static inline void hook_write(struct cgit_filter *filter, ssize_t (*new_write)(struct cgit_filter *base, const void *buf, size_t count)) +static inline void hook_write(struct cgit_filter *filter, + filter_write_fn write_fn) { - /* We want to avoid buggy nested patterns. */ + // Filters cannot nest, because there is one stdout and one hook, so a + // second one would strand the first. assert(filter_write == NULL); assert(current_write_filter == NULL); current_write_filter = filter; - filter_write = new_write; + filter_write = write_fn; } static inline void unhook_write(void) @@ -176,102 +175,113 @@ static inline void unhook_write(void) current_write_filter = NULL; } -struct lua_filter { - struct cgit_filter base; - char *script_file; - lua_State *lua_state; -}; - -static void error_lua_filter(struct lua_filter *filter) +static void die_lua_error(struct lua_filter *filter) { - die("Lua error in %s: %s", filter->script_file, lua_tostring(filter->lua_state, -1)); + die("Lua error in %s: %s", filter->script_file, + lua_tostring(filter->lua_state, -1)); lua_pop(filter->lua_state, 1); } -static ssize_t write_lua_filter(struct cgit_filter *base, const void *buf, size_t count) +static ssize_t write_lua_filter(struct cgit_filter *base, const void *buf, + size_t count) { struct lua_filter *filter = (struct lua_filter *)base; lua_getglobal(filter->lua_state, "filter_write"); lua_pushlstring(filter->lua_state, buf, count); if (lua_pcall(filter->lua_state, 1, 0, 0)) { - error_lua_filter(filter); + die_lua_error(filter); errno = EIO; return -1; } return count; } -static inline int hook_lua_filter(lua_State *lua_state, void (*fn)(const char *txt)) +/* + * Output a script asks for belongs on the page and not back in its own filter, + * so the hook comes off around the call. The zero returned is Lua's count of + * values pushed for the script, not a success code. + */ +static inline int emit_unfiltered(lua_State *lua_state, + void (*emit)(const char *text)) { - const char *str; - ssize_t (*save_filter_write)(struct cgit_filter *base, const void *buf, size_t count); - struct cgit_filter *save_filter; + const char *text; + filter_write_fn saved_write; + struct cgit_filter *saved_filter; - str = lua_tostring(lua_state, 1); - if (!str) + text = lua_tostring(lua_state, 1); + if (!text) return 0; - save_filter_write = filter_write; - save_filter = current_write_filter; + saved_write = filter_write; + saved_filter = current_write_filter; unhook_write(); - fn(str); - // fn buffers, so empty it while the hook is still off. Re-hooking first - // would send the page's own bytes back into the filter that asked for - // them to be written. + emit(text); + // emit leaves bytes in the html buffer, so empty it while the hook is + // still off. Re-hooking first would send the page's own bytes back into + // the filter. html_flush(); - hook_write(save_filter, save_filter_write); + hook_write(saved_filter, saved_write); return 0; } -static int html_lua_filter(lua_State *lua_state) +static int script_html(lua_State *lua_state) { - return hook_lua_filter(lua_state, html); + return emit_unfiltered(lua_state, html); } -static int html_txt_lua_filter(lua_State *lua_state) +static int script_html_txt(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_txt); + return emit_unfiltered(lua_state, html_txt); } -static int html_attr_lua_filter(lua_State *lua_state) +static int script_html_attr(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_attr); + return emit_unfiltered(lua_state, html_attr); } -static int html_url_path_lua_filter(lua_State *lua_state) +static int script_html_url_path(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_url_path); + return emit_unfiltered(lua_state, html_url_path); } -static int html_url_arg_lua_filter(lua_State *lua_state) +static int script_html_url_arg(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_url_arg); + return emit_unfiltered(lua_state, html_url_arg); } -static int html_include_lua_filter(lua_State *lua_state) +/* + * html_include returns whether it found the file, and calling it through a + * pointer that claims it returns nothing would be undefined, so the result is + * dropped in a wrapper of the right shape. + */ +static void include_and_discard_result(const char *filename) { - return hook_lua_filter(lua_state, (void (*)(const char *))html_include); + html_include(filename); } -static void cleanup_lua_filter(struct cgit_filter *base) +static int script_html_include(lua_State *lua_state) { - struct lua_filter *filter = (struct lua_filter *)base; - - if (!filter->lua_state) - return; - - lua_close(filter->lua_state); - filter->lua_state = NULL; - if (filter->script_file) { - free(filter->script_file); - filter->script_file = NULL; - } + return emit_unfiltered(lua_state, include_and_discard_result); } +static const struct { + const char *name; + lua_CFunction fn; +} script_globals[] = { + { "html", script_html }, + { "html_txt", script_html_txt }, + { "html_attr", script_html_attr }, + { "html_url_path", script_html_url_path }, + { "html_url_arg", script_html_url_arg }, + { "html_include", script_html_include }, +}; + static int init_lua_filter(struct lua_filter *filter) { + size_t i; + if (filter->lua_state) return 0; @@ -280,21 +290,13 @@ static int init_lua_filter(struct lua_filter *filter) luaL_openlibs(filter->lua_state); - lua_pushcfunction(filter->lua_state, html_lua_filter); - lua_setglobal(filter->lua_state, "html"); - lua_pushcfunction(filter->lua_state, html_txt_lua_filter); - lua_setglobal(filter->lua_state, "html_txt"); - lua_pushcfunction(filter->lua_state, html_attr_lua_filter); - lua_setglobal(filter->lua_state, "html_attr"); - lua_pushcfunction(filter->lua_state, html_url_path_lua_filter); - lua_setglobal(filter->lua_state, "html_url_path"); - lua_pushcfunction(filter->lua_state, html_url_arg_lua_filter); - lua_setglobal(filter->lua_state, "html_url_arg"); - lua_pushcfunction(filter->lua_state, html_include_lua_filter); - lua_setglobal(filter->lua_state, "html_include"); + for (i = 0; i < ARRAY_SIZE(script_globals); i++) { + lua_pushcfunction(filter->lua_state, script_globals[i].fn); + lua_setglobal(filter->lua_state, script_globals[i].name); + } if (luaL_dofile(filter->lua_state, filter->script_file)) { - error_lua_filter(filter); + die_lua_error(filter); lua_close(filter->lua_state); filter->lua_state = NULL; return 1; @@ -316,7 +318,7 @@ static int open_lua_filter(struct cgit_filter *base, va_list ap) for (i = 0; i < filter->base.argument_count; ++i) lua_pushstring(filter->lua_state, va_arg(ap, char *)); if (lua_pcall(filter->lua_state, filter->base.argument_count, 0, 0)) { - error_lua_filter(filter); + die_lua_error(filter); return 1; } return 0; @@ -329,7 +331,7 @@ static int close_lua_filter(struct cgit_filter *base) lua_getglobal(filter->lua_state, "filter_close"); if (lua_pcall(filter->lua_state, 0, 1, 0)) { - error_lua_filter(filter); + die_lua_error(filter); ret = -1; } else { ret = lua_tonumber(filter->lua_state, -1); @@ -340,12 +342,27 @@ static int close_lua_filter(struct cgit_filter *base) return ret; } -static void fprintf_lua_filter(struct cgit_filter *base, FILE *f, const char *prefix) +static void fprintf_lua_filter(struct cgit_filter *base, FILE *f, + const char *prefix) { struct lua_filter *filter = (struct lua_filter *)base; fprintf(f, "%slua:%s\n", prefix, filter->script_file); } +static void cleanup_lua_filter(struct cgit_filter *base) +{ + struct lua_filter *filter = (struct lua_filter *)base; + + if (!filter->lua_state) + return; + + lua_close(filter->lua_state); + filter->lua_state = NULL; + if (filter->script_file) { + free(filter->script_file); + filter->script_file = NULL; + } +} static struct cgit_filter *new_lua_filter(const char *cmd, int argument_count) { @@ -362,10 +379,8 @@ static struct cgit_filter *new_lua_filter(const char *cmd, int argument_count) return &filter->base; } - #endif - int cgit_open_filter(struct cgit_filter *filter, ...) { int result; @@ -391,16 +406,37 @@ int cgit_close_filter(struct cgit_filter *filter) return filter->close(filter); } -void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, const char *prefix) +void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, + const char *prefix) { filter->fprintfp(filter, f, prefix); } +static inline void cleanup_filter(struct cgit_filter *filter) +{ + if (filter && filter->cleanup) + filter->cleanup(filter); +} +void cgit_cleanup_filters(void) +{ + int i; + cleanup_filter(ctx.cfg.about_filter); + cleanup_filter(ctx.cfg.commit_filter); + cleanup_filter(ctx.cfg.source_filter); + cleanup_filter(ctx.cfg.email_filter); + cleanup_filter(ctx.cfg.auth_filter); + for (i = 0; i < cgit_repolist.count; ++i) { + cleanup_filter(cgit_repolist.repos[i].about_filter); + cleanup_filter(cgit_repolist.repos[i].commit_filter); + cleanup_filter(cgit_repolist.repos[i].source_filter); + cleanup_filter(cgit_repolist.repos[i].email_filter); + } +} static const struct { const char *prefix; - struct cgit_filter *(*ctor)(const char *cmd, int argument_count); + struct cgit_filter *(*create)(const char *cmd, int argument_count); } filter_specs[] = { { "exec", new_exec_filter }, #ifndef NO_LUA @@ -419,42 +455,38 @@ struct cgit_filter *cgit_new_filter(const char *cmd, filter_type filtertype) return NULL; colon = strchr(cmd, ':'); - len = colon - cmd; - /* - * In case we're running on Windows, don't allow a single letter before - * the colon. - */ + // Measured only once there is a colon to measure against, because + // subtracting from a null pointer is not something C defines. + len = colon ? (size_t)(colon - cmd) : 0; + // A single letter before the colon is a Windows drive letter, not a + // filter prefix. if (len == 1) colon = NULL; switch (filtertype) { - case AUTH: - argument_count = 12; - break; - - case EMAIL: - argument_count = 2; - break; - - case SOURCE: - case ABOUT: - argument_count = 1; - break; - - case COMMIT: - default: - argument_count = 0; - break; + case AUTH: + argument_count = 12; + break; + case EMAIL: + argument_count = 2; + break; + case SOURCE: + case ABOUT: + argument_count = 1; + break; + case COMMIT: + default: + argument_count = 0; + break; } - /* If no prefix is given, exec filter is the default. */ if (!colon) return new_exec_filter(cmd, argument_count); for (i = 0; i < ARRAY_SIZE(filter_specs); i++) { if (len == strlen(filter_specs[i].prefix) && !strncmp(filter_specs[i].prefix, cmd, len)) - return filter_specs[i].ctor(colon + 1, argument_count); + return filter_specs[i].create(colon + 1, argument_count); } die("Invalid filter type: %.*s", (int) len, cmd); diff --git a/source/filter.h b/source/filter.h new file mode 100644 index 0000000..861578c --- /dev/null +++ b/source/filter.h @@ -0,0 +1,54 @@ +/* + * Output filters, the hook that lets a site pass page fragments through an + * external program or an embedded Lua script on their way to the browser. A + * filter takes over stdout between an open and a close call, so at most one + * can be active at a time. The exec and Lua backends both hide behind the + * struct cgit_filter call table, so callers never learn which is in use. + */ + +#ifndef CGIT_FILTER_H +#define CGIT_FILTER_H + +#include +#include + +typedef enum { + ABOUT, COMMIT, SOURCE, EMAIL, AUTH +} filter_type; + +struct cgit_filter { + int (*open)(struct cgit_filter *, va_list ap); + int (*close)(struct cgit_filter *); + void (*fprintfp)(struct cgit_filter *, FILE *, const char *prefix); + void (*cleanup)(struct cgit_filter *); + int argument_count; +}; + +struct cgit_exec_filter { + struct cgit_filter base; + char *cmd; + char **argv; + int old_stdout; + int pid; +}; + +extern struct cgit_filter *cgit_new_filter(const char *cmd, + filter_type filtertype); +extern void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, + char **argv); + +/* + * Redirect stdout into the filter and pass it the variadic arguments its type + * expects, then hand stdout back and return the filter's exit status. + */ +extern int cgit_open_filter(struct cgit_filter *filter, ...); +extern int cgit_close_filter(struct cgit_filter *filter); + +// Describe the filter the way cgitrc would have configured it. +extern void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, + const char *prefix); + +extern void cgit_init_filters(void); +extern void cgit_cleanup_filters(void); + +#endif // CGIT_FILTER_H diff --git a/source/gen-version.sh b/source/gen-version.sh index fb7d459..b28e987 100755 --- a/source/gen-version.sh +++ b/source/gen-version.sh @@ -1,27 +1,27 @@ #!/bin/sh +# Writes a "CGIT_VERSION = ..." line into the output file, which the build +# then includes. Run it from the project root, as gen-version.sh +# [output], so that git describe reports cgit rather than the bundled git +# submodule. The output file defaults to VERSION. -# Writes a "CGIT_VERSION = ..." line to the output file (default: VERSION), -# for the build to include. Run from the project root so `git describe` acts -# on cgit rather than the bundled Git submodule. -# -# Usage: gen-version.sh [output-file] +version=$1 +output=${2:-VERSION} -V=$1 -OUT=${2:-VERSION} - -# Prefer `git describe` when run at the top of the cgit working tree, -# keeping the fallback when it fails, as in a shallow tagless clone. +# Prefer what git describe reports, but only at the top of the cgit working +# tree, and keep the fallback when it has nothing to say, as in a shallow or +# tagless clone. if test "$(git rev-parse --git-dir 2>/dev/null)" = '.git' then - D=$(git describe --abbrev=4 HEAD 2>/dev/null) - test -n "$D" && V=$D + described=$(git describe --abbrev=4 HEAD 2>/dev/null) + test -n "$described" && version=$described fi -new="CGIT_VERSION = $V" -old=$(cat "$OUT" 2>/dev/null) +new="CGIT_VERSION = $version" +old=$(cat "$output" 2>/dev/null) -# Exit if the version is already up to date. +# Leaving the file untouched when nothing changed keeps make from rebuilding +# everything that depends on it. test "$old" = "$new" && exit 0 -echo "$new" > "$OUT" -cat "$OUT" +echo "$new" > "$output" +cat "$output" diff --git a/source/html.c b/source/html.c index 6e023d1..58ccf3b 100644 --- a/source/html.c +++ b/source/html.c @@ -1,20 +1,22 @@ -/* html.c: helper functions for html output - * - * Copyright (C) 2006-2014 cgit Development Team - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The output layer every cgit page is built with, holding the escaping rules + * for page text, attribute values, URL paths and query arguments along with + * the small formatting helpers the rest of the code prints through. What is + * written here is gathered into one buffer and handed to stdout in whole + * blocks, because a page is made of a great many small fragments and a write + * apiece spent more time in the kernel than rendering the page did. cgit + * shares stdout with the filters it runs and with git itself, so that buffer + * has to be emptied wherever another writer takes over. */ #include "cgit.h" #include "html.h" -#include "url.h" + #define HTML_WRITE_BUFSIZE (64 * 1024) -static char html_buf[HTML_WRITE_BUFSIZE]; -static size_t html_buflen; -/* Percent-encoding of each character, except: a-zA-Z0-9!$()*,./:;@- */ -static const char* url_escape_table[256] = { +// The percent encoding for each byte, with NULL marking the bytes a URL may +// carry as themselves. Those are the letters, the digits, and !$()*,-./:;@[]_~ +static const char *url_escape_table[256] = { "%00", "%01", "%02", "%03", "%04", "%05", "%06", "%07", "%08", "%09", "%0a", "%0b", "%0c", "%0d", "%0e", "%0f", "%10", "%11", "%12", "%13", "%14", "%15", "%16", "%17", @@ -49,24 +51,42 @@ static const char* url_escape_table[256] = { "%f8", "%f9", "%fa", "%fb", "%fc", "%fd", "%fe", "%ff" }; +static char out_buf[HTML_WRITE_BUFSIZE]; +static size_t out_len; +static struct strbuf *capture; + +static void write_out(const char *data, size_t size) +{ + // A blob, a snapshot or a patch reaches this with a size well past what + // one write can move onto a pipe, so a short write is ordinary rather + // than an error and has to be resumed instead of reported. + if (write_in_full(STDOUT_FILENO, data, size) < 0) + die_errno("write error on html output"); +} + +/* + * Format into one of a rotating set of static buffers, so that a few results + * can be alive at once, for example as several arguments to one call. The slot + * count has to stay a power of two for the wrap below. + */ char *cgit_fmt(const char *format, ...) { static char buf[8][1024]; - static int bufidx; + static int slot; int len; va_list args; - bufidx++; - bufidx &= 7; + slot++; + slot &= ARRAY_SIZE(buf) - 1; va_start(args, format); - len = vsnprintf(buf[bufidx], sizeof(buf[bufidx]), format, args); + len = vsnprintf(buf[slot], sizeof(buf[slot]), format, args); va_end(args); - if (len < 0 || (size_t)len >= sizeof(buf[bufidx])) { + if (len < 0 || (size_t)len >= sizeof(buf[slot])) { fprintf(stderr, "[html.c] string truncated: %s\n", format); exit(1); } - return buf[bufidx]; + return buf[slot]; } char *cgit_fmtalloc(const char *format, ...) @@ -81,77 +101,44 @@ char *cgit_fmtalloc(const char *format, ...) return strbuf_detach(&sb, NULL); } -/* - * Page output is collected here and written out in whole buffers. A page is - * built from a great many small fragments, and writing each one cost a syscall - * apiece: a tree listing spent more time entering the kernel than generating - * anything. - * - * cgit does not own stdout by itself, so the buffer has to be emptied before - * anything else writes there. Those points are a filter taking over stdout, - * the header block that precedes output produced by git itself, the cache - * slot being measured, and process exit. - */ -static void html_write(const char *data, size_t size) -{ - // A blob, snapshot or patch reaches this with a size well past what one - // write can move onto a pipe, so a short write is ordinary rather than - // an error and has to be resumed instead of reported. - if (write_in_full(STDOUT_FILENO, data, size) < 0) - die_errno("write error on html output"); -} - void html_flush(void) { - size_t len = html_buflen; + size_t len = out_len; if (!len) return; - // Clear the length first: html_write can die, and the error page it - // produces would otherwise try to flush the same bytes again. - html_buflen = 0; - html_write(html_buf, len); + // Clear the length first, because write_out can die and the error page + // it produces would otherwise try to flush the same bytes again. + out_len = 0; + write_out(out_buf, len); } -/* - * While a capture is in effect, page output is collected into the caller's - * buffer instead of being written. The diff view uses this to render a file's - * body at the point it already has the file open, rather than walking the whole - * tree a second time to produce what the diffstat above it has to be printed - * before. - * - * A capture must not span anything that writes to stdout by another route, a - * filter in particular, since that output would escape the capture. - */ -static struct strbuf *html_capture; - void html_capture_begin(struct strbuf *sb) { html_flush(); - html_capture = sb; + capture = sb; } void html_capture_end(void) { - html_capture = NULL; + capture = NULL; } void html_raw(const char *data, size_t size) { - if (html_capture) { - strbuf_add(html_capture, data, size); + if (capture) { + strbuf_add(capture, data, size); return; } if (size >= HTML_WRITE_BUFSIZE) { - // Nothing is gained by copying a blob through the buffer. html_flush(); - html_write(data, size); + write_out(data, size); return; } - if (html_buflen + size > HTML_WRITE_BUFSIZE) + if (out_len + size > HTML_WRITE_BUFSIZE) html_flush(); - memcpy(html_buf + html_buflen, data, size); - html_buflen += size; + memcpy(out_buf + out_len, data, size); + out_len += size; } void html(const char *txt) @@ -162,13 +149,13 @@ void html(const char *txt) void htmlf(const char *format, ...) { va_list args; - struct strbuf buf = STRBUF_INIT; + struct strbuf sb = STRBUF_INIT; va_start(args, format); - strbuf_vaddf(&buf, format, args); + strbuf_vaddf(&sb, format, args); va_end(args); - html(buf.buf); - strbuf_release(&buf); + html(sb.buf); + strbuf_release(&sb); } void html_txtf(const char *format, ...) @@ -182,14 +169,14 @@ void html_txtf(const char *format, ...) void html_vtxtf(const char *format, va_list ap) { - va_list cp; - struct strbuf buf = STRBUF_INIT; - - va_copy(cp, ap); - strbuf_vaddf(&buf, format, cp); - va_end(cp); - html_txt(buf.buf); - strbuf_release(&buf); + va_list copy; + struct strbuf sb = STRBUF_INIT; + + va_copy(copy, ap); + strbuf_vaddf(&sb, format, copy); + va_end(copy); + html_txt(sb.buf); + strbuf_release(&sb); } void html_txt(const char *txt) @@ -200,40 +187,40 @@ void html_txt(const char *txt) ssize_t html_ntxt(const char *txt, size_t len) { - const char *t = txt; - ssize_t slen; + const char *p = txt; + ssize_t left; if (len > SSIZE_MAX) return -1; - slen = (ssize_t) len; - while (t && *t && slen--) { - int c = *t; + left = (ssize_t) len; + while (p && *p && left--) { + int c = *p; if (c == '<' || c == '>' || c == '&') { - html_raw(txt, t - txt); + html_raw(txt, p - txt); if (c == '>') html(">"); else if (c == '<') html("<"); else if (c == '&') html("&"); - txt = t + 1; + txt = p + 1; } - t++; + p++; } - if (t != txt) - html_raw(txt, t - txt); - return slen; + if (p != txt) + html_raw(txt, p - txt); + return left; } -void html_attrf(const char *fmt, ...) +void html_attrf(const char *format, ...) { - va_list ap; + va_list args; struct strbuf sb = STRBUF_INIT; - va_start(ap, fmt); - strbuf_vaddf(&sb, fmt, ap); - va_end(ap); + va_start(args, format); + strbuf_vaddf(&sb, format, args); + va_end(args); html_attr(sb.buf); strbuf_release(&sb); @@ -241,11 +228,11 @@ void html_attrf(const char *fmt, ...) void html_attr(const char *txt) { - const char *t = txt; - while (t && *t) { - int c = *t; + const char *p = txt; + while (p && *p) { + int c = *p; if (c == '<' || c == '>' || c == '\'' || c == '\"' || c == '&') { - html_raw(txt, t - txt); + html_raw(txt, p - txt); if (c == '>') html(">"); else if (c == '<') @@ -256,74 +243,76 @@ void html_attr(const char *txt) html("""); else if (c == '&') html("&"); - txt = t + 1; + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); } void html_url_path(const char *txt) { - const char *t = txt; - while (t && *t) { - unsigned char c = *t; - const char *e = url_escape_table[c]; - if (e && c != '+' && c != '&') { - html_raw(txt, t - txt); - html(e); - txt = t + 1; + const char *p = txt; + while (p && *p) { + unsigned char c = *p; + const char *esc = url_escape_table[c]; + // The table is shared with the query string case, where a plus + // means a space and an ampersand separates parameters. Neither + // carries that meaning in a path. + if (esc && c != '+' && c != '&') { + html_raw(txt, p - txt); + html(esc); + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); } void html_url_arg(const char *txt) { - const char *t = txt; - while (t && *t) { - unsigned char c = *t; - const char *e = url_escape_table[c]; + const char *p = txt; + while (p && *p) { + unsigned char c = *p; + const char *esc = url_escape_table[c]; if (c == ' ') - e = "+"; - if (e) { - html_raw(txt, t - txt); - html(e); - txt = t + 1; + esc = "+"; + if (esc) { + html_raw(txt, p - txt); + html(esc); + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); } void html_header_arg_in_quotes(const char *txt) { - const char *t = txt; - while (t && *t) { - unsigned char c = *t; - const char *e = NULL; + const char *p = txt; + while (p && *p) { + unsigned char c = *p; + const char *esc = NULL; if (c == '\\') - e = "\\\\"; + esc = "\\\\"; else if (c == '\r') - e = "\\r"; + esc = "\\r"; else if (c == '\n') - e = "\\n"; + esc = "\\n"; else if (c == '"') - e = "\\\""; - if (e) { - html_raw(txt, t - txt); - html(e); - txt = t + 1; + esc = "\\\""; + if (esc) { + html_raw(txt, p - txt); + html(esc); + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); - } void html_hidden(const char *name, const char *value) @@ -335,7 +324,8 @@ void html_hidden(const char *name, const char *value) html("'/>"); } -void html_option(const char *value, const char *text, const char *selected_value) +void html_option(const char *value, const char *text, + const char *selected_value) { html("