diff options
| author | Bryce Kwon <bryce@brycekwon.com> | |
|---|---|---|
| committer | Bryce Kwon <bryce@brycekwon.com> | |
| commit | ||
| parent | ||
| tree | ||
| download | ||
Restyle the sources and fix the audit's findings
Diffstat (limited to 'source')
60 files changed, 6213 insertions, 5166 deletions
diff --git a/source/cache.c b/source/cache.c index c6d0427..d6e450a 100644 --- a/source/cache.c +++ b/source/cache.c @@ -1,33 +1,49 @@ -/* cache.c: cache management - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) - * - * - * The cache is just a directory structure where each file is a cache slot, - * and each filename is based on the hash of some key (e.g. the cgit url). - * Each file contains the full key followed by the cached content for that - * key. - * +/* + * The cache that lets a repeated request be answered from disk instead of + * being rendered again. A slot is one file named after the hash of the + * request key, holding that key and then the page it rendered to, and a lock + * file beside it is where a replacement page is written before being renamed + * over the slot. Only the process holding that lock rebuilds a slot, so a + * request arriving while a stale slot is being rebuilt is served the stale + * page, and a request with no usable slot at all renders straight to the + * client without caching anything. */ -#include "cgit.h" #include "cache.h" +#include "cgit.h" #include "html.h" +#include "shared.h" #ifdef HAVE_LINUX_SENDFILE #include <sys/sendfile.h> #endif +// One read of a slot file. The stored key has to be recognised out of a +// single such read, so this also bounds how long a cacheable key can be, see +// key_fits_slot. #define CACHE_BUFSIZE (1024 * 4) -/* Crude implementation of 32-bit FNV-1 hash algorithm, - * see http://www.isthe.com/chongo/tech/comp/fnv/ for details - * about the magic numbers. - */ + +// A slot is named by this many hex digits of the key hash, which is also how +// cache_ls tells slots from the lock files sitting beside them. +#define SLOT_NAME_LEN 8 + +// The 32 bit FNV-1 offset basis and prime. #define FNV_OFFSET 0x811c9dc5 #define FNV_PRIME 0x01000193 +/* + * Cache trouble goes to stderr, which under CGI is the web server's error + * log, so that it cannot land in the middle of the page being written to + * stdout. + */ +__attribute__((format (printf,1,2))) +static void log_error(const char *format, ...) +{ + va_list args; + va_start(args, format); + vfprintf(stderr, format, args); + va_end(args); +} + struct cache_slot { const char *key; size_t keylen; @@ -35,56 +51,56 @@ struct cache_slot { cache_fill_fn fn; int cache_fd; int lock_fd; - int stdout_fd; - const char *cache_name; - const char *lock_name; - int match; - struct stat cache_st; - int bufsize; + int saved_stdout; + const char *path; + const char *lock_path; + int key_matches; + // The slot as it was when it was opened, or the lock file once + // fill_slot has written a page into it. + struct stat st; + // How much of the slot was read into buf, not the size of buf. + int buflen; char buf[CACHE_BUFSIZE]; }; -/* Open an existing cache slot and fill the cache buffer with - * (part of) the content of the cache file. Return 0 on success - * and errno otherwise. - */ static int open_slot(struct cache_slot *slot) { - char *bufz; - ssize_t bufkeylen = -1; + char *nul; + ssize_t keylen = -1; - slot->cache_fd = open(slot->cache_name, O_RDONLY); + slot->cache_fd = open(slot->path, O_RDONLY); if (slot->cache_fd == -1) return errno; - if (fstat(slot->cache_fd, &slot->cache_st)) + if (fstat(slot->cache_fd, &slot->st)) return errno; - slot->bufsize = xread(slot->cache_fd, slot->buf, sizeof(slot->buf)); - if (slot->bufsize < 0) + slot->buflen = xread(slot->cache_fd, slot->buf, sizeof(slot->buf)); + if (slot->buflen < 0) return errno; - bufz = memchr(slot->buf, 0, slot->bufsize); - if (bufz) - bufkeylen = bufz - slot->buf; + nul = memchr(slot->buf, 0, slot->buflen); + if (nul) + keylen = nul - slot->buf; if (slot->key) - slot->match = bufkeylen >= 0 && (size_t)bufkeylen == slot->keylen && - !memcmp(slot->key, slot->buf, bufkeylen + 1); + slot->key_matches = keylen >= 0 && + (size_t)keylen == slot->keylen && + !memcmp(slot->key, slot->buf, keylen + 1); return 0; } -/* A key longer than the buffer above can never be read back, so a slot keyed - * on one would never match and every such request would regenerate its page - * while still writing a slot nothing can use. Those requests skip the cache - * instead. */ +/* + * A key longer than the buffer above can never be read back by open_slot, so + * a slot keyed on one would never match and every such request would + * regenerate its page while still writing a slot nothing can use. + */ static int key_fits_slot(const char *key) { return strlen(key) + 1 <= CACHE_BUFSIZE; } -/* Close the active cache slot */ static int close_slot(struct cache_slot *slot) { int err = 0; @@ -97,7 +113,6 @@ static int close_slot(struct cache_slot *slot) return err; } -/* Print the content of the active cache slot (but skip the key). */ static int print_slot(struct cache_slot *slot) { off_t off; @@ -108,7 +123,7 @@ static int print_slot(struct cache_slot *slot) off = slot->keylen + 1; #ifdef HAVE_LINUX_SENDFILE - size = slot->cache_st.st_size; + size = slot->st.st_size; do { ssize_t ret; @@ -116,7 +131,10 @@ static int print_slot(struct cache_slot *slot) if (ret < 0) { if (errno == EAGAIN || errno == EINTR) continue; - /* Fall back to read/write on EINVAL or ENOSYS */ + // EINVAL and ENOSYS mean this kernel or this pair of + // descriptors cannot do sendfile at all, so fall back + // to the read and write loop rather than fail the + // request. if (errno == EINVAL || errno == ENOSYS) break; return errno; @@ -141,30 +159,41 @@ static int print_slot(struct cache_slot *slot) } while (1); } -/* Check if the slot has expired */ +static int serve_slot(struct cache_slot *slot) +{ + int err; + + err = print_slot(slot); + if (err) + log_error("[cgit] error printing cache %s: %s (%d)\n", + slot->path, + strerror(err), + err); + return err; +} + static int is_expired(struct cache_slot *slot) { if (slot->ttl < 0) return 0; - else - return slot->cache_st.st_mtime + slot->ttl * 60 < time(NULL); + return slot->st.st_mtime + slot->ttl * SECONDS_PER_MINUTE < time(NULL); } -/* Check if the slot has been modified since we opened it. - * NB: If stat() fails, we pretend the file is modified. +/* + * A stat that fails counts as modified, so that the caller leaves alone a file + * it was unable to look at. */ static int is_modified(struct cache_slot *slot) { - struct stat st; + struct stat current; - if (stat(slot->cache_name, &st)) + if (stat(slot->path, ¤t)) return 1; - return (st.st_ino != slot->cache_st.st_ino || - st.st_mtime != slot->cache_st.st_mtime || - st.st_size != slot->cache_st.st_size); + return (current.st_ino != slot->st.st_ino || + current.st_mtime != slot->st.st_mtime || + current.st_size != slot->st.st_size); } -/* Close an open lockfile */ static int close_lock(struct cache_slot *slot) { int err = 0; @@ -177,9 +206,11 @@ static int close_lock(struct cache_slot *slot) return err; } -/* Create a lockfile used to store the generated content for a cache - * slot, and write the slot key + \0 into it. - * Returns 0 on success and errno otherwise. +/* + * The lock file becomes the slot once it is renamed, so it has to open with + * the key the same way a slot does. The lock is taken without blocking, + * because failing to get it is how a second process learns that this slot is + * already being rebuilt, so it returns an errno instead of waiting. */ static int lock_slot(struct cache_slot *slot) { @@ -190,7 +221,7 @@ static int lock_slot(struct cache_slot *slot) .l_len = 0, }; - slot->lock_fd = open(slot->lock_name, O_RDWR | O_CREAT, + slot->lock_fd = open(slot->lock_path, O_RDWR | O_CREAT, S_IRUSR | S_IWUSR); if (slot->lock_fd == -1) return errno; @@ -200,6 +231,8 @@ static int lock_slot(struct cache_slot *slot) slot->lock_fd = -1; return saved_errno; } + // A run that died before its rename leaves the lock file behind, so + // start from empty now that nobody else can be writing it. if (ftruncate(slot->lock_fd, 0) < 0) return errno; if (xwrite(slot->lock_fd, slot->key, slot->keylen + 1) < 0) @@ -207,24 +240,19 @@ static int lock_slot(struct cache_slot *slot) return 0; } -/* Release the current lockfile. If `replace_old_slot` is set the - * lockfile replaces the old cache slot, otherwise the lockfile is - * just deleted. - */ static int unlock_slot(struct cache_slot *slot, int replace_old_slot) { int err; if (replace_old_slot) - err = rename(slot->lock_name, slot->cache_name); + err = rename(slot->lock_path, slot->path); else - err = unlink(slot->lock_name); + err = unlink(slot->lock_path); - /* Restore stdout and close the temporary FD. */ - if (slot->stdout_fd >= 0) { - dup2(slot->stdout_fd, STDOUT_FILENO); - close(slot->stdout_fd); - slot->stdout_fd = -1; + if (slot->saved_stdout >= 0) { + dup2(slot->saved_stdout, STDOUT_FILENO); + close(slot->saved_stdout); + slot->saved_stdout = -1; } if (err) @@ -233,48 +261,84 @@ static int unlock_slot(struct cache_slot *slot, int replace_old_slot) return 0; } -/* Generate the content for the current cache slot by redirecting - * stdout to the lock-fd and invoking the callback function +// Only one slot is ever being filled at a time, so a single pointer is enough +// for cache_abandon_fill to find its way back to the client. +static struct cache_slot *slot_being_filled; + +void cache_abandon_fill(void) +{ + struct cache_slot *slot = slot_being_filled; + + if (!slot) + return; + slot_being_filled = NULL; + + // Emptied while stdout still points at the lock file, so the half + // rendered page goes into the file about to be removed rather than + // reaching the client ahead of whatever is written next. + html_flush(); + + if (slot->saved_stdout >= 0) { + dup2(slot->saved_stdout, STDOUT_FILENO); + close(slot->saved_stdout); + slot->saved_stdout = -1; + } + unlink(slot->lock_path); +} + +/* + * Renders with stdout pointed at the lock file, and on success or failure + * alike it is unlock_slot that gives stdout back. */ static int fill_slot(struct cache_slot *slot) { - /* Preserve stdout */ - slot->stdout_fd = dup(STDOUT_FILENO); - if (slot->stdout_fd == -1) + slot->saved_stdout = dup(STDOUT_FILENO); + if (slot->saved_stdout == -1) return errno; - /* Redirect stdout to lockfile */ if (dup2(slot->lock_fd, STDOUT_FILENO) == -1) return errno; - /* Generate cache content */ + slot_being_filled = slot; slot->fn(); + slot_being_filled = NULL; - /* Make sure any buffered data is flushed to the file */ + // The page is sitting in html.c's buffer and then in stdio's, and all + // of it has to reach the lock file before that file is renamed into + // place. html_flush(); if (fflush(stdout)) return errno; - /* update stat info */ - if (fstat(slot->lock_fd, &slot->cache_st)) + // print_slot takes the length of what it copies from here, and what + // it copies after a fill is the lock file rather than the old slot. + if (fstat(slot->lock_fd, &slot->st)) return errno; return 0; } -unsigned long cache_hash_str(const char *str) +/* + * Giving up is always a valid outcome, because the caller still has the + * expired copy open and can serve that. + */ +static void refresh_slot(struct cache_slot *slot) { - unsigned long h = FNV_OFFSET; - unsigned char *s = (unsigned char *)str; - - if (!s) - return h; + if (lock_slot(slot)) + return; - while (*s) { - h *= FNV_PRIME; - h ^= *s++; + // If another process replaced the slot between open_slot and + // lock_slot, the copy already open is served rather than the newer + // one, which would mean opening that file and comparing the key in it, + // not worth a second descriptor and read on every expiry. + if (is_modified(slot) || fill_slot(slot)) { + unlock_slot(slot, 0); + close_lock(slot); + } else { + close_slot(slot); + unlock_slot(slot, 1); + slot->cache_fd = slot->lock_fd; } - return h; } static int process_slot(struct cache_slot *slot) @@ -282,216 +346,180 @@ static int process_slot(struct cache_slot *slot) int err; err = open_slot(slot); - if (!err && slot->match) { - if (is_expired(slot)) { - if (!lock_slot(slot)) { - /* If the cachefile has been replaced between - * `open_slot` and `lock_slot`, we'll just - * serve the stale content from the original - * cachefile. This way we avoid pruning the - * newly generated slot. The same code-path - * is chosen if fill_slot() fails for some - * reason. - * - * TODO? check if the new slot contains the - * same key as the old one, since we would - * prefer to serve the newest content. - * This will require us to open yet another - * file-descriptor and read and compare the - * key from the new file, so for now we're - * lazy and just ignore the new file. - */ - if (is_modified(slot) || fill_slot(slot)) { - unlock_slot(slot, 0); - close_lock(slot); - } else { - close_slot(slot); - unlock_slot(slot, 1); - slot->cache_fd = slot->lock_fd; - } - } - } - if ((err = print_slot(slot)) != 0) { - cache_log("[cgit] error printing cache %s: %s (%d)\n", - slot->cache_name, - strerror(err), - err); - } + if (!err && slot->key_matches) { + if (is_expired(slot)) + refresh_slot(slot); + err = serve_slot(slot); close_slot(slot); return err; } - /* If the cache slot does not exist (or its key doesn't match the - * current key), lets try to create a new cache slot for this - * request. If this fails (for whatever reason), lets just generate - * the content without caching it and fool the caller to believe - * everything worked out (but print a warning on stdout). - */ - + // If any part of creating a slot fails the page is still rendered + // straight to the client and the caller is told the request succeeded, + // because it did. close_slot(slot); if ((err = lock_slot(slot)) != 0) { - cache_log("[cgit] Unable to lock slot %s: %s (%d)\n", - slot->lock_name, strerror(err), err); + log_error("[cgit] Unable to lock slot %s: %s (%d)\n", + slot->lock_path, strerror(err), err); slot->fn(); return 0; } if ((err = fill_slot(slot)) != 0) { - cache_log("[cgit] Unable to fill slot %s: %s (%d)\n", - slot->lock_name, strerror(err), err); + log_error("[cgit] Unable to fill slot %s: %s (%d)\n", + slot->lock_path, strerror(err), err); unlock_slot(slot, 0); close_lock(slot); slot->fn(); return 0; } - // We've got a valid cache slot in the lock file, which - // is about to replace the old cache slot. But if we - // release the lockfile and then try to open the new cache - // slot, we might get a race condition with a concurrent - // writer for the same cache slot (with a different key). - // Lets avoid such a race by just printing the content of - // the lock file. + + // Opening the slot by name after the rename could land on a file a + // concurrent writer put there for a different key, so what gets + // printed is the descriptor still open on the lock file. slot->cache_fd = slot->lock_fd; unlock_slot(slot, 1); - if ((err = print_slot(slot)) != 0) { - cache_log("[cgit] error printing cache %s: %s (%d)\n", - slot->cache_name, - strerror(err), - err); - } + err = serve_slot(slot); close_slot(slot); return err; } -/* Print cached content to stdout, generate the content if necessary. */ +// The result lives in a static buffer and is only good until the next call. +static char *format_time(const char *format, time_t when) +{ + static char buf[64]; + struct tm tm; + + if (!when) + return NULL; + gmtime_r(&when, &tm); + strftime(buf, sizeof(buf) - 1, format, &tm); + return buf; +} + +/* + * The accumulator is an unsigned long rather than a fixed 32 bit type, so on a + * 64 bit host this is not the published FNV-1 value. All that decides is which + * slot a key lands in, and nothing outside a single build has to agree on the + * answer. + */ +unsigned long cache_hash_str(const char *str) +{ + unsigned long h = FNV_OFFSET; + unsigned char *s = (unsigned char *)str; + + if (!s) + return h; + + while (*s) { + h *= FNV_PRIME; + h ^= *s++; + } + return h; +} + int cache_process(int size, const char *path, const char *key, int ttl, cache_fill_fn fn) { unsigned long hash; int i; - struct strbuf filename = STRBUF_INIT; - struct strbuf lockname = STRBUF_INIT; + struct strbuf slot_path = STRBUF_INIT; + struct strbuf lock_path = STRBUF_INIT; struct cache_slot slot; int result; - /* If the cache is disabled, just generate the content */ if (size <= 0 || ttl == 0) { fn(); return 0; } - /* Verify input, calculate filenames */ if (!path) { - cache_log("[cgit] Cache path not specified, caching is disabled\n"); + log_error("[cgit] Cache path not specified, caching is disabled\n"); fn(); return 0; } if (!key) key = ""; if (!key_fits_slot(key)) { - cache_log("[cgit] Cache key too long for a slot, caching is " + log_error("[cgit] Cache key too long for a slot, caching is " "disabled for this request\n"); fn(); return 0; } hash = cache_hash_str(key) % size; - strbuf_addstr(&filename, path); - strbuf_ensure_end(&filename, '/'); - for (i = 0; i < 8; i++) { - strbuf_addf(&filename, "%x", (unsigned char)(hash & 0xf)); + strbuf_addstr(&slot_path, path); + strbuf_ensure_end(&slot_path, '/'); + for (i = 0; i < SLOT_NAME_LEN; i++) { + strbuf_addf(&slot_path, "%x", (unsigned char)(hash & 0xf)); hash >>= 4; } - strbuf_addbuf(&lockname, &filename); - strbuf_addstr(&lockname, ".lock"); + strbuf_addbuf(&lock_path, &slot_path); + strbuf_addstr(&lock_path, ".lock"); slot.fn = fn; slot.ttl = ttl; - slot.stdout_fd = -1; - slot.cache_name = filename.buf; - slot.lock_name = lockname.buf; + slot.saved_stdout = -1; + slot.path = slot_path.buf; + slot.lock_path = lock_path.buf; slot.key = key; slot.keylen = strlen(key); result = process_slot(&slot); - strbuf_release(&filename); - strbuf_release(&lockname); + strbuf_release(&slot_path); + strbuf_release(&lock_path); return result; } -/* Return a strftime formatted date/time - * NB: the result from this function is to shared memory - */ -static char *sprintftime(const char *format, time_t time) -{ - static char buf[64]; - struct tm tm; - - if (!time) - return NULL; - gmtime_r(&time, &tm); - strftime(buf, sizeof(buf)-1, format, &tm); - return buf; -} - int cache_ls(const char *path) { DIR *dir; struct dirent *ent; int err = 0; + // A NULL key leaves open_slot with nothing to compare against, so + // every slot it opens is simply read. struct cache_slot slot = { NULL }; - struct strbuf fullname = STRBUF_INIT; + struct strbuf slot_path = STRBUF_INIT; size_t prefixlen; char *nul; int keylen; if (!path) { - cache_log("[cgit] cache path not specified\n"); + log_error("[cgit] cache path not specified\n"); return -1; } dir = opendir(path); if (!dir) { err = errno; - cache_log("[cgit] unable to open path %s: %s (%d)\n", + log_error("[cgit] unable to open path %s: %s (%d)\n", path, strerror(err), err); return err; } - strbuf_addstr(&fullname, path); - strbuf_ensure_end(&fullname, '/'); - prefixlen = fullname.len; + strbuf_addstr(&slot_path, path); + strbuf_ensure_end(&slot_path, '/'); + prefixlen = slot_path.len; while ((ent = readdir(dir)) != NULL) { - if (strlen(ent->d_name) != 8) + if (strlen(ent->d_name) != SLOT_NAME_LEN) continue; - strbuf_setlen(&fullname, prefixlen); - strbuf_addstr(&fullname, ent->d_name); - slot.cache_name = fullname.buf; + strbuf_setlen(&slot_path, prefixlen); + strbuf_addstr(&slot_path, ent->d_name); + slot.path = slot_path.buf; if ((err = open_slot(&slot)) != 0) { - cache_log("[cgit] unable to open path %s: %s (%d)\n", - fullname.buf, strerror(err), err); + log_error("[cgit] unable to open path %s: %s (%d)\n", + slot_path.buf, strerror(err), err); continue; } - // The stored key is NUL-terminated within the buffer, but a - // truncated or corrupt slot may not be. Bound the print to - // what was read so %s cannot run off the end. - nul = memchr(slot.buf, 0, slot.bufsize); - keylen = nul ? (int)(nul - slot.buf) : slot.bufsize; + // A truncated or corrupt slot may hold no NUL, so the print is + // bounded by what was read and cannot run off the end. + nul = memchr(slot.buf, 0, slot.buflen); + keylen = nul ? (int)(nul - slot.buf) : slot.buflen; htmlf("%s %s %10"PRIuMAX" %.*s\n", - fullname.buf, - sprintftime("%Y-%m-%d %H:%M:%S", - slot.cache_st.st_mtime), - (uintmax_t)slot.cache_st.st_size, + slot_path.buf, + format_time("%Y-%m-%d %H:%M:%S", + slot.st.st_mtime), + (uintmax_t)slot.st.st_size, keylen, slot.buf); close_slot(&slot); } closedir(dir); - strbuf_release(&fullname); + strbuf_release(&slot_path); return 0; } - -/* Print a message to stdout */ -void cache_log(const char *format, ...) -{ - va_list args; - va_start(args, format); - vfprintf(stderr, format, args); - va_end(args); -} - diff --git a/source/cache.h b/source/cache.h index f8018be..db77999 100644 --- a/source/cache.h +++ b/source/cache.h @@ -1,6 +1,7 @@ /* - * Since git has it's own cache.h which we include, - * lets test on CGIT_CACHE_H to avoid confusion + * The front door to cgit's page cache. A caller hands over a key identifying + * the request and a callback that renders the page, and gets back either the + * copy already on disk or a freshly rendered one. */ #ifndef CGIT_CACHE_H @@ -8,30 +9,30 @@ typedef void (*cache_fill_fn)(void); - -/* Print cached content to stdout, generate the content if necessary. - * - * Parameters - * size max number of cache files - * path directory used to store cache files - * key the key used to lookup cache files - * ttl max cache time in seconds for this key - * fn content generator function for this key - * - * Return value - * 0 indicates success, everything else is an error +/* + * Write the page for key to stdout, taking it from the cache when a fresh + * slot for that key is there and rendering it through fn when it is not. size + * is how many slots the cache may use and path is the directory holding them. + * ttl is how many minutes a slot for this key stays fresh, where a negative + * ttl never expires and a ttl of zero skips the cache for this request. + * Returns 0 when the page was written, and an errno value when it was not. */ extern int cache_process(int size, const char *path, const char *key, int ttl, cache_fill_fn fn); - -/* List info about all cache entries on stdout */ +// Write one line per cache slot to stdout, giving its path, modification +// time, size and key. extern int cache_ls(const char *path); -/* Print a message to stdout */ -__attribute__((format (printf,1,2))) -extern void cache_log(const char *format, ...); +/* + * Give up on the slot being filled, discarding what has been rendered into it + * and putting stdout back on the client. Anything that ends a request part way + * through rendering has to call this before it writes what the visitor should + * see, because until then stdout is the cache file and the visitor is on + * course to receive nothing at all. Does nothing when no slot is being filled. + */ +extern void cache_abandon_fill(void); extern unsigned long cache_hash_str(const char *str); -#endif /* CGIT_CACHE_H */ +#endif // CGIT_CACHE_H diff --git a/source/cgit.c b/source/cgit.c index 7cce3d3..5c11a93 100644 --- a/source/cgit.c +++ b/source/cgit.c This diff is too large to be rendered inline. View it on its own page. diff --git a/source/cgit.h b/source/cgit.h index b05876e..eadb680 100644 --- a/source/cgit.h +++ b/source/cgit.h @@ -1,95 +1,71 @@ +/* + * The vocabulary shared by every part of cgit. It pulls in the git headers + * the renderers are written against, declares the record types they pass + * around, and declares the single global ctx holding the parsed request, the + * effective configuration and the repository being served. Anything that + * belongs to one module is declared in that module's own header instead. + */ + #ifndef CGIT_H #define CGIT_H -#include <stdbool.h> - +// git-compat-util.h sets the feature test macros that git's own headers and +// the system headers are then compiled against, so it has to come first. #include <git-compat-util.h> #include <archive.h> +#include <blame.h> #include <commit.h> +#include <config.h> #include <date.h> -#include <diffcore.h> #include <diff.h> +#include <diffcore.h> #include <environment.h> #include <graph.h> #include <grep.h> #include <hex.h> #include <log-tree.h> #include <notes.h> -#include <object.h> #include <object-name.h> +#include <object.h> #include <odb.h> +#include <packfile.h> #include <path.h> #include <refs.h> #include <revision.h> #include <setup.h> +#include <stdbool.h> #include <string-list.h> #include <strvec.h> #include <tag.h> #include <tree.h> +#include <url.h> #include <utf8.h> +#include <version.h> #include <wrapper.h> #include <xdiff-interface.h> #include <xdiff/xdiff.h> -/* Add isgraph(x) to Git's sane ctype support (see git-compat-util.h) */ -#undef isgraph -#define isgraph(x) (isprint((x)) && !isspace((x))) - - -/* - * Limits used for relative dates - */ -#define TM_MIN 60 -#define TM_HOUR (TM_MIN * 60) -#define TM_DAY (TM_HOUR * 24) -#define TM_WEEK (TM_DAY * 7) -#define TM_YEAR (TM_DAY * 365) -#define TM_MONTH (TM_YEAR / 12.0) - - -/* - * Default encoding - */ +// Every page cgit renders is UTF-8, so commit text in another encoding is +// converted on the way in. #define PAGE_ENCODING "UTF-8" -#define BIT(x) (1U << (x)) - -/* - * Longest search string a request may supply. The filter bar matches its - * query against every item a page lists, so this bounds the work one request - * can ask for. It is not a configuration knob, in the same way as the ofs - * ceiling in querystring_cb. - */ -#define CGIT_MAX_SEARCH_LEN 512 - -typedef void (*configfn)(const char *name, const char *value); -typedef void (*filepair_fn)(struct diff_filepair *pair); -typedef void (*linediff_fn)(char *line, int len); +// The month is a double because a twelfth of a year is not a whole number of +// seconds. +#define SECONDS_PER_MINUTE 60 +#define SECONDS_PER_HOUR (SECONDS_PER_MINUTE * 60) +#define SECONDS_PER_DAY (SECONDS_PER_HOUR * 24) +#define SECONDS_PER_WEEK (SECONDS_PER_DAY * 7) +#define SECONDS_PER_YEAR (SECONDS_PER_DAY * 365) +#define SECONDS_PER_MONTH (SECONDS_PER_YEAR / 12.0) typedef enum { DIFF_UNIFIED, DIFF_SSDIFF, DIFF_STATONLY } diff_type; -typedef enum { - ABOUT, COMMIT, SOURCE, EMAIL, AUTH -} filter_type; - -struct cgit_filter { - int (*open)(struct cgit_filter *, va_list ap); - int (*close)(struct cgit_filter *); - void (*fprintfp)(struct cgit_filter *, FILE *, const char *prefix); - void (*cleanup)(struct cgit_filter *); - int argument_count; -}; - -struct cgit_exec_filter { - struct cgit_filter base; - char *cmd; - char **argv; - int old_stdout; - int pid; -}; +// Filters are only ever held by pointer here, so their layout stays private +// to filter.h. +struct cgit_filter; struct cgit_repo { char *url; @@ -140,11 +116,13 @@ struct commitinfo { struct commit *commit; char *author; char *author_email; - unsigned long author_date; + // git's own width for a commit timestamp, which is wider than a long + // on a 32 bit host and on Windows. + timestamp_t author_date; int author_tz; char *committer; char *committer_email; - unsigned long committer_date; + timestamp_t committer_date; int committer_tz; char *subject; char *msg; @@ -154,7 +132,7 @@ struct commitinfo { struct taginfo { char *tagger; char *tagger_email; - unsigned long tagger_date; + timestamp_t tagger_date; int tagger_tz; char *msg; }; @@ -224,7 +202,7 @@ struct cgit_config { char *script_name; char *section; char *repository_sort; - char *virtual_root; /* Always ends with '/'. */ + char *virtual_root; // Always ends with a slash. char *strict_export; struct date_mode date_mode; int cache_size; @@ -316,7 +294,9 @@ struct cgit_environment { const char *server_port; const char *http_cookie; const char *http_referer; - unsigned int content_length; + // A byte count, so size_t rather than a fixed width type that would + // wrap at four gigabytes on the way in. + size_t content_length; int authenticated; }; @@ -326,97 +306,17 @@ struct cgit_context { struct cgit_config cfg; struct cgit_repo *repo; struct cgit_page page; - // Set once the repository is known to hold no commits, so the page - // header can drop the tabs that cannot render anything. + // Set when the repository holds no commits, so the page header can drop + // the tabs that cannot render anything. int empty_repo; }; -typedef int (*write_archive_fn_t)(const char *, const char *); - -struct cgit_snapshot_format { - const char *suffix; - const char *mimetype; - write_archive_fn_t write_func; -}; - extern const char *cgit_version; extern struct cgit_repolist cgit_repolist; extern struct cgit_context ctx; -extern const struct cgit_snapshot_format cgit_snapshot_formats[]; -extern char *cgit_default_repo_desc; -extern struct cgit_repo *cgit_add_repo(const char *url); -extern struct cgit_repo *cgit_get_repoinfo(const char *url); extern void cgit_repo_config(struct cgit_repo *repo, const char *name, const char *value); -extern int cgit_die_unless_zero(int result, const char *msg); -extern int cgit_die_unless_positive(int result, const char *msg); -extern int cgit_die_unless_non_negative(int result, const char *msg); - -extern char *cgit_trim_end(const char *str, char c); -extern char *cgit_ensure_end(const char *str, char c); - -extern void strbuf_ensure_end(struct strbuf *sb, char c); - -extern void cgit_add_ref(struct reflist *list, struct refinfo *ref); -extern void cgit_free_reflist_inner(struct reflist *list); -extern int cgit_refs_cb(const struct reference *ref, void *cb_data); - -extern void cgit_free_commitinfo(struct commitinfo *info); -extern void cgit_free_taginfo(struct taginfo *info); - -void cgit_diff_tree_cb(struct diff_queue_struct *q, - struct diff_options *options, void *data); - -extern int cgit_diff_files(const struct object_id *old_oid, - const struct object_id *new_oid, - unsigned long *old_size, unsigned long *new_size, - int *binary, int context, int ignorews, - linediff_fn fn); - -extern void cgit_diff_tree(const struct object_id *old_oid, - const struct object_id *new_oid, - filepair_fn fn, const char *prefix, int ignorews); - -extern void cgit_diff_commit(struct commit *commit, filepair_fn fn, - const char *prefix); - -__attribute__((format (printf,1,2))) -extern char *cgit_fmt(const char *format,...); - -__attribute__((format (printf,1,2))) -extern char *cgit_fmtalloc(const char *format,...); - -extern struct commitinfo *cgit_parse_commit(struct commit *commit); -extern struct taginfo *cgit_parse_tag(struct tag *tag); -extern void cgit_parse_url(const char *url); - -extern const char *cgit_repobasename(const char *reponame); - -extern int cgit_parse_snapshots_mask(const char *str); -extern void cgit_parse_date_format(const char *format, struct date_mode *mode); -extern const struct object_id *cgit_snapshot_get_sig(const char *ref, - const struct cgit_snapshot_format *f); -extern unsigned cgit_snapshot_format_bit(const struct cgit_snapshot_format *f); - -extern int cgit_open_filter(struct cgit_filter *filter, ...); -extern int cgit_close_filter(struct cgit_filter *filter); -extern void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, const char *prefix); -extern void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, char **argv); -extern struct cgit_filter *cgit_new_filter(const char *cmd, filter_type filtertype); -extern void cgit_cleanup_filters(void); -extern void cgit_init_filters(void); - -extern void cgit_prepare_repo_env(struct cgit_repo * repo); - -extern int cgit_read_first_line(const char *path, char **buf, size_t *size); - -extern char *cgit_strdup_first_line(const char *txt); - -extern char *cgit_expand_macros(const char *txt); - -extern char *cgit_get_mimetype_for_filename(const char *filename); - -#endif /* CGIT_H */ +#endif // CGIT_H diff --git a/source/cgit.mk b/source/cgit.mk index e0a3312..1cee1fb 100644 --- a/source/cgit.mk +++ b/source/cgit.mk @@ -1,34 +1,30 @@ -# This Makefile runs inside the bundled Git tree (vendor/git) so that it -# can reuse Git's build variables and platform detection. The top-level -# Makefile invokes it as: +# The rules that compile and link cgit itself. They run with the bundled Git +# tree as the working directory so that Git's build variables and platform +# detection can be reused, which is why every path back into the project leads +# through CGIT_ROOT. The top level Makefile invokes it roughly as # # make -C vendor/git -f ../../source/cgit.mk ../../build/cgit # -# so every path back into the project reaches through CGIT_ROOT ("../.."): -# sources come from ../../source and all build output goes to ../../build. +# so sources are read from ../../source and all output is written to +# ../../build. include Makefile -# Locations relative to vendor/git, where this file is run. SRCDIR and -# BUILDDIR name the root-relative subdirs (matching the top-level Makefile and -# used by the version recipe, which cds to the root); CGIT_SRC and CGIT_BUILD -# are their full paths from here. +# SRCDIR and BUILDDIR are named relative to the project root, matching the top +# level Makefile, because the version recipe changes into the root before using +# them. CGIT_SRC and CGIT_BUILD are the same two directories reached from +# vendor/git, where everything else here runs. CGIT_ROOT = ../.. SRCDIR = source BUILDDIR = build CGIT_SRC = $(CGIT_ROOT)/$(SRCDIR) CGIT_BUILD = $(CGIT_ROOT)/$(BUILDDIR) -# Emit zero-initialised globals as plain definitions instead of common -# symbols, the default everywhere but Apple clang. ld64 otherwise derives -# a 32 KB alignment from the size of git's 64 KB packet_buffer and warns -# about reducing it on every macOS link. Use override so a command-line -# CFLAGS (as tools/release-build.sh passes) still keeps the flag. -override CFLAGS += -fno-common - +# Read again here because a sub make inherits only the variables the top level +# Makefile exports, which leaves out the build options this file reads. -include $(CGIT_ROOT)/cgit.conf -# The CGIT_* variables are inherited from the top-level Makefile. - +# CGIT_VERSION and the other CGIT_ values used below come from the top level +# Makefile, which exports them, rather than being defined in this file. $(CGIT_BUILD)/VERSION: force-version @mkdir -p $(CGIT_BUILD)/ @cd $(CGIT_ROOT) && '$(SHELL_PATH_SQ)' $(SRCDIR)/gen-version.sh "$(CGIT_VERSION)" $(BUILDDIR)/VERSION @@ -37,12 +33,12 @@ $(CGIT_BUILD)/VERSION: force-version # The language the cgit sources are written in. Both GCC and Clang default to # this today, so pinning it changes nothing now and stops the meaning of the -# sources drifting when a compiler moves its default on (GCC 15 defaults to -# gnu23). The GNU dialect rather than plain c17 because git's headers use GNU -# extensions, and because dlsym cannot be used through a conforming cast. -# -# Only the cgit objects are held to this. Git keeps whatever its own build -# decides, which on some platforms is a different standard again. +# sources drifting when a compiler moves its default on, as GCC 15 did by +# defaulting to gnu23. The GNU dialect rather than plain c17 because git's +# headers use GNU extensions, and because dlsym cannot be used through a +# conforming cast. Only the cgit objects are held to this, and Git keeps +# whatever its own build decides, which on some platforms is a different +# standard again. CGIT_STD ?= gnu17 # CGIT_CFLAGS is tracked separately so that changing it does not force a @@ -52,8 +48,8 @@ CGIT_CFLAGS += -DCGIT_CONFIG='"$(CGIT_CONFIG)"' CGIT_CFLAGS += -DCGIT_SCRIPT_NAME='"$(CGIT_SCRIPT_NAME)"' CGIT_CFLAGS += -DCGIT_CACHE_ROOT='"$(CACHE_ROOT)"' -# Reaches only the cgit objects, so a caller can tighten the build (CI passes -# -Werror) without holding git's own sources to the same standard. +# Reaches only the cgit objects, so a caller can tighten the build, the way CI +# passes -Werror, without holding git's own sources to the same standard. CGIT_CFLAGS += $(CGIT_EXTRA_CFLAGS) PKG_CONFIG ?= pkg-config @@ -88,12 +84,15 @@ endif endif -# Add -ldl to linker flags on systems that commonly use GNU libc. +# The filters reach libc's write through dlsym, which lives in a library of its +# own on the systems that use GNU libc. ifneq (,$(filter $(uname_S),Linux GNU GNU/kFreeBSD)) CGIT_LIBS += -ldl endif -# glibc 2.1+ offers sendfile which the most common C library on Linux +# The cache sends a stored page straight to stdout with sendfile where it can. +# Only Linux is assumed to offer it, so everywhere else falls back to reading +# and writing the file by hand. ifeq ($(uname_S),Linux) HAVE_LINUX_SENDFILE = YesPlease endif @@ -105,7 +104,7 @@ endif CGIT_OBJ_NAMES += cgit.o CGIT_OBJ_NAMES += cache.o CGIT_OBJ_NAMES += cmd.o -CGIT_OBJ_NAMES += configfile.o +CGIT_OBJ_NAMES += config.o CGIT_OBJ_NAMES += filter.o CGIT_OBJ_NAMES += html.o CGIT_OBJ_NAMES += parsing.o @@ -140,8 +139,8 @@ $(CGIT_VERSION_OBJS): $(CGIT_BUILD)/VERSION $(CGIT_VERSION_OBJS): EXTRA_CPPFLAGS = \ -DCGIT_VERSION='"$(CGIT_VERSION)"' -# Git handles dependencies using ":=" so dependencies in CGIT_OBJS are not -# handled by that and we must handle them ourselves. +# Git builds its list of dependency files with := before this file adds the +# cgit objects, so those are missing from it and have to be picked up here. cgit_dep_files := $(foreach f,$(CGIT_OBJS),$(dir $f).depend/$(notdir $f).d) cgit_dep_files_present := $(wildcard $(cgit_dep_files)) ifneq ($(cgit_dep_files_present),) diff --git a/source/cmd.c b/source/cmd.c index c17d8d9..456a937 100644 --- a/source/cmd.c +++ b/source/cmd.c @@ -1,15 +1,14 @@ -/* cmd.c: the cgit command dispatcher - * - * Copyright (C) 2006-2017 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * One thunk per page, each pulling the parts of the request its renderer + * needs out of the global ctx, plus the table that maps a page name onto its + * thunk. Keeping the thunks here lets every renderer take ordinary arguments + * instead of reaching into the request state itself. */ +#include "cache.h" #include "cgit.h" #include "cmd.h" -#include "cache.h" -#include "ui-shared.h" +#include "html.h" #include "ui-atom.h" #include "ui-blame.h" #include "ui-blob.h" @@ -21,50 +20,56 @@ #include "ui-plain.h" #include "ui-refs.h" #include "ui-repolist.h" +#include "ui-shared.h" #include "ui-snapshot.h" #include "ui-stats.h" #include "ui-summary.h" #include "ui-tag.h" #include "ui-tree.h" -#define def_cmd(name, want_repo, want_vpath, is_clone) \ - {#name, name##_fn, want_repo, want_vpath, is_clone} -static void HEAD_fn(void) +static void head_fn(void) { cgit_clone_head(); } -static void atom_fn(void) +static void about_fn(void) { - cgit_print_atom(ctx.qry.head, ctx.qry.path, ctx.cfg.max_atom_items); + char *currenturl, *redirect; + size_t path_info_len; + + if (!ctx.repo) { + cgit_print_site_readme(); + return; + } + + // The about page resolves relative links against its own URL, so it + // only works with a trailing slash. + path_info_len = ctx.env.path_info ? strlen(ctx.env.path_info) : 0; + if (!ctx.qry.path && + (!ctx.qry.url || !*ctx.qry.url || + ctx.qry.url[strlen(ctx.qry.url) - 1] != '/') && + (!path_info_len || ctx.env.path_info[path_info_len - 1] != '/')) { + currenturl = cgit_currenturl(); + redirect = cgit_fmtalloc("%s/", currenturl); + cgit_redirect(redirect, true); + free(currenturl); + free(redirect); + } else if (ctx.repo->readme.nr) { + cgit_print_repo_readme(ctx.qry.path); + } else if (ctx.repo->homepage) { + cgit_redirect(ctx.repo->homepage, false); + } else { + currenturl = cgit_currenturl(); + redirect = cgit_fmtalloc("%s../", currenturl); + cgit_redirect(redirect, false); + free(currenturl); + free(redirect); + } } -static void about_fn(void) +static void atom_fn(void) { - if (ctx.repo) { - size_t path_info_len = ctx.env.path_info ? strlen(ctx.env.path_info) : 0; - if (!ctx.qry.path && - (!ctx.qry.url || !*ctx.qry.url || - ctx.qry.url[strlen(ctx.qry.url) - 1] != '/') && - (!path_info_len || ctx.env.path_info[path_info_len - 1] != '/')) { - char *currenturl = cgit_currenturl(); - char *redirect = cgit_fmtalloc("%s/", currenturl); - cgit_redirect(redirect, true); - free(currenturl); - free(redirect); - } else if (ctx.repo->readme.nr) - cgit_print_repo_readme(ctx.qry.path); - else if (ctx.repo->homepage) - cgit_redirect(ctx.repo->homepage, false); - else { - char *currenturl = cgit_currenturl(); - char *redirect = cgit_fmtalloc("%s../", currenturl); - cgit_redirect(redirect, false); - free(currenturl); - free(redirect); - } - } else - cgit_print_site_readme(); + cgit_print_atom(ctx.qry.head, ctx.qry.path, ctx.cfg.max_atom_items); } static void blame_fn(void) @@ -90,11 +95,6 @@ static void diff_fn(void) cgit_print_diff(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path, 1, 0); } -static void rawdiff_fn(void) -{ - cgit_print_diff(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path, 1, 1); -} - static void info_fn(void) { cgit_clone_info(); @@ -110,8 +110,8 @@ static void log_fn(void) static void ls_cache_fn(void) { - /* The listing exposes the cache path and the URLs other visitors - * requested, so it stays off unless an admin opts in. */ + // The listing names the cache path and the URLs other visitors asked + // for, so it stays off until an administrator opts in. if (!ctx.cfg.enable_cache_list) { cgit_print_error_page(404, "Not found", "Not found"); return; @@ -127,11 +127,6 @@ static void objects_fn(void) cgit_clone_objects(); } -static void repolist_fn(void) -{ - cgit_print_repolist(); -} - static void patch_fn(void) { cgit_print_patch(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path); @@ -142,11 +137,21 @@ static void plain_fn(void) cgit_print_plain(); } +static void rawdiff_fn(void) +{ + cgit_print_diff(ctx.qry.oid, ctx.qry.oid2, ctx.qry.path, 1, 1); +} + static void refs_fn(void) { cgit_print_refs(); } +static void repolist_fn(void) +{ + cgit_print_repolist(); +} + static void snapshot_fn(void) { cgit_print_snapshot(ctx.qry.head, ctx.qry.oid, ctx.qry.path, @@ -176,31 +181,34 @@ static void tree_fn(void) cgit_print_tree(ctx.qry.oid, ctx.qry.path); } -struct cgit_cmd *cgit_get_cmd(void) +// The name is what a request spells as p=, so these strings are part of the +// URL space and cannot be renamed. +static const struct cgit_cmd commands[] = { + { .name = "HEAD", .fn = head_fn, .want_repo = 1, .is_clone = 1 }, + { .name = "about", .fn = about_fn }, + { .name = "atom", .fn = atom_fn, .want_repo = 1 }, + { .name = "blame", .fn = blame_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "blob", .fn = blob_fn, .want_repo = 1 }, + { .name = "commit", .fn = commit_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "diff", .fn = diff_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "info", .fn = info_fn, .want_repo = 1, .is_clone = 1 }, + { .name = "log", .fn = log_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "ls_cache", .fn = ls_cache_fn }, + { .name = "objects", .fn = objects_fn, .want_repo = 1, .is_clone = 1 }, + { .name = "patch", .fn = patch_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "plain", .fn = plain_fn, .want_repo = 1 }, + { .name = "rawdiff", .fn = rawdiff_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "refs", .fn = refs_fn, .want_repo = 1 }, + { .name = "repolist", .fn = repolist_fn }, + { .name = "snapshot", .fn = snapshot_fn, .want_repo = 1 }, + { .name = "stats", .fn = stats_fn, .want_repo = 1, .want_vpath = 1 }, + { .name = "summary", .fn = summary_fn, .want_repo = 1 }, + { .name = "tag", .fn = tag_fn, .want_repo = 1 }, + { .name = "tree", .fn = tree_fn, .want_repo = 1, .want_vpath = 1 }, +}; + +const struct cgit_cmd *cgit_get_cmd(void) { - static struct cgit_cmd cmds[] = { - def_cmd(HEAD, 1, 0, 1), - def_cmd(atom, 1, 0, 0), - def_cmd(about, 0, 0, 0), - def_cmd(blame, 1, 1, 0), - def_cmd(blob, 1, 0, 0), - def_cmd(commit, 1, 1, 0), - def_cmd(diff, 1, 1, 0), - def_cmd(info, 1, 0, 1), - def_cmd(log, 1, 1, 0), - def_cmd(ls_cache, 0, 0, 0), - def_cmd(objects, 1, 0, 1), - def_cmd(patch, 1, 1, 0), - def_cmd(plain, 1, 0, 0), - def_cmd(rawdiff, 1, 1, 0), - def_cmd(refs, 1, 0, 0), - def_cmd(repolist, 0, 0, 0), - def_cmd(snapshot, 1, 0, 0), - def_cmd(stats, 1, 1, 0), - def_cmd(summary, 1, 0, 0), - def_cmd(tag, 1, 0, 0), - def_cmd(tree, 1, 1, 0), - }; size_t i; if (ctx.qry.page == NULL) { @@ -210,8 +218,8 @@ struct cgit_cmd *cgit_get_cmd(void) ctx.qry.page = "repolist"; } - for (i = 0; i < ARRAY_SIZE(cmds); i++) - if (!strcmp(ctx.qry.page, cmds[i].name)) - return &cmds[i]; + for (i = 0; i < ARRAY_SIZE(commands); i++) + if (!strcmp(ctx.qry.page, commands[i].name)) + return &commands[i]; return NULL; } diff --git a/source/cmd.h b/source/cmd.h index 6249b1d..7299028 100644 --- a/source/cmd.h +++ b/source/cmd.h @@ -1,5 +1,12 @@ -#ifndef CMD_H -#define CMD_H +/* + * The dispatch table that maps the page name in a request to the function + * rendering it. Each entry also records what the request needs resolved + * first, so the caller can look up the repository or the path within the tree + * and reject a bad request once, rather than every page repeating the check. + */ + +#ifndef CGIT_CMD_H +#define CGIT_CMD_H typedef void (*cgit_cmd_fn)(void); @@ -11,6 +18,7 @@ struct cgit_cmd { is_clone:1; }; -extern struct cgit_cmd *cgit_get_cmd(void); +// The entry naming ctx.qry.page, or NULL when no page goes by that name. +extern const struct cgit_cmd *cgit_get_cmd(void); -#endif /* CMD_H */ +#endif // CGIT_CMD_H diff --git a/source/config.c b/source/config.c new file mode 100644 index 0000000..6a5d136 --- /dev/null +++ b/source/config.c @@ -0,0 +1,113 @@ +/* + * The reader for cgit's config files, which are the main cgitrc, any file it + * pulls in with an include line, the cgitrc that sits beside a scanned + * repository, and the cached repolist. A file is a sequence of name=value + * lines, and each pair is handed to a callback that decides what it means, so + * nothing here knows a single key by name. A value runs to the end of its line + * and keeps whatever spacing it has, there is no quoting. A line whose first + * non blank character is a hash or a semicolon is a comment, and blank lines + * are ignored. + */ + +#include "cgit.h" +#include "config.h" + +#define MAX_INCLUDE_NESTING 8 + +/* + * Folds away a carriage return that sits right before a newline so a file + * saved with DOS line endings parses the same as one saved with Unix endings. + */ +static int next_char(FILE *f) +{ + int c = fgetc(f); + + if (c == '\r') { + c = fgetc(f); + if (c != '\n') { + ungetc(c, f); + c = '\r'; + } + } + return c; +} + +static void skip_line(FILE *f) +{ + int c; + + while ((c = next_char(f)) && c != '\n' && c != EOF) + ; +} + +static int next_content_char(FILE *f) +{ + int c = next_char(f); + + for (;;) { + if (c == EOF) + return EOF; + if (c == '#' || c == ';') + skip_line(f); + else if (!isspace(c)) + return c; + c = next_char(f); + } +} + +static int read_entry(FILE *f, struct strbuf *name, struct strbuf *value) +{ + int c; + + strbuf_reset(value); + + for (;;) { + c = next_content_char(f); + if (c == EOF) + return 0; + + strbuf_reset(name); + while (c != '=' && c != '\n' && c != EOF) { + strbuf_addch(name, c); + c = next_char(f); + } + if (c == EOF) + return 0; + // Dropping just the line with no equals sign keeps one typo + // from hiding the rest of the file. + if (c != '=') + continue; + + c = next_char(f); + while (c != '\n' && c != EOF) { + strbuf_addch(value, c); + c = next_char(f); + } + + return 1; + } +} + +int config_file_parse(const char *filename, config_file_value_fn fn) +{ + static int nesting; + struct strbuf name = STRBUF_INIT; + struct strbuf value = STRBUF_INIT; + FILE *f; + + // An include line calls back into here, so a file that includes itself, + // directly or round a longer loop, would recurse until the stack gave + // out. + if (nesting > MAX_INCLUDE_NESTING) + return -1; + if (!(f = fopen(filename, "r"))) + return -1; + nesting++; + while (read_entry(f, &name, &value)) + fn(name.buf, value.buf); + nesting--; + fclose(f); + strbuf_release(&name); + strbuf_release(&value); + return 0; +} diff --git a/source/config.h b/source/config.h new file mode 100644 index 0000000..ab42e2f --- /dev/null +++ b/source/config.h @@ -0,0 +1,16 @@ +/* + * The interface to cgit's config file reader. A caller names a file and passes + * a callback that receives every name and value in it, in the order they + * appear. + */ + +#ifndef CGIT_CONFIG_H +#define CGIT_CONFIG_H + +#include "cgit.h" + +typedef void (*config_file_value_fn)(const char *name, const char *value); + +extern int config_file_parse(const char *filename, config_file_value_fn fn); + +#endif // CGIT_CONFIG_H diff --git a/source/configfile.c b/source/configfile.c deleted file mode 100644 index 9bf2da1..0000000 --- a/source/configfile.c +++ /dev/null @@ -1,100 +0,0 @@ -/* configfile.c: parsing of config files - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) - */ - -#include <git-compat-util.h> -#include "configfile.h" - -static int next_char(FILE *f) -{ - int c = fgetc(f); - if (c == '\r') { - c = fgetc(f); - if (c != '\n') { - ungetc(c, f); - c = '\r'; - } - } - return c; -} - -static void skip_line(FILE *f) -{ - int c; - - while ((c = next_char(f)) && c != '\n' && c != EOF) - ; -} - -static int read_config_line(FILE *f, struct strbuf *name, struct strbuf *value) -{ - int c; - - strbuf_reset(value); - - for (;;) { - c = next_char(f); - - // Skip comments and preceding spaces. - for (;;) { - if (c == EOF) - return 0; - else if (c == '#' || c == ';') - skip_line(f); - else if (!isspace(c)) - break; - c = next_char(f); - } - - // Read variable name. - strbuf_reset(name); - while (c != '=') { - if (c == EOF) - return 0; - // A line without '=' is malformed. Skip it and keep - // reading so a typo does not drop the rest of the file. - if (c == '\n') - break; - strbuf_addch(name, c); - c = next_char(f); - } - if (c != '=') - continue; - - // Read variable value. - c = next_char(f); - while (c != '\n' && c != EOF) { - strbuf_addch(value, c); - c = next_char(f); - } - - return 1; - } -} - -int parse_configfile(const char *filename, configfile_value_fn fn) -{ - static int nesting; - struct strbuf name = STRBUF_INIT; - struct strbuf value = STRBUF_INIT; - FILE *f; - - /* cancel deeply nested include-commands */ - if (nesting > 8) - return -1; - if (!(f = fopen(filename, "r"))) - return -1; - nesting++; - while (read_config_line(f, &name, &value)) - fn(name.buf, value.buf); - nesting--; - fclose(f); - strbuf_release(&name); - strbuf_release(&value); - return 0; -} - diff --git a/source/configfile.h b/source/configfile.h deleted file mode 100644 index af7ca19..0000000 --- a/source/configfile.h +++ /dev/null @@ -1,10 +0,0 @@ -#ifndef CONFIGFILE_H -#define CONFIGFILE_H - -#include "cgit.h" - -typedef void (*configfile_value_fn)(const char *name, const char *value); - -extern int parse_configfile(const char *filename, configfile_value_fn fn); - -#endif /* CONFIGFILE_H */ diff --git a/source/filter.c b/source/filter.c index f195534..3c0c7e4 100644 --- a/source/filter.c +++ b/source/filter.c @@ -1,13 +1,16 @@ -/* filter.c: filter framework functions - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The two backends behind a cgit filter and the plumbing that hands stdout to + * whichever one is open. An exec filter forks a program and feeds it the page + * through a pipe, while a Lua filter runs a script inside cgit and feeds it + * the page through a write that has been interposed in front of libc's. A + * cgitrc command picks its backend with an exec or lua prefix, and a command + * with no prefix is taken as a program to run. */ #include "cgit.h" +#include "filter.h" #include "html.h" +#include "shared.h" #ifndef NO_LUA #include <dlfcn.h> #include <lua.h> @@ -15,32 +18,10 @@ #include <lauxlib.h> #endif -static inline void reap_filter(struct cgit_filter *filter) -{ - if (filter && filter->cleanup) - filter->cleanup(filter); -} - -void cgit_cleanup_filters(void) -{ - int i; - reap_filter(ctx.cfg.about_filter); - reap_filter(ctx.cfg.commit_filter); - reap_filter(ctx.cfg.source_filter); - reap_filter(ctx.cfg.email_filter); - reap_filter(ctx.cfg.auth_filter); - for (i = 0; i < cgit_repolist.count; ++i) { - reap_filter(cgit_repolist.repos[i].about_filter); - reap_filter(cgit_repolist.repos[i].commit_filter); - reap_filter(cgit_repolist.repos[i].source_filter); - reap_filter(cgit_repolist.repos[i].email_filter); - } -} - static int open_exec_filter(struct cgit_filter *base, va_list ap) { struct cgit_exec_filter *filter = (struct cgit_exec_filter *)base; - int pipe_fh[2]; + int pipefd[2]; int i; for (i = 0; i < filter->base.argument_count; i++) @@ -48,19 +29,21 @@ static int open_exec_filter(struct cgit_filter *base, va_list ap) filter->old_stdout = cgit_die_unless_positive(dup(STDOUT_FILENO), "Unable to duplicate STDOUT"); - cgit_die_unless_zero(pipe(pipe_fh), "Unable to create pipe to subprocess"); - filter->pid = cgit_die_unless_non_negative(fork(), "Unable to create subprocess"); + cgit_die_unless_zero(pipe(pipefd), + "Unable to create pipe to subprocess"); + filter->pid = cgit_die_unless_non_negative(fork(), + "Unable to create subprocess"); if (filter->pid == 0) { - close(pipe_fh[1]); - cgit_die_unless_non_negative(dup2(pipe_fh[0], STDIN_FILENO), + close(pipefd[1]); + cgit_die_unless_non_negative(dup2(pipefd[0], STDIN_FILENO), "Unable to use pipe as STDIN"); execvp(filter->cmd, filter->argv); die_errno("Unable to exec subprocess %s", filter->cmd); } - close(pipe_fh[0]); - cgit_die_unless_non_negative(dup2(pipe_fh[1], STDOUT_FILENO), + close(pipefd[0]); + cgit_die_unless_non_negative(dup2(pipefd[1], STDOUT_FILENO), "Unable to use pipe as STDOUT"); - close(pipe_fh[1]); + close(pipefd[1]); return 0; } @@ -83,10 +66,10 @@ done: for (i = 0; i < filter->base.argument_count; i++) filter->argv[i + 1] = NULL; return WEXITSTATUS(exit_status); - } -static void fprintf_exec_filter(struct cgit_filter *base, FILE *f, const char *prefix) +static void fprintf_exec_filter(struct cgit_filter *base, FILE *f, + const char *prefix) { struct cgit_exec_filter *filter = (struct cgit_exec_filter *)base; fprintf(f, "%sexec:%s\n", prefix, filter->cmd); @@ -107,21 +90,23 @@ static void cleanup_exec_filter(struct cgit_filter *base) static struct cgit_filter *new_exec_filter(const char *cmd, int argument_count) { - struct cgit_exec_filter *f; - int args_size = 0; + struct cgit_exec_filter *filter; + int argv_size; - f = xmalloc(sizeof(*f)); - /* We leave argv for now and assign it below. */ - cgit_exec_filter_init(f, cgit_strdup_first_line(cmd), NULL); - f->base.argument_count = argument_count; - args_size = (2 + argument_count) * sizeof(char *); - f->argv = xmalloc(args_size); - memset(f->argv, 0, args_size); - f->argv[0] = f->cmd; - return &f->base; + filter = xmalloc(sizeof(*filter)); + cgit_exec_filter_init(filter, cgit_strdup_first_line(cmd), NULL); + filter->base.argument_count = argument_count; + // argv is the command, then a slot per argument, then the NULL that + // execvp needs to find the end. + argv_size = (2 + argument_count) * sizeof(char *); + filter->argv = xmalloc(argv_size); + memset(filter->argv, 0, argv_size); + filter->argv[0] = filter->cmd; + return &filter->base; } -void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, char **argv) +void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, + char **argv) { memset(filter, 0, sizeof(*filter)); filter->base.open = open_exec_filter; @@ -130,7 +115,6 @@ void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, char **ar filter->base.cleanup = cleanup_exec_filter; filter->cmd = cmd; filter->argv = argv; - /* The argument count for open_filter is zero by default, unless called from new_filter, above. */ filter->base.argument_count = 0; } @@ -141,8 +125,17 @@ void cgit_init_filters(void) #endif #ifndef NO_LUA +struct lua_filter { + struct cgit_filter base; + char *script_file; + lua_State *lua_state; +}; + +typedef ssize_t (*filter_write_fn)(struct cgit_filter *base, const void *buf, + size_t count); + static ssize_t (*libc_write)(int fd, const void *buf, size_t count); -static ssize_t (*filter_write)(struct cgit_filter *base, const void *buf, size_t count) = NULL; +static filter_write_fn filter_write = NULL; static struct cgit_filter *current_write_filter = NULL; void cgit_init_filters(void) @@ -152,6 +145,10 @@ void cgit_init_filters(void) die("Could not locate libc's write function"); } +/* + * Interposing on write is what lets a Lua filter see output produced by code + * that has no idea a filter is running. + */ ssize_t write(int fd, const void *buf, size_t count) { if (fd != STDOUT_FILENO || !filter_write) @@ -159,13 +156,15 @@ ssize_t write(int fd, const void *buf, size_t count) return filter_write(current_write_filter, buf, count); } -static inline void hook_write(struct cgit_filter *filter, ssize_t (*new_write)(struct cgit_filter *base, const void *buf, size_t count)) +static inline void hook_write(struct cgit_filter *filter, + filter_write_fn write_fn) { - /* We want to avoid buggy nested patterns. */ + // Filters cannot nest, because there is one stdout and one hook, so a + // second one would strand the first. assert(filter_write == NULL); assert(current_write_filter == NULL); current_write_filter = filter; - filter_write = new_write; + filter_write = write_fn; } static inline void unhook_write(void) @@ -176,102 +175,113 @@ static inline void unhook_write(void) current_write_filter = NULL; } -struct lua_filter { - struct cgit_filter base; - char *script_file; - lua_State *lua_state; -}; - -static void error_lua_filter(struct lua_filter *filter) +static void die_lua_error(struct lua_filter *filter) { - die("Lua error in %s: %s", filter->script_file, lua_tostring(filter->lua_state, -1)); + die("Lua error in %s: %s", filter->script_file, + lua_tostring(filter->lua_state, -1)); lua_pop(filter->lua_state, 1); } -static ssize_t write_lua_filter(struct cgit_filter *base, const void *buf, size_t count) +static ssize_t write_lua_filter(struct cgit_filter *base, const void *buf, + size_t count) { struct lua_filter *filter = (struct lua_filter *)base; lua_getglobal(filter->lua_state, "filter_write"); lua_pushlstring(filter->lua_state, buf, count); if (lua_pcall(filter->lua_state, 1, 0, 0)) { - error_lua_filter(filter); + die_lua_error(filter); errno = EIO; return -1; } return count; } -static inline int hook_lua_filter(lua_State *lua_state, void (*fn)(const char *txt)) +/* + * Output a script asks for belongs on the page and not back in its own filter, + * so the hook comes off around the call. The zero returned is Lua's count of + * values pushed for the script, not a success code. + */ +static inline int emit_unfiltered(lua_State *lua_state, + void (*emit)(const char *text)) { - const char *str; - ssize_t (*save_filter_write)(struct cgit_filter *base, const void *buf, size_t count); - struct cgit_filter *save_filter; + const char *text; + filter_write_fn saved_write; + struct cgit_filter *saved_filter; - str = lua_tostring(lua_state, 1); - if (!str) + text = lua_tostring(lua_state, 1); + if (!text) return 0; - save_filter_write = filter_write; - save_filter = current_write_filter; + saved_write = filter_write; + saved_filter = current_write_filter; unhook_write(); - fn(str); - // fn buffers, so empty it while the hook is still off. Re-hooking first - // would send the page's own bytes back into the filter that asked for - // them to be written. + emit(text); + // emit leaves bytes in the html buffer, so empty it while the hook is + // still off. Re-hooking first would send the page's own bytes back into + // the filter. html_flush(); - hook_write(save_filter, save_filter_write); + hook_write(saved_filter, saved_write); return 0; } -static int html_lua_filter(lua_State *lua_state) +static int script_html(lua_State *lua_state) { - return hook_lua_filter(lua_state, html); + return emit_unfiltered(lua_state, html); } -static int html_txt_lua_filter(lua_State *lua_state) +static int script_html_txt(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_txt); + return emit_unfiltered(lua_state, html_txt); } -static int html_attr_lua_filter(lua_State *lua_state) +static int script_html_attr(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_attr); + return emit_unfiltered(lua_state, html_attr); } -static int html_url_path_lua_filter(lua_State *lua_state) +static int script_html_url_path(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_url_path); + return emit_unfiltered(lua_state, html_url_path); } -static int html_url_arg_lua_filter(lua_State *lua_state) +static int script_html_url_arg(lua_State *lua_state) { - return hook_lua_filter(lua_state, html_url_arg); + return emit_unfiltered(lua_state, html_url_arg); } -static int html_include_lua_filter(lua_State *lua_state) +/* + * html_include returns whether it found the file, and calling it through a + * pointer that claims it returns nothing would be undefined, so the result is + * dropped in a wrapper of the right shape. + */ +static void include_and_discard_result(const char *filename) { - return hook_lua_filter(lua_state, (void (*)(const char *))html_include); + html_include(filename); } -static void cleanup_lua_filter(struct cgit_filter *base) +static int script_html_include(lua_State *lua_state) { - struct lua_filter *filter = (struct lua_filter *)base; - - if (!filter->lua_state) - return; - - lua_close(filter->lua_state); - filter->lua_state = NULL; - if (filter->script_file) { - free(filter->script_file); - filter->script_file = NULL; - } + return emit_unfiltered(lua_state, include_and_discard_result); } +static const struct { + const char *name; + lua_CFunction fn; +} script_globals[] = { + { "html", script_html }, + { "html_txt", script_html_txt }, + { "html_attr", script_html_attr }, + { "html_url_path", script_html_url_path }, + { "html_url_arg", script_html_url_arg }, + { "html_include", script_html_include }, +}; + static int init_lua_filter(struct lua_filter *filter) { + size_t i; + if (filter->lua_state) return 0; @@ -280,21 +290,13 @@ static int init_lua_filter(struct lua_filter *filter) luaL_openlibs(filter->lua_state); - lua_pushcfunction(filter->lua_state, html_lua_filter); - lua_setglobal(filter->lua_state, "html"); - lua_pushcfunction(filter->lua_state, html_txt_lua_filter); - lua_setglobal(filter->lua_state, "html_txt"); - lua_pushcfunction(filter->lua_state, html_attr_lua_filter); - lua_setglobal(filter->lua_state, "html_attr"); - lua_pushcfunction(filter->lua_state, html_url_path_lua_filter); - lua_setglobal(filter->lua_state, "html_url_path"); - lua_pushcfunction(filter->lua_state, html_url_arg_lua_filter); - lua_setglobal(filter->lua_state, "html_url_arg"); - lua_pushcfunction(filter->lua_state, html_include_lua_filter); - lua_setglobal(filter->lua_state, "html_include"); + for (i = 0; i < ARRAY_SIZE(script_globals); i++) { + lua_pushcfunction(filter->lua_state, script_globals[i].fn); + lua_setglobal(filter->lua_state, script_globals[i].name); + } if (luaL_dofile(filter->lua_state, filter->script_file)) { - error_lua_filter(filter); + die_lua_error(filter); lua_close(filter->lua_state); filter->lua_state = NULL; return 1; @@ -316,7 +318,7 @@ static int open_lua_filter(struct cgit_filter *base, va_list ap) for (i = 0; i < filter->base.argument_count; ++i) lua_pushstring(filter->lua_state, va_arg(ap, char *)); if (lua_pcall(filter->lua_state, filter->base.argument_count, 0, 0)) { - error_lua_filter(filter); + die_lua_error(filter); return 1; } return 0; @@ -329,7 +331,7 @@ static int close_lua_filter(struct cgit_filter *base) lua_getglobal(filter->lua_state, "filter_close"); if (lua_pcall(filter->lua_state, 0, 1, 0)) { - error_lua_filter(filter); + die_lua_error(filter); ret = -1; } else { ret = lua_tonumber(filter->lua_state, -1); @@ -340,12 +342,27 @@ static int close_lua_filter(struct cgit_filter *base) return ret; } -static void fprintf_lua_filter(struct cgit_filter *base, FILE *f, const char *prefix) +static void fprintf_lua_filter(struct cgit_filter *base, FILE *f, + const char *prefix) { struct lua_filter *filter = (struct lua_filter *)base; fprintf(f, "%slua:%s\n", prefix, filter->script_file); } +static void cleanup_lua_filter(struct cgit_filter *base) +{ + struct lua_filter *filter = (struct lua_filter *)base; + + if (!filter->lua_state) + return; + + lua_close(filter->lua_state); + filter->lua_state = NULL; + if (filter->script_file) { + free(filter->script_file); + filter->script_file = NULL; + } +} static struct cgit_filter *new_lua_filter(const char *cmd, int argument_count) { @@ -362,10 +379,8 @@ static struct cgit_filter *new_lua_filter(const char *cmd, int argument_count) return &filter->base; } - #endif - int cgit_open_filter(struct cgit_filter *filter, ...) { int result; @@ -391,16 +406,37 @@ int cgit_close_filter(struct cgit_filter *filter) return filter->close(filter); } -void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, const char *prefix) +void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, + const char *prefix) { filter->fprintfp(filter, f, prefix); } +static inline void cleanup_filter(struct cgit_filter *filter) +{ + if (filter && filter->cleanup) + filter->cleanup(filter); +} +void cgit_cleanup_filters(void) +{ + int i; + cleanup_filter(ctx.cfg.about_filter); + cleanup_filter(ctx.cfg.commit_filter); + cleanup_filter(ctx.cfg.source_filter); + cleanup_filter(ctx.cfg.email_filter); + cleanup_filter(ctx.cfg.auth_filter); + for (i = 0; i < cgit_repolist.count; ++i) { + cleanup_filter(cgit_repolist.repos[i].about_filter); + cleanup_filter(cgit_repolist.repos[i].commit_filter); + cleanup_filter(cgit_repolist.repos[i].source_filter); + cleanup_filter(cgit_repolist.repos[i].email_filter); + } +} static const struct { const char *prefix; - struct cgit_filter *(*ctor)(const char *cmd, int argument_count); + struct cgit_filter *(*create)(const char *cmd, int argument_count); } filter_specs[] = { { "exec", new_exec_filter }, #ifndef NO_LUA @@ -419,42 +455,38 @@ struct cgit_filter *cgit_new_filter(const char *cmd, filter_type filtertype) return NULL; colon = strchr(cmd, ':'); - len = colon - cmd; - /* - * In case we're running on Windows, don't allow a single letter before - * the colon. - */ + // Measured only once there is a colon to measure against, because + // subtracting from a null pointer is not something C defines. + len = colon ? (size_t)(colon - cmd) : 0; + // A single letter before the colon is a Windows drive letter, not a + // filter prefix. if (len == 1) colon = NULL; switch (filtertype) { - case AUTH: - argument_count = 12; - break; - - case EMAIL: - argument_count = 2; - break; - - case SOURCE: - case ABOUT: - argument_count = 1; - break; - - case COMMIT: - default: - argument_count = 0; - break; + case AUTH: + argument_count = 12; + break; + case EMAIL: + argument_count = 2; + break; + case SOURCE: + case ABOUT: + argument_count = 1; + break; + case COMMIT: + default: + argument_count = 0; + break; } - /* If no prefix is given, exec filter is the default. */ if (!colon) return new_exec_filter(cmd, argument_count); for (i = 0; i < ARRAY_SIZE(filter_specs); i++) { if (len == strlen(filter_specs[i].prefix) && !strncmp(filter_specs[i].prefix, cmd, len)) - return filter_specs[i].ctor(colon + 1, argument_count); + return filter_specs[i].create(colon + 1, argument_count); } die("Invalid filter type: %.*s", (int) len, cmd); diff --git a/source/filter.h b/source/filter.h new file mode 100644 index 0000000..861578c --- /dev/null +++ b/source/filter.h @@ -0,0 +1,54 @@ +/* + * Output filters, the hook that lets a site pass page fragments through an + * external program or an embedded Lua script on their way to the browser. A + * filter takes over stdout between an open and a close call, so at most one + * can be active at a time. The exec and Lua backends both hide behind the + * struct cgit_filter call table, so callers never learn which is in use. + */ + +#ifndef CGIT_FILTER_H +#define CGIT_FILTER_H + +#include <stdarg.h> +#include <stdio.h> + +typedef enum { + ABOUT, COMMIT, SOURCE, EMAIL, AUTH +} filter_type; + +struct cgit_filter { + int (*open)(struct cgit_filter *, va_list ap); + int (*close)(struct cgit_filter *); + void (*fprintfp)(struct cgit_filter *, FILE *, const char *prefix); + void (*cleanup)(struct cgit_filter *); + int argument_count; +}; + +struct cgit_exec_filter { + struct cgit_filter base; + char *cmd; + char **argv; + int old_stdout; + int pid; +}; + +extern struct cgit_filter *cgit_new_filter(const char *cmd, + filter_type filtertype); +extern void cgit_exec_filter_init(struct cgit_exec_filter *filter, char *cmd, + char **argv); + +/* + * Redirect stdout into the filter and pass it the variadic arguments its type + * expects, then hand stdout back and return the filter's exit status. + */ +extern int cgit_open_filter(struct cgit_filter *filter, ...); +extern int cgit_close_filter(struct cgit_filter *filter); + +// Describe the filter the way cgitrc would have configured it. +extern void cgit_fprintf_filter(struct cgit_filter *filter, FILE *f, + const char *prefix); + +extern void cgit_init_filters(void); +extern void cgit_cleanup_filters(void); + +#endif // CGIT_FILTER_H diff --git a/source/gen-version.sh b/source/gen-version.sh index fb7d459..b28e987 100755 --- a/source/gen-version.sh +++ b/source/gen-version.sh @@ -1,27 +1,27 @@ #!/bin/sh +# Writes a "CGIT_VERSION = ..." line into the output file, which the build +# then includes. Run it from the project root, as gen-version.sh <fallback> +# [output], so that git describe reports cgit rather than the bundled git +# submodule. The output file defaults to VERSION. -# Writes a "CGIT_VERSION = ..." line to the output file (default: VERSION), -# for the build to include. Run from the project root so `git describe` acts -# on cgit rather than the bundled Git submodule. -# -# Usage: gen-version.sh <fallback-version> [output-file] +version=$1 +output=${2:-VERSION} -V=$1 -OUT=${2:-VERSION} - -# Prefer `git describe` when run at the top of the cgit working tree, -# keeping the fallback when it fails, as in a shallow tagless clone. +# Prefer what git describe reports, but only at the top of the cgit working +# tree, and keep the fallback when it has nothing to say, as in a shallow or +# tagless clone. if test "$(git rev-parse --git-dir 2>/dev/null)" = '.git' then - D=$(git describe --abbrev=4 HEAD 2>/dev/null) - test -n "$D" && V=$D + described=$(git describe --abbrev=4 HEAD 2>/dev/null) + test -n "$described" && version=$described fi -new="CGIT_VERSION = $V" -old=$(cat "$OUT" 2>/dev/null) +new="CGIT_VERSION = $version" +old=$(cat "$output" 2>/dev/null) -# Exit if the version is already up to date. +# Leaving the file untouched when nothing changed keeps make from rebuilding +# everything that depends on it. test "$old" = "$new" && exit 0 -echo "$new" > "$OUT" -cat "$OUT" +echo "$new" > "$output" +cat "$output" diff --git a/source/html.c b/source/html.c index 6e023d1..58ccf3b 100644 --- a/source/html.c +++ b/source/html.c @@ -1,20 +1,22 @@ -/* html.c: helper functions for html output - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The output layer every cgit page is built with, holding the escaping rules + * for page text, attribute values, URL paths and query arguments along with + * the small formatting helpers the rest of the code prints through. What is + * written here is gathered into one buffer and handed to stdout in whole + * blocks, because a page is made of a great many small fragments and a write + * apiece spent more time in the kernel than rendering the page did. cgit + * shares stdout with the filters it runs and with git itself, so that buffer + * has to be emptied wherever another writer takes over. */ #include "cgit.h" #include "html.h" -#include "url.h" + #define HTML_WRITE_BUFSIZE (64 * 1024) -static char html_buf[HTML_WRITE_BUFSIZE]; -static size_t html_buflen; -/* Percent-encoding of each character, except: a-zA-Z0-9!$()*,./:;@- */ -static const char* url_escape_table[256] = { +// The percent encoding for each byte, with NULL marking the bytes a URL may +// carry as themselves. Those are the letters, the digits, and !$()*,-./:;@[]_~ +static const char *url_escape_table[256] = { "%00", "%01", "%02", "%03", "%04", "%05", "%06", "%07", "%08", "%09", "%0a", "%0b", "%0c", "%0d", "%0e", "%0f", "%10", "%11", "%12", "%13", "%14", "%15", "%16", "%17", @@ -49,24 +51,42 @@ static const char* url_escape_table[256] = { "%f8", "%f9", "%fa", "%fb", "%fc", "%fd", "%fe", "%ff" }; +static char out_buf[HTML_WRITE_BUFSIZE]; +static size_t out_len; +static struct strbuf *capture; + +static void write_out(const char *data, size_t size) +{ + // A blob, a snapshot or a patch reaches this with a size well past what + // one write can move onto a pipe, so a short write is ordinary rather + // than an error and has to be resumed instead of reported. + if (write_in_full(STDOUT_FILENO, data, size) < 0) + die_errno("write error on html output"); +} + +/* + * Format into one of a rotating set of static buffers, so that a few results + * can be alive at once, for example as several arguments to one call. The slot + * count has to stay a power of two for the wrap below. + */ char *cgit_fmt(const char *format, ...) { static char buf[8][1024]; - static int bufidx; + static int slot; int len; va_list args; - bufidx++; - bufidx &= 7; + slot++; + slot &= ARRAY_SIZE(buf) - 1; va_start(args, format); - len = vsnprintf(buf[bufidx], sizeof(buf[bufidx]), format, args); + len = vsnprintf(buf[slot], sizeof(buf[slot]), format, args); va_end(args); - if (len < 0 || (size_t)len >= sizeof(buf[bufidx])) { + if (len < 0 || (size_t)len >= sizeof(buf[slot])) { fprintf(stderr, "[html.c] string truncated: %s\n", format); exit(1); } - return buf[bufidx]; + return buf[slot]; } char *cgit_fmtalloc(const char *format, ...) @@ -81,77 +101,44 @@ char *cgit_fmtalloc(const char *format, ...) return strbuf_detach(&sb, NULL); } -/* - * Page output is collected here and written out in whole buffers. A page is - * built from a great many small fragments, and writing each one cost a syscall - * apiece: a tree listing spent more time entering the kernel than generating - * anything. - * - * cgit does not own stdout by itself, so the buffer has to be emptied before - * anything else writes there. Those points are a filter taking over stdout, - * the header block that precedes output produced by git itself, the cache - * slot being measured, and process exit. - */ -static void html_write(const char *data, size_t size) -{ - // A blob, snapshot or patch reaches this with a size well past what one - // write can move onto a pipe, so a short write is ordinary rather than - // an error and has to be resumed instead of reported. - if (write_in_full(STDOUT_FILENO, data, size) < 0) - die_errno("write error on html output"); -} - void html_flush(void) { - size_t len = html_buflen; + size_t len = out_len; if (!len) return; - // Clear the length first: html_write can die, and the error page it - // produces would otherwise try to flush the same bytes again. - html_buflen = 0; - html_write(html_buf, len); + // Clear the length first, because write_out can die and the error page + // it produces would otherwise try to flush the same bytes again. + out_len = 0; + write_out(out_buf, len); } -/* - * While a capture is in effect, page output is collected into the caller's - * buffer instead of being written. The diff view uses this to render a file's - * body at the point it already has the file open, rather than walking the whole - * tree a second time to produce what the diffstat above it has to be printed - * before. - * - * A capture must not span anything that writes to stdout by another route, a - * filter in particular, since that output would escape the capture. - */ -static struct strbuf *html_capture; - void html_capture_begin(struct strbuf *sb) { html_flush(); - html_capture = sb; + capture = sb; } void html_capture_end(void) { - html_capture = NULL; + capture = NULL; } void html_raw(const char *data, size_t size) { - if (html_capture) { - strbuf_add(html_capture, data, size); + if (capture) { + strbuf_add(capture, data, size); return; } if (size >= HTML_WRITE_BUFSIZE) { - // Nothing is gained by copying a blob through the buffer. html_flush(); - html_write(data, size); + write_out(data, size); return; } - if (html_buflen + size > HTML_WRITE_BUFSIZE) + if (out_len + size > HTML_WRITE_BUFSIZE) html_flush(); - memcpy(html_buf + html_buflen, data, size); - html_buflen += size; + memcpy(out_buf + out_len, data, size); + out_len += size; } void html(const char *txt) @@ -162,13 +149,13 @@ void html(const char *txt) void htmlf(const char *format, ...) { va_list args; - struct strbuf buf = STRBUF_INIT; + struct strbuf sb = STRBUF_INIT; va_start(args, format); - strbuf_vaddf(&buf, format, args); + strbuf_vaddf(&sb, format, args); va_end(args); - html(buf.buf); - strbuf_release(&buf); + html(sb.buf); + strbuf_release(&sb); } void html_txtf(const char *format, ...) @@ -182,14 +169,14 @@ void html_txtf(const char *format, ...) void html_vtxtf(const char *format, va_list ap) { - va_list cp; - struct strbuf buf = STRBUF_INIT; + va_list copy; + struct strbuf sb = STRBUF_INIT; - va_copy(cp, ap); - strbuf_vaddf(&buf, format, cp); - va_end(cp); - html_txt(buf.buf); - strbuf_release(&buf); + va_copy(copy, ap); + strbuf_vaddf(&sb, format, copy); + va_end(copy); + html_txt(sb.buf); + strbuf_release(&sb); } void html_txt(const char *txt) @@ -200,40 +187,40 @@ void html_txt(const char *txt) ssize_t html_ntxt(const char *txt, size_t len) { - const char *t = txt; - ssize_t slen; + const char *p = txt; + ssize_t left; if (len > SSIZE_MAX) return -1; - slen = (ssize_t) len; - while (t && *t && slen--) { - int c = *t; + left = (ssize_t) len; + while (p && *p && left--) { + int c = *p; if (c == '<' || c == '>' || c == '&') { - html_raw(txt, t - txt); + html_raw(txt, p - txt); if (c == '>') html(">"); else if (c == '<') html("<"); else if (c == '&') html("&"); - txt = t + 1; + txt = p + 1; } - t++; + p++; } - if (t != txt) - html_raw(txt, t - txt); - return slen; + if (p != txt) + html_raw(txt, p - txt); + return left; } -void html_attrf(const char *fmt, ...) +void html_attrf(const char *format, ...) { - va_list ap; + va_list args; struct strbuf sb = STRBUF_INIT; - va_start(ap, fmt); - strbuf_vaddf(&sb, fmt, ap); - va_end(ap); + va_start(args, format); + strbuf_vaddf(&sb, format, args); + va_end(args); html_attr(sb.buf); strbuf_release(&sb); @@ -241,11 +228,11 @@ void html_attrf(const char *fmt, ...) void html_attr(const char *txt) { - const char *t = txt; - while (t && *t) { - int c = *t; + const char *p = txt; + while (p && *p) { + int c = *p; if (c == '<' || c == '>' || c == '\'' || c == '\"' || c == '&') { - html_raw(txt, t - txt); + html_raw(txt, p - txt); if (c == '>') html(">"); else if (c == '<') @@ -256,74 +243,76 @@ void html_attr(const char *txt) html("""); else if (c == '&') html("&"); - txt = t + 1; + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); } void html_url_path(const char *txt) { - const char *t = txt; - while (t && *t) { - unsigned char c = *t; - const char *e = url_escape_table[c]; - if (e && c != '+' && c != '&') { - html_raw(txt, t - txt); - html(e); - txt = t + 1; + const char *p = txt; + while (p && *p) { + unsigned char c = *p; + const char *esc = url_escape_table[c]; + // The table is shared with the query string case, where a plus + // means a space and an ampersand separates parameters. Neither + // carries that meaning in a path. + if (esc && c != '+' && c != '&') { + html_raw(txt, p - txt); + html(esc); + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); } void html_url_arg(const char *txt) { - const char *t = txt; - while (t && *t) { - unsigned char c = *t; - const char *e = url_escape_table[c]; + const char *p = txt; + while (p && *p) { + unsigned char c = *p; + const char *esc = url_escape_table[c]; if (c == ' ') - e = "+"; - if (e) { - html_raw(txt, t - txt); - html(e); - txt = t + 1; + esc = "+"; + if (esc) { + html_raw(txt, p - txt); + html(esc); + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); } void html_header_arg_in_quotes(const char *txt) { - const char *t = txt; - while (t && *t) { - unsigned char c = *t; - const char *e = NULL; + const char *p = txt; + while (p && *p) { + unsigned char c = *p; + const char *esc = NULL; if (c == '\\') - e = "\\\\"; + esc = "\\\\"; else if (c == '\r') - e = "\\r"; + esc = "\\r"; else if (c == '\n') - e = "\\n"; + esc = "\\n"; else if (c == '"') - e = "\\\""; - if (e) { - html_raw(txt, t - txt); - html(e); - txt = t + 1; + esc = "\\\""; + if (esc) { + html_raw(txt, p - txt); + html(esc); + txt = p + 1; } - t++; + p++; } - if (t != txt) + if (p != txt) html(txt); - } void html_hidden(const char *name, const char *value) @@ -335,7 +324,8 @@ void html_hidden(const char *name, const char *value) html("'/>"); } -void html_option(const char *value, const char *text, const char *selected_value) +void html_option(const char *value, const char *text, + const char *selected_value) { html("<option value='"); html_attr(value); @@ -375,6 +365,10 @@ void html_link_close(void) html("</a>"); } +/* + * Render one permission triplet, so a caller prints a whole mode by passing it + * shifted right by six, then by three, then unshifted. + */ void html_fileperm(unsigned short mode) { htmlf("%c%c%c", (mode & 4 ? 'r' : '-'), @@ -392,20 +386,21 @@ int html_include(const char *filename) filename, strerror(errno), errno); return -1; } - while ((len = fread(buf, 1, 4096, f)) > 0) + while ((len = fread(buf, 1, sizeof(buf), f)) > 0) html_raw(buf, len); fclose(f); return 0; } -void http_parse_querystring(const char *txt, void (*fn)(const char *name, const char *value)) +void http_parse_querystring(const char *txt, + void (*fn)(const char *name, const char *value)) { - const char *t = txt; + const char *p = txt; - while (t && *t) { - char *name = url_decode_parameter_name(&t); + while (p && *p) { + char *name = url_decode_parameter_name(&p); if (*name) { - char *value = url_decode_parameter_value(&t); + char *value = url_decode_parameter_value(&p); fn(name, value); free(value); } diff --git a/source/html.h b/source/html.h index 0818ab5..ecc7141 100644 --- a/source/html.h +++ b/source/html.h @@ -1,53 +1,83 @@ -#ifndef HTML_H -#define HTML_H +/* + * The printing side of cgit, which every page is written through. Text is + * handed in as plain strings and the call a renderer picks decides how it is + * escaped for where it lands, whether that is page text, an attribute value, a + * URL path, a query argument or a quoted HTTP header. Output is buffered + * rather than written as it is produced, so the flush and capture calls below + * are part of the contract for anything that reaches stdout by another route. + */ + +#ifndef CGIT_HTML_H +#define CGIT_HTML_H #include "cgit.h" +// How much of a generated run to build before handing it to html_raw. Growing +// past this gains nothing, since html_raw gathers what it is given into a +// buffer of its own, and a run built whole would instead be sized by the file +// it came from, which the blob limits do not bound. +#define HTML_BATCH (64 * 1024) + extern void html_raw(const char *txt, size_t size); extern void html(const char *txt); -/* How much of a generated run to build before handing it to html_raw. Batching - * is what removes the write per line, and html_raw already gathers what it is - * given into one buffer, so nothing is gained by growing past this. A run built - * whole would instead be sized by the file it was generated from, which the - * blob limits do not bound. */ -#define HTML_BATCH (64 * 1024) - -/* Write out whatever page output is still buffered. Call before anything - * other than html_raw writes to stdout. */ +/* + * Write out whatever page output is still buffered. Call this before anything + * other than html_raw writes to stdout. + */ extern void html_flush(void); -/* Collect page output into sb rather than writing it, until html_capture_end. - * Must not span a filter or anything else that writes to stdout directly. */ +/* + * Collect page output into sb rather than writing it, until html_capture_end. + * A capture must not span a filter or anything else that writes to stdout + * directly, since that output would escape the capture. + */ extern void html_capture_begin(struct strbuf *sb); extern void html_capture_end(void); __attribute__((format (printf,1,2))) -extern void htmlf(const char *format,...); +extern void htmlf(const char *format, ...); __attribute__((format (printf,1,2))) -extern void html_txtf(const char *format,...); +extern void html_txtf(const char *format, ...); __attribute__((format (printf,1,0))) extern void html_vtxtf(const char *format, va_list ap); __attribute__((format (printf,1,2))) -extern void html_attrf(const char *format,...); +extern void html_attrf(const char *format, ...); extern void html_txt(const char *txt); + +/* + * Escape at most len bytes of txt. The return is what is left of that budget, + * and a caller prints an ellipsis when it comes back negative. + */ extern ssize_t html_ntxt(const char *txt, size_t len); + extern void html_attr(const char *txt); extern void html_url_path(const char *txt); extern void html_url_arg(const char *txt); extern void html_header_arg_in_quotes(const char *txt); extern void html_hidden(const char *name, const char *value); -extern void html_option(const char *value, const char *text, const char *selected_value); + +__attribute__((format (printf,1,2))) +extern char *cgit_fmt(const char *format, ...); + +__attribute__((format (printf,1,2))) +extern char *cgit_fmtalloc(const char *format, ...); + +extern void html_option(const char *value, const char *text, + const char *selected_value); extern void html_intoption(int value, const char *text, int selected_value); -extern void html_link_open(const char *url, const char *title, const char *class); +extern void html_link_open(const char *url, const char *title, + const char *class); extern void html_link_close(void); extern void html_fileperm(unsigned short mode); extern int html_include(const char *filename); -extern void http_parse_querystring(const char *txt, void (*fn)(const char *name, const char *value)); +extern void http_parse_querystring(const char *txt, + void (*fn)(const char *name, + const char *value)); -#endif /* HTML_H */ +#endif // CGIT_HTML_H diff --git a/source/parsing.c b/source/parsing.c index 0d63b51..de2798a 100644 --- a/source/parsing.c +++ b/source/parsing.c @@ -1,127 +1,61 @@ -/* parsing.c: parsing of config files - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The two kinds of raw text cgit reads before it can render a page, the path + * of the incoming request and the body of a commit or a tag object. Reading + * the request path settles which repository and which page the request works + * on. Commit and tag objects arrive as header lines, a blank line, and then + * the message, so the readers here walk the headers, copy out the fields the + * pages need, and re-encode the message into the page encoding. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" +#include "parsing.h" +#include "shared.h" -/* - * url syntax: [repo ['/' cmd [ '/' path]]] - * repo: any valid repo url, may contain '/' - * cmd: log | commit | diff | tree | view | blob | snapshot - * path: any valid path, may contain '/' - * - */ -void cgit_parse_url(const char *url) -{ - char *c, *cmd, *p, *buf; - struct cgit_repo *repo; - - if (!url || url[0] == '\0') - return; - - ctx.qry.page = NULL; - ctx.repo = cgit_get_repoinfo(url); - if (ctx.repo) { - ctx.qry.repo = ctx.repo->url; - return; - } - - buf = xstrdup(url); - cmd = NULL; - c = strchr(buf, '/'); - while (c) { - c[0] = '\0'; - repo = cgit_get_repoinfo(buf); - if (repo) { - ctx.repo = repo; - cmd = c; - } - c[0] = '/'; - c = strchr(c + 1, '/'); - } - - if (ctx.repo) { - ctx.qry.repo = ctx.repo->url; - p = strchr(cmd + 1, '/'); - if (p) { - p[0] = '\0'; - if (p[1]) - ctx.qry.path = cgit_trim_end(p + 1, '/'); - } - if (cmd[1]) - ctx.qry.page = xstrdup(cmd + 1); - } - free(buf); -} - -static char *substr(const char *head, const char *tail) +static char *substr(const char *start, const char *end) { size_t len; char *buf; - if (tail < head) + if (end < start) return xstrdup(""); - // head points into the commit buffer, so strlcpy would measure the - // whole remaining commit just to copy a name or a subject off the front - // of it. - len = tail - head; + // start points into the object buffer, so strlcpy would measure the + // whole rest of the object to copy a name off the front of it. + len = end - start; buf = xmalloc(len + 1); - memcpy(buf, head, len); + memcpy(buf, start, len); buf[len] = '\0'; return buf; } -static void parse_user(const char *t, char **name, char **email, unsigned long *date, int *tz) +static void parse_user(const char *line, char **name, char **email, + timestamp_t *date, int *tz) { struct ident_split ident; - unsigned email_len; + struct strbuf address = STRBUF_INIT; + ptrdiff_t email_len; - if (!split_ident_line(&ident, t, strchrnul(t, '\n') - t)) { + if (!split_ident_line(&ident, line, strchrnul(line, '\n') - line)) { *name = substr(ident.name_begin, ident.name_end); + // Assembled rather than formatted. The length is a pointer + // difference, and the precision of a %.*s conversion has to be + // an int, which cannot hold one on a 64 bit host. email_len = ident.mail_end - ident.mail_begin; - *email = xmalloc(strlen("<") + email_len + strlen(">") + 1); - xsnprintf(*email, email_len + 3, "<%.*s>", email_len, ident.mail_begin); + strbuf_addch(&address, '<'); + if (email_len > 0) + strbuf_add(&address, ident.mail_begin, email_len); + strbuf_addch(&address, '>'); + *email = strbuf_detach(&address, NULL); if (ident.date_begin) - *date = strtoul(ident.date_begin, NULL, 10); + *date = parse_timestamp(ident.date_begin, NULL, 10); if (ident.tz_begin) *tz = atoi(ident.tz_begin); } } -#ifdef NO_ICONV -#define reencode(a, b, c) -#else -static const char *reencode(char **txt, const char *src_enc, const char *dst_enc) -{ - char *tmp; - - if (!txt) - return NULL; - - if (!*txt || !src_enc || !dst_enc) - return *txt; - - /* no encoding needed if src_enc equals dst_enc */ - if (!strcasecmp(src_enc, dst_enc)) - return *txt; - - tmp = reencode_string(*txt, dst_enc, src_enc); - if (tmp) { - free(*txt); - *txt = tmp; - } - return *txt; -} -#endif - static const char *next_header_line(const char *p) { p = strchr(p, '\n'); @@ -135,17 +69,89 @@ static int end_of_header(const char *p) return !p || (*p == '\n'); } +/* + * A git built without iconv still offers reencode_string, where it always + * fails, so the field is left in whatever encoding it arrived in. + */ +static const char *reencode(char **text, const char *from, const char *to) +{ + char *converted; + + if (!text) + return NULL; + + if (!*text || !from || !to) + return *text; + + if (!strcasecmp(from, to)) + return *text; + + converted = reencode_string(*text, to, from); + if (converted) { + free(*text); + *text = converted; + } + return *text; +} + +/* + * A repository url may itself contain slashes, so every slash-separated + * prefix is looked up and the longest one that names a repository wins. + */ +void cgit_parse_url(const char *url) +{ + char *buf, *slash, *repo_end, *page_end; + struct cgit_repo *repo; + + if (!url || url[0] == '\0') + return; + + ctx.qry.page = NULL; + ctx.repo = cgit_get_repoinfo(url); + if (ctx.repo) { + ctx.qry.repo = ctx.repo->url; + return; + } + + buf = xstrdup(url); + repo_end = NULL; + slash = strchr(buf, '/'); + while (slash) { + slash[0] = '\0'; + repo = cgit_get_repoinfo(buf); + if (repo) { + ctx.repo = repo; + repo_end = slash; + } + slash[0] = '/'; + slash = strchr(slash + 1, '/'); + } + + if (ctx.repo) { + ctx.qry.repo = ctx.repo->url; + page_end = strchr(repo_end + 1, '/'); + if (page_end) { + page_end[0] = '\0'; + if (page_end[1]) + ctx.qry.path = cgit_trim_end(page_end + 1, '/'); + } + if (repo_end[1]) + ctx.qry.page = xstrdup(repo_end + 1); + } + free(buf); +} + struct commitinfo *cgit_parse_commit(struct commit *commit) { - struct commitinfo *ret; + struct commitinfo *info; const char *p = repo_get_commit_buffer(the_repository, commit, NULL); - const char *t; + const char *eol; - ret = xcalloc(1, sizeof(struct commitinfo)); - ret->commit = commit; + info = xcalloc(1, sizeof(struct commitinfo)); + info->commit = commit; if (!p) - return ret; + return info; if (!skip_prefix(p, "tree ", &p)) die("Bad commit: %s", oid_to_hex(&commit->object.oid)); @@ -155,27 +161,29 @@ struct commitinfo *cgit_parse_commit(struct commit *commit) p += the_hash_algo->hexsz + 1; if (p && skip_prefix(p, "author ", &p)) { - parse_user(p, &ret->author, &ret->author_email, - &ret->author_date, &ret->author_tz); + parse_user(p, &info->author, &info->author_email, + &info->author_date, &info->author_tz); p = next_header_line(p); } if (p && skip_prefix(p, "committer ", &p)) { - parse_user(p, &ret->committer, &ret->committer_email, - &ret->committer_date, &ret->committer_tz); + parse_user(p, &info->committer, &info->committer_email, + &info->committer_date, &info->committer_tz); p = next_header_line(p); } if (p && skip_prefix(p, "encoding ", &p)) { - t = strchr(p, '\n'); - if (t) { - ret->msg_encoding = substr(p, t + 1); - p = t + 1; + eol = strchr(p, '\n'); + if (eol) { + info->msg_encoding = substr(p, eol + 1); + p = eol + 1; } } - if (!ret->msg_encoding) - ret->msg_encoding = xstrdup("UTF-8"); + // Git only writes the header when the message is in something other + // than UTF-8, so its absence means UTF-8. + if (!info->msg_encoding) + info->msg_encoding = xstrdup("UTF-8"); while (!end_of_header(p)) p = next_header_line(p); @@ -183,27 +191,28 @@ struct commitinfo *cgit_parse_commit(struct commit *commit) p++; if (p) { - t = strchrnul(p, '\n'); - ret->subject = substr(p, t); - while (*t == '\n') - t++; - ret->msg = xstrdup(t); + eol = strchrnul(p, '\n'); + info->subject = substr(p, eol); + while (*eol == '\n') + eol++; + info->msg = xstrdup(eol); } else { - // A crafted commit can end right after its headers with no - // message. Keep subject and msg as empty strings so callers - // can treat them as text unconditionally. - ret->subject = xstrdup(""); - ret->msg = xstrdup(""); + // Reached when an object is truncated mid header, which + // leaves nothing at all after them. Callers render subject + // and msg as text without checking, so they get empty + // strings rather than NULL. + info->subject = xstrdup(""); + info->msg = xstrdup(""); } - reencode(&ret->author, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->author_email, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->committer, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->committer_email, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->subject, ret->msg_encoding, PAGE_ENCODING); - reencode(&ret->msg, ret->msg_encoding, PAGE_ENCODING); + reencode(&info->author, info->msg_encoding, PAGE_ENCODING); + reencode(&info->author_email, info->msg_encoding, PAGE_ENCODING); + reencode(&info->committer, info->msg_encoding, PAGE_ENCODING); + reencode(&info->committer_email, info->msg_encoding, PAGE_ENCODING); + reencode(&info->subject, info->msg_encoding, PAGE_ENCODING); + reencode(&info->msg, info->msg_encoding, PAGE_ENCODING); - return ret; + return info; } struct taginfo *cgit_parse_tag(struct tag *tag) @@ -212,18 +221,19 @@ struct taginfo *cgit_parse_tag(struct tag *tag) enum object_type type; unsigned long size; const char *p; - struct taginfo *ret = NULL; + struct taginfo *info = NULL; - data = odb_read_object(the_repository->objects, &tag->object.oid, &type, &size); + data = odb_read_object(the_repository->objects, &tag->object.oid, + &type, &size); if (!data || type != OBJ_TAG) goto cleanup; - ret = xcalloc(1, sizeof(struct taginfo)); + info = xcalloc(1, sizeof(struct taginfo)); for (p = data; !end_of_header(p); p = next_header_line(p)) { if (skip_prefix(p, "tagger ", &p)) { - parse_user(p, &ret->tagger, &ret->tagger_email, - &ret->tagger_date, &ret->tagger_tz); + parse_user(p, &info->tagger, &info->tagger_email, + &info->tagger_date, &info->tagger_tz); } } @@ -231,9 +241,9 @@ struct taginfo *cgit_parse_tag(struct tag *tag) p++; if (p && *p) - ret->msg = xstrdup(p); + info->msg = xstrdup(p); cleanup: free(data); - return ret; + return info; } diff --git a/source/parsing.h b/source/parsing.h new file mode 100644 index 0000000..2230eaa --- /dev/null +++ b/source/parsing.h @@ -0,0 +1,20 @@ +/* + * Readers that turn raw git object bodies and the incoming request URL into + * the records cgit renders from. Commit and tag objects arrive as flat text, + * so the work here is splitting off the headers and re-encoding the remaining + * free text into the page encoding. + */ + +#ifndef CGIT_PARSING_H +#define CGIT_PARSING_H + +#include "cgit.h" + +// Split PATH_INFO into the repository and the page it names, storing both in +// ctx.qry. +extern void cgit_parse_url(const char *url); + +extern struct commitinfo *cgit_parse_commit(struct commit *commit); +extern struct taginfo *cgit_parse_tag(struct tag *tag); + +#endif // CGIT_PARSING_H diff --git a/source/scan-tree.c b/source/scan-tree.c index 89f65c9..4a1444f 100644 --- a/source/scan-tree.c +++ b/source/scan-tree.c @@ -1,86 +1,120 @@ -/* scan-tree.c - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * Discovery of repositories on disk, so that an administrator can point cgit + * at a directory instead of naming every repository in cgitrc. The tree below + * that directory is walked for bare repositories and working trees, or when a + * project list is configured only the paths it names are visited. Every + * repository found is registered under its path relative to the root of the + * scan, with owner, description and section taken from the files git keeps + * beside it, and a cgitrc in the repository has the last word over all of them. */ #include "cgit.h" +#include "config.h" #include "scan-tree.h" -#include "configfile.h" -#include "html.h" -#include <config.h> +#include "shared.h" -// The description git puts in every freshly created repository. +// Git writes this line into the description file of every repository it +// creates, so a repository still carrying it gets cgit's own default instead. static const char *default_git_desc = "Unnamed repository; edit this file 'description' to name the repository."; -static struct cgit_repo *repo; +// The config callbacks are handed only a name and a value, so the repository +// being filled in waits here for the length of one add_repo call. +static struct cgit_repo *current_repo; -/* return 1 if path contains a objects/ directory and a HEAD file */ -static int is_git_dir(const char *path) +static int stat_entry(const char *dir, const char *name, struct stat *st) { - struct stat st; - struct strbuf pathbuf = STRBUF_INIT; - int result = 0; - - strbuf_addf(&pathbuf, "%s/objects", path); - if (stat(pathbuf.buf, &st)) { - if (errno != ENOENT) - fprintf(stderr, "Error checking path %s: %s (%d)\n", - path, strerror(errno), errno); - goto out; - } - if (!S_ISDIR(st.st_mode)) - goto out; - - strbuf_reset(&pathbuf); - strbuf_addf(&pathbuf, "%s/HEAD", path); - if (stat(pathbuf.buf, &st)) { - if (errno != ENOENT) - fprintf(stderr, "Error checking path %s: %s (%d)\n", - path, strerror(errno), errno); - goto out; - } - if (!S_ISREG(st.st_mode)) - goto out; + struct strbuf path = STRBUF_INIT; + int err; - result = 1; -out: - strbuf_release(&pathbuf); - return result; + strbuf_addf(&path, "%s/%s", dir, name); + err = stat(path.buf, st); + // A missing entry is the ordinary answer for a directory that is not a + // repository, so only some other failure is worth reporting. + if (err && errno != ENOENT) + fprintf(stderr, "Error checking path %s: %s (%d)\n", + dir, strerror(errno), errno); + strbuf_release(&path); + return err; } -static void scan_tree_repo_config(const char *name, const char *value) +static int is_git_dir(const char *path) { - cgit_repo_config(repo, name, value); + struct stat st; + + if (stat_entry(path, "objects", &st) || !S_ISDIR(st.st_mode)) + return 0; + if (stat_entry(path, "HEAD", &st) || !S_ISREG(st.st_mode)) + return 0; + return 1; } -static int gitconfig_config(const char *key, const char *value, +static int apply_gitconfig(const char *key, const char *value, const __attribute__((unused)) struct config_context *cfg_ctx, void *cb) { const char *name; if (!strcmp(key, "gitweb.owner")) - cgit_repo_config(repo, "owner", value); + cgit_repo_config(current_repo, "owner", value); else if (!strcmp(key, "gitweb.description")) - cgit_repo_config(repo, "desc", value); + cgit_repo_config(current_repo, "desc", value); else if (!strcmp(key, "gitweb.category")) - cgit_repo_config(repo, "section", value); + cgit_repo_config(current_repo, "section", value); else if (!strcmp(key, "gitweb.homepage")) - cgit_repo_config(repo, "homepage", value); + cgit_repo_config(current_repo, "homepage", value); else if (skip_prefix(key, "cgit.", &name)) - cgit_repo_config(repo, name, value); + cgit_repo_config(current_repo, name, value); return 0; } -static char *xstrrchr(char *s, char *from, int c) +static void apply_cgitrc(const char *name, const char *value) { - while (from >= s && *from != c) + cgit_repo_config(current_repo, name, value); +} + +static char *find_char_back(char *start, char *from, int c) +{ + while (from >= start && *from != c) from--; - return from < s ? NULL : from; + return from < start ? NULL : from; +} + +/* + * A positive depth counts separators from the left of the path and a negative + * one counts back from the right. + */ +static char *section_slash(struct strbuf *relpath, int depth) +{ + char *slash; + + if (depth > 0) { + slash = relpath->buf - 1; + while (slash && depth && (slash = strchr(slash + 1, '/'))) + depth--; + } else { + slash = relpath->buf + relpath->len; + while (slash && depth && + (slash = find_char_back(relpath->buf, slash - 1, '/'))) + depth++; + } + return slash && !depth ? slash : NULL; +} + +static void set_section_from_path(struct strbuf *relpath, int depth) +{ + char *slash = section_slash(relpath, depth); + + if (!slash) + return; + *slash = '\0'; + current_repo->section = cgit_strdup_first_line(relpath->buf); + *slash = '/'; + if (starts_with(current_repo->name, current_repo->section)) { + current_repo->name += strlen(current_repo->section); + if (*current_repo->name == '/') + current_repo->name++; + } } static void add_repo(const char *base, struct strbuf *path) @@ -88,10 +122,9 @@ static void add_repo(const char *base, struct strbuf *path) struct stat st; struct passwd *pwd; size_t pathlen; - struct strbuf rel = STRBUF_INIT; - char *p, *slash; - int n; - size_t size; + struct strbuf relpath = STRBUF_INIT; + char *comma; + size_t desc_size; if (stat(path->buf, &st)) { fprintf(stderr, "Error accessing %s: %s (%d)\n", @@ -104,7 +137,7 @@ static void add_repo(const char *base, struct strbuf *path) if (ctx.cfg.strict_export) { strbuf_addstr(path, ctx.cfg.strict_export); - if(stat(path->buf, &st)) + if (stat(path->buf, &st)) return; strbuf_setlen(path, pathlen); } @@ -115,86 +148,79 @@ static void add_repo(const char *base, struct strbuf *path) strbuf_setlen(path, pathlen); if (!starts_with(path->buf, base)) - strbuf_addbuf(&rel, path); + strbuf_addbuf(&relpath, path); else - strbuf_addstr(&rel, path->buf + strlen(base) + 1); + strbuf_addstr(&relpath, path->buf + strlen(base) + 1); // Drop the trailing slash added above before testing for "/.git", since // with it still attached the suffix never matches and every ordinary // working tree is named "repo/.git" instead of "repo". - if (rel.len && rel.buf[rel.len - 1] == '/') - strbuf_setlen(&rel, rel.len - 1); - if (rel.len >= 5 && !strcmp(rel.buf + rel.len - 5, "/.git")) - strbuf_setlen(&rel, rel.len - 5); + if (relpath.len && relpath.buf[relpath.len - 1] == '/') + strbuf_setlen(&relpath, relpath.len - 1); + if (relpath.len >= 5 && !strcmp(relpath.buf + relpath.len - 5, "/.git")) + strbuf_setlen(&relpath, relpath.len - 5); - repo = cgit_add_repo(rel.buf); + current_repo = cgit_add_repo(relpath.buf); if (ctx.cfg.enable_git_config) { strbuf_addstr(path, "config"); - git_config_from_file(gitconfig_config, path->buf, NULL); + git_config_from_file(apply_gitconfig, path->buf, NULL); strbuf_setlen(path, pathlen); } if (ctx.cfg.remove_suffix) { size_t urllen; - strip_suffix(repo->url, ".git", &urllen); - strip_suffix_mem(repo->url, &urllen, "/"); - repo->url[urllen] = '\0'; + strip_suffix(current_repo->url, ".git", &urllen); + strip_suffix_mem(current_repo->url, &urllen, "/"); + current_repo->url[urllen] = '\0'; } - repo->path = cgit_strdup_first_line(path->buf); - while (!repo->owner) { + current_repo->path = cgit_strdup_first_line(path->buf); + while (!current_repo->owner) { if ((pwd = getpwuid(st.st_uid)) == NULL) { fprintf(stderr, "Error reading owner-info for %s: %s (%d)\n", path->buf, strerror(errno), errno); break; } + // A gecos field puts the owner's name in front of a comma + // separated list of office and phone details. if (pwd->pw_gecos) - if ((p = strchr(pwd->pw_gecos, ','))) - *p = '\0'; - repo->owner = cgit_strdup_first_line(pwd->pw_gecos ? pwd->pw_gecos : pwd->pw_name); + if ((comma = strchr(pwd->pw_gecos, ','))) + *comma = '\0'; + current_repo->owner = cgit_strdup_first_line( + pwd->pw_gecos ? pwd->pw_gecos : pwd->pw_name); } - if (repo->desc == cgit_default_repo_desc || !repo->desc) { + if (current_repo->desc == cgit_default_repo_desc || !current_repo->desc) { strbuf_addstr(path, "description"); if (!stat(path->buf, &st)) - cgit_read_first_line(path->buf, &repo->desc, &size); + cgit_read_first_line(path->buf, ¤t_repo->desc, + &desc_size); strbuf_setlen(path, pathlen); - // Git writes this line into every repository it creates, so it - // describes nothing. Treat it as no description at all rather - // than repeating it down the whole index. - if (repo->desc && !strcmp(repo->desc, default_git_desc)) { - free(repo->desc); - repo->desc = cgit_default_repo_desc; + if (current_repo->desc && + !strcmp(current_repo->desc, default_git_desc)) { + free(current_repo->desc); + current_repo->desc = cgit_default_repo_desc; } } - if (ctx.cfg.section_from_path) { - n = ctx.cfg.section_from_path; - if (n > 0) { - slash = rel.buf - 1; - while (slash && n && (slash = strchr(slash + 1, '/'))) - n--; - } else { - slash = rel.buf + rel.len; - while (slash && n && (slash = xstrrchr(rel.buf, slash - 1, '/'))) - n++; - } - if (slash && !n) { - *slash = '\0'; - repo->section = cgit_strdup_first_line(rel.buf); - *slash = '/'; - if (starts_with(repo->name, repo->section)) { - repo->name += strlen(repo->section); - if (*repo->name == '/') - repo->name++; - } - } - } + if (ctx.cfg.section_from_path) + set_section_from_path(&relpath, ctx.cfg.section_from_path); strbuf_addstr(path, "cgitrc"); if (!stat(path->buf, &st)) - parse_configfile(path->buf, &scan_tree_repo_config); + config_file_parse(path->buf, &apply_cgitrc); - strbuf_release(&rel); + strbuf_release(&relpath); +} + +static int should_scan(const struct dirent *ent) +{ + if (ent->d_name[0] != '.') + return 1; + if (ent->d_name[1] == '\0') + return 0; + if (ent->d_name[1] == '.' && ent->d_name[2] == '\0') + return 0; + return ctx.cfg.scan_hidden_path; } static void scan_path(const char *base, const char *path) @@ -211,7 +237,7 @@ static void scan_path(const char *base, const char *path) return; } - strbuf_add(&pathbuf, path, strlen(path)); + strbuf_add(&pathbuf, path, pathlen); if (is_git_dir(pathbuf.buf)) { add_repo(base, &pathbuf); goto end; @@ -221,20 +247,12 @@ static void scan_path(const char *base, const char *path) add_repo(base, &pathbuf); goto end; } - /* - * Add one because we don't want to lose the trailing '/' when we - * reset the length of pathbuf in the loop below. - */ + // Take in the '/' that "/.git" left in the buffer, since the loop below + // truncates to this length and then appends an entry name straight on. pathlen++; while ((ent = readdir(dir)) != NULL) { - if (ent->d_name[0] == '.') { - if (ent->d_name[1] == '\0') - continue; - if (ent->d_name[1] == '.' && ent->d_name[2] == '\0') - continue; - if (!ctx.cfg.scan_hidden_path) - continue; - } + if (!should_scan(ent)) + continue; strbuf_setlen(&pathbuf, pathlen); strbuf_addstr(&pathbuf, ent->d_name); if (stat(pathbuf.buf, &st)) { diff --git a/source/scan-tree.h b/source/scan-tree.h index def0a7a..515037a 100644 --- a/source/scan-tree.h +++ b/source/scan-tree.h @@ -1,2 +1,16 @@ +/* + * Discovery of repositories on disk, so that an administrator can point cgit + * at a directory instead of naming every repository in cgitrc. The tree below + * that directory is walked for bare repositories and working trees, or when a + * project list is configured only the paths it names are visited. Every + * repository found is added to the shared repository list under its path + * relative to the root of the scan. + */ + +#ifndef CGIT_SCAN_TREE_H +#define CGIT_SCAN_TREE_H + extern void scan_projects(const char *path, const char *projectsfile); extern void scan_tree(const char *path); + +#endif // CGIT_SCAN_TREE_H diff --git a/source/shared.c b/source/shared.c index 259802a..e6dd011 100644 --- a/source/shared.c +++ b/source/shared.c This diff is too large to be rendered inline. View it on its own page. diff --git a/source/shared.h b/source/shared.h new file mode 100644 index 0000000..9f4cba3 --- /dev/null +++ b/source/shared.h @@ -0,0 +1,92 @@ +/* + * The helpers that belong to no single page. Repository registration, the + * lifetimes of the reference and commit records the renderers build, thin + * wrappers over git's diff machinery, and a handful of string, date and + * mimetype utilities all live here because several renderers need them. The + * two process-wide globals, ctx and cgit_repolist, are defined here as well. + */ + +#ifndef CGIT_SHARED_H +#define CGIT_SHARED_H + +#include "cgit.h" + +extern char *cgit_default_repo_desc; + +typedef void (*filepair_fn)(struct diff_filepair *pair); +typedef void (*linediff_fn)(char *line, int len); + +/* + * Each of these returns its argument untouched when the syscall it wraps + * succeeded, and dies with the message and errno otherwise. + */ +extern int cgit_die_unless_zero(int result, const char *msg); +extern int cgit_die_unless_positive(int result, const char *msg); +extern int cgit_die_unless_non_negative(int result, const char *msg); + +/* + * Append a repository to cgit_repolist and return it, seeded from the current + * global configuration. The returned pointer is invalidated by the next call, + * because the list grows by reallocation. + */ +extern struct cgit_repo *cgit_add_repo(const char *url); +extern struct cgit_repo *cgit_get_repoinfo(const char *url); + +// Export the CGIT_REPO_* variables that filters and hooks read. +extern void cgit_prepare_repo_env(struct cgit_repo *repo); + +extern void cgit_free_reflist_inner(struct reflist *list); + +// Matches git's reference iteration callback, and appends each ref it is +// given to the struct reflist passed as cb_data. +extern int cgit_refs_cb(const struct reference *ref, void *cb_data); + +extern void cgit_free_commitinfo(struct commitinfo *info); +extern void cgit_free_taginfo(struct taginfo *info); + +extern void cgit_diff_tree_cb(struct diff_queue_struct *q, + struct diff_options *options, void *data); + +/* + * Diff two blobs and hand fn one line at a time. Returns non-zero when either + * blob could not be read. A blob that is binary or larger than max-blob-size + * sets binary instead, and produces no lines, so the caller can offer a link + * in place of the content. + */ +extern int cgit_diff_files(const struct object_id *old_oid, + const struct object_id *new_oid, + unsigned long *old_size, unsigned long *new_size, + int *binary, int context, int ignorews, + linediff_fn fn); + +extern void cgit_diff_tree(const struct object_id *old_oid, + const struct object_id *new_oid, + filepair_fn fn, const char *prefix, int ignorews); +extern void cgit_diff_commit(struct commit *commit, filepair_fn fn, + const char *prefix); + +// Accept only the date formats cgit documents, leaving mode unchanged for +// anything else. +extern void cgit_parse_date_format(const char *format, struct date_mode *mode); + +// Copy str without its trailing runs of c, returning NULL if nothing is left. +extern char *cgit_trim_end(const char *str, char c); +extern char *cgit_ensure_end(const char *str, char c); +extern void strbuf_ensure_end(struct strbuf *sb, char c); + +/* + * Read the first line of a file into a newly allocated, zero terminated + * buffer. Returns 0 on success and an errno value otherwise. + */ +extern int cgit_read_first_line(const char *path, char **buf, size_t *size); +extern char *cgit_strdup_first_line(const char *text); + +/* + * Replace every $token in text with the matching environment variable. The + * result is a static buffer, so a caller that needs to keep it must copy it. + */ +extern char *cgit_expand_macros(const char *text); + +extern char *cgit_get_mimetype_for_filename(const char *filename); + +#endif // CGIT_SHARED_H diff --git a/source/ui-atom.c b/source/ui-atom.c index f968e24..aa9237a 100644 --- a/source/ui-atom.c +++ b/source/ui-atom.c @@ -1,24 +1,57 @@ -/* ui-atom.c: functions for atom feeds - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The Atom feed for a repository, which lists recent commits so a reader can + * follow the project from a feed reader instead of the browsable pages. The + * response is XML rather than a page, so this file writes its own HTTP headers + * and never goes through the shared HTML layout. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-atom.h" #include "html.h" +#include "parsing.h" +#include "shared.h" +#include "ui-atom.h" #include "ui-shared.h" -static void add_entry(struct commit *commit, const char *host) +/* + * Atom timestamps have to be RFC 3339, so a feed ignores the date-format and + * local-time settings the browsable pages honour. The zero is the timezone + * offset, which pins every feed date to UTC. + */ +static const char *feed_date(timestamp_t when) +{ + return show_date(when, 0, date_mode_from_type(DATE_ISO8601_STRICT)); +} + +/* + * cgit keeps an address with the angle brackets it was written with, while + * Atom wants the bare address. + */ +static void print_email(const char *email) +{ + char *copy = xstrdup(email); + char *start, *end; + + start = strchr(copy, '<'); + if (start) + start++; + else + start = copy; + end = strchr(start, '>'); + if (end) + *end = '\0'; + + html("<email>"); + html_txt(start); + html("</email>\n"); + free(copy); +} + +static void print_entry(struct commit *commit, const char *host) { - char delim = '&'; - char *hex; - char *mail, *t, *t2; struct commitinfo *info; + char *hex; info = cgit_parse_commit(commit); hex = oid_to_hex(&commit->object.oid); @@ -27,8 +60,7 @@ static void add_entry(struct commit *commit, const char *host) html_txt(info->subject); html("</title>\n"); html("<updated>"); - html_txt(show_date(info->committer_date, 0, - date_mode_from_type(DATE_ISO8601_STRICT))); + html_txt(feed_date(info->committer_date)); html("</updated>\n"); html("<author>\n"); if (info->author) { @@ -36,33 +68,24 @@ static void add_entry(struct commit *commit, const char *host) html_txt(info->author); html("</name>\n"); } - if (info->author_email && !ctx.cfg.noplainemail) { - mail = xstrdup(info->author_email); - t = strchr(mail, '<'); - if (t) - t++; - else - t = mail; - t2 = strchr(t, '>'); - if (t2) - *t2 = '\0'; - html("<email>"); - html_txt(t); - html("</email>\n"); - free(mail); - } + if (info->author_email && !ctx.cfg.noplainemail) + print_email(info->author_email); html("</author>\n"); html("<published>"); - html_txt(show_date(info->author_date, 0, - date_mode_from_type(DATE_ISO8601_STRICT))); + html_txt(feed_date(info->author_date)); html("</published>\n"); if (host) { char *pageurl; + char delim = '&'; + html("<link rel='alternate' type='text/html' href='"); html(cgit_httpscheme()); html_attr(host); pageurl = cgit_pageurl(ctx.repo->url, "commit", NULL); html_attr(pageurl); + // Without a virtual root the page url is already a query + // string, so the commit id continues it instead of opening + // one. if (ctx.cfg.virtual_root) delim = '?'; html_attrf("%cid=%s", delim, hex); @@ -79,15 +102,16 @@ static void add_entry(struct commit *commit, const char *host) cgit_free_commitinfo(info); } - void cgit_print_atom(char *tip, const char *path, int max_count) { char *host; + // setup_revisions reads a command line, so the first slot is the + // unused program name and parsing starts at the second. const char *argv[] = {NULL, tip, NULL, NULL, NULL}; struct commit *commit; struct rev_info rev; int argc = 2; - bool first = true; + bool need_updated = true; if (ctx.qry.show_all) argv[1] = "--all"; @@ -143,22 +167,23 @@ void cgit_print_atom(char *tip, const char *path, int max_count) free(repourl); } while ((commit = get_revision(&rev)) != NULL) { - if (first) { + if (need_updated) { html("<updated>"); - html_txt(show_date(commit->date, 0, - date_mode_from_type(DATE_ISO8601_STRICT))); + html_txt(feed_date(commit->date)); html("</updated>\n"); - first = false; + need_updated = false; } - add_entry(commit, host); + print_entry(commit, host); + // release_commit_memory frees the parent list without clearing + // the pointer to it, so drop the dangling reference here. release_commit_memory(the_repository->parsed_objects, commit); commit->parents = NULL; } - if (first) { - /* An empty feed still needs one feed-level <updated>. */ + if (need_updated) { + // Atom makes a feed level updated mandatory, and an empty feed + // has no commit to take one from. html("<updated>"); - html_txt(show_date(time(NULL), 0, - date_mode_from_type(DATE_ISO8601_STRICT))); + html_txt(feed_date(time(NULL))); html("</updated>\n"); } html("</feed>\n"); diff --git a/source/ui-atom.h b/source/ui-atom.h index dda953b..71549e8 100644 --- a/source/ui-atom.h +++ b/source/ui-atom.h @@ -1,6 +1,11 @@ -#ifndef UI_ATOM_H -#define UI_ATOM_H +/* + * The Atom feed page, which serves a repository's recent commits as XML for a + * feed reader rather than as one of the browsable pages. + */ + +#ifndef CGIT_UI_ATOM_H +#define CGIT_UI_ATOM_H extern void cgit_print_atom(char *tip, const char *path, int max_count); -#endif +#endif // CGIT_UI_ATOM_H diff --git a/source/ui-blame.c b/source/ui-blame.c index 132121b..80512a0 100644 --- a/source/ui-blame.c +++ b/source/ui-blame.c @@ -1,25 +1,42 @@ -/* ui-blame.c: functions for blame output - * - * Copyright (C) 2006-2017 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The blame page, which shows a file next to the commit that last touched + * each of its lines. git returns blame as runs of neighbouring lines that + * share a commit, and the page turns every run into one block in each column + * so the hashes, the line numbers and the striped background all line up with + * the source. Only a regular file can be blamed, so a path naming a folder is + * turned away. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-blame.h" +#include "filter.h" #include "html.h" +#include "parsing.h" +#include "shared.h" +#include "ui-blame.h" #include "ui-shared.h" -#include "strvec.h" -#include "blame.h" +// A tab in the rendered source runs on to the next multiple of this. The +// stylesheet leaves tab-size alone, so the measurement has to match what the +// browser does on its own rather than anything cgit picks. +#define TAB_WIDTH 8 -/* Blame coalesces neighbouring lines from one commit, but a commit that - * touched several separate parts of the file still comes back once per part. - * Each of those repeats a full commit parse, so the rendered detail is kept - * and looked up by object id. */ +enum blame_target { + TARGET_MISSING, + TARGET_FILE, + TARGET_FOLDER, +}; + +struct walk_tree_context { + char *rev; + int match_baselen; + enum blame_target found; +}; + +// A commit that touched several separate parts of the file comes back once +// per part, and each repeat would parse the commit again, so the rendered +// detail is cached here and looked up by object id. static struct string_list suspect_details = STRING_LIST_INIT_DUP; static void free_suspect_details(void) @@ -31,8 +48,40 @@ static void free_suspect_details(void) string_list_clear(&suspect_details, 0); } -/* Write out a run of one repeated character. A single blame entry can cover - * the whole file, so the run is handed on in batches rather than built whole. */ +/* + * The scoreboard keeps a pointer to revs, so the caller owns both and has to + * keep them alive together. + */ +static void run_blame(struct blame_scoreboard *sb, struct rev_info *revs, + const char *path, const char *rev) +{ + struct strvec argv = STRVEC_INIT; + struct blame_origin *origin; + + strvec_push(&argv, "blame"); + strvec_push(&argv, rev); + repo_init_revisions(the_repository, revs, NULL); + revs->diffopt.flags.allow_textconv = 1; + setup_revisions(argv.nr, argv.v, revs, NULL); + init_scoreboard(sb); + sb->revs = revs; + sb->repo = the_repository; + sb->path = path; + setup_scoreboard(sb, &origin); + origin->suspects = blame_entry_prepend(NULL, 0, sb->num_lines, origin); + prio_queue_put(&sb->commits, origin->commit); + blame_origin_decref(origin); + sb->ent = NULL; + sb->path = path; + assign_blame(sb, 0); + blame_sort_final(sb); + blame_coalesce(sb); +} + +/* + * A single blame entry can cover the whole file, so the run is handed on in + * batches rather than built whole. + */ static void emit_chars(char ch, unsigned long count) { struct strbuf run = STRBUF_INIT; @@ -48,7 +97,10 @@ static void emit_chars(char ch, unsigned long count) strbuf_release(&run); } -static char *emit_suspect_detail(struct blame_origin *suspect) +/* + * The returned string belongs to suspect_details, not the caller. + */ +static char *suspect_detail(struct blame_origin *suspect) { struct commitinfo *info; struct strbuf detail = STRBUF_INIT; @@ -87,18 +139,21 @@ static char *emit_suspect_detail(struct blame_origin *suspect) return cached->util; } -static void emit_blame_entry_hash(struct blame_entry *ent) +static void emit_entry_hash(struct blame_entry *ent) { struct blame_origin *suspect = ent->suspect; struct object_id *oid = &suspect->commit->object.oid; + const char *detail = suspect_detail(suspect); - const char *detail = emit_suspect_detail(suspect); html("<span class='oid'>"); - cgit_commit_link(repo_find_unique_abbrev(the_repository, oid, DEFAULT_ABBREV), detail, - NULL, ctx.qry.head, oid_to_hex(oid), suspect->path); + cgit_commit_link(repo_find_unique_abbrev(the_repository, oid, + DEFAULT_ABBREV), + detail, NULL, ctx.qry.head, oid_to_hex(oid), + suspect->path); html("</span>"); - if (!repo_parse_commit(the_repository, suspect->commit) && suspect->commit->parents) { + if (!repo_parse_commit(the_repository, suspect->commit) && + suspect->commit->parents) { struct commit *parent = suspect->commit->parents->item; html(" "); @@ -107,17 +162,30 @@ static void emit_blame_entry_hash(struct blame_entry *ent) suspect->path); } - // Batched rather than one write per line of the entry. + // The stripes only line up across the columns if each column gives an + // entry the same height, so pad out to the lines the entry covers. emit_chars('\n', ent->num_lines); } -static void emit_blame_entry_linenumber(struct blame_entry *ent) +static void emit_hashes(struct blame_scoreboard *sb) +{ + struct blame_entry *ent; + + html("<td class='hashes'>"); + for (ent = sb->ent; ent; ent = ent->next) { + html("<div class='alt'><pre>"); + emit_entry_hash(ent); + html("</pre></div>"); + } + html("</td>\n"); +} + +static void emit_entry_linenumbers(struct blame_entry *ent) { const char *numberfmt = "<a id='n%1$d' href='#n%1$d'>%1$d</a>\n"; struct strbuf numbers = STRBUF_INIT; int lineno = ent->lno; - // Batched rather than a formatted write per line of the entry. while (lineno < ent->lno + ent->num_lines) { strbuf_addf(&numbers, numberfmt, ++lineno); if (numbers.len >= HTML_BATCH) { @@ -129,50 +197,84 @@ static void emit_blame_entry_linenumber(struct blame_entry *ent) strbuf_release(&numbers); } -static void emit_blame_entry_line_background(struct blame_scoreboard *sb, - struct blame_entry *ent) +static void emit_linenumbers(struct blame_scoreboard *sb) +{ + struct blame_entry *ent; + + html("<td class='linenumbers'>"); + for (ent = sb->ent; ent; ent = ent->next) { + html("<div class='alt'><pre>"); + emit_entry_linenumbers(ent); + html("</pre></div>"); + } + html("</td>\n"); +} + +static size_t line_width(struct blame_scoreboard *sb, int line) +{ + const char *pos = blame_nth_line(sb, line); + const char *end = blame_nth_line(sb, line + 1); + size_t width = 0; + + while (pos < end) { + width++; + if (*pos++ == '\t') + width = (width + TAB_WIDTH - 1) & ~(TAB_WIDTH - 1); + } + return width; +} + +/* + * The stylesheet takes the source pre out of flow and positions it over these + * blocks, so nothing else gives the cell a size and each block has to be + * padded to the height and the width of the lines it stands behind. + */ +static void emit_entry_background(struct blame_scoreboard *sb, + struct blame_entry *ent) { + size_t widest = 2; int line; - size_t len, maxlen = 2; - const char* pos, *endpos; for (line = ent->lno; line < ent->lno + ent->num_lines; line++) { - pos = blame_nth_line(sb, line); - endpos = blame_nth_line(sb, line + 1); - len = 0; - while (pos < endpos) { - len++; - if (*pos++ == '\t') - len = (len + 7) & ~7; - } - if (len > maxlen) - maxlen = len; + size_t width = line_width(sb, line); + + if (width > widest) + widest = width; } - // The widest line decides the padding, so the entry has to be measured - // before any of it is written. The newlines do not depend on that - // measurement, so both runs go out batched once it is known. emit_chars('\n', ent->num_lines); - emit_chars(' ', maxlen - 1); + emit_chars(' ', widest - 1); } -struct walk_tree_context { - char *curr_rev; - int match_baselen; - int state; -}; +/* + * Frees each entry on the way past, so this has to be the last pass over the + * scoreboard's entries. + */ +static void emit_line_backgrounds(struct blame_scoreboard *sb) +{ + struct blame_entry *ent = sb->ent; + + html("<div>"); + while (ent) { + struct blame_entry *next = ent->next; -static void print_object(const struct object_id *oid, const char *path, - const char *basename, const char *rev) + html("<div class='alt'><pre>"); + emit_entry_background(sb, ent); + html("</pre></div>"); + free(ent); + ent = next; + } + html("</div>"); +} + +static void print_blame_page(const struct object_id *oid, const char *path, + const char *filename, const char *rev) { enum object_type type; char *buf; unsigned long size; - struct strvec rev_argv = STRVEC_INIT; struct rev_info revs; struct blame_scoreboard sb; - struct blame_origin *o; - struct blame_entry *ent = NULL; type = odb_read_object_info(the_repository->objects, oid, &size); if (type == OBJ_BAD) { @@ -194,24 +296,7 @@ static void print_object(const struct object_id *oid, const char *path, return; } - strvec_push(&rev_argv, "blame"); - strvec_push(&rev_argv, rev); - repo_init_revisions(the_repository, &revs, NULL); - revs.diffopt.flags.allow_textconv = 1; - setup_revisions(rev_argv.nr, rev_argv.v, &revs, NULL); - init_scoreboard(&sb); - sb.revs = &revs; - sb.repo = the_repository; - sb.path = path; - setup_scoreboard(&sb, &o); - o->suspects = blame_entry_prepend(NULL, 0, sb.num_lines, o); - prio_queue_put(&sb.commits, o->commit); - blame_origin_decref(o); - sb.ent = NULL; - sb.path = path; - assign_blame(&sb, 0); - blame_sort_final(&sb); - blame_coalesce(&sb); + run_blame(&sb, &revs, path, rev); cgit_set_title_from_path(path); @@ -228,46 +313,19 @@ static void print_object(const struct object_id *oid, const char *path, } html("<table class='blame blob'>\n<tr>\n"); - /* Commit hashes */ - html("<td class='hashes'>"); - for (ent = sb.ent; ent; ent = ent->next) { - html("<div class='alt'><pre>"); - emit_blame_entry_hash(ent); - html("</pre></div>"); - } - html("</td>\n"); + emit_hashes(&sb); - /* Line numbers */ - if (ctx.cfg.enable_tree_linenumbers) { - html("<td class='linenumbers'>"); - for (ent = sb.ent; ent; ent = ent->next) { - html("<div class='alt'><pre>"); - emit_blame_entry_linenumber(ent); - html("</pre></div>"); - } - html("</td>\n"); - } + if (ctx.cfg.enable_tree_linenumbers) + emit_linenumbers(&sb); html("<td class='lines'><div>"); - - /* Colored bars behind lines */ - html("<div>"); - for (ent = sb.ent; ent; ) { - struct blame_entry *e = ent->next; - html("<div class='alt'><pre>"); - emit_blame_entry_line_background(&sb, ent); - html("</pre></div>"); - free(ent); - ent = e; - } - html("</div>"); + emit_line_backgrounds(&sb); free((void *)sb.final_buf); - /* Lines */ html("<pre><code>"); if (ctx.repo->source_filter) { - char *filter_arg = xstrdup(basename); + char *filter_arg = xstrdup(filename); cgit_open_filter(ctx.repo->source_filter, filter_arg); html_raw(buf, size); cgit_close_filter(ctx.repo->source_filter); @@ -282,35 +340,34 @@ static void print_object(const struct object_id *oid, const char *path, html("</tr>\n</table>\n"); cleanup: - /* The binary and oversized branches jump here with the layout still - * open, so close it on every path rather than only the normal one. */ cgit_print_layout_end(); free_suspect_details(); free(buf); } static int walk_tree(const struct object_id *oid, struct strbuf *base, - const char *pathname, unsigned mode, void *cbdata) + const char *pathname, unsigned mode, void *data) { - struct walk_tree_context *walk_tree_ctx = cbdata; + struct walk_tree_context *walk = data; // match_baselen is -1 when no path was given, which no length equals. - if (walk_tree_ctx->match_baselen >= 0 && - base->len == (size_t)walk_tree_ctx->match_baselen) { + if (walk->match_baselen >= 0 && + base->len == (size_t)walk->match_baselen) { if (S_ISREG(mode)) { - struct strbuf buffer = STRBUF_INIT; - strbuf_addbuf(&buffer, base); - strbuf_addstr(&buffer, pathname); - print_object(oid, buffer.buf, pathname, - walk_tree_ctx->curr_rev); - strbuf_release(&buffer); - walk_tree_ctx->state = 1; + struct strbuf fullpath = STRBUF_INIT; + + strbuf_addbuf(&fullpath, base); + strbuf_addstr(&fullpath, pathname); + print_blame_page(oid, fullpath.buf, pathname, + walk->rev); + strbuf_release(&fullpath); + walk->found = TARGET_FILE; } else if (S_ISDIR(mode)) { - walk_tree_ctx->state = 2; + walk->found = TARGET_FOLDER; } } else if (base->len < INT_MAX - && (int)base->len > walk_tree_ctx->match_baselen) { - walk_tree_ctx->state = 2; + && (int)base->len > walk->match_baselen) { + walk->found = TARGET_FOLDER; } else if (S_ISDIR(mode)) { return READ_TREE_RECURSIVE; } @@ -319,9 +376,10 @@ static int walk_tree(const struct object_id *oid, struct strbuf *base, static int basedir_len(const char *path) { - const char *p = strrchr(path, '/'); - if (p) - return p - path + 1; + const char *slash = strrchr(path, '/'); + + if (slash) + return slash - path + 1; return 0; } @@ -330,16 +388,20 @@ void cgit_print_blame(void) const char *rev = ctx.qry.oid; struct object_id oid; struct commit *commit; + int path_len = ctx.qry.path ? strlen(ctx.qry.path) : 0; + // nowildcard_len matches len so git treats the path as literal rather + // than as a glob, which is what a request naming one file means. struct pathspec_item path_items = { .match = ctx.qry.path, - .len = ctx.qry.path ? strlen(ctx.qry.path) : 0 + .len = path_len, + .nowildcard_len = path_len }; struct pathspec paths = { .nr = 1, .items = &path_items }; - struct walk_tree_context walk_tree_ctx = { - .state = 0 + struct walk_tree_context walk = { + .found = TARGET_MISSING }; if (!rev) @@ -357,17 +419,17 @@ void cgit_print_blame(void) return; } - walk_tree_ctx.curr_rev = xstrdup(rev); - walk_tree_ctx.match_baselen = (path_items.match) ? - basedir_len(path_items.match) : -1; + walk.rev = xstrdup(rev); + walk.match_baselen = path_items.match ? + basedir_len(path_items.match) : -1; read_tree(the_repository, repo_get_commit_tree(the_repository, commit), - &paths, walk_tree, &walk_tree_ctx); - if (!walk_tree_ctx.state) + &paths, walk_tree, &walk); + if (walk.found == TARGET_MISSING) cgit_print_error_page(404, "Not found", "Not found"); - else if (walk_tree_ctx.state == 2) + else if (walk.found == TARGET_FOLDER) cgit_print_error_page(404, "No blame for folders", "Blame is not available for folders."); - free(walk_tree_ctx.curr_rev); + free(walk.rev); } diff --git a/source/ui-blame.h b/source/ui-blame.h index 5b97e03..f406698 100644 --- a/source/ui-blame.h +++ b/source/ui-blame.h @@ -1,6 +1,13 @@ -#ifndef UI_BLAME_H -#define UI_BLAME_H +/* + * The blame page, which pairs every line of a file with the commit that last + * changed it. It is one repository command among those in cmd.c, reached with + * a revision and a path in the query string, and a repository only offers it + * when enable-blame is set. + */ + +#ifndef CGIT_UI_BLAME_H +#define CGIT_UI_BLAME_H extern void cgit_print_blame(void); -#endif /* UI_BLAME_H */ +#endif // CGIT_UI_BLAME_H diff --git a/source/ui-blob.c b/source/ui-blob.c index 92edcf1..beb968b 100644 --- a/source/ui-blob.c +++ b/source/ui-blob.c @@ -1,16 +1,17 @@ -/* ui-blob.c: show blob content - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * Reads a blob out of the object database and writes its bytes to the client. + * The blob is named either directly by object id or by a path walked out of a + * commit's tree. The blob page serves those bytes as the whole response, under + * headers that stop a browser treating repository content as markup, while the + * summary page instead drops a readme's contents into a page it is already + * building. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-blob.h" #include "html.h" +#include "ui-blob.h" #include "ui-shared.h" struct walk_tree_context { @@ -20,93 +21,110 @@ struct walk_tree_context { unsigned int file_only:1; }; +/* + * read_tree reads the return value as a direction rather than a status, so + * READ_TREE_RECURSIVE means step into this entry and zero means step over it. + */ static int walk_tree(const struct object_id *oid, struct strbuf *base, - const char *pathname, unsigned mode, void *cbdata) + const char *pathname, unsigned mode, void *context) { - struct walk_tree_context *walk_tree_ctx = cbdata; + struct walk_tree_context *walk = context; - if (walk_tree_ctx->file_only && !S_ISREG(mode)) + if (walk->file_only && !S_ISREG(mode)) return READ_TREE_RECURSIVE; - if (strncmp(base->buf, walk_tree_ctx->match_path, base->len) - || strcmp(walk_tree_ctx->match_path + base->len, pathname)) + if (strncmp(base->buf, walk->match_path, base->len) + || strcmp(walk->match_path + base->len, pathname)) return READ_TREE_RECURSIVE; - oidcpy(walk_tree_ctx->matched_oid, oid); - walk_tree_ctx->found_path = 1; + oidcpy(walk->matched_oid, oid); + walk->found_path = 1; return 0; } -int cgit_ref_path_exists(const char *path, const char *ref, int file_only) +/* + * oid comes in naming a commit and goes out naming the blob found at path, + * which both callers rely on. The path has to be writable because a pathspec + * item does not hold a const string. + */ +static int find_path_oid(struct object_id *oid, char *path, int file_only) { - struct object_id oid; - unsigned long size; - struct pathspec_item path_items = { - .match = xstrdup(path), - .len = strlen(path) + struct commit *commit = lookup_commit_reference(the_repository, oid); + // nowildcard_len matching len makes git treat the path as literal + // rather than as a glob. + struct pathspec_item item = { + .match = path, + .len = strlen(path), + .nowildcard_len = strlen(path) }; struct pathspec paths = { .nr = 1, - .items = &path_items + .items = &item }; - struct walk_tree_context walk_tree_ctx = { + struct walk_tree_context walk = { .match_path = path, - .matched_oid = &oid, + .matched_oid = oid, .found_path = 0, .file_only = file_only }; + read_tree(the_repository, repo_get_commit_tree(the_repository, commit), + &paths, walk_tree, &walk); + return walk.found_path; +} + +/* + * Callers ask before reading the object, because the point of the limit is to + * keep a huge blob out of memory rather than to notice it once it is already + * there. + */ +static int over_size_limit(unsigned long size) +{ + return ctx.cfg.max_blob_size && + size / 1024 > (unsigned long)ctx.cfg.max_blob_size; +} + +int cgit_ref_path_exists(const char *path, const char *ref, int file_only) +{ + struct object_id oid; + unsigned long size; + char *path_copy = xstrdup(path); + int found = 0; + if (repo_get_oid(the_repository, ref, &oid)) goto done; if (odb_read_object_info(the_repository->objects, &oid, &size) != OBJ_COMMIT) goto done; - read_tree(the_repository, - repo_get_commit_tree(the_repository, lookup_commit_reference(the_repository, &oid)), - &paths, walk_tree, &walk_tree_ctx); + found = find_path_oid(&oid, path_copy, file_only); done: - free(path_items.match); - return walk_tree_ctx.found_path; + free(path_copy); + return found; } int cgit_print_file(char *path, const char *head, int file_only, int html_escape) { struct object_id oid; enum object_type type; - char *buf; unsigned long size; - struct commit *commit; - struct pathspec_item path_items = { - .match = path, - .len = strlen(path) - }; - struct pathspec paths = { - .nr = 1, - .items = &path_items - }; - struct walk_tree_context walk_tree_ctx = { - .match_path = path, - .matched_oid = &oid, - .found_path = 0, - .file_only = file_only - }; + char *buf; if (repo_get_oid(the_repository, head, &oid)) return -1; type = odb_read_object_info(the_repository->objects, &oid, &size); if (type == OBJ_COMMIT) { - commit = lookup_commit_reference(the_repository, &oid); - read_tree(the_repository, repo_get_commit_tree(the_repository, commit), - &paths, walk_tree, &walk_tree_ctx); - if (!walk_tree_ctx.found_path) + if (!find_path_oid(&oid, path, file_only)) return -1; type = odb_read_object_info(the_repository->objects, &oid, &size); } if (type == OBJ_BAD) return -1; - if (ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size) + if (over_size_limit(size)) return -1; buf = odb_read_object(the_repository->objects, &oid, &type, &size); if (!buf) return -1; + + // html_txt wants a terminated string, and git leaves a spare byte + // past every object it reads, so this write stays in the allocation. buf[size] = '\0'; if (html_escape) html_txt(buf); @@ -120,23 +138,8 @@ void cgit_print_blob(const char *hex, char *path, const char *head, int file_onl { struct object_id oid; enum object_type type; - char *buf; unsigned long size; - struct commit *commit; - struct pathspec_item path_items = { - .match = path, - .len = path ? strlen(path) : 0 - }; - struct pathspec paths = { - .nr = 1, - .items = &path_items - }; - struct walk_tree_context walk_tree_ctx = { - .match_path = path, - .matched_oid = &oid, - .found_path = 0, - .file_only = file_only - }; + char *buf; if (hex) { if (get_oid_hex(hex, &oid)) { @@ -154,11 +157,8 @@ void cgit_print_blob(const char *hex, char *path, const char *head, int file_onl type = odb_read_object_info(the_repository->objects, &oid, &size); - if ((!hex) && type == OBJ_COMMIT && path) { - commit = lookup_commit_reference(the_repository, &oid); - read_tree(the_repository, repo_get_commit_tree(the_repository, commit), - &paths, walk_tree, &walk_tree_ctx); - if (!walk_tree_ctx.found_path) { + if (!hex && type == OBJ_COMMIT && path) { + if (!find_path_oid(&oid, path, file_only)) { cgit_print_error_page(404, "Not found", "Path not found: %s", path); return; @@ -172,8 +172,7 @@ void cgit_print_blob(const char *hex, char *path, const char *head, int file_onl return; } - /* Reject an oversized object before reading it whole into memory. */ - if (ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size) { + if (over_size_limit(size)) { cgit_print_error_page(413, "Too large", "Object size (%luKB) exceeds limit (%dKB)", size / 1024, ctx.cfg.max_blob_size); @@ -194,6 +193,10 @@ void cgit_print_blob(const char *hex, char *path, const char *head, int file_onl ctx.page.mimetype = "text/plain"; ctx.page.filename = path; + // The bytes are whatever the repository holds, so the browser is told + // not to guess a type of its own from them and not to load anything + // they reference. Both must go out before cgit_print_http_headers, + // which closes the header block. html("X-Content-Type-Options: nosniff\n"); html("Content-Security-Policy: default-src 'none'\n"); cgit_print_http_headers(); diff --git a/source/ui-blob.h b/source/ui-blob.h index efbc94e..eb3dbd1 100644 --- a/source/ui-blob.h +++ b/source/ui-blob.h @@ -1,8 +1,18 @@ -#ifndef UI_BLOB_H -#define UI_BLOB_H +/* + * Declarations for reading a blob out of the object database and writing it to + * the client. A caller either hands the whole response over to the blob's own + * bytes and headers, or drops a file's contents into a page cgit is already + * building, as the summary page does for a readme. + */ -extern int cgit_ref_path_exists(const char *path, const char *ref, int file_only); -extern int cgit_print_file(char *path, const char *head, int file_only, int html_escape); -extern void cgit_print_blob(const char *hex, char *path, const char *head, int file_only); +#ifndef CGIT_UI_BLOB_H +#define CGIT_UI_BLOB_H -#endif /* UI_BLOB_H */ +extern int cgit_ref_path_exists(const char *path, const char *ref, + int file_only); +extern int cgit_print_file(char *path, const char *head, int file_only, + int html_escape); +extern void cgit_print_blob(const char *hex, char *path, const char *head, + int file_only); + +#endif // CGIT_UI_BLOB_H diff --git a/source/ui-clone.c b/source/ui-clone.c index 71bb2ed..70e4446 100644 --- a/source/ui-clone.c +++ b/source/ui-clone.c @@ -1,40 +1,64 @@ -/* ui-clone.c: functions for http cloning, based on - * git's http-backend.c by Shawn O. Pearce - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The dumb HTTP transport, which lets a client clone by fetching plain files + * rather than by talking to a server side helper. It answers the requests + * such a client makes, the ref listing under info, the loose objects and pack + * files under objects, and HEAD, each written as raw bytes rather than as a + * page. The two listings a clone starts from are built per request, so a + * repository serves without anyone having run git update-server-info over it. + * The shape of it all follows git's own http-backend. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-clone.h" #include "html.h" +#include "ui-clone.h" #include "ui-shared.h" -#include "packfile.h" -static int print_ref_info(const struct reference *ref, void *cb_data) +/* + * A client reading this listing expects an annotated tag to be followed by a + * second line, marked with ^{}, naming the object the tag peels to, the + * format git update-server-info writes into info/refs. + */ +static int print_ref(const struct reference *ref, void *cb_data) { struct object *obj; - if (!(obj = parse_object(the_repository, ref->oid))) + obj = parse_object(the_repository, ref->oid); + if (!obj) return 0; htmlf("%s\t%s\n", oid_to_hex(ref->oid), ref->name); if (obj->type == OBJ_TAG) { - if (!(obj = deref_tag(the_repository, obj, ref->name, 0))) + obj = deref_tag(the_repository, obj, ref->name, 0); + if (!obj) return 0; htmlf("%s\t%s^{}\n", oid_to_hex(&obj->oid), ref->name); } return 0; } +/* + * A path ending in a slash is handed back whole, where git's own + * pack_basename would return the empty string. + */ +static const char *last_path_component(const char *path) +{ + const char *slash = strrchr(path, '/'); + + if (slash && slash[1] != '\0') + return slash + 1; + return path; +} + +/* + * Only the packs this repository holds itself are listed, because one + * borrowed from an alternate is not reachable below this URL and a client + * told about it would come back for a file that is not there. + */ static void print_pack_info(void) { struct odb_source *source; - char *offset; ctx.page.mimetype = "text/plain"; ctx.page.filename = "objects/info/packs"; @@ -42,21 +66,38 @@ static void print_pack_info(void) odb_reprepare(the_repository->objects); for (source = the_repository->objects->sources; source; source = source->next) { struct odb_source_files *files = odb_source_files_downcast(source); - struct packfile_list_entry *e; - for (e = files->packed->packs.head; e; e = e->next) { - struct packed_git *p = e->pack; - if (p->pack_local) { - offset = strrchr(p->pack_name, '/'); - if (offset && offset[1] != '\0') - ++offset; - else - offset = p->pack_name; - htmlf("P %s\n", offset); - } + struct packfile_list_entry *entry; + // Asked for through the accessor rather than read off the list, + // because a pack the multi-pack-index already covers joins that + // list only when the accessor loads it, and reading the field + // directly leaves those packs unfindable. + for (entry = packfile_store_get_packs(files->packed); entry; + entry = entry->next) { + struct packed_git *pack = entry->pack; + if (pack->pack_local) + htmlf("P %s\n", + last_path_component(pack->pack_name)); } } } +/* + * Beyond the obvious directory traversal, the strict character set heads off + * other funny business, for example the file name quirks of the Cygwin port. + */ +static int path_is_safe(const char *path) +{ + const char *p; + + for (p = path; *p; ++p) { + if (*p == '.' && *(p + 1) == '.') + return 0; + if (!isalnum((unsigned char)*p) && *p != '/' && *p != '.' && *p != '-') + return 0; + } + return 1; +} + static void send_file(const char *path) { struct stat st; @@ -75,6 +116,8 @@ static void send_file(const char *path) return; } ctx.page.mimetype = "application/octet-stream"; + // Offer the file under its path inside the repository, so the layout + // of the server's disk stays out of the download name. ctx.page.filename = path; skip_prefix(path, ctx.repo->path, &ctx.page.filename); skip_prefix(ctx.page.filename, "/", &ctx.page.filename); @@ -93,12 +136,12 @@ void cgit_clone_info(void) ctx.page.filename = "info/refs"; cgit_print_http_headers(); refs_for_each_ref(get_main_ref_store(the_repository), - print_ref_info, NULL); + print_ref, NULL); } void cgit_clone_objects(void) { - char *p, *path; + char *path; if (!ctx.qry.path) goto err; @@ -108,16 +151,8 @@ void cgit_clone_objects(void) return; } - /* Avoid directory traversal by forbidding "..", but also work around - * other funny business by just specifying a fairly strict format. For - * example, now we don't have to stress out about the Cygwin port. - */ - for (p = ctx.qry.path; *p; ++p) { - if (*p == '.' && *(p + 1) == '.') - goto err; - if (!isalnum((unsigned char)*p) && *p != '/' && *p != '.' && *p != '-') - goto err; - } + if (!path_is_safe(ctx.qry.path)) + goto err; path = repo_git_path(the_repository, "objects/%s", ctx.qry.path); send_file(path); diff --git a/source/ui-clone.h b/source/ui-clone.h index 3e460a3..71deff2 100644 --- a/source/ui-clone.h +++ b/source/ui-clone.h @@ -1,8 +1,15 @@ -#ifndef UI_CLONE_H -#define UI_CLONE_H +/* + * The entry points that serve a dumb HTTP clone, one for each kind of file + * such a client fetches. Each writes raw bytes rather than a page, and cmd.c + * flags them as clone requests so a site with http cloning turned off can + * refuse all of them in one place. + */ -void cgit_clone_info(void); -void cgit_clone_objects(void); -void cgit_clone_head(void); +#ifndef CGIT_UI_CLONE_H +#define CGIT_UI_CLONE_H -#endif /* UI_CLONE_H */ +extern void cgit_clone_info(void); +extern void cgit_clone_objects(void); +extern void cgit_clone_head(void); + +#endif // CGIT_UI_CLONE_H diff --git a/source/ui-commit.c b/source/ui-commit.c index 4d6d4da..f9051c1 100644 --- a/source/ui-commit.c +++ b/source/ui-commit.c @@ -1,29 +1,93 @@ -/* ui-commit.c: generate commit view - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The commit page, which shows one commit on its own and is where the log and + * the ref listings link. It renders the idents, the object ids, the message + * and any note, then hands off to ui-diff.c for the diff against the first + * parent. Free text and addresses are written through the repository's commit + * and email filters, so a site can rewrite either on the way out. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-commit.h" +#include "filter.h" #include "html.h" -#include "ui-shared.h" +#include "parsing.h" +#include "shared.h" +#include "ui-commit.h" #include "ui-diff.h" #include "ui-log.h" +#include "ui-shared.h" -void cgit_print_commit(char *hex, const char *prefix) +// The diff below the message is always taken against the first parent alone, +// which says little about a merge of many branches, so a commit with this +// many parents or more is shown without a diff at all. +#define OCTOPUS_PARENTS 3 + +static void print_ident_row(const char *role, const char *name, + const char *email, timestamp_t date, int tz) +{ + htmlf("<tr><th>%s</th><td>", role); + cgit_open_filter(ctx.repo->email_filter, email, "commit"); + html_txt(name); + if (!ctx.cfg.noplainemail) { + html(" "); + html_txt(email); + } + cgit_close_filter(ctx.repo->email_filter); + html("</td><td class='right'>"); + html_txt(show_date(date, tz, cgit_date_mode(DATE_ISO8601))); + html("</td></tr>\n"); +} + +static int print_parent_rows(struct commit *commit, const char *rev, + const char *prefix) { - struct commit *commit, *parent; - struct commitinfo *info, *parent_info; struct commit_list *p; + struct commit *parent; + const char *parent_hex, *label; + int parents = 0; + + for (p = commit->parents; p; p = p->next) { + parent = lookup_commit_reference(the_repository, + &p->item->object.oid); + if (!parent) { + html("<tr><td colspan='3'>"); + cgit_print_error("Error reading parent commit"); + html("</td></tr>"); + continue; + } + html("<tr><th>parent</th>" + "<td colspan='2' class='oid'>"); + parent_hex = label = oid_to_hex(&p->item->object.oid); + if (ctx.repo->enable_subject_links) + label = cgit_parse_commit(parent)->subject; + cgit_commit_link(label, NULL, NULL, ctx.qry.head, parent_hex, + prefix); + html(" ("); + cgit_diff_link("diff", NULL, NULL, ctx.qry.head, rev, + oid_to_hex(&p->item->object.oid), prefix); + html(")</td></tr>"); + parents++; + } + return parents; +} + +static void print_filtered_text(const char *text) +{ + cgit_open_filter(ctx.repo->commit_filter); + html_txt(text); + cgit_close_filter(ctx.repo->commit_filter); +} + +void cgit_print_commit(char *hex, const char *prefix) +{ + struct commit *commit; + struct commitinfo *info; struct strbuf notes = STRBUF_INIT; struct object_id oid; - char *tmp, *tmp2; - int parents = 0; + const char *commit_hex, *first_parent; + char *tree_rev; + int parents; if (!hex) hex = ctx.qry.head; @@ -48,102 +112,66 @@ void cgit_print_commit(char *hex, const char *prefix) ctx.page.title = cgit_fmtalloc("%s - %s", info->subject, ctx.page.title); cgit_print_layout_start(); cgit_print_diff_ctrls(); + html("<table summary='commit info' class='commit-info'>\n"); - html("<tr><th>author</th><td>"); - cgit_open_filter(ctx.repo->email_filter, info->author_email, "commit"); - html_txt(info->author); - if (!ctx.cfg.noplainemail) { - html(" "); - html_txt(info->author_email); - } - cgit_close_filter(ctx.repo->email_filter); - html("</td><td class='right'>"); - html_txt(show_date(info->author_date, info->author_tz, - cgit_date_mode(DATE_ISO8601))); - html("</td></tr>\n"); - html("<tr><th>committer</th><td>"); - cgit_open_filter(ctx.repo->email_filter, info->committer_email, "commit"); - html_txt(info->committer); - if (!ctx.cfg.noplainemail) { - html(" "); - html_txt(info->committer_email); - } - cgit_close_filter(ctx.repo->email_filter); - html("</td><td class='right'>"); - html_txt(show_date(info->committer_date, info->committer_tz, - cgit_date_mode(DATE_ISO8601))); - html("</td></tr>\n"); + print_ident_row("author", info->author, info->author_email, + info->author_date, info->author_tz); + print_ident_row("committer", info->committer, info->committer_email, + info->committer_date, info->committer_tz); + html("<tr><th>commit</th><td colspan='2' class='oid'>"); - tmp = oid_to_hex(&commit->object.oid); - cgit_commit_link(tmp, NULL, NULL, ctx.qry.head, tmp, prefix); + commit_hex = oid_to_hex(&commit->object.oid); + cgit_commit_link(commit_hex, NULL, NULL, ctx.qry.head, commit_hex, + prefix); html(" ("); - cgit_patch_link("patch", NULL, NULL, NULL, tmp, prefix); + cgit_patch_link("patch", NULL, NULL, NULL, commit_hex, prefix); html(")</td></tr>\n"); + html("<tr><th>tree</th><td colspan='2' class='oid'>"); - tmp = xstrdup(hex); + tree_rev = xstrdup(hex); cgit_tree_link(oid_to_hex(get_commit_tree_oid(commit)), NULL, NULL, - ctx.qry.head, tmp, NULL); + ctx.qry.head, tree_rev, NULL); if (prefix) { html(" /"); - cgit_tree_link(prefix, NULL, NULL, ctx.qry.head, tmp, prefix); + cgit_tree_link(prefix, NULL, NULL, ctx.qry.head, tree_rev, + prefix); } - free(tmp); + free(tree_rev); html("</td></tr>\n"); - for (p = commit->parents; p; p = p->next) { - parent = lookup_commit_reference(the_repository, &p->item->object.oid); - if (!parent) { - html("<tr><td colspan='3'>"); - cgit_print_error("Error reading parent commit"); - html("</td></tr>"); - continue; - } - html("<tr><th>parent</th>" - "<td colspan='2' class='oid'>"); - tmp = tmp2 = oid_to_hex(&p->item->object.oid); - if (ctx.repo->enable_subject_links) { - parent_info = cgit_parse_commit(parent); - tmp2 = parent_info->subject; - } - cgit_commit_link(tmp2, NULL, NULL, ctx.qry.head, tmp, prefix); - html(" ("); - cgit_diff_link("diff", NULL, NULL, ctx.qry.head, hex, - oid_to_hex(&p->item->object.oid), prefix); - html(")</td></tr>"); - parents++; - } + + parents = print_parent_rows(commit, hex, prefix); + if (ctx.repo->snapshots) { html("<tr><th>download</th><td colspan='2' class='oid'>"); cgit_print_snapshot_links(ctx.repo, hex, "<br/>"); html("</td></tr>"); } html("</table>\n"); + html("<div class='commit-subject'>"); - cgit_open_filter(ctx.repo->commit_filter); - html_txt(info->subject); - cgit_close_filter(ctx.repo->commit_filter); + print_filtered_text(info->subject); cgit_print_commit_decorations(commit); html("</div>"); html("<div class='commit-msg'>"); - cgit_open_filter(ctx.repo->commit_filter); - html_txt(info->msg); - cgit_close_filter(ctx.repo->commit_filter); + print_filtered_text(info->msg); html("</div>"); if (notes.len != 0) { html("<div class='notes-header'>Notes</div>"); html("<div class='notes'>"); - cgit_open_filter(ctx.repo->commit_filter); - html_txt(notes.buf); - cgit_close_filter(ctx.repo->commit_filter); + print_filtered_text(notes.buf); html("</div>"); html("<div class='notes-footer'></div>"); } - if (parents < 3) { + + if (parents < OCTOPUS_PARENTS) { if (parents) - tmp = oid_to_hex(&commit->parents->item->object.oid); + first_parent = + oid_to_hex(&commit->parents->item->object.oid); else - tmp = NULL; - cgit_print_diff(ctx.qry.oid, tmp, prefix, 0, 0); + first_parent = NULL; + cgit_print_diff(ctx.qry.oid, first_parent, prefix, 0, 0); } + strbuf_release(¬es); cgit_free_commitinfo(info); cgit_print_layout_end(); diff --git a/source/ui-commit.h b/source/ui-commit.h index 8198b4b..0cc0cb0 100644 --- a/source/ui-commit.h +++ b/source/ui-commit.h @@ -1,6 +1,12 @@ -#ifndef UI_COMMIT_H -#define UI_COMMIT_H +/* + * The commit page, which shows one commit on its own and is where the log and + * the ref listings link. It takes the commit to show plus an optional path + * that narrows the diff printed below the message. + */ + +#ifndef CGIT_UI_COMMIT_H +#define CGIT_UI_COMMIT_H extern void cgit_print_commit(char *hex, const char *prefix); -#endif /* UI_COMMIT_H */ +#endif // CGIT_UI_COMMIT_H diff --git a/source/ui-diff.c b/source/ui-diff.c index a0790f7..09e2790 100644 --- a/source/ui-diff.c +++ b/source/ui-diff.c This diff is too large to be rendered inline. View it on its own page. diff --git a/source/ui-diff.h b/source/ui-diff.h index 39264a1..67d33ba 100644 --- a/source/ui-diff.h +++ b/source/ui-diff.h @@ -1,5 +1,19 @@ -#ifndef UI_DIFF_H -#define UI_DIFF_H +/* + * The page that shows what one revision changed against another, and the + * options panel that sits above it. The side by side renderer in ui-ssdiff + * builds a link on every line number, so the two revisions being compared and + * the file pair being rendered are published here for it rather than kept + * private to the page. + */ + +#ifndef CGIT_UI_DIFF_H +#define CGIT_UI_DIFF_H + +#include "cgit.h" + +// Widest context the options panel offers, and so the widest a request may +// ask for, since the query string is clamped against it before the page runs. +#define MAX_DIFF_CONTEXT_LINES 40 extern void cgit_print_diff_ctrls(void); @@ -12,4 +26,4 @@ extern struct diff_filespec *cgit_get_current_new_file(void); extern struct object_id old_rev_oid[1]; extern struct object_id new_rev_oid[1]; -#endif /* UI_DIFF_H */ +#endif // CGIT_UI_DIFF_H diff --git a/source/ui-empty.c b/source/ui-empty.c index cb91fbc..56ca6e1 100644 --- a/source/ui-empty.c +++ b/source/ui-empty.c @@ -1,15 +1,14 @@ /* - * ui-empty.c: page shown for a repository that holds no commits - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) + * The page a repository with no commits falls back to, an error line and then + * the clone urls someone needs in order to push the first commit. cgit.c + * prints this in place of whatever page was asked for as soon as it finds + * there is no branch to resolve, so none of the ref driven pages ever run + * against an empty repository. */ #include "cgit.h" -#include "ui-empty.h" #include "html.h" +#include "ui-empty.h" #include "ui-shared.h" static void print_clone_url(const char *url) @@ -31,6 +30,8 @@ void cgit_print_empty_repo(void) { cgit_print_error("Repository seems to be empty"); + // cgit_add_clone_urls draws on just these two, so with neither set the + // table below would be a Clone heading with no rows. if (!ctx.repo->clone_url && !ctx.cfg.clone_prefix) return; diff --git a/source/ui-empty.h b/source/ui-empty.h index dda303a..25ae049 100644 --- a/source/ui-empty.h +++ b/source/ui-empty.h @@ -1,6 +1,13 @@ -#ifndef UI_EMPTY_H -#define UI_EMPTY_H +/* + * The page shown for a repository that holds no commits, an error line saying + * it looks empty followed by the clone urls a first push would use. It is not + * one of the commands in cmd.c, because cgit.c reaches it directly when a + * repository has no branch for the requested page to work from. + */ + +#ifndef CGIT_UI_EMPTY_H +#define CGIT_UI_EMPTY_H extern void cgit_print_empty_repo(void); -#endif /* UI_EMPTY_H */ +#endif // CGIT_UI_EMPTY_H diff --git a/source/ui-log.c b/source/ui-log.c index 5eb9a74..f80ed16 100644 --- a/source/ui-log.c +++ b/source/ui-log.c This diff is too large to be rendered inline. View it on its own page. diff --git a/source/ui-log.h b/source/ui-log.h index 562524b..ecfe85d 100644 --- a/source/ui-log.h +++ b/source/ui-log.h @@ -1,9 +1,16 @@ -#ifndef UI_LOG_H -#define UI_LOG_H +/* + * The listing of commits, shown by the log page and by the block of recent + * commits on a repository's summary page. The commit page borrows the + * decorations alone, so that branch and tag labels look the same wherever a + * commit is named. + */ + +#ifndef CGIT_UI_LOG_H +#define CGIT_UI_LOG_H extern void cgit_print_log(const char *tip, int ofs, int cnt, char *grep, char *pattern, const char *path, int pager, int commit_graph, int commit_sort); extern void cgit_print_commit_decorations(struct commit *commit); -#endif /* UI_LOG_H */ +#endif // CGIT_UI_LOG_H diff --git a/source/ui-patch.c b/source/ui-patch.c index 29432ac..6f75b6a 100644 --- a/source/ui-patch.c +++ b/source/ui-patch.c @@ -1,83 +1,107 @@ -/* ui-patch.c: generate patch view - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The patch page, which serves a commit or a range of commits as plain text + * in the mail format git format-patch writes, so that a change read in cgit + * can be fed straight to git am. It is one of the repository commands in + * cmd.c, taking the newer revision from id, the older one from id2, and an + * optional path that narrows the diff. A merge carries no single patch and so + * drops out of a range, and max-patch-count bounds how many commits one + * request may emit. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-patch.h" #include "html.h" +#include "ui-diff.h" +#include "ui-patch.h" #include "ui-shared.h" -/* two commit hashes with two dots in between and termination */ -#define REV_RANGE_LEN 2 * GIT_MAX_HEXSZ + 3 +// Room for two hex object ids, the two dots between them, and the null byte. +#define REV_RANGE_LEN (2 * GIT_MAX_HEXSZ + 3) -void cgit_print_patch(const char *new_rev, const char *old_rev, - const char *prefix) +// Where the format argument sits in the walk arguments below. +#define FORMAT_ARG 2 + +/* + * A root commit has no parent, so the old end is left null and the caller asks + * for that one commit instead. + */ +static int resolve_range(const char *new_rev, const char *old_rev, + struct object_id *new_oid, struct object_id *old_oid) { - struct rev_info rev; struct commit *commit; - struct object_id new_rev_oid, old_rev_oid; - char rev_range[REV_RANGE_LEN]; - const char *rev_argv[] = { NULL, "--reverse", "--format=email", rev_range, "--", prefix, NULL }; - int rev_argc = ARRAY_SIZE(rev_argv) - 1; - char *patchname; - if (!prefix) - rev_argc--; - - if (!new_rev) - new_rev = ctx.qry.head; - - if (repo_get_oid(the_repository, new_rev, &new_rev_oid)) { + if (repo_get_oid(the_repository, new_rev, new_oid)) { cgit_print_error_page(404, "Not found", "Bad object id: %s", new_rev); - return; + return -1; } - commit = lookup_commit_reference(the_repository, &new_rev_oid); + commit = lookup_commit_reference(the_repository, new_oid); if (!commit) { cgit_print_error_page(404, "Not found", "Bad commit reference: %s", new_rev); - return; + return -1; } if (old_rev) { - if (repo_get_oid(the_repository, old_rev, &old_rev_oid)) { + if (repo_get_oid(the_repository, old_rev, old_oid)) { cgit_print_error_page(404, "Not found", "Bad object id: %s", old_rev); - return; + return -1; } - if (!lookup_commit_reference(the_repository, &old_rev_oid)) { + if (!lookup_commit_reference(the_repository, old_oid)) { cgit_print_error_page(404, "Not found", "Bad commit reference: %s", old_rev); - return; + return -1; } } else if (commit->parents && commit->parents->item) { - oidcpy(&old_rev_oid, &commit->parents->item->object.oid); + oidcpy(old_oid, &commit->parents->item->object.oid); } else { - oidclr(&old_rev_oid, the_repository->hash_algo); + oidclr(old_oid, the_repository->hash_algo); } + return 0; +} + +void cgit_print_patch(const char *new_rev, const char *old_rev, + const char *prefix) +{ + struct rev_info rev; + struct commit *commit; + struct object_id new_oid, old_oid; + char rev_range[REV_RANGE_LEN]; + // setup_revisions reads these the way git reads a command line, so the + // first entry stands in for the program name and is skipped, and the + // array has to stay null terminated because the path after the double + // dash is picked up past the count. + const char *rev_argv[] = { NULL, "--reverse", "--format=email", + rev_range, "--", prefix, NULL }; + int rev_argc = ARRAY_SIZE(rev_argv) - 1; + + if (!prefix) + rev_argc--; + + if (!new_rev) + new_rev = ctx.qry.head; + + if (resolve_range(new_rev, old_rev, &new_oid, &old_oid)) + return; - if (is_null_oid(&old_rev_oid)) { - memcpy(rev_range, oid_to_hex(&new_rev_oid), the_hash_algo->hexsz + 1); + if (is_null_oid(&old_oid)) { + memcpy(rev_range, oid_to_hex(&new_oid), the_hash_algo->hexsz + 1); } else { - xsnprintf(rev_range, REV_RANGE_LEN, "%s..%s", oid_to_hex(&old_rev_oid), - oid_to_hex(&new_rev_oid)); + xsnprintf(rev_range, REV_RANGE_LEN, "%s..%s", + oid_to_hex(&old_oid), oid_to_hex(&new_oid)); } - patchname = cgit_fmt("%s.patch", rev_range); ctx.page.mimetype = "text/plain"; - ctx.page.filename = patchname; + ctx.page.filename = cgit_fmt("%s.patch", rev_range); cgit_print_http_headers(); if (ctx.cfg.noplainemail) { - rev_argv[2] = "--format=format:From %H Mon Sep 17 00:00:00 " - "2001%nFrom: %an%nDate: %aD%n%w(78,0,1)Subject: " - "%s%n%n%w(0)%b"; + rev_argv[FORMAT_ARG] = + "--format=format:From %H Mon Sep 17 00:00:00 " + "2001%nFrom: %an%nDate: %aD%n%w(78,0,1)Subject: " + "%s%n%n%w(0)%b"; } repo_init_revisions(the_repository, &rev, NULL); @@ -89,17 +113,25 @@ void cgit_print_patch(const char *new_rev, const char *old_rev, rev.diffopt.output_format |= DIFF_FORMAT_DIFFSTAT | DIFF_FORMAT_PATCH | DIFF_FORMAT_SUMMARY; if (prefix) - rev.diffopt.stat_sep = cgit_fmt("(limited to '%s')\n\n", prefix); + // Allocated rather than formatted into cgit_fmt's fixed + // buffer, because the path comes from the request and a long + // one would abort the process here, with the headers for a + // successful response already on the wire. + rev.diffopt.stat_sep = cgit_fmtalloc("(limited to '%s')\n\n", + prefix); setup_revisions(rev_argc, rev_argv, &rev, NULL); - // A single commit resolves to a parent..commit range, so this only - // bounds an explicit id/id2 range and keeps one request from - // emitting a patch for the entire history. + // A single commit resolves to a range starting at its parent, so this + // only ever cuts an explicit range short and keeps one request from + // emitting a patch for the whole history. if (ctx.cfg.max_patch_count > 0) rev.max_count = ctx.cfg.max_patch_count; prepare_revision_walk(&rev); while ((commit = get_revision(&rev)) != NULL) { log_tree_commit(&rev, commit); + // Two dashes and a space is the mail signature separator, so + // git am and mail readers drop the version note below rather + // than carrying it into the commit message. printf("-- \ncgit %s\n\n", cgit_version); } } diff --git a/source/ui-patch.h b/source/ui-patch.h index 7a6cacd..4dfe8bc 100644 --- a/source/ui-patch.h +++ b/source/ui-patch.h @@ -1,7 +1,14 @@ -#ifndef UI_PATCH_H -#define UI_PATCH_H +/* + * The patch page, which hands back a commit or a range of commits as a plain + * text mail patch that git am can apply. Any of the newer revision, the older + * one, and the path that limits the diff may be null, meaning the current + * head, the first parent, and the whole tree. + */ + +#ifndef CGIT_UI_PATCH_H +#define CGIT_UI_PATCH_H extern void cgit_print_patch(const char *new_rev, const char *old_rev, const char *prefix); -#endif /* UI_PATCH_H */ +#endif // CGIT_UI_PATCH_H diff --git a/source/ui-plain.c b/source/ui-plain.c index 8085382..9ce25b3 100644 --- a/source/ui-plain.c +++ b/source/ui-plain.c @@ -1,23 +1,53 @@ -/* ui-plain.c: functions for output of plain blobs by path - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The plain page, which hands over a repository's own bytes rather than + * rendering a view of them. A file is written out whole under a content type + * guessed from its name, though a repository that has not enabled html serving + * keeps only the types a browser will not act on. A directory, or a request + * carrying no path, is answered with a bare document of links to the entries + * below it rather than with one of cgit's themed pages. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-plain.h" #include "html.h" +#include "shared.h" +#include "ui-plain.h" #include "ui-shared.h" +/* + * A listing is opened by the entry that matched and closed only once the walk + * is over, so the end of the page has to tell the three cases apart. + */ +enum response { + RESPONSE_NONE, + RESPONSE_BLOB, + RESPONSE_LISTING +}; + struct walk_tree_context { - int match_baselen; - int match; + // Length of the directory part of the requested path, slash included, + // and -1 when no path was requested so that no base length can equal + // it. + int dir_len; + enum response response; }; +/* + * Everything below text/ and application/ can carry markup or script that a + * browser would run against the site, so only PDF is let back through. + */ +static int is_unsafe_type(const char *mimetype) +{ + return (starts_with(mimetype, "text/") || + starts_with(mimetype, "application/")) && + strcmp(mimetype, "application/pdf"); +} + +/* + * A nonzero return says the response has been written, error pages included, + * so the walk does not go on to report the path as missing. + */ static int print_object(const struct object_id *oid, const char *path) { enum object_type type; @@ -30,8 +60,10 @@ static int print_object(const struct object_id *oid, const char *path) return 1; } - /* Reject an oversized object before reading it whole into memory. */ - if (ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size) { + // The limit counts kilobytes and is checked before the read, so a huge + // blob is kept out of memory rather than noticed once it is there. + if (ctx.cfg.max_blob_size && + size / 1024 > (unsigned long)ctx.cfg.max_blob_size) { cgit_print_error_page(413, "Too large", "Object size (%luKB) exceeds limit (%dKB)", size / 1024, ctx.cfg.max_blob_size); @@ -48,13 +80,14 @@ static int print_object(const struct object_id *oid, const char *path) ctx.page.mimetype = mimetype; if (!ctx.repo->enable_html_serving) { + // The bytes are whatever the repository holds, so the browser + // is told not to guess a type of its own and not to load + // anything they reference. Both lines must go out before + // cgit_print_http_headers, which closes the header block. html("X-Content-Type-Options: nosniff\n"); html("Content-Security-Policy: default-src 'none'\n"); - if (mimetype) { - /* Built-in white list allows PDF and everything that isn't text/ and application/ */ - if ((!strncmp(mimetype, "text/", 5) || !strncmp(mimetype, "application/", 12)) && strcmp(mimetype, "application/pdf")) - ctx.page.mimetype = NULL; - } + if (mimetype && is_unsafe_type(mimetype)) + ctx.page.mimetype = NULL; } if (!ctx.page.mimetype) { @@ -75,7 +108,7 @@ static int print_object(const struct object_id *oid, const char *path) return 1; } -static char *buildpath(const char *base, int baselen, const char *path) +static char *build_path(const char *base, int baselen, const char *path) { if (path[0]) return cgit_fmtalloc("%.*s%s/", baselen, base, path); @@ -86,20 +119,25 @@ static char *buildpath(const char *base, int baselen, const char *path) static void print_dir(const struct object_id *oid, const char *base, int baselen, const char *path) { - char *fullpath, *slash; + char *fullpath; + const char *leading_slash; size_t len; - fullpath = buildpath(base, baselen, path); - slash = (fullpath[0] == '/' ? "" : "/"); + fullpath = build_path(base, baselen, path); + leading_slash = (fullpath[0] == '/' ? "" : "/"); ctx.page.etag = oid_to_hex(oid); cgit_print_http_headers(); - htmlf("<html><head><title>%s", slash); + htmlf("<html><head><title>%s", leading_slash); html_txt(fullpath); - htmlf("</title></head>\n<body>\n<h2>%s", slash); + htmlf("</title></head>\n<body>\n<h2>%s", leading_slash); html_txt(fullpath); html("</h2>\n<ul>\n"); len = strlen(fullpath); if (len > 1) { + char *slash; + + // Nothing left to drop means the parent is the root, which + // cgit_plain_link is asked for with a null path. fullpath[len - 1] = 0; slash = strrchr(fullpath, '/'); if (slash) @@ -121,13 +159,13 @@ static void print_dir_entry(const struct object_id *oid, const char *base, { char *fullpath; - fullpath = buildpath(base, baselen, path); + fullpath = build_path(base, baselen, path); if (!S_ISDIR(mode) && !S_ISGITLINK(mode)) fullpath[strlen(fullpath) - 1] = 0; html(" <li>"); - if (S_ISGITLINK(mode)) { + if (S_ISGITLINK(mode)) cgit_submodule_link(NULL, fullpath, oid_to_hex(oid)); - } else + else cgit_plain_link(path, NULL, NULL, ctx.qry.head, ctx.qry.oid, fullpath); html("</li>\n"); @@ -139,25 +177,27 @@ static void print_dir_tail(void) html(" </ul>\n</body></html>\n"); } +/* + * read_tree reads the return value as a direction rather than a status, so + * READ_TREE_RECURSIVE means step into this entry and zero means step over it. + */ static int walk_tree(const struct object_id *oid, struct strbuf *base, - const char *pathname, unsigned mode, void *cbdata) + const char *pathname, unsigned mode, void *context) { - struct walk_tree_context *walk_tree_ctx = cbdata; + struct walk_tree_context *walk = context; - // match_baselen is -1 when no path was given, which no length equals. - if (walk_tree_ctx->match_baselen >= 0 && - base->len == (size_t)walk_tree_ctx->match_baselen) { + if (walk->dir_len >= 0 && base->len == (size_t)walk->dir_len) { if (S_ISREG(mode) || S_ISLNK(mode)) { if (print_object(oid, pathname)) - walk_tree_ctx->match = 1; + walk->response = RESPONSE_BLOB; } else if (S_ISDIR(mode)) { print_dir(oid, base->buf, base->len, pathname); - walk_tree_ctx->match = 2; + walk->response = RESPONSE_LISTING; return READ_TREE_RECURSIVE; } - } else if (base->len < INT_MAX && (int)base->len > walk_tree_ctx->match_baselen) { + } else if (base->len < INT_MAX && (int)base->len > walk->dir_len) { print_dir_entry(oid, base->buf, base->len, pathname, mode); - walk_tree_ctx->match = 2; + walk->response = RESPONSE_LISTING; } else if (S_ISDIR(mode)) { return READ_TREE_RECURSIVE; } @@ -165,11 +205,11 @@ static int walk_tree(const struct object_id *oid, struct strbuf *base, return 0; } -static int basedir_len(const char *path) +static int dir_prefix_len(const char *path) { - const char *p = strrchr(path, '/'); - if (p) - return p - path + 1; + const char *slash = strrchr(path, '/'); + if (slash) + return slash - path + 1; return 0; } @@ -178,16 +218,22 @@ void cgit_print_plain(void) const char *rev = ctx.qry.oid; struct object_id oid; struct commit *commit; + int path_len = ctx.qry.path ? strlen(ctx.qry.path) : 0; + // A hand built pathspec leaves nowildcard_len at zero, which tells git + // the match may be a glob. It would then hand this walk every entry a + // pattern like * matches, and each one would be answered with its own + // set of HTTP headers inside the body of the first. struct pathspec_item path_items = { .match = ctx.qry.path, - .len = ctx.qry.path ? strlen(ctx.qry.path) : 0 + .len = path_len, + .nowildcard_len = path_len }; struct pathspec paths = { .nr = 1, .items = &path_items }; - struct walk_tree_context walk_tree_ctx = { - .match = 0 + struct walk_tree_context walk = { + .response = RESPONSE_NONE }; if (!rev) @@ -203,17 +249,19 @@ void cgit_print_plain(void) return; } if (!path_items.match) { + // The walk is never handed an entry for the top of the tree + // itself, so the listing it would have opened is opened here. path_items.match = ""; - walk_tree_ctx.match_baselen = -1; + walk.dir_len = -1; print_dir(get_commit_tree_oid(commit), "", 0, ""); - walk_tree_ctx.match = 2; + walk.response = RESPONSE_LISTING; + } else { + walk.dir_len = dir_prefix_len(path_items.match); } - else - walk_tree_ctx.match_baselen = basedir_len(path_items.match); read_tree(the_repository, repo_get_commit_tree(the_repository, commit), - &paths, walk_tree, &walk_tree_ctx); - if (!walk_tree_ctx.match) + &paths, walk_tree, &walk); + if (walk.response == RESPONSE_NONE) cgit_print_error_page(404, "Not found", "Not found"); - else if (walk_tree_ctx.match == 2) + else if (walk.response == RESPONSE_LISTING) print_dir_tail(); } diff --git a/source/ui-plain.h b/source/ui-plain.h index 5bff07b..ec1eb7d 100644 --- a/source/ui-plain.h +++ b/source/ui-plain.h @@ -1,6 +1,12 @@ -#ifndef UI_PLAIN_H -#define UI_PLAIN_H +/* + * The plain page, which serves a path out of a repository as the bytes the + * repository holds, or as a bare listing of links when the path names a + * directory. The tree and blame views link to it for the raw file. + */ + +#ifndef CGIT_UI_PLAIN_H +#define CGIT_UI_PLAIN_H extern void cgit_print_plain(void); -#endif /* UI_PLAIN_H */ +#endif // CGIT_UI_PLAIN_H diff --git a/source/ui-refs.c b/source/ui-refs.c index 84b929c..d179cdf 100644 --- a/source/ui-refs.c +++ b/source/ui-refs.c @@ -1,33 +1,38 @@ -/* ui-refs.c: browse symbolic refs - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The branch and tag listings, shown as sections of the summary page and as + * the whole of a repository's refs page. Every row is one ref beside + * the commit or tag object it points at, ordered newest first, with branches + * then reordered by name unless branch-sort asks for age. A section longer + * than max-ref-count is cut short and ends in a link to a dedicated heads or + * tags page, which walks the same list one offset at a time. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-refs.h" +#include "filter.h" #include "html.h" +#include "shared.h" +#include "ui-refs.h" #include "ui-shared.h" -static inline int cmp_age(int age1, int age2) -{ - /* age1 and age2 are assumed to be non-negative */ - return age2 - age1; -} - -static int cmp_ref_name(const void *a, const void *b) -{ - struct refinfo *r1 = *(struct refinfo **)a; - struct refinfo *r2 = *(struct refinfo **)b; - - return strcmp(r1->refname, r2->refname); -} +/* + * The slice of a sorted ref list that one page shows, with end one past the + * last row. size is what a full page holds, so it also decides whether the + * list needs a pager. + */ +struct ref_page { + int size; + int start; + int end; +}; -static int get_ref_age(struct refinfo *ref) +/* + * The tagger date and the committer date live in a union in struct refinfo and + * only the member matching the object type is ever filled, so assuming a ref + * points at a commit reads past the end of the smaller struct. + */ +static timestamp_t ref_date(struct refinfo *ref) { if (!ref->object) return 0; @@ -40,22 +45,53 @@ static int get_ref_age(struct refinfo *ref) return 0; } -// tag and commit share a union in struct refinfo and only the member matching -// the object type is ever filled, so the date has to be reached through -// get_ref_age rather than by assuming a branch points at a commit. Reading the -// wrong member ran off the end of the smaller struct. -static int cmp_ref_age(const void *a, const void *b) +static int cmp_date(const void *a, const void *b) +{ + struct refinfo *ref1 = *(struct refinfo **)a; + struct refinfo *ref2 = *(struct refinfo **)b; + timestamp_t date1 = ref_date(ref1), date2 = ref_date(ref2); + + // Compared rather than subtracted, at the width the dates are stored + // at. A commit may carry any timestamp, so a difference that does not + // fit an int would leave qsort with a contradictory ordering. + if (date1 < date2) + return 1; + if (date1 > date2) + return -1; + return 0; +} + +static int cmp_name(const void *a, const void *b) { - struct refinfo *r1 = *(struct refinfo **)a; - struct refinfo *r2 = *(struct refinfo **)b; + struct refinfo *ref1 = *(struct refinfo **)a; + struct refinfo *ref2 = *(struct refinfo **)b; + + return strcmp(ref1->refname, ref2->refname); +} + +static void collect_branches(struct reflist *list) +{ + list->refs = NULL; + list->alloc = list->count = 0; + refs_for_each_branch_ref(get_main_ref_store(the_repository), + cgit_refs_cb, list); + if (ctx.repo->enable_remote_branches) + refs_for_each_remote_ref(get_main_ref_store(the_repository), + cgit_refs_cb, list); +} - return cmp_age(get_ref_age(r1), get_ref_age(r2)); +static void print_branch_header(void) +{ + html("<tr class='nohover'><th class='left'>Branch</th>" + "<th class='left'>Commit message</th>" + "<th class='left col-author'>Author</th>" + "<th class='left' colspan='2'>Age</th></tr>\n"); } static int print_branch(struct refinfo *ref) { struct commitinfo *info = ref->commit; - char *name = (char *)ref->refname; + const char *name = ref->refname; if (!info) return 1; @@ -80,6 +116,14 @@ static int print_branch(struct refinfo *ref) return 0; } +static void collect_tags(struct reflist *list) +{ + list->refs = NULL; + list->alloc = list->count = 0; + refs_for_each_tag_ref(get_main_ref_store(the_repository), + cgit_refs_cb, list); +} + static void print_tag_header(void) { html("<tr class='nohover'><th class='left'>Tag</th>" @@ -90,13 +134,15 @@ static void print_tag_header(void) static int print_tag(struct refinfo *ref) { - struct tag *tag = NULL; struct taginfo *info = NULL; - char *name = (char *)ref->refname; + const char *name = ref->refname; struct object *obj = ref->object; + // A lightweight tag has no tag object, so the author and age columns + // below fall back to the commit. if (obj->type == OBJ_TAG) { - tag = (struct tag *)obj; + struct tag *tag = (struct tag *)obj; + obj = tag->tagged; info = ref->tag; if (!info) @@ -141,16 +187,16 @@ static void print_refs_link(const char *path) html("</td></tr>"); } -/* Prev/next links for the dedicated branch and tag pages, so each - * category pages independently instead of sharing one endless page. */ static void print_ref_pager(int ofs, int pagesize, int count, const char *path) { char *url; html("<tr class='nohover'><td colspan='5' class='refs-pager'>"); if (ofs > 0) { + int prev_ofs = ofs > pagesize ? ofs - pagesize : 0; + url = cgit_pageurl(ctx.qry.repo, cgit_fmt("refs/%s", path), - cgit_fmt("ofs=%d", ofs > pagesize ? ofs - pagesize : 0)); + cgit_fmt("ofs=%d", prev_ofs)); html("<a href='"); html_attr(url); html("'>[prev]</a> "); @@ -169,132 +215,116 @@ static void print_ref_pager(int ofs, int pagesize, int count, const char *path) html("</td></tr>"); } -static void collect_branches(struct reflist *list) +static struct ref_page page_bounds(int pagesize, int count) { - list->refs = NULL; - list->alloc = list->count = 0; - refs_for_each_branch_ref(get_main_ref_store(the_repository), - cgit_refs_cb, list); - if (ctx.repo->enable_remote_branches) - refs_for_each_remote_ref(get_main_ref_store(the_repository), - cgit_refs_cb, list); -} + struct ref_page page; -static void print_branch_header(void) -{ - html("<tr class='nohover'><th class='left'>Branch</th>" - "<th class='left'>Commit message</th>" - "<th class='left col-author'>Author</th>" - "<th class='left' colspan='2'>Age</th></tr>\n"); + page.size = (pagesize <= 0 || pagesize > count) ? count : pagesize; + page.start = ctx.qry.ofs > 0 ? ctx.qry.ofs : 0; + if (page.start > count) + page.start = count; + page.end = page.start + page.size < count ? + page.start + page.size : count; + return page; } -void cgit_print_branches(int maxcount) +/* + * Unlike the capped sections, the whole list is sorted before a page is cut + * out of it, so a branch keeps its place no matter which page it lands on. + */ +static void print_branches_page(int pagesize) { struct reflist list; + struct ref_page page; int i; print_branch_header(); collect_branches(&list); - if (maxcount == 0 || maxcount > list.count) - maxcount = list.count; - - qsort(list.refs, list.count, sizeof(*list.refs), cmp_ref_age); + qsort(list.refs, list.count, sizeof(*list.refs), cmp_date); if (ctx.repo->branch_sort == 0) - qsort(list.refs, maxcount, sizeof(*list.refs), cmp_ref_name); + qsort(list.refs, list.count, sizeof(*list.refs), cmp_name); - for (i = 0; i < maxcount; i++) + page = page_bounds(pagesize, list.count); + + for (i = page.start; i < page.end; i++) print_branch(list.refs[i]); - if (maxcount < list.count) - print_refs_link("heads"); + if (page.size < list.count) + print_ref_pager(page.start, page.size, list.count, "heads"); cgit_free_reflist_inner(&list); } -void cgit_print_tags(int maxcount) +static void print_tags_page(int pagesize) { struct reflist list; + struct ref_page page; int i; - list.refs = NULL; - list.alloc = list.count = 0; - refs_for_each_tag_ref(get_main_ref_store(the_repository), - cgit_refs_cb, &list); + collect_tags(&list); if (list.count == 0) return; - qsort(list.refs, list.count, sizeof(*list.refs), cmp_ref_age); - if (!maxcount) - maxcount = list.count; - else if (maxcount > list.count) - maxcount = list.count; + qsort(list.refs, list.count, sizeof(*list.refs), cmp_date); + + page = page_bounds(pagesize, list.count); + print_tag_header(); - for (i = 0; i < maxcount; i++) + for (i = page.start; i < page.end; i++) print_tag(list.refs[i]); - if (maxcount < list.count) - print_refs_link("tags"); + if (page.size < list.count) + print_ref_pager(page.start, page.size, list.count, "tags"); cgit_free_reflist_inner(&list); } -/* The dedicated branch page lists everything, a page at a time. The - * whole list is name-sorted (or age-sorted per branch-sort) so the - * order is stable across pages. */ -static void print_branches_page(int pagesize) +void cgit_print_branches(int maxcount) { struct reflist list; - int i, ofs, end; + int i; print_branch_header(); collect_branches(&list); - qsort(list.refs, list.count, sizeof(*list.refs), cmp_ref_age); - if (ctx.repo->branch_sort == 0) - qsort(list.refs, list.count, sizeof(*list.refs), cmp_ref_name); + if (maxcount == 0 || maxcount > list.count) + maxcount = list.count; - if (pagesize <= 0 || pagesize > list.count) - pagesize = list.count; - ofs = ctx.qry.ofs > 0 ? ctx.qry.ofs : 0; - if (ofs > list.count) - ofs = list.count; - end = ofs + pagesize < list.count ? ofs + pagesize : list.count; + // The date sort covers the list, the name sort only the rows about to + // be shown, so the section holds the newest branches rather than the + // first ones by name. + qsort(list.refs, list.count, sizeof(*list.refs), cmp_date); + if (ctx.repo->branch_sort == 0) + qsort(list.refs, maxcount, sizeof(*list.refs), cmp_name); - for (i = ofs; i < end; i++) + for (i = 0; i < maxcount; i++) print_branch(list.refs[i]); - if (pagesize < list.count) - print_ref_pager(ofs, pagesize, list.count, "heads"); + if (maxcount < list.count) + print_refs_link("heads"); cgit_free_reflist_inner(&list); } -static void print_tags_page(int pagesize) +void cgit_print_tags(int maxcount) { struct reflist list; - int i, ofs, end; + int i; - list.refs = NULL; - list.alloc = list.count = 0; - refs_for_each_tag_ref(get_main_ref_store(the_repository), - cgit_refs_cb, &list); + collect_tags(&list); if (list.count == 0) return; - qsort(list.refs, list.count, sizeof(*list.refs), cmp_ref_age); - - if (pagesize <= 0 || pagesize > list.count) - pagesize = list.count; - ofs = ctx.qry.ofs > 0 ? ctx.qry.ofs : 0; - if (ofs > list.count) - ofs = list.count; - end = ofs + pagesize < list.count ? ofs + pagesize : list.count; - + qsort(list.refs, list.count, sizeof(*list.refs), cmp_date); + if (!maxcount) + maxcount = list.count; + else if (maxcount > list.count) + maxcount = list.count; print_tag_header(); - for (i = ofs; i < end; i++) + for (i = 0; i < maxcount; i++) print_tag(list.refs[i]); - if (pagesize < list.count) - print_ref_pager(ofs, pagesize, list.count, "tags"); + if (maxcount < list.count) + print_refs_link("tags"); cgit_free_reflist_inner(&list); } @@ -309,8 +339,6 @@ void cgit_print_refs(void) else if (ctx.qry.path && starts_with(ctx.qry.path, "tags")) print_tags_page(ctx.cfg.max_ref_count); else { - /* The combined page caps each section, with the [...] rows - * leading to the dedicated pages above. */ cgit_print_branches(ctx.cfg.max_ref_count); html("<tr class='nohover'><td colspan='5'> </td></tr>"); cgit_print_tags(ctx.cfg.max_ref_count); diff --git a/source/ui-refs.h b/source/ui-refs.h index 1d4a54a..c66050b 100644 --- a/source/ui-refs.h +++ b/source/ui-refs.h @@ -1,8 +1,25 @@ -#ifndef UI_REFS_H -#define UI_REFS_H +/* + * The branch and tag listings of a repository. The two section printers write + * rows into a table the caller has already opened, which is how the summary + * page carries a short version of each above its recent log. + */ +#ifndef CGIT_UI_REFS_H +#define CGIT_UI_REFS_H + +/* + * Rows for the newest branches or tags, at most maxcount of them, where a + * maxcount of zero means every one. A section that leaves refs out ends in a + * link to the matching form of the refs page. + */ extern void cgit_print_branches(int maxcount); extern void cgit_print_tags(int maxcount); + +/* + * The whole refs page. Without a path it holds a capped section of each kind, + * and with a path of heads or tags it becomes a page of that one kind alone, + * walked an offset at a time. + */ extern void cgit_print_refs(void); -#endif /* UI_REFS_H */ +#endif // CGIT_UI_REFS_H diff --git a/source/ui-repolist.c b/source/ui-repolist.c index e2c4abf..3bdae37 100644 --- a/source/ui-repolist.c +++ b/source/ui-repolist.c @@ -1,113 +1,137 @@ -/* ui-repolist.c: functions for generating the repolist page - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The repository index, which is the page a cgit site opens on, and the site + * readme the about page falls back to when a request names no repository. + * Every repository cgitrc registered is a candidate row, cut down to those + * matching the search terms and the url prefix the request carried, ordered + * either by the column the reader asked for or by section, and split into + * pages of max-repo-count rows. */ #include "cgit.h" -#include "ui-repolist.h" +#include "filter.h" #include "html.h" +#include "shared.h" +#include "ui-repolist.h" #include "ui-shared.h" +// A section heading spans the whole table, so this has to stay in step with +// the headings print_header_row emits. Name, description and idle are always +// there, owner and links are added when cgitrc asks for them. +#define BASE_COLUMNS 3 + +struct sort_column { + const char *name; + int (*cmp)(const void *a, const void *b); +}; + static time_t read_agefile(const char *path) { time_t result; size_t size; char *buf = NULL; - struct strbuf date_buf = STRBUF_INIT; + struct strbuf date = STRBUF_INIT; if (cgit_read_first_line(path, &buf, &size)) { free(buf); return 0; } - if (parse_date(buf, &date_buf) == 0) - result = strtoul(date_buf.buf, NULL, 10); + // parse_date writes the timestamp followed by its timezone offset, so + // only the seconds at the front are wanted here. + if (parse_date(buf, &date) == 0) + result = strtoul(date.buf, NULL, 10); else result = 0; free(buf); - strbuf_release(&date_buf); + strbuf_release(&date); return result; } +/* + * The returned name can point into head, which the caller has to keep alive + * for as long as it uses it. + */ +static const char *tip_branch(const struct cgit_repo *repo, struct strbuf *head) +{ + struct strbuf path = STRBUF_INIT; + const char *branch = repo->defbranch; + + if (branch) + return branch; + + strbuf_addf(&path, "%s/HEAD", repo->path); + if (strbuf_read_file(head, path.buf, 0) > 0) { + strbuf_rtrim(head); + if (!skip_prefix(head->buf, "ref: refs/heads/", &branch)) + branch = NULL; + // HEAD belongs to the repository, so a crafted target such as + // "ref: refs/heads/../../.." must not let the caller's stat + // walk outside it. Git forbids ".." in a ref name anyway. + if (branch && strstr(branch, "..")) + branch = NULL; + } + strbuf_release(&path); + + return branch ? branch : "master"; +} + static int get_repo_modtime(const struct cgit_repo *repo, time_t *mtime) { struct strbuf path = STRBUF_INIT; struct strbuf head = STRBUF_INIT; - struct stat s; - struct cgit_repo *r = (struct cgit_repo *)repo; + struct stat st; const char *branch; + // The comparators below are handed const repositories, but the answer + // is kept in the repository itself so one page stats it only once. + struct cgit_repo *writable = (struct cgit_repo *)repo; if (repo->mtime != -1) { *mtime = repo->mtime; return 1; } strbuf_addf(&path, "%s/%s", repo->path, ctx.cfg.agefile); - if (stat(path.buf, &s) == 0) { + if (stat(path.buf, &st) == 0) { *mtime = read_agefile(path.buf); if (*mtime) { - r->mtime = *mtime; + writable->mtime = *mtime; goto end; } } - /* Stat the tip of the default branch. Prefer a configured defbranch, - * otherwise read HEAD so a repo on "main" (or any branch name) is - * handled, not only "master". - */ - branch = repo->defbranch; - if (!branch) { - strbuf_reset(&path); - strbuf_addf(&path, "%s/HEAD", repo->path); - if (strbuf_read_file(&head, path.buf, 0) > 0) { - strbuf_rtrim(&head); - if (!skip_prefix(head.buf, "ref: refs/heads/", &branch)) - branch = NULL; - /* HEAD is repo-controlled, so a crafted target such as - * "ref: refs/heads/../../.." must not let the stat() - * below walk outside the repository. Git forbids ".." - * in ref names anyway. */ - if (branch && strstr(branch, "..")) - branch = NULL; - } - if (!branch) - branch = "master"; - } - + branch = tip_branch(repo, &head); strbuf_reset(&path); strbuf_addf(&path, "%s/refs/heads/%s", repo->path, branch); - if (stat(path.buf, &s) == 0) { - *mtime = s.st_mtime; - r->mtime = *mtime; + if (stat(path.buf, &st) == 0) { + *mtime = st.st_mtime; + writable->mtime = *mtime; goto end; } strbuf_reset(&path); - strbuf_addf(&path, "%s/%s", repo->path, "packed-refs"); - if (stat(path.buf, &s) == 0) { - *mtime = s.st_mtime; - r->mtime = *mtime; + strbuf_addf(&path, "%s/packed-refs", repo->path); + if (stat(path.buf, &st) == 0) { + *mtime = st.st_mtime; + writable->mtime = *mtime; goto end; } *mtime = 0; - r->mtime = *mtime; + writable->mtime = *mtime; end: strbuf_release(&path); strbuf_release(&head); - return (r->mtime != 0); + return (writable->mtime != 0); } static void print_modtime(struct cgit_repo *repo) { - time_t t; - if (get_repo_modtime(repo, &t)) - cgit_print_age(t, 0, -1); + time_t mtime; + + if (get_repo_modtime(repo, &mtime)) + cgit_print_age(mtime, 0, -1); } -static int is_match(struct cgit_repo *repo) +static int matches_search(struct cgit_repo *repo) { if (!ctx.qry.search) return 1; @@ -122,7 +146,7 @@ static int is_match(struct cgit_repo *repo) return 0; } -static int is_in_url(struct cgit_repo *repo) +static int matches_url(struct cgit_repo *repo) { if (!ctx.qry.url) return 1; @@ -135,7 +159,7 @@ static int is_visible(struct cgit_repo *repo) { if (repo->hide || repo->ignore) return 0; - if (!(is_match(repo) && is_in_url(repo))) + if (!(matches_search(repo) && matches_url(repo))) return 0; return 1; } @@ -151,14 +175,14 @@ static int any_repos_visible(void) return 0; } -// The index url is the same for every heading and every row, so the caller -// works it out once rather than building and freeing one per cell. -static void print_sort_header(const char *title, const char *sort, - const char *currenturl) +// currenturl is passed in because it is the same for every heading and every +// row, and working it out here would mean an allocation and a free per cell. +static void print_column_header(const char *title, const char *column, + const char *currenturl) { - htmlf("<th class='left col-%s'><a href='", sort); + htmlf("<th class='left col-%s'><a href='", column); html_attr(currenturl); - htmlf("?s=%s", sort); + htmlf("?s=%s", column); if (ctx.qry.search) { html("&q="); html_url_arg(ctx.qry.search); @@ -166,26 +190,83 @@ static void print_sort_header(const char *title, const char *sort, htmlf("'>%s</a></th>", title); } -static void print_header(const char *currenturl) +static void print_header_row(const char *currenturl) { html("<tr class='nohover'>"); - print_sort_header("Name", "name", currenturl); - print_sort_header("Description", "desc", currenturl); + print_column_header("Name", "name", currenturl); + print_column_header("Description", "desc", currenturl); if (ctx.cfg.enable_index_owner) - print_sort_header("Owner", "owner", currenturl); - print_sort_header("Idle", "idle", currenturl); + print_column_header("Owner", "owner", currenturl); + print_column_header("Idle", "idle", currenturl); if (ctx.cfg.enable_index_links) html("<th class='left col-links'>Links</th>"); html("</tr>\n"); } +static int section_changed(const char *section, const char *last) +{ + if (!section && !last) + return 0; + if (!section || !last) + return 1; + return strcmp(section, last) != 0; +} -static void print_pager(int items, int pagelen, char *search, char *sort) +static void print_section_row(const char *section, int columns) +{ + htmlf("<tr class='nohover-highlight'><td colspan='%d' class='reposection'>", + columns); + html_txt(section); + html("</td></tr>"); +} + +static void print_repo_row(const char *currenturl, int sublevel) +{ + char *repourl; + + htmlf("<tr><td class='col-name %s'>", + sublevel ? "sublevel-repo" : "toplevel-repo"); + cgit_summary_link(ctx.repo->name, NULL, NULL, NULL); + html("</td><td class='col-desc'>"); + repourl = cgit_repourl(ctx.repo->url); + html_link_open(repourl, NULL, NULL); + free(repourl); + if (html_ntxt(ctx.repo->desc, ctx.cfg.max_repodesc_len) < 0) + html("..."); + html_link_close(); + html("</td>"); + if (ctx.cfg.enable_index_owner) { + html("<td class='col-owner'>"); + html("<a href='"); + html_attr(currenturl); + html("?q="); + html_url_arg(ctx.repo->owner); + html("'>"); + html_txt(ctx.repo->owner); + html("</a>"); + html("</td>"); + } + html("<td class='col-idle'>"); + print_modtime(ctx.repo); + html("</td>"); + if (ctx.cfg.enable_index_links) { + html("<td class='col-links'>"); + cgit_summary_link("summary", NULL, "button", NULL); + cgit_log_link("log", NULL, "button", NULL, NULL, NULL, + 0, NULL, NULL, ctx.qry.showmsg, 0); + cgit_tree_link("tree", NULL, "button", NULL, NULL, NULL); + html("</td>"); + } + html("</tr>\n"); +} + +static void print_pager(int total, int pagelen, char *search, char *sort) { int i, ofs; char *class = NULL; + html("<ul class='pager'>"); - for (i = 0, ofs = 0; ofs < items; i++, ofs = i * pagelen) { + for (i = 0, ofs = 0; ofs < total; i++, ofs = i * pagelen) { class = (ctx.qry.ofs == ofs) ? "current" : NULL; html("<li>"); cgit_index_link(cgit_fmt("[%d]", i + 1), cgit_fmt("Page %d", i + 1), @@ -195,7 +276,7 @@ static void print_pager(int items, int pagelen, char *search, char *sort) html("</ul>"); } -static int cmp(const char *s1, const char *s2) +static int cmp_str(const char *s1, const char *s2) { if (s1 && s2) { if (ctx.cfg.case_sensitive_sort) @@ -210,44 +291,31 @@ static int cmp(const char *s1, const char *s2) return 0; } -static int sort_name(const void *a, const void *b) +static int cmp_name(const void *a, const void *b) { const struct cgit_repo *r1 = a; const struct cgit_repo *r2 = b; - return cmp(r1->name, r2->name); + return cmp_str(r1->name, r2->name); } -static int sort_desc(const void *a, const void *b) +static int cmp_desc(const void *a, const void *b) { const struct cgit_repo *r1 = a; const struct cgit_repo *r2 = b; - return cmp(r1->desc, r2->desc); + return cmp_str(r1->desc, r2->desc); } -static int sort_owner(const void *a, const void *b) +static int cmp_owner(const void *a, const void *b) { const struct cgit_repo *r1 = a; const struct cgit_repo *r2 = b; - return cmp(r1->owner, r2->owner); + return cmp_str(r1->owner, r2->owner); } -/* Resolve every repository's modification time up front. get_repo_modtime - * caches into the repo it is given, but qsort moves those structs around while - * it sorts, so a comparator that fills the cache loses most of what it stored - * and stats the same repository again and again. */ -static void resolve_modtimes(void) -{ - time_t t; - int i; - - for (i = 0; i < cgit_repolist.count; i++) - get_repo_modtime(&cgit_repolist.repos[i], &t); -} - -static int sort_idle(const void *a, const void *b) +static int cmp_idle(const void *a, const void *b) { const struct cgit_repo *r1 = a; const struct cgit_repo *r2 = b; @@ -256,8 +324,9 @@ static int sort_idle(const void *a, const void *b) t1 = t2 = 0; get_repo_modtime(r1, &t1); get_repo_modtime(r2, &t2); - /* Return the sign only; a truncated 64-bit time_t difference could - * flip and make the comparator inconsistent. */ + // Only the sign is returned, because a 64-bit difference truncated + // into an int could come back with the wrong sign and leave the + // ordering inconsistent. if (t2 > t1) return 1; if (t2 < t1) @@ -265,61 +334,70 @@ static int sort_idle(const void *a, const void *b) return 0; } -static int sort_section(const void *a, const void *b) +static int cmp_section(const void *a, const void *b) { const struct cgit_repo *r1 = a; const struct cgit_repo *r2 = b; int result; - result = cmp(r1->section, r2->section); + result = cmp_str(r1->section, r2->section); if (!result) { if (!strcmp(ctx.cfg.repository_sort, "age")) - result = sort_idle(r1, r2); + result = cmp_idle(r1, r2); if (!result) - result = cmp(r1->name, r2->name); + result = cmp_str(r1->name, r2->name); } return result; } -struct sortcolumn { - const char *name; - int (*fn)(const void *a, const void *b); -}; +/* + * get_repo_modtime caches into the repository it is handed, but qsort moves + * those structs around as it works, so a comparator left to fill the cache + * loses most of what it stored and stats the same repository over and over. + */ +static void resolve_modtimes(void) +{ + time_t t; + int i; + + for (i = 0; i < cgit_repolist.count; i++) + get_repo_modtime(&cgit_repolist.repos[i], &t); +} -static const struct sortcolumn sortcolumn[] = { - {"section", sort_section}, - {"name", sort_name}, - {"desc", sort_desc}, - {"owner", sort_owner}, - {"idle", sort_idle}, +static const struct sort_column sort_columns[] = { + {"section", cmp_section}, + {"name", cmp_name}, + {"desc", cmp_desc}, + {"owner", cmp_owner}, + {"idle", cmp_idle}, {NULL, NULL} }; static int sort_repolist(char *field) { - const struct sortcolumn *column; + const struct sort_column *column; - for (column = &sortcolumn[0]; column->name; column++) { + for (column = &sort_columns[0]; column->name; column++) { if (strcmp(field, column->name)) continue; - if (column->fn == sort_idle || column->fn == sort_section) + if (column->cmp == cmp_idle || column->cmp == cmp_section) resolve_modtimes(); qsort(cgit_repolist.repos, cgit_repolist.count, - sizeof(struct cgit_repo), column->fn); + sizeof(struct cgit_repo), column->cmp); return 1; } return 0; } - void cgit_print_repolist(void) { - int i, columns = 3, hits = 0, header = 0; char *last_section = NULL; - char *section; - char *repourl; char *currenturl; - int sorted = 0; + int columns = BASE_COLUMNS; + int column_sorted = 0; + int hits = 0; + int shown = 0; + int i; if (!any_repos_visible()) { cgit_print_error_page(404, "Not found", "No repositories found"); @@ -337,71 +415,38 @@ void cgit_print_repolist(void) cgit_print_pageheader(); if (ctx.qry.sort) - sorted = sort_repolist(ctx.qry.sort); + column_sorted = sort_repolist(ctx.qry.sort); else if (ctx.cfg.section_sort) sort_repolist("section"); currenturl = cgit_currenturl(); html("<table class='list nowrap repolist'>"); for (i = 0; i < cgit_repolist.count; i++) { + char *section; + ctx.repo = &cgit_repolist.repos[i]; if (!is_visible(ctx.repo)) continue; hits++; + // Rows outside the page are stepped over rather than broken + // out of, because the pager below is sized from the total. if (hits <= ctx.qry.ofs) continue; - if (hits > ctx.qry.ofs + ctx.cfg.max_repo_count) + // Written as a subtraction because max-repo-count of zero + // means unlimited and is held as INT_MAX, which any positive + // offset would overflow if it were added to instead. + if (hits - ctx.qry.ofs > ctx.cfg.max_repo_count) continue; - if (!header++) - print_header(currenturl); + if (!shown++) + print_header_row(currenturl); section = ctx.repo->section; if (section && !strcmp(section, "")) section = NULL; - if (!sorted && - ((last_section == NULL && section != NULL) || - (last_section != NULL && section == NULL) || - (last_section != NULL && section != NULL && - strcmp(section, last_section)))) { - htmlf("<tr class='nohover-highlight'><td colspan='%d' class='reposection'>", - columns); - html_txt(section); - html("</td></tr>"); + if (!column_sorted && section_changed(section, last_section)) { + print_section_row(section, columns); last_section = section; } - htmlf("<tr><td class='col-name %s'>", - !sorted && section ? "sublevel-repo" : "toplevel-repo"); - cgit_summary_link(ctx.repo->name, NULL, NULL, NULL); - html("</td><td class='col-desc'>"); - repourl = cgit_repourl(ctx.repo->url); - html_link_open(repourl, NULL, NULL); - free(repourl); - if (html_ntxt(ctx.repo->desc, ctx.cfg.max_repodesc_len) < 0) - html("..."); - html_link_close(); - html("</td>"); - if (ctx.cfg.enable_index_owner) { - html("<td class='col-owner'>"); - html("<a href='"); - html_attr(currenturl); - html("?q="); - html_url_arg(ctx.repo->owner); - html("'>"); - html_txt(ctx.repo->owner); - html("</a>"); - html("</td>"); - } - html("<td class='col-idle'>"); - print_modtime(ctx.repo); - html("</td>"); - if (ctx.cfg.enable_index_links) { - html("<td class='col-links'>"); - cgit_summary_link("summary", NULL, "button", NULL); - cgit_log_link("log", NULL, "button", NULL, NULL, NULL, - 0, NULL, NULL, ctx.qry.showmsg, 0); - cgit_tree_link("tree", NULL, "button", NULL, NULL, NULL); - html("</td>"); - } - html("</tr>\n"); + print_repo_row(currenturl, !column_sorted && section); } html("</table>"); if (hits > ctx.cfg.max_repo_count) diff --git a/source/ui-repolist.h b/source/ui-repolist.h index 1b6b322..294e13f 100644 --- a/source/ui-repolist.h +++ b/source/ui-repolist.h @@ -1,7 +1,14 @@ -#ifndef UI_REPOLIST_H -#define UI_REPOLIST_H +/* + * The two pages cgit renders without a repository in hand. One is the index + * of every repository the configuration knows about, which is where a request + * naming neither a repository nor a page lands. The other is the site wide + * readme, which the about page shows when no repository was selected. + */ + +#ifndef CGIT_UI_REPOLIST_H +#define CGIT_UI_REPOLIST_H extern void cgit_print_repolist(void); extern void cgit_print_site_readme(void); -#endif /* UI_REPOLIST_H */ +#endif // CGIT_UI_REPOLIST_H diff --git a/source/ui-shared.c b/source/ui-shared.c index 23da286..938f292 100644 --- a/source/ui-shared.c +++ b/source/ui-shared.c This diff is too large to be rendered inline. View it on its own page. diff --git a/source/ui-shared.h b/source/ui-shared.h index b6f2797..c266951 100644 --- a/source/ui-shared.h +++ b/source/ui-shared.h @@ -1,9 +1,25 @@ -#ifndef UI_SHARED_H -#define UI_SHARED_H +/* + * The page furniture and the link builders every renderer shares. A page + * handler renders its own body and calls in here for all that surrounds it, + * which is the HTTP headers, the document head and footer, the header block + * with its tabs and search form, the error page, and an anchor to any other + * cgit page. Whether a site addresses a request as a path below a virtual root + * or as a query string is settled behind these calls, so that nothing else has + * to know which of the two it is looking at. + */ +#ifndef CGIT_UI_SHARED_H +#define CGIT_UI_SHARED_H + +#include "cgit.h" extern const char *cgit_httpscheme(void); extern char *cgit_hosturl(void); -extern const char *cgit_rooturl(void); + +/* + * The URL of the request being answered, which cgit_currentfullurl gives with + * the query string on the end as well, minus the url argument that the path + * was taken from. + */ extern char *cgit_currenturl(void); extern char *cgit_currentfullurl(void); extern const char *cgit_loginurl(void); @@ -13,10 +29,17 @@ extern char *cgit_fileurl(const char *reponame, const char *pagename, extern char *cgit_pageurl(const char *reponame, const char *pagename, const char *query); +// Call fn once for every URL this repository can be cloned from. extern void cgit_add_clone_urls(void (*fn)(const char *)); +/* + * Each of these writes one anchor to the page and nothing around it, with name + * as the text a reader sees. A NULL title or class leaves that attribute out, + * and what follows them is the state the target page should open in. + */ extern void cgit_index_link(const char *name, const char *title, - const char *class, const char *pattern, const char *sort, int ofs, int always_root); + const char *class, const char *pattern, + const char *sort, int ofs, int always_root); extern void cgit_summary_link(const char *name, const char *title, const char *class, const char *head); extern void cgit_tag_link(const char *name, const char *title, @@ -43,9 +66,6 @@ extern void cgit_patch_link(const char *name, const char *title, extern void cgit_refs_link(const char *name, const char *title, const char *class, const char *head, const char *rev, const char *path); -extern void cgit_snapshot_link(const char *name, const char *title, - const char *class, const char *head, - const char *rev, const char *archivename); extern void cgit_diff_link(const char *name, const char *title, const char *class, const char *head, const char *new_rev, const char *old_rev, @@ -63,26 +83,42 @@ extern void cgit_print_layout_end(void); __attribute__((format (printf,1,2))) extern void cgit_print_error(const char *fmt, ...); -__attribute__((format (printf,1,0))) -extern void cgit_vprint_error(const char *fmt, va_list ap); extern struct date_mode cgit_date_mode(enum date_mode_type type); + +/* + * Print the time t, which was recorded in the timezone tz, as an age. An age + * further back than max_relative is given as a calendar date instead, and a + * negative max_relative stays relative however old it is. A site that turns + * relative dates off gets the date either way. + */ extern void cgit_print_age(time_t t, int tz, time_t max_relative); extern void cgit_print_http_headers(void); extern void cgit_redirect(const char *url, bool permanent); extern void cgit_print_docstart(void); extern void cgit_print_docend(void); __attribute__((format (printf,3,4))) -extern void cgit_print_error_page(int code, const char *msg, const char *fmt, ...); -extern void cgit_vprint_error_page(int code, const char *msg, const char *fmt, va_list ap); +extern void cgit_print_error_page(int code, const char *msg, const char *fmt, + ...); +extern void cgit_vprint_error_page(int code, const char *msg, const char *fmt, + va_list ap); extern void cgit_print_pageheader(void); extern void cgit_print_filemode(unsigned short mode); -extern void cgit_compose_snapshot_prefix(struct strbuf *filename, - const char *base, const char *ref); extern void cgit_print_snapshot_links(const struct cgit_repo *repo, const char *ref, const char *separator); + +// The name a snapshot of this repository is downloaded under, which is the +// configured prefix or the last component of the repository URL. extern const char *cgit_snapshot_prefix(const struct cgit_repo *repo); + +/* + * Carry the current request into a form as hidden inputs, so that submitting + * it changes only what the form itself asks about. The branch and the search + * terms are carried only when the caller asks for them, since a form with a + * field of its own for either would otherwise submit that name twice. + */ extern void cgit_add_hidden_formfields(int incl_head, int incl_search, const char *page); +// Put path in front of the page title, its last component first. extern void cgit_set_title_from_path(const char *path); -#endif /* UI_SHARED_H */ +#endif // CGIT_UI_SHARED_H diff --git a/source/ui-snapshot.c b/source/ui-snapshot.c index 97472c5..0bab8ea 100644 --- a/source/ui-snapshot.c +++ b/source/ui-snapshot.c @@ -1,27 +1,37 @@ -/* ui-snapshot.c: generate snapshot of a commit - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * Serves a commit as a downloadable archive. Writing the archive is git's + * work, so each format here is a thin wrapper that either calls write_archive + * directly or pipes its tar output through an external compressor. The table + * of formats also lives here, and since a repository's snapshots mask is one + * bit per table position, that order is part of what cgitrc means. When the + * request carries no revision the file name is worked backwards to find one, + * so that a link to cgit-1.2.tar.gz can serve the tag it was named after. A + * name ending in .asc asks for the detached signature kept under refs/notes + * instead of the archive itself. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-snapshot.h" +#include "filter.h" #include "html.h" #include "ui-shared.h" +#include "ui-snapshot.h" + +#define SIG_SUFFIX ".asc" -static int write_archive_type(const char *format, const char *hex, const char *prefix) +static int write_archive_format(const char *format_arg, const char *hex, + const char *prefix) { struct strvec argv = STRVEC_INIT; - const char **nargv; + const char **args; int result; + strvec_push(&argv, "snapshot"); - strvec_push(&argv, format); + strvec_push(&argv, format_arg); if (prefix) { struct strbuf buf = STRBUF_INIT; + strbuf_addstr(&buf, prefix); strbuf_addch(&buf, '/'); strvec_push(&argv, "--prefix"); @@ -29,44 +39,40 @@ static int write_archive_type(const char *format, const char *hex, const char *p strbuf_release(&buf); } strvec_push(&argv, hex); - /* - * Now we need to copy the pointers to arguments into a new - * structure because write_archive will rearrange its arguments - * which may result in duplicated/missing entries causing leaks - * or double-frees in strvec_clear. - */ - nargv = xmalloc(sizeof(char *) * (argv.nr + 1)); - /* strvec guarantees a trailing NULL entry. */ - memcpy(nargv, argv.v, sizeof(char *) * (argv.nr + 1)); - result = write_archive(argv.nr, nargv, NULL, the_repository, NULL, 0); + // write_archive rearranges the argv it is handed, which would leave + // strvec_clear leaking or double freeing, so it gets a copy of the + // pointers, one entry longer than argv.nr for the trailing NULL. + args = xmalloc(sizeof(char *) * (argv.nr + 1)); + memcpy(args, argv.v, sizeof(char *) * (argv.nr + 1)); + + result = write_archive(argv.nr, args, NULL, the_repository, NULL, 0); strvec_clear(&argv); - free(nargv); + free(args); return result; } static int write_tar_archive(const char *hex, const char *prefix) { - return write_archive_type("--format=tar", hex, prefix); + return write_archive_format("--format=tar", hex, prefix); } static int write_zip_archive(const char *hex, const char *prefix) { - return write_archive_type("--format=zip", hex, prefix); + return write_archive_format("--format=zip", hex, prefix); } -static int write_compressed_tar_archive(const char *hex, - const char *prefix, - char *filter_argv[]) +static int write_compressed_tar_archive(const char *hex, const char *prefix, + char *argv[]) { - int rv; - struct cgit_exec_filter f; - cgit_exec_filter_init(&f, filter_argv[0], filter_argv); + struct cgit_exec_filter filter; + int result; - cgit_open_filter(&f.base); - rv = write_tar_archive(hex, prefix); - cgit_close_filter(&f.base); - return rv; + cgit_exec_filter_init(&filter, argv[0], argv); + cgit_open_filter(&filter.base); + result = write_tar_archive(hex, prefix); + cgit_close_filter(&filter.base); + return result; } static int write_tar_gzip_archive(const char *hex, const char *prefix) @@ -95,14 +101,15 @@ static int write_tar_xz_archive(const char *hex, const char *prefix) static int write_tar_zstd_archive(const char *hex, const char *prefix) { - // Stay single-threaded like the other compressors so one request - // cannot pin every core. -T0 would fan out across all of them. + // Single-threaded like the other compressors, since -T0 would let one + // request pin every core. char *argv[] = { "zstd", NULL }; return write_compressed_tar_archive(hex, prefix, argv); } +// ui-shared.c reads entry zero as the tar whose signature stands in for every +// tar variant, so this order cannot change. const struct cgit_snapshot_format cgit_snapshot_formats[] = { - /* .tar must remain the 0 index */ { ".tar", "application/x-tar", write_tar_archive }, { ".tar.gz", "application/x-gzip", write_tar_gzip_archive }, { ".tar.bz2", "application/x-bzip2", write_tar_bzip2_archive }, @@ -113,48 +120,77 @@ const struct cgit_snapshot_format cgit_snapshot_formats[] = { { NULL } }; -static struct notes_tree snapshot_sig_notes[ARRAY_SIZE(cgit_snapshot_formats)]; +// Each tree is read the first time a signature for that format is asked for, +// then kept for the life of the process. +static struct notes_tree sig_notes[ARRAY_SIZE(cgit_snapshot_formats)]; -const struct object_id *cgit_snapshot_get_sig(const char *ref, - const struct cgit_snapshot_format *f) +static size_t format_index(const struct cgit_snapshot_format *f) { - struct notes_tree *tree; - struct object_id oid; - - if (repo_get_oid(the_repository, ref, &oid)) - return NULL; - - tree = &snapshot_sig_notes[f - &cgit_snapshot_formats[0]]; - if (!tree->initialized) { - struct strbuf notes_ref = STRBUF_INIT; + return f - cgit_snapshot_formats; +} - strbuf_addf(¬es_ref, "refs/notes/signatures/%s", - f->suffix + 1); +static const struct cgit_snapshot_format *find_format(const char *filename) +{ + const struct cgit_snapshot_format *f; - init_notes(tree, notes_ref.buf, combine_notes_ignore, 0); - strbuf_release(¬es_ref); + for (f = cgit_snapshot_formats; f->suffix; f++) { + if (ends_with(filename, f->suffix)) + return f; } - - return get_note(tree, &oid); + return NULL; } -static const struct cgit_snapshot_format *get_format(const char *filename) +static int resolves(const char *rev) { - const struct cgit_snapshot_format *fmt; + struct object_id oid; - for (fmt = cgit_snapshot_formats; fmt->suffix; fmt++) { - if (ends_with(filename, fmt->suffix)) - return fmt; - } - return NULL; + return repo_get_oid(the_repository, rev, &oid) == 0; } -unsigned cgit_snapshot_format_bit(const struct cgit_snapshot_format *f) +static const char *ref_from_filename(const struct cgit_repo *repo, + const char *filename, + const struct cgit_snapshot_format *format) { - return BIT(f - &cgit_snapshot_formats[0]); + struct strbuf rev = STRBUF_INIT; + const char *repo_prefix; + int found = 1; + + strbuf_addstr(&rev, filename); + strbuf_setlen(&rev, rev.len - strlen(format->suffix)); + + if (resolves(rev.buf)) + goto out; + + repo_prefix = cgit_snapshot_prefix(repo); + if (starts_with(rev.buf, repo_prefix)) { + const char *rest = rev.buf + strlen(repo_prefix); + + while (rest && (*rest == '-' || *rest == '_')) + rest++; + strbuf_splice(&rev, 0, rest - rev.buf, "", 0); + } + + if (resolves(rev.buf)) + goto out; + + // A tag is often written with a leading v while the file named after it + // is not, so that is tried last. + strbuf_insert(&rev, 0, "v", 1); + if (resolves(rev.buf)) + goto out; + + strbuf_splice(&rev, 0, 1, "V", 1); + if (resolves(rev.buf)) + goto out; + + found = 0; + strbuf_release(&rev); + +out: + return found ? strbuf_detach(&rev, NULL) : NULL; } -static int make_snapshot(const struct cgit_snapshot_format *format, +static int send_snapshot(const struct cgit_snapshot_format *format, const char *hex, const char *prefix, const char *filename) { @@ -179,9 +215,9 @@ static int make_snapshot(const struct cgit_snapshot_format *format, return 0; } -static int write_sig(const struct cgit_snapshot_format *format, - const char *hex, const char *archive, - const char *filename) +static int send_sig(const struct cgit_snapshot_format *format, + const char *hex, const char *archive_name, + const char *sig_filename) { const struct object_id *note = cgit_snapshot_get_sig(hex, format); enum object_type type; @@ -190,7 +226,7 @@ static int write_sig(const struct cgit_snapshot_format *format, if (!note) { cgit_print_error_page(404, "Not found", - "No signature for %s", archive); + "No signature for %s", archive_name); return 0; } @@ -200,11 +236,14 @@ static int write_sig(const struct cgit_snapshot_format *format, return 0; } + // The body is whatever bytes the note holds, so these go out ahead of + // the usual headers to stop a browser sniffing it into a type it will + // act on. html("X-Content-Type-Options: nosniff\n"); html("Content-Security-Policy: default-src 'none'\n"); ctx.page.etag = oid_to_hex(note); ctx.page.mimetype = xstrdup("application/pgp-signature"); - ctx.page.filename = xstrdup(filename); + ctx.page.filename = xstrdup(sig_filename); cgit_print_http_headers(); html_raw(buf, size); @@ -212,64 +251,74 @@ static int write_sig(const struct cgit_snapshot_format *format, return 0; } -/* Try to guess the requested revision from the requested snapshot name. - * First the format extension is stripped, e.g. "cgit-0.7.2.tar.gz" become - * "cgit-0.7.2". If this is a valid commit object name we've got a winner. - * Otherwise, if the snapshot name has a prefix matching the result from - * repo_basename(), we strip the basename and any following '-' and '_' - * characters ("cgit-0.7.2" -> "0.7.2") and check the resulting name once - * more. If this still isn't a valid commit object name, we check if pre- - * pending a 'v' or a 'V' to the remaining snapshot name ("0.7.2" -> - * "v0.7.2") gives us something valid. - */ -static const char *get_ref_from_filename(const struct cgit_repo *repo, - const char *filename, - const struct cgit_snapshot_format *format) +const struct object_id *cgit_snapshot_get_sig(const char *ref, + const struct cgit_snapshot_format *f) { - const char *reponame; + struct notes_tree *tree; struct object_id oid; - struct strbuf snapshot = STRBUF_INIT; - int result = 1; - strbuf_addstr(&snapshot, filename); - strbuf_setlen(&snapshot, snapshot.len - strlen(format->suffix)); + if (repo_get_oid(the_repository, ref, &oid)) + return NULL; - if (repo_get_oid(the_repository, snapshot.buf, &oid) == 0) - goto out; + tree = &sig_notes[format_index(f)]; + if (!tree->initialized) { + struct strbuf notes_ref = STRBUF_INIT; + + // Signatures live under the format suffix with the leading dot + // dropped, so plain tar is refs/notes/signatures/tar. + strbuf_addf(¬es_ref, "refs/notes/signatures/%s", + f->suffix + 1); - reponame = cgit_snapshot_prefix(repo); - if (starts_with(snapshot.buf, reponame)) { - const char *new_start = snapshot.buf; - new_start += strlen(reponame); - while (new_start && (*new_start == '-' || *new_start == '_')) - new_start++; - strbuf_splice(&snapshot, 0, new_start - snapshot.buf, "", 0); + init_notes(tree, notes_ref.buf, combine_notes_ignore, 0); + strbuf_release(¬es_ref); } - if (repo_get_oid(the_repository, snapshot.buf, &oid) == 0) - goto out; + return get_note(tree, &oid); +} - strbuf_insert(&snapshot, 0, "v", 1); - if (repo_get_oid(the_repository, snapshot.buf, &oid) == 0) - goto out; +unsigned cgit_snapshot_format_bit(const struct cgit_snapshot_format *f) +{ + return 1U << format_index(f); +} - strbuf_splice(&snapshot, 0, 1, "V", 1); - if (repo_get_oid(the_repository, snapshot.buf, &oid) == 0) - goto out; +int cgit_parse_snapshots_mask(const char *str) +{ + struct string_list tokens = STRING_LIST_INIT_DUP; + struct string_list_item *item; + const struct cgit_snapshot_format *f; + int mask = 0; - result = 0; - strbuf_release(&snapshot); + // A number is the legacy form of this setting and is still taken + // ahead of the named forms below. + if (atoi(str)) + return 1; -out: - return result ? strbuf_detach(&snapshot, NULL) : NULL; + if (strcmp(str, "all") == 0) + return INT_MAX; + + string_list_split(&tokens, str, " ", -1); + string_list_remove_empty_items(&tokens, 0); + + for_each_string_list_item(item, &tokens) { + for (f = cgit_snapshot_formats; f->suffix; f++) { + if (!strcmp(item->string, f->suffix) || + !strcmp(item->string, f->suffix + 1)) { + mask |= cgit_snapshot_format_bit(f); + break; + } + } + } + + string_list_clear(&tokens, 0); + return mask; } void cgit_print_snapshot(const char *head, const char *hex, const char *filename, int dwim) { - const struct cgit_snapshot_format* f; + const struct cgit_snapshot_format *f; const char *sig_filename = NULL; - char *adj_filename = NULL; + char *archive_name = NULL; char *prefix = NULL; if (!filename) { @@ -278,25 +327,24 @@ void cgit_print_snapshot(const char *head, const char *hex, return; } - if (ends_with(filename, ".asc")) { + if (ends_with(filename, SIG_SUFFIX)) { sig_filename = filename; - - /* Strip ".asc" from filename for common format processing */ - adj_filename = xstrdup(filename); - adj_filename[strlen(adj_filename) - 4] = '\0'; - filename = adj_filename; + archive_name = xstrdup(filename); + archive_name[strlen(archive_name) - strlen(SIG_SUFFIX)] = '\0'; + filename = archive_name; } - f = get_format(filename); - if (!f || (!sig_filename && !(ctx.repo->snapshots & cgit_snapshot_format_bit(f)))) { + f = find_format(filename); + if (!f || (!sig_filename && + !(ctx.repo->snapshots & cgit_snapshot_format_bit(f)))) { cgit_print_error_page(400, "Bad request", "Unsupported snapshot format: %s", filename); return; } if (!hex && dwim) { - hex = get_ref_from_filename(ctx.repo, filename, f); - if (hex == NULL) { + hex = ref_from_filename(ctx.repo, filename, f); + if (!hex) { cgit_print_error_page(404, "Not found", "Not found"); return; } @@ -311,10 +359,10 @@ void cgit_print_snapshot(const char *head, const char *hex, prefix = xstrdup(cgit_snapshot_prefix(ctx.repo)); if (sig_filename) - write_sig(f, hex, filename, sig_filename); + send_sig(f, hex, filename, sig_filename); else - make_snapshot(f, hex, prefix, filename); + send_snapshot(f, hex, prefix, filename); free(prefix); - free(adj_filename); + free(archive_name); } diff --git a/source/ui-snapshot.h b/source/ui-snapshot.h index a8deec3..9f1ec02 100644 --- a/source/ui-snapshot.h +++ b/source/ui-snapshot.h @@ -1,7 +1,41 @@ -#ifndef UI_SNAPSHOT_H -#define UI_SNAPSHOT_H +/* + * The snapshot page, which serves a commit as a downloadable archive. It owns + * the table of formats cgit knows how to write, so the config reader and the + * link builders both reach through here to ask which formats a repository + * offers. Each format has a bit in the per-repository snapshots mask, taken + * from its position in that table. + */ + +#ifndef CGIT_UI_SNAPSHOT_H +#define CGIT_UI_SNAPSHOT_H + +#include "cgit.h" + +typedef int (*write_archive_fn_t)(const char *hex, const char *prefix); + +struct cgit_snapshot_format { + const char *suffix; + const char *mimetype; + write_archive_fn_t write_func; +}; + +// Terminated by an entry with a NULL suffix. +extern const struct cgit_snapshot_format cgit_snapshot_formats[]; + +extern unsigned cgit_snapshot_format_bit(const struct cgit_snapshot_format *f); + +/* + * Turn a cgitrc snapshots value into a mask over cgit_snapshot_formats. A + * plain number is the legacy form meaning every format, as is the word all, + * otherwise the value is a space separated list of suffixes. + */ +extern int cgit_parse_snapshots_mask(const char *str); + +// The detached signature stored for this format under refs/notes, if any. +extern const struct object_id *cgit_snapshot_get_sig( + const char *ref, const struct cgit_snapshot_format *f); extern void cgit_print_snapshot(const char *head, const char *hex, const char *filename, int dwim); -#endif /* UI_SNAPSHOT_H */ +#endif // CGIT_UI_SNAPSHOT_H diff --git a/source/ui-ssdiff.c b/source/ui-ssdiff.c index e4472c4..2ea7021 100644 --- a/source/ui-ssdiff.c +++ b/source/ui-ssdiff.c @@ -1,185 +1,131 @@ +/* + * The side by side rendering of a diff, the layout a reader gets instead of + * the unified listing. ui-diff.c drives it, handing over each line xdiff + * produces and calling in around the header and the footer of every file. + * Removed and added lines are held back until their run ends, so that the two + * sides can be paired into a row apiece. When a run has as many removals as + * additions the paired lines are compared character by character, and what + * differs within them is marked. + */ + #include "cgit.h" -#include "ui-ssdiff.h" #include "html.h" -#include "ui-shared.h" #include "ui-diff.h" +#include "ui-shared.h" +#include "ui-ssdiff.h" -extern int use_ssdiff; - -static int current_old_line, current_new_line; -static int **L = NULL; +// The stylesheet sets no tab-size and tabs are expanded here rather than left +// to the browser, so this has to be the width a browser would pick on its own. +#define TAB_WIDTH 8 -struct deferred_lines { +// One line held back until its run ends, owning the copy taken of it. +struct deferred_line { int line_no; char *line; - struct deferred_lines *next; + struct deferred_line *next; }; -static struct deferred_lines *deferred_old, *deferred_old_last; -static struct deferred_lines *deferred_new, *deferred_new_last; +static int current_old_line, current_new_line; +static int **lcs_table; +static struct deferred_line *deferred_old, *deferred_old_last; +static struct deferred_line *deferred_new, *deferred_new_last; /* - * The table does not need clearing between calls. The fill below assigns - * every cell in [0,m] x [0,n] before anything reads it: a read only happens - * where both lines still have a character, so it never reaches past row m or - * column n, and the loops run downwards so the neighbour is always already - * written. Clearing the whole table on every changed line pair cost more than - * the comparison it was preparing for. + * The table is reused by every comparison and nothing clears it in between, + * because the fill in longest_common_subsequence works back from the far + * corner and writes every cell it goes on to read. Clearing it for each pair + * of lines cost more than the comparison it was preparing for. */ static void create_lcs_table(void) { int i; - if (L != NULL) + if (lcs_table) return; - // xcalloc will die if we ran out of memory; - // not very helpful for debugging - L = (int**)xcalloc(MAX_SSDIFF_M, sizeof(int *)); - *L = (int*)xcalloc(MAX_SSDIFF_SIZE, sizeof(int)); - - for (i = 1; i < MAX_SSDIFF_M; i++) { - L[i] = *L + i * MAX_SSDIFF_N; - } + lcs_table = xcalloc(MAX_SSDIFF_M, sizeof(int *)); + lcs_table[0] = xcalloc(MAX_SSDIFF_SIZE, sizeof(int)); + for (i = 1; i < MAX_SSDIFF_M; i++) + lcs_table[i] = lcs_table[0] + i * MAX_SSDIFF_N; } -static char *longest_common_subsequence(char *A, char *B) +/* + * The characters the two lines have in common, in order, which the caller + * owns. A line too long for the table gets NULL back and is shown whole + * instead. + */ +static char *longest_common_subsequence(const char *old_line, + const char *new_line) { - int i, j, ri; - int m = strlen(A); - int n = strlen(B); - int tmp1, tmp2; - int lcs_length; - char *result; + int old_len = strlen(old_line); + int new_len = strlen(new_line); + int i, j, pos, lcs_len; + char *lcs; - // We bail if the lines are too long - if (m >= MAX_SSDIFF_M || n >= MAX_SSDIFF_N) + if (old_len >= MAX_SSDIFF_M || new_len >= MAX_SSDIFF_N) return NULL; create_lcs_table(); - for (i = m; i >= 0; i--) { - for (j = n; j >= 0; j--) { - if (A[i] == '\0' || B[j] == '\0') { - L[i][j] = 0; - } else if (A[i] == B[j]) { - L[i][j] = 1 + L[i + 1][j + 1]; + for (i = old_len; i >= 0; i--) { + for (j = new_len; j >= 0; j--) { + if (old_line[i] == '\0' || new_line[j] == '\0') { + lcs_table[i][j] = 0; + } else if (old_line[i] == new_line[j]) { + lcs_table[i][j] = 1 + lcs_table[i + 1][j + 1]; } else { - tmp1 = L[i + 1][j]; - tmp2 = L[i][j + 1]; - L[i][j] = (tmp1 > tmp2 ? tmp1 : tmp2); + int drop_old = lcs_table[i + 1][j]; + int drop_new = lcs_table[i][j + 1]; + + lcs_table[i][j] = (drop_old > drop_new ? + drop_old : drop_new); } } } - lcs_length = L[0][0]; - result = xmalloc(lcs_length + 2); - memset(result, 0, sizeof(*result) * (lcs_length + 2)); + lcs_len = lcs_table[0][0]; + lcs = xmalloc(lcs_len + 2); + memset(lcs, 0, sizeof(*lcs) * (lcs_len + 2)); - ri = 0; + pos = 0; i = 0; j = 0; - while (i < m && j < n) { - if (A[i] == B[j]) { - result[ri] = A[i]; - ri += 1; + while (i < old_len && j < new_len) { + if (old_line[i] == new_line[j]) { + lcs[pos] = old_line[i]; + pos += 1; i += 1; j += 1; - } else if (L[i + 1][j] >= L[i][j + 1]) { + } else if (lcs_table[i + 1][j] >= lcs_table[i][j + 1]) { i += 1; } else { j += 1; } } - return result; + return lcs; } -static int line_from_hunk(char *line, char type) -{ - char *p; - long res; - - p = strchr(line, type); - if (p == NULL) - return 0; - p += 1; - // git omits the length when a hunk covers a single line, as in - // "@@ -1 +1 @@", so the number runs to whatever follows it rather than - // to a comma that may belong to the other side of the header or be - // missing altogether. - res = strtol(p, NULL, 10); - if (res < 0 || res > INT_MAX) - return 0; - return (int)res; -} - -static char *replace_tabs(char *line) +/* + * The line with its tabs expanded, which the caller owns. Appending as the + * line is walked replaces a loop that rescanned the rest of the input at every + * tab, which made a tab heavy line quadratic in its own length. + */ +static char *expand_tabs(const char *line) { struct strbuf out = STRBUF_INIT; const char *p; - // Each tab runs to the next eight-column stop. Walking the line once - // and appending replaces a loop that measured the result and rescanned - // the rest of the input at every tab, which made a tab-heavy line - // quadratic in its own length. for (p = line; *p; p++) { if (*p == '\t') - strbuf_addchars(&out, ' ', 8 - (out.len % 8)); + strbuf_addchars(&out, ' ', + TAB_WIDTH - (out.len % TAB_WIDTH)); else strbuf_addch(&out, *p); } return strbuf_detach(&out, NULL); } -static int calc_deferred_lines(struct deferred_lines *start) -{ - struct deferred_lines *item = start; - int result = 0; - while (item) { - result += 1; - item = item->next; - } - return result; -} - -static void deferred_old_add(char *line, int line_no) -{ - struct deferred_lines *item = xmalloc(sizeof(struct deferred_lines)); - item->line = xstrdup(line); - item->line_no = line_no; - item->next = NULL; - if (deferred_old) { - deferred_old_last->next = item; - deferred_old_last = item; - } else { - deferred_old = deferred_old_last = item; - } -} - -static void deferred_new_add(char *line, int line_no) -{ - struct deferred_lines *item = xmalloc(sizeof(struct deferred_lines)); - item->line = xstrdup(line); - item->line_no = line_no; - item->next = NULL; - if (deferred_new) { - deferred_new_last->next = item; - deferred_new_last = item; - } else { - deferred_new = deferred_new_last = item; - } -} - -/* The item owns the copy of the line taken when it was deferred, so both go - * together. print_ssdiff_line only reads the line and frees what it derives - * from it, and never keeps the pointer it was handed. */ -static void free_deferred(struct deferred_lines *item) -{ - free(item->line); - free(item); -} - static void flush_run(struct strbuf *run) { if (!run->len) @@ -188,186 +134,240 @@ static void flush_run(struct strbuf *run) strbuf_reset(run); } -static void print_part_with_lcs(const char *class, char *line, char *lcs) +/* + * A stretch that the other side does not share is escaped in one call because + * escaping a character at a time sent every character of every changed line + * through the output path on its own, which dominated this page. + */ +static void print_line_with_lcs(const char *class, const char *line, + const char *lcs) { - int line_len = strlen(line); - int i, j; - int same = 1; + int len = strlen(line); + int in_common = 1; + int matched = 0; + int i; struct strbuf run = STRBUF_INIT; - // Collect each stretch that is wholly inside or wholly outside the - // common subsequence and escape it in one go. Escaping a character at a - // time meant a write syscall per character of every changed line, which - // dominated this page. - j = 0; - for (i = 0; i < line_len; i++) { - if (same) { - if (line[i] == lcs[j]) - j += 1; + for (i = 0; i < len; i++) { + if (in_common) { + if (line[i] == lcs[matched]) + matched += 1; else { - same = 0; + in_common = 0; flush_run(&run); htmlf("<span class='%s'>", class); } - } else if (line[i] == lcs[j]) { - same = 1; + } else if (line[i] == lcs[matched]) { + in_common = 1; flush_run(&run); html("</span>"); - j += 1; + matched += 1; } strbuf_addch(&run, line[i]); } flush_run(&run); - if (!same) + if (!in_common) html("</span>"); strbuf_release(&run); } -static void print_ssdiff_line(const char *class, - int old_line_no, - char *old_line, - int new_line_no, - char *new_line, int individual_chars) +static void print_lineno_cell(struct diff_filespec *file, + const struct object_id *rev, int line_no) +{ + struct strbuf path = STRBUF_INIT; + char *anchor, *query, *fileurl; + const char *rev_hex; + + anchor = cgit_fmt("n%d", line_no); + rev_hex = is_null_oid(&file->oid) ? "HEAD" : oid_to_hex(rev); + query = cgit_fmt("id=%s#%s", rev_hex, anchor); + // The path is repository content, so percent-encode it before it lands + // raw in the href. + if (file->path) + strbuf_add_percentencode(&path, file->path, 0); + fileurl = cgit_fileurl(ctx.repo->url, "tree", path.buf, query); + html("<td class='lineno'><a href='"); + html(fileurl); + htmlf("'>%s</a>", anchor + 1); + html("</td>"); + free(fileurl); + strbuf_release(&path); +} + +static void print_row(const char *class, + int old_line_no, char *old_line, + int new_line_no, char *new_line, + int highlight_chars) { char *lcs = NULL; + // The first byte of a line is the marker xdiff put on it, not text. if (old_line) - old_line = replace_tabs(old_line + 1); + old_line = expand_tabs(old_line + 1); if (new_line) - new_line = replace_tabs(new_line + 1); - if (individual_chars && old_line && new_line) + new_line = expand_tabs(new_line + 1); + if (highlight_chars && old_line && new_line) lcs = longest_common_subsequence(old_line, new_line); html("<tr>\n"); if (old_line_no > 0) { - struct diff_filespec *old_file = cgit_get_current_old_file(); - char *lineno_str = cgit_fmt("n%d", old_line_no); - char *id_str = cgit_fmt("id=%s#%s", is_null_oid(&old_file->oid)?"HEAD":oid_to_hex(old_rev_oid), lineno_str); - struct strbuf path = STRBUF_INIT; - char *fileurl; - // The file path is repository content, so percent-encode it - // before it lands raw in the href below. - if (old_file->path) - strbuf_add_percentencode(&path, old_file->path, 0); - fileurl = cgit_fileurl(ctx.repo->url, "tree", path.buf, id_str); - html("<td class='lineno'><a href='"); - html(fileurl); - htmlf("'>%s</a>", lineno_str + 1); - html("</td>"); + print_lineno_cell(cgit_get_current_old_file(), old_rev_oid, + old_line_no); htmlf("<td class='%s'>", class); - free(fileurl); - strbuf_release(&path); } else if (old_line) htmlf("<td class='lineno'></td><td class='%s'>", class); else htmlf("<td class='lineno'></td><td class='%s_dark'>", class); if (old_line) { if (lcs) - print_part_with_lcs("del", old_line, lcs); + print_line_with_lcs("del", old_line, lcs); else html_txt(old_line); } html("</td>\n"); if (new_line_no > 0) { - struct diff_filespec *new_file = cgit_get_current_new_file(); - char *lineno_str = cgit_fmt("n%d", new_line_no); - char *id_str = cgit_fmt("id=%s#%s", is_null_oid(&new_file->oid)?"HEAD":oid_to_hex(new_rev_oid), lineno_str); - struct strbuf path = STRBUF_INIT; - char *fileurl; - // The file path is repository content, so percent-encode it - // before it lands raw in the href below. - if (new_file->path) - strbuf_add_percentencode(&path, new_file->path, 0); - fileurl = cgit_fileurl(ctx.repo->url, "tree", path.buf, id_str); - html("<td class='lineno'><a href='"); - html(fileurl); - htmlf("'>%s</a>", lineno_str + 1); - html("</td>"); + print_lineno_cell(cgit_get_current_new_file(), new_rev_oid, + new_line_no); htmlf("<td class='%s'>", class); - free(fileurl); - strbuf_release(&path); } else if (new_line) htmlf("<td class='lineno'></td><td class='%s'>", class); else htmlf("<td class='lineno'></td><td class='%s_dark'>", class); if (new_line) { if (lcs) - print_part_with_lcs("add", new_line, lcs); + print_line_with_lcs("add", new_line, lcs); else html_txt(new_line); } html("</td></tr>"); - if (lcs) - free(lcs); - if (new_line) - free(new_line); - if (old_line) - free(old_line); + free(lcs); + free(new_line); + free(old_line); +} + +static void defer_line(struct deferred_line **head, + struct deferred_line **last, + const char *line, int line_no) +{ + struct deferred_line *item = xmalloc(sizeof(*item)); + + item->line = xstrdup(line); + item->line_no = line_no; + item->next = NULL; + if (*head) + (*last)->next = item; + else + *head = item; + *last = item; +} + +/* + * print_row frees only what it derives from the line it is handed and never + * keeps the pointer itself, so the item can go as soon as its row is written. + */ +static void free_deferred(struct deferred_line *item) +{ + free(item->line); + free(item); +} + +static int count_deferred(struct deferred_line *item) +{ + int count = 0; + + while (item) { + count += 1; + item = item->next; + } + return count; } static void print_deferred_old_lines(void) { - struct deferred_lines *iter_old, *tmp; - iter_old = deferred_old; - while (iter_old) { - print_ssdiff_line("del", iter_old->line_no, - iter_old->line, -1, NULL, 0); - tmp = iter_old->next; - free_deferred(iter_old); - iter_old = tmp; + struct deferred_line *item = deferred_old; + struct deferred_line *next; + + while (item) { + print_row("del", item->line_no, item->line, -1, NULL, 0); + next = item->next; + free_deferred(item); + item = next; } } static void print_deferred_new_lines(void) { - struct deferred_lines *iter_new, *tmp; - iter_new = deferred_new; - while (iter_new) { - print_ssdiff_line("add", -1, NULL, - iter_new->line_no, iter_new->line, 0); - tmp = iter_new->next; - free_deferred(iter_new); - iter_new = tmp; + struct deferred_line *item = deferred_new; + struct deferred_line *next; + + while (item) { + print_row("add", -1, NULL, item->line_no, item->line, 0); + next = item->next; + free_deferred(item); + item = next; } } +/* + * Pairing a removal with an addition only stands for anything when the two + * runs are the same length, which is why the character marking is offered only + * then. + */ static void print_deferred_changed_lines(void) { - struct deferred_lines *iter_old, *iter_new, *tmp; - int n_old_lines = calc_deferred_lines(deferred_old); - int n_new_lines = calc_deferred_lines(deferred_new); - int individual_chars = (n_old_lines == n_new_lines ? 1 : 0); + struct deferred_line *old_item = deferred_old; + struct deferred_line *new_item = deferred_new; + struct deferred_line *next; + int highlight_chars; - iter_old = deferred_old; - iter_new = deferred_new; - while (iter_old || iter_new) { - if (iter_old && iter_new) - print_ssdiff_line("changed", iter_old->line_no, - iter_old->line, - iter_new->line_no, iter_new->line, - individual_chars); - else if (iter_old) - print_ssdiff_line("changed", iter_old->line_no, - iter_old->line, -1, NULL, 0); - else if (iter_new) - print_ssdiff_line("changed", -1, NULL, - iter_new->line_no, iter_new->line, 0); - if (iter_old) { - tmp = iter_old->next; - free_deferred(iter_old); - iter_old = tmp; + highlight_chars = count_deferred(old_item) == count_deferred(new_item); + while (old_item || new_item) { + if (old_item && new_item) + print_row("changed", old_item->line_no, + old_item->line, new_item->line_no, + new_item->line, highlight_chars); + else if (old_item) + print_row("changed", old_item->line_no, + old_item->line, -1, NULL, 0); + else if (new_item) + print_row("changed", -1, NULL, + new_item->line_no, new_item->line, 0); + if (old_item) { + next = old_item->next; + free_deferred(old_item); + old_item = next; } - if (iter_new) { - tmp = iter_new->next; - free_deferred(iter_new); - iter_new = tmp; + if (new_item) { + next = new_item->next; + free_deferred(new_item); + new_item = next; } } } -void cgit_ssdiff_print_deferred_lines(void) +/* + * git leaves the length out when a hunk covers a single line, as in + * "@@ -1 +1 @@", so the number runs to whatever follows it rather than to a + * comma that may belong to the other side or be missing altogether. + */ +static int hunk_start_line(const char *hunk, char marker) +{ + const char *p; + long line_no; + + p = strchr(hunk, marker); + if (p == NULL) + return 0; + p += 1; + line_no = strtol(p, NULL, 10); + if (line_no < 0 || line_no > INT_MAX) + return 0; + return (int)line_no; +} + +static void print_deferred_lines(void) { if (!deferred_old && !deferred_new) return; @@ -382,29 +382,34 @@ void cgit_ssdiff_print_deferred_lines(void) } /* - * print a single line returned from xdiff + * The length counts the byte that ends the line, and the buffer belongs to the + * caller, so that byte is only swapped for a NUL while the row is written and + * is put back before returning. */ void cgit_ssdiff_line_cb(char *line, int len) { - char c = line[len - 1]; + char terminator = line[len - 1]; + line[len - 1] = '\0'; if (line[0] == '@') { - current_old_line = line_from_hunk(line, '-'); - current_new_line = line_from_hunk(line, '+'); + current_old_line = hunk_start_line(line, '-'); + current_new_line = hunk_start_line(line, '+'); } if (line[0] == ' ') { if (deferred_old || deferred_new) - cgit_ssdiff_print_deferred_lines(); - print_ssdiff_line("ctx", current_old_line, line, - current_new_line, line, 0); + print_deferred_lines(); + print_row("ctx", current_old_line, line, + current_new_line, line, 0); current_old_line += 1; current_new_line += 1; } else if (line[0] == '+') { - deferred_new_add(line, current_new_line); + defer_line(&deferred_new, &deferred_new_last, line, + current_new_line); current_new_line += 1; } else if (line[0] == '-') { - deferred_old_add(line, current_old_line); + defer_line(&deferred_old, &deferred_old_last, line, + current_old_line); current_old_line += 1; } else if (line[0] == '@') { html("<tr><td colspan='4' class='hunk'>"); @@ -415,7 +420,7 @@ void cgit_ssdiff_line_cb(char *line, int len) html_txt(line); html("</td></tr>"); } - line[len - 1] = c; + line[len - 1] = terminator; } void cgit_ssdiff_header_begin(void) @@ -434,6 +439,6 @@ void cgit_ssdiff_header_end(void) void cgit_ssdiff_footer(void) { if (deferred_old || deferred_new) - cgit_ssdiff_print_deferred_lines(); + print_deferred_lines(); html("<tr><td class='foot' colspan='4'></td></tr>"); } diff --git a/source/ui-ssdiff.h b/source/ui-ssdiff.h index 11f2714..f6489c8 100644 --- a/source/ui-ssdiff.h +++ b/source/ui-ssdiff.h @@ -1,9 +1,16 @@ -#ifndef UI_SSDIFF_H -#define UI_SSDIFF_H - /* - * ssdiff line limits + * The side by side diff view, which ui-diff.c renders through when a reader + * asks for that layout rather than the unified one. Every call below is made + * as ui-diff.c walks a diff, so nothing here is a page of its own. */ + +#ifndef CGIT_UI_SSDIFF_H +#define CGIT_UI_SSDIFF_H + +// The longest old line and the longest new line the view will compare +// character by character. The comparison fills a cell for every pair of +// characters, so its cost is the product of the two, and a line past the limit +// is shown whole instead. #ifndef MAX_SSDIFF_M #define MAX_SSDIFF_M 128 #endif @@ -13,8 +20,6 @@ #endif #define MAX_SSDIFF_SIZE ((MAX_SSDIFF_M) * (MAX_SSDIFF_N)) -extern void cgit_ssdiff_print_deferred_lines(void); - extern void cgit_ssdiff_line_cb(char *line, int len); extern void cgit_ssdiff_header_begin(void); @@ -22,4 +27,4 @@ extern void cgit_ssdiff_header_end(void); extern void cgit_ssdiff_footer(void); -#endif /* UI_SSDIFF_H */ +#endif // CGIT_UI_SSDIFF_H diff --git a/source/ui-stats.c b/source/ui-stats.c index 20e9a30..5436907 100644 --- a/source/ui-stats.c +++ b/source/ui-stats.c This diff is too large to be rendered inline. View it on its own page. diff --git a/source/ui-stats.h b/source/ui-stats.h index 0e61b03..718e335 100644 --- a/source/ui-stats.h +++ b/source/ui-stats.h @@ -1,5 +1,13 @@ -#ifndef UI_STATS_H -#define UI_STATS_H +/* + * The statistics page, which counts commits per author over a window of + * recent weeks, months, quarters or years and sizes the tip of the branch by + * language. The windows are described here rather than inside the page, + * because cgit.c resolves a repository's max-stats setting against the same + * set. + */ + +#ifndef CGIT_UI_STATS_H +#define CGIT_UI_STATS_H #include "cgit.h" @@ -7,22 +15,36 @@ struct cgit_period { const char code; const char *name; int max_periods; + + // How many periods a page shows side by side. int count; - /* Convert a tm value to the first day in the period */ + // Move a tm back to the first day of the period it falls in. void (*trunc)(struct tm *tm); - /* Update tm value to start of next/previous period */ + // Step a truncated tm one whole period back or forward. void (*dec)(struct tm *tm); void (*inc)(struct tm *tm); - /* Pretty-print a tm value */ + // Label the period a tm falls in. The buffer is reused, so a caller + // that keeps a label copies it. char *(*pretty)(struct tm *tm); }; -extern int cgit_find_stats_period(const char *expr, const struct cgit_period **period); +/* + * Look a period up by its one letter code or by its full name, storing the + * match through period when that is not NULL. The return is a one based index + * into the table of periods, or 0 when nothing matches, and because that + * table runs from the finest window to the coarsest, the index is what + * max-stats is compared against. + */ +extern int cgit_find_stats_period(const char *expr, + const struct cgit_period **period); + +// The name of the period at a one based index, or an empty string when the +// index names no period. extern const char *cgit_find_stats_periodname(int idx); extern void cgit_show_stats(void); -#endif /* UI_STATS_H */ +#endif // CGIT_UI_STATS_H diff --git a/source/ui-summary.c b/source/ui-summary.c index 2169048..ada6b1b 100644 --- a/source/ui-summary.c +++ b/source/ui-summary.c @@ -1,32 +1,49 @@ -/* ui-summary.c: functions for generating repo summary page - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The repository summary page and the about page. The summary stacks the + * branch list, the tag list and the head of the log into one table and ends + * with the clone urls, reusing the listings ui-refs.c and ui-log.c draw + * elsewhere. The about page renders the readme that config parsing picked for + * the repository, either through the configured about filter or escaped as + * plain text, and it can serve a file sitting beside that readme so links + * inside the readme resolve. */ #include "cgit.h" -#include "ui-summary.h" +#include "filter.h" #include "html.h" +#include "shared.h" #include "ui-blob.h" #include "ui-log.h" #include "ui-plain.h" #include "ui-refs.h" #include "ui-shared.h" +#include "ui-summary.h" + +// Age, commit message and author, the three columns a log row always has. +// ui-log.c adds one more for each of the two optional counts, and the rows +// this page stretches across the table have to match that width. +#define LOG_BASE_COLUMNS 3 -static int urls; +static int clone_urls_printed; -static void print_url(const char *url) +static int log_columns(void) { - int columns = 3; + int columns = LOG_BASE_COLUMNS; if (ctx.repo->enable_log_filecount) columns++; if (ctx.repo->enable_log_linecount) columns++; + return columns; +} + +static void print_clone_url(const char *url) +{ + int columns = log_columns(); - if (urls++ == 0) { + // cgit_add_clone_urls may call back no times at all, so the heading + // waits for a first url rather than being printed ahead of the walk. + if (clone_urls_printed++ == 0) { htmlf("<tr class='nohover'><td colspan='%d'> </td></tr>", columns); htmlf("<tr class='nohover'><th class='left' colspan='%d'>Clone</th></tr>\n", columns); } @@ -40,43 +57,38 @@ static void print_url(const char *url) html("</a></td></tr>\n"); } -void cgit_print_summary(void) +/* + * Without the separator boundary a sibling directory that merely shares the + * base as a name prefix, such as repo.git-backup beside repo.git, would pass. + */ +static int path_within(const char *base, const char *path) { - int columns = 3; - - if (ctx.repo->enable_log_filecount) - columns++; - if (ctx.repo->enable_log_linecount) - columns++; + size_t len = strlen(base); - cgit_print_layout_start(); - html("<table summary='repository info' class='list nowrap'>"); - cgit_print_branches(ctx.cfg.summary_branches); - htmlf("<tr class='nohover'><td colspan='%d'> </td></tr>", columns); - cgit_print_tags(ctx.cfg.summary_tags); - if (ctx.cfg.summary_log > 0) { - htmlf("<tr class='nohover'><td colspan='%d'> </td></tr>", columns); - cgit_print_log(ctx.qry.head, 0, ctx.cfg.summary_log, NULL, - NULL, NULL, 0, 0, 0); - } - urls = 0; - cgit_add_clone_urls(print_url); - html("</table>"); - cgit_print_layout_end(); + return starts_with(path, base) && + (path[len] == '\0' || path[len] == '/'); } -/* The caller must free the return value. */ -static char* append_readme_path(const char *filename, const char *ref, const char *path) +/* + * Returns a path the caller must free, or NULL when the request cannot be + * served. A null ref means the readme is a file on the server's disk rather + * than a path inside a ref, and such a readme is confined to its own + * directory, so one named without a directory has nothing to confine it to + * and is refused. + */ +static char *resolve_about_path(const char *filename, const char *ref, + const char *path) { - char *file, *base_dir, *full_path, *resolved_base = NULL, *resolved_full = NULL; - /* If a subpath is specified for the about page, make it relative - * to the directory containing the configured readme. */ + char *copy, *base_dir, *full_path; + char *resolved_base = NULL, *resolved_full = NULL; - file = xstrdup(filename); - base_dir = dirname(file); + // dirname is allowed to write into its argument and to return a + // pointer into it, so base_dir borrows from a copy we keep alive. + copy = xstrdup(filename); + base_dir = dirname(copy); if (!strcmp(base_dir, ".") || !strcmp(base_dir, "..")) { if (!ref) { - free(file); + free(copy); return NULL; } full_path = xstrdup(path); @@ -86,32 +98,48 @@ static char* append_readme_path(const char *filename, const char *ref, const cha if (!ref) { resolved_base = realpath(base_dir, NULL); resolved_full = realpath(full_path, NULL); - /* Require a path-separator boundary after the base so a sibling - * directory that merely shares the base as a name prefix (say - * repo.git-backup next to repo.git) cannot pass the check. */ if (!resolved_base || !resolved_full || - !starts_with(resolved_full, resolved_base) || - (resolved_full[strlen(resolved_base)] != '\0' && - resolved_full[strlen(resolved_base)] != '/')) { + !path_within(resolved_base, resolved_full)) { free(full_path); full_path = NULL; } } - free(file); + free(copy); free(resolved_base); free(resolved_full); return full_path; } +void cgit_print_summary(void) +{ + int columns = log_columns(); + + cgit_print_layout_start(); + html("<table summary='repository info' class='list nowrap'>"); + cgit_print_branches(ctx.cfg.summary_branches); + htmlf("<tr class='nohover'><td colspan='%d'> </td></tr>", columns); + cgit_print_tags(ctx.cfg.summary_tags); + if (ctx.cfg.summary_log > 0) { + htmlf("<tr class='nohover'><td colspan='%d'> </td></tr>", columns); + cgit_print_log(ctx.qry.head, 0, ctx.cfg.summary_log, NULL, + NULL, NULL, 0, 0, 0); + } + clone_urls_printed = 0; + cgit_add_clone_urls(print_clone_url); + html("</table>"); + cgit_print_layout_end(); +} + void cgit_print_repo_readme(const char *path) { char *filename, *ref, *mimetype; int free_filename = 0; mimetype = cgit_get_mimetype_for_filename(path); - if (mimetype && (!strncmp(mimetype, "image/", 6) || !strncmp(mimetype, "video/", 6))) { + if (mimetype && (starts_with(mimetype, "image/") || + starts_with(mimetype, "video/"))) { ctx.page.mimetype = mimetype; ctx.page.charset = NULL; cgit_print_plain(); @@ -129,21 +157,17 @@ void cgit_print_repo_readme(const char *path) if (path) { free_filename = 1; - filename = append_readme_path(filename, ref, path); + filename = resolve_about_path(filename, ref, path); if (!filename) goto done; } html("<div id='summary'>"); if (!ctx.repo->about_filter) { - /* No about-filter is configured, so there is nothing to turn - * the readme source into safe HTML. Escape it rather than serve - * repo content raw, which would let an untrusted repository - * inject script into the about page. The pre keeps the line - * structure of the text, which bare escaped output loses. Point - * about-filter at the bundled about-render.lua to render a - * markdown or man readme instead, see cgitrc.5.txt. - */ + // With no about-filter configured there is nothing to turn the + // readme source into safe HTML, so it is escaped rather than + // served raw, which would let an untrusted repository put + // script on this page. html("<pre class='plaintext'>"); if (ref) { cgit_print_file(filename, ref, 1, 1); @@ -155,9 +179,8 @@ void cgit_print_repo_readme(const char *path) } html("</pre>"); } else { - /* An about-filter is configured and is responsible for turning - * the source into safe HTML, so pass it through the filter raw. - */ + // The filter is what makes the source safe here, so it is fed + // through unescaped. cgit_open_filter(ctx.repo->about_filter, filename); if (ref) cgit_print_file(filename, ref, 1, 0); diff --git a/source/ui-summary.h b/source/ui-summary.h index cba696a..676cc47 100644 --- a/source/ui-summary.h +++ b/source/ui-summary.h @@ -1,7 +1,14 @@ -#ifndef UI_SUMMARY_H -#define UI_SUMMARY_H +/* + * Declarations for the two pages a repository opens with, the summary of its + * refs and recent commits and the about page built from its readme. cmd.c + * reaches both, and it hands the about page whatever path follows the page + * name in the url, so a readme can link to a file beside it. + */ + +#ifndef CGIT_UI_SUMMARY_H +#define CGIT_UI_SUMMARY_H extern void cgit_print_summary(void); extern void cgit_print_repo_readme(const char *path); -#endif /* UI_SUMMARY_H */ +#endif // CGIT_UI_SUMMARY_H diff --git a/source/ui-tag.c b/source/ui-tag.c index 7d01c5a..d09ff2b 100644 --- a/source/ui-tag.c +++ b/source/ui-tag.c @@ -1,43 +1,118 @@ -/* ui-tag.c: display a tag - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The tag page, which shows a single ref under refs/tags and is where the + * branch and tag listings link. An annotated tag is a git object in its own + * right, so its page carries the tagger, the date and the message, while a + * lightweight tag is only a name for a commit and gets a shorter table. Both + * forms link the object the tag points at and, when the repository enables + * snapshots, offer the tree at that tag for download as an archive. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-tag.h" +#include "filter.h" #include "html.h" +#include "parsing.h" +#include "shared.h" #include "ui-shared.h" +#include "ui-tag.h" + +static void print_object_row(struct object *obj) +{ + html("<tr><td>tagged object</td><td class='oid'>"); + cgit_object_link(obj); + html("</td></tr>\n"); +} + +static void print_download_row(const char *revname) +{ + html("<tr><th>download</th><td class='oid'>"); + cgit_print_snapshot_links(ctx.repo, revname, "<br/>"); + html("</td></tr>"); +} -static void print_tag_content(char *buf) +/* + * Cutting the subject out terminates the message in place, which is safe only + * because the caller frees it straight after. + */ +static void print_message(char *msg) { - char *p; + char *newline; - if (!buf) + if (!msg) return; html("<div class='commit-subject'>"); - p = strchr(buf, '\n'); - if (p) - *p = '\0'; - html_txt(buf); + newline = strchr(msg, '\n'); + if (newline) + *newline = '\0'; + html_txt(msg); html("</div>"); - if (p) { + if (newline) { html("<div class='commit-msg'>"); - html_txt(++p); + html_txt(newline + 1); html("</div>"); } } -static void print_download_links(char *revname) +static void print_annotated_tag(const char *revname, + const struct object_id *oid) { - html("<tr><th>download</th><td class='oid'>"); - cgit_print_snapshot_links(ctx.repo, revname, "<br/>"); - html("</td></tr>"); + struct tag *tag; + struct taginfo *info; + + tag = lookup_tag(the_repository, oid); + if (!tag || parse_tag(the_repository, tag) || + !(info = cgit_parse_tag(tag))) { + cgit_print_error_page(500, "Internal server error", + "Bad tag object: %s", revname); + return; + } + + cgit_print_layout_start(); + html("<table class='commit-info'>\n"); + html("<tr><td>tag name</td><td>"); + html_txt(revname); + htmlf(" (%s)</td></tr>\n", oid_to_hex(oid)); + if (info->tagger_date > 0) { + html("<tr><td>tag date</td><td>"); + html_txt(show_date(info->tagger_date, info->tagger_tz, + cgit_date_mode(DATE_ISO8601))); + html("</td></tr>\n"); + } + if (info->tagger) { + html("<tr><td>tagged by</td><td>"); + cgit_open_filter(ctx.repo->email_filter, info->tagger_email, + "tag"); + html_txt(info->tagger); + if (info->tagger_email && !ctx.cfg.noplainemail) { + html(" "); + html_txt(info->tagger_email); + } + cgit_close_filter(ctx.repo->email_filter); + html("</td></tr>\n"); + } + print_object_row(tag->tagged); + if (ctx.repo->snapshots) + print_download_row(revname); + html("</table>\n"); + print_message(info->msg); + cgit_print_layout_end(); + cgit_free_taginfo(info); +} + +static void print_lightweight_tag(const char *revname, struct object *obj) +{ + cgit_print_layout_start(); + html("<table class='commit-info'>\n"); + html("<tr><td>tag name</td><td>"); + html_txt(revname); + html("</td></tr>\n"); + print_object_row(obj); + if (ctx.repo->snapshots) + print_download_row(revname); + html("</table>\n"); + cgit_print_layout_end(); } void cgit_print_tag(char *revname) @@ -61,61 +136,10 @@ void cgit_print_tag(char *revname) "Bad object id: %s", oid_to_hex(&oid)); goto cleanup; } - if (obj->type == OBJ_TAG) { - struct tag *tag; - struct taginfo *info; - - tag = lookup_tag(the_repository, &oid); - if (!tag || parse_tag(the_repository, tag) || !(info = cgit_parse_tag(tag))) { - cgit_print_error_page(500, "Internal server error", - "Bad tag object: %s", revname); - goto cleanup; - } - cgit_print_layout_start(); - html("<table class='commit-info'>\n"); - html("<tr><td>tag name</td><td>"); - html_txt(revname); - htmlf(" (%s)</td></tr>\n", oid_to_hex(&oid)); - if (info->tagger_date > 0) { - html("<tr><td>tag date</td><td>"); - html_txt(show_date(info->tagger_date, info->tagger_tz, - cgit_date_mode(DATE_ISO8601))); - html("</td></tr>\n"); - } - if (info->tagger) { - html("<tr><td>tagged by</td><td>"); - cgit_open_filter(ctx.repo->email_filter, info->tagger_email, "tag"); - html_txt(info->tagger); - if (info->tagger_email && !ctx.cfg.noplainemail) { - html(" "); - html_txt(info->tagger_email); - } - cgit_close_filter(ctx.repo->email_filter); - html("</td></tr>\n"); - } - html("<tr><td>tagged object</td><td class='oid'>"); - cgit_object_link(tag->tagged); - html("</td></tr>\n"); - if (ctx.repo->snapshots) - print_download_links(revname); - html("</table>\n"); - print_tag_content(info->msg); - cgit_print_layout_end(); - cgit_free_taginfo(info); - } else { - cgit_print_layout_start(); - html("<table class='commit-info'>\n"); - html("<tr><td>tag name</td><td>"); - html_txt(revname); - html("</td></tr>\n"); - html("<tr><td>tagged object</td><td class='oid'>"); - cgit_object_link(obj); - html("</td></tr>\n"); - if (ctx.repo->snapshots) - print_download_links(revname); - html("</table>\n"); - cgit_print_layout_end(); - } + if (obj->type == OBJ_TAG) + print_annotated_tag(revname, &oid); + else + print_lightweight_tag(revname, obj); cleanup: strbuf_release(&fullref); diff --git a/source/ui-tag.h b/source/ui-tag.h index d295cdc..864d29c 100644 --- a/source/ui-tag.h +++ b/source/ui-tag.h @@ -1,6 +1,12 @@ -#ifndef UI_TAG_H -#define UI_TAG_H +/* + * The tag page, which shows a single ref under refs/tags on its own. It is + * one repository command among those in cmd.c, reached with the tag name in + * the id query string parameter, and the branch and tag listings link to it. + */ + +#ifndef CGIT_UI_TAG_H +#define CGIT_UI_TAG_H extern void cgit_print_tag(char *revname); -#endif /* UI_TAG_H */ +#endif // CGIT_UI_TAG_H diff --git a/source/ui-tree.c b/source/ui-tree.c index 64be0b6..0054b0c 100644 --- a/source/ui-tree.c +++ b/source/ui-tree.c @@ -1,76 +1,105 @@ -/* ui-tree.c: functions for tree output - * - * Copyright (C) 2006-2017 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) +/* + * The tree page, which renders one level of a repository at a revision as a + * table of the folders and files in it, or the contents of the file when the + * path names one. A file is shown as numbered source, passed through the + * repository's source filter when it has one, or as a hex dump when its bytes + * look binary. A folder whose only entry is another folder is followed, and + * both are named in the same row, so a long chain of single folders does not + * cost a page each. */ - #define USE_THE_REPOSITORY_VARIABLE +#define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" -#include "ui-tree.h" +#include "filter.h" #include "html.h" #include "ui-shared.h" +#include "ui-tree.h" -/* Bytes shown per row of the binary hex dump. */ #define HEXDUMP_ROW_BYTES 32 +#define HEXDUMP_ROW_HALF (HEXDUMP_ROW_BYTES / 2) +#define HEXDUMP_GAP_WIDTH 4 -struct tree_ls_entry { +enum walk_state { + WALK_LOOKING, + WALK_LISTING, + WALK_BLOB_SHOWN, + WALK_ERROR_SHOWN, +}; + +struct ls_entry { struct object_id oid; char *name; unsigned mode; }; struct walk_tree_context { - char *curr_rev; + char *rev; char *match_path; - int state; - struct tree_ls_entry *entries; + enum walk_state state; + struct ls_entry *entries; size_t entries_nr, entries_alloc; }; -static void print_text_buffer(const char *name, char *buf, unsigned long size) +/* + * The count runs past one as soon as a second entry or a file turns up, which + * ends the descent. + */ +struct only_child { + struct strbuf *path; + struct object_id oid; + char *name; + size_t count; +}; + +/* + * A formatted write per line meant a syscall and a temporary buffer for every + * line of the file, so the anchors are handed over in batches. Building the + * column whole was rejected because it would come to several times the size of + * the blob. + */ +static void print_linenumbers(const char *buf, unsigned long size) { - unsigned long lineno, idx; const char *numberfmt = "<a id='n%1$d' href='#n%1$d'>%1$d</a>\n"; + struct strbuf numbers = STRBUF_INIT; + unsigned long lineno = 0, idx = 0; - html("<table summary='blob content' class='blob'>\n"); - - if (ctx.cfg.enable_tree_linenumbers) { - struct strbuf numbers = STRBUF_INIT; + if (size) { + strbuf_addf(&numbers, numberfmt, ++lineno); - html("<tr><td class='linenumbers'><pre>"); - idx = 0; - lineno = 0; - - // Build the column in batches. A formatted write per line meant - // a syscall and a temporary buffer for every line of the file, - // which is most of what rendering a large blob cost, and the - // whole column at once would come to several times the blob. - if (size) { - strbuf_addf(&numbers, numberfmt, ++lineno); - while (idx < size - 1) { // skip absolute last newline - if (buf[idx] == '\n') { - strbuf_addf(&numbers, numberfmt, ++lineno); - if (numbers.len >= HTML_BATCH) { - html_raw(numbers.buf, numbers.len); - strbuf_reset(&numbers); - } + // The newline that ends the last line must not open a line of + // its own, so the final byte is left out of the scan. + while (idx < size - 1) { + if (buf[idx] == '\n') { + strbuf_addf(&numbers, numberfmt, ++lineno); + if (numbers.len >= HTML_BATCH) { + html_raw(numbers.buf, numbers.len); + strbuf_reset(&numbers); } - idx++; } - html_raw(numbers.buf, numbers.len); + idx++; } - strbuf_release(&numbers); - html("</pre></td>\n"); + html_raw(numbers.buf, numbers.len); } - else { + strbuf_release(&numbers); +} + +static void print_text_buffer(const char *filename, char *buf, + unsigned long size) +{ + html("<table summary='blob content' class='blob'>\n"); + + if (ctx.cfg.enable_tree_linenumbers) { + html("<tr><td class='linenumbers'><pre>"); + print_linenumbers(buf, size); + html("</pre></td>\n"); + } else { html("<tr>\n"); } if (ctx.repo->source_filter) { - char *filter_arg = xstrdup(name); + char *filter_arg = xstrdup(filename); + html("<td class='lines'><pre><code>"); cgit_open_filter(ctx.repo->source_filter, filter_arg); html_raw(buf, size); @@ -80,38 +109,47 @@ static void print_text_buffer(const char *name, char *buf, unsigned long size) return; } - /* No source filter is configured, so serve the text plain. Syntax - * highlighting ships as an optional source filter in custom/extensions/, - * keeping language knowledge out of the core. */ + // Syntax highlighting ships as one of the filters under + // custom/extensions, which keeps knowledge of languages out of cgit. html("<td class='lines'><pre><code>"); html_txt(buf); html("</code></pre></td></tr></table>\n"); } +/* + * git's ctype macros are locale free, but there is no isgraph among them, so + * the dump works one out for itself. + */ +static int is_graphic(unsigned char ch) +{ + return isprint(ch) && !isspace(ch); +} + static void print_binary_buffer(char *buf, unsigned long size) { - unsigned long ofs, idx; + unsigned long offset, idx; char ascii[HEXDUMP_ROW_BYTES + 1]; struct strbuf row = STRBUF_INIT; html("<table summary='blob content' class='bin-blob'>\n"); html("<tr><th>ofs</th><th>hex dump</th><th>ascii</th></tr>"); - for (ofs = 0; ofs < size; ofs += HEXDUMP_ROW_BYTES, buf += HEXDUMP_ROW_BYTES) { - // One write per row rather than one per byte. At the default - // blob limit the per-byte form spent almost all of its time in - // the kernel, which made a single request for a large binary - // blob far more expensive than the page it produced. + + // At the default blob size limit a write per byte spent almost all of + // its time in the kernel, so a row goes out in one write. + for (offset = 0; offset < size; + offset += HEXDUMP_ROW_BYTES, buf += HEXDUMP_ROW_BYTES) { strbuf_reset(&row); - strbuf_addf(&row, "<tr><td class='right'>%04lx</td><td class='hex'>", ofs); - for (idx = 0; idx < HEXDUMP_ROW_BYTES && ofs + idx < size; idx++) - strbuf_addf(&row, "%*s%02x", - idx == 16 ? 4 : 1, "", - buf[idx] & 0xff); + strbuf_addf(&row, "<tr><td class='right'>%04lx</td><td class='hex'>", offset); + for (idx = 0; idx < HEXDUMP_ROW_BYTES && offset + idx < size; idx++) { + int gap = idx == HEXDUMP_ROW_HALF ? HEXDUMP_GAP_WIDTH : 1; + + strbuf_addf(&row, "%*s%02x", gap, "", buf[idx] & 0xff); + } strbuf_addstr(&row, " </td><td class='hex'>"); html_raw(row.buf, row.len); - for (idx = 0; idx < HEXDUMP_ROW_BYTES && ofs + idx < size; idx++) - ascii[idx] = isgraph((unsigned char)buf[idx]) ? buf[idx] : '.'; + for (idx = 0; idx < HEXDUMP_ROW_BYTES && offset + idx < size; idx++) + ascii[idx] = is_graphic(buf[idx]) ? buf[idx] : '.'; ascii[idx] = '\0'; html_txt(ascii); html("</td></tr>\n"); @@ -120,9 +158,13 @@ static void print_binary_buffer(char *buf, unsigned long size) html("</table>\n"); } -/* Returns 1 if it opened the page layout for the caller to close, or 0 if - * it emitted a complete standalone error page. */ -static int print_object(const struct object_id *oid, const char *path, const char *basename, const char *rev) +/* + * Answer whether the page layout was left open for the caller to close. A + * false means a complete error page went out in place of the blob, so nothing + * may be added to it. + */ +static bool print_object(const struct object_id *oid, const char *path, + const char *filename, const char *rev) { enum object_type type; char *buf; @@ -133,22 +175,21 @@ static int print_object(const struct object_id *oid, const char *path, const cha if (type == OBJ_BAD) { cgit_print_error_page(404, "Not found", "Bad object name: %s", oid_to_hex(oid)); - return 0; + return false; } - /* Reject an oversized object before reading it whole into memory. */ if (ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size) { cgit_print_error_page(413, "Too large", "blob size (%luKB) exceeds display size limit (%dKB)", size / 1024, ctx.cfg.max_blob_size); - return 0; + return false; } buf = odb_read_object(the_repository->objects, oid, &type, &size); if (!buf) { cgit_print_error_page(500, "Internal server error", "Error reading object %s", oid_to_hex(oid)); - return 0; + return false; } is_binary = buffer_is_binary(buf, size); @@ -168,46 +209,37 @@ static int print_object(const struct object_id *oid, const char *path, const cha if (is_binary) print_binary_buffer(buf, size); else - print_text_buffer(basename, buf, size); + print_text_buffer(filename, buf, size); free(buf); - return 1; + return true; } -struct single_tree_ctx { - struct strbuf *path; - struct object_id oid; - char *name; - size_t count; -}; - -static int single_tree_cb(const struct object_id *oid, struct strbuf *base, - const char *pathname, unsigned mode, void *cbdata) +static int only_child_cb(const struct object_id *oid, struct strbuf *base, + const char *pathname, unsigned mode, void *data) { - // Not named ctx: that is the global request context, and shadowing it - // here would hide it from anything added to this function later. - struct single_tree_ctx *tree_ctx = cbdata; + struct only_child *child = data; - if (++tree_ctx->count > 1) + if (++child->count > 1) return -1; if (!S_ISDIR(mode)) { - tree_ctx->count = 2; + child->count = 2; return -1; } - tree_ctx->name = xstrdup(pathname); - oidcpy(&tree_ctx->oid, oid); - strbuf_addf(tree_ctx->path, "/%s", pathname); + child->name = xstrdup(pathname); + oidcpy(&child->oid, oid); + strbuf_addf(child->path, "/%s", pathname); return 0; } -static void write_tree_link(const struct object_id *oid, char *name, +static void print_dir_chain(const struct object_id *oid, char *name, char *rev, struct strbuf *fullpath) { size_t initial_length = fullpath->len; struct tree *tree; - struct single_tree_ctx tree_ctx = { + struct only_child child = { .path = fullpath, .count = 1, }; @@ -215,34 +247,34 @@ static void write_tree_link(const struct object_id *oid, char *name, .nr = 0 }; - oidcpy(&tree_ctx.oid, oid); + oidcpy(&child.oid, oid); - while (tree_ctx.count == 1) { + while (child.count == 1) { cgit_tree_link(name, NULL, "ls-dir", ctx.qry.head, rev, fullpath->buf); - tree = lookup_tree(the_repository, &tree_ctx.oid); + tree = lookup_tree(the_repository, &child.oid); if (!tree) return; - free(tree_ctx.name); - tree_ctx.name = NULL; - tree_ctx.count = 0; + free(child.name); + child.name = NULL; + child.count = 0; - read_tree(the_repository, tree, &paths, single_tree_cb, &tree_ctx); + read_tree(the_repository, tree, &paths, only_child_cb, &child); - if (tree_ctx.count != 1) + if (child.count != 1) break; html(" / "); - name = tree_ctx.name; + name = child.name; } strbuf_setlen(fullpath, initial_length); } -static void render_ls_item(const struct object_id *oid, const char *pathname, - unsigned mode, struct walk_tree_context *walk_tree_ctx) +static void print_ls_row(const struct object_id *oid, const char *pathname, + unsigned mode, struct walk_tree_context *walk) { char *name; struct strbuf fullpath = STRBUF_INIT; @@ -259,9 +291,11 @@ static void render_ls_item(const struct object_id *oid, const char *pathname, if (!S_ISGITLINK(mode)) { type = odb_read_object_info(the_repository->objects, oid, &size); if (type == OBJ_BAD) { - htmlf("<tr><td colspan='3'>Bad object: %s %s</td></tr>", - name, - oid_to_hex(oid)); + // The name comes from the tree, so it can hold + // anything a commit was allowed to record. + html("<tr><td colspan='3'>Bad object: "); + html_txt(name); + htmlf(" %s</td></tr>", oid_to_hex(oid)); goto cleanup; } } @@ -272,15 +306,14 @@ static void render_ls_item(const struct object_id *oid, const char *pathname, if (S_ISGITLINK(mode)) { cgit_submodule_link("ls-mod", fullpath.buf, oid_to_hex(oid)); } else if (S_ISDIR(mode)) { - write_tree_link(oid, name, walk_tree_ctx->curr_rev, - &fullpath); + print_dir_chain(oid, name, walk->rev, &fullpath); } else { char *ext = strrchr(name, '.'); strbuf_addstr(&class, "ls-blob"); if (ext) strbuf_addf(&class, " %s", ext + 1); cgit_tree_link(name, NULL, class.buf, ctx.qry.head, - walk_tree_ctx->curr_rev, fullpath.buf); + walk->rev, fullpath.buf); } if (S_ISLNK(mode)) { html(" -> "); @@ -293,7 +326,7 @@ static void render_ls_item(const struct object_id *oid, const char *pathname, strbuf_addf(&linkpath, "/../%s", buf); strbuf_normalize_path(&linkpath); cgit_tree_link(buf, NULL, class.buf, ctx.qry.head, - walk_tree_ctx->curr_rev, linkpath.buf); + walk->rev, linkpath.buf); free(buf); strbuf_release(&linkpath); } @@ -301,17 +334,17 @@ static void render_ls_item(const struct object_id *oid, const char *pathname, html("<td>"); cgit_log_link("log", NULL, "button", ctx.qry.head, - walk_tree_ctx->curr_rev, fullpath.buf, 0, NULL, NULL, + walk->rev, fullpath.buf, 0, NULL, NULL, ctx.qry.showmsg, 0); if (ctx.repo->enable_stats) cgit_stats_link("stats", NULL, "button", ctx.qry.head, fullpath.buf); if (!S_ISGITLINK(mode)) cgit_plain_link("plain", NULL, "button", ctx.qry.head, - walk_tree_ctx->curr_rev, fullpath.buf); + walk->rev, fullpath.buf); if (!S_ISDIR(mode) && ctx.repo->enable_blame) cgit_blame_link("blame", NULL, "button", ctx.qry.head, - walk_tree_ctx->curr_rev, fullpath.buf); + walk->rev, fullpath.buf); html("</td></tr>\n"); cleanup: @@ -321,50 +354,50 @@ cleanup: } static int ls_item(const struct object_id *oid, struct strbuf *base, - const char *pathname, unsigned mode, void *cbdata) + const char *pathname, unsigned mode, void *data) { - struct walk_tree_context *walk_tree_ctx = cbdata; + struct walk_tree_context *walk = data; - /* When grouping directories first, collect the entries now and render - * them once the whole level has been read (see ls_flush). */ + // With folders grouped first, a row cannot go out as it arrives, + // because git hands the level over in name order. if (ctx.cfg.enable_tree_group_dirs) { - struct tree_ls_entry *e; - ALLOC_GROW(walk_tree_ctx->entries, walk_tree_ctx->entries_nr + 1, - walk_tree_ctx->entries_alloc); - e = &walk_tree_ctx->entries[walk_tree_ctx->entries_nr++]; - oidcpy(&e->oid, oid); - e->name = xstrdup(pathname); - e->mode = mode; + struct ls_entry *entry; + + ALLOC_GROW(walk->entries, walk->entries_nr + 1, + walk->entries_alloc); + entry = &walk->entries[walk->entries_nr++]; + oidcpy(&entry->oid, oid); + entry->name = xstrdup(pathname); + entry->mode = mode; return 0; } - render_ls_item(oid, pathname, mode, walk_tree_ctx); + print_ls_row(oid, pathname, mode, walk); return 0; } -/* Render any entries collected by ls_item, directories first and then the - * rest, keeping git's ordering within each group. A no-op unless directory - * grouping is enabled, in which case nothing was collected. */ -static void ls_flush(struct walk_tree_context *walk_tree_ctx) +static void ls_flush(struct walk_tree_context *walk) { size_t i; - for (i = 0; i < walk_tree_ctx->entries_nr; i++) { - struct tree_ls_entry *e = &walk_tree_ctx->entries[i]; - if (S_ISDIR(e->mode)) - render_ls_item(&e->oid, e->name, e->mode, walk_tree_ctx); + for (i = 0; i < walk->entries_nr; i++) { + struct ls_entry *entry = &walk->entries[i]; + + if (S_ISDIR(entry->mode)) + print_ls_row(&entry->oid, entry->name, entry->mode, walk); } - for (i = 0; i < walk_tree_ctx->entries_nr; i++) { - struct tree_ls_entry *e = &walk_tree_ctx->entries[i]; - if (!S_ISDIR(e->mode)) - render_ls_item(&e->oid, e->name, e->mode, walk_tree_ctx); + for (i = 0; i < walk->entries_nr; i++) { + struct ls_entry *entry = &walk->entries[i]; + + if (!S_ISDIR(entry->mode)) + print_ls_row(&entry->oid, entry->name, entry->mode, walk); } - for (i = 0; i < walk_tree_ctx->entries_nr; i++) - free(walk_tree_ctx->entries[i].name); - free(walk_tree_ctx->entries); - walk_tree_ctx->entries = NULL; - walk_tree_ctx->entries_nr = walk_tree_ctx->entries_alloc = 0; + for (i = 0; i < walk->entries_nr; i++) + free(walk->entries[i].name); + free(walk->entries); + walk->entries = NULL; + walk->entries_nr = walk->entries_alloc = 0; } static void ls_head(void) @@ -385,7 +418,8 @@ static void ls_tail(void) cgit_print_layout_end(); } -static void ls_tree(const struct object_id *oid, const char *path, struct walk_tree_context *walk_tree_ctx) +static void ls_tree(const struct object_id *oid, const char *path, + struct walk_tree_context *walk) { struct tree *tree; struct pathspec paths = { @@ -400,64 +434,64 @@ static void ls_tree(const struct object_id *oid, const char *path, struct walk_t } ls_head(); - read_tree(the_repository, tree, &paths, ls_item, walk_tree_ctx); - ls_flush(walk_tree_ctx); + read_tree(the_repository, tree, &paths, ls_item, walk); + ls_flush(walk); ls_tail(); } static int walk_tree(const struct object_id *oid, struct strbuf *base, - const char *pathname, unsigned mode, void *cbdata) + const char *pathname, unsigned mode, void *data) { - struct walk_tree_context *walk_tree_ctx = cbdata; + struct walk_tree_context *walk = data; - if (walk_tree_ctx->state == 0) { - struct strbuf buffer = STRBUF_INIT; + if (walk->state == WALK_LOOKING) { + struct strbuf fullpath = STRBUF_INIT; - strbuf_addbuf(&buffer, base); - strbuf_addstr(&buffer, pathname); - if (strcmp(walk_tree_ctx->match_path, buffer.buf)) + strbuf_addbuf(&fullpath, base); + strbuf_addstr(&fullpath, pathname); + if (strcmp(walk->match_path, fullpath.buf)) return READ_TREE_RECURSIVE; if (S_ISDIR(mode)) { - walk_tree_ctx->state = 1; - cgit_set_title_from_path(buffer.buf); - strbuf_release(&buffer); + walk->state = WALK_LISTING; + cgit_set_title_from_path(fullpath.buf); + strbuf_release(&fullpath); ls_head(); return READ_TREE_RECURSIVE; } else { - /* state 2: layout left open for us to close; state 3: - * print_object already emitted a standalone error page. */ - walk_tree_ctx->state = - print_object(oid, buffer.buf, pathname, - walk_tree_ctx->curr_rev) ? 2 : 3; - strbuf_release(&buffer); + bool shown = print_object(oid, fullpath.buf, + pathname, walk->rev); + + walk->state = shown ? WALK_BLOB_SHOWN : WALK_ERROR_SHOWN; + strbuf_release(&fullpath); return 0; } } - ls_item(oid, base, pathname, mode, walk_tree_ctx); + ls_item(oid, base, pathname, mode, walk); return 0; } -/* - * Show a tree or a blob - * rev: the commit pointing at the root tree object - * path: path to tree or blob - */ +// Either argument may be null, in which case the head of the current query +// stands in for the revision and the listing starts at the root of the tree. void cgit_print_tree(const char *rev, char *path) { struct object_id oid; struct commit *commit; + int path_len = path ? strlen(path) : 0; + // nowildcard_len matches len so git treats the path as literal rather + // than as a glob. struct pathspec_item path_items = { .match = path, - .len = path ? strlen(path) : 0 + .len = path_len, + .nowildcard_len = path_len }; struct pathspec paths = { .nr = path ? 1 : 0, .items = &path_items }; - struct walk_tree_context walk_tree_ctx = { + struct walk_tree_context walk = { .match_path = path, - .state = 0 + .state = WALK_LOOKING }; if (!rev) @@ -475,24 +509,25 @@ void cgit_print_tree(const char *rev, char *path) return; } - walk_tree_ctx.curr_rev = xstrdup(rev); + walk.rev = xstrdup(rev); if (path == NULL) { - ls_tree(get_commit_tree_oid(commit), NULL, &walk_tree_ctx); + ls_tree(get_commit_tree_oid(commit), NULL, &walk); goto cleanup; } read_tree(the_repository, repo_get_commit_tree(the_repository, commit), - &paths, walk_tree, &walk_tree_ctx); - if (walk_tree_ctx.state == 1) { - ls_flush(&walk_tree_ctx); + &paths, walk_tree, &walk); + if (walk.state == WALK_LISTING) { + ls_flush(&walk); ls_tail(); - } else if (walk_tree_ctx.state == 2) + } else if (walk.state == WALK_BLOB_SHOWN) cgit_print_layout_end(); - else if (walk_tree_ctx.state == 0) + else if (walk.state == WALK_LOOKING) cgit_print_error_page(404, "Not found", "Path not found"); - /* state 3: print_object already emitted a complete error page */ + // WALK_ERROR_SHOWN is left alone, since the error page print_object + // put out is already complete. cleanup: - free(walk_tree_ctx.curr_rev); + free(walk.rev); } diff --git a/source/ui-tree.h b/source/ui-tree.h index bbd34e3..f99e085 100644 --- a/source/ui-tree.h +++ b/source/ui-tree.h @@ -1,6 +1,13 @@ -#ifndef UI_TREE_H -#define UI_TREE_H +/* + * The tree page, which browses a repository at a revision, listing the folder + * a path selects or showing the contents of the file it names. It is one of + * the repository commands in cmd.c, and it is where the plain, blame, log and + * stats views of a path are linked from. + */ + +#ifndef CGIT_UI_TREE_H +#define CGIT_UI_TREE_H extern void cgit_print_tree(const char *rev, char *path); -#endif /* UI_TREE_H */ +#endif // CGIT_UI_TREE_H |
