diff options
Diffstat (limited to '')
| -rw-r--r-- | source/cache.c | 462 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
1 file changed, 245 insertions, 217 deletions
diff --git a/source/cache.c b/source/cache.c index c6d0427..d6e450a 100644 --- a/source/cache.c +++ b/source/cache.c @@ -1,33 +1,49 @@ -/* cache.c: cache management - * - * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com> - * - * Licensed under GNU General Public License v2 - * (see LICENSE.txt for full license text) - * - * - * The cache is just a directory structure where each file is a cache slot, - * and each filename is based on the hash of some key (e.g. the cgit url). - * Each file contains the full key followed by the cached content for that - * key. - * +/* + * The cache that lets a repeated request be answered from disk instead of + * being rendered again. A slot is one file named after the hash of the + * request key, holding that key and then the page it rendered to, and a lock + * file beside it is where a replacement page is written before being renamed + * over the slot. Only the process holding that lock rebuilds a slot, so a + * request arriving while a stale slot is being rebuilt is served the stale + * page, and a request with no usable slot at all renders straight to the + * client without caching anything. */ -#include "cgit.h" #include "cache.h" +#include "cgit.h" #include "html.h" +#include "shared.h" #ifdef HAVE_LINUX_SENDFILE #include <sys/sendfile.h> #endif +// One read of a slot file. The stored key has to be recognised out of a +// single such read, so this also bounds how long a cacheable key can be, see +// key_fits_slot. #define CACHE_BUFSIZE (1024 * 4) -/* Crude implementation of 32-bit FNV-1 hash algorithm, - * see http://www.isthe.com/chongo/tech/comp/fnv/ for details - * about the magic numbers. - */ + +// A slot is named by this many hex digits of the key hash, which is also how +// cache_ls tells slots from the lock files sitting beside them. +#define SLOT_NAME_LEN 8 + +// The 32 bit FNV-1 offset basis and prime. #define FNV_OFFSET 0x811c9dc5 #define FNV_PRIME 0x01000193 +/* + * Cache trouble goes to stderr, which under CGI is the web server's error + * log, so that it cannot land in the middle of the page being written to + * stdout. + */ +__attribute__((format (printf,1,2))) +static void log_error(const char *format, ...) +{ + va_list args; + va_start(args, format); + vfprintf(stderr, format, args); + va_end(args); +} + struct cache_slot { const char *key; size_t keylen; @@ -35,56 +51,56 @@ struct cache_slot { cache_fill_fn fn; int cache_fd; int lock_fd; - int stdout_fd; - const char *cache_name; - const char *lock_name; - int match; - struct stat cache_st; - int bufsize; + int saved_stdout; + const char *path; + const char *lock_path; + int key_matches; + // The slot as it was when it was opened, or the lock file once + // fill_slot has written a page into it. + struct stat st; + // How much of the slot was read into buf, not the size of buf. + int buflen; char buf[CACHE_BUFSIZE]; }; -/* Open an existing cache slot and fill the cache buffer with - * (part of) the content of the cache file. Return 0 on success - * and errno otherwise. - */ static int open_slot(struct cache_slot *slot) { - char *bufz; - ssize_t bufkeylen = -1; + char *nul; + ssize_t keylen = -1; - slot->cache_fd = open(slot->cache_name, O_RDONLY); + slot->cache_fd = open(slot->path, O_RDONLY); if (slot->cache_fd == -1) return errno; - if (fstat(slot->cache_fd, &slot->cache_st)) + if (fstat(slot->cache_fd, &slot->st)) return errno; - slot->bufsize = xread(slot->cache_fd, slot->buf, sizeof(slot->buf)); - if (slot->bufsize < 0) + slot->buflen = xread(slot->cache_fd, slot->buf, sizeof(slot->buf)); + if (slot->buflen < 0) return errno; - bufz = memchr(slot->buf, 0, slot->bufsize); - if (bufz) - bufkeylen = bufz - slot->buf; + nul = memchr(slot->buf, 0, slot->buflen); + if (nul) + keylen = nul - slot->buf; if (slot->key) - slot->match = bufkeylen >= 0 && (size_t)bufkeylen == slot->keylen && - !memcmp(slot->key, slot->buf, bufkeylen + 1); + slot->key_matches = keylen >= 0 && + (size_t)keylen == slot->keylen && + !memcmp(slot->key, slot->buf, keylen + 1); return 0; } -/* A key longer than the buffer above can never be read back, so a slot keyed - * on one would never match and every such request would regenerate its page - * while still writing a slot nothing can use. Those requests skip the cache - * instead. */ +/* + * A key longer than the buffer above can never be read back by open_slot, so + * a slot keyed on one would never match and every such request would + * regenerate its page while still writing a slot nothing can use. + */ static int key_fits_slot(const char *key) { return strlen(key) + 1 <= CACHE_BUFSIZE; } -/* Close the active cache slot */ static int close_slot(struct cache_slot *slot) { int err = 0; @@ -97,7 +113,6 @@ static int close_slot(struct cache_slot *slot) return err; } -/* Print the content of the active cache slot (but skip the key). */ static int print_slot(struct cache_slot *slot) { off_t off; @@ -108,7 +123,7 @@ static int print_slot(struct cache_slot *slot) off = slot->keylen + 1; #ifdef HAVE_LINUX_SENDFILE - size = slot->cache_st.st_size; + size = slot->st.st_size; do { ssize_t ret; @@ -116,7 +131,10 @@ static int print_slot(struct cache_slot *slot) if (ret < 0) { if (errno == EAGAIN || errno == EINTR) continue; - /* Fall back to read/write on EINVAL or ENOSYS */ + // EINVAL and ENOSYS mean this kernel or this pair of + // descriptors cannot do sendfile at all, so fall back + // to the read and write loop rather than fail the + // request. if (errno == EINVAL || errno == ENOSYS) break; return errno; @@ -141,30 +159,41 @@ static int print_slot(struct cache_slot *slot) } while (1); } -/* Check if the slot has expired */ +static int serve_slot(struct cache_slot *slot) +{ + int err; + + err = print_slot(slot); + if (err) + log_error("[cgit] error printing cache %s: %s (%d)\n", + slot->path, + strerror(err), + err); + return err; +} + static int is_expired(struct cache_slot *slot) { if (slot->ttl < 0) return 0; - else - return slot->cache_st.st_mtime + slot->ttl * 60 < time(NULL); + return slot->st.st_mtime + slot->ttl * SECONDS_PER_MINUTE < time(NULL); } -/* Check if the slot has been modified since we opened it. - * NB: If stat() fails, we pretend the file is modified. +/* + * A stat that fails counts as modified, so that the caller leaves alone a file + * it was unable to look at. */ static int is_modified(struct cache_slot *slot) { - struct stat st; + struct stat current; - if (stat(slot->cache_name, &st)) + if (stat(slot->path, ¤t)) return 1; - return (st.st_ino != slot->cache_st.st_ino || - st.st_mtime != slot->cache_st.st_mtime || - st.st_size != slot->cache_st.st_size); + return (current.st_ino != slot->st.st_ino || + current.st_mtime != slot->st.st_mtime || + current.st_size != slot->st.st_size); } -/* Close an open lockfile */ static int close_lock(struct cache_slot *slot) { int err = 0; @@ -177,9 +206,11 @@ static int close_lock(struct cache_slot *slot) return err; } -/* Create a lockfile used to store the generated content for a cache - * slot, and write the slot key + \0 into it. - * Returns 0 on success and errno otherwise. +/* + * The lock file becomes the slot once it is renamed, so it has to open with + * the key the same way a slot does. The lock is taken without blocking, + * because failing to get it is how a second process learns that this slot is + * already being rebuilt, so it returns an errno instead of waiting. */ static int lock_slot(struct cache_slot *slot) { @@ -190,7 +221,7 @@ static int lock_slot(struct cache_slot *slot) .l_len = 0, }; - slot->lock_fd = open(slot->lock_name, O_RDWR | O_CREAT, + slot->lock_fd = open(slot->lock_path, O_RDWR | O_CREAT, S_IRUSR | S_IWUSR); if (slot->lock_fd == -1) return errno; @@ -200,6 +231,8 @@ static int lock_slot(struct cache_slot *slot) slot->lock_fd = -1; return saved_errno; } + // A run that died before its rename leaves the lock file behind, so + // start from empty now that nobody else can be writing it. if (ftruncate(slot->lock_fd, 0) < 0) return errno; if (xwrite(slot->lock_fd, slot->key, slot->keylen + 1) < 0) @@ -207,24 +240,19 @@ static int lock_slot(struct cache_slot *slot) return 0; } -/* Release the current lockfile. If `replace_old_slot` is set the - * lockfile replaces the old cache slot, otherwise the lockfile is - * just deleted. - */ static int unlock_slot(struct cache_slot *slot, int replace_old_slot) { int err; if (replace_old_slot) - err = rename(slot->lock_name, slot->cache_name); + err = rename(slot->lock_path, slot->path); else - err = unlink(slot->lock_name); + err = unlink(slot->lock_path); - /* Restore stdout and close the temporary FD. */ - if (slot->stdout_fd >= 0) { - dup2(slot->stdout_fd, STDOUT_FILENO); - close(slot->stdout_fd); - slot->stdout_fd = -1; + if (slot->saved_stdout >= 0) { + dup2(slot->saved_stdout, STDOUT_FILENO); + close(slot->saved_stdout); + slot->saved_stdout = -1; } if (err) @@ -233,48 +261,84 @@ static int unlock_slot(struct cache_slot *slot, int replace_old_slot) return 0; } -/* Generate the content for the current cache slot by redirecting - * stdout to the lock-fd and invoking the callback function +// Only one slot is ever being filled at a time, so a single pointer is enough +// for cache_abandon_fill to find its way back to the client. +static struct cache_slot *slot_being_filled; + +void cache_abandon_fill(void) +{ + struct cache_slot *slot = slot_being_filled; + + if (!slot) + return; + slot_being_filled = NULL; + + // Emptied while stdout still points at the lock file, so the half + // rendered page goes into the file about to be removed rather than + // reaching the client ahead of whatever is written next. + html_flush(); + + if (slot->saved_stdout >= 0) { + dup2(slot->saved_stdout, STDOUT_FILENO); + close(slot->saved_stdout); + slot->saved_stdout = -1; + } + unlink(slot->lock_path); +} + +/* + * Renders with stdout pointed at the lock file, and on success or failure + * alike it is unlock_slot that gives stdout back. */ static int fill_slot(struct cache_slot *slot) { - /* Preserve stdout */ - slot->stdout_fd = dup(STDOUT_FILENO); - if (slot->stdout_fd == -1) + slot->saved_stdout = dup(STDOUT_FILENO); + if (slot->saved_stdout == -1) return errno; - /* Redirect stdout to lockfile */ if (dup2(slot->lock_fd, STDOUT_FILENO) == -1) return errno; - /* Generate cache content */ + slot_being_filled = slot; slot->fn(); + slot_being_filled = NULL; - /* Make sure any buffered data is flushed to the file */ + // The page is sitting in html.c's buffer and then in stdio's, and all + // of it has to reach the lock file before that file is renamed into + // place. html_flush(); if (fflush(stdout)) return errno; - /* update stat info */ - if (fstat(slot->lock_fd, &slot->cache_st)) + // print_slot takes the length of what it copies from here, and what + // it copies after a fill is the lock file rather than the old slot. + if (fstat(slot->lock_fd, &slot->st)) return errno; return 0; } -unsigned long cache_hash_str(const char *str) +/* + * Giving up is always a valid outcome, because the caller still has the + * expired copy open and can serve that. + */ +static void refresh_slot(struct cache_slot *slot) { - unsigned long h = FNV_OFFSET; - unsigned char *s = (unsigned char *)str; - - if (!s) - return h; + if (lock_slot(slot)) + return; - while (*s) { - h *= FNV_PRIME; - h ^= *s++; + // If another process replaced the slot between open_slot and + // lock_slot, the copy already open is served rather than the newer + // one, which would mean opening that file and comparing the key in it, + // not worth a second descriptor and read on every expiry. + if (is_modified(slot) || fill_slot(slot)) { + unlock_slot(slot, 0); + close_lock(slot); + } else { + close_slot(slot); + unlock_slot(slot, 1); + slot->cache_fd = slot->lock_fd; } - return h; } static int process_slot(struct cache_slot *slot) @@ -282,216 +346,180 @@ static int process_slot(struct cache_slot *slot) int err; err = open_slot(slot); - if (!err && slot->match) { - if (is_expired(slot)) { - if (!lock_slot(slot)) { - /* If the cachefile has been replaced between - * `open_slot` and `lock_slot`, we'll just - * serve the stale content from the original - * cachefile. This way we avoid pruning the - * newly generated slot. The same code-path - * is chosen if fill_slot() fails for some - * reason. - * - * TODO? check if the new slot contains the - * same key as the old one, since we would - * prefer to serve the newest content. - * This will require us to open yet another - * file-descriptor and read and compare the - * key from the new file, so for now we're - * lazy and just ignore the new file. - */ - if (is_modified(slot) || fill_slot(slot)) { - unlock_slot(slot, 0); - close_lock(slot); - } else { - close_slot(slot); - unlock_slot(slot, 1); - slot->cache_fd = slot->lock_fd; - } - } - } - if ((err = print_slot(slot)) != 0) { - cache_log("[cgit] error printing cache %s: %s (%d)\n", - slot->cache_name, - strerror(err), - err); - } + if (!err && slot->key_matches) { + if (is_expired(slot)) + refresh_slot(slot); + err = serve_slot(slot); close_slot(slot); return err; } - /* If the cache slot does not exist (or its key doesn't match the - * current key), lets try to create a new cache slot for this - * request. If this fails (for whatever reason), lets just generate - * the content without caching it and fool the caller to believe - * everything worked out (but print a warning on stdout). - */ - + // If any part of creating a slot fails the page is still rendered + // straight to the client and the caller is told the request succeeded, + // because it did. close_slot(slot); if ((err = lock_slot(slot)) != 0) { - cache_log("[cgit] Unable to lock slot %s: %s (%d)\n", - slot->lock_name, strerror(err), err); + log_error("[cgit] Unable to lock slot %s: %s (%d)\n", + slot->lock_path, strerror(err), err); slot->fn(); return 0; } if ((err = fill_slot(slot)) != 0) { - cache_log("[cgit] Unable to fill slot %s: %s (%d)\n", - slot->lock_name, strerror(err), err); + log_error("[cgit] Unable to fill slot %s: %s (%d)\n", + slot->lock_path, strerror(err), err); unlock_slot(slot, 0); close_lock(slot); slot->fn(); return 0; } - // We've got a valid cache slot in the lock file, which - // is about to replace the old cache slot. But if we - // release the lockfile and then try to open the new cache - // slot, we might get a race condition with a concurrent - // writer for the same cache slot (with a different key). - // Lets avoid such a race by just printing the content of - // the lock file. + + // Opening the slot by name after the rename could land on a file a + // concurrent writer put there for a different key, so what gets + // printed is the descriptor still open on the lock file. slot->cache_fd = slot->lock_fd; unlock_slot(slot, 1); - if ((err = print_slot(slot)) != 0) { - cache_log("[cgit] error printing cache %s: %s (%d)\n", - slot->cache_name, - strerror(err), - err); - } + err = serve_slot(slot); close_slot(slot); return err; } -/* Print cached content to stdout, generate the content if necessary. */ +// The result lives in a static buffer and is only good until the next call. +static char *format_time(const char *format, time_t when) +{ + static char buf[64]; + struct tm tm; + + if (!when) + return NULL; + gmtime_r(&when, &tm); + strftime(buf, sizeof(buf) - 1, format, &tm); + return buf; +} + +/* + * The accumulator is an unsigned long rather than a fixed 32 bit type, so on a + * 64 bit host this is not the published FNV-1 value. All that decides is which + * slot a key lands in, and nothing outside a single build has to agree on the + * answer. + */ +unsigned long cache_hash_str(const char *str) +{ + unsigned long h = FNV_OFFSET; + unsigned char *s = (unsigned char *)str; + + if (!s) + return h; + + while (*s) { + h *= FNV_PRIME; + h ^= *s++; + } + return h; +} + int cache_process(int size, const char *path, const char *key, int ttl, cache_fill_fn fn) { unsigned long hash; int i; - struct strbuf filename = STRBUF_INIT; - struct strbuf lockname = STRBUF_INIT; + struct strbuf slot_path = STRBUF_INIT; + struct strbuf lock_path = STRBUF_INIT; struct cache_slot slot; int result; - /* If the cache is disabled, just generate the content */ if (size <= 0 || ttl == 0) { fn(); return 0; } - /* Verify input, calculate filenames */ if (!path) { - cache_log("[cgit] Cache path not specified, caching is disabled\n"); + log_error("[cgit] Cache path not specified, caching is disabled\n"); fn(); return 0; } if (!key) key = ""; if (!key_fits_slot(key)) { - cache_log("[cgit] Cache key too long for a slot, caching is " + log_error("[cgit] Cache key too long for a slot, caching is " "disabled for this request\n"); fn(); return 0; } hash = cache_hash_str(key) % size; - strbuf_addstr(&filename, path); - strbuf_ensure_end(&filename, '/'); - for (i = 0; i < 8; i++) { - strbuf_addf(&filename, "%x", (unsigned char)(hash & 0xf)); + strbuf_addstr(&slot_path, path); + strbuf_ensure_end(&slot_path, '/'); + for (i = 0; i < SLOT_NAME_LEN; i++) { + strbuf_addf(&slot_path, "%x", (unsigned char)(hash & 0xf)); hash >>= 4; } - strbuf_addbuf(&lockname, &filename); - strbuf_addstr(&lockname, ".lock"); + strbuf_addbuf(&lock_path, &slot_path); + strbuf_addstr(&lock_path, ".lock"); slot.fn = fn; slot.ttl = ttl; - slot.stdout_fd = -1; - slot.cache_name = filename.buf; - slot.lock_name = lockname.buf; + slot.saved_stdout = -1; + slot.path = slot_path.buf; + slot.lock_path = lock_path.buf; slot.key = key; slot.keylen = strlen(key); result = process_slot(&slot); - strbuf_release(&filename); - strbuf_release(&lockname); + strbuf_release(&slot_path); + strbuf_release(&lock_path); return result; } -/* Return a strftime formatted date/time - * NB: the result from this function is to shared memory - */ -static char *sprintftime(const char *format, time_t time) -{ - static char buf[64]; - struct tm tm; - - if (!time) - return NULL; - gmtime_r(&time, &tm); - strftime(buf, sizeof(buf)-1, format, &tm); - return buf; -} - int cache_ls(const char *path) { DIR *dir; struct dirent *ent; int err = 0; + // A NULL key leaves open_slot with nothing to compare against, so + // every slot it opens is simply read. struct cache_slot slot = { NULL }; - struct strbuf fullname = STRBUF_INIT; + struct strbuf slot_path = STRBUF_INIT; size_t prefixlen; char *nul; int keylen; if (!path) { - cache_log("[cgit] cache path not specified\n"); + log_error("[cgit] cache path not specified\n"); return -1; } dir = opendir(path); if (!dir) { err = errno; - cache_log("[cgit] unable to open path %s: %s (%d)\n", + log_error("[cgit] unable to open path %s: %s (%d)\n", path, strerror(err), err); return err; } - strbuf_addstr(&fullname, path); - strbuf_ensure_end(&fullname, '/'); - prefixlen = fullname.len; + strbuf_addstr(&slot_path, path); + strbuf_ensure_end(&slot_path, '/'); + prefixlen = slot_path.len; while ((ent = readdir(dir)) != NULL) { - if (strlen(ent->d_name) != 8) + if (strlen(ent->d_name) != SLOT_NAME_LEN) continue; - strbuf_setlen(&fullname, prefixlen); - strbuf_addstr(&fullname, ent->d_name); - slot.cache_name = fullname.buf; + strbuf_setlen(&slot_path, prefixlen); + strbuf_addstr(&slot_path, ent->d_name); + slot.path = slot_path.buf; if ((err = open_slot(&slot)) != 0) { - cache_log("[cgit] unable to open path %s: %s (%d)\n", - fullname.buf, strerror(err), err); + log_error("[cgit] unable to open path %s: %s (%d)\n", + slot_path.buf, strerror(err), err); continue; } - // The stored key is NUL-terminated within the buffer, but a - // truncated or corrupt slot may not be. Bound the print to - // what was read so %s cannot run off the end. - nul = memchr(slot.buf, 0, slot.bufsize); - keylen = nul ? (int)(nul - slot.buf) : slot.bufsize; + // A truncated or corrupt slot may hold no NUL, so the print is + // bounded by what was read and cannot run off the end. + nul = memchr(slot.buf, 0, slot.buflen); + keylen = nul ? (int)(nul - slot.buf) : slot.buflen; htmlf("%s %s %10"PRIuMAX" %.*s\n", - fullname.buf, - sprintftime("%Y-%m-%d %H:%M:%S", - slot.cache_st.st_mtime), - (uintmax_t)slot.cache_st.st_size, + slot_path.buf, + format_time("%Y-%m-%d %H:%M:%S", + slot.st.st_mtime), + (uintmax_t)slot.st.st_size, keylen, slot.buf); close_slot(&slot); } closedir(dir); - strbuf_release(&fullname); + strbuf_release(&slot_path); return 0; } - -/* Print a message to stdout */ -void cache_log(const char *format, ...) -{ - va_list args; - va_start(args, format); - vfprintf(stderr, format, args); - va_end(args); -} - |
