diff options
context:
space:
mode:
Diffstat (limited to '')
-rw-r--r--source/cache.c462
1 file changed, 245 insertions, 217 deletions
diff --git a/source/cache.c b/source/cache.c
index c6d0427..d6e450a 100644
--- a/source/cache.c
+++ b/source/cache.c
@@ -1,33 +1,49 @@
-/* cache.c: cache management
- *
- * Copyright (C) 2006-2014 cgit Development Team <cgit@lists.zx2c4.com>
- *
- * Licensed under GNU General Public License v2
- * (see LICENSE.txt for full license text)
- *
- *
- * The cache is just a directory structure where each file is a cache slot,
- * and each filename is based on the hash of some key (e.g. the cgit url).
- * Each file contains the full key followed by the cached content for that
- * key.
- *
+/*
+ * The cache that lets a repeated request be answered from disk instead of
+ * being rendered again. A slot is one file named after the hash of the
+ * request key, holding that key and then the page it rendered to, and a lock
+ * file beside it is where a replacement page is written before being renamed
+ * over the slot. Only the process holding that lock rebuilds a slot, so a
+ * request arriving while a stale slot is being rebuilt is served the stale
+ * page, and a request with no usable slot at all renders straight to the
+ * client without caching anything.
*/
-#include "cgit.h"
#include "cache.h"
+#include "cgit.h"
#include "html.h"
+#include "shared.h"
#ifdef HAVE_LINUX_SENDFILE
#include <sys/sendfile.h>
#endif
+// One read of a slot file. The stored key has to be recognised out of a
+// single such read, so this also bounds how long a cacheable key can be, see
+// key_fits_slot.
#define CACHE_BUFSIZE (1024 * 4)
-/* Crude implementation of 32-bit FNV-1 hash algorithm,
- * see http://www.isthe.com/chongo/tech/comp/fnv/ for details
- * about the magic numbers.
- */
+
+// A slot is named by this many hex digits of the key hash, which is also how
+// cache_ls tells slots from the lock files sitting beside them.
+#define SLOT_NAME_LEN 8
+
+// The 32 bit FNV-1 offset basis and prime.
#define FNV_OFFSET 0x811c9dc5
#define FNV_PRIME 0x01000193
+/*
+ * Cache trouble goes to stderr, which under CGI is the web server's error
+ * log, so that it cannot land in the middle of the page being written to
+ * stdout.
+ */
+__attribute__((format (printf,1,2)))
+static void log_error(const char *format, ...)
+{
+ va_list args;
+ va_start(args, format);
+ vfprintf(stderr, format, args);
+ va_end(args);
+}
+
struct cache_slot {
const char *key;
size_t keylen;
@@ -35,56 +51,56 @@ struct cache_slot {
cache_fill_fn fn;
int cache_fd;
int lock_fd;
- int stdout_fd;
- const char *cache_name;
- const char *lock_name;
- int match;
- struct stat cache_st;
- int bufsize;
+ int saved_stdout;
+ const char *path;
+ const char *lock_path;
+ int key_matches;
+ // The slot as it was when it was opened, or the lock file once
+ // fill_slot has written a page into it.
+ struct stat st;
+ // How much of the slot was read into buf, not the size of buf.
+ int buflen;
char buf[CACHE_BUFSIZE];
};
-/* Open an existing cache slot and fill the cache buffer with
- * (part of) the content of the cache file. Return 0 on success
- * and errno otherwise.
- */
static int open_slot(struct cache_slot *slot)
{
- char *bufz;
- ssize_t bufkeylen = -1;
+ char *nul;
+ ssize_t keylen = -1;
- slot->cache_fd = open(slot->cache_name, O_RDONLY);
+ slot->cache_fd = open(slot->path, O_RDONLY);
if (slot->cache_fd == -1)
return errno;
- if (fstat(slot->cache_fd, &slot->cache_st))
+ if (fstat(slot->cache_fd, &slot->st))
return errno;
- slot->bufsize = xread(slot->cache_fd, slot->buf, sizeof(slot->buf));
- if (slot->bufsize < 0)
+ slot->buflen = xread(slot->cache_fd, slot->buf, sizeof(slot->buf));
+ if (slot->buflen < 0)
return errno;
- bufz = memchr(slot->buf, 0, slot->bufsize);
- if (bufz)
- bufkeylen = bufz - slot->buf;
+ nul = memchr(slot->buf, 0, slot->buflen);
+ if (nul)
+ keylen = nul - slot->buf;
if (slot->key)
- slot->match = bufkeylen >= 0 && (size_t)bufkeylen == slot->keylen &&
- !memcmp(slot->key, slot->buf, bufkeylen + 1);
+ slot->key_matches = keylen >= 0 &&
+ (size_t)keylen == slot->keylen &&
+ !memcmp(slot->key, slot->buf, keylen + 1);
return 0;
}
-/* A key longer than the buffer above can never be read back, so a slot keyed
- * on one would never match and every such request would regenerate its page
- * while still writing a slot nothing can use. Those requests skip the cache
- * instead. */
+/*
+ * A key longer than the buffer above can never be read back by open_slot, so
+ * a slot keyed on one would never match and every such request would
+ * regenerate its page while still writing a slot nothing can use.
+ */
static int key_fits_slot(const char *key)
{
return strlen(key) + 1 <= CACHE_BUFSIZE;
}
-/* Close the active cache slot */
static int close_slot(struct cache_slot *slot)
{
int err = 0;
@@ -97,7 +113,6 @@ static int close_slot(struct cache_slot *slot)
return err;
}
-/* Print the content of the active cache slot (but skip the key). */
static int print_slot(struct cache_slot *slot)
{
off_t off;
@@ -108,7 +123,7 @@ static int print_slot(struct cache_slot *slot)
off = slot->keylen + 1;
#ifdef HAVE_LINUX_SENDFILE
- size = slot->cache_st.st_size;
+ size = slot->st.st_size;
do {
ssize_t ret;
@@ -116,7 +131,10 @@ static int print_slot(struct cache_slot *slot)
if (ret < 0) {
if (errno == EAGAIN || errno == EINTR)
continue;
- /* Fall back to read/write on EINVAL or ENOSYS */
+ // EINVAL and ENOSYS mean this kernel or this pair of
+ // descriptors cannot do sendfile at all, so fall back
+ // to the read and write loop rather than fail the
+ // request.
if (errno == EINVAL || errno == ENOSYS)
break;
return errno;
@@ -141,30 +159,41 @@ static int print_slot(struct cache_slot *slot)
} while (1);
}
-/* Check if the slot has expired */
+static int serve_slot(struct cache_slot *slot)
+{
+ int err;
+
+ err = print_slot(slot);
+ if (err)
+ log_error("[cgit] error printing cache %s: %s (%d)\n",
+ slot->path,
+ strerror(err),
+ err);
+ return err;
+}
+
static int is_expired(struct cache_slot *slot)
{
if (slot->ttl < 0)
return 0;
- else
- return slot->cache_st.st_mtime + slot->ttl * 60 < time(NULL);
+ return slot->st.st_mtime + slot->ttl * SECONDS_PER_MINUTE < time(NULL);
}
-/* Check if the slot has been modified since we opened it.
- * NB: If stat() fails, we pretend the file is modified.
+/*
+ * A stat that fails counts as modified, so that the caller leaves alone a file
+ * it was unable to look at.
*/
static int is_modified(struct cache_slot *slot)
{
- struct stat st;
+ struct stat current;
- if (stat(slot->cache_name, &st))
+ if (stat(slot->path, &current))
return 1;
- return (st.st_ino != slot->cache_st.st_ino ||
- st.st_mtime != slot->cache_st.st_mtime ||
- st.st_size != slot->cache_st.st_size);
+ return (current.st_ino != slot->st.st_ino ||
+ current.st_mtime != slot->st.st_mtime ||
+ current.st_size != slot->st.st_size);
}
-/* Close an open lockfile */
static int close_lock(struct cache_slot *slot)
{
int err = 0;
@@ -177,9 +206,11 @@ static int close_lock(struct cache_slot *slot)
return err;
}
-/* Create a lockfile used to store the generated content for a cache
- * slot, and write the slot key + \0 into it.
- * Returns 0 on success and errno otherwise.
+/*
+ * The lock file becomes the slot once it is renamed, so it has to open with
+ * the key the same way a slot does. The lock is taken without blocking,
+ * because failing to get it is how a second process learns that this slot is
+ * already being rebuilt, so it returns an errno instead of waiting.
*/
static int lock_slot(struct cache_slot *slot)
{
@@ -190,7 +221,7 @@ static int lock_slot(struct cache_slot *slot)
.l_len = 0,
};
- slot->lock_fd = open(slot->lock_name, O_RDWR | O_CREAT,
+ slot->lock_fd = open(slot->lock_path, O_RDWR | O_CREAT,
S_IRUSR | S_IWUSR);
if (slot->lock_fd == -1)
return errno;
@@ -200,6 +231,8 @@ static int lock_slot(struct cache_slot *slot)
slot->lock_fd = -1;
return saved_errno;
}
+ // A run that died before its rename leaves the lock file behind, so
+ // start from empty now that nobody else can be writing it.
if (ftruncate(slot->lock_fd, 0) < 0)
return errno;
if (xwrite(slot->lock_fd, slot->key, slot->keylen + 1) < 0)
@@ -207,24 +240,19 @@ static int lock_slot(struct cache_slot *slot)
return 0;
}
-/* Release the current lockfile. If `replace_old_slot` is set the
- * lockfile replaces the old cache slot, otherwise the lockfile is
- * just deleted.
- */
static int unlock_slot(struct cache_slot *slot, int replace_old_slot)
{
int err;
if (replace_old_slot)
- err = rename(slot->lock_name, slot->cache_name);
+ err = rename(slot->lock_path, slot->path);
else
- err = unlink(slot->lock_name);
+ err = unlink(slot->lock_path);
- /* Restore stdout and close the temporary FD. */
- if (slot->stdout_fd >= 0) {
- dup2(slot->stdout_fd, STDOUT_FILENO);
- close(slot->stdout_fd);
- slot->stdout_fd = -1;
+ if (slot->saved_stdout >= 0) {
+ dup2(slot->saved_stdout, STDOUT_FILENO);
+ close(slot->saved_stdout);
+ slot->saved_stdout = -1;
}
if (err)
@@ -233,48 +261,84 @@ static int unlock_slot(struct cache_slot *slot, int replace_old_slot)
return 0;
}
-/* Generate the content for the current cache slot by redirecting
- * stdout to the lock-fd and invoking the callback function
+// Only one slot is ever being filled at a time, so a single pointer is enough
+// for cache_abandon_fill to find its way back to the client.
+static struct cache_slot *slot_being_filled;
+
+void cache_abandon_fill(void)
+{
+ struct cache_slot *slot = slot_being_filled;
+
+ if (!slot)
+ return;
+ slot_being_filled = NULL;
+
+ // Emptied while stdout still points at the lock file, so the half
+ // rendered page goes into the file about to be removed rather than
+ // reaching the client ahead of whatever is written next.
+ html_flush();
+
+ if (slot->saved_stdout >= 0) {
+ dup2(slot->saved_stdout, STDOUT_FILENO);
+ close(slot->saved_stdout);
+ slot->saved_stdout = -1;
+ }
+ unlink(slot->lock_path);
+}
+
+/*
+ * Renders with stdout pointed at the lock file, and on success or failure
+ * alike it is unlock_slot that gives stdout back.
*/
static int fill_slot(struct cache_slot *slot)
{
- /* Preserve stdout */
- slot->stdout_fd = dup(STDOUT_FILENO);
- if (slot->stdout_fd == -1)
+ slot->saved_stdout = dup(STDOUT_FILENO);
+ if (slot->saved_stdout == -1)
return errno;
- /* Redirect stdout to lockfile */
if (dup2(slot->lock_fd, STDOUT_FILENO) == -1)
return errno;
- /* Generate cache content */
+ slot_being_filled = slot;
slot->fn();
+ slot_being_filled = NULL;
- /* Make sure any buffered data is flushed to the file */
+ // The page is sitting in html.c's buffer and then in stdio's, and all
+ // of it has to reach the lock file before that file is renamed into
+ // place.
html_flush();
if (fflush(stdout))
return errno;
- /* update stat info */
- if (fstat(slot->lock_fd, &slot->cache_st))
+ // print_slot takes the length of what it copies from here, and what
+ // it copies after a fill is the lock file rather than the old slot.
+ if (fstat(slot->lock_fd, &slot->st))
return errno;
return 0;
}
-unsigned long cache_hash_str(const char *str)
+/*
+ * Giving up is always a valid outcome, because the caller still has the
+ * expired copy open and can serve that.
+ */
+static void refresh_slot(struct cache_slot *slot)
{
- unsigned long h = FNV_OFFSET;
- unsigned char *s = (unsigned char *)str;
-
- if (!s)
- return h;
+ if (lock_slot(slot))
+ return;
- while (*s) {
- h *= FNV_PRIME;
- h ^= *s++;
+ // If another process replaced the slot between open_slot and
+ // lock_slot, the copy already open is served rather than the newer
+ // one, which would mean opening that file and comparing the key in it,
+ // not worth a second descriptor and read on every expiry.
+ if (is_modified(slot) || fill_slot(slot)) {
+ unlock_slot(slot, 0);
+ close_lock(slot);
+ } else {
+ close_slot(slot);
+ unlock_slot(slot, 1);
+ slot->cache_fd = slot->lock_fd;
}
- return h;
}
static int process_slot(struct cache_slot *slot)
@@ -282,216 +346,180 @@ static int process_slot(struct cache_slot *slot)
int err;
err = open_slot(slot);
- if (!err && slot->match) {
- if (is_expired(slot)) {
- if (!lock_slot(slot)) {
- /* If the cachefile has been replaced between
- * `open_slot` and `lock_slot`, we'll just
- * serve the stale content from the original
- * cachefile. This way we avoid pruning the
- * newly generated slot. The same code-path
- * is chosen if fill_slot() fails for some
- * reason.
- *
- * TODO? check if the new slot contains the
- * same key as the old one, since we would
- * prefer to serve the newest content.
- * This will require us to open yet another
- * file-descriptor and read and compare the
- * key from the new file, so for now we're
- * lazy and just ignore the new file.
- */
- if (is_modified(slot) || fill_slot(slot)) {
- unlock_slot(slot, 0);
- close_lock(slot);
- } else {
- close_slot(slot);
- unlock_slot(slot, 1);
- slot->cache_fd = slot->lock_fd;
- }
- }
- }
- if ((err = print_slot(slot)) != 0) {
- cache_log("[cgit] error printing cache %s: %s (%d)\n",
- slot->cache_name,
- strerror(err),
- err);
- }
+ if (!err && slot->key_matches) {
+ if (is_expired(slot))
+ refresh_slot(slot);
+ err = serve_slot(slot);
close_slot(slot);
return err;
}
- /* If the cache slot does not exist (or its key doesn't match the
- * current key), lets try to create a new cache slot for this
- * request. If this fails (for whatever reason), lets just generate
- * the content without caching it and fool the caller to believe
- * everything worked out (but print a warning on stdout).
- */
-
+ // If any part of creating a slot fails the page is still rendered
+ // straight to the client and the caller is told the request succeeded,
+ // because it did.
close_slot(slot);
if ((err = lock_slot(slot)) != 0) {
- cache_log("[cgit] Unable to lock slot %s: %s (%d)\n",
- slot->lock_name, strerror(err), err);
+ log_error("[cgit] Unable to lock slot %s: %s (%d)\n",
+ slot->lock_path, strerror(err), err);
slot->fn();
return 0;
}
if ((err = fill_slot(slot)) != 0) {
- cache_log("[cgit] Unable to fill slot %s: %s (%d)\n",
- slot->lock_name, strerror(err), err);
+ log_error("[cgit] Unable to fill slot %s: %s (%d)\n",
+ slot->lock_path, strerror(err), err);
unlock_slot(slot, 0);
close_lock(slot);
slot->fn();
return 0;
}
- // We've got a valid cache slot in the lock file, which
- // is about to replace the old cache slot. But if we
- // release the lockfile and then try to open the new cache
- // slot, we might get a race condition with a concurrent
- // writer for the same cache slot (with a different key).
- // Lets avoid such a race by just printing the content of
- // the lock file.
+
+ // Opening the slot by name after the rename could land on a file a
+ // concurrent writer put there for a different key, so what gets
+ // printed is the descriptor still open on the lock file.
slot->cache_fd = slot->lock_fd;
unlock_slot(slot, 1);
- if ((err = print_slot(slot)) != 0) {
- cache_log("[cgit] error printing cache %s: %s (%d)\n",
- slot->cache_name,
- strerror(err),
- err);
- }
+ err = serve_slot(slot);
close_slot(slot);
return err;
}
-/* Print cached content to stdout, generate the content if necessary. */
+// The result lives in a static buffer and is only good until the next call.
+static char *format_time(const char *format, time_t when)
+{
+ static char buf[64];
+ struct tm tm;
+
+ if (!when)
+ return NULL;
+ gmtime_r(&when, &tm);
+ strftime(buf, sizeof(buf) - 1, format, &tm);
+ return buf;
+}
+
+/*
+ * The accumulator is an unsigned long rather than a fixed 32 bit type, so on a
+ * 64 bit host this is not the published FNV-1 value. All that decides is which
+ * slot a key lands in, and nothing outside a single build has to agree on the
+ * answer.
+ */
+unsigned long cache_hash_str(const char *str)
+{
+ unsigned long h = FNV_OFFSET;
+ unsigned char *s = (unsigned char *)str;
+
+ if (!s)
+ return h;
+
+ while (*s) {
+ h *= FNV_PRIME;
+ h ^= *s++;
+ }
+ return h;
+}
+
int cache_process(int size, const char *path, const char *key, int ttl,
cache_fill_fn fn)
{
unsigned long hash;
int i;
- struct strbuf filename = STRBUF_INIT;
- struct strbuf lockname = STRBUF_INIT;
+ struct strbuf slot_path = STRBUF_INIT;
+ struct strbuf lock_path = STRBUF_INIT;
struct cache_slot slot;
int result;
- /* If the cache is disabled, just generate the content */
if (size <= 0 || ttl == 0) {
fn();
return 0;
}
- /* Verify input, calculate filenames */
if (!path) {
- cache_log("[cgit] Cache path not specified, caching is disabled\n");
+ log_error("[cgit] Cache path not specified, caching is disabled\n");
fn();
return 0;
}
if (!key)
key = "";
if (!key_fits_slot(key)) {
- cache_log("[cgit] Cache key too long for a slot, caching is "
+ log_error("[cgit] Cache key too long for a slot, caching is "
"disabled for this request\n");
fn();
return 0;
}
hash = cache_hash_str(key) % size;
- strbuf_addstr(&filename, path);
- strbuf_ensure_end(&filename, '/');
- for (i = 0; i < 8; i++) {
- strbuf_addf(&filename, "%x", (unsigned char)(hash & 0xf));
+ strbuf_addstr(&slot_path, path);
+ strbuf_ensure_end(&slot_path, '/');
+ for (i = 0; i < SLOT_NAME_LEN; i++) {
+ strbuf_addf(&slot_path, "%x", (unsigned char)(hash & 0xf));
hash >>= 4;
}
- strbuf_addbuf(&lockname, &filename);
- strbuf_addstr(&lockname, ".lock");
+ strbuf_addbuf(&lock_path, &slot_path);
+ strbuf_addstr(&lock_path, ".lock");
slot.fn = fn;
slot.ttl = ttl;
- slot.stdout_fd = -1;
- slot.cache_name = filename.buf;
- slot.lock_name = lockname.buf;
+ slot.saved_stdout = -1;
+ slot.path = slot_path.buf;
+ slot.lock_path = lock_path.buf;
slot.key = key;
slot.keylen = strlen(key);
result = process_slot(&slot);
- strbuf_release(&filename);
- strbuf_release(&lockname);
+ strbuf_release(&slot_path);
+ strbuf_release(&lock_path);
return result;
}
-/* Return a strftime formatted date/time
- * NB: the result from this function is to shared memory
- */
-static char *sprintftime(const char *format, time_t time)
-{
- static char buf[64];
- struct tm tm;
-
- if (!time)
- return NULL;
- gmtime_r(&time, &tm);
- strftime(buf, sizeof(buf)-1, format, &tm);
- return buf;
-}
-
int cache_ls(const char *path)
{
DIR *dir;
struct dirent *ent;
int err = 0;
+ // A NULL key leaves open_slot with nothing to compare against, so
+ // every slot it opens is simply read.
struct cache_slot slot = { NULL };
- struct strbuf fullname = STRBUF_INIT;
+ struct strbuf slot_path = STRBUF_INIT;
size_t prefixlen;
char *nul;
int keylen;
if (!path) {
- cache_log("[cgit] cache path not specified\n");
+ log_error("[cgit] cache path not specified\n");
return -1;
}
dir = opendir(path);
if (!dir) {
err = errno;
- cache_log("[cgit] unable to open path %s: %s (%d)\n",
+ log_error("[cgit] unable to open path %s: %s (%d)\n",
path, strerror(err), err);
return err;
}
- strbuf_addstr(&fullname, path);
- strbuf_ensure_end(&fullname, '/');
- prefixlen = fullname.len;
+ strbuf_addstr(&slot_path, path);
+ strbuf_ensure_end(&slot_path, '/');
+ prefixlen = slot_path.len;
while ((ent = readdir(dir)) != NULL) {
- if (strlen(ent->d_name) != 8)
+ if (strlen(ent->d_name) != SLOT_NAME_LEN)
continue;
- strbuf_setlen(&fullname, prefixlen);
- strbuf_addstr(&fullname, ent->d_name);
- slot.cache_name = fullname.buf;
+ strbuf_setlen(&slot_path, prefixlen);
+ strbuf_addstr(&slot_path, ent->d_name);
+ slot.path = slot_path.buf;
if ((err = open_slot(&slot)) != 0) {
- cache_log("[cgit] unable to open path %s: %s (%d)\n",
- fullname.buf, strerror(err), err);
+ log_error("[cgit] unable to open path %s: %s (%d)\n",
+ slot_path.buf, strerror(err), err);
continue;
}
- // The stored key is NUL-terminated within the buffer, but a
- // truncated or corrupt slot may not be. Bound the print to
- // what was read so %s cannot run off the end.
- nul = memchr(slot.buf, 0, slot.bufsize);
- keylen = nul ? (int)(nul - slot.buf) : slot.bufsize;
+ // A truncated or corrupt slot may hold no NUL, so the print is
+ // bounded by what was read and cannot run off the end.
+ nul = memchr(slot.buf, 0, slot.buflen);
+ keylen = nul ? (int)(nul - slot.buf) : slot.buflen;
htmlf("%s %s %10"PRIuMAX" %.*s\n",
- fullname.buf,
- sprintftime("%Y-%m-%d %H:%M:%S",
- slot.cache_st.st_mtime),
- (uintmax_t)slot.cache_st.st_size,
+ slot_path.buf,
+ format_time("%Y-%m-%d %H:%M:%S",
+ slot.st.st_mtime),
+ (uintmax_t)slot.st.st_size,
keylen, slot.buf);
close_slot(&slot);
}
closedir(dir);
- strbuf_release(&fullname);
+ strbuf_release(&slot_path);
return 0;
}
-
-/* Print a message to stdout */
-void cache_log(const char *format, ...)
-{
- va_list args;
- va_start(args, format);
- vfprintf(stderr, format, args);
- va_end(args);
-}
-