/* * The plain page, which hands over a repository's own bytes rather than * rendering a view of them. A file is written out whole under a content type * guessed from its name, though a repository that has not enabled html serving * keeps only the types a browser will not act on. A directory, or a request * carrying no path, is answered with a bare document of links to the entries * below it rather than with one of cgit's themed pages. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" #include "html.h" #include "shared.h" #include "ui-plain.h" #include "ui-shared.h" // A listing is opened by the entry that matched and closed only once the walk // is over, so the end of the page has to tell the three cases apart. enum response { RESPONSE_NONE, RESPONSE_BLOB, RESPONSE_LISTING }; struct walk_tree_context { // Length of the directory part of the requested path, slash included, // and -1 when no path was requested so that no base length can equal // it. int dir_len; enum response response; }; /* * Everything below text/ and application/ can carry markup or script that a * browser would run against the site, so only PDF is let back through. */ static int is_unsafe_type(const char *mimetype) { return (starts_with(mimetype, "text/") || starts_with(mimetype, "application/")) && strcmp(mimetype, "application/pdf"); } /* * Writes the response for the object, error pages included, so the walk * does not go on to report the path as missing. */ static void print_object(const struct object_id *oid, const char *path) { enum object_type type; char *buf, *mimetype; unsigned long size; type = odb_read_object_info(the_repository->objects, oid, &size); if (type == OBJ_BAD) { cgit_print_error_page(404, "Not Found", "Not found"); return; } // The limit counts kilobytes and is checked before the read, so a huge // blob is kept out of memory rather than noticed once it is there. if (ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size) { cgit_print_error_page(413, "Content Too Large", "Object size (%luKB) exceeds limit (%dKB)", size / 1024, ctx.cfg.max_blob_size); return; } buf = odb_read_object(the_repository->objects, oid, &type, &size); if (!buf) { cgit_print_error_page(404, "Not Found", "Not found"); return; } mimetype = cgit_get_mimetype_for_filename(path); ctx.page.mimetype = mimetype; if (!ctx.repo->enable_html_serving) { ctx.page.untrusted = 1; if (mimetype && is_unsafe_type(mimetype)) ctx.page.mimetype = NULL; } if (!ctx.page.mimetype) { if (buffer_is_binary(buf, size)) { ctx.page.mimetype = "application/octet-stream"; ctx.page.charset = NULL; } else { ctx.page.mimetype = "text/plain"; } } ctx.page.filename = path; ctx.page.size = size; cgit_print_http_headers(); html_raw(buf, size); free(mimetype); free(buf); } static char *build_path(const char *base, int baselen, const char *path) { if (path[0]) return cgit_fmtalloc("%.*s%s/", baselen, base, path); else return cgit_fmtalloc("%.*s/", baselen, base); } static void print_dir(const char *base, int baselen, const char *path) { char *fullpath; const char *leading_slash; size_t len; fullpath = build_path(base, baselen, path); leading_slash = (fullpath[0] == '/' ? "" : "/"); cgit_print_http_headers(); // A full document of its own, and without the doctype and charset // the browser would parse it in quirks mode. html("\n\n\n"); html("\n"); htmlf("%s", leading_slash); html_txt(fullpath); htmlf("\n\n\n

%s", leading_slash); html_txt(fullpath); html("

\n\n\n\n"); } /* * read_tree reads the return value as a direction rather than a status, so * READ_TREE_RECURSIVE means step into this entry and zero means step over it. */ static int walk_tree(const struct object_id *oid, struct strbuf *base, const char *pathname, unsigned mode, void *context) { struct walk_tree_context *walk = context; if (walk->dir_len >= 0 && base->len == (size_t)walk->dir_len) { if (S_ISREG(mode) || S_ISLNK(mode)) { print_object(oid, pathname); walk->response = RESPONSE_BLOB; } else if (S_ISDIR(mode)) { print_dir(base->buf, base->len, pathname); walk->response = RESPONSE_LISTING; return READ_TREE_RECURSIVE; } } else if (base->len < INT_MAX && (int)base->len > walk->dir_len) { print_dir_entry(oid, base->buf, base->len, pathname, mode); walk->response = RESPONSE_LISTING; } else if (S_ISDIR(mode)) { return READ_TREE_RECURSIVE; } return 0; } static int dir_prefix_len(const char *path) { const char *slash = strrchr(path, '/'); if (slash) return slash - path + 1; return 0; } void cgit_print_plain(void) { const char *rev = ctx.qry.oid; struct object_id oid; struct commit *commit; int path_len = ctx.qry.path ? strlen(ctx.qry.path) : 0; // nowildcard_len matches len so git treats the path as literal. As a // glob, every entry a pattern like * matches would be answered with // its own HTTP headers inside the body of the first. struct pathspec_item path_items = { .match = ctx.qry.path, .len = path_len, .nowildcard_len = path_len }; struct pathspec paths = { .nr = 1, .items = &path_items }; struct walk_tree_context walk = { .response = RESPONSE_NONE }; if (!rev) rev = ctx.qry.head; if (repo_get_oid(the_repository, rev, &oid)) { cgit_print_error_page(404, "Not Found", "Not found"); return; } commit = lookup_commit_reference(the_repository, &oid); if (!commit || repo_parse_commit(the_repository, commit)) { cgit_print_error_page(404, "Not Found", "Not found"); return; } if (!path_items.match) { // The walk is never handed an entry for the top of the tree // itself, so the listing it would have opened is opened here. path_items.match = ""; walk.dir_len = -1; print_dir("", 0, ""); walk.response = RESPONSE_LISTING; } else { walk.dir_len = dir_prefix_len(path_items.match); } read_tree(the_repository, repo_get_commit_tree(the_repository, commit), &paths, walk_tree, &walk); if (walk.response == RESPONSE_NONE) cgit_print_error_page(404, "Not Found", "Not found"); else if (walk.response == RESPONSE_LISTING) print_dir_tail(); }