/* * Reads a blob out of the object database and writes its bytes to the client. * The blob is named either directly by object id or by a path walked out of a * commit's tree. The blob page serves those bytes as the whole response, under * headers that stop a browser treating repository content as markup, while the * summary page instead drops a readme's contents into a page it is already * building. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" #include "html.h" #include "ui-blob.h" #include "ui-shared.h" struct walk_tree_context { const char *match_path; struct object_id *matched_oid; unsigned int found_path:1; unsigned int file_only:1; }; /* * read_tree reads the return value as a direction rather than a status, so * READ_TREE_RECURSIVE means step into this entry and zero means step over it. */ static int walk_tree(const struct object_id *oid, struct strbuf *base, const char *pathname, unsigned mode, void *context) { struct walk_tree_context *walk = context; if (walk->file_only && !S_ISREG(mode)) return READ_TREE_RECURSIVE; if (strncmp(base->buf, walk->match_path, base->len) || strcmp(walk->match_path + base->len, pathname)) return READ_TREE_RECURSIVE; oidcpy(walk->matched_oid, oid); walk->found_path = 1; return 0; } /* * oid comes in naming a commit and goes out naming the blob found at path, * which both callers rely on. The path has to be writable because a pathspec * item does not hold a const string. */ static int find_path_oid(struct object_id *oid, char *path, int file_only) { struct commit *commit = lookup_commit_reference(the_repository, oid); // nowildcard_len matching len makes git treat the path as literal // rather than as a glob. struct pathspec_item item = { .match = path, .len = strlen(path), .nowildcard_len = strlen(path) }; struct pathspec paths = { .nr = 1, .items = &item }; struct walk_tree_context walk = { .match_path = path, .matched_oid = oid, .found_path = 0, .file_only = file_only }; read_tree(the_repository, repo_get_commit_tree(the_repository, commit), &paths, walk_tree, &walk); return walk.found_path; } /* * Callers ask before reading the object, because the point of the limit is to * keep a huge blob out of memory rather than to notice it once it is already * there. */ static int over_size_limit(unsigned long size) { return ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size; } int cgit_ref_path_exists(const char *path, const char *ref, int file_only) { struct object_id oid; unsigned long size; char *path_copy = xstrdup(path); int found = 0; if (repo_get_oid(the_repository, ref, &oid)) goto done; if (odb_read_object_info(the_repository->objects, &oid, &size) != OBJ_COMMIT) goto done; found = find_path_oid(&oid, path_copy, file_only); done: free(path_copy); return found; } int cgit_print_file(char *path, const char *head, int file_only, int html_escape) { struct object_id oid; enum object_type type; unsigned long size; char *buf; if (repo_get_oid(the_repository, head, &oid)) return -1; type = odb_read_object_info(the_repository->objects, &oid, &size); if (type == OBJ_COMMIT) { if (!find_path_oid(&oid, path, file_only)) return -1; type = odb_read_object_info(the_repository->objects, &oid, &size); } if (type == OBJ_BAD) return -1; if (over_size_limit(size)) return -1; buf = odb_read_object(the_repository->objects, &oid, &type, &size); if (!buf) return -1; // html_txt wants a terminated string, and git leaves a spare byte // past every object it reads, so this write stays in the allocation. buf[size] = '\0'; if (html_escape) html_txt(buf); else html_raw(buf, size); free(buf); return 0; } void cgit_print_blob(const char *hex, char *path, const char *head, int file_only) { struct object_id oid; enum object_type type; unsigned long size; char *buf; if (hex) { if (get_oid_hex(hex, &oid)) { cgit_print_error_page(400, "Bad request", "Bad hex value: %s", hex); return; } } else { if (repo_get_oid(the_repository, head, &oid)) { cgit_print_error_page(404, "Not found", "Bad ref: %s", head); return; } } type = odb_read_object_info(the_repository->objects, &oid, &size); if (!hex && type == OBJ_COMMIT && path) { if (!find_path_oid(&oid, path, file_only)) { cgit_print_error_page(404, "Not found", "Path not found: %s", path); return; } type = odb_read_object_info(the_repository->objects, &oid, &size); } if (type == OBJ_BAD) { cgit_print_error_page(404, "Not found", "Bad object name: %s", hex ? hex : path); return; } if (over_size_limit(size)) { cgit_print_error_page(413, "Too large", "Object size (%luKB) exceeds limit (%dKB)", size / 1024, ctx.cfg.max_blob_size); return; } buf = odb_read_object(the_repository->objects, &oid, &type, &size); if (!buf) { cgit_print_error_page(500, "Internal server error", "Error reading object %s", hex ? hex : path); return; } buf[size] = '\0'; if (buffer_is_binary(buf, size)) ctx.page.mimetype = "application/octet-stream"; else ctx.page.mimetype = "text/plain"; ctx.page.filename = path; // The bytes are whatever the repository holds, so the browser is told // not to guess a type of its own from them and not to load anything // they reference. Both must go out before cgit_print_http_headers, // which closes the header block. html("X-Content-Type-Options: nosniff\n"); html("Content-Security-Policy: default-src 'none'\n"); cgit_print_http_headers(); html_raw(buf, size); free(buf); }