/* * Reads a blob out of the object database and writes its bytes to the client. * The blob is named either directly by object id or by a path walked out of a * commit's tree. The blob page serves those bytes as the whole response, under * headers that stop a browser treating repository content as markup, while the * summary page instead drops a readme's contents into a page it is already * building. */ #define USE_THE_REPOSITORY_VARIABLE #include "cgit.h" #include "html.h" #include "ui-blob.h" #include "ui-shared.h" struct walk_tree_context { const char *match_path; struct object_id *matched_oid; unsigned int found_path:1; unsigned int file_only:1; }; /* * An annotated tag names a commit through its tag object, and a path is looked * up in the commit's tree, not in the tag. */ static void peel_to_commit(struct object_id *oid) { struct commit *commit = lookup_commit_reference_gently(the_repository, oid, 1); if (commit) oidcpy(oid, &commit->object.oid); } /* * read_tree reads the return value as a direction rather than a status, so * READ_TREE_RECURSIVE means step into this entry and zero means step over it. */ static int walk_tree(const struct object_id *oid, struct strbuf *base, const char *pathname, unsigned mode, void *context) { struct walk_tree_context *walk = context; // Stepping into a submodule entry would have git look its commit up // in this repository, which does not hold it, and die. if (walk->file_only && !S_ISREG(mode)) return S_ISDIR(mode) ? READ_TREE_RECURSIVE : 0; if ( strncmp(base->buf, walk->match_path, base->len) || strcmp(walk->match_path + base->len, pathname) ) return READ_TREE_RECURSIVE; oidcpy(walk->matched_oid, oid); walk->found_path = 1; return 0; } /* * oid comes in naming a commit and goes out naming the blob found at path, * which both callers rely on. The path has to be writable because a pathspec * item does not hold a const string. */ static int find_path_oid(struct object_id *oid, char *path, int file_only) { struct commit *commit = lookup_commit_reference(the_repository, oid); struct tree *tree; // nowildcard_len matching len makes git treat the path as literal // rather than as a glob. struct pathspec_item item = { .match = path, .len = strlen(path), .nowildcard_len = strlen(path) }; struct pathspec paths = { .nr = 1, .items = &item }; struct walk_tree_context walk = { .match_path = path, .matched_oid = oid, .found_path = 0, .file_only = file_only }; // A commit git cannot parse, one with a broken tree line for example, // comes back null and has no tree to search. if (!commit) return 0; tree = repo_get_commit_tree(the_repository, commit); if (!tree) return 0; read_tree(the_repository, tree, &paths, walk_tree, &walk); return walk.found_path; } /* * Callers ask before reading the object, so a huge blob stays out of * memory rather than being noticed once already there. */ static int over_size_limit(unsigned long size) { return ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size; } int cgit_ref_path_exists(const char *path, const char *ref, int file_only) { struct object_id oid; unsigned long size; char *path_copy = xstrdup(path); int found = 0; if (repo_get_oid(the_repository, ref, &oid)) goto done; if (odb_read_object_info(the_repository->objects, &oid, &size) != OBJ_COMMIT) goto done; found = find_path_oid(&oid, path_copy, file_only); done: free(path_copy); return found; } int cgit_print_file(char *path, const char *head, int file_only, int html_escape) { struct object_id oid; enum object_type type; unsigned long size; char *buf; if (repo_get_oid(the_repository, head, &oid)) return -1; peel_to_commit(&oid); type = odb_read_object_info(the_repository->objects, &oid, &size); if (type == OBJ_COMMIT) { if (!find_path_oid(&oid, path, file_only)) return -1; type = odb_read_object_info(the_repository->objects, &oid, &size); } if (type == OBJ_BAD) return -1; if (over_size_limit(size)) return -1; buf = odb_read_object(the_repository->objects, &oid, &type, &size); if (!buf) return -1; // html_txt wants a terminated string, and git leaves a spare byte // past every object it reads, so this write stays in the allocation. buf[size] = '\0'; if (html_escape) { // html_txt stops at a NUL, so a blob holding one is written // segment by segment with a question mark standing in for // each NUL byte, rather than silently cut short. const char *p = buf, *end = buf + size; while (p < end) { html_txt(p); p += strlen(p); while (p < end && !*p) { html("?"); p++; } } } else { html_raw(buf, size); } free(buf); return 0; } void cgit_print_blob(const char *hex, char *path, const char *head, int file_only) { struct object_id oid; enum object_type type; unsigned long size; char *buf; if (hex) { if (get_oid_hex(hex, &oid)) { cgit_print_error_page(404, "Not Found", "Bad object id: %s", hex); return; } } else if (repo_get_oid(the_repository, head, &oid)) { cgit_print_error_page(404, "Not Found", "Bad object id: %s", head); return; } type = odb_read_object_info(the_repository->objects, &oid, &size); // A tag is peeled to its commit, so a path can be looked up in the // tree whether the id or the head named the tag. The type is read // first because peeling a blob would read it whole just to learn that // it is one. if (type == OBJ_TAG) { peel_to_commit(&oid); type = odb_read_object_info(the_repository->objects, &oid, &size); } if (type == OBJ_COMMIT && path) { if (!find_path_oid(&oid, path, file_only)) { cgit_print_error_page(404, "Not Found", "Path not found: %s", path); return; } type = odb_read_object_info(the_repository->objects, &oid, &size); } if (type == OBJ_BAD) { cgit_print_error_page(404, "Not Found", "Bad object id: %s", hex ? hex : path); return; } if (over_size_limit(size)) { cgit_print_error_page(413, "Content Too Large", "Object size (%luKB) exceeds limit (%dKB)", size / 1024, ctx.cfg.max_blob_size); return; } buf = odb_read_object(the_repository->objects, &oid, &type, &size); if (!buf) { cgit_print_error_page(500, "Internal Server Error", "Unable to read object %s", hex ? hex : path); return; } buf[size] = '\0'; if (buffer_is_binary(buf, size)) ctx.page.mimetype = "application/octet-stream"; else ctx.page.mimetype = "text/plain"; ctx.page.filename = path; ctx.page.untrusted = 1; cgit_print_http_headers(); html_raw(buf, size); free(buf); }