blob: 6d02051c27a2709b9c63c52066b34d870b643a34 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
/*
 * The plain page, which hands over a repository's own bytes rather than
 * rendering a view of them. A file is written out whole under a content type
 * guessed from its name, though a repository that has not enabled html serving
 * keeps only the types a browser will not act on. A directory, or a request
 * carrying no path, is answered with a bare document of links to the entries
 * below it rather than with one of cgit's themed pages.
 */

#define USE_THE_REPOSITORY_VARIABLE

#include "cgit.h"
#include "html.h"
#include "shared.h"
#include "ui-plain.h"
#include "ui-shared.h"

// A listing is opened by the entry that matched and closed only once the walk
// is over, so the end of the page has to tell the three cases apart.
enum response {
	RESPONSE_NONE,
	RESPONSE_BLOB,
	RESPONSE_LISTING
};

struct walk_tree_context {
	// Length of the directory part of the requested path, slash included,
	// and -1 when no path was requested so that no base length can equal
	// it.
	int dir_len;
	enum response response;
};

/*
 * Everything below text/ and application/ can carry markup or script that a
 * browser would run against the site, so only PDF is let back through.
 */
static int is_unsafe_type(const char *mimetype)
{
	return (starts_with(mimetype, "text/") ||
		starts_with(mimetype, "application/")) &&
		strcmp(mimetype, "application/pdf");
}

/*
 * Writes the response for the object, error pages included, so the walk
 * does not go on to report the path as missing.
 */
static void print_object(const struct object_id *oid, const char *path)
{
	enum object_type type;
	char *buf, *mimetype;
	unsigned long size;

	type = odb_read_object_info(the_repository->objects, oid, &size);
	if (type == OBJ_BAD) {
		cgit_print_error_page(404, "Not Found", "Not found");
		return;
	}

	// The limit counts kilobytes and is checked before the read, so a huge
	// blob is kept out of memory rather than noticed once it is there.
	if (ctx.cfg.max_blob_size && size / 1024 > (unsigned long)ctx.cfg.max_blob_size) {
		cgit_print_error_page(413, "Content Too Large", "Blob size (%lu KB) exceeds the limit (%d KB)",
			size / 1024, ctx.cfg.max_blob_size);
		return;
	}

	buf = odb_read_object(the_repository->objects, oid, &type, &size);
	if (!buf) {
		cgit_print_error_page(404, "Not Found", "Not found");
		return;
	}

	mimetype = cgit_get_mimetype_for_filename(path);
	ctx.page.mimetype = mimetype;

	if (!ctx.repo->enable_html_serving) {
		ctx.page.untrusted = 1;
		if (mimetype && is_unsafe_type(mimetype))
			ctx.page.mimetype = NULL;
	}

	if (!ctx.page.mimetype)
		ctx.page.mimetype = buffer_is_binary(buf, size) ? "application/octet-stream" : "text/plain";
	ctx.page.filename = path;
	ctx.page.size = size;
	cgit_print_http_headers();
	html_raw(buf, size);
	free(mimetype);
	free(buf);
}

static char *build_path(const char *base, int baselen, const char *path)
{
	if (path[0])
		return cgit_fmtalloc("%.*s%s/", baselen, base, path);
	else
		return cgit_fmtalloc("%.*s/", baselen, base);
}

static void print_dir(const char *base, int baselen, const char *path)
{
	char *fullpath;
	const char *leading_slash;
	size_t len;

	fullpath = build_path(base, baselen, path);
	leading_slash = (fullpath[0] == '/' ? "" : "/");
	cgit_print_http_headers();
	// A full document of its own, and without the doctype and charset
	// the browser would parse it in quirks mode.
	html("<!DOCTYPE html>\n<html lang='en'>\n<head>\n");
	html("<meta charset='UTF-8'>\n");
	htmlf("<title>%s", leading_slash);
	html_txt(fullpath);
	htmlf("</title>\n</head>\n<body>\n<h2>%s", leading_slash);
	html_txt(fullpath);
	html("</h2>\n<ul>\n");
	len = strlen(fullpath);
	if (len > 1) {
		char *slash;

		// Nothing left to drop means the parent is the root, which
		// cgit_plain_link is asked for with a null path.
		fullpath[len - 1] = 0;
		slash = strrchr(fullpath, '/');
		if (slash) {
			*(slash + 1) = 0;
		} else {
			free(fullpath);
			fullpath = NULL;
		}
		html("<li>");
		cgit_plain_link("../", NULL, NULL, ctx.qry.head, ctx.qry.oid, fullpath);
		html("</li>\n");
	}
	free(fullpath);
}

static void print_dir_entry(const struct object_id *oid, const char *base,
	int baselen, const char *path, unsigned mode)
{
	char *fullpath;

	fullpath = build_path(base, baselen, path);
	if (!S_ISDIR(mode) && !S_ISGITLINK(mode))
		fullpath[strlen(fullpath) - 1] = 0;
	html("<li>");
	if (S_ISGITLINK(mode))
		cgit_submodule_link(NULL, fullpath, oid_to_hex(oid));
	else
		cgit_plain_link(path, NULL, NULL, ctx.qry.head, ctx.qry.oid, fullpath);
	html("</li>\n");
	free(fullpath);
}

static void print_dir_tail(void)
{
	html("</ul>\n</body>\n</html>\n");
}

/*
 * read_tree reads the return value as a direction rather than a status, so
 * READ_TREE_RECURSIVE means step into this entry and zero means step over it.
 */
static int walk_tree(const struct object_id *oid, struct strbuf *base,
	const char *pathname, unsigned mode, void *context)
{
	struct walk_tree_context *walk = context;

	if (walk->dir_len >= 0 && base->len == (size_t)walk->dir_len) {
		if (S_ISREG(mode) || S_ISLNK(mode)) {
			print_object(oid, pathname);
			walk->response = RESPONSE_BLOB;
		} else if (S_ISDIR(mode)) {
			print_dir(base->buf, base->len, pathname);
			walk->response = RESPONSE_LISTING;
			return READ_TREE_RECURSIVE;
		}
	} else if (base->len < INT_MAX && (int)base->len > walk->dir_len) {
		print_dir_entry(oid, base->buf, base->len, pathname, mode);
		walk->response = RESPONSE_LISTING;
	} else if (S_ISDIR(mode)) {
		return READ_TREE_RECURSIVE;
	}

	return 0;
}

static int dir_prefix_len(const char *path)
{
	const char *slash = strrchr(path, '/');

	if (slash)
		return slash - path + 1;
	return 0;
}

void cgit_print_plain(void)
{
	const char *rev = ctx.qry.oid;
	struct object_id oid;
	struct commit *commit;
	int path_len = ctx.qry.path ? strlen(ctx.qry.path) : 0;
	// nowildcard_len matches len so git treats the path as literal. As a
	// glob, every entry a pattern like * matches would be answered with
	// its own HTTP headers inside the body of the first.
	struct pathspec_item path_items = {
		.match = ctx.qry.path,
		.len = path_len,
		.nowildcard_len = path_len
	};
	struct pathspec paths = {
		.nr = 1,
		.items = &path_items
	};
	struct walk_tree_context walk = {
		.response = RESPONSE_NONE
	};

	if (!rev)
		rev = ctx.qry.head;

	if (repo_get_oid(the_repository, rev, &oid)) {
		cgit_print_error_page(404, "Not Found", "Not found");
		return;
	}
	commit = lookup_commit_reference(the_repository, &oid);
	if (!commit || repo_parse_commit(the_repository, commit)) {
		cgit_print_error_page(404, "Not Found", "Not found");
		return;
	}
	if (!path_items.match) {
		// The walk is never handed an entry for the top of the tree
		// itself, so the listing it would have opened is opened here.
		path_items.match = "";
		walk.dir_len = -1;
		print_dir("", 0, "");
		walk.response = RESPONSE_LISTING;
	} else {
		walk.dir_len = dir_prefix_len(path_items.match);
	}
	read_tree(the_repository, repo_get_commit_tree(the_repository, commit), &paths, walk_tree, &walk);
	if (walk.response == RESPONSE_NONE)
		cgit_print_error_page(404, "Not Found", "Not found");
	else if (walk.response == RESPONSE_LISTING)
		print_dir_tail();
}