From 39c8cd2ae5eadec313661227055fe6d07246ca4e Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Wed, 15 Jul 2026 11:12:30 -1000 Subject: Render README markdown in the browser cgit had no markdown support of its own, so a readme was rendered through an external python filter or not at all. Escaping the source and formatting it in cgit.js keeps the work in the browser like the blob highlighter, and the page stays readable as plain text without scripting. --- assets/cgit.css | 90 +++++++++++ assets/cgit.js | 161 +++++++++++++++++++ cgitrc.5.txt | 17 +- extensions/html-converters/md2html | 304 ------------------------------------ extensions/html-converters/rst2html | 2 - source/cgit.c | 3 + source/cgit.h | 1 + source/ui-blob.c | 7 +- source/ui-blob.h | 2 +- source/ui-summary.c | 48 ++++-- 10 files changed, 313 insertions(+), 322 deletions(-) delete mode 100755 extensions/html-converters/md2html delete mode 100755 extensions/html-converters/rst2html diff --git a/assets/cgit.css b/assets/cgit.css index 96192e4..10fcf66 100644 --- a/assets/cgit.css +++ b/assets/cgit.css @@ -496,6 +496,96 @@ div#cgit div#summary pre { overflow-x: auto; } +div#cgit .markdown { + line-height: 1.6; + overflow-wrap: break-word; +} + +div#cgit .markdown > :first-child { + margin-top: 0; +} + +div#cgit .markdown > :last-child { + margin-bottom: 0; +} + +div#cgit .markdown h1, +div#cgit .markdown h2, +div#cgit .markdown h3, +div#cgit .markdown h4, +div#cgit .markdown h5, +div#cgit .markdown h6 { + margin: 1.2em 0 0.5em; + font-weight: bold; + line-height: 1.25; +} + +div#cgit .markdown h1 { font-size: 1.6em; } +div#cgit .markdown h2 { + font-size: 1.35em; + border-bottom: 1px solid var(--border); + padding-bottom: 0.2em; +} +div#cgit .markdown h3 { font-size: 1.15em; } +div#cgit .markdown h4 { font-size: 1em; } +div#cgit .markdown h5, +div#cgit .markdown h6 { font-size: 0.9em; color: var(--muted); } + +div#cgit .markdown p { margin: 0.7em 0; } +div#cgit .markdown a { color: var(--link); } + +div#cgit .markdown ul, +div#cgit .markdown ol { margin: 0.7em 0; padding-left: 2em; } +div#cgit .markdown li { margin: 0.2em 0; } + +div#cgit .markdown blockquote { + margin: 0.7em 0; + padding: 0.1em 1em; + color: var(--muted); + border-left: 3px solid var(--border-mid); +} + +div#cgit .markdown hr { + border: none; + border-top: 1px solid var(--border); + margin: 1.2em 0; +} + +div#cgit .markdown code { + background: var(--surface-2); + border: 1px solid var(--border); + border-radius: 3px; + padding: 0.1em 0.35em; + font-size: 0.95em; +} + +div#cgit .markdown pre { + background: var(--surface-2); + border: 1px solid var(--border); + border-radius: 4px; + padding: 0.7em 0.9em; + overflow-x: auto; +} + +div#cgit .markdown pre code { + background: none; + border: none; + padding: 0; +} + +div#cgit .markdown table { + border-collapse: collapse; + margin: 0.7em 0; +} + +div#cgit .markdown th, +div#cgit .markdown td { + border: 1px solid var(--border); + padding: 0.35em 0.7em; +} + +div#cgit .markdown th { background: var(--surface-2); } + div#cgit table#downloads { float: right; border-collapse: collapse; diff --git a/assets/cgit.js b/assets/cgit.js index cda167e..4a0d088 100644 --- a/assets/cgit.js +++ b/assets/cgit.js @@ -123,6 +123,167 @@ document.addEventListener("DOMContentLoaded", function () { })(); +/* Built-in Markdown rendering for the about page. When no about-filter is + * configured, cgit escapes a markdown readme into a data-markdown container + * (see cgit_print_repo_readme) and this renders a deliberately small, safe + * subset client-side: headings, lists, blockquotes, rules, fenced and inline + * code, pipe tables, links, images and emphasis. Every run of text is escaped + * before any markup is added, and link and image URLs are restricted to http, + * https, mailto and relative targets, so a hostile readme cannot inject markup + * or scripts. Fenced code keeps its data-lang so the highlighter below styles + * it. Without JavaScript the escaped source stays readable as plain text. + * + * This is intentionally a subset, not CommonMark: no reference links, raw HTML + * passthrough, nested lists or setext headings. Configure an about-filter to + * replace it, or set enable-markdown=0 to turn it off. */ + +(function () { + +var MAX_BYTES = 400000; + +function esc(s) { + return s.replace(/&/g, "&").replace(//g, ">"); +} + +function escAttr(s) { + return esc(s).replace(/"/g, """).replace(/'/g, "'"); +} + +/* Return the url if its scheme is safe, else "". Whitespace and control bytes + * are stripped before the scheme is read because browsers ignore them when + * resolving it, so "java\nscript:..." must still be caught as javascript. */ +function safeUrl(url) { + url = (url || "").replace(/[\u0000-\u0020]+/g, ""); + var scheme = /^([a-z][a-z0-9+.\-]*):/i.exec(url); + if (scheme && !/^(https?|mailto)$/i.test(scheme[1])) + return ""; + return url; +} + +function link(text, url, image) { + var u = safeUrl(url); + if (!u) + return image ? esc("![" + text + "]") : inline(text); + if (image) + return "" + escAttr(text) + ""; + return "" + inline(text) + ""; +} + +/* Inline rendering over one block of text. Scans to the next marker character + * and bulk-escapes the plain text in between, so it stays roughly linear. */ +function inline(s) { + var out = "", i = 0, n = s.length, marker = /[`!\[*_]/g, m, rest; + while (i < n) { + marker.lastIndex = i; + m = marker.exec(s); + if (!m) { out += esc(s.slice(i)); break; } + if (m.index > i) { out += esc(s.slice(i, m.index)); i = m.index; } + rest = s.slice(i); + if ((m = /^`([^`]+)`/.exec(rest))) + out += "" + esc(m[1]) + ""; + else if ((m = /^!\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)/.exec(rest))) + out += link(m[1], m[2], true); + else if ((m = /^\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)/.exec(rest))) + out += link(m[1], m[2], false); + else if ((m = /^(\*\*|__)([\s\S]+?)\1/.exec(rest))) + out += "" + inline(m[2]) + ""; + else if ((m = /^(\*|_)([^\s][\s\S]*?)\1/.exec(rest))) + out += "" + inline(m[2]) + ""; + else { out += esc(s.charAt(i)); i++; continue; } + i += m[0].length; + } + return out; +} + +function cells(row) { + return row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|").map(function (c) { + return c.trim(); + }); +} + +function render(src) { + var lines = src.replace(/\r\n?/g, "\n").split("\n"); + var out = "", i = 0, n = lines.length, line, m, k; + while (i < n) { + line = lines[i]; + if (/^\s*$/.test(line)) { i++; continue; } + if ((m = /^\s*(`{3,}|~{3,})\s*([\w.+#-]*)/.exec(line))) { + var fence = m[1].charAt(0) === "`" ? /^\s*`{3,}\s*$/ : /^\s*~{3,}\s*$/; + var lang = m[2], code = ""; + for (i++; i < n && !fence.test(lines[i]); i++) + code += lines[i] + "\n"; + i++; + out += "
" + esc(code) + "
"; + continue; + } + if ((m = /^(#{1,6})\s+(.*?)\s*#*\s*$/.exec(line))) { + k = m[1].length; + out += "" + inline(m[2]) + ""; + i++; continue; + } + if (/^\s*([-*_])(\s*\1){2,}\s*$/.test(line)) { out += "
"; i++; continue; } + if (/^\s*>/.test(line)) { + var q = ""; + for (; i < n && /^\s*>/.test(lines[i]); i++) + q += lines[i].replace(/^\s*>\s?/, "") + "\n"; + out += "
" + render(q) + "
"; + continue; + } + if (line.indexOf("|") >= 0 && i + 1 < n && + /^\s*\|?(\s*:?-+:?\s*\|)+\s*:?-+:?\s*\|?\s*$/.test(lines[i + 1])) { + var head = cells(line), t = ""; + for (k = 0; k < head.length; k++) + t += ""; + t += ""; + for (i += 2; i < n && lines[i].indexOf("|") >= 0 && !/^\s*$/.test(lines[i]); i++) { + var row = cells(lines[i]); + t += ""; + for (k = 0; k < row.length; k++) + t += ""; + t += ""; + } + out += t + "
" + inline(head[k]) + "
" + inline(row[k]) + "
"; + continue; + } + if (/^\s*([-*+]|\d+[.)])\s+/.test(line)) { + var ordered = /^\s*\d/.test(line), tag = ordered ? "ol" : "ul"; + out += "<" + tag + ">"; + for (; i < n && (m = /^\s*([-*+]|\d+[.)])\s+(.*)$/.exec(lines[i])); i++) { + if ((/\d/.test(m[1])) !== ordered) break; + out += "
  • " + inline(m[2]) + "
  • "; + } + out += ""; + continue; + } + /* Always consume the current line so i advances even when it + * matched none of the block branches above. */ + var para = lines[i++]; + for (; i < n && !/^\s*$/.test(lines[i]) && + !/^\s*(#{1,6}\s|>|`{3,}|~{3,}|([-*+]|\d+[.)])\s)/.test(lines[i]); i++) + para += "\n" + lines[i]; + out += "

    " + inline(para).replace(/\n/g, "
    ") + "

    "; + } + return out; +} + +document.addEventListener("DOMContentLoaded", function () { + var nodes = document.querySelectorAll("div#cgit [data-markdown]"), i, el, text; + for (i = 0; i < nodes.length; i++) { + el = nodes[i]; + text = el.textContent; + if (!text || text.length > MAX_BYTES) + continue; + try { + el.innerHTML = render(text); + } catch (e) { + /* leave the escaped source in place on any failure */ + } + } +}, false); + +})(); + /* Built-in syntax highlighting for the blob view. When no server-side * source filter is configured, cgit tags the element with * data-lang set to the file's extension (or bare name, so Makefile and diff --git a/cgitrc.5.txt b/cgitrc.5.txt index fd420f1..b5315cc 100644 --- a/cgitrc.5.txt +++ b/cgitrc.5.txt @@ -31,8 +31,9 @@ about-filter:: about pages (both top-level and for each repository). The command will get the content of the about-file on its STDIN, the name of the file as the first argument, and the STDOUT from the command will be - included verbatim on the about page. Default value: none. See - also: "FILTER API". + included verbatim on the about page. Default value: none. When no + about-filter is set, a markdown readme is instead rendered by the + bundled cgit.js (see enable-markdown). See also: "FILTER API". agefile:: Specifies a path, relative to each repository path, which can be used @@ -229,6 +230,14 @@ enable-tree-group-dirs:: in git's own order, with directories and files intermixed by name. Default value: "0". +enable-markdown:: + Flag which, when set to "1", renders a markdown readme on the about + page. cgit escapes the source and a small renderer bundled in cgit.js + formats it in the browser, so the page stays readable as plain text + when scripting is off. It applies only when no about-filter handles the + file, and only to names ending in .md, .markdown, .mkd or .mdown. + Default value: "1". + favicon:: Url used as link to a shortcut icon for cgit. It is suggested to use the value "/favicon.ico" since certain browsers will ignore other @@ -918,8 +927,8 @@ mimetype.svg=image/svg+xml # Source code is highlighted in the browser by the bundled cgit.js, so no # source-filter is needed here. Set one only to override that. -# Format markdown, restructuredtext, manpages, text files, and html files -# through the right converters +# Markdown about pages render client-side by default (see enable-markdown), +# so a filter is only needed for other formats such as manpages. about-filter=/var/www/cgit/filters/about-formatting.sh ## diff --git a/extensions/html-converters/md2html b/extensions/html-converters/md2html deleted file mode 100755 index 59f43a8..0000000 --- a/extensions/html-converters/md2html +++ /dev/null @@ -1,304 +0,0 @@ -#!/usr/bin/env python3 -import markdown -import sys -import io -from pygments.formatters import HtmlFormatter -from markdown.extensions.toc import TocExtension -sys.stdin = io.TextIOWrapper(sys.stdin.buffer, encoding='utf-8') -sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding='utf-8') -sys.stdout.write(''' - -''') -sys.stdout.write("
    ") -sys.stdout.flush() -# Note: you may want to run this through bleach for sanitization -markdown.markdownFromFile( - output_format="html5", - extensions=[ - "markdown.extensions.fenced_code", - "markdown.extensions.codehilite", - "markdown.extensions.tables", - "markdown.extensions.sane_lists", - TocExtension(anchorlink=True)], - extension_configs={ - "markdown.extensions.codehilite":{"css_class":"highlight"}}) -sys.stdout.write("
    ") diff --git a/extensions/html-converters/rst2html b/extensions/html-converters/rst2html deleted file mode 100755 index 02d90f8..0000000 --- a/extensions/html-converters/rst2html +++ /dev/null @@ -1,2 +0,0 @@ -#!/bin/bash -exec rst2html.py --template <(echo -e "%(stylesheet)s\n%(body_pre_docinfo)s\n%(docinfo)s\n%(body)s") diff --git a/source/cgit.c b/source/cgit.c index 29b2131..8a7027c 100644 --- a/source/cgit.c +++ b/source/cgit.c @@ -201,6 +201,8 @@ static void config_cb(const char *name, const char *value) ctx.cfg.enable_tree_linenumbers = atoi(value); else if (!strcmp(name, "enable-tree-group-dirs")) ctx.cfg.enable_tree_group_dirs = atoi(value); + else if (!strcmp(name, "enable-markdown")) + ctx.cfg.enable_markdown = atoi(value); else if (!strcmp(name, "enable-git-config")) ctx.cfg.enable_git_config = atoi(value); else if (!strcmp(name, "enable-cache-list")) @@ -394,6 +396,7 @@ static void prepare_context(void) ctx.cfg.enable_http_clone = 1; ctx.cfg.enable_index_owner = 1; ctx.cfg.enable_tree_linenumbers = 1; + ctx.cfg.enable_markdown = 1; ctx.cfg.enable_git_config = 0; ctx.cfg.max_repo_count = 50; ctx.cfg.max_commit_count = 50; diff --git a/source/cgit.h b/source/cgit.h index ab7b284..3cd0317 100644 --- a/source/cgit.h +++ b/source/cgit.h @@ -243,6 +243,7 @@ struct cgit_config { int enable_html_serving; int enable_tree_linenumbers; int enable_tree_group_dirs; + int enable_markdown; int enable_git_config; int enable_cache_list; int local_time; diff --git a/source/ui-blob.c b/source/ui-blob.c index bc91656..7720a28 100644 --- a/source/ui-blob.c +++ b/source/ui-blob.c @@ -67,7 +67,7 @@ done: return walk_tree_ctx.found_path; } -int cgit_print_file(char *path, const char *head, int file_only) +int cgit_print_file(char *path, const char *head, int file_only, int html_escape) { struct object_id oid; enum object_type type; @@ -106,7 +106,10 @@ int cgit_print_file(char *path, const char *head, int file_only) if (!buf) return -1; buf[size] = '\0'; - html_raw(buf, size); + if (html_escape) + html_txt(buf); + else + html_raw(buf, size); free(buf); return 0; } diff --git a/source/ui-blob.h b/source/ui-blob.h index 16847b2..efbc94e 100644 --- a/source/ui-blob.h +++ b/source/ui-blob.h @@ -2,7 +2,7 @@ #define UI_BLOB_H extern int cgit_ref_path_exists(const char *path, const char *ref, int file_only); -extern int cgit_print_file(char *path, const char *head, int file_only); +extern int cgit_print_file(char *path, const char *head, int file_only, int html_escape); extern void cgit_print_blob(const char *hex, char *path, const char *head, int file_only); #endif /* UI_BLOB_H */ diff --git a/source/ui-summary.c b/source/ui-summary.c index 947812a..7e533e1 100644 --- a/source/ui-summary.c +++ b/source/ui-summary.c @@ -99,6 +99,17 @@ static char* append_readme_path(const char *filename, const char *ref, const cha return full_path; } +static int readme_is_markdown(const char *filename) +{ + const char *ext = strrchr(filename, '.'); + + if (!ext || !ext[1]) + return 0; + ext++; + return !strcasecmp(ext, "md") || !strcasecmp(ext, "markdown") || + !strcasecmp(ext, "mkd") || !strcasecmp(ext, "mdown"); +} + void cgit_print_repo_readme(const char *path) { char *filename, *ref, *mimetype; @@ -128,16 +139,35 @@ void cgit_print_repo_readme(const char *path) goto done; } - /* Print the calculated readme, either from the git repo or from the - * filesystem, while applying the about-filter. - */ html("
    "); - cgit_open_filter(ctx.repo->about_filter, filename); - if (ref) - cgit_print_file(filename, ref, 1); - else - html_include(filename); - cgit_close_filter(ctx.repo->about_filter); + if (!ctx.repo->about_filter && ctx.cfg.enable_markdown && + readme_is_markdown(filename)) { + /* No about-filter is set, so hand the markdown source to the + * built-in client-side renderer in cgit.js. The source is + * escaped here and rendered in the browser, and it degrades to + * readable plain text when scripting is off. + */ + html("
    "); + if (ref) { + cgit_print_file(filename, ref, 1, 1); + } else { + struct strbuf sb = STRBUF_INIT; + if (strbuf_read_file(&sb, filename, 0) >= 0) + html_txt(sb.buf); + strbuf_release(&sb); + } + html("
    "); + } else { + /* Otherwise print the readme through the about-filter, or raw + * when none is configured. + */ + cgit_open_filter(ctx.repo->about_filter, filename); + if (ref) + cgit_print_file(filename, ref, 1, 0); + else + html_include(filename); + cgit_close_filter(ctx.repo->about_filter); + } html("
    "); if (free_filename) -- cgit v2.8.0