From 5104f90310a2028e50667ea12d9e75e820fbbd36 Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Sat, 25 Jul 2026 15:03:48 -1000 Subject: Replace the browser markdown renderer with a filter The readme is now escaped plain text unless `about-filter` points at the new `about-render.lua`, which renders markdown, man pages and plain text server-side. `enable-markdown` goes away with the renderer. --- README.txt | 2 + assets/cgit.css | 8 - assets/cgit.js | 164 ------------ cgitrc.5.txt | 19 +- examples/cgitrc | 19 +- extensions/about-render.lua | 592 ++++++++++++++++++++++++++++++++++++++++++++ source/cgit.c | 3 - source/cgit.h | 1 - source/ui-summary.c | 34 +-- tests/t0200-security.sh | 4 +- 10 files changed, 613 insertions(+), 233 deletions(-) create mode 100644 extensions/about-render.lua diff --git a/README.txt b/README.txt index d3b32ec..d385d31 100644 --- a/README.txt +++ b/README.txt @@ -96,6 +96,8 @@ header lists the exact install commands for its own dependencies. `luaossl`. * The syntax highlighter (`syntax-highlight.lua`) needs `lpeg` and a Scintillua lexer set. +* The about-page renderer (`about-render.lua`) needs `lpeg` for markdown and + man pages. Plain text needs only Lua. These filters target Lua 5.1 through 5.4 and LuaJIT. `luaossl` has no Lua 5.5 build, so build cgit against 5.1 to 5.4 if you use the auth or email filters. diff --git a/assets/cgit.css b/assets/cgit.css index 3255068..ba7d7b1 100644 --- a/assets/cgit.css +++ b/assets/cgit.css @@ -520,14 +520,6 @@ div#cgit pre.plaintext { margin: 0; } -/* Markdown source before the client-side renderer has run, and for good - * when scripting is off. Keeps the line structure so it reads as text. */ -div#cgit .markdown[data-markdown] { - white-space: pre-wrap; - overflow-wrap: anywhere; - font-family: var(--font-mono); -} - div#cgit .markdown { line-height: 1.6; overflow-wrap: break-word; diff --git a/assets/cgit.js b/assets/cgit.js index 10a34cd..6fe53ba 100644 --- a/assets/cgit.js +++ b/assets/cgit.js @@ -194,167 +194,3 @@ document.addEventListener("DOMContentLoaded", function () { }, false); })(); - -/* Built-in Markdown rendering for the about page. When no about-filter is - * configured, cgit escapes a markdown readme into a data-markdown container - * (see cgit_print_repo_readme) and this renders a deliberately small, safe - * subset client-side: headings, lists, blockquotes, rules, fenced and inline - * code, pipe tables, links, images and emphasis. Every run of text is escaped - * before any markup is added, and link and image URLs are restricted to http, - * https, mailto and relative targets, so a hostile readme cannot inject markup - * or scripts. Fenced code carries its language in data-lang as a styling - * hook. Without JavaScript the escaped source stays readable as plain text. - * - * This is intentionally a subset, not CommonMark: no reference links, raw HTML - * passthrough, nested lists or setext headings. Configure an about-filter to - * replace it, or set enable-markdown=0 to turn it off. */ - -(function () { - -var MAX_BYTES = 400000; - -function esc(s) { - return s.replace(/&/g, "&").replace(//g, ">"); -} - -function escAttr(s) { - return esc(s).replace(/"/g, """).replace(/'/g, "'"); -} - -/* Return the url if its scheme is safe, else "". Whitespace and control bytes - * are stripped before the scheme is read because browsers ignore them when - * resolving it, so "java\nscript:..." must still be caught as javascript. */ -function safeUrl(url) { - url = (url || "").replace(/[\u0000-\u0020]+/g, ""); - var scheme = /^([a-z][a-z0-9+.\-]*):/i.exec(url); - if (scheme && !/^(https?|mailto)$/i.test(scheme[1])) - return ""; - return url; -} - -function link(text, url, image) { - var u = safeUrl(url); - if (!u) - return image ? esc("![" + text + "]") : inline(text); - if (image) - return "" + escAttr(text) + ""; - return "" + inline(text) + ""; -} - -/* Inline rendering over one block of text. Scans to the next marker character - * and bulk-escapes the plain text in between, so it stays roughly linear. */ -function inline(s) { - var out = "", i = 0, n = s.length, marker = /[`!\[*_]/g, m, rest; - while (i < n) { - marker.lastIndex = i; - m = marker.exec(s); - if (!m) { out += esc(s.slice(i)); break; } - if (m.index > i) { out += esc(s.slice(i, m.index)); i = m.index; } - rest = s.slice(i); - if ((m = /^`([^`]+)`/.exec(rest))) - out += "" + esc(m[1]) + ""; - else if ((m = /^!\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)/.exec(rest))) - out += link(m[1], m[2], true); - else if ((m = /^\[([^\]]*)\]\(\s*([^)\s]+)[^)]*\)/.exec(rest))) - out += link(m[1], m[2], false); - else if ((m = /^(\*\*|__)([\s\S]+?)\1/.exec(rest))) - out += "" + inline(m[2]) + ""; - else if ((m = /^(\*|_)([^\s][\s\S]*?)\1/.exec(rest))) - out += "" + inline(m[2]) + ""; - else { out += esc(s.charAt(i)); i++; continue; } - i += m[0].length; - } - return out; -} - -function cells(row) { - return row.trim().replace(/^\|/, "").replace(/\|$/, "").split("|").map(function (c) { - return c.trim(); - }); -} - -function render(src) { - var lines = src.replace(/\r\n?/g, "\n").split("\n"); - var out = "", i = 0, n = lines.length, line, m, k; - while (i < n) { - line = lines[i]; - if (/^\s*$/.test(line)) { i++; continue; } - if ((m = /^\s*(`{3,}|~{3,})\s*([\w.+#-]*)/.exec(line))) { - var fence = m[1].charAt(0) === "`" ? /^\s*`{3,}\s*$/ : /^\s*~{3,}\s*$/; - var lang = m[2], code = ""; - for (i++; i < n && !fence.test(lines[i]); i++) - code += lines[i] + "\n"; - i++; - out += "
" + esc(code) + "
"; - continue; - } - if ((m = /^(#{1,6})\s+(.*?)\s*#*\s*$/.exec(line))) { - k = m[1].length; - out += "" + inline(m[2]) + ""; - i++; continue; - } - if (/^\s*([-*_])(\s*\1){2,}\s*$/.test(line)) { out += "
"; i++; continue; } - if (/^\s*>/.test(line)) { - var q = ""; - for (; i < n && /^\s*>/.test(lines[i]); i++) - q += lines[i].replace(/^\s*>\s?/, "") + "\n"; - out += "
" + render(q) + "
"; - continue; - } - if (line.indexOf("|") >= 0 && i + 1 < n && - /^\s*\|?(\s*:?-+:?\s*\|)+\s*:?-+:?\s*\|?\s*$/.test(lines[i + 1])) { - var head = cells(line), t = ""; - for (k = 0; k < head.length; k++) - t += ""; - t += ""; - for (i += 2; i < n && lines[i].indexOf("|") >= 0 && !/^\s*$/.test(lines[i]); i++) { - var row = cells(lines[i]); - t += ""; - for (k = 0; k < row.length; k++) - t += ""; - t += ""; - } - out += t + "
" + inline(head[k]) + "
" + inline(row[k]) + "
"; - continue; - } - if (/^\s*([-*+]|\d+[.)])\s+/.test(line)) { - var ordered = /^\s*\d/.test(line), tag = ordered ? "ol" : "ul"; - out += "<" + tag + ">"; - for (; i < n && (m = /^\s*([-*+]|\d+[.)])\s+(.*)$/.exec(lines[i])); i++) { - if ((/\d/.test(m[1])) !== ordered) break; - out += "
  • " + inline(m[2]) + "
  • "; - } - out += ""; - continue; - } - /* Always consume the current line so i advances even when it - * matched none of the block branches above. */ - var para = lines[i++]; - for (; i < n && !/^\s*$/.test(lines[i]) && - !/^\s*(#{1,6}\s|>|`{3,}|~{3,}|([-*+]|\d+[.)])\s)/.test(lines[i]); i++) - para += "\n" + lines[i]; - out += "

    " + inline(para).replace(/\n/g, "
    ") + "

    "; - } - return out; -} - -document.addEventListener("DOMContentLoaded", function () { - var nodes = document.querySelectorAll("div#cgit [data-markdown]"), i, el, text; - for (i = 0; i < nodes.length; i++) { - el = nodes[i]; - text = el.textContent; - if (!text || text.length > MAX_BYTES) - continue; - try { - el.innerHTML = render(text); - /* The attribute doubles as the style hook for the - * unrendered source, so drop it once rendered. */ - el.removeAttribute("data-markdown"); - } catch (e) { - /* leave the escaped source in place on any failure */ - } - } -}, false); - -})(); diff --git a/cgitrc.5.txt b/cgitrc.5.txt index 049b234..32f8b86 100644 --- a/cgitrc.5.txt +++ b/cgitrc.5.txt @@ -32,8 +32,9 @@ about-filter:: get the content of the about-file on its STDIN, the name of the file as the first argument, and the STDOUT from the command will be included verbatim on the about page. Default value: none. When no - about-filter is set, a markdown readme is instead rendered by the - bundled cgit.js (see enable-markdown). See also: "FILTER API". + about-filter is set, the readme is escaped and served as plain text. + A bundled Lua filter, about-render.lua, renders markdown, man pages + and plain text when about-filter points at it. See also: "FILTER API". agefile:: Specifies a path, relative to each repository path, which can be used @@ -241,14 +242,6 @@ enable-tree-group-dirs:: in git's own order, with directories and files intermixed by name. Default value: "0". -enable-markdown:: - Flag which, when set to "1", renders a markdown readme on the about - page. cgit escapes the source and a small renderer bundled in cgit.js - formats it in the browser, so the page stays readable as plain text - when scripting is off. It applies only when no about-filter handles the - file, and only to names ending in .md, .markdown, .mkd or .mdown. - Default value: "1". - favicon:: Url used as link to the icon for cgit. Any path works, but keeping the value "/favicon.ico" is still worthwhile, since clients that do @@ -955,9 +948,9 @@ mimetype.svg=image/svg+xml # extensions/syntax-highlight.lua highlights through the Scintillua lexers. # source-filter=lua:/usr/share/cgit/extensions/syntax-highlight.lua -# Markdown about pages render client-side by default (see enable-markdown). -# For other formats such as manpages, point about-filter at your own script. -# about-filter=/var/www/cgit/filters/my-about-formatter +# About pages are escaped plain text unless an about-filter is set. The shipped +# extensions/about-render.lua renders markdown, man pages and plain text. +# about-filter=lua:/usr/share/cgit/extensions/about-render.lua ## ## Search for these files in the root of the default branch of repositories diff --git a/examples/cgitrc b/examples/cgitrc index 5626b58..bb87a9b 100644 --- a/examples/cgitrc +++ b/examples/cgitrc @@ -11,9 +11,9 @@ # directory, or list repositories by hand in the per-repository section at # the end of this file. # -# In this fork, markdown readmes are rendered in the browser, so no -# about-filter is needed for them. Source syntax highlighting is optional -# and ships as a source-filter, see the filter section below. +# About pages (markdown, man and plain-text readmes) render through the +# bundled about-render.lua about-filter, see the filter section below. Source +# syntax highlighting is optional and ships as a source-filter there too. # # One key=value pair per line. Lines starting with # are comments. @@ -237,10 +237,6 @@ enable-tree-linenumbers=1 # Default is 0. enable-tree-group-dirs=0 -# Render a markdown readme client-side on the about page. Values are 0 or 1. -# Default is 1. -enable-markdown=1 - # Default maximum statistics period. Leaving this unset disables statistics. # Values are week, month, quarter or year. Default is unset. #max-stats=week @@ -288,10 +284,10 @@ enable-http-clone=1 # 0 or 1. Default is 0. enable-filter-overrides=0 -# Filter command used to format about-page content. Markdown already renders -# in the browser, so this is only for other formats such as man pages. Value -# is a command optionally prefixed with exec or lua. Default is none. -#about-filter=exec:/path/to/your-command +# Filter command used to format about-page content. The bundled about-render.lua +# renders markdown, man pages and plain text. Value is a command optionally +# prefixed with exec or lua. Default is none. +#about-filter=lua:/usr/share/cgit/extensions/about-render.lua # Filter command used to format commit messages, for example to turn object # names and issue numbers into links. Value is a command optionally prefixed @@ -563,4 +559,3 @@ noplainemail=0 # enable-filter-overrides is 1. Value is a command. Default is the global # email-filter value. #repo.email-filter=lua:/usr/share/cgit/extensions/email-gravatar.lua - diff --git a/extensions/about-render.lua b/extensions/about-render.lua new file mode 100644 index 0000000..073b531 --- /dev/null +++ b/extensions/about-render.lua @@ -0,0 +1,592 @@ +-- Server-side about-page rendering for cgit, used with the about-filter setting +-- in cgitrc and the lua: prefix so it runs in cgit's embedded interpreter with +-- no per-request process. +-- +-- about-filter=lua:/usr/lib/cgit/filters/about-render.lua +-- +-- One filter renders the three baseline about formats, chosen by the readme's +-- file extension. +-- +-- markdown .md .markdown .mkd .mdown +-- man page .1 through .9 and .man +-- plain text everything else, and the fallback for anything that fails +-- +-- Add a format by giving it a render function and a row in the handler table +-- near the foot of this file. Nothing else needs to change. +-- +-- SUPPORTED LUA +-- +-- Lua 5.1 through 5.4 and LuaJIT. +-- +-- REQUIREMENTS +-- +-- lpeg, the parsing module, for the Lua cgit is linked against. Markdown and +-- man pages are parsed with lpeg grammars, so without lpeg both fall back to +-- escaped plain text. It ships with cgit's syntax highlighter too, so a cgit +-- that colours source already has it. +-- +-- # Debian and Ubuntu +-- sudo apt install lua-lpeg +-- # Fedora +-- sudo dnf install lua-lpeg +-- # Alpine +-- sudo apk add lua5.1-lpeg +-- # or with LuaRocks, matched to your Lua version +-- sudo luarocks --lua-version 5.1 install lpeg +-- +-- Plain text needs only Lua itself. +-- +-- SECURITY +-- +-- The about page renders untrusted repository content, so the safety rule is +-- that every run of text reaches the page through cgit's own html_txt, every +-- attribute value through html_attr, and every link or image target through +-- the safe_url scheme allowlist below before html_attr. Only http, https, +-- mailto and relative targets survive, so a hostile readme cannot inject markup +-- or a javascript: url. Because all output is routed through those sinks by +-- construction, a bug in the parser can only mis-render, never inject. +-- +-- LIMITATIONS +-- +-- Markdown is a deliberate subset, not CommonMark. Headings, thematic breaks, +-- fenced and inline code, blockquotes, pipe tables, single-level lists, links, +-- images and emphasis. No reference links, no raw HTML passthrough, no nested +-- lists, no setext headings. Emphasis does not span a hard line break inside a +-- paragraph. Man rendering covers the common macros (section headings, filled +-- and no-fill paragraphs, bold and italic, font escapes) and drops the rest. +-- cgit sends filter output through a C string sink that stops at the first NUL +-- byte, so content past a NUL is truncated. +-- +-- OUTPUT +-- +-- Markdown is wrapped in
    , man pages in +--
    , plain text in
    , all
    +-- styled by assets/cgit.css.
    +
    +local ok_lpeg, lpeg = pcall(require, "lpeg")
    +
    +
    +-- ===== configuration ===================================================
    +
    +-- Readmes larger than this are served as escaped plain text rather than parsed.
    +local max_bytes = 512 * 1024
    +
    +-- Ceiling on blockquote and emphasis nesting, so a hostile readme of stacked
    +-- markers cannot drive the recursive parser into a stack overflow.
    +local max_depth = 24
    +
    +local filename, chunks = "", {}
    +
    +
    +-- ===== helpers =========================================================
    +
    +local function trim(s)
    +	return (s:gsub("^%s+", ""):gsub("%s+$", ""))
    +end
    +
    +-- Split on newlines after normalising CRLF and CR, returning the lines without
    +-- their terminators. A trailing newline yields a final empty line, which every
    +-- caller treats as blank.
    +local function split_lines(s)
    +	s = s:gsub("\r\n?", "\n")
    +	local lines, start = {}, 1
    +	while true do
    +		local nl = s:find("\n", start, true)
    +		if not nl then
    +			lines[#lines + 1] = s:sub(start)
    +			return lines
    +		end
    +		lines[#lines + 1] = s:sub(start, nl - 1)
    +		start = nl + 1
    +	end
    +end
    +
    +-- Split a table row into trimmed cells, dropping one optional leading and
    +-- trailing pipe.
    +local function split_cells(row)
    +	row = trim(row):gsub("^|", ""):gsub("|$", "")
    +	local cells = {}
    +	for c in (row .. "|"):gmatch("(.-)|") do
    +		cells[#cells + 1] = trim(c)
    +	end
    +	return cells
    +end
    +
    +-- Return the url if its scheme is safe, else nil. Control and whitespace bytes
    +-- are stripped anywhere first because a browser ignores them when resolving the
    +-- scheme, so "java\nscript:" must still be caught as javascript. The class is
    +-- %c%s rather than a literal 0x00 range so it is safe on Lua 5.1, where an
    +-- embedded zero byte ends a pattern.
    +local function safe_url(url)
    +	url = url:gsub("[%c%s]", "")
    +	-- A leading "//" or "/\" is scheme-relative, which a browser resolves to
    +	-- another origin, so it is not the local target the parser meant to allow.
    +	if url:sub(1, 2) == "//" or url:sub(1, 2) == "/\\" then
    +		return nil
    +	end
    +	local scheme = url:match("^(%a[%w%+%.%-]*):")
    +	if scheme then
    +		scheme = scheme:lower()
    +		if scheme ~= "http" and scheme ~= "https" and scheme ~= "mailto" then
    +			return nil
    +		end
    +	end
    +	return url
    +end
    +
    +
    +-- ===== emit ============================================================
    +-- html, html_txt and html_attr are injected by cgit's lua filter host. Tag
    +-- scaffolding is a literal argument to html; every value that came from the
    +-- repository goes through html_txt, html_attr or safe_url.
    +
    +local emit_inline
    +
    +emit_inline = function(list)
    +	for _, nd in ipairs(list) do
    +		local t = nd.t
    +		if t == "text" then
    +			html_txt(nd.v)
    +		elseif t == "code" then
    +			html(""); html_txt(nd.v); html("")
    +		elseif t == "strong" then
    +			html(""); emit_inline(nd.kids); html("")
    +		elseif t == "em" then
    +			html(""); emit_inline(nd.kids); html("")
    +		elseif t == "link" then
    +			local u = safe_url(nd.url)
    +			if u then
    +				html("")
    +				emit_inline(nd.kids)
    +				html("")
    +			else
    +				emit_inline(nd.kids)
    +			end
    +		elseif t == "image" then
    +			local u = safe_url(nd.url)
    +			if u then
    +				html(""); html_attr(nd.alt); html("")
    +			else
    +				html_txt("![" .. nd.alt .. "]")
    +			end
    +		end
    +	end
    +end
    +
    +local emit_blocks
    +
    +emit_blocks = function(blocks)
    +	for _, b in ipairs(blocks) do
    +		local t = b.t
    +		if t == "heading" then
    +			html("")
    +			emit_inline(b.kids)
    +			html("")
    +		elseif t == "hr" then
    +			html("
    ") + elseif t == "code" then + html("
    "); html_txt(b.text); html("
    ") + elseif t == "blockquote" then + html("
    "); emit_blocks(b.blocks); html("
    ") + elseif t == "table" then + html("") + for _, c in ipairs(b.head) do + html("") + end + html("") + for _, row in ipairs(b.rows) do + html("") + for _, c in ipairs(row) do + html("") + end + html("") + end + html("
    "); emit_inline(c); html("
    "); emit_inline(c); html("
    ") + elseif t == "list" then + local tag = b.ordered and "ol" or "ul" + html("<" .. tag .. ">") + for _, item in ipairs(b.items) do + html("
  • "); emit_inline(item); html("
  • ") + end + html("") + elseif t == "para" then + html("

    ") + for k, line in ipairs(b.lines) do + if k > 1 then html("
    ") end + emit_inline(line) + end + html("

    ") + end + end +end + + +-- ===== plain text ====================================================== + +local function render_plaintext(text) + html("
    ")
    +	html_txt(text)
    +	html("
    ") +end + + +-- ===== markdown and man ================================================ +-- Both are parsed with lpeg into the block/inline node trees emit_blocks and +-- emit_inline above consume. Parsing is pure and builds a tree; emit is the +-- only step that writes output, so a parse failure falls back to plain text +-- without leaving half a page behind. + +local render_markdown, render_man + +if ok_lpeg then + local P, R, S, C, Ct = lpeg.P, lpeg.R, lpeg.S, lpeg.C, lpeg.Ct + + local sp = S(" \t") + local ws = S(" \t\r\n\f\v") + local eol = P(-1) + + -- ----- markdown inline ----- + local marker = S("`![*_") + local urlchar = 1 - S(")") - ws + -- Bound the link and image inner scans. Without a cap a readme of unclosed + -- brackets ("[[[[...") makes each position scan to end of line for a "]" + -- that never comes, which is quadratic. Real link text and urls sit far + -- under this, and anything longer simply renders as plain text. + local CAP = 512 + + local parse_inline + local inline_depth = 0 + + local function node_text(s) return { t = "text", v = s } end + local function node_code(s) return { t = "code", v = s } end + local function node_strong(s) return { t = "strong", kids = parse_inline(s) } end + local function node_em(s) return { t = "em", kids = parse_inline(s) } end + local function node_link(text, url) return { t = "link", kids = parse_inline(text), url = url } end + local function node_image(alt, url) return { t = "image", alt = alt, url = url } end + + local inline_grammar = Ct(( + (P("`") * C((1 - P("`")) ^ 1) * P("`")) / node_code + + (P("![") * C((1 - P("]")) ^ (-CAP)) * P("]") * P("(") * sp ^ 0 + * C(urlchar * urlchar ^ (-CAP)) * (1 - P(")")) ^ (-CAP) * P(")")) / node_image + + (P("[") * C((1 - P("]")) ^ (-CAP)) * P("]") * P("(") * sp ^ 0 + * C(urlchar * urlchar ^ (-CAP)) * (1 - P(")")) ^ (-CAP) * P(")")) / node_link + + (P("**") * C((1 - P("**")) ^ 1) * P("**")) / node_strong + + (P("__") * C((1 - P("__")) ^ 1) * P("__")) / node_strong + + (P("*") * C((1 - ws) * (1 - P("*")) ^ 0) * P("*")) / node_em + + (P("_") * C((1 - ws) * (1 - P("_")) ^ 0) * P("_")) / node_em + + C((1 - marker) ^ 1) / node_text + + C(P(1)) / node_text + ) ^ 0) + + parse_inline = function(s) + if inline_depth >= max_depth then + return { node_text(s) } + end + inline_depth = inline_depth + 1 + local nodes = inline_grammar:match(s) or { node_text(s) } + inline_depth = inline_depth - 1 + return nodes + end + + -- ----- markdown blocks ----- + -- Line matchers. Each is anchored at the start of a single line and returns + -- its captures, or nil when the line is not of that kind. + local langchar = R("az", "AZ", "09") + S("_.+#-") + local m_heading = C(P("#") * P("#") ^ -5) * sp ^ 1 * C(P(1) ^ 0) + local function rule(c) return sp ^ 0 * P(c) * (sp ^ 0 * P(c)) ^ 2 * sp ^ 0 * eol end + local m_hr = rule("-") + rule("*") + rule("_") + local m_fence = sp ^ 0 * (C(P("`") ^ 3) + C(P("~") ^ 3)) * sp ^ 0 * C(langchar ^ 0) + local m_close_bt = sp ^ 0 * P("`") ^ 3 * sp ^ 0 * eol + local m_close_ti = sp ^ 0 * P("~") ^ 3 * sp ^ 0 * eol + local m_bq = sp ^ 0 * P(">") + local bullet = S("-*+") + local number = R("09") ^ 1 * S(".)") + local m_item = sp ^ 0 * C(bullet + number) * sp ^ 1 * C(P(1) ^ 0) + local dcell = sp ^ 0 * P(":") ^ -1 * P("-") ^ 1 * P(":") ^ -1 * sp ^ 0 + -- A delimiter row is dash-cells joined by pipes. Require at least one pipe, + -- a leading one or one between cells, so a bare rule of dashes stays a + -- thematic break and an ordinary paragraph line is never taken for a table. + local pipe = P("|") + local m_tdelim = sp ^ 0 * ( + pipe * dcell * (pipe * dcell) ^ 0 * pipe ^ -1 + + dcell * (pipe * dcell) ^ 1 * pipe ^ -1 + ) * sp ^ 0 * eol + local m_blockstart = sp ^ 0 * ((P("#") * P("#") ^ -5 * sp) + P(">") + + P("`") ^ 3 + P("~") ^ 3 + ((bullet + number) * sp)) + + local function ordered(mark) return mark:match("%d") ~= nil end + + local parse_blocks + + parse_blocks = function(lines, depth) + local blocks, i, n = {}, 1, #lines + while i <= n do + local line = lines[i] + local fence, lang = m_fence:match(line) + local hashes, htext = m_heading:match(line) + local mark = m_item:match(line) + if line:match("^%s*$") then + i = i + 1 + elseif fence then + local close = (fence:sub(1, 1) == "`") and m_close_bt or m_close_ti + local code = {} + i = i + 1 + while i <= n and not close:match(lines[i]) do + code[#code + 1] = lines[i]; i = i + 1 + end + i = i + 1 + local body = table.concat(code, "\n") + if #code > 0 then body = body .. "\n" end + blocks[#blocks + 1] = { t = "code", lang = lang, text = body } + elseif hashes then + blocks[#blocks + 1] = { + t = "heading", + level = #hashes, + kids = parse_inline((htext:gsub("%s*#*%s*$", ""))), + } + i = i + 1 + elseif m_hr:match(line) then + blocks[#blocks + 1] = { t = "hr" } + i = i + 1 + elseif m_bq:match(line) then + local inner = {} + while i <= n and m_bq:match(lines[i]) do + inner[#inner + 1] = (lines[i]:gsub("^%s*>%s?", "")) + i = i + 1 + end + if depth < max_depth then + blocks[#blocks + 1] = { t = "blockquote", blocks = parse_blocks(inner, depth + 1) } + else + local para = {} + for _, l in ipairs(inner) do para[#para + 1] = parse_inline(l) end + blocks[#blocks + 1] = { t = "para", lines = para } + end + elseif i + 1 <= n and line:find("|", 1, true) and m_tdelim:match(lines[i + 1]) then + local head = {} + for _, c in ipairs(split_cells(line)) do head[#head + 1] = parse_inline(c) end + local rows = {} + i = i + 2 + while i <= n and lines[i]:find("|", 1, true) and not lines[i]:match("^%s*$") do + local row = {} + for _, c in ipairs(split_cells(lines[i])) do row[#row + 1] = parse_inline(c) end + rows[#rows + 1] = row + i = i + 1 + end + blocks[#blocks + 1] = { t = "table", head = head, rows = rows } + elseif mark then + local is_ol = ordered(mark) + local items = {} + while i <= n do + local mk, ct = m_item:match(lines[i]) + if not mk or ordered(mk) ~= is_ol then break end + items[#items + 1] = parse_inline(ct) + i = i + 1 + end + blocks[#blocks + 1] = { t = "list", ordered = is_ol, items = items } + else + local para = { parse_inline(line) } + i = i + 1 + while i <= n and not lines[i]:match("^%s*$") and not m_blockstart:match(lines[i]) do + para[#para + 1] = parse_inline(lines[i]); i = i + 1 + end + blocks[#blocks + 1] = { t = "para", lines = para } + end + end + return blocks + end + + render_markdown = function(text) + if #text > max_bytes then + return render_plaintext(text) + end + local ok, blocks = pcall(parse_blocks, split_lines(text), 0) + if not ok then + return render_plaintext(text) + end + html("
    ") + -- Emit is wrapped so a parser bug can at worst truncate the page, never + -- error out and let the outer fallback duplicate what was already sent. + pcall(emit_blocks, blocks) + html("
    ") + end + + -- ----- man pages ----- + -- Macro lines are matched with lpeg, and the roff inline font and character + -- escapes are an lpeg grammar producing a flat token list that a fold turns + -- into the same inline nodes markdown emits. Bold and italic are the current + -- font, which carries across a run, so the inline pass tokenises and folds + -- rather than nesting the way markdown does. + local m_macro = S(".'") * sp ^ 0 * C((1 - sp) ^ 1) * sp ^ 0 * C(P(1) ^ 0) + + local one_font = { B = "B", I = "I" } + local two_font = { CB = "B", BI = "B", CI = "I" } + local man_chars = { + aq = "'", cq = "'", oq = "'", dq = '"', lq = '"', rq = '"', + hy = "-", en = "-", em = "-", + } + local esc = P("\\") + local function font_tok(f) return { k = "font", f = f } end + local function text_tok(v) return { k = "text", v = v } end + local skip_tok = { k = "skip" } + + local man_inline_grammar = Ct(( + (esc * P("f") * P("(") * C(P(1) * P(1))) / function(nm) return font_tok(two_font[nm] or "R") end + + (esc * P("f") * P("[") * C((1 - P("]")) ^ 0) * P("]")) / function(nm) return font_tok(one_font[nm] or "R") end + + (esc * P("f") * C(P(1))) / function(f) return font_tok(one_font[f] or "R") end + + (esc * P("-")) / function() return text_tok("-") end + + (esc * P("e")) / function() return text_tok("\\") end + + (esc * P(" ")) / function() return text_tok(" ") end + + (esc * S("&|^")) / function() return skip_tok end + + (esc * P("(") * C(P(1) * P(1))) / function(nm) return text_tok(man_chars[nm] or "") end + + (esc * P("*") * (P("(") * P(1) * P(1) + P(1))) / function() return skip_tok end + + (esc * C(P(1))) / text_tok + + esc / function() return skip_tok end + + C((1 - esc) ^ 1) / text_tok + ) ^ 0) + + local function man_inline(s) + local toks = man_inline_grammar:match(s) or {} + local nodes, buf, font = {}, {}, "R" + local function flush() + if #buf == 0 then return end + local txt = table.concat(buf); buf = {} + if txt == "" then return end + if font == "B" then + nodes[#nodes + 1] = { t = "strong", kids = { { t = "text", v = txt } } } + elseif font == "I" then + nodes[#nodes + 1] = { t = "em", kids = { { t = "text", v = txt } } } + else + nodes[#nodes + 1] = { t = "text", v = txt } + end + end + for _, tk in ipairs(toks) do + if tk.k == "text" then + buf[#buf + 1] = tk.v + elseif tk.k == "font" then + flush(); font = tk.f + end + end + flush() + return nodes + end + + local function parse_man(text) + local lines = split_lines(text) + local blocks = {} + local para, pre, nofill = nil, nil, false + local function flush_para() + if para and #para > 0 then + blocks[#blocks + 1] = { t = "para", lines = para } + end + para = nil + end + local function flush_pre() + if pre then + local body = table.concat(pre, "\n") + if #pre > 0 then body = body .. "\n" end + blocks[#blocks + 1] = { t = "code", text = body } + pre = nil + end + end + local function add_line(nodes) + para = para or {} + para[#para + 1] = nodes + end + for _, line in ipairs(lines) do + local macro, rest = m_macro:match(line) + if rest then rest = (rest:gsub('^"(.*)"$', "%1")) end + if nofill then + if macro == "fi" then + nofill = false; flush_pre() + else + pre[#pre + 1] = line + end + elseif line:match("^%s*$") then + flush_para() + elseif macro and macro:sub(1, 1) == "\\" then + -- roff comment, dropped + elseif macro == "SH" or macro == "SS" then + flush_para() + blocks[#blocks + 1] = { + t = "heading", + level = (macro == "SH") and 2 or 3, + kids = man_inline(rest), + } + elseif macro == "nf" then + flush_para(); nofill = true; pre = {} + elseif macro == "fi" then + flush_pre() + elseif macro == "B" or macro == "BR" or macro == "RB" + or macro == "BI" or macro == "IB" then + add_line({ { t = "strong", kids = man_inline(rest) } }) + elseif macro == "I" or macro == "IR" or macro == "RI" then + add_line({ { t = "em", kids = man_inline(rest) } }) + elseif macro == "TH" or macro == "PP" or macro == "LP" + or macro == "P" or macro == "TP" or macro == "IP" + or macro == "HP" or macro == "RS" or macro == "RE" + or macro == "sp" or macro == "br" then + flush_para() + elseif macro then + if rest ~= "" then add_line(man_inline(rest)) end + else + add_line(man_inline(line)) + end + end + flush_para(); flush_pre() + return blocks + end + + render_man = function(text) + if #text > max_bytes then + return render_plaintext(text) + end + local ok, blocks = pcall(parse_man, text) + if not ok then + return render_plaintext(text) + end + html("
    ") + pcall(emit_blocks, blocks) + html("
    ") + end +else + -- No lpeg, so markdown and man readmes read as escaped source, not nothing. + render_markdown = render_plaintext + render_man = render_plaintext +end + + +-- ===== dispatch ======================================================== + +local handlers = {} +for _, ext in ipairs({ "md", "markdown", "mkd", "mdown" }) do + handlers[ext] = render_markdown +end +for _, ext in ipairs({ "1", "2", "3", "4", "5", "6", "7", "8", "9", "man" }) do + handlers[ext] = render_man +end + +function filter_open(name) + filename = name or "" + chunks = {} +end + +function filter_write(str) + chunks[#chunks + 1] = str +end + +function filter_close() + local text = table.concat(chunks) + chunks = {} + local ext = (filename:match("%.([^.]+)$") or ""):lower() + local render = handlers[ext] or render_plaintext + local ok = pcall(render, text) + if not ok then + render_plaintext(text) + end + return 0 +end diff --git a/source/cgit.c b/source/cgit.c index 5af2232..74600a7 100644 --- a/source/cgit.c +++ b/source/cgit.c @@ -211,8 +211,6 @@ static void config_cb(const char *name, const char *value) ctx.cfg.enable_tree_linenumbers = atoi(value); else if (!strcmp(name, "enable-tree-group-dirs")) ctx.cfg.enable_tree_group_dirs = atoi(value); - else if (!strcmp(name, "enable-markdown")) - ctx.cfg.enable_markdown = atoi(value); else if (!strcmp(name, "enable-git-config")) ctx.cfg.enable_git_config = atoi(value); else if (!strcmp(name, "enable-cache-list")) @@ -420,7 +418,6 @@ static void prepare_context(void) ctx.cfg.enable_http_clone = 1; ctx.cfg.enable_index_owner = 1; ctx.cfg.enable_tree_linenumbers = 1; - ctx.cfg.enable_markdown = 1; ctx.cfg.enable_git_config = 0; ctx.cfg.max_repo_count = 50; ctx.cfg.max_commit_count = 50; diff --git a/source/cgit.h b/source/cgit.h index b7caaf5..966851c 100644 --- a/source/cgit.h +++ b/source/cgit.h @@ -243,7 +243,6 @@ struct cgit_config { int enable_html_serving; int enable_tree_linenumbers; int enable_tree_group_dirs; - int enable_markdown; int enable_git_config; int enable_cache_list; int local_time; diff --git a/source/ui-summary.c b/source/ui-summary.c index 8dd191b..80d1e5b 100644 --- a/source/ui-summary.c +++ b/source/ui-summary.c @@ -105,17 +105,6 @@ static char* append_readme_path(const char *filename, const char *ref, const cha return full_path; } -static int readme_is_markdown(const char *filename) -{ - const char *ext = strrchr(filename, '.'); - - if (!ext || !ext[1]) - return 0; - ext++; - return !strcasecmp(ext, "md") || !strcasecmp(ext, "markdown") || - !strcasecmp(ext, "mkd") || !strcasecmp(ext, "mdown"); -} - void cgit_print_repo_readme(const char *path) { char *filename, *ref, *mimetype; @@ -146,29 +135,14 @@ void cgit_print_repo_readme(const char *path) } html("
    "); - if (!ctx.repo->about_filter && ctx.cfg.enable_markdown && - readme_is_markdown(filename)) { - /* No about-filter is set, so hand the markdown source to the - * built-in client-side renderer in cgit.js. The source is - * escaped here and rendered in the browser, and it degrades to - * readable plain text when scripting is off. - */ - html("
    "); - if (ref) { - cgit_print_file(filename, ref, 1, 1); - } else { - struct strbuf sb = STRBUF_INIT; - if (strbuf_read_file(&sb, filename, 0) >= 0) - html_txt(sb.buf); - strbuf_release(&sb); - } - html("
    "); - } else if (!ctx.repo->about_filter) { + if (!ctx.repo->about_filter) { /* No about-filter is configured, so there is nothing to turn * the readme source into safe HTML. Escape it rather than serve * repo content raw, which would let an untrusted repository * inject script into the about page. The pre keeps the line - * structure of the text, which bare escaped output loses. + * structure of the text, which bare escaped output loses. Point + * about-filter at the bundled about-render.lua to render a + * markdown or man readme instead, see cgitrc.5.txt. */ html("
    ");
     		if (ref) {
    diff --git a/tests/t0200-security.sh b/tests/t0200-security.sh
    index 83cdbfd..425708e 100644
    --- a/tests/t0200-security.sh
    +++ b/tests/t0200-security.sh
    @@ -73,7 +73,7 @@ test_expect_success 'a small blob is still served' '
     '
     
     # --- Readme rendering escapes untrusted repository content ------------------
    -test_expect_success 'markdown readme is escaped and marked for the client' '
    +test_expect_success 'markdown readme without a filter is escaped as plain text' '
     	{
     		echo "virtual-root=/" &&
     		echo "cache-size=0" &&
    @@ -82,7 +82,7 @@ test_expect_success 'markdown readme is escaped and marked for the client' '
     		echo "repo.readme=master:README.md"
     	} >secmdrc &&
     	CGIT_CONFIG="$PWD/secmdrc" QUERY_STRING="url=md/about/" cgit >tmp &&
    -	grep "data-markdown" tmp &&
    +	grep "pre class=.plaintext." tmp &&
     	grep "<script>" tmp &&
     	! grep "" tmp
     '
    -- 
    cgit v2.8.0