diff options
context:
space:
mode:
authorBryce Kwon <bryce@brycekwon.com>
committerBryce Kwon <bryce@brycekwon.com>
commit
parent
tree
download
Replace the browser markdown renderer with a filter
The readme is now escaped plain text unless `about-filter` points at the new `about-render.lua`, which renders markdown, man pages and plain text server-side. `enable-markdown` goes away with the renderer.
Diffstat (limited to 'extensions/about-render.lua')
-rw-r--r--extensions/about-render.lua592
1 file changed, 592 insertions, 0 deletions
diff --git a/extensions/about-render.lua b/extensions/about-render.lua
new file mode 100644
index 0000000..073b531
--- /dev/null
+++ b/extensions/about-render.lua
@@ -0,0 +1,592 @@
+-- Server-side about-page rendering for cgit, used with the about-filter setting
+-- in cgitrc and the lua: prefix so it runs in cgit's embedded interpreter with
+-- no per-request process.
+--
+-- about-filter=lua:/usr/lib/cgit/filters/about-render.lua
+--
+-- One filter renders the three baseline about formats, chosen by the readme's
+-- file extension.
+--
+-- markdown .md .markdown .mkd .mdown
+-- man page .1 through .9 and .man
+-- plain text everything else, and the fallback for anything that fails
+--
+-- Add a format by giving it a render function and a row in the handler table
+-- near the foot of this file. Nothing else needs to change.
+--
+-- SUPPORTED LUA
+--
+-- Lua 5.1 through 5.4 and LuaJIT.
+--
+-- REQUIREMENTS
+--
+-- lpeg, the parsing module, for the Lua cgit is linked against. Markdown and
+-- man pages are parsed with lpeg grammars, so without lpeg both fall back to
+-- escaped plain text. It ships with cgit's syntax highlighter too, so a cgit
+-- that colours source already has it.
+--
+-- # Debian and Ubuntu
+-- sudo apt install lua-lpeg
+-- # Fedora
+-- sudo dnf install lua-lpeg
+-- # Alpine
+-- sudo apk add lua5.1-lpeg
+-- # or with LuaRocks, matched to your Lua version
+-- sudo luarocks --lua-version 5.1 install lpeg
+--
+-- Plain text needs only Lua itself.
+--
+-- SECURITY
+--
+-- The about page renders untrusted repository content, so the safety rule is
+-- that every run of text reaches the page through cgit's own html_txt, every
+-- attribute value through html_attr, and every link or image target through
+-- the safe_url scheme allowlist below before html_attr. Only http, https,
+-- mailto and relative targets survive, so a hostile readme cannot inject markup
+-- or a javascript: url. Because all output is routed through those sinks by
+-- construction, a bug in the parser can only mis-render, never inject.
+--
+-- LIMITATIONS
+--
+-- Markdown is a deliberate subset, not CommonMark. Headings, thematic breaks,
+-- fenced and inline code, blockquotes, pipe tables, single-level lists, links,
+-- images and emphasis. No reference links, no raw HTML passthrough, no nested
+-- lists, no setext headings. Emphasis does not span a hard line break inside a
+-- paragraph. Man rendering covers the common macros (section headings, filled
+-- and no-fill paragraphs, bold and italic, font escapes) and drops the rest.
+-- cgit sends filter output through a C string sink that stops at the first NUL
+-- byte, so content past a NUL is truncated.
+--
+-- OUTPUT
+--
+-- Markdown is wrapped in <div class='markdown'>, man pages in
+-- <div class='markdown manpage'>, plain text in <pre class='plaintext'>, all
+-- styled by assets/cgit.css.
+
+local ok_lpeg, lpeg = pcall(require, "lpeg")
+
+
+-- ===== configuration ===================================================
+
+-- Readmes larger than this are served as escaped plain text rather than parsed.
+local max_bytes = 512 * 1024
+
+-- Ceiling on blockquote and emphasis nesting, so a hostile readme of stacked
+-- markers cannot drive the recursive parser into a stack overflow.
+local max_depth = 24
+
+local filename, chunks = "", {}
+
+
+-- ===== helpers =========================================================
+
+local function trim(s)
+ return (s:gsub("^%s+", ""):gsub("%s+$", ""))
+end
+
+-- Split on newlines after normalising CRLF and CR, returning the lines without
+-- their terminators. A trailing newline yields a final empty line, which every
+-- caller treats as blank.
+local function split_lines(s)
+ s = s:gsub("\r\n?", "\n")
+ local lines, start = {}, 1
+ while true do
+ local nl = s:find("\n", start, true)
+ if not nl then
+ lines[#lines + 1] = s:sub(start)
+ return lines
+ end
+ lines[#lines + 1] = s:sub(start, nl - 1)
+ start = nl + 1
+ end
+end
+
+-- Split a table row into trimmed cells, dropping one optional leading and
+-- trailing pipe.
+local function split_cells(row)
+ row = trim(row):gsub("^|", ""):gsub("|$", "")
+ local cells = {}
+ for c in (row .. "|"):gmatch("(.-)|") do
+ cells[#cells + 1] = trim(c)
+ end
+ return cells
+end
+
+-- Return the url if its scheme is safe, else nil. Control and whitespace bytes
+-- are stripped anywhere first because a browser ignores them when resolving the
+-- scheme, so "java\nscript:" must still be caught as javascript. The class is
+-- %c%s rather than a literal 0x00 range so it is safe on Lua 5.1, where an
+-- embedded zero byte ends a pattern.
+local function safe_url(url)
+ url = url:gsub("[%c%s]", "")
+ -- A leading "//" or "/\" is scheme-relative, which a browser resolves to
+ -- another origin, so it is not the local target the parser meant to allow.
+ if url:sub(1, 2) == "//" or url:sub(1, 2) == "/\\" then
+ return nil
+ end
+ local scheme = url:match("^(%a[%w%+%.%-]*):")
+ if scheme then
+ scheme = scheme:lower()
+ if scheme ~= "http" and scheme ~= "https" and scheme ~= "mailto" then
+ return nil
+ end
+ end
+ return url
+end
+
+
+-- ===== emit ============================================================
+-- html, html_txt and html_attr are injected by cgit's lua filter host. Tag
+-- scaffolding is a literal argument to html; every value that came from the
+-- repository goes through html_txt, html_attr or safe_url.
+
+local emit_inline
+
+emit_inline = function(list)
+ for _, nd in ipairs(list) do
+ local t = nd.t
+ if t == "text" then
+ html_txt(nd.v)
+ elseif t == "code" then
+ html("<code>"); html_txt(nd.v); html("</code>")
+ elseif t == "strong" then
+ html("<strong>"); emit_inline(nd.kids); html("</strong>")
+ elseif t == "em" then
+ html("<em>"); emit_inline(nd.kids); html("</em>")
+ elseif t == "link" then
+ local u = safe_url(nd.url)
+ if u then
+ html("<a href='"); html_attr(u); html("'>")
+ emit_inline(nd.kids)
+ html("</a>")
+ else
+ emit_inline(nd.kids)
+ end
+ elseif t == "image" then
+ local u = safe_url(nd.url)
+ if u then
+ html("<img src='"); html_attr(u)
+ html("' alt='"); html_attr(nd.alt); html("'/>")
+ else
+ html_txt("![" .. nd.alt .. "]")
+ end
+ end
+ end
+end
+
+local emit_blocks
+
+emit_blocks = function(blocks)
+ for _, b in ipairs(blocks) do
+ local t = b.t
+ if t == "heading" then
+ html("<h" .. b.level .. ">")
+ emit_inline(b.kids)
+ html("</h" .. b.level .. ">")
+ elseif t == "hr" then
+ html("<hr/>")
+ elseif t == "code" then
+ html("<pre><code")
+ if b.lang and b.lang ~= "" then
+ html(" data-lang='"); html_attr(b.lang); html("'")
+ end
+ html(">"); html_txt(b.text); html("</code></pre>")
+ elseif t == "blockquote" then
+ html("<blockquote>"); emit_blocks(b.blocks); html("</blockquote>")
+ elseif t == "table" then
+ html("<table><thead><tr>")
+ for _, c in ipairs(b.head) do
+ html("<th>"); emit_inline(c); html("</th>")
+ end
+ html("</tr></thead><tbody>")
+ for _, row in ipairs(b.rows) do
+ html("<tr>")
+ for _, c in ipairs(row) do
+ html("<td>"); emit_inline(c); html("</td>")
+ end
+ html("</tr>")
+ end
+ html("</tbody></table>")
+ elseif t == "list" then
+ local tag = b.ordered and "ol" or "ul"
+ html("<" .. tag .. ">")
+ for _, item in ipairs(b.items) do
+ html("<li>"); emit_inline(item); html("</li>")
+ end
+ html("</" .. tag .. ">")
+ elseif t == "para" then
+ html("<p>")
+ for k, line in ipairs(b.lines) do
+ if k > 1 then html("<br/>") end
+ emit_inline(line)
+ end
+ html("</p>")
+ end
+ end
+end
+
+
+-- ===== plain text ======================================================
+
+local function render_plaintext(text)
+ html("<pre class='plaintext'>")
+ html_txt(text)
+ html("</pre>")
+end
+
+
+-- ===== markdown and man ================================================
+-- Both are parsed with lpeg into the block/inline node trees emit_blocks and
+-- emit_inline above consume. Parsing is pure and builds a tree; emit is the
+-- only step that writes output, so a parse failure falls back to plain text
+-- without leaving half a page behind.
+
+local render_markdown, render_man
+
+if ok_lpeg then
+ local P, R, S, C, Ct = lpeg.P, lpeg.R, lpeg.S, lpeg.C, lpeg.Ct
+
+ local sp = S(" \t")
+ local ws = S(" \t\r\n\f\v")
+ local eol = P(-1)
+
+ -- ----- markdown inline -----
+ local marker = S("`![*_")
+ local urlchar = 1 - S(")") - ws
+ -- Bound the link and image inner scans. Without a cap a readme of unclosed
+ -- brackets ("[[[[...") makes each position scan to end of line for a "]"
+ -- that never comes, which is quadratic. Real link text and urls sit far
+ -- under this, and anything longer simply renders as plain text.
+ local CAP = 512
+
+ local parse_inline
+ local inline_depth = 0
+
+ local function node_text(s) return { t = "text", v = s } end
+ local function node_code(s) return { t = "code", v = s } end
+ local function node_strong(s) return { t = "strong", kids = parse_inline(s) } end
+ local function node_em(s) return { t = "em", kids = parse_inline(s) } end
+ local function node_link(text, url) return { t = "link", kids = parse_inline(text), url = url } end
+ local function node_image(alt, url) return { t = "image", alt = alt, url = url } end
+
+ local inline_grammar = Ct((
+ (P("`") * C((1 - P("`")) ^ 1) * P("`")) / node_code
+ + (P("![") * C((1 - P("]")) ^ (-CAP)) * P("]") * P("(") * sp ^ 0
+ * C(urlchar * urlchar ^ (-CAP)) * (1 - P(")")) ^ (-CAP) * P(")")) / node_image
+ + (P("[") * C((1 - P("]")) ^ (-CAP)) * P("]") * P("(") * sp ^ 0
+ * C(urlchar * urlchar ^ (-CAP)) * (1 - P(")")) ^ (-CAP) * P(")")) / node_link
+ + (P("**") * C((1 - P("**")) ^ 1) * P("**")) / node_strong
+ + (P("__") * C((1 - P("__")) ^ 1) * P("__")) / node_strong
+ + (P("*") * C((1 - ws) * (1 - P("*")) ^ 0) * P("*")) / node_em
+ + (P("_") * C((1 - ws) * (1 - P("_")) ^ 0) * P("_")) / node_em
+ + C((1 - marker) ^ 1) / node_text
+ + C(P(1)) / node_text
+ ) ^ 0)
+
+ parse_inline = function(s)
+ if inline_depth >= max_depth then
+ return { node_text(s) }
+ end
+ inline_depth = inline_depth + 1
+ local nodes = inline_grammar:match(s) or { node_text(s) }
+ inline_depth = inline_depth - 1
+ return nodes
+ end
+
+ -- ----- markdown blocks -----
+ -- Line matchers. Each is anchored at the start of a single line and returns
+ -- its captures, or nil when the line is not of that kind.
+ local langchar = R("az", "AZ", "09") + S("_.+#-")
+ local m_heading = C(P("#") * P("#") ^ -5) * sp ^ 1 * C(P(1) ^ 0)
+ local function rule(c) return sp ^ 0 * P(c) * (sp ^ 0 * P(c)) ^ 2 * sp ^ 0 * eol end
+ local m_hr = rule("-") + rule("*") + rule("_")
+ local m_fence = sp ^ 0 * (C(P("`") ^ 3) + C(P("~") ^ 3)) * sp ^ 0 * C(langchar ^ 0)
+ local m_close_bt = sp ^ 0 * P("`") ^ 3 * sp ^ 0 * eol
+ local m_close_ti = sp ^ 0 * P("~") ^ 3 * sp ^ 0 * eol
+ local m_bq = sp ^ 0 * P(">")
+ local bullet = S("-*+")
+ local number = R("09") ^ 1 * S(".)")
+ local m_item = sp ^ 0 * C(bullet + number) * sp ^ 1 * C(P(1) ^ 0)
+ local dcell = sp ^ 0 * P(":") ^ -1 * P("-") ^ 1 * P(":") ^ -1 * sp ^ 0
+ -- A delimiter row is dash-cells joined by pipes. Require at least one pipe,
+ -- a leading one or one between cells, so a bare rule of dashes stays a
+ -- thematic break and an ordinary paragraph line is never taken for a table.
+ local pipe = P("|")
+ local m_tdelim = sp ^ 0 * (
+ pipe * dcell * (pipe * dcell) ^ 0 * pipe ^ -1
+ + dcell * (pipe * dcell) ^ 1 * pipe ^ -1
+ ) * sp ^ 0 * eol
+ local m_blockstart = sp ^ 0 * ((P("#") * P("#") ^ -5 * sp) + P(">")
+ + P("`") ^ 3 + P("~") ^ 3 + ((bullet + number) * sp))
+
+ local function ordered(mark) return mark:match("%d") ~= nil end
+
+ local parse_blocks
+
+ parse_blocks = function(lines, depth)
+ local blocks, i, n = {}, 1, #lines
+ while i <= n do
+ local line = lines[i]
+ local fence, lang = m_fence:match(line)
+ local hashes, htext = m_heading:match(line)
+ local mark = m_item:match(line)
+ if line:match("^%s*$") then
+ i = i + 1
+ elseif fence then
+ local close = (fence:sub(1, 1) == "`") and m_close_bt or m_close_ti
+ local code = {}
+ i = i + 1
+ while i <= n and not close:match(lines[i]) do
+ code[#code + 1] = lines[i]; i = i + 1
+ end
+ i = i + 1
+ local body = table.concat(code, "\n")
+ if #code > 0 then body = body .. "\n" end
+ blocks[#blocks + 1] = { t = "code", lang = lang, text = body }
+ elseif hashes then
+ blocks[#blocks + 1] = {
+ t = "heading",
+ level = #hashes,
+ kids = parse_inline((htext:gsub("%s*#*%s*$", ""))),
+ }
+ i = i + 1
+ elseif m_hr:match(line) then
+ blocks[#blocks + 1] = { t = "hr" }
+ i = i + 1
+ elseif m_bq:match(line) then
+ local inner = {}
+ while i <= n and m_bq:match(lines[i]) do
+ inner[#inner + 1] = (lines[i]:gsub("^%s*>%s?", ""))
+ i = i + 1
+ end
+ if depth < max_depth then
+ blocks[#blocks + 1] = { t = "blockquote", blocks = parse_blocks(inner, depth + 1) }
+ else
+ local para = {}
+ for _, l in ipairs(inner) do para[#para + 1] = parse_inline(l) end
+ blocks[#blocks + 1] = { t = "para", lines = para }
+ end
+ elseif i + 1 <= n and line:find("|", 1, true) and m_tdelim:match(lines[i + 1]) then
+ local head = {}
+ for _, c in ipairs(split_cells(line)) do head[#head + 1] = parse_inline(c) end
+ local rows = {}
+ i = i + 2
+ while i <= n and lines[i]:find("|", 1, true) and not lines[i]:match("^%s*$") do
+ local row = {}
+ for _, c in ipairs(split_cells(lines[i])) do row[#row + 1] = parse_inline(c) end
+ rows[#rows + 1] = row
+ i = i + 1
+ end
+ blocks[#blocks + 1] = { t = "table", head = head, rows = rows }
+ elseif mark then
+ local is_ol = ordered(mark)
+ local items = {}
+ while i <= n do
+ local mk, ct = m_item:match(lines[i])
+ if not mk or ordered(mk) ~= is_ol then break end
+ items[#items + 1] = parse_inline(ct)
+ i = i + 1
+ end
+ blocks[#blocks + 1] = { t = "list", ordered = is_ol, items = items }
+ else
+ local para = { parse_inline(line) }
+ i = i + 1
+ while i <= n and not lines[i]:match("^%s*$") and not m_blockstart:match(lines[i]) do
+ para[#para + 1] = parse_inline(lines[i]); i = i + 1
+ end
+ blocks[#blocks + 1] = { t = "para", lines = para }
+ end
+ end
+ return blocks
+ end
+
+ render_markdown = function(text)
+ if #text > max_bytes then
+ return render_plaintext(text)
+ end
+ local ok, blocks = pcall(parse_blocks, split_lines(text), 0)
+ if not ok then
+ return render_plaintext(text)
+ end
+ html("<div class='markdown'>")
+ -- Emit is wrapped so a parser bug can at worst truncate the page, never
+ -- error out and let the outer fallback duplicate what was already sent.
+ pcall(emit_blocks, blocks)
+ html("</div>")
+ end
+
+ -- ----- man pages -----
+ -- Macro lines are matched with lpeg, and the roff inline font and character
+ -- escapes are an lpeg grammar producing a flat token list that a fold turns
+ -- into the same inline nodes markdown emits. Bold and italic are the current
+ -- font, which carries across a run, so the inline pass tokenises and folds
+ -- rather than nesting the way markdown does.
+ local m_macro = S(".'") * sp ^ 0 * C((1 - sp) ^ 1) * sp ^ 0 * C(P(1) ^ 0)
+
+ local one_font = { B = "B", I = "I" }
+ local two_font = { CB = "B", BI = "B", CI = "I" }
+ local man_chars = {
+ aq = "'", cq = "'", oq = "'", dq = '"', lq = '"', rq = '"',
+ hy = "-", en = "-", em = "-",
+ }
+ local esc = P("\\")
+ local function font_tok(f) return { k = "font", f = f } end
+ local function text_tok(v) return { k = "text", v = v } end
+ local skip_tok = { k = "skip" }
+
+ local man_inline_grammar = Ct((
+ (esc * P("f") * P("(") * C(P(1) * P(1))) / function(nm) return font_tok(two_font[nm] or "R") end
+ + (esc * P("f") * P("[") * C((1 - P("]")) ^ 0) * P("]")) / function(nm) return font_tok(one_font[nm] or "R") end
+ + (esc * P("f") * C(P(1))) / function(f) return font_tok(one_font[f] or "R") end
+ + (esc * P("-")) / function() return text_tok("-") end
+ + (esc * P("e")) / function() return text_tok("\\") end
+ + (esc * P(" ")) / function() return text_tok(" ") end
+ + (esc * S("&|^")) / function() return skip_tok end
+ + (esc * P("(") * C(P(1) * P(1))) / function(nm) return text_tok(man_chars[nm] or "") end
+ + (esc * P("*") * (P("(") * P(1) * P(1) + P(1))) / function() return skip_tok end
+ + (esc * C(P(1))) / text_tok
+ + esc / function() return skip_tok end
+ + C((1 - esc) ^ 1) / text_tok
+ ) ^ 0)
+
+ local function man_inline(s)
+ local toks = man_inline_grammar:match(s) or {}
+ local nodes, buf, font = {}, {}, "R"
+ local function flush()
+ if #buf == 0 then return end
+ local txt = table.concat(buf); buf = {}
+ if txt == "" then return end
+ if font == "B" then
+ nodes[#nodes + 1] = { t = "strong", kids = { { t = "text", v = txt } } }
+ elseif font == "I" then
+ nodes[#nodes + 1] = { t = "em", kids = { { t = "text", v = txt } } }
+ else
+ nodes[#nodes + 1] = { t = "text", v = txt }
+ end
+ end
+ for _, tk in ipairs(toks) do
+ if tk.k == "text" then
+ buf[#buf + 1] = tk.v
+ elseif tk.k == "font" then
+ flush(); font = tk.f
+ end
+ end
+ flush()
+ return nodes
+ end
+
+ local function parse_man(text)
+ local lines = split_lines(text)
+ local blocks = {}
+ local para, pre, nofill = nil, nil, false
+ local function flush_para()
+ if para and #para > 0 then
+ blocks[#blocks + 1] = { t = "para", lines = para }
+ end
+ para = nil
+ end
+ local function flush_pre()
+ if pre then
+ local body = table.concat(pre, "\n")
+ if #pre > 0 then body = body .. "\n" end
+ blocks[#blocks + 1] = { t = "code", text = body }
+ pre = nil
+ end
+ end
+ local function add_line(nodes)
+ para = para or {}
+ para[#para + 1] = nodes
+ end
+ for _, line in ipairs(lines) do
+ local macro, rest = m_macro:match(line)
+ if rest then rest = (rest:gsub('^"(.*)"$', "%1")) end
+ if nofill then
+ if macro == "fi" then
+ nofill = false; flush_pre()
+ else
+ pre[#pre + 1] = line
+ end
+ elseif line:match("^%s*$") then
+ flush_para()
+ elseif macro and macro:sub(1, 1) == "\\" then
+ -- roff comment, dropped
+ elseif macro == "SH" or macro == "SS" then
+ flush_para()
+ blocks[#blocks + 1] = {
+ t = "heading",
+ level = (macro == "SH") and 2 or 3,
+ kids = man_inline(rest),
+ }
+ elseif macro == "nf" then
+ flush_para(); nofill = true; pre = {}
+ elseif macro == "fi" then
+ flush_pre()
+ elseif macro == "B" or macro == "BR" or macro == "RB"
+ or macro == "BI" or macro == "IB" then
+ add_line({ { t = "strong", kids = man_inline(rest) } })
+ elseif macro == "I" or macro == "IR" or macro == "RI" then
+ add_line({ { t = "em", kids = man_inline(rest) } })
+ elseif macro == "TH" or macro == "PP" or macro == "LP"
+ or macro == "P" or macro == "TP" or macro == "IP"
+ or macro == "HP" or macro == "RS" or macro == "RE"
+ or macro == "sp" or macro == "br" then
+ flush_para()
+ elseif macro then
+ if rest ~= "" then add_line(man_inline(rest)) end
+ else
+ add_line(man_inline(line))
+ end
+ end
+ flush_para(); flush_pre()
+ return blocks
+ end
+
+ render_man = function(text)
+ if #text > max_bytes then
+ return render_plaintext(text)
+ end
+ local ok, blocks = pcall(parse_man, text)
+ if not ok then
+ return render_plaintext(text)
+ end
+ html("<div class='markdown manpage'>")
+ pcall(emit_blocks, blocks)
+ html("</div>")
+ end
+else
+ -- No lpeg, so markdown and man readmes read as escaped source, not nothing.
+ render_markdown = render_plaintext
+ render_man = render_plaintext
+end
+
+
+-- ===== dispatch ========================================================
+
+local handlers = {}
+for _, ext in ipairs({ "md", "markdown", "mkd", "mdown" }) do
+ handlers[ext] = render_markdown
+end
+for _, ext in ipairs({ "1", "2", "3", "4", "5", "6", "7", "8", "9", "man" }) do
+ handlers[ext] = render_man
+end
+
+function filter_open(name)
+ filename = name or ""
+ chunks = {}
+end
+
+function filter_write(str)
+ chunks[#chunks + 1] = str
+end
+
+function filter_close()
+ local text = table.concat(chunks)
+ chunks = {}
+ local ext = (filename:match("%.([^.]+)$") or ""):lower()
+ local render = handlers[ext] or render_plaintext
+ local ok = pcall(render, text)
+ if not ok then
+ render_plaintext(text)
+ end
+ return 0
+end