-- Server-side rendering of a repository's about page, named by the -- about-filter setting in cgitrc and run inside cgit's embedded Lua -- interpreter so a readme costs no extra process per request. -- Markdown, man pages and plain text are the three formats, chosen from the -- readme's file extension, and anything that fails to parse falls back to -- escaped plain text. Adding a format takes a render function and a row in -- the handler table at the foot of this file, and nothing else. The wrappers -- it emits carry the classes that assets/cgit.css styles. It runs on Lua 5.1 -- through 5.4 and LuaJIT. -- -- about-filter=lua:/usr/lib/cgit/filters/about-render.lua -- Markdown and man pages are parsed with lpeg grammars, so where the parsing -- module is missing both fall back to escaped plain text and only plain text -- still renders. It has to be built for the Lua cgit is linked against. -- cgit's syntax highlighter needs it too, so a cgit that colours source -- already has it. -- -- # Debian and Ubuntu -- sudo apt install lua-lpeg -- # Fedora -- sudo dnf install lua-lpeg -- # Alpine -- sudo apk add lua5.1-lpeg -- # or with LuaRocks, matched to your Lua version -- sudo luarocks --lua-version 5.1 install lpeg local has_lpeg, lpeg = pcall(require, "lpeg") -- Readmes larger than this are served as escaped plain text rather than -- parsed. local max_bytes = 512 * 1024 -- Ceiling on blockquote and emphasis nesting, so a hostile readme of stacked -- markers cannot drive the recursive parser into a stack overflow. local max_depth = 24 local filename, chunks = "", {} local function trim(text) return (text:gsub("^%s+", ""):gsub("%s+$", "")) end -- Split on newlines after normalising CRLF and CR, returning the lines -- without their terminators. A trailing newline yields a final empty line, -- which every caller treats as blank. local function split_lines(text) text = text:gsub("\r\n?", "\n") local lines, start = {}, 1 while true do local newline = text:find("\n", start, true) if not newline then lines[#lines + 1] = text:sub(start) return lines end lines[#lines + 1] = text:sub(start, newline - 1) start = newline + 1 end end -- Split a table row into trimmed cells, dropping one optional leading and one -- optional trailing pipe. local function split_cells(row) row = trim(row):gsub("^|", ""):gsub("|$", "") local cells = {} for cell in (row .. "|"):gmatch("(.-)|") do cells[#cells + 1] = trim(cell) end return cells end -- Return the URL if its scheme is safe, else nil. Control and whitespace -- bytes are stripped anywhere first because a browser ignores them when -- resolving the scheme, so "java\nscript:" must still be caught as -- javascript. The class is %c%s rather than a literal 0x00 range so it is -- safe on Lua 5.1, where an embedded zero byte ends a pattern. local function safe_url(url) url = url:gsub("[%c%s]", "") -- A leading "//" or "/\" is scheme-relative, which a browser -- resolves to another origin, so it is not the local target the -- parser meant to allow. if url:sub(1, 2) == "//" or url:sub(1, 2) == "/\\" then return nil end local scheme = url:match("^(%a[%w%+%.%-]*):") if scheme then scheme = scheme:lower() if scheme ~= "http" and scheme ~= "https" and scheme ~= "mailto" then return nil end end return url end -- The about page renders untrusted repository content, so the rule the two -- functions below keep is that every run of text reaches the page through -- cgit's own html_txt, every attribute value through html_attr, and every -- link or image target through safe_url before html_attr. Those three come -- from cgit's Lua filter host rather than from here, and the only thing -- passed to html is literal tag scaffolding. Because all output is routed -- through those sinks by construction, a bug in the parser can only -- mis-render, never inject markup or a URL with a javascript scheme. local emit_inline emit_inline = function(list) for _, node in ipairs(list) do local kind = node.kind if kind == "text" then html_txt(node.text) elseif kind == "code" then html(""); html_txt(node.text); html("") elseif kind == "strong" then html(""); emit_inline(node.kids); html("") elseif kind == "em" then html(""); emit_inline(node.kids); html("") elseif kind == "link" then local target = safe_url(node.url) if target then html("") emit_inline(node.kids) html("") else emit_inline(node.kids) end elseif kind == "image" then local target = safe_url(node.url) if target then html(""); html_attr(node.alt); html("") else html_txt("![" .. node.alt .. "]") end end end end local emit_blocks emit_blocks = function(blocks) for _, block in ipairs(blocks) do local kind = block.kind if kind == "heading" then html("") emit_inline(block.kids) html("") elseif kind == "hr" then html("
") elseif kind == "code" then html("
"); html_txt(block.text); html("
") elseif kind == "blockquote" then html("
") emit_blocks(block.blocks) html("
") elseif kind == "table" then html("") for _, cell in ipairs(block.head) do html("") end html("") for _, row in ipairs(block.rows) do html("") for _, cell in ipairs(row) do html("") end html("") end html("
"); emit_inline(cell); html("
"); emit_inline(cell); html("
") elseif kind == "list" then local tag = block.ordered and "ol" or "ul" html("<" .. tag .. ">") for _, item in ipairs(block.items) do html("
  • "); emit_inline(item); html("
  • ") end html("") elseif kind == "para" then html("

    ") for i, line in ipairs(block.lines) do if i > 1 then html("
    ") end emit_inline(line) end html("

    ") end end end local function render_plaintext(text) html("
    ")
    	html_txt(text)
    	html("
    ") end -- Markdown here is a deliberate subset rather than CommonMark, covering -- headings, thematic breaks, fenced and inline code, blockquotes, pipe -- tables, single-level lists, links, images and emphasis, and leaving out -- reference links, raw HTML passthrough, nested lists and setext headings. -- Emphasis does not span a hard line break inside a paragraph. Man rendering -- covers the common macros, that is section headings, filled and no-fill -- paragraphs, bold and italic and the font escapes, and drops the rest. Both -- are parsed with lpeg into the node trees emit_blocks and emit_inline above -- consume, and parsing only builds a tree, so a parse that fails falls back -- to plain text without leaving half a page behind. local render_markdown, render_man if has_lpeg then local P, R, S, C, Ct = lpeg.P, lpeg.R, lpeg.S, lpeg.C, lpeg.Ct local space = S(" \t") local whitespace = S(" \t\r\n\f\v") local eol = P(-1) local marker = S("`![*_") local url_char = 1 - S(")") - whitespace -- Bound the link and image inner scans. Without a cap a readme of -- unclosed brackets ("[[[[...") makes every position scan to the -- end of the line for a "]" that never comes, which is quadratic. -- Real link text and urls sit far under this, and anything longer -- simply renders as plain text. local max_scan = 512 local parse_inline local inline_depth = 0 local function node_text(text) return { kind = "text", text = text } end local function node_code(text) return { kind = "code", text = text } end local function node_strong(text) return { kind = "strong", kids = parse_inline(text) } end local function node_em(text) return { kind = "em", kids = parse_inline(text) } end local function node_link(text, url) return { kind = "link", kids = parse_inline(text), url = url } end local function node_image(alt, url) return { kind = "image", alt = alt, url = url } end local inline_grammar = Ct(( (P("`") * C((1 - P("`")) ^ 1) * P("`")) / node_code + (P("![") * C((1 - P("]")) ^ (-max_scan)) * P("]") * P("(") * space ^ 0 * C(url_char * url_char ^ (-max_scan)) * (1 - P(")")) ^ (-max_scan) * P(")")) / node_image + (P("[") * C((1 - P("]")) ^ (-max_scan)) * P("]") * P("(") * space ^ 0 * C(url_char * url_char ^ (-max_scan)) * (1 - P(")")) ^ (-max_scan) * P(")")) / node_link + (P("**") * C((1 - P("**")) ^ 1) * P("**")) / node_strong + (P("__") * C((1 - P("__")) ^ 1) * P("__")) / node_strong + (P("*") * C((1 - whitespace) * (1 - P("*")) ^ 0) * P("*")) / node_em + (P("_") * C((1 - whitespace) * (1 - P("_")) ^ 0) * P("_")) / node_em + C((1 - marker) ^ 1) / node_text + C(P(1)) / node_text ) ^ 0) parse_inline = function(s) if inline_depth >= max_depth then return { node_text(s) } end inline_depth = inline_depth + 1 local nodes = inline_grammar:match(s) or { node_text(s) } inline_depth = inline_depth - 1 return nodes end -- Line matchers. Each is anchored at the start of a single line and -- returns its captures, or nil when the line is not of that kind. local lang_char = R("az", "AZ", "09") + S("_.+#-") local heading_line = C(P("#") * P("#") ^ -5) * space ^ 1 * C(P(1) ^ 0) local function thematic(mark) return space ^ 0 * P(mark) * (space ^ 0 * P(mark)) ^ 2 * space ^ 0 * eol end local break_line = thematic("-") + thematic("*") + thematic("_") local fence_line = space ^ 0 * (C(P("`") ^ 3) + C(P("~") ^ 3)) * space ^ 0 * C(lang_char ^ 0) local close_backtick = space ^ 0 * P("`") ^ 3 * space ^ 0 * eol local close_tilde = space ^ 0 * P("~") ^ 3 * space ^ 0 * eol local quote_line = space ^ 0 * P(">") local bullet = S("-*+") local number = R("09") ^ 1 * S(".)") local item_line = space ^ 0 * C(bullet + number) * space ^ 1 * C(P(1) ^ 0) local dash_cell = space ^ 0 * P(":") ^ -1 * P("-") ^ 1 * P(":") ^ -1 * space ^ 0 -- A delimiter row is dash-cells joined by pipes. Require at least one -- pipe, a leading one or one between cells, so a bare rule of dashes -- stays a thematic break and an ordinary paragraph line is never taken -- for a table. local pipe = P("|") local table_delimiter = space ^ 0 * ( pipe * dash_cell * (pipe * dash_cell) ^ 0 * pipe ^ -1 + dash_cell * (pipe * dash_cell) ^ 1 * pipe ^ -1 ) * space ^ 0 * eol local block_start = space ^ 0 * ((P("#") * P("#") ^ -5 * space) + P(">") + P("`") ^ 3 + P("~") ^ 3 + ((bullet + number) * space)) local function ordered(mark) return mark:match("%d") ~= nil end local parse_blocks parse_blocks = function(lines, depth) local blocks, i, n = {}, 1, #lines while i <= n do local line = lines[i] local fence, lang = fence_line:match(line) local hashes, title = heading_line:match(line) local mark = item_line:match(line) if line:match("^%s*$") then i = i + 1 elseif fence then local close = (fence:sub(1, 1) == "`") and close_backtick or close_tilde local code = {} i = i + 1 while i <= n and not close:match(lines[i]) do code[#code + 1] = lines[i]; i = i + 1 end i = i + 1 local body = table.concat(code, "\n") if #code > 0 then body = body .. "\n" end blocks[#blocks + 1] = { kind = "code", lang = lang, text = body, } elseif hashes then blocks[#blocks + 1] = { kind = "heading", level = #hashes, kids = parse_inline((title:gsub("%s*#*%s*$", ""))), } i = i + 1 elseif break_line:match(line) then blocks[#blocks + 1] = { kind = "hr" } i = i + 1 elseif quote_line:match(line) then local inner = {} while i <= n and quote_line:match(lines[i]) do inner[#inner + 1] = (lines[i]:gsub("^%s*>%s?", "")) i = i + 1 end if depth < max_depth then blocks[#blocks + 1] = { kind = "blockquote", blocks = parse_blocks(inner, depth + 1), } else local paragraph = {} for _, quoted in ipairs(inner) do paragraph[#paragraph + 1] = parse_inline(quoted) end blocks[#blocks + 1] = { kind = "para", lines = paragraph, } end elseif i + 1 <= n and line:find("|", 1, true) and table_delimiter:match(lines[i + 1]) then local head = {} for _, cell in ipairs(split_cells(line)) do head[#head + 1] = parse_inline(cell) end local rows = {} i = i + 2 while i <= n and lines[i]:find("|", 1, true) and not lines[i]:match("^%s*$") do local row = {} for _, cell in ipairs(split_cells(lines[i])) do row[#row + 1] = parse_inline(cell) end rows[#rows + 1] = row i = i + 1 end blocks[#blocks + 1] = { kind = "table", head = head, rows = rows, } elseif mark then local is_ordered = ordered(mark) local items = {} while i <= n do local item_mark, content = item_line:match(lines[i]) if not item_mark or ordered(item_mark) ~= is_ordered then break end items[#items + 1] = parse_inline(content) i = i + 1 end blocks[#blocks + 1] = { kind = "list", ordered = is_ordered, items = items, } else local paragraph = { parse_inline(line) } i = i + 1 while i <= n and not lines[i]:match("^%s*$") and not block_start:match(lines[i]) do paragraph[#paragraph + 1] = parse_inline(lines[i]) i = i + 1 end blocks[#blocks + 1] = { kind = "para", lines = paragraph } end end return blocks end render_markdown = function(text) if #text > max_bytes then return render_plaintext(text) end local ok, blocks = pcall(parse_blocks, split_lines(text), 0) if not ok then return render_plaintext(text) end html("
    ") -- Emitting is wrapped so a bug there can at worst -- truncate the page, never raise and let the outer -- fallback duplicate what has already been sent. pcall(emit_blocks, blocks) html("
    ") end -- Macro lines are matched with lpeg, and the roff inline font and -- character escapes are a grammar producing a flat token list that -- a fold turns into the same inline nodes markdown emits. Bold and -- italic are the current font, which carries across a run rather -- than being opened and closed, so the inline pass tokenises and -- folds instead of nesting the way markdown does. local macro_line = S(".'") * space ^ 0 * C((1 - space) ^ 1) * space ^ 0 * C(P(1) ^ 0) local one_font = { B = "B", I = "I" } local two_font = { CB = "B", BI = "B", CI = "I" } local man_chars = { aq = "'", cq = "'", oq = "'", dq = '"', lq = '"', rq = '"', hy = "-", en = "-", em = "-", } local backslash = P("\\") local function font_token(name) return { kind = "font", font = name } end local function text_token(text) return { kind = "text", text = text } end local skip_token = { kind = "skip" } local man_inline_grammar = Ct(( (backslash * P("f") * P("(") * C(P(1) * P(1))) / function(name) return font_token(two_font[name] or "R") end + (backslash * P("f") * P("[") * C((1 - P("]")) ^ 0) * P("]")) / function(name) return font_token(one_font[name] or "R") end + (backslash * P("f") * C(P(1))) / function(name) return font_token(one_font[name] or "R") end + (backslash * P("-")) / function() return text_token("-") end + (backslash * P("e")) / function() return text_token("\\") end + (backslash * P(" ")) / function() return text_token(" ") end + (backslash * S("&|^")) / function() return skip_token end + (backslash * P("(") * C(P(1) * P(1))) / function(name) return text_token(man_chars[name] or "") end + (backslash * P("*") * (P("(") * P(1) * P(1) + P(1))) / function() return skip_token end + (backslash * C(P(1))) / text_token + backslash / function() return skip_token end + C((1 - backslash) ^ 1) / text_token ) ^ 0) local function man_inline(text) local tokens = man_inline_grammar:match(text) or {} local nodes, parts, font = {}, {}, "R" local function flush() if #parts == 0 then return end local run = table.concat(parts); parts = {} if run == "" then return end if font == "B" then nodes[#nodes + 1] = { kind = "strong", kids = { { kind = "text", text = run } }, } elseif font == "I" then nodes[#nodes + 1] = { kind = "em", kids = { { kind = "text", text = run } }, } else nodes[#nodes + 1] = { kind = "text", text = run } end end for _, token in ipairs(tokens) do if token.kind == "text" then parts[#parts + 1] = token.text elseif token.kind == "font" then flush(); font = token.font end end flush() return nodes end local function parse_man(text) local lines = split_lines(text) local blocks = {} local paragraph, nofill_lines, nofill = nil, nil, false local function flush_paragraph() if paragraph and #paragraph > 0 then blocks[#blocks + 1] = { kind = "para", lines = paragraph } end paragraph = nil end local function flush_nofill() if nofill_lines then local body = table.concat(nofill_lines, "\n") if #nofill_lines > 0 then body = body .. "\n" end blocks[#blocks + 1] = { kind = "code", text = body } nofill_lines = nil end end local function add_line(nodes) paragraph = paragraph or {} paragraph[#paragraph + 1] = nodes end for _, line in ipairs(lines) do local macro, rest = macro_line:match(line) if rest then rest = (rest:gsub('^"(.*)"$', "%1")) end if nofill then if macro == "fi" then nofill = false; flush_nofill() else nofill_lines[#nofill_lines + 1] = line end elseif line:match("^%s*$") then flush_paragraph() elseif macro and macro:sub(1, 1) == "\\" then -- A roff comment, dropped. elseif macro == "SH" or macro == "SS" then flush_paragraph() blocks[#blocks + 1] = { kind = "heading", level = (macro == "SH") and 2 or 3, kids = man_inline(rest), } elseif macro == "nf" then flush_paragraph(); nofill = true; nofill_lines = {} elseif macro == "fi" then flush_nofill() elseif macro == "B" or macro == "BR" or macro == "RB" or macro == "BI" or macro == "IB" then add_line({ { kind = "strong", kids = man_inline(rest) } }) elseif macro == "I" or macro == "IR" or macro == "RI" then add_line({ { kind = "em", kids = man_inline(rest) } }) elseif macro == "TH" or macro == "PP" or macro == "LP" or macro == "P" or macro == "TP" or macro == "IP" or macro == "HP" or macro == "RS" or macro == "RE" or macro == "sp" or macro == "br" then flush_paragraph() elseif macro then if rest ~= "" then add_line(man_inline(rest)) end else add_line(man_inline(line)) end end flush_paragraph(); flush_nofill() return blocks end render_man = function(text) if #text > max_bytes then return render_plaintext(text) end local ok, blocks = pcall(parse_man, text) if not ok then return render_plaintext(text) end html("
    ") pcall(emit_blocks, blocks) html("
    ") end else -- No lpeg, so markdown and man readmes read as escaped source, not -- nothing. render_markdown = render_plaintext render_man = render_plaintext end local handlers = {} for _, ext in ipairs({ "md", "markdown", "mkd", "mdown" }) do handlers[ext] = render_markdown end for _, ext in ipairs({ "1", "2", "3", "4", "5", "6", "7", "8", "9", "man" }) do handlers[ext] = render_man end function filter_open(name) filename = name or "" chunks = {} end function filter_write(str) chunks[#chunks + 1] = str end -- cgit takes filter output through a C string sink that stops at the first -- NUL byte, so a readme holding one is truncated there. function filter_close() local text = table.concat(chunks) chunks = {} local ext = (filename:match("%.([^.]+)$") or ""):lower() local render = handlers[ext] or render_plaintext local ok = pcall(render, text) if not ok then render_plaintext(text) end return 0 end