diff options
context:
space:
mode:
Diffstat (limited to 'custom/extensions/about-render.lua')
-rw-r--r--custom/extensions/about-render.lua91
1 file changed, 37 insertions, 54 deletions
diff --git a/custom/extensions/about-render.lua b/custom/extensions/about-render.lua
index ea13746..f4e7685 100644
--- a/custom/extensions/about-render.lua
+++ b/custom/extensions/about-render.lua
@@ -1,20 +1,15 @@
-- Server-side rendering of a repository's about page, named by the
-- about-filter setting in cgitrc and run inside cgit's embedded Lua
--- interpreter so a readme costs no extra process per request.
--- Markdown, man pages and plain text are the three formats, chosen from the
--- readme's file extension, and anything that fails to parse falls back to
--- escaped plain text. Adding a format takes a render function and a row in
--- the handler table at the foot of this file, and nothing else. The wrappers
--- it emits carry the classes that assets/cgit.css styles. It runs on Lua 5.1
--- through 5.4 and LuaJIT.
+-- interpreter so a readme costs no extra process per request. Markdown, man
+-- pages and plain text are the three formats, chosen from the readme's file
+-- extension, and anything that fails to parse falls back to escaped plain
+-- text. The wrappers it emits carry the classes that assets/cgit.css styles.
+-- It runs on Lua 5.1 through 5.4 and LuaJIT.
--
--- about-filter=lua:/usr/lib/cgit/filters/about-render.lua
+-- about-filter=lua:/usr/local/lib/cgit/filters/about-render.lua
--- Markdown and man pages are parsed with lpeg grammars, so where the parsing
--- module is missing both fall back to escaped plain text and only plain text
--- still renders. It has to be built for the Lua cgit is linked against.
--- cgit's syntax highlighter needs it too, so a cgit that colours source
--- already has it.
+-- Markdown and man pages need lpeg, built for the Lua cgit is linked
+-- against. Without it both fall back to escaped plain text.
--
-- # Debian and Ubuntu
-- sudo apt install lua-lpeg
@@ -42,9 +37,8 @@ local function trim(text)
return (text:gsub("^%s+", ""):gsub("%s+$", ""))
end
--- Split on newlines after normalising CRLF and CR, returning the lines
--- without their terminators. A trailing newline yields a final empty line,
--- which every caller treats as blank.
+-- Normalise CRLF and CR to newlines, then split. A trailing newline yields
+-- a final empty line, which every caller treats as blank.
local function split_lines(text)
text = text:gsub("\r\n?", "\n")
local lines, start = {}, 1
@@ -59,8 +53,6 @@ local function split_lines(text)
end
end
--- Split a table row into trimmed cells, dropping one optional leading and one
--- optional trailing pipe.
local function split_cells(row)
row = trim(row):gsub("^|", ""):gsub("|$", "")
local cells = {}
@@ -70,11 +62,10 @@ local function split_cells(row)
return cells
end
--- Return the URL if its scheme is safe, else nil. Control and whitespace
--- bytes are stripped anywhere first because a browser ignores them when
--- resolving the scheme, so "java\nscript:" must still be caught as
--- javascript. The class is %c%s rather than a literal 0x00 range so it is
--- safe on Lua 5.1, where an embedded zero byte ends a pattern.
+-- Control and whitespace bytes are stripped before the scheme check because
+-- a browser ignores them when resolving it, so "java\nscript:" must still be
+-- caught as javascript. The class is %c%s rather than a literal 0x00 range
+-- so it is safe on Lua 5.1, where an embedded zero byte ends a pattern.
local function safe_url(url)
url = url:gsub("[%c%s]", "")
-- A leading "//" or "/\" is scheme-relative, which a browser
@@ -94,14 +85,10 @@ local function safe_url(url)
end
--- The about page renders untrusted repository content, so the rule the two
--- functions below keep is that every run of text reaches the page through
--- cgit's own html_txt, every attribute value through html_attr, and every
--- link or image target through safe_url before html_attr. Those three come
--- from cgit's Lua filter host rather than from here, and the only thing
--- passed to html is literal tag scaffolding. Because all output is routed
--- through those sinks by construction, a bug in the parser can only
--- mis-render, never inject markup or a URL with a javascript scheme.
+-- The page renders untrusted repository content. Every run of text goes out
+-- through cgit's html_txt, every attribute value through html_attr, and
+-- every link or image target through safe_url before html_attr. html() only
+-- ever gets literal tag scaffolding.
local emit_inline
@@ -198,16 +185,12 @@ local function render_plaintext(text)
end
--- Markdown here is a deliberate subset rather than CommonMark, covering
--- headings, thematic breaks, fenced and inline code, blockquotes, pipe
--- tables, single-level lists, links, images and emphasis, and leaving out
--- reference links, raw HTML passthrough, nested lists and setext headings.
--- Emphasis does not span a hard line break inside a paragraph. Man rendering
--- covers the common macros, that is section headings, filled and no-fill
--- paragraphs, bold and italic and the font escapes, and drops the rest. Both
--- are parsed with lpeg into the node trees emit_blocks and emit_inline above
--- consume, and parsing only builds a tree, so a parse that fails falls back
--- to plain text without leaving half a page behind.
+-- Markdown is a deliberate subset rather than CommonMark, leaving out
+-- reference links, raw HTML passthrough, nested lists and setext headings,
+-- and emphasis does not span a hard line break. Man rendering covers the
+-- common macros and drops the rest. Both parse into a node tree before
+-- anything is emitted, so a failed parse falls back to plain text without
+-- leaving half a page behind.
local render_markdown, render_man
@@ -223,8 +206,6 @@ if has_lpeg then
-- Bound the link and image inner scans. Without a cap a readme of
-- unclosed brackets ("[[[[...") makes every position scan to the
-- end of the line for a "]" that never comes, which is quadratic.
- -- Real link text and urls sit far under this, and anything longer
- -- simply renders as plain text.
local max_scan = 512
local parse_inline
@@ -275,8 +256,6 @@ if has_lpeg then
return nodes
end
- -- Line matchers. Each is anchored at the start of a single line and
- -- returns its captures, or nil when the line is not of that kind.
local lang_char = R("az", "AZ", "09") + S("_.+#-")
local heading_line = C(P("#") * P("#") ^ -5) * space ^ 1 * C(P(1) ^ 0)
local function thematic(mark)
@@ -434,20 +413,24 @@ if has_lpeg then
html("</div>")
end
- -- Macro lines are matched with lpeg, and the roff inline font and
- -- character escapes are a grammar producing a flat token list that
- -- a fold turns into the same inline nodes markdown emits. Bold and
- -- italic are the current font, which carries across a run rather
- -- than being opened and closed, so the inline pass tokenises and
- -- folds instead of nesting the way markdown does.
+ -- Roff fonts are a current state carried across a run rather than
+ -- opened and closed, so the inline pass tokenises to a flat list
+ -- and folds it into the nodes markdown emits, instead of nesting.
local macro_line = S(".'") * space ^ 0 * C((1 - space) ^ 1)
* space ^ 0 * C(P(1) ^ 0)
local one_font = { B = "B", I = "I" }
local two_font = { CB = "B", BI = "B", CI = "I" }
local man_chars = {
- aq = "'", cq = "'", oq = "'", dq = '"', lq = '"', rq = '"',
- hy = "-", en = "-", em = "-",
+ aq = "'",
+ cq = "'",
+ oq = "'",
+ dq = '"',
+ lq = '"',
+ rq = '"',
+ hy = "-",
+ en = "-",
+ em = "-",
}
local backslash = P("\\")
local function font_token(name) return { kind = "font", font = name } end
@@ -610,7 +593,7 @@ function filter_write(str)
end
-- cgit takes filter output through a C string sink that stops at the first
--- NUL byte, so a readme holding one is truncated there.
+-- NUL byte, so a write holding one loses everything from the NUL on.
function filter_close()
local text = table.concat(chunks)
chunks = {}