diff options
Diffstat (limited to 'custom/extensions/about-render.lua')
| -rw-r--r-- | custom/extensions/about-render.lua | 585 | |||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||||
1 file changed, 315 insertions, 270 deletions
diff --git a/custom/extensions/about-render.lua b/custom/extensions/about-render.lua index c4c1c93..ea13746 100644 --- a/custom/extensions/about-render.lua +++ b/custom/extensions/about-render.lua @@ -1,29 +1,20 @@ --- Server-side about-page rendering for cgit, used with the about-filter setting --- in cgitrc and the lua: prefix so it runs in cgit's embedded interpreter with --- no per-request process. +-- Server-side rendering of a repository's about page, named by the +-- about-filter setting in cgitrc and run inside cgit's embedded Lua +-- interpreter so a readme costs no extra process per request. +-- Markdown, man pages and plain text are the three formats, chosen from the +-- readme's file extension, and anything that fails to parse falls back to +-- escaped plain text. Adding a format takes a render function and a row in +-- the handler table at the foot of this file, and nothing else. The wrappers +-- it emits carry the classes that assets/cgit.css styles. It runs on Lua 5.1 +-- through 5.4 and LuaJIT. -- -- about-filter=lua:/usr/lib/cgit/filters/about-render.lua --- --- One filter renders the three baseline about formats, chosen by the readme's --- file extension. --- --- markdown .md .markdown .mkd .mdown --- man page .1 through .9 and .man --- plain text everything else, and the fallback for anything that fails --- --- Add a format by giving it a render function and a row in the handler table --- near the foot of this file. Nothing else needs to change. --- --- SUPPORTED LUA --- --- Lua 5.1 through 5.4 and LuaJIT. --- --- REQUIREMENTS --- --- lpeg, the parsing module, for the Lua cgit is linked against. Markdown and --- man pages are parsed with lpeg grammars, so without lpeg both fall back to --- escaped plain text. It ships with cgit's syntax highlighter too, so a cgit --- that colours source already has it. + +-- Markdown and man pages are parsed with lpeg grammars, so where the parsing +-- module is missing both fall back to escaped plain text and only plain text +-- still renders. It has to be built for the Lua cgit is linked against. +-- cgit's syntax highlighter needs it too, so a cgit that colours source +-- already has it. -- -- # Debian and Ubuntu -- sudo apt install lua-lpeg @@ -33,40 +24,11 @@ -- sudo apk add lua5.1-lpeg -- # or with LuaRocks, matched to your Lua version -- sudo luarocks --lua-version 5.1 install lpeg --- --- Plain text needs only Lua itself. --- --- SECURITY --- --- The about page renders untrusted repository content, so the safety rule is --- that every run of text reaches the page through cgit's own html_txt, every --- attribute value through html_attr, and every link or image target through --- the safe_url scheme allowlist below before html_attr. Only http, https, --- mailto and relative targets survive, so a hostile readme cannot inject markup --- or a javascript: url. Because all output is routed through those sinks by --- construction, a bug in the parser can only mis-render, never inject. --- --- LIMITATIONS --- --- Markdown is a deliberate subset, not CommonMark. Headings, thematic breaks, --- fenced and inline code, blockquotes, pipe tables, single-level lists, links, --- images and emphasis. No reference links, no raw HTML passthrough, no nested --- lists, no setext headings. Emphasis does not span a hard line break inside a --- paragraph. Man rendering covers the common macros (section headings, filled --- and no-fill paragraphs, bold and italic, font escapes) and drops the rest. --- cgit sends filter output through a C string sink that stops at the first NUL --- byte, so content past a NUL is truncated. --- --- OUTPUT --- --- Markdown is wrapped in <div class='markdown'>, man pages in --- <div class='markdown manpage'>, plain text in <pre class='plaintext'>, all --- styled by assets/cgit.css. +local has_lpeg, lpeg = pcall(require, "lpeg") -local ok_lpeg, lpeg = pcall(require, "lpeg") - --- Readmes larger than this are served as escaped plain text rather than parsed. +-- Readmes larger than this are served as escaped plain text rather than +-- parsed. local max_bytes = 512 * 1024 -- Ceiling on blockquote and emphasis nesting, so a hostile readme of stacked @@ -76,47 +38,48 @@ local max_depth = 24 local filename, chunks = "", {} -local function trim(s) - return (s:gsub("^%s+", ""):gsub("%s+$", "")) +local function trim(text) + return (text:gsub("^%s+", ""):gsub("%s+$", "")) end --- Split on newlines after normalising CRLF and CR, returning the lines without --- their terminators. A trailing newline yields a final empty line, which every --- caller treats as blank. -local function split_lines(s) - s = s:gsub("\r\n?", "\n") +-- Split on newlines after normalising CRLF and CR, returning the lines +-- without their terminators. A trailing newline yields a final empty line, +-- which every caller treats as blank. +local function split_lines(text) + text = text:gsub("\r\n?", "\n") local lines, start = {}, 1 while true do - local nl = s:find("\n", start, true) - if not nl then - lines[#lines + 1] = s:sub(start) + local newline = text:find("\n", start, true) + if not newline then + lines[#lines + 1] = text:sub(start) return lines end - lines[#lines + 1] = s:sub(start, nl - 1) - start = nl + 1 + lines[#lines + 1] = text:sub(start, newline - 1) + start = newline + 1 end end --- Split a table row into trimmed cells, dropping one optional leading and --- trailing pipe. +-- Split a table row into trimmed cells, dropping one optional leading and one +-- optional trailing pipe. local function split_cells(row) row = trim(row):gsub("^|", ""):gsub("|$", "") local cells = {} - for c in (row .. "|"):gmatch("(.-)|") do - cells[#cells + 1] = trim(c) + for cell in (row .. "|"):gmatch("(.-)|") do + cells[#cells + 1] = trim(cell) end return cells end --- Return the url if its scheme is safe, else nil. Control and whitespace bytes --- are stripped anywhere first because a browser ignores them when resolving the --- scheme, so "java\nscript:" must still be caught as javascript. The class is --- %c%s rather than a literal 0x00 range so it is safe on Lua 5.1, where an --- embedded zero byte ends a pattern. +-- Return the URL if its scheme is safe, else nil. Control and whitespace +-- bytes are stripped anywhere first because a browser ignores them when +-- resolving the scheme, so "java\nscript:" must still be caught as +-- javascript. The class is %c%s rather than a literal 0x00 range so it is +-- safe on Lua 5.1, where an embedded zero byte ends a pattern. local function safe_url(url) url = url:gsub("[%c%s]", "") - -- A leading "//" or "/\" is scheme-relative, which a browser resolves to - -- another origin, so it is not the local target the parser meant to allow. + -- A leading "//" or "/\" is scheme-relative, which a browser + -- resolves to another origin, so it is not the local target the + -- parser meant to allow. if url:sub(1, 2) == "//" or url:sub(1, 2) == "/\\" then return nil end @@ -131,39 +94,44 @@ local function safe_url(url) end --- html, html_txt and html_attr are injected by cgit's lua filter host. Tag --- scaffolding is a literal argument to html; every value that came from the --- repository goes through html_txt, html_attr or safe_url. +-- The about page renders untrusted repository content, so the rule the two +-- functions below keep is that every run of text reaches the page through +-- cgit's own html_txt, every attribute value through html_attr, and every +-- link or image target through safe_url before html_attr. Those three come +-- from cgit's Lua filter host rather than from here, and the only thing +-- passed to html is literal tag scaffolding. Because all output is routed +-- through those sinks by construction, a bug in the parser can only +-- mis-render, never inject markup or a URL with a javascript scheme. local emit_inline emit_inline = function(list) - for _, nd in ipairs(list) do - local t = nd.t - if t == "text" then - html_txt(nd.v) - elseif t == "code" then - html("<code>"); html_txt(nd.v); html("</code>") - elseif t == "strong" then - html("<strong>"); emit_inline(nd.kids); html("</strong>") - elseif t == "em" then - html("<em>"); emit_inline(nd.kids); html("</em>") - elseif t == "link" then - local u = safe_url(nd.url) - if u then - html("<a href='"); html_attr(u); html("'>") - emit_inline(nd.kids) + for _, node in ipairs(list) do + local kind = node.kind + if kind == "text" then + html_txt(node.text) + elseif kind == "code" then + html("<code>"); html_txt(node.text); html("</code>") + elseif kind == "strong" then + html("<strong>"); emit_inline(node.kids); html("</strong>") + elseif kind == "em" then + html("<em>"); emit_inline(node.kids); html("</em>") + elseif kind == "link" then + local target = safe_url(node.url) + if target then + html("<a href='"); html_attr(target); html("'>") + emit_inline(node.kids) html("</a>") else - emit_inline(nd.kids) + emit_inline(node.kids) end - elseif t == "image" then - local u = safe_url(nd.url) - if u then - html("<img src='"); html_attr(u) - html("' alt='"); html_attr(nd.alt); html("'/>") + elseif kind == "image" then + local target = safe_url(node.url) + if target then + html("<img src='"); html_attr(target) + html("' alt='"); html_attr(node.alt); html("'>") else - html_txt("![" .. nd.alt .. "]") + html_txt("![" .. node.alt .. "]") end end end @@ -172,47 +140,49 @@ end local emit_blocks emit_blocks = function(blocks) - for _, b in ipairs(blocks) do - local t = b.t - if t == "heading" then - html("<h" .. b.level .. ">") - emit_inline(b.kids) - html("</h" .. b.level .. ">") - elseif t == "hr" then - html("<hr/>") - elseif t == "code" then + for _, block in ipairs(blocks) do + local kind = block.kind + if kind == "heading" then + html("<h" .. block.level .. ">") + emit_inline(block.kids) + html("</h" .. block.level .. ">") + elseif kind == "hr" then + html("<hr>") + elseif kind == "code" then html("<pre><code") - if b.lang and b.lang ~= "" then - html(" data-lang='"); html_attr(b.lang); html("'") + if block.lang and block.lang ~= "" then + html(" data-lang='"); html_attr(block.lang); html("'") end - html(">"); html_txt(b.text); html("</code></pre>") - elseif t == "blockquote" then - html("<blockquote>"); emit_blocks(b.blocks); html("</blockquote>") - elseif t == "table" then + html(">"); html_txt(block.text); html("</code></pre>") + elseif kind == "blockquote" then + html("<blockquote>") + emit_blocks(block.blocks) + html("</blockquote>") + elseif kind == "table" then html("<table><thead><tr>") - for _, c in ipairs(b.head) do - html("<th>"); emit_inline(c); html("</th>") + for _, cell in ipairs(block.head) do + html("<th>"); emit_inline(cell); html("</th>") end html("</tr></thead><tbody>") - for _, row in ipairs(b.rows) do + for _, row in ipairs(block.rows) do html("<tr>") - for _, c in ipairs(row) do - html("<td>"); emit_inline(c); html("</td>") + for _, cell in ipairs(row) do + html("<td>"); emit_inline(cell); html("</td>") end html("</tr>") end html("</tbody></table>") - elseif t == "list" then - local tag = b.ordered and "ol" or "ul" + elseif kind == "list" then + local tag = block.ordered and "ol" or "ul" html("<" .. tag .. ">") - for _, item in ipairs(b.items) do + for _, item in ipairs(block.items) do html("<li>"); emit_inline(item); html("</li>") end html("</" .. tag .. ">") - elseif t == "para" then + elseif kind == "para" then html("<p>") - for k, line in ipairs(b.lines) do - if k > 1 then html("<br/>") end + for i, line in ipairs(block.lines) do + if i > 1 then html("<br>") end emit_inline(line) end html("</p>") @@ -228,48 +198,69 @@ local function render_plaintext(text) end --- Both are parsed with lpeg into the block/inline node trees emit_blocks and --- emit_inline above consume. Parsing is pure and builds a tree; emit is the --- only step that writes output, so a parse failure falls back to plain text --- without leaving half a page behind. +-- Markdown here is a deliberate subset rather than CommonMark, covering +-- headings, thematic breaks, fenced and inline code, blockquotes, pipe +-- tables, single-level lists, links, images and emphasis, and leaving out +-- reference links, raw HTML passthrough, nested lists and setext headings. +-- Emphasis does not span a hard line break inside a paragraph. Man rendering +-- covers the common macros, that is section headings, filled and no-fill +-- paragraphs, bold and italic and the font escapes, and drops the rest. Both +-- are parsed with lpeg into the node trees emit_blocks and emit_inline above +-- consume, and parsing only builds a tree, so a parse that fails falls back +-- to plain text without leaving half a page behind. local render_markdown, render_man -if ok_lpeg then +if has_lpeg then local P, R, S, C, Ct = lpeg.P, lpeg.R, lpeg.S, lpeg.C, lpeg.Ct - local sp = S(" \t") - local ws = S(" \t\r\n\f\v") + local space = S(" \t") + local whitespace = S(" \t\r\n\f\v") local eol = P(-1) local marker = S("`![*_") - local urlchar = 1 - S(")") - ws - -- Bound the link and image inner scans. Without a cap a readme of unclosed - -- brackets ("[[[[...") makes each position scan to end of line for a "]" - -- that never comes, which is quadratic. Real link text and urls sit far - -- under this, and anything longer simply renders as plain text. - local CAP = 512 + local url_char = 1 - S(")") - whitespace + -- Bound the link and image inner scans. Without a cap a readme of + -- unclosed brackets ("[[[[...") makes every position scan to the + -- end of the line for a "]" that never comes, which is quadratic. + -- Real link text and urls sit far under this, and anything longer + -- simply renders as plain text. + local max_scan = 512 local parse_inline local inline_depth = 0 - local function node_text(s) return { t = "text", v = s } end - local function node_code(s) return { t = "code", v = s } end - local function node_strong(s) return { t = "strong", kids = parse_inline(s) } end - local function node_em(s) return { t = "em", kids = parse_inline(s) } end - local function node_link(text, url) return { t = "link", kids = parse_inline(text), url = url } end - local function node_image(alt, url) return { t = "image", alt = alt, url = url } end + local function node_text(text) + return { kind = "text", text = text } + end + local function node_code(text) + return { kind = "code", text = text } + end + local function node_strong(text) + return { kind = "strong", kids = parse_inline(text) } + end + local function node_em(text) + return { kind = "em", kids = parse_inline(text) } + end + local function node_link(text, url) + return { kind = "link", kids = parse_inline(text), url = url } + end + local function node_image(alt, url) + return { kind = "image", alt = alt, url = url } + end local inline_grammar = Ct(( (P("`") * C((1 - P("`")) ^ 1) * P("`")) / node_code - + (P("![") * C((1 - P("]")) ^ (-CAP)) * P("]") * P("(") * sp ^ 0 - * C(urlchar * urlchar ^ (-CAP)) * (1 - P(")")) ^ (-CAP) * P(")")) / node_image - + (P("[") * C((1 - P("]")) ^ (-CAP)) * P("]") * P("(") * sp ^ 0 - * C(urlchar * urlchar ^ (-CAP)) * (1 - P(")")) ^ (-CAP) * P(")")) / node_link + + (P("![") * C((1 - P("]")) ^ (-max_scan)) * P("]") * P("(") + * space ^ 0 * C(url_char * url_char ^ (-max_scan)) + * (1 - P(")")) ^ (-max_scan) * P(")")) / node_image + + (P("[") * C((1 - P("]")) ^ (-max_scan)) * P("]") * P("(") + * space ^ 0 * C(url_char * url_char ^ (-max_scan)) + * (1 - P(")")) ^ (-max_scan) * P(")")) / node_link + (P("**") * C((1 - P("**")) ^ 1) * P("**")) / node_strong + (P("__") * C((1 - P("__")) ^ 1) * P("__")) / node_strong - + (P("*") * C((1 - ws) * (1 - P("*")) ^ 0) * P("*")) / node_em - + (P("_") * C((1 - ws) * (1 - P("_")) ^ 0) * P("_")) / node_em + + (P("*") * C((1 - whitespace) * (1 - P("*")) ^ 0) * P("*")) / node_em + + (P("_") * C((1 - whitespace) * (1 - P("_")) ^ 0) * P("_")) / node_em + C((1 - marker) ^ 1) / node_text + C(P(1)) / node_text ) ^ 0) @@ -284,30 +275,36 @@ if ok_lpeg then return nodes end - -- Line matchers. Each is anchored at the start of a single line and returns - -- its captures, or nil when the line is not of that kind. - local langchar = R("az", "AZ", "09") + S("_.+#-") - local m_heading = C(P("#") * P("#") ^ -5) * sp ^ 1 * C(P(1) ^ 0) - local function rule(c) return sp ^ 0 * P(c) * (sp ^ 0 * P(c)) ^ 2 * sp ^ 0 * eol end - local m_hr = rule("-") + rule("*") + rule("_") - local m_fence = sp ^ 0 * (C(P("`") ^ 3) + C(P("~") ^ 3)) * sp ^ 0 * C(langchar ^ 0) - local m_close_bt = sp ^ 0 * P("`") ^ 3 * sp ^ 0 * eol - local m_close_ti = sp ^ 0 * P("~") ^ 3 * sp ^ 0 * eol - local m_bq = sp ^ 0 * P(">") + -- Line matchers. Each is anchored at the start of a single line and + -- returns its captures, or nil when the line is not of that kind. + local lang_char = R("az", "AZ", "09") + S("_.+#-") + local heading_line = C(P("#") * P("#") ^ -5) * space ^ 1 * C(P(1) ^ 0) + local function thematic(mark) + return space ^ 0 * P(mark) * (space ^ 0 * P(mark)) ^ 2 + * space ^ 0 * eol + end + local break_line = thematic("-") + thematic("*") + thematic("_") + local fence_line = space ^ 0 * (C(P("`") ^ 3) + C(P("~") ^ 3)) + * space ^ 0 * C(lang_char ^ 0) + local close_backtick = space ^ 0 * P("`") ^ 3 * space ^ 0 * eol + local close_tilde = space ^ 0 * P("~") ^ 3 * space ^ 0 * eol + local quote_line = space ^ 0 * P(">") local bullet = S("-*+") local number = R("09") ^ 1 * S(".)") - local m_item = sp ^ 0 * C(bullet + number) * sp ^ 1 * C(P(1) ^ 0) - local dcell = sp ^ 0 * P(":") ^ -1 * P("-") ^ 1 * P(":") ^ -1 * sp ^ 0 - -- A delimiter row is dash-cells joined by pipes. Require at least one pipe, - -- a leading one or one between cells, so a bare rule of dashes stays a - -- thematic break and an ordinary paragraph line is never taken for a table. + local item_line = space ^ 0 * C(bullet + number) * space ^ 1 * C(P(1) ^ 0) + local dash_cell = space ^ 0 * P(":") ^ -1 * P("-") ^ 1 * P(":") ^ -1 + * space ^ 0 + -- A delimiter row is dash-cells joined by pipes. Require at least one + -- pipe, a leading one or one between cells, so a bare rule of dashes + -- stays a thematic break and an ordinary paragraph line is never taken + -- for a table. local pipe = P("|") - local m_tdelim = sp ^ 0 * ( - pipe * dcell * (pipe * dcell) ^ 0 * pipe ^ -1 - + dcell * (pipe * dcell) ^ 1 * pipe ^ -1 - ) * sp ^ 0 * eol - local m_blockstart = sp ^ 0 * ((P("#") * P("#") ^ -5 * sp) + P(">") - + P("`") ^ 3 + P("~") ^ 3 + ((bullet + number) * sp)) + local table_delimiter = space ^ 0 * ( + pipe * dash_cell * (pipe * dash_cell) ^ 0 * pipe ^ -1 + + dash_cell * (pipe * dash_cell) ^ 1 * pipe ^ -1 + ) * space ^ 0 * eol + local block_start = space ^ 0 * ((P("#") * P("#") ^ -5 * space) + P(">") + + P("`") ^ 3 + P("~") ^ 3 + ((bullet + number) * space)) local function ordered(mark) return mark:match("%d") ~= nil end @@ -317,13 +314,14 @@ if ok_lpeg then local blocks, i, n = {}, 1, #lines while i <= n do local line = lines[i] - local fence, lang = m_fence:match(line) - local hashes, htext = m_heading:match(line) - local mark = m_item:match(line) + local fence, lang = fence_line:match(line) + local hashes, title = heading_line:match(line) + local mark = item_line:match(line) if line:match("^%s*$") then i = i + 1 elseif fence then - local close = (fence:sub(1, 1) == "`") and m_close_bt or m_close_ti + local close = (fence:sub(1, 1) == "`") + and close_backtick or close_tilde local code = {} i = i + 1 while i <= n and not close:match(lines[i]) do @@ -332,59 +330,89 @@ if ok_lpeg then i = i + 1 local body = table.concat(code, "\n") if #code > 0 then body = body .. "\n" end - blocks[#blocks + 1] = { t = "code", lang = lang, text = body } + blocks[#blocks + 1] = { + kind = "code", + lang = lang, + text = body, + } elseif hashes then blocks[#blocks + 1] = { - t = "heading", + kind = "heading", level = #hashes, - kids = parse_inline((htext:gsub("%s*#*%s*$", ""))), + kids = parse_inline((title:gsub("%s*#*%s*$", ""))), } i = i + 1 - elseif m_hr:match(line) then - blocks[#blocks + 1] = { t = "hr" } + elseif break_line:match(line) then + blocks[#blocks + 1] = { kind = "hr" } i = i + 1 - elseif m_bq:match(line) then + elseif quote_line:match(line) then local inner = {} - while i <= n and m_bq:match(lines[i]) do + while i <= n and quote_line:match(lines[i]) do inner[#inner + 1] = (lines[i]:gsub("^%s*>%s?", "")) i = i + 1 end if depth < max_depth then - blocks[#blocks + 1] = { t = "blockquote", blocks = parse_blocks(inner, depth + 1) } + blocks[#blocks + 1] = { + kind = "blockquote", + blocks = parse_blocks(inner, depth + 1), + } else - local para = {} - for _, l in ipairs(inner) do para[#para + 1] = parse_inline(l) end - blocks[#blocks + 1] = { t = "para", lines = para } + local paragraph = {} + for _, quoted in ipairs(inner) do + paragraph[#paragraph + 1] = parse_inline(quoted) + end + blocks[#blocks + 1] = { + kind = "para", + lines = paragraph, + } end - elseif i + 1 <= n and line:find("|", 1, true) and m_tdelim:match(lines[i + 1]) then + elseif i + 1 <= n and line:find("|", 1, true) + and table_delimiter:match(lines[i + 1]) then local head = {} - for _, c in ipairs(split_cells(line)) do head[#head + 1] = parse_inline(c) end + for _, cell in ipairs(split_cells(line)) do + head[#head + 1] = parse_inline(cell) + end local rows = {} i = i + 2 - while i <= n and lines[i]:find("|", 1, true) and not lines[i]:match("^%s*$") do + while i <= n and lines[i]:find("|", 1, true) + and not lines[i]:match("^%s*$") do local row = {} - for _, c in ipairs(split_cells(lines[i])) do row[#row + 1] = parse_inline(c) end + for _, cell in ipairs(split_cells(lines[i])) do + row[#row + 1] = parse_inline(cell) + end rows[#rows + 1] = row i = i + 1 end - blocks[#blocks + 1] = { t = "table", head = head, rows = rows } + blocks[#blocks + 1] = { + kind = "table", + head = head, + rows = rows, + } elseif mark then - local is_ol = ordered(mark) + local is_ordered = ordered(mark) local items = {} while i <= n do - local mk, ct = m_item:match(lines[i]) - if not mk or ordered(mk) ~= is_ol then break end - items[#items + 1] = parse_inline(ct) + local item_mark, content = item_line:match(lines[i]) + if not item_mark or ordered(item_mark) ~= is_ordered then + break + end + items[#items + 1] = parse_inline(content) i = i + 1 end - blocks[#blocks + 1] = { t = "list", ordered = is_ol, items = items } + blocks[#blocks + 1] = { + kind = "list", + ordered = is_ordered, + items = items, + } else - local para = { parse_inline(line) } + local paragraph = { parse_inline(line) } i = i + 1 - while i <= n and not lines[i]:match("^%s*$") and not m_blockstart:match(lines[i]) do - para[#para + 1] = parse_inline(lines[i]); i = i + 1 + while i <= n and not lines[i]:match("^%s*$") + and not block_start:match(lines[i]) do + paragraph[#paragraph + 1] = parse_inline(lines[i]) + i = i + 1 end - blocks[#blocks + 1] = { t = "para", lines = para } + blocks[#blocks + 1] = { kind = "para", lines = paragraph } end end return blocks @@ -399,18 +427,21 @@ if ok_lpeg then return render_plaintext(text) end html("<div class='markdown'>") - -- Emit is wrapped so a parser bug can at worst truncate the page, never - -- error out and let the outer fallback duplicate what was already sent. + -- Emitting is wrapped so a bug there can at worst + -- truncate the page, never raise and let the outer + -- fallback duplicate what has already been sent. pcall(emit_blocks, blocks) html("</div>") end - -- Macro lines are matched with lpeg, and the roff inline font and character - -- escapes are an lpeg grammar producing a flat token list that a fold turns - -- into the same inline nodes markdown emits. Bold and italic are the current - -- font, which carries across a run, so the inline pass tokenises and folds - -- rather than nesting the way markdown does. - local m_macro = S(".'") * sp ^ 0 * C((1 - sp) ^ 1) * sp ^ 0 * C(P(1) ^ 0) + -- Macro lines are matched with lpeg, and the roff inline font and + -- character escapes are a grammar producing a flat token list that + -- a fold turns into the same inline nodes markdown emits. Bold and + -- italic are the current font, which carries across a run rather + -- than being opened and closed, so the inline pass tokenises and + -- folds instead of nesting the way markdown does. + local macro_line = S(".'") * space ^ 0 * C((1 - space) ^ 1) + * space ^ 0 * C(P(1) ^ 0) local one_font = { B = "B", I = "I" } local two_font = { CB = "B", BI = "B", CI = "I" } @@ -418,46 +449,57 @@ if ok_lpeg then aq = "'", cq = "'", oq = "'", dq = '"', lq = '"', rq = '"', hy = "-", en = "-", em = "-", } - local esc = P("\\") - local function font_tok(f) return { k = "font", f = f } end - local function text_tok(v) return { k = "text", v = v } end - local skip_tok = { k = "skip" } + local backslash = P("\\") + local function font_token(name) return { kind = "font", font = name } end + local function text_token(text) return { kind = "text", text = text } end + local skip_token = { kind = "skip" } local man_inline_grammar = Ct(( - (esc * P("f") * P("(") * C(P(1) * P(1))) / function(nm) return font_tok(two_font[nm] or "R") end - + (esc * P("f") * P("[") * C((1 - P("]")) ^ 0) * P("]")) / function(nm) return font_tok(one_font[nm] or "R") end - + (esc * P("f") * C(P(1))) / function(f) return font_tok(one_font[f] or "R") end - + (esc * P("-")) / function() return text_tok("-") end - + (esc * P("e")) / function() return text_tok("\\") end - + (esc * P(" ")) / function() return text_tok(" ") end - + (esc * S("&|^")) / function() return skip_tok end - + (esc * P("(") * C(P(1) * P(1))) / function(nm) return text_tok(man_chars[nm] or "") end - + (esc * P("*") * (P("(") * P(1) * P(1) + P(1))) / function() return skip_tok end - + (esc * C(P(1))) / text_tok - + esc / function() return skip_tok end - + C((1 - esc) ^ 1) / text_tok + (backslash * P("f") * P("(") * C(P(1) * P(1))) + / function(name) return font_token(two_font[name] or "R") end + + (backslash * P("f") * P("[") * C((1 - P("]")) ^ 0) * P("]")) + / function(name) return font_token(one_font[name] or "R") end + + (backslash * P("f") * C(P(1))) + / function(name) return font_token(one_font[name] or "R") end + + (backslash * P("-")) / function() return text_token("-") end + + (backslash * P("e")) / function() return text_token("\\") end + + (backslash * P(" ")) / function() return text_token(" ") end + + (backslash * S("&|^")) / function() return skip_token end + + (backslash * P("(") * C(P(1) * P(1))) + / function(name) return text_token(man_chars[name] or "") end + + (backslash * P("*") * (P("(") * P(1) * P(1) + P(1))) + / function() return skip_token end + + (backslash * C(P(1))) / text_token + + backslash / function() return skip_token end + + C((1 - backslash) ^ 1) / text_token ) ^ 0) - local function man_inline(s) - local toks = man_inline_grammar:match(s) or {} - local nodes, buf, font = {}, {}, "R" + local function man_inline(text) + local tokens = man_inline_grammar:match(text) or {} + local nodes, parts, font = {}, {}, "R" local function flush() - if #buf == 0 then return end - local txt = table.concat(buf); buf = {} - if txt == "" then return end + if #parts == 0 then return end + local run = table.concat(parts); parts = {} + if run == "" then return end if font == "B" then - nodes[#nodes + 1] = { t = "strong", kids = { { t = "text", v = txt } } } + nodes[#nodes + 1] = { + kind = "strong", + kids = { { kind = "text", text = run } }, + } elseif font == "I" then - nodes[#nodes + 1] = { t = "em", kids = { { t = "text", v = txt } } } + nodes[#nodes + 1] = { + kind = "em", + kids = { { kind = "text", text = run } }, + } else - nodes[#nodes + 1] = { t = "text", v = txt } + nodes[#nodes + 1] = { kind = "text", text = run } end end - for _, tk in ipairs(toks) do - if tk.k == "text" then - buf[#buf + 1] = tk.v - elseif tk.k == "font" then - flush(); font = tk.f + for _, token in ipairs(tokens) do + if token.kind == "text" then + parts[#parts + 1] = token.text + elseif token.kind == "font" then + flush(); font = token.font end end flush() @@ -467,66 +509,66 @@ if ok_lpeg then local function parse_man(text) local lines = split_lines(text) local blocks = {} - local para, pre, nofill = nil, nil, false - local function flush_para() - if para and #para > 0 then - blocks[#blocks + 1] = { t = "para", lines = para } + local paragraph, nofill_lines, nofill = nil, nil, false + local function flush_paragraph() + if paragraph and #paragraph > 0 then + blocks[#blocks + 1] = { kind = "para", lines = paragraph } end - para = nil + paragraph = nil end - local function flush_pre() - if pre then - local body = table.concat(pre, "\n") - if #pre > 0 then body = body .. "\n" end - blocks[#blocks + 1] = { t = "code", text = body } - pre = nil + local function flush_nofill() + if nofill_lines then + local body = table.concat(nofill_lines, "\n") + if #nofill_lines > 0 then body = body .. "\n" end + blocks[#blocks + 1] = { kind = "code", text = body } + nofill_lines = nil end end local function add_line(nodes) - para = para or {} - para[#para + 1] = nodes + paragraph = paragraph or {} + paragraph[#paragraph + 1] = nodes end for _, line in ipairs(lines) do - local macro, rest = m_macro:match(line) + local macro, rest = macro_line:match(line) if rest then rest = (rest:gsub('^"(.*)"$', "%1")) end if nofill then if macro == "fi" then - nofill = false; flush_pre() + nofill = false; flush_nofill() else - pre[#pre + 1] = line + nofill_lines[#nofill_lines + 1] = line end elseif line:match("^%s*$") then - flush_para() + flush_paragraph() elseif macro and macro:sub(1, 1) == "\\" then - -- roff comment, dropped + -- A roff comment, dropped. elseif macro == "SH" or macro == "SS" then - flush_para() + flush_paragraph() blocks[#blocks + 1] = { - t = "heading", + kind = "heading", level = (macro == "SH") and 2 or 3, kids = man_inline(rest), } elseif macro == "nf" then - flush_para(); nofill = true; pre = {} + flush_paragraph(); nofill = true; nofill_lines = {} elseif macro == "fi" then - flush_pre() + flush_nofill() elseif macro == "B" or macro == "BR" or macro == "RB" or macro == "BI" or macro == "IB" then - add_line({ { t = "strong", kids = man_inline(rest) } }) + add_line({ { kind = "strong", kids = man_inline(rest) } }) elseif macro == "I" or macro == "IR" or macro == "RI" then - add_line({ { t = "em", kids = man_inline(rest) } }) + add_line({ { kind = "em", kids = man_inline(rest) } }) elseif macro == "TH" or macro == "PP" or macro == "LP" or macro == "P" or macro == "TP" or macro == "IP" or macro == "HP" or macro == "RS" or macro == "RE" or macro == "sp" or macro == "br" then - flush_para() + flush_paragraph() elseif macro then if rest ~= "" then add_line(man_inline(rest)) end else add_line(man_inline(line)) end end - flush_para(); flush_pre() + flush_paragraph(); flush_nofill() return blocks end @@ -543,7 +585,8 @@ if ok_lpeg then html("</div>") end else - -- No lpeg, so markdown and man readmes read as escaped source, not nothing. + -- No lpeg, so markdown and man readmes read as escaped source, not + -- nothing. render_markdown = render_plaintext render_man = render_plaintext end @@ -566,6 +609,8 @@ function filter_write(str) chunks[#chunks + 1] = str end +-- cgit takes filter output through a C string sink that stops at the first +-- NUL byte, so a readme holding one is truncated there. function filter_close() local text = table.concat(chunks) chunks = {} |
