#!/usr/bin/env python3 """Runs the built cgit binary behind a local HTTP port for development. cgit is a CGI program that reads a request out of environment variables and writes a response to stdout, so each request here becomes one run of the binary with the same environment the server configs in custom/servers/ set up. Requests for the files in assets are answered from disk instead, and a cgitrc is expected, by default the one in the working directory. Nothing outside the standard library is needed, and nothing here is meant to face a network wider than loopback. """ from __future__ import annotations import argparse import mimetypes import os import subprocess import sys from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer from pathlib import Path from typing import NamedTuple, cast from urllib.parse import unquote REPO_ROOT = Path(__file__).resolve().parent.parent # Bounds so a single request cannot exhaust the preview server. It only ever # serves one browser on loopback, so these are generous. MAX_BODY_BYTES = 8 * 1024 * 1024 CGI_TIMEOUT = 60 STATIC_SUFFIXES = ( ".css", ".js", ".png", ".ico", ".gif", ".jpg", ".jpeg", ".svg", ".webp", ".txt", ".woff", ".woff2", ) # The request headers cgit looks at, paired with the CGI variable each one # has to arrive as. PASSED_HEADERS = ( ("Cookie", "HTTP_COOKIE"), ("Referer", "HTTP_REFERER"), ("Content-Type", "CONTENT_TYPE"), ) class CgiResponse(NamedTuple): status: int reason: str headers: list[tuple[str, str]] body: bytes def split_cgi_output(output: bytes) -> CgiResponse: """Split a raw CGI response into its status, headers and body. The header block ends at the first blank line. cgit ends its lines with a bare LF while a filter writing its own headers may use CRLF, so whichever terminator appears earliest is the real one. Looking for CRLF everywhere before falling back to LF would instead find the first CRLF in the body, and a commit message with DOS line endings has one, which swallowed the whole document head into the headers. Splitting on the separator rather than on a non-empty body keeps a legitimately empty body, such as the one a redirect leaves behind, from being read as headers. """ ends = [(at, len(sep)) for sep in (b"\r\n\r\n", b"\n\n") if (at := output.find(sep)) >= 0] if ends: at, seplen = min(ends) header_block, body = output[:at], output[at + seplen:] else: header_block, body = output, b"" status, reason = 200, "OK" headers: list[tuple[str, str]] = [] for line in header_block.replace(b"\r\n", b"\n").split(b"\n"): if not line.strip(): continue raw_name, _, raw_value = line.partition(b":") name = raw_name.strip().decode("latin-1") value = raw_value.strip().decode("latin-1") if name.lower() != "status": headers.append((name, value)) continue # For example "Status: 404 Not Found", where the reason phrase is # optional. A value that will not parse leaves the 200 OK default # whole rather than pairing a stale code with a new phrase. code, _, phrase = value.partition(" ") try: status = int(code) except ValueError: continue reason = phrase.strip() return CgiResponse(status, reason, headers, body) class CgitHandler(BaseHTTPRequestHandler): server_version: str = "cgit-preview" @property def preview(self) -> CgitServer: """The owning server, typed so the paths it carries are visible.""" return cast("CgitServer", self.server) def do_GET(self) -> None: self.respond() def do_HEAD(self) -> None: self.respond() def do_POST(self) -> None: self.respond() def respond(self) -> None: raw_path, _, query = self.path.partition("?") # A real web server hands the CGI a decoded PATH_INFO, so decode here # too and keep a repository whose name needs escaping working. path = unquote(raw_path) asset = self.locate_asset(path) if asset is None: self.run_cgit(path, query) else: self.send_asset(asset) def locate_asset(self, path: str) -> Path | None: """Return the file backing a root-level asset request, or None. Only bare names at the root qualify, which is what the configs in custom/servers/ allow as well, so a repository file such as /myrepo/tree/cgit.css still reaches cgit rather than 404ing on disk. """ name = path.lstrip("/") if not name or "/" in name: return None if not name.lower().endswith(STATIC_SUFFIXES): return None candidate = (self.preview.data_dir / name).resolve() if candidate.parent != self.preview.data_dir or not candidate.is_file(): return None return candidate def send_asset(self, path: Path) -> None: data = path.read_bytes() content_type = (mimetypes.guess_type(path.name)[0] or "application/octet-stream") self.send_response(200) self.send_header("Content-Type", content_type) self.send_header("Content-Length", str(len(data))) self.end_headers() if self.command != "HEAD": _ = self.wfile.write(data) def read_body(self) -> bytes | None: """Return the request body, or None if it exceeds the cap.""" try: length = int(self.headers.get("Content-Length", 0)) except ValueError: return b"" if length <= 0: return b"" if length > MAX_BODY_BYTES: return None return self.rfile.read(length) def cgi_environ(self, path: str, query: str, body: bytes) -> dict[str, str]: env = dict(os.environ) env.update( GATEWAY_INTERFACE="CGI/1.1", SERVER_PROTOCOL="HTTP/1.1", SERVER_SOFTWARE=self.server_version, SERVER_NAME=self.preview.server_name, SERVER_PORT=str(self.preview.server_port), REQUEST_METHOD=self.command, REQUEST_URI=self.path, # An empty SCRIPT_NAME puts cgit at the root of the URL space, the # same as the server configs in custom/servers/ do. SCRIPT_NAME="", PATH_INFO=path, QUERY_STRING=query, HTTP_HOST=self.headers.get("Host", "localhost"), CGIT_CONFIG=str(self.preview.config), ) for header, variable in PASSED_HEADERS: value = self.headers.get(header) if value: env[variable] = value if body: env["CONTENT_LENGTH"] = str(len(body)) return env def run_cgit(self, path: str, query: str) -> None: body = self.read_body() if body is None: self.send_error(413, "Request body too large") return try: result = subprocess.run( [str(self.preview.cgit)], input=body, env=self.cgi_environ(path, query, body), stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=CGI_TIMEOUT, ) except subprocess.TimeoutExpired: self.send_error(504, f"cgit did not finish within {CGI_TIMEOUT}s") return # cgit reports config problems on stderr while still exiting 0, so # relay it either way rather than only on failure. if result.stderr: _ = sys.stderr.write(result.stderr.decode("latin-1", "replace")) if result.returncode != 0: self.send_error(500, f"cgit exited with status {result.returncode}") return response = split_cgi_output(result.stdout) self.send_response(response.status, response.reason) for name, value in response.headers: self.send_header(name, value) if not any(name.lower() == "content-length" for name, _ in response.headers): self.send_header("Content-Length", str(len(response.body))) self.end_headers() if self.command != "HEAD": _ = self.wfile.write(response.body) def log_message(self, format: str, *args: object) -> None: # The signature matches BaseHTTPRequestHandler, so the parameter here # shadows the builtin of the same name. _ = sys.stderr.write(f" {format % args}\n") class CgitServer(ThreadingHTTPServer): """Holds the paths the handler needs, so none are attached later on.""" def __init__(self, address: tuple[str, int], config: Path, cgit: Path, data_dir: Path) -> None: self.config: Path = config self.cgit: Path = cgit self.data_dir: Path = data_dir super().__init__(address, CgitHandler) class Options(argparse.Namespace): """Typed view of the command line, since Namespace is otherwise untyped. The defaults live here and are handed to add_argument below, so the two cannot drift apart. """ host: str = "127.0.0.1" port: int = 8080 config: str = "cgitrc" cgit: str = str(REPO_ROOT / "build" / "cgit") data: str = str(REPO_ROOT / "assets") def parse_args(argv: list[str] | None = None) -> Options: parser = argparse.ArgumentParser(description="Preview cgit locally.") _ = parser.add_argument("--config", default=Options.config, help="path to cgitrc (default: ./cgitrc)") _ = parser.add_argument("--port", type=int, default=Options.port) _ = parser.add_argument("--host", default=Options.host) _ = parser.add_argument("--cgit", default=Options.cgit, help="path to the cgit binary") _ = parser.add_argument("--data", default=Options.data, help="directory holding cgit.css, cgit.js, images") return parser.parse_args(argv, namespace=Options()) def main() -> None: options = parse_args() config = Path(options.config).resolve() cgit = Path(options.cgit).resolve() data_dir = Path(options.data).resolve() for label, path in ( ("config", config), ("cgit binary", cgit), ("data directory", data_dir), ): if not path.exists(): sys.exit(f"error: {label} not found: {path}") # Pin the two types the page depends on, so the preview matches what the # server configs send rather than whatever is in this machine's mime table. mimetypes.add_type("text/css", ".css") mimetypes.add_type("text/javascript", ".js") server = CgitServer((options.host, options.port), config, cgit, data_dir) banner = (f"cgit preview serving http://{options.host}:{options.port}/\n" f" config: {config}\n" f" press Ctrl-C to stop\n") _ = sys.stderr.write(banner) try: server.serve_forever() except KeyboardInterrupt: _ = sys.stderr.write("\nstopped\n") finally: server.server_close() if __name__ == "__main__": main()