From b1668d1284482042ea2b7940f247fcea973e1ab2 Mon Sep 17 00:00:00 2001 From: Bryce Kwon Date: Wed, 12 Aug 2026 20:52:59 -1000 Subject: Split preview server output at the first blank line --- tools/serve.py | 94 ++++++++++++++++++++++++++++++---------------------------- 1 file changed, 49 insertions(+), 45 deletions(-) (limited to 'tools') diff --git a/tools/serve.py b/tools/serve.py index 99ff02f..2b2f045 100755 --- a/tools/serve.py +++ b/tools/serve.py @@ -1,20 +1,13 @@ #!/usr/bin/env python3 -"""Local preview server for cgit. - -cgit is a CGI program. It reads the request from environment variables and -writes an HTTP response to stdout. This wraps it in a small stdlib-only HTTP -server so the interface can be previewed in a browser during development, -without configuring Apache or nginx. It is a development aid, not a production -server. - -Static assets (cgit.css, cgit.js, images) are served straight from disk. Every -other request is handed to the cgit binary as CGI, with the same environment -the web server configs in custom/servers/ set up. - -Usage: - python3 tools/serve.py --config path/to/cgitrc [--port 8080] - -If --config is omitted, ./cgitrc in the current directory is used. +"""Runs the built cgit binary behind a local HTTP port for development. + +cgit is a CGI program that reads a request out of environment variables and +writes a response to stdout, so each request here becomes one run of the +binary with the same environment the server configs in custom/servers/ set +up. Requests for the files in assets are answered from disk instead, and a +cgitrc is expected, by default the one in the working directory. Nothing +outside the standard library is needed, and nothing here is meant to face a +network wider than loopback. """ from __future__ import annotations @@ -31,16 +24,16 @@ from urllib.parse import unquote REPO_ROOT = Path(__file__).resolve().parent.parent -# Bounds so a single request cannot exhaust the dev server. It only ever -# serves a browser on loopback, so these are generous. +# Bounds so a single request cannot exhaust the preview server. It only ever +# serves one browser on loopback, so these are generous. MAX_BODY_BYTES = 8 * 1024 * 1024 CGI_TIMEOUT = 60 -# Suffixes eligible to be served off disk. Everything else is a cgit URL. STATIC_SUFFIXES = (".css", ".js", ".png", ".ico", ".gif", ".jpg", ".jpeg", ".svg", ".webp", ".txt", ".woff", ".woff2") -# Request headers cgit reads, and the CGI variable each arrives as. +# The request headers cgit looks at, paired with the CGI variable each one +# has to arrive as. PASSED_HEADERS = ( ("Cookie", "HTTP_COOKIE"), ("Referer", "HTTP_REFERER"), @@ -55,21 +48,29 @@ class CgiResponse(NamedTuple): body: bytes -def split_cgi_output(raw: bytes) -> CgiResponse: +def split_cgi_output(output: bytes) -> CgiResponse: """Split a raw CGI response into its status, headers and body. - cgit ends its header block with a blank CRLF line, but a lua filter that - writes its own headers may use a bare LF, so both terminators are - accepted. Splitting on the separator rather than on a non-empty body keeps - a legitimately empty body, such as a 304, from being read as headers. + The header block ends at the first blank line. cgit ends its lines with a + bare LF while a filter writing its own headers may use CRLF, so whichever + terminator appears earliest is the real one. Looking for CRLF everywhere + before falling back to LF would instead find the first CRLF in the body, + and a commit message with DOS line endings has one, which swallowed the + whole document head into the headers. Splitting on the separator rather + than on a non-empty body keeps a legitimately empty body, such as the one + a redirect leaves behind, from being read as headers. """ - blob, separator, body = raw.partition(b"\r\n\r\n") - if not separator: - blob, separator, body = raw.partition(b"\n\n") + ends = [(at, len(sep)) for sep in (b"\r\n\r\n", b"\n\n") + if (at := output.find(sep)) >= 0] + if ends: + at, seplen = min(ends) + header_block, body = output[:at], output[at + seplen:] + else: + header_block, body = output, b"" status, reason = 200, "OK" headers: list[tuple[str, str]] = [] - for line in blob.replace(b"\r\n", b"\n").split(b"\n"): + for line in header_block.replace(b"\r\n", b"\n").split(b"\n"): if not line.strip(): continue raw_name, _, raw_value = line.partition(b":") @@ -78,9 +79,9 @@ def split_cgi_output(raw: bytes) -> CgiResponse: if name.lower() != "status": headers.append((name, value)) continue - # "Status: 404 Not Found", where the reason phrase is optional. A - # value that will not parse leaves the 200 OK default whole rather - # than pairing a stale code with a new phrase. + # For example "Status: 404 Not Found", where the reason phrase is + # optional. A value that will not parse leaves the 200 OK default + # whole rather than pairing a stale code with a new phrase. code, _, phrase = value.partition(" ") try: status = int(code) @@ -95,7 +96,7 @@ class CgitHandler(BaseHTTPRequestHandler): @property def preview(self) -> CgitServer: - """The owning server, narrowed from BaseServer for the paths it holds.""" + """The owning server, typed so the paths it carries are visible.""" return cast("CgitServer", self.server) def do_GET(self) -> None: @@ -137,7 +138,8 @@ class CgitHandler(BaseHTTPRequestHandler): def send_asset(self, path: Path) -> None: data = path.read_bytes() - content_type = mimetypes.guess_type(path.name)[0] or "application/octet-stream" + content_type = (mimetypes.guess_type(path.name)[0] + or "application/octet-stream") self.send_response(200) self.send_header("Content-Type", content_type) self.send_header("Content-Length", str(len(data))) @@ -214,19 +216,21 @@ class CgitHandler(BaseHTTPRequestHandler): self.send_response(response.status, response.reason) for name, value in response.headers: self.send_header(name, value) - if not any(name.lower() == "content-length" for name, _ in response.headers): + if not any(name.lower() == "content-length" + for name, _ in response.headers): self.send_header("Content-Length", str(len(response.body))) self.end_headers() if self.command != "HEAD": _ = self.wfile.write(response.body) def log_message(self, format: str, *args: object) -> None: - # Named to match BaseHTTPRequestHandler, shadowing the builtin. + # The signature matches BaseHTTPRequestHandler, so the parameter here + # shadows the builtin of the same name. _ = sys.stderr.write(f" {format % args}\n") class CgitServer(ThreadingHTTPServer): - """Holds the paths the handler needs, so none are attached after the fact.""" + """Holds the paths the handler needs, so none are attached later on.""" def __init__(self, address: tuple[str, int], config: Path, cgit: Path, data_dir: Path) -> None: @@ -264,11 +268,11 @@ def parse_args(argv: list[str] | None = None) -> Options: def main() -> None: - opts = parse_args() + options = parse_args() - config = Path(opts.config).resolve() - cgit = Path(opts.cgit).resolve() - data_dir = Path(opts.data).resolve() + config = Path(options.config).resolve() + cgit = Path(options.cgit).resolve() + data_dir = Path(options.data).resolve() for label, path in (("config", config), ("cgit binary", cgit), ("data directory", data_dir)): if not path.exists(): @@ -279,17 +283,17 @@ def main() -> None: mimetypes.add_type("text/css", ".css") mimetypes.add_type("text/javascript", ".js") - httpd = CgitServer((opts.host, opts.port), config, cgit, data_dir) - banner = (f"cgit preview serving http://{opts.host}:{opts.port}/\n" + server = CgitServer((options.host, options.port), config, cgit, data_dir) + banner = (f"cgit preview serving http://{options.host}:{options.port}/\n" f" config: {config}\n" f" press Ctrl-C to stop\n") _ = sys.stderr.write(banner) try: - httpd.serve_forever() + server.serve_forever() except KeyboardInterrupt: _ = sys.stderr.write("\nstopped\n") finally: - httpd.server_close() + server.server_close() if __name__ == "__main__": -- cgit v2.8.0