blob: a39d03f38f9a1be10b677e8e5a45e852ed8ae446 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
#!/usr/bin/env python3
"""Runs the built cgit binary behind a local HTTP port for development.

cgit is a CGI program that reads a request out of environment variables and
writes a response to stdout, so each request here becomes one run of the
binary with the same environment the server configs in custom/servers/ set
up. Requests for the files in assets are answered from disk instead, and a
cgitrc is expected, by default the one in the working directory. Nothing
outside the standard library is needed, and nothing here is meant to face a
network wider than loopback.
"""

from __future__ import annotations

import argparse
import mimetypes
import os
import subprocess
import sys
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from typing import NamedTuple, cast
from urllib.parse import unquote

REPO_ROOT = Path(__file__).resolve().parent.parent

# Bounds so a single request cannot exhaust the preview server. It only ever
# serves one browser on loopback, so these are generous.
MAX_BODY_BYTES = 8 * 1024 * 1024
CGI_TIMEOUT = 60

STATIC_SUFFIXES = (
    ".css",
    ".js",
    ".png",
    ".ico",
    ".gif",
    ".jpg",
    ".jpeg",
    ".svg",
    ".webp",
    ".txt",
    ".woff",
    ".woff2",
)

# The request headers cgit looks at, paired with the CGI variable each one
# has to arrive as.
PASSED_HEADERS = (
    ("Cookie", "HTTP_COOKIE"),
    ("Referer", "HTTP_REFERER"),
    ("Content-Type", "CONTENT_TYPE"),
)


class CgiResponse(NamedTuple):
    status: int
    reason: str
    headers: list[tuple[str, str]]
    body: bytes


def split_cgi_output(output: bytes) -> CgiResponse:
    """Split a raw CGI response into its status, headers and body.

    The header block ends at the first blank line. cgit ends its lines with a
    bare LF while a filter writing its own headers may use CRLF, so whichever
    terminator appears earliest is the real one. Looking for CRLF everywhere
    before falling back to LF would instead find the first CRLF in the body,
    and a commit message with DOS line endings has one, which swallowed the
    whole document head into the headers. Splitting on the separator rather
    than on a non-empty body keeps a legitimately empty body, such as the one
    a redirect leaves behind, from being read as headers.
    """
    ends = [(at, len(sep)) for sep in (b"\r\n\r\n", b"\n\n")
            if (at := output.find(sep)) >= 0]
    if ends:
        at, seplen = min(ends)
        header_block, body = output[:at], output[at + seplen:]
    else:
        header_block, body = output, b""

    status, reason = 200, "OK"
    headers: list[tuple[str, str]] = []
    for line in header_block.replace(b"\r\n", b"\n").split(b"\n"):
        if not line.strip():
            continue
        raw_name, _, raw_value = line.partition(b":")
        name = raw_name.strip().decode("latin-1")
        value = raw_value.strip().decode("latin-1")
        if name.lower() != "status":
            headers.append((name, value))
            continue
        # For example "Status: 404 Not Found", where the reason phrase is
        # optional. A value that will not parse leaves the 200 OK default
        # whole rather than pairing a stale code with a new phrase.
        code, _, phrase = value.partition(" ")
        try:
            status = int(code)
        except ValueError:
            continue
        reason = phrase.strip()
    return CgiResponse(status, reason, headers, body)


class CgitHandler(BaseHTTPRequestHandler):
    server_version: str = "cgit-preview"

    @property
    def preview(self) -> CgitServer:
        """The owning server, typed so the paths it carries are visible."""
        return cast("CgitServer", self.server)

    def do_GET(self) -> None:
        self.respond()

    def do_HEAD(self) -> None:
        self.respond()

    def do_POST(self) -> None:
        self.respond()

    def respond(self) -> None:
        raw_path, _, query = self.path.partition("?")
        # A real web server hands the CGI a decoded PATH_INFO, so decode here
        # too and keep a repository whose name needs escaping working.
        path = unquote(raw_path)
        asset = self.locate_asset(path)
        if asset is None:
            self.run_cgit(path, query)
        else:
            self.send_asset(asset)

    def locate_asset(self, path: str) -> Path | None:
        """Return the file backing a root-level asset request, or None.

        Only bare names at the root qualify, which is what the configs in
        custom/servers/ allow as well, so a repository file such as
        /myrepo/tree/cgit.css still reaches cgit rather than 404ing on disk.
        """
        name = path.lstrip("/")
        if not name or "/" in name:
            return None
        if not name.lower().endswith(STATIC_SUFFIXES):
            return None
        candidate = (self.preview.data_dir / name).resolve()
        if candidate.parent != self.preview.data_dir or not candidate.is_file():
            return None
        return candidate

    def send_asset(self, path: Path) -> None:
        data = path.read_bytes()
        content_type = (mimetypes.guess_type(path.name)[0]
                        or "application/octet-stream")
        self.send_response(200)
        self.send_header("Content-Type", content_type)
        self.send_header("Content-Length", str(len(data)))
        self.end_headers()
        if self.command != "HEAD":
            _ = self.wfile.write(data)

    def read_body(self) -> bytes | None:
        """Return the request body, or None if it exceeds the cap."""
        try:
            length = int(self.headers.get("Content-Length", 0))
        except ValueError:
            return b""
        if length <= 0:
            return b""
        if length > MAX_BODY_BYTES:
            return None
        return self.rfile.read(length)

    def cgi_environ(self, path: str, query: str, body: bytes) -> dict[str, str]:
        env = dict(os.environ)
        env.update(
            GATEWAY_INTERFACE="CGI/1.1",
            SERVER_PROTOCOL="HTTP/1.1",
            SERVER_SOFTWARE=self.server_version,
            SERVER_NAME=self.preview.server_name,
            SERVER_PORT=str(self.preview.server_port),
            REQUEST_METHOD=self.command,
            REQUEST_URI=self.path,
            # An empty SCRIPT_NAME puts cgit at the root of the URL space, the
            # same as the server configs in custom/servers/ do.
            SCRIPT_NAME="",
            PATH_INFO=path,
            QUERY_STRING=query,
            HTTP_HOST=self.headers.get("Host", "localhost"),
            CGIT_CONFIG=str(self.preview.config),
        )
        for header, variable in PASSED_HEADERS:
            value = self.headers.get(header)
            if value:
                env[variable] = value
        if body:
            env["CONTENT_LENGTH"] = str(len(body))
        return env

    def run_cgit(self, path: str, query: str) -> None:
        body = self.read_body()
        if body is None:
            self.send_error(413, "Request body too large")
            return

        try:
            result = subprocess.run(
                [str(self.preview.cgit)],
                input=body,
                env=self.cgi_environ(path, query, body),
                stdout=subprocess.PIPE,
                stderr=subprocess.PIPE,
                timeout=CGI_TIMEOUT,
            )
        except subprocess.TimeoutExpired:
            self.send_error(504, f"cgit did not finish within {CGI_TIMEOUT}s")
            return

        # cgit reports config problems on stderr while still exiting 0, so
        # relay it either way rather than only on failure.
        if result.stderr:
            _ = sys.stderr.write(result.stderr.decode("latin-1", "replace"))
        if result.returncode != 0:
            self.send_error(500, f"cgit exited with status {result.returncode}")
            return

        response = split_cgi_output(result.stdout)
        self.send_response(response.status, response.reason)
        for name, value in response.headers:
            self.send_header(name, value)
        if not any(name.lower() == "content-length"
                   for name, _ in response.headers):
            self.send_header("Content-Length", str(len(response.body)))
        self.end_headers()
        if self.command != "HEAD":
            _ = self.wfile.write(response.body)

    def log_message(self, format: str, *args: object) -> None:
        # The signature matches BaseHTTPRequestHandler, so the parameter here
        # shadows the builtin of the same name.
        _ = sys.stderr.write(f"  {format % args}\n")


class CgitServer(ThreadingHTTPServer):
    """Holds the paths the handler needs, so none are attached later on."""

    def __init__(self, address: tuple[str, int], config: Path, cgit: Path,
                 data_dir: Path) -> None:
        self.config: Path = config
        self.cgit: Path = cgit
        self.data_dir: Path = data_dir
        super().__init__(address, CgitHandler)


class Options(argparse.Namespace):
    """Typed view of the command line, since Namespace is otherwise untyped.

    The defaults live here and are handed to add_argument below, so the two
    cannot drift apart.
    """

    host: str = "127.0.0.1"
    port: int = 8080
    config: str = "cgitrc"
    cgit: str = str(REPO_ROOT / "build" / "cgit")
    data: str = str(REPO_ROOT / "assets")


def parse_args(argv: list[str] | None = None) -> Options:
    parser = argparse.ArgumentParser(description="Preview cgit locally.")
    _ = parser.add_argument("--config", default=Options.config,
                            help="path to cgitrc (default: ./cgitrc)")
    _ = parser.add_argument("--port", type=int, default=Options.port)
    _ = parser.add_argument("--host", default=Options.host)
    _ = parser.add_argument("--cgit", default=Options.cgit,
                            help="path to the cgit binary")
    _ = parser.add_argument("--data", default=Options.data,
                            help="directory holding cgit.css, cgit.js, images")
    return parser.parse_args(argv, namespace=Options())


def main() -> None:
    options = parse_args()

    config = Path(options.config).resolve()
    cgit = Path(options.cgit).resolve()
    data_dir = Path(options.data).resolve()
    for label, path in (
        ("config", config),
        ("cgit binary", cgit),
        ("data directory", data_dir),
    ):
        if not path.exists():
            sys.exit(f"error: {label} not found: {path}")

    # Pin the two types the page depends on, so the preview matches what the
    # server configs send rather than whatever is in this machine's mime table.
    mimetypes.add_type("text/css", ".css")
    mimetypes.add_type("text/javascript", ".js")

    server = CgitServer((options.host, options.port), config, cgit, data_dir)
    banner = (f"cgit preview serving http://{options.host}:{options.port}/\n"
              f"  config: {config}\n"
              f"  press Ctrl-C to stop\n")
    _ = sys.stderr.write(banner)
    try:
        server.serve_forever()
    except KeyboardInterrupt:
        _ = sys.stderr.write("\nstopped\n")
    finally:
        server.server_close()


if __name__ == "__main__":
    main()