blob: 99ff02f1ff0c0b1ff9313bdbf14a35eddc10728e (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
#!/usr/bin/env python3
"""Local preview server for cgit.

cgit is a CGI program. It reads the request from environment variables and
writes an HTTP response to stdout. This wraps it in a small stdlib-only HTTP
server so the interface can be previewed in a browser during development,
without configuring Apache or nginx. It is a development aid, not a production
server.

Static assets (cgit.css, cgit.js, images) are served straight from disk. Every
other request is handed to the cgit binary as CGI, with the same environment
the web server configs in custom/servers/ set up.

Usage:
    python3 tools/serve.py --config path/to/cgitrc [--port 8080]

If --config is omitted, ./cgitrc in the current directory is used.
"""

from __future__ import annotations

import argparse
import mimetypes
import os
import subprocess
import sys
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from typing import NamedTuple, cast
from urllib.parse import unquote

REPO_ROOT = Path(__file__).resolve().parent.parent

# Bounds so a single request cannot exhaust the dev server. It only ever
# serves a browser on loopback, so these are generous.
MAX_BODY_BYTES = 8 * 1024 * 1024
CGI_TIMEOUT = 60

# Suffixes eligible to be served off disk. Everything else is a cgit URL.
STATIC_SUFFIXES = (".css", ".js", ".png", ".ico", ".gif", ".jpg", ".jpeg",
                   ".svg", ".webp", ".txt", ".woff", ".woff2")

# Request headers cgit reads, and the CGI variable each arrives as.
PASSED_HEADERS = (
    ("Cookie", "HTTP_COOKIE"),
    ("Referer", "HTTP_REFERER"),
    ("Content-Type", "CONTENT_TYPE"),
)


class CgiResponse(NamedTuple):
    status: int
    reason: str
    headers: list[tuple[str, str]]
    body: bytes


def split_cgi_output(raw: bytes) -> CgiResponse:
    """Split a raw CGI response into its status, headers and body.

    cgit ends its header block with a blank CRLF line, but a lua filter that
    writes its own headers may use a bare LF, so both terminators are
    accepted. Splitting on the separator rather than on a non-empty body keeps
    a legitimately empty body, such as a 304, from being read as headers.
    """
    blob, separator, body = raw.partition(b"\r\n\r\n")
    if not separator:
        blob, separator, body = raw.partition(b"\n\n")

    status, reason = 200, "OK"
    headers: list[tuple[str, str]] = []
    for line in blob.replace(b"\r\n", b"\n").split(b"\n"):
        if not line.strip():
            continue
        raw_name, _, raw_value = line.partition(b":")
        name = raw_name.strip().decode("latin-1")
        value = raw_value.strip().decode("latin-1")
        if name.lower() != "status":
            headers.append((name, value))
            continue
        # "Status: 404 Not Found", where the reason phrase is optional. A
        # value that will not parse leaves the 200 OK default whole rather
        # than pairing a stale code with a new phrase.
        code, _, phrase = value.partition(" ")
        try:
            status = int(code)
        except ValueError:
            continue
        reason = phrase.strip()
    return CgiResponse(status, reason, headers, body)


class CgitHandler(BaseHTTPRequestHandler):
    server_version: str = "cgit-preview"

    @property
    def preview(self) -> CgitServer:
        """The owning server, narrowed from BaseServer for the paths it holds."""
        return cast("CgitServer", self.server)

    def do_GET(self) -> None:
        self.respond()

    def do_HEAD(self) -> None:
        self.respond()

    def do_POST(self) -> None:
        self.respond()

    def respond(self) -> None:
        raw_path, _, query = self.path.partition("?")
        # A real web server hands the CGI a decoded PATH_INFO, so decode here
        # too and keep a repository whose name needs escaping working.
        path = unquote(raw_path)
        asset = self.locate_asset(path)
        if asset is None:
            self.run_cgit(path, query)
        else:
            self.send_asset(asset)

    def locate_asset(self, path: str) -> Path | None:
        """Return the file backing a root-level asset request, or None.

        Only bare names at the root qualify, which is what the configs in
        custom/servers/ allow as well, so a repository file such as
        /myrepo/tree/cgit.css still reaches cgit rather than 404ing on disk.
        """
        name = path.lstrip("/")
        if not name or "/" in name:
            return None
        if not name.lower().endswith(STATIC_SUFFIXES):
            return None
        candidate = (self.preview.data_dir / name).resolve()
        if candidate.parent != self.preview.data_dir or not candidate.is_file():
            return None
        return candidate

    def send_asset(self, path: Path) -> None:
        data = path.read_bytes()
        content_type = mimetypes.guess_type(path.name)[0] or "application/octet-stream"
        self.send_response(200)
        self.send_header("Content-Type", content_type)
        self.send_header("Content-Length", str(len(data)))
        self.end_headers()
        if self.command != "HEAD":
            _ = self.wfile.write(data)

    def read_body(self) -> bytes | None:
        """Return the request body, or None if it exceeds the cap."""
        try:
            length = int(self.headers.get("Content-Length", 0))
        except ValueError:
            return b""
        if length <= 0:
            return b""
        if length > MAX_BODY_BYTES:
            return None
        return self.rfile.read(length)

    def cgi_environ(self, path: str, query: str, body: bytes) -> dict[str, str]:
        env = dict(os.environ)
        env.update(
            GATEWAY_INTERFACE="CGI/1.1",
            SERVER_PROTOCOL="HTTP/1.1",
            SERVER_SOFTWARE=self.server_version,
            SERVER_NAME=self.preview.server_name,
            SERVER_PORT=str(self.preview.server_port),
            REQUEST_METHOD=self.command,
            REQUEST_URI=self.path,
            # An empty SCRIPT_NAME puts cgit at the root of the URL space, the
            # same as the server configs in custom/servers/ do.
            SCRIPT_NAME="",
            PATH_INFO=path,
            QUERY_STRING=query,
            HTTP_HOST=self.headers.get("Host", "localhost"),
            CGIT_CONFIG=str(self.preview.config),
        )
        for header, variable in PASSED_HEADERS:
            value = self.headers.get(header)
            if value:
                env[variable] = value
        if body:
            env["CONTENT_LENGTH"] = str(len(body))
        return env

    def run_cgit(self, path: str, query: str) -> None:
        body = self.read_body()
        if body is None:
            self.send_error(413, "Request body too large")
            return

        try:
            result = subprocess.run(
                [str(self.preview.cgit)],
                input=body,
                env=self.cgi_environ(path, query, body),
                stdout=subprocess.PIPE,
                stderr=subprocess.PIPE,
                timeout=CGI_TIMEOUT,
            )
        except subprocess.TimeoutExpired:
            self.send_error(504, f"cgit did not finish within {CGI_TIMEOUT}s")
            return

        # cgit reports config problems on stderr while still exiting 0, so
        # relay it either way rather than only on failure.
        if result.stderr:
            _ = sys.stderr.write(result.stderr.decode("latin-1", "replace"))
        if result.returncode != 0:
            self.send_error(500, f"cgit exited with status {result.returncode}")
            return

        response = split_cgi_output(result.stdout)
        self.send_response(response.status, response.reason)
        for name, value in response.headers:
            self.send_header(name, value)
        if not any(name.lower() == "content-length" for name, _ in response.headers):
            self.send_header("Content-Length", str(len(response.body)))
        self.end_headers()
        if self.command != "HEAD":
            _ = self.wfile.write(response.body)

    def log_message(self, format: str, *args: object) -> None:
        # Named to match BaseHTTPRequestHandler, shadowing the builtin.
        _ = sys.stderr.write(f"  {format % args}\n")


class CgitServer(ThreadingHTTPServer):
    """Holds the paths the handler needs, so none are attached after the fact."""

    def __init__(self, address: tuple[str, int], config: Path, cgit: Path,
                 data_dir: Path) -> None:
        self.config: Path = config
        self.cgit: Path = cgit
        self.data_dir: Path = data_dir
        super().__init__(address, CgitHandler)


class Options(argparse.Namespace):
    """Typed view of the command line, since Namespace is otherwise untyped.

    The defaults live here and are handed to add_argument below, so the two
    cannot drift apart.
    """

    host: str = "127.0.0.1"
    port: int = 8080
    config: str = "cgitrc"
    cgit: str = str(REPO_ROOT / "build" / "cgit")
    data: str = str(REPO_ROOT / "assets")


def parse_args(argv: list[str] | None = None) -> Options:
    parser = argparse.ArgumentParser(description="Preview cgit locally.")
    _ = parser.add_argument("--config", default=Options.config,
                            help="path to cgitrc (default: ./cgitrc)")
    _ = parser.add_argument("--port", type=int, default=Options.port)
    _ = parser.add_argument("--host", default=Options.host)
    _ = parser.add_argument("--cgit", default=Options.cgit,
                            help="path to the cgit binary")
    _ = parser.add_argument("--data", default=Options.data,
                            help="directory holding cgit.css, cgit.js, images")
    return parser.parse_args(argv, namespace=Options())


def main() -> None:
    opts = parse_args()

    config = Path(opts.config).resolve()
    cgit = Path(opts.cgit).resolve()
    data_dir = Path(opts.data).resolve()
    for label, path in (("config", config), ("cgit binary", cgit),
                        ("data directory", data_dir)):
        if not path.exists():
            sys.exit(f"error: {label} not found: {path}")

    # Pin the two types the page depends on, so the preview matches what the
    # server configs send rather than whatever is in this machine's mime table.
    mimetypes.add_type("text/css", ".css")
    mimetypes.add_type("text/javascript", ".js")

    httpd = CgitServer((opts.host, opts.port), config, cgit, data_dir)
    banner = (f"cgit preview serving http://{opts.host}:{opts.port}/\n"
              f"  config: {config}\n"
              f"  press Ctrl-C to stop\n")
    _ = sys.stderr.write(banner)
    try:
        httpd.serve_forever()
    except KeyboardInterrupt:
        _ = sys.stderr.write("\nstopped\n")
    finally:
        httpd.server_close()


if __name__ == "__main__":
    main()