gitoria
All repositories: gitoria
18.4 KB
// Shared HTTP utilities — compiled into each HTTP plugin at build time// Not a standalone plugin .so, but a Zig module imported by hl:http1, hl:http2, hl:http3const std = @import("std");const api = @import("plugin_api");pub const HlString = api.HlString;pub fn hlStr(s: []const u8) HlString {return .{ .ptr = s.ptr, .len = s.len };}pub fn statusReason(code: u16) []const u8 {return switch (code) {200 => "OK",201 => "Created",204 => "No Content",301 => "Moved Permanently",302 => "Found",304 => "Not Modified",400 => "Bad Request",401 => "Unauthorized",403 => "Forbidden",404 => "Not Found",405 => "Method Not Allowed",409 => "Conflict",413 => "Payload Too Large",415 => "Unsupported Media Type",422 => "Unprocessable Entity",429 => "Too Many Requests",500 => "Internal Server Error",502 => "Bad Gateway",503 => "Service Unavailable",504 => "Gateway Timeout",else => "OK",};}/// Decode percent-encoded URL components (%20 → space, etc.)pub fn urlDecode(allocator: anytype, input: []const u8) ![]u8 {var result = try allocator.alloc(u8, input.len);var out: usize = 0;var i: usize = 0;while (i < input.len) {if (input[i] == '%' and i + 2 < input.len) {const hi = hexVal(input[i + 1]) orelse {result[out] = input[i];out += 1;i += 1;continue;};const lo = hexVal(input[i + 2]) orelse {result[out] = input[i];out += 1;i += 1;continue;};result[out] = (@as(u8, hi) << 4) | @as(u8, lo);out += 1;i += 3;} else if (input[i] == '+') {result[out] = ' ';out += 1;i += 1;} else {result[out] = input[i];out += 1;i += 1;}}return result[0..out];}fn hexVal(c: u8) ?u4 {if (c >= '0' and c <= '9') return @intCast(c - '0');if (c >= 'a' and c <= 'f') return @intCast(c - 'a' + 10);if (c >= 'A' and c <= 'F') return @intCast(c - 'A' + 10);return null;}/// Split a query string into key-value pairs./// Returns allocated slices that must be freed by the caller.pub fn parseQueryString(alloc: anytype, query: []const u8) ![]QueryParam {var params: @import("std").ArrayListUnmanaged(QueryParam) = .empty;var iter = @import("std").mem.splitScalar(u8, query, '&');while (iter.next()) |pair| {if (pair.len == 0) continue;if (@import("std").mem.indexOfScalar(u8, pair, '=')) |eq| {const key = try urlDecode(alloc, pair[0..eq]);const val = try urlDecode(alloc, pair[eq + 1 ..]);try params.append(alloc, .{ .key = key, .value = val });} else {const key = try urlDecode(alloc, pair);try params.append(alloc, .{ .key = key, .value = &[_]u8{} });}}return params.toOwnedSlice(alloc);}pub const QueryParam = struct {key: []u8,value: []u8,};/// Lowercase a header key in-place (HTTP/1.1 headers are case-insensitive)pub fn toLowercase(buf: []u8) void {for (buf) |*ch| {if (ch.* >= 'A' and ch.* <= 'Z') {ch.* = ch.* + 32;}}}// ---------------------------------------------------------------------------// RESPONSE BODY COMPRESSION (creator, 2026-08-15) — a TRANSPORT concern, like// TLS: nobody writes compression in app code; a server honors Accept-Encoding.// ONE implementation here (decision D8's pattern), each plugin adds only its// own framing. gzip via Zig std flate — no external library.// ---------------------------------------------------------------------------/// Below this a gzip header+footer is overhead, not savings (canonical ~860/// from Akamai's measurements; MTU-fitting responses gain nothing).pub const compress_min_len = 860;/// The texty content types worth compressing, decided from the Content-Type/// VALUE. Already-compressed formats (images, fonts, archives) get bigger, not/// smaller. hl:http2 and hl:http3 carry headers as key/value pairs rather than/// h1's raw blob, so this is the half they can ask.pub fn compressibleTypeValue(value: []const u8) bool {const v = std.mem.trim(u8, value, " ");return std.mem.startsWith(u8, v, "text/") orstd.mem.startsWith(u8, v, "application/javascript") orstd.mem.startsWith(u8, v, "application/json") orstd.mem.startsWith(u8, v, "image/svg");}/// The same question asked of h1's assembled custom-header text.pub fn compressibleContentType(headers_blob: []const u8) bool {var it = std.mem.splitSequence(u8, headers_blob, "\r\n");while (it.next()) |line| {if (line.len > 13 and std.ascii.eqlIgnoreCase(line[0..13], "content-type:")) {return compressibleTypeValue(line[13..]);}if (line.len > 17 and std.ascii.eqlIgnoreCase(line[0..17], "content-encoding:")) return false;}// no Content-Type in custom headers → the plugin's default is text/htmlreturn true;}/// gzip `body`; null when compression does not pay (or fails — the caller then/// sends the identity bytes, which is always correct).pub fn gzipBody(allocator: std.mem.Allocator, body: []const u8) ?[]u8 {if (body.len < compress_min_len) return null;// Compress.init asserts `output.buffer.len > 8`, and a plain// Allocating.init starts EMPTY — the assert then panics on a worker thread// and kills the server after the first compressed response (found the hard// way 2026-08-15). initCapacity gives the writer its buffer up front.var out: std.Io.Writer.Allocating = std.Io.Writer.Allocating.initCapacity(allocator, 4096) catch return null;defer out.deinit();const window = allocator.alloc(u8, std.compress.flate.max_window_len) catch return null;defer allocator.free(window);// .level_6 — the documented setting; Options carries no defaults.var c = std.compress.flate.Compress.init(&out.writer, window, .gzip, .level_6) catch return null;c.writer.writeAll(body) catch return null;c.finish() catch return null;if (out.writer.buffered().len >= body.len) return null; // grew: send identityreturn allocator.dupe(u8, out.writer.buffered()) catch null;}// ── The mission-125 BELL: one eventfd per loop source ────────────────────────//// RESTORED 2026-08-29 (mission 257). `hl:fetch` and `hl:smtp` both call these by// name, and both had been UNBUILDABLE since the 2026-08-20 recovery replay// dropped them from this file — nothing noticed, because `native/build.zig` did// not build either plugin. See `native/src/loop_wait.zig` for the contract they// implement: the loop owns one epoll set over every source's `wake_fd`, adds it// LEVEL-triggered, and drains the counter itself right after `epoll_wait`. A// producer therefore only ever has to create the fd and write to it.//// Raw syscalls rather than libc, so a plugin that does not link libc (hl:http1)// can still ring a bell./// A source's wake fd: a non-blocking eventfd counter. `-1` on failure, which is/// the ABI's "no signal, keep polling" value — a plugin that cannot get an fd/// degrades to the polled path instead of failing to load.pub fn makeWakeFd() i32 {const rc = std.os.linux.eventfd(0, std.os.linux.EFD.NONBLOCK | std.os.linux.EFD.CLOEXEC);if (@as(isize, @bitCast(rc)) < 0) return -1;return @intCast(rc);}/// RING IT. Coalesced, duplicated or early wakes are harmless by design (a bell,/// not a routing key); only a wake that is never sent can be a bug — so producers/// ring generously and never reason about whether the loop "needs" it.pub fn ringWake(fd: i32) void {if (fd < 0) return;const one: u64 = 1;_ = std.os.linux.write(fd, @ptrCast(&one), @sizeOf(u64));}// ================================================================================// WHY A BIND FAILED — IN WORDS, AND WITH THE HOLDER'S NAME (mission 286)// ================================================================================// All three transports used to print the same six words — `bind() failed on port// N` — and drop the errno on the floor. Six words cannot be acted on: a port that// is taken, a port below 1024 without CAP_NET_BIND_SERVICE, and an address that is// not one of this machine's are three different mistakes with three different// fixes, and the kernel had already said which one it was.//// So the errno is named, and when it is EADDRINUSE the holder is looked up the way// an operator would look it up: `/proc/net/{tcp,tcp6,udp,udp6}` for the socket on// that port, then `/proc/<pid>/fd` for whoever owns its inode. It says "a process// this one cannot see" rather than guessing when the holder belongs to another// user — `/proc/<pid>/fd` is unreadable then, and a wrong name is worse than none.//// This costs nothing on the happy path: it runs once, on a boot that is about to// be refused.fn errnoWord(e: u32) []const u8 {return switch (e) {13 => "EACCES",22 => "EINVAL",24 => "EMFILE",97 => "EAFNOSUPPORT",98 => "EADDRINUSE",99 => "EADDRNOTAVAIL",else => "errno",};}fn errnoWhy(e: u32) []const u8 {return switch (e) {13 => "permission denied — a port below 1024 needs CAP_NET_BIND_SERVICE or root",22 => "invalid argument — this socket is already bound",24 => "too many open files — the process fd limit is reached",97 => "address family not supported",98 => "the port is already in use",99 => "that address is not one of this machine's",else => "the kernel refused the bind",};}fn hexNibble(c: u8) ?u8 {return switch (c) {'0'...'9' => c - '0','a'...'f' => c - 'a' + 10,'A'...'F' => c - 'A' + 10,else => null,};}fn hexValue(s: []const u8) ?u64 {if (s.len == 0 or s.len > 16) return null;var v: u64 = 0;for (s) |c| v = (v << 4) | (hexNibble(c) orelse return null);return v;}/// `/proc/net/*`'s local_address column ("0100007F:1F90") as text the operator/// recognises. IPv4 is four little-endian bytes; anything longer is printed as the/// raw hex word rather than mis-rendered as a v6 address it may not be.fn writeProcAddr(out: []u8, hex: []const u8, port: u16) []const u8 {if (hex.len == 8) {const v = hexValue(hex) orelse return "";const b: [4]u8 = .{@truncate(v & 0xff),@truncate((v >> 8) & 0xff),@truncate((v >> 16) & 0xff),@truncate((v >> 24) & 0xff),};return std.fmt.bufPrint(out, "{d}.{d}.{d}.{d}:{d}", .{ b[0], b[1], b[2], b[3], port }) catch "";}return std.fmt.bufPrint(out, "[{s}]:{d}", .{ hex, port }) catch "";}// RAW SYSCALLS, for the same reason the wake bell above uses them: this file is// compiled INTO three plugins, `std.fs` in Zig 0.16 wants an `Io` interface these// plugins do not carry, and `/proc` is read once, at a boot that is failing.const lin = std.os.linux;fn sysOpen(path: [*:0]const u8, dir: bool) ?i32 {const rc = lin.open(path, .{ .ACCMODE = .RDONLY, .CLOEXEC = true, .DIRECTORY = dir }, 0);const fd: i64 = @bitCast(@as(u64, rc));if (fd < 0) return null;return @intCast(fd);}fn sysClose(fd: i32) void {_ = lin.close(fd);}const SocketOnPort = struct { inode: u64, addr_hex: [40]u8, addr_len: usize };/// The socket sitting on `port`, out of the four `/proc/net` tables. For TCP only a/// LISTEN socket (`st == 0A`) can block a bind — mission 285 measured that, squat by/// squat: TIME_WAIT and ESTABLISHED are tolerated because both sides set/// SO_REUSEADDR. UDP has no states worth filtering.fn socketOnPort(port: u16, udp: bool) ?SocketOnPort {const tables: [2][*:0]const u8 = if (udp).{ "/proc/net/udp", "/proc/net/udp6" }else.{ "/proc/net/tcp", "/proc/net/tcp6" };for (tables) |path| {const fd = sysOpen(path, false) orelse continue;defer sysClose(fd);var chunk: [8192]u8 = undefined;var line: [512]u8 = undefined;var line_len: usize = 0;while (true) {const rc = lin.read(fd, &chunk, chunk.len);const got: i64 = @bitCast(@as(u64, rc));if (got <= 0) break;const n: usize = @intCast(got);for (chunk[0..n]) |ch| {if (ch != '\n') {if (line_len < line.len) {line[line_len] = ch;line_len += 1;}continue;}if (parseProcNetLine(line[0..line_len], port, udp)) |hit| return hit;line_len = 0;}}if (line_len > 0) {if (parseProcNetLine(line[0..line_len], port, udp)) |hit| return hit;}}return null;}/// One `/proc/net/*` row: `sl local rem st tx:rx tr:when retrnsmt uid timeout inode`.fn parseProcNetLine(line: []const u8, port: u16, udp: bool) ?SocketOnPort {var it = std.mem.tokenizeAny(u8, line, " \t");_ = it.next() orelse return null; // slconst local = it.next() orelse return null;_ = it.next() orelse return null; // rem_addressconst st = it.next() orelse return null;if (!udp and !std.mem.eql(u8, st, "0A")) return null;const colon = std.mem.lastIndexOfScalar(u8, local, ':') orelse return null;const lport = hexValue(local[colon + 1 ..]) orelse return null;if (lport != port) return null;var skipped: usize = 0;while (skipped < 5) : (skipped += 1) _ = it.next() orelse return null;const inode = std.fmt.parseInt(u64, it.next() orelse return null, 10) catch return null;const hex = local[0..colon];var hit = SocketOnPort{ .inode = inode, .addr_hex = undefined, .addr_len = 0 };if (hex.len > hit.addr_hex.len) return null;@memcpy(hit.addr_hex[0..hex.len], hex);hit.addr_len = hex.len;return hit;}/// Whoever has that socket inode open. An unreadable `/proc/<pid>/fd` is SKIPPED,/// not reported as absent: a holder owned by another user is invisible from here,/// and the caller says exactly that instead of naming the wrong process.fn pidHolding(inode: u64, name_out: []u8, name_len: *usize) ?u32 {const proc_fd = sysOpen("/proc", true) orelse return null;defer sysClose(proc_fd);var want: [48]u8 = undefined;const want_link = std.fmt.bufPrint(&want, "socket:[{d}]", .{inode}) catch return null;var ents: [8192]u8 align(8) = undefined;while (true) {const rc = lin.getdents64(proc_fd, &ents, ents.len);const got: i64 = @bitCast(@as(u64, rc));if (got <= 0) return null;var off: usize = 0;while (off < @as(usize, @intCast(got))) {const ent: *align(1) const lin.dirent64 = @ptrCast(&ents[off]);off += ent.reclen;const name_ptr: [*:0]const u8 = @ptrCast(&ent.name);const pid = std.fmt.parseInt(u32, std.mem.span(name_ptr), 10) catch continue;if (fdsHold(pid, want_link)) {name_len.* = readComm(pid, name_out);return pid;}}}}/// Does this pid have that socket open? An unreadable `/proc/<pid>/fd` answers/// false — the process belongs to another user and is simply not visible.fn fdsHold(pid: u32, want_link: []const u8) bool {var path: [64:0]u8 = undefined;const dir_path = std.fmt.bufPrintZ(&path, "/proc/{d}/fd", .{pid}) catch return false;const dir_fd = sysOpen(dir_path, true) orelse return false;defer sysClose(dir_fd);var ents: [4096]u8 align(8) = undefined;while (true) {const rc = lin.getdents64(dir_fd, &ents, ents.len);const got: i64 = @bitCast(@as(u64, rc));if (got <= 0) return false;var off: usize = 0;while (off < @as(usize, @intCast(got))) {const ent: *align(1) const lin.dirent64 = @ptrCast(&ents[off]);off += ent.reclen;const name_ptr: [*:0]const u8 = @ptrCast(&ent.name);var lbuf: [64]u8 = undefined;const lrc = lin.readlinkat(dir_fd, name_ptr, &lbuf, lbuf.len);const llen: i64 = @bitCast(@as(u64, lrc));if (llen <= 0) continue;if (std.mem.eql(u8, lbuf[0..@intCast(llen)], want_link)) return true;}}}fn readComm(pid: u32, out: []u8) usize {var path: [64:0]u8 = undefined;const p = std.fmt.bufPrintZ(&path, "/proc/{d}/comm", .{pid}) catch return 0;const fd = sysOpen(p, false) orelse return 0;defer sysClose(fd);const rc = lin.read(fd, out.ptr, out.len);const got: i64 = @bitCast(@as(u64, rc));if (got <= 0) return 0;var len: usize = @intCast(got);while (len > 0 and (out[len - 1] == '\n' or out[len - 1] == '\r')) len -= 1;return len;}/// The sentence a transport appends to `bind() failed on port N`. It never fails: a/// buffer too small or a `/proc` that will not answer degrades to the part that IS/// known, and the errno alone already says more than the old message did.pub fn bindFailureDetail(out: []u8, rc: usize, port: u16, udp: bool) []const u8 {const signed: i64 = @bitCast(@as(u64, rc));const errno: u32 = if (signed < 0 and signed > -4096) @intCast(-signed) else 0;var w: usize = 0;const head = std.fmt.bufPrint(out[w..], "{s} — {s}", .{ errnoWord(errno), errnoWhy(errno) }) catch return out[0..w];w += head.len;if (errno != 98) return out[0..w];const sock = socketOnPort(port, udp) orelse {const gone = std.fmt.bufPrint(out[w..], "; nothing holds it now — the listener came and went", .{}) catch return out[0..w];return out[0 .. w + gone.len];};var abuf: [80]u8 = undefined;const addr = writeProcAddr(&abuf, sock.addr_hex[0..sock.addr_len], port);var comm: [32]u8 = undefined;var comm_len: usize = 0;if (pidHolding(sock.inode, &comm, &comm_len)) |pid| {const tail = std.fmt.bufPrint(out[w..], "; held by pid {d} ({s}) on {s}", .{ pid, comm[0..comm_len], addr }) catch return out[0..w];return out[0 .. w + tail.len];}const tail = std.fmt.bufPrint(out[w..], "; held on {s} by a process this one cannot see (another user)", .{addr}) catch return out[0..w];return out[0 .. w + tail.len];}
Branches
- mainmain branch