diff --git a/lang/runtime/el_runtime.c b/lang/runtime/el_runtime.c index a1cbce5..d430a72 100644 --- a/lang/runtime/el_runtime.c +++ b/lang/runtime/el_runtime.c @@ -40,6 +40,7 @@ #include #include #include +#include /* SIGPIPE disposition: a hung-up client must not kill us */ #include /* dlsym for http_set_handler fallback */ #include #include @@ -1335,10 +1336,63 @@ static const char* http_reason_phrase(int status) { } } -/* Best-effort send with retry on partial writes. */ +/* A DISCONNECTING CLIENT MUST NOT KILL THE SERVER (2026-08-16). + * + * There was no SIGPIPE handling anywhere in this runtime: no signal disposition, + * no MSG_NOSIGNAL, no SO_NOSIGPIPE, and send() called with bare flags. The + * default disposition of SIGPIPE is to TERMINATE THE PROCESS, so any client that + * hung up mid-response — a curl that hit its timeout, a browser tab closed + * during a large read, a proxy giving up — took the whole engram down with it. + * + * Measured on the live instance: 18 boots in the log, and `launchctl list` + * reporting the previous exit for ai.neuron.engram as -13, i.e. killed by + * signal 13 = SIGPIPE. Reproduced by the cause: pulling /api/nodes/list (26 MB) + * with a client-side timeout. launchd's KeepAlive then restarts it, so the + * failure looks like a mysterious restart rather than a crash, and the graph + * silently reloads under whatever was mid-flight. + * + * This is an exemption in the §8 sense: the write never checked whether the + * peer was still there, and the consequence of not checking was fatal rather + * than merely wrong. + * + * Two layers, because neither alone is portable: + * - SO_NOSIGPIPE per socket (Darwin/BSD) and MSG_NOSIGNAL per send (Linux), + * so the signal is never raised for socket writes in the first place. + * - A process-wide SIG_IGN as the backstop for platforms/paths with neither, + * installed once and idempotent. With the signal ignored, send() returns + * -1/EPIPE and the existing error path closes the connection. */ +#ifndef MSG_NOSIGNAL +#define MSG_NOSIGNAL 0 +#endif + +static void el_ignore_sigpipe_once(void) { + static int done = 0; + if (done) return; + done = 1; +#ifndef _WIN32 + signal(SIGPIPE, SIG_IGN); +#endif +} + +/* Per-socket suppression where the platform offers it. Best-effort: a failure + * here is not fatal because el_ignore_sigpipe_once() already covers the case. */ +static void el_sock_nosigpipe(int fd) { +#if defined(SO_NOSIGPIPE) + int on = 1; + setsockopt(fd, SOL_SOCKET, SO_NOSIGPIPE, &on, sizeof(on)); +#else + (void)fd; +#endif +} + +/* Best-effort send with retry on partial writes. EPIPE/ECONNRESET are a client + * that left, not a server fault: return -1 so the caller closes the connection, + * and never let it reach the process as a signal. */ static int http_send_all(int fd, const char* p, size_t left) { + el_ignore_sigpipe_once(); while (left > 0) { - ssize_t w = send(fd, p, left, 0); + ssize_t w = send(fd, p, left, MSG_NOSIGNAL); + if (w < 0 && errno == EINTR) continue; if (w <= 0) return -1; p += w; left -= (size_t)w; } @@ -1788,6 +1842,7 @@ void http_serve(el_val_t port, el_val_t handler) { pthread_mutex_unlock(&_http_conn_mu); HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); if (!arg) { el_closesocket(cfd); continue; } + el_sock_nosigpipe(cfd); arg->fd = cfd; pthread_t tid; if (pthread_create(&tid, NULL, http_worker, arg) != 0) { @@ -1834,6 +1889,7 @@ static void* _http_serve_async_loop(void* raw) { pthread_mutex_unlock(&_http_conn_mu); HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); if (!arg) { close(cfd); continue; } + el_sock_nosigpipe(cfd); arg->fd = cfd; pthread_t tid; if (pthread_create(&tid, NULL, http_worker, arg) != 0) { @@ -2134,6 +2190,7 @@ void http_serve_v2(el_val_t port, el_val_t handler) { pthread_mutex_unlock(&_http_conn_mu); HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg)); if (!arg) { el_closesocket(cfd); continue; } + el_sock_nosigpipe(cfd); arg->fd = cfd; pthread_t tid; if (pthread_create(&tid, NULL, http_worker_v2, arg) != 0) {