Compare commits

...

1 Commits

Author SHA1 Message Date
Neuron cace6a5ebf runtime: a disconnecting client must not kill the server
El SDK CI - dev / build-and-test (pull_request) Failing after 13m45s
There was no SIGPIPE handling anywhere in this runtime: no signal
disposition, no MSG_NOSIGNAL, no SO_NOSIGPIPE, and send() called with bare
flags. The default disposition of SIGPIPE is to TERMINATE THE PROCESS, so
any client that hangs up mid-response takes the whole engram with it.

MEASURED, and it is not hypothetical. Production has restarted 254 times
since 2026-08-13T19:37 at a flat ~10 minute cadence:

  17:05:18  17:15:29  17:25:38  17:35:50  17:46:00  17:56:10  18:06:22  18:16:30

Intervals of 10m09s-10m12s, not 10m00s. That excess is the whole story:
ai.neuron.engram-tick has StartInterval 600, and engram-tick.sh:13 calls

  curl -s -m10 -X POST .../api/tick

The beat does not finish within 10s over 13,634 nodes, so curl waits its
full timeout and closes. The engram then writes the tick response to a dead
socket, takes SIGPIPE, and dies. launchd KeepAlive restarts it, so the
failure presents as a mysterious restart rather than a crash — and
~/.neuron/logs/engram.log records nothing but "[http] listening on" 254
times, with no exit reason. launchctl list confirms the last exit as -13.

Root cause is one level out: consolidation had no owner, so an external
ticker was created to poke it, and the ticker is what kills it. The fix
here does not address that; it makes the process survivable while it is
addressed.

Two layers, because neither alone is portable:
  - SO_NOSIGPIPE per accepted socket (Darwin/BSD) and MSG_NOSIGNAL per send
    (Linux), so the signal is never raised for socket writes at all.
  - A process-wide SIG_IGN backstop, installed once and idempotent, for
    platforms and paths with neither. With the signal ignored, send()
    returns -1/EPIPE and the existing error path closes the connection.

Also retries send() on EINTR, which the previous loop treated as fatal.

This is an exemption in the sense of lang/spec §8: the write never checked
whether the peer was still there, and the consequence of not checking was
fatal rather than merely wrong.
2026-08-16 13:25:14 -05:00
+59 -2
View File
@@ -40,6 +40,7 @@
#include <sys/stat.h>
#include <netinet/in.h>
#include <arpa/inet.h>
#include <signal.h> /* SIGPIPE disposition: a hung-up client must not kill us */
#include <dlfcn.h> /* dlsym for http_set_handler fallback */
#include <unistd.h>
#include <fcntl.h>
@@ -1335,10 +1336,63 @@ static const char* http_reason_phrase(int status) {
}
}
/* Best-effort send with retry on partial writes. */
/* A DISCONNECTING CLIENT MUST NOT KILL THE SERVER (2026-08-16).
*
* There was no SIGPIPE handling anywhere in this runtime: no signal disposition,
* no MSG_NOSIGNAL, no SO_NOSIGPIPE, and send() called with bare flags. The
* default disposition of SIGPIPE is to TERMINATE THE PROCESS, so any client that
* hung up mid-response a curl that hit its timeout, a browser tab closed
* during a large read, a proxy giving up took the whole engram down with it.
*
* Measured on the live instance: 18 boots in the log, and `launchctl list`
* reporting the previous exit for ai.neuron.engram as -13, i.e. killed by
* signal 13 = SIGPIPE. Reproduced by the cause: pulling /api/nodes/list (26 MB)
* with a client-side timeout. launchd's KeepAlive then restarts it, so the
* failure looks like a mysterious restart rather than a crash, and the graph
* silently reloads under whatever was mid-flight.
*
* This is an exemption in the §8 sense: the write never checked whether the
* peer was still there, and the consequence of not checking was fatal rather
* than merely wrong.
*
* Two layers, because neither alone is portable:
* - SO_NOSIGPIPE per socket (Darwin/BSD) and MSG_NOSIGNAL per send (Linux),
* so the signal is never raised for socket writes in the first place.
* - A process-wide SIG_IGN as the backstop for platforms/paths with neither,
* installed once and idempotent. With the signal ignored, send() returns
* -1/EPIPE and the existing error path closes the connection. */
#ifndef MSG_NOSIGNAL
#define MSG_NOSIGNAL 0
#endif
static void el_ignore_sigpipe_once(void) {
static int done = 0;
if (done) return;
done = 1;
#ifndef _WIN32
signal(SIGPIPE, SIG_IGN);
#endif
}
/* Per-socket suppression where the platform offers it. Best-effort: a failure
* here is not fatal because el_ignore_sigpipe_once() already covers the case. */
static void el_sock_nosigpipe(int fd) {
#if defined(SO_NOSIGPIPE)
int on = 1;
setsockopt(fd, SOL_SOCKET, SO_NOSIGPIPE, &on, sizeof(on));
#else
(void)fd;
#endif
}
/* Best-effort send with retry on partial writes. EPIPE/ECONNRESET are a client
* that left, not a server fault: return -1 so the caller closes the connection,
* and never let it reach the process as a signal. */
static int http_send_all(int fd, const char* p, size_t left) {
el_ignore_sigpipe_once();
while (left > 0) {
ssize_t w = send(fd, p, left, 0);
ssize_t w = send(fd, p, left, MSG_NOSIGNAL);
if (w < 0 && errno == EINTR) continue;
if (w <= 0) return -1;
p += w; left -= (size_t)w;
}
@@ -1788,6 +1842,7 @@ void http_serve(el_val_t port, el_val_t handler) {
pthread_mutex_unlock(&_http_conn_mu);
HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg));
if (!arg) { el_closesocket(cfd); continue; }
el_sock_nosigpipe(cfd);
arg->fd = cfd;
pthread_t tid;
if (pthread_create(&tid, NULL, http_worker, arg) != 0) {
@@ -1834,6 +1889,7 @@ static void* _http_serve_async_loop(void* raw) {
pthread_mutex_unlock(&_http_conn_mu);
HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg));
if (!arg) { close(cfd); continue; }
el_sock_nosigpipe(cfd);
arg->fd = cfd;
pthread_t tid;
if (pthread_create(&tid, NULL, http_worker, arg) != 0) {
@@ -2134,6 +2190,7 @@ void http_serve_v2(el_val_t port, el_val_t handler) {
pthread_mutex_unlock(&_http_conn_mu);
HttpWorkerArg* arg = malloc(sizeof(HttpWorkerArg));
if (!arg) { el_closesocket(cfd); continue; }
el_sock_nosigpipe(cfd);
arg->fd = cfd;
pthread_t tid;
if (pthread_create(&tid, NULL, http_worker_v2, arg) != 0) {