| Author | Andy Green <andy@warmcat.com> 2026-09-06 07:37 UTC | | Committer | Andy Green <andy@warmcat.com> 2026-09-06 07:37 UTC | | Tree | cde837dcfbdc6de2685740c28fae3a81d45e43f7 Raw Patch | | | web: browser tx backpressure and connection cap, fixes F-004 | web: browser tx backpressure and connection cap, fixes F-004
An unauthenticated peer could open ws connections to /sai/browse, stop
reading from the socket, and on each connection spam
com.warmcat.sai.taskinfo with an empty task_hash: every request queued a
full fresh overview payload into pss->raw_tx with no backpressure (only
the log-batch path checked the backlog), up to the deliberately-raised
32MiB per-pss buflist ceiling. A few hundred stalled connections against
a populated db then drive sai-web into an OOM kill.
Three complementary bounds:
- saiw_ws_browser_queue_REQUIRES_LWS_PRE() itself now enforces a
high-water mark (SAIW_BROWSER_TX_HWM, 8MiB, above the largest legit
single composed message) for every producer path: past it, the peer is
not consuming, further queuing is refused and the connection is shed
with lws_wsi_close(LWS_TO_KILL_ASYNC) so the close does not depend on
the socket ever becoming writable again.
- saiw_browser_queue_overview() defers composition while the backlog is
over SAIW_BROWSER_TX_DEFER (100KiB, the threshold the log path already
used) and retries from a 250ms sul, mirroring saiw_retry_logs: a
stalled connection gets at most one overview in flight instead of one
per incoming request. The sul is cancelled on connection close.
- LWS_CALLBACK_FILTER_PROTOCOL_CONNECTION sheds new browser connections
above SAIW_BROWSER_MAX_CONNS (100), so worst-case queued tx memory is
cap x HWM instead of unbounded in the connection count.
|
diff --git a/src/web/w-comms.c b/src/web/w-comms.c
index c378d56..9300272 100644
--- a/src/web/w-comms.c
+++ b/src/web/w-comms.c
@@ -62,6 +62,15 @@ static const char * const well_known[] = {
"/login"
};
+/*
+ * Cap on simultaneously-connected browser wss. Each connected browser can
+ * hold a queued tx backlog of up to SAIW_BROWSER_TX_HWM bytes (w-ws-browser.c)
+ * while it drains, so the count has to be bounded for worst-case memory to
+ * stay bounded too. Browsers shed at the cap simply reconnect when a slot
+ * frees.
+ */
+#define SAIW_BROWSER_MAX_CONNS 100
+
int
saiw_task_cancel(struct vhd *vhd, const char *task_uuid)
{
@@ -479,7 +488,7 @@ http_resp:
/*
* ws connections from builders and browsers
*/
- case LWS_CALLBACK_FILTER_PROTOCOL_CONNECTION:
+ case LWS_CALLBACK_FILTER_PROTOCOL_CONNECTION:
n = lws_hdr_copy(wsi, (char *)buf, sizeof(buf) - 1,
WSI_TOKEN_GET_URI);
@@ -498,6 +507,19 @@ http_resp:
}
/*
+ * Cap concurrent browser connections: the overview is
+ * public by design, so anyone can hold wss open, and
+ * each holds a bounded tx backlog while draining. At
+ * the cap, shed new connections (they retry).
+ */
+ if (vhd && vhd->browsers.count >= SAIW_BROWSER_MAX_CONNS) {
+ lwsl_wsi_notice(wsi,
+ "Shedding browser conn: at %u conns cap",
+ (unsigned int)vhd->browsers.count);
+ return 1;
+ }
+
+ /*
* Security: Cross-Site WebSocket Hijacking (CSWSH).
*
* Browser auth here is cookie-based JWT. A malicious page
@@ -720,6 +742,7 @@ http_resp:
saiw_browser_state_changed(pss, 0);
lws_dll2_remove(&pss->subs_list);
lws_sul_cancel(&pss->sul_logcache);
+ lws_sul_cancel(&pss->sul_overview);
for (n = 0; n < 4; n++) {
if (pss->last_bps[n])
diff --git a/src/web/w-private.h b/src/web/w-private.h
index a86dd3d..4f6bef7 100644
--- a/src/web/w-private.h
+++ b/src/web/w-private.h
@@ -115,6 +115,7 @@ struct pss { struct vhd *vhd;
lws_dll2_owner_t logs_owner;
lws_sorted_usec_list_t sul_logcache;
+ lws_sorted_usec_list_t sul_overview;
lws_struct_args_t a;
union {
@@ -157,6 +158,7 @@ struct pss { struct vhd *vhd;
unsigned int bulk_binary_data:1;
unsigned int toggle_favour_sch:1;
unsigned int resolved_task_offset:1;
+ unsigned int tx_shed:1;
uint8_t wants_builder_info;
sai_auth_state_t auth_state;
};
diff --git a/src/web/w-ws-browser.c b/src/web/w-ws-browser.c
index 7c57ae5..fdcc16b 100644
--- a/src/web/w-ws-browser.c
+++ b/src/web/w-ws-browser.c
@@ -187,12 +187,46 @@ enum sai_overview_state {
SOS_TASKS,
};
+/*
+ * Tx backpressure thresholds for browser connections.
+ *
+ * DEFER: producers that can retry (overview / log batches) hold off while
+ * the connection is this backed-up, and retry from a sul once it has
+ * drained below; matches the threshold the log path already used.
+ *
+ * HWM: if the backlog is still above this, the peer is simply not
+ * consuming (zero TCP window). It is set above the largest legit
+ * single composed message (an unscoped overview on a populated db
+ * can be a few MB), so reaching it means repeated messages have
+ * piled up undrained; we shed the connection instead of letting it
+ * pin memory. The raw_tx sanity limit (32MiB) remains the hard
+ * ceiling behind this.
+ */
+#define SAIW_BROWSER_TX_DEFER (100 * 1024)
+#define SAIW_BROWSER_TX_HWM (8 * 1024 * 1024)
+
int
saiw_ws_browser_queue_REQUIRES_LWS_PRE(struct pss *pss, const void *buf,
size_t len, enum lws_write_protocol flags)
{
int *pi = (int *)((const char *)buf - sizeof(int)), r = 0;
+ if (lws_buflist2_total_len(&pss->raw_tx) > SAIW_BROWSER_TX_HWM) {
+ /*
+ * The browser stopped reading and everything queued for it is
+ * still sitting here. Stop appending and have the connection
+ * closed rather than keep allocating for it.
+ */
+ if (!pss->tx_shed) {
+ pss->tx_shed = 1;
+ lwsl_wsi_notice(pss->wsi,
+ "tx backlog over HWM, shedding conn");
+ lws_wsi_close(pss->wsi, LWS_TO_KILL_ASYNC);
+ }
+
+ return 1;
+ }
+
*pi = (int)flags;
if (lws_buflist2_append_segment(&pss->raw_tx, buf - sizeof(int), len + sizeof(int)) < 0) {
@@ -1150,6 +1184,14 @@ saiw_retry_logs(lws_sorted_usec_list_t *sul)
saiw_broadcast_logs_batch(pss->vhd, pss);
}
+static void
+saiw_retry_overview(lws_sorted_usec_list_t *sul)
+{
+ struct pss *pss = lws_container_of(sul, struct pss, sul_overview);
+
+ saiw_browser_queue_overview(pss->vhd, pss);
+}
+
int
saiw_broadcast_logs_batch(struct vhd *vhd, struct pss *pss)
{
@@ -1158,7 +1200,7 @@ saiw_broadcast_logs_batch(struct vhd *vhd, struct pss *pss)
if (!pss->subs_list.owner)
return 0;
- if (lws_buflist2_total_len(&pss->raw_tx) > 100 * 1024) {
+ if (lws_buflist2_total_len(&pss->raw_tx) > SAIW_BROWSER_TX_DEFER) {
lws_sul_schedule(vhd->context, 0, &pss->sul_logcache,
saiw_retry_logs, 250 * LWS_US_PER_MS);
return 0;
@@ -1384,6 +1426,20 @@ saiw_browser_queue_overview(struct vhd *vhd, struct pss *pss)
int n;
size_t w;
+ if (lws_buflist2_total_len(&pss->raw_tx) > SAIW_BROWSER_TX_DEFER) {
+ /*
+ * Our own tx towards this browser is backed-up (he is not
+ * draining, or a previous overview is still in flight). The
+ * overview can be megabytes on a populated db, so composing
+ * another one now would just pile it onto the backlog; come
+ * back from a timer when it has drained.
+ */
+ lws_sul_schedule(vhd->context, 0, &pss->sul_overview,
+ saiw_retry_overview, 250 * LWS_US_PER_MS);
+
+ return 0;
+ }
+
filt[0] = '\0';
esc[0] = '\0';
n = -6;
|