Project homepage Mailing List  Warmcat.com  API Docs  Github Mirror 
    npro  
 Modern all-safe Rust Network Protocol library supporting h1, h2, h3, ws, wt sans-IO and with socket IO + tls
git clone https://npro.rs/repo/npro
 
root / src / web / CMakeLists.txt
Author[]Andy Green <andy@warmcat.com> 2026-10-05 04:32 UTC
Committer[]Andy Green <andy@warmcat.com> 2026-10-05 04:56 UTC
Tree3ee304d96c5eba3f7f5f92cb66a2c99b9f3dc729   Raw Patch
 
sai-virt: survive a stale libvirt connection and VMs that die or leak
sai-virt: survive a stale libvirt connection and VMs that die or leak

sai-virt opened its libvirt connection once and never checked it.  If
the connection went stale (the same thing virt-manager suffered, needing
a disconnect / reconnect to see the domains again), destroys failed, but
the result was ignored: the VM record was freed and its index reused
while the real domain kept running, idle.  The next spawn of that name
then collided with the leftover domain, and since a failed spawn just
gave up and sai-server only resends pending tasks when they change,
nothing ever retried, so the platform stopped being built until the
leftover was forced off by hand.

 - reopen the libvirt connection on demand when it's no longer alive
 - distinguish "domain doesn't exist" from "couldn't find out", so a
   connection problem never looks like a vanished VM
 - destroy reports failure; the VM record (and so its name, and its
   share of max_vms) is kept and the destroy retried until confirmed
 - spawn destroys a stale domain squatting on the name it picked, and
   cleans up its overlay if it fails part way
 - at startup, destroy transient sai-vm-<plat>-<n> domains left by a
   previous run, they'd never be reaped otherwise
 - a 15s watchdog reaps VMs whose domain is gone, shut off, crashed or
   paused on an I/O error (eg, /dev/shm full), and retries spawning
 - /auto-power-off replies "ACK:" like sai-power, so the builder takes
   the power-off path instead of treating it as a NAK; unknown VMs NAK
 - tolerate a few late /stay polls (90s) before killing a busy VM, the
   watchdog now covers ones that really died
 - one place tears VMs down, cancelling both timers, so a timeout can no
   longer free a VM with a pending delayed destroy

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
diff --git a/src/virt/v-http-api.c b/src/virt/v-http-api.c index 91f3887..56904b0 100644 --- a/src/virt/v-http-api.c +++ b/src/virt/v-http-api.c @@ -14,6 +14,48 @@ #include "v-private.h" +static saiv_vm_t * +saiv_find_vm(const char *name) +{ + lws_start_foreach_dll(struct lws_dll2 *, d, virt.plat_owner.head) { + saiv_plat_t *vp = lws_container_of(d, saiv_plat_t, list); + + lws_start_foreach_dll(struct lws_dll2 *, v, vp->vm_owner.head) { + saiv_vm_t *vm = lws_container_of(v, saiv_vm_t, list); + + if (!strcmp(vm->name, name)) + return vm; + } lws_end_foreach_dll(v); + } lws_end_foreach_dll(d); + + return NULL; +} + +static int +saiv_http_reply_text(struct lws *wsi, const char *text) +{ + uint8_t buf[LWS_PRE + 512], *start = buf + LWS_PRE, *p = start, + *end = buf + sizeof(buf); + size_t len = strlen(text); + + if (lws_add_http_header_status(wsi, HTTP_STATUS_OK, &p, end) || + lws_add_http_header_by_token(wsi, WSI_TOKEN_HTTP_CONTENT_TYPE, + (unsigned char *)"text/plain", 10, &p, end) || + lws_add_http_header_content_length(wsi, len, &p, end) || + lws_finalize_http_header(wsi, &p, end)) + return -1; + + if (lws_write(wsi, start, lws_ptr_diff_size_t(p, start), + LWS_WRITE_HTTP_HEADERS) < 0) + return -1; + + if (lws_write(wsi, (uint8_t *)text, len, LWS_WRITE_HTTP_FINAL) != + (int)len) + return -1; + + return -1; /* hang up */ +} + int callback_virt_http(struct lws *wsi, enum lws_callback_reasons reason, void *user, void *in, size_t len) @@ -25,73 +67,41 @@ callback_virt_http(struct lws *wsi, enum lws_callback_reasons reason, case LWS_CALLBACK_HTTP: path = (const char *)in; if (len > 16 && !strncmp(path, "/auto-power-off/", 16)) { + saiv_vm_t *found_vm; + lws_strncpy(vm_id, path + 16, sizeof(vm_id)); lwsl_notice("%s: Received auto-power-off for %s\n", __func__, vm_id); - saiv_vm_t *found_vm = NULL; - lws_start_foreach_dll(struct lws_dll2 *, d, virt.plat_owner.head) { - saiv_plat_t *vp = lws_container_of(d, saiv_plat_t, list); - lws_start_foreach_dll(struct lws_dll2 *, v, vp->vm_owner.head) { - saiv_vm_t *vm = lws_container_of(v, saiv_vm_t, list); - if (!strcmp(vm->name, vm_id)) { - found_vm = vm; - break; - } - } lws_end_foreach_dll(v); - if (found_vm) - break; - } lws_end_foreach_dll(d); - - if (found_vm) { - /* Delay destruction by 2s so sai-builder can cleanly flush its TCP FIN to sai-server */ - lws_sul_schedule(virt.context, 0, &found_vm->sul_destroy, - saiv_vm_destroy_cb, 2 * LWS_US_PER_SEC); - } - - lws_return_http_status(wsi, HTTP_STATUS_OK, NULL); - return -1; /* hang up */ + found_vm = saiv_find_vm(vm_id); + if (!found_vm) + /* builder must not wait for a power-off that won't come */ + return saiv_http_reply_text(wsi, "NAK: unknown VM"); + + /* Delay destruction by 2s so sai-builder can cleanly flush its TCP FIN to sai-server */ + lws_sul_schedule(virt.context, 0, &found_vm->sul_destroy, + saiv_vm_destroy_cb, 2 * LWS_US_PER_SEC); + + /* sai-builder only proceeds on an "ACK:" reply, like sai-power's */ + return saiv_http_reply_text(wsi, "ACK: destroying VM in 2s"); } if (len > 6 && !strncmp(path, "/stay/", 6)) { + saiv_vm_t *found_vm; + lws_strncpy(vm_id, path + 6, sizeof(vm_id)); - lwsl_notice("%s: Received stay request for %s. Replying '0' (do not stay, proceed with auto-power-off grace period)\n", __func__, vm_id); - - saiv_vm_t *found_vm = NULL; - lws_start_foreach_dll(struct lws_dll2 *, d, virt.plat_owner.head) { - saiv_plat_t *vp = lws_container_of(d, saiv_plat_t, list); - lws_start_foreach_dll(struct lws_dll2 *, v, vp->vm_owner.head) { - saiv_vm_t *vm = lws_container_of(v, saiv_vm_t, list); - if (!strcmp(vm->name, vm_id)) { - found_vm = vm; - break; - } - } lws_end_foreach_dll(v); - if (found_vm) - break; - } lws_end_foreach_dll(d); - - if (found_vm) { + lwsl_info("%s: stay request for %s\n", __func__, vm_id); + + found_vm = saiv_find_vm(vm_id); + if (found_vm) /* Extend the safety timeout since the VM is alive and communicating */ lws_sul_schedule(virt.context, 0, &found_vm->sul_timeout, - saiv_vm_timeout_cb, 30 * LWS_US_PER_SEC); - } + saiv_vm_timeout_cb, SAIV_VM_STAY_TIMEOUT_US); + else + lwsl_warn("%s: stay request from unknown VM %s\n", + __func__, vm_id); /* We never return stay = true for ephemeral VMs */ - uint8_t buf[LWS_PRE + 256], *p = buf + LWS_PRE, *end = p + 256; - - if (lws_add_http_header_status(wsi, HTTP_STATUS_OK, &p, end)) return -1; - if (lws_add_http_header_by_token(wsi, WSI_TOKEN_HTTP_CONTENT_TYPE, - (unsigned char *)"text/plain", 10, &p, end)) return -1; - if (lws_add_http_header_content_length(wsi, 1, &p, end)) return -1; - if (lws_finalize_http_header(wsi, &p, end)) return -1; - - if (lws_write(wsi, buf + LWS_PRE, (size_t)(p - (buf + LWS_PRE)), LWS_WRITE_HTTP_HEADERS) < 0) - return -1; - - uint8_t stay_res = '0'; - if (lws_write(wsi, &stay_res, 1, LWS_WRITE_HTTP_FINAL) != 1) - return -1; - return -1; /* hang up */ + return saiv_http_reply_text(wsi, "0"); } lws_return_http_status(wsi, HTTP_STATUS_NOT_FOUND, NULL); diff --git a/src/virt/v-libvirt.c b/src/virt/v-libvirt.c index 86aefe7..21a44a8 100644 --- a/src/virt/v-libvirt.c +++ b/src/virt/v-libvirt.c @@ -14,10 +14,11 @@ #include <stdio.h> #include <stdlib.h> #include <libvirt/libvirt.h> +#include <libvirt/virterror.h> #include "v-private.h" -virConnectPtr conn; +static virConnectPtr conn; static char * replace_string(const char *orig, const char *rep, const char *with) @@ -72,15 +73,132 @@ strip_xml_tags(char *xml, const char *start_tag, const char *end_tag) } } +/* + * libvirtd / virtqemud can restart, or our connection can otherwise go stale, + * under us. A dead connection stays dead, so check it and reopen on demand + * before every operation. + */ + +static virConnectPtr +saiv_libvirt_conn(void) +{ + if (conn) { + if (virConnectIsAlive(conn) == 1) + return conn; + + lwsl_warn("%s: libvirt connection is dead, reopening\n", + __func__); + virConnectClose(conn); + conn = NULL; + } + + conn = virConnectOpen("qemu:///system"); + if (!conn) + lwsl_err("%s: Failed to open connection to qemu:///system\n", + __func__); + + return conn; +} + +/* + * After a failed lookup / action, distinguish "the domain doesn't exist" from + * "we couldn't find out" (eg, connection trouble) + */ + +static int +saiv_libvirt_no_domain(void) +{ + virErrorPtr e = virGetLastError(); + + return e && e->code == VIR_ERR_NO_DOMAIN; +} + +static void +saiv_libvirt_delete_overlay(virConnectPtr c, const char *vm_name) +{ + virStoragePoolPtr pool; + virStorageVolPtr vol; + char vol_name[128]; + + pool = virStoragePoolLookupByName(c, "sai_shm"); + if (!pool) + return; + + lws_snprintf(vol_name, sizeof(vol_name), "%s.qcow2", vm_name); + vol = virStorageVolLookupByName(pool, vol_name); + if (vol) { + if (virStorageVolDelete(vol, 0) < 0) + lwsl_err("%s: failed to delete overlay %s\n", + __func__, vol_name); + virStorageVolFree(vol); + } + virStoragePoolFree(pool); +} + +/* + * Is this a domain name we would generate, ie, "sai-vm-<plat name>-<n>"? + */ + +static int +saiv_libvirt_name_is_ours(struct sai_virt *virt, const char *name) +{ + char pfx[96]; + size_t n; + + lws_start_foreach_dll(struct lws_dll2 *, d, virt->plat_owner.head) { + saiv_plat_t *vp = lws_container_of(d, saiv_plat_t, list); + const char *q; + + n = (size_t)lws_snprintf(pfx, sizeof(pfx), "sai-vm-%s-", + vp->name); + if (strncmp(name, pfx, n) || !name[n]) + continue; + + for (q = name + n; *q >= '0' && *q <= '9'; q++) + ; + if (!*q) + return 1; + } lws_end_foreach_dll(d); + + return 0; +} + static int ops_libvirt_init(struct sai_virt *virt) { - conn = virConnectOpen("qemu:///system"); - if (!conn) { - lwsl_err("Failed to open connection to qemu:///system\n"); + virDomainPtr *doms = NULL; + virConnectPtr c; + int n, i; + + c = saiv_libvirt_conn(); + if (!c) return 1; + + /* + * VMs left running by a previous sai-virt instance are unknown to us: + * their /stay and /auto-power-off would be ignored, so they would run + * forever, and their names would clash with what we spawn. + */ + + n = virConnectListAllDomains(c, &doms, + VIR_CONNECT_LIST_DOMAINS_TRANSIENT); + for (i = 0; i < n; i++) { + const char *name = virDomainGetName(doms[i]); + + if (name && saiv_libvirt_name_is_ours(virt, name)) { + lwsl_warn("%s: destroying orphaned VM %s\n", + __func__, name); + if (virDomainDestroy(doms[i]) < 0) + lwsl_err("%s: failed to destroy %s\n", + __func__, name); + saiv_libvirt_delete_overlay(c, name); + } + virDomainFree(doms[i]); } + free(doms); + lwsl_notice("%s: libvirt ops initialized\n", __func__); + return 0; } @@ -90,6 +208,7 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) virDomainPtr dom; virStoragePoolPtr pool; virStorageVolPtr vol; + virConnectPtr c; char *xml, *xml2, *xml3; char vol_xml[1024]; char overlay_path[256]; @@ -102,13 +221,33 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) lwsl_notice("%s: Spawning ephemeral VM %s for platform: %s (base %s)\n", __func__, vm->name, vm->plat->name, vm->plat->base_image); - if (!conn) + c = saiv_libvirt_conn(); + if (!c) return 1; + /* + * We pick a name nothing of ours is using... if the hypervisor still + * has a domain by that name, it's a leftover nobody will ever clean + * up, and it would make the create fail + */ + dom = virDomainLookupByName(c, vm->name); + if (dom) { + lwsl_warn("%s: stale domain %s exists, destroying it\n", + __func__, vm->name); + if (virDomainDestroy(dom) < 0 && !saiv_libvirt_no_domain() && + virDomainIsActive(dom) != 0) { + lwsl_err("%s: unable to destroy stale domain %s\n", + __func__, vm->name); + virDomainFree(dom); + return 1; + } + virDomainFree(dom); + } + /* 1. Ensure the /dev/shm storage pool exists */ - pool = virStoragePoolLookupByName(conn, "sai_shm"); + pool = virStoragePoolLookupByName(c, "sai_shm"); if (!pool) { - pool = virStoragePoolCreateXML(conn, shm_pool_xml, 0); + pool = virStoragePoolCreateXML(c, shm_pool_xml, 0); if (!pool) { lwsl_err("Failed to create transient shm storage pool\n"); return 1; @@ -166,10 +305,10 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) lws_snprintf(overlay_path, sizeof(overlay_path), "/dev/shm/%s.qcow2", vm->name); /* 3. Get base domain XML and manipulate it */ - dom = virDomainLookupByName(conn, vm->plat->name); + dom = virDomainLookupByName(c, vm->plat->name); if (!dom) { lwsl_err("Failed to find base domain %s\n", vm->plat->name); - return 1; + goto bail; } xml = virDomainGetXMLDesc(dom, 0); @@ -177,7 +316,7 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) if (!xml) { lwsl_err("Failed to get XML for base domain\n"); - return 1; + goto bail; } /* Replace <name>base</name> with <name>vm->name</name> */ @@ -194,7 +333,7 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) if (!xml3) { lwsl_err("Failed to manipulate XML\n"); - return 1; + goto bail; } /* Inject qemu namespace into <domain> */ @@ -236,7 +375,7 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) if (!xml5) { lwsl_err("Failed to manipulate XML\n"); - return 1; + goto bail; } /* Remove UUID so libvirt generates a new one, avoiding conflicts with the base VM */ @@ -245,55 +384,111 @@ ops_libvirt_spawn(struct sai_virt *virt, struct saiv_vm *vm) strip_xml_tags(xml5, "<mac address=", "/>"); /* 4. Boot the transient domain */ - dom = virDomainCreateXML(conn, xml5, 0); + dom = virDomainCreateXML(c, xml5, 0); free(xml5); if (!dom) { lwsl_err("Failed to create transient domain %s\n", vm->name); - return 1; + goto bail; } virDomainFree(dom); lwsl_notice("Successfully spawned ephemeral VM %s\n", vm->name); return 0; + +bail: + saiv_libvirt_delete_overlay(c, vm->name); + + return 1; } +/* + * Returns 0 only if the domain is confirmed gone (and its overlay deleted). + * Otherwise the caller must keep the VM's name reserved and retry later. + */ + static int ops_libvirt_destroy(struct sai_virt *virt, struct saiv_vm *vm) { virDomainPtr dom; - virStoragePoolPtr pool; - virStorageVolPtr vol; - char vol_name[128]; + virConnectPtr c; lwsl_notice("%s: Destroying ephemeral VM: %s\n", __func__, vm->name); - if (!conn) + c = saiv_libvirt_conn(); + if (!c) return 1; - dom = virDomainLookupByName(conn, vm->name); + dom = virDomainLookupByName(c, vm->name); if (dom) { - virDomainDestroy(dom); + if (virDomainDestroy(dom) < 0 && !saiv_libvirt_no_domain() && + virDomainIsActive(dom) != 0) { + lwsl_err("%s: failed to destroy %s\n", __func__, + vm->name); + virDomainFree(dom); + return 1; + } virDomainFree(dom); } else { - lwsl_warn("Domain %s not found during destroy\n", vm->name); + if (!saiv_libvirt_no_domain()) { + lwsl_err("%s: unable to look up %s\n", __func__, + vm->name); + return 1; + } + lwsl_notice("%s: domain %s already gone\n", __func__, + vm->name); } - pool = virStoragePoolLookupByName(conn, "sai_shm"); - if (pool) { - lws_snprintf(vol_name, sizeof(vol_name), "%s.qcow2", vm->name); - vol = virStorageVolLookupByName(pool, vol_name); - if (vol) { - virStorageVolDelete(vol, 0); - virStorageVolFree(vol); - } else { - lwsl_warn("Volume %s not found in pool sai_shm\n", vol_name); + saiv_libvirt_delete_overlay(c, vm->name); + + return 0; +} + +static int +ops_libvirt_alive(struct sai_virt *virt, struct saiv_vm *vm) +{ + int state, reason, r = 1; + virDomainPtr dom; + virConnectPtr c; + + c = saiv_libvirt_conn(); + if (!c) + return -1; + + dom = virDomainLookupByName(c, vm->name); + if (!dom) + return saiv_libvirt_no_domain() ? 0 : -1; + + if (virDomainGetState(dom, &state, &reason, 0) < 0) { + r = saiv_libvirt_no_domain() ? 0 : -1; + goto out; + } + + switch (state) { + case VIR_DOMAIN_SHUTOFF: + case VIR_DOMAIN_CRASHED: + r = 0; + break; + case VIR_DOMAIN_PAUSED: + /* it won't progress again, it's no use to anybody */ + if (reason == VIR_DOMAIN_PAUSED_IOERROR) { + lwsl_err("%s: %s paused on I/O error (is the overlay " + "storage in /dev/shm full?)\n", __func__, + vm->name); + r = 0; } - virStoragePoolFree(pool); + if (reason == VIR_DOMAIN_PAUSED_CRASHED) { + lwsl_err("%s: %s guest crashed\n", __func__, vm->name); + r = 0; + } + break; } - return 0; +out: + virDomainFree(dom); + + return r; } const sai_virt_ops_t ops_libvirt = { @@ -301,4 +496,5 @@ const sai_virt_ops_t ops_libvirt = { .init = ops_libvirt_init, .spawn = ops_libvirt_spawn, .destroy = ops_libvirt_destroy, + .alive = ops_libvirt_alive, }; diff --git a/src/virt/v-private.h b/src/virt/v-private.h index 7223762..470cba1 100644 --- a/src/virt/v-private.h +++ b/src/virt/v-private.h @@ -24,8 +24,19 @@ typedef struct sai_virt_ops { int (*init)(struct sai_virt *virt); int (*spawn)(struct sai_virt *virt, struct saiv_vm *vm); int (*destroy)(struct sai_virt *virt, struct saiv_vm *vm); + /* 1 = running, 0 = gone / can't make progress, -1 = can't tell */ + int (*alive)(struct sai_virt *virt, struct saiv_vm *vm); } sai_virt_ops_t; +/* a VM that never contacts us at all is given up on after this */ +#define SAIV_VM_FIRST_CONTACT_US (5 * 60 * LWS_US_PER_SEC) +/* builders poll /stay every 20s, tolerate a few late polls under load */ +#define SAIV_VM_STAY_TIMEOUT_US (90 * LWS_US_PER_SEC) +/* how often we check our VMs still exist, and retry failed spawns */ +#define SAIV_WATCH_INTERVAL_US (15 * LWS_US_PER_SEC) +/* retry interval for a VM the hypervisor didn't confirm destroyed */ +#define SAIV_DESTROY_RETRY_US (10 * LWS_US_PER_SEC) + typedef struct saiv_plat { lws_dll2_t list; char name[64]; @@ -58,6 +69,8 @@ struct sai_virt { const sai_virt_ops_t *ops; + lws_sorted_usec_list_t sul_watch; + int running_vms; int max_vms; @@ -103,6 +116,9 @@ void saiv_vm_destroy_cb(lws_sorted_usec_list_t *sul); void +saiv_watch_cb(lws_sorted_usec_list_t *sul); + +void saiv_try_spawn(void); #endif diff --git a/src/virt/v-sai.c b/src/virt/v-sai.c index 460209d..1633a1c 100644 --- a/src/virt/v-sai.c +++ b/src/virt/v-sai.c @@ -95,9 +95,6 @@ int main(int argc, const char **argv) return 1; } - virt.ops = &ops_libvirt; - virt.ops->init(&virt); - virt.vhost = lws_create_vhost(virt.context, &info); if (!virt.vhost) { lwsl_err("lws init failed\n"); @@ -107,12 +104,25 @@ int main(int argc, const char **argv) /* Parse platforms from /etc/sai/virt/conf.d */ saiv_config(&virt, "/etc/sai/virt/conf.d"); + /* + * After the platforms are known, so it can recognize VMs left over + * from a previous run. If the hypervisor isn't reachable now, the + * ops reconnect on demand later. + */ + virt.ops = &ops_libvirt; + virt.ops->init(&virt); + /* Parse global configuration from /etc/sai/virt/conf */ saiv_config_global(&virt, "/etc/sai/virt/conf"); + lws_sul_schedule(virt.context, 0, &virt.sul_watch, saiv_watch_cb, + SAIV_WATCH_INTERVAL_US); + while (!lws_service(virt.context, 0) && !interrupted) ; + lws_sul_cancel(&virt.sul_watch); + lws_start_foreach_dll_safe(struct lws_dll2 *, d, d1, virt.sai_server_owner.head) { saiv_server_t *s = lws_container_of(d, saiv_server_t, list); lws_ss_destroy(&s->ss); @@ -130,6 +140,7 @@ int main(int argc, const char **argv) virt.ops->destroy(&virt, vm); lws_dll2_remove(v); lws_sul_cancel(&vm->sul_timeout); + lws_sul_cancel(&vm->sul_destroy); free(vm); } lws_end_foreach_dll_safe(v, v1); diff --git a/src/virt/v-ws-server.c b/src/virt/v-ws-server.c index fb63f77..38099d0 100644 --- a/src/virt/v-ws-server.c +++ b/src/virt/v-ws-server.c @@ -30,22 +30,39 @@ saiv_server_tx(void *userobj, lws_ss_tx_ordinal_t ord, uint8_t *buf, size_t *len return r; } -void -saiv_vm_destroy_cb(lws_sorted_usec_list_t *sul) -{ - saiv_vm_t *vm = lws_container_of(sul, saiv_vm_t, sul_destroy); - - lwsl_notice("%s: delayed destruction of %s executing\n", __func__, vm->name); - - if (virt.ops) - virt.ops->destroy(&virt, vm); +/* + * Tear down a VM and forget it. If the hypervisor didn't confirm it's gone, + * keep the record, so its name stays reserved and it still counts against + * max_vms, and try again shortly. Callers should saiv_try_spawn() after. + */ - virt.running_vms--; +static void +saiv_vm_reap(saiv_vm_t *vm, const char *why) +{ + lwsl_notice("%s: %s: %s\n", __func__, vm->name, why); + + if (virt.ops && virt.ops->destroy(&virt, vm)) { + lwsl_err("%s: %s not confirmed destroyed, retrying in %ds\n", + __func__, vm->name, + (int)(SAIV_DESTROY_RETRY_US / LWS_US_PER_SEC)); + lws_sul_schedule(virt.context, 0, &vm->sul_destroy, + saiv_vm_destroy_cb, SAIV_DESTROY_RETRY_US); + return; + } - lws_dll2_remove(&vm->list); lws_sul_cancel(&vm->sul_timeout); + lws_sul_cancel(&vm->sul_destroy); + lws_dll2_remove(&vm->list); + virt.running_vms--; free(vm); +} + +void +saiv_vm_destroy_cb(lws_sorted_usec_list_t *sul) +{ + saiv_vm_t *vm = lws_container_of(sul, saiv_vm_t, sul_destroy); + saiv_vm_reap(vm, "delayed destruction executing"); saiv_try_spawn(); } @@ -56,15 +73,39 @@ saiv_vm_timeout_cb(lws_sorted_usec_list_t *sul) lwsl_err("%s: VM %s timed out, purging\n", __func__, vm->name); - if (virt.ops) - virt.ops->destroy(&virt, vm); + saiv_vm_reap(vm, "timed out"); + saiv_try_spawn(); +} - virt.running_vms--; +/* + * Periodically check the VMs we think we have still exist and can make + * progress, so we notice ones that died or got stuck without telling us. + * + * sai-server only sends us pending tasks when they change, so this is also + * what retries spawning after a failure. + */ - lws_dll2_remove(&vm->list); - free(vm); +void +saiv_watch_cb(lws_sorted_usec_list_t *sul) +{ + if (virt.ops && virt.ops->alive) { + lws_start_foreach_dll(struct lws_dll2 *, d, virt.plat_owner.head) { + saiv_plat_t *vp = lws_container_of(d, saiv_plat_t, list); + + lws_start_foreach_dll_safe(struct lws_dll2 *, v, v1, + vp->vm_owner.head) { + saiv_vm_t *vm = lws_container_of(v, saiv_vm_t, list); + + if (!virt.ops->alive(&virt, vm)) + saiv_vm_reap(vm, "domain is gone or stuck"); + } lws_end_foreach_dll_safe(v, v1); + } lws_end_foreach_dll(d); + } saiv_try_spawn(); + + lws_sul_schedule(virt.context, 0, &virt.sul_watch, saiv_watch_cb, + SAIV_WATCH_INTERVAL_US); } void @@ -171,6 +212,7 @@ saiv_try_spawn(void) virt.running_vms++; winner->wait_magnification = 0; if (virt.ops->spawn(&virt, vm)) { + /* saiv_watch_cb() will try again */ lwsl_err("%s: Failed to spawn VM %s\n", __func__, vm->name); lws_dll2_remove(&vm->list); free(vm); @@ -180,8 +222,10 @@ saiv_try_spawn(void) /* Clean up if it never connects and terminates itself */ lws_sul_schedule(virt.context, 0, &vm->sul_timeout, - saiv_vm_timeout_cb, 5 * 60 * LWS_US_PER_SEC); /* 5 min */ - } + saiv_vm_timeout_cb, + SAIV_VM_FIRST_CONTACT_US); + } else + break; /* OOM: don't spin */ } else { break; }
Page fetched 0s ago, creation time: 3ms (vhost etag hits: 0%, cache hits: 0%)