Author: Andy Green Date: Mon Oct 05 11:40:47 2026 +0100 builder: keep a started task's job dir between its steps The job dir hold covering the gap between a task's steps lapses after two hours unrenewed, after which disk pressure may reclaim the dir. If the next step's offer keeps getting refused for lack of space, the gap can easily be that long, although the server is still trying to give us the task. Renew the hold each time we refuse a later step of a task. Also count held job dirs as being in use when deciding whether we're idle. A builder between two steps of a task has no nspawn, so it could report itself idle and be powered down; on a sai-virt VM that throws away the whole disk, and the next step then comes to the fresh VM of the same name with no src/ tree. Co-Authored-By: Claude Opus 5.5 diff --git a/src/builder/b-deletion.c b/src/builder/b-deletion.c index f17bd3d..18dc5a5 100644 --- a/src/builder/b-deletion.c +++ b/src/builder/b-deletion.c @@ -584,6 +584,28 @@ saib_jobdir_is_held(const char *vn) return 1; } +/* + * How many job dirs are held for tasks we are still building... expired holds + * are dropped on the way + */ + +unsigned int +saib_jobdir_holds_live(void) +{ + unsigned int n = 0; + + lws_start_foreach_dll_safe(struct lws_dll2 *, p, p1, + builder.jobdir_hold_owner.head) { + saib_jobdir_hold_t *h = lws_container_of(p, saib_jobdir_hold_t, list); + + if (saib_jobdir_is_held(h->vn)) + n++; + + } lws_end_foreach_dll_safe(p, p1); + + return n; +} + void saib_jobdir_holds_destroy(void) { diff --git a/src/builder/b-power.c b/src/builder/b-power.c index 1c49793..0692dc4 100644 --- a/src/builder/b-power.c +++ b/src/builder/b-power.c @@ -591,6 +591,7 @@ saib_power_shutdown(void) int saib_reassess_idle_situation(void) { + unsigned int held; char in_use = 0; if (builder.stay) { @@ -622,6 +623,18 @@ saib_reassess_idle_situation(void) } lws_end_foreach_dll(d); } lws_end_foreach_dll(mp); + held = saib_jobdir_holds_live(); + if (held) { + /* + * Tasks we started are between steps. Going down would lose + * their job dirs (a sai-virt VM's disk goes with it) and the + * next step would come to a builder with no src/ tree. + */ + lwsl_notice("%s: %u job dirs held for started tasks\n", + __func__, held); + in_use = 1; + } + if (builder.shell_owner.head) { lwsl_notice("%s: builder has %d active shell sessions\n", __func__, builder.shell_owner.count); diff --git a/src/builder/b-private.h b/src/builder/b-private.h index d75d243..0971367 100644 --- a/src/builder/b-private.h +++ b/src/builder/b-private.h @@ -549,6 +549,8 @@ int saib_jobdir_is_held(const char *vn); void saib_jobdir_holds_destroy(void); +unsigned int +saib_jobdir_holds_live(void); /* * The job dir name for a task: the first 4 and last 4 chars of its uuid. Both diff --git a/src/builder/b-task.c b/src/builder/b-task.c index bef6bed..7eb17ea 100644 --- a/src/builder/b-task.c +++ b/src/builder/b-task.c @@ -1567,6 +1567,20 @@ saib_consider_allocating_task(struct sai_plat_server *spm, lws_struct_args_t *a, goto idle_decline; lwsl_warn("%s: builder rejects offered task\n", __func__); + + if (task->build_step > 0) { + char vn[16]; + + /* + * It's a later step of a task we started, the server + * keeps it for us and will offer it again. Its job + * dir isn't abandoned however long we keep saying not + * yet, so don't let its hold lapse meanwhile. + */ + saib_task_jobdir_vn(vn, sizeof(vn), task->uuid); + saib_jobdir_hold(vn); + } + if (saib_queue_task_status_update(sp, spm, task, 0, SAI_TASK_REASON_BUSY)) { lwsl_notice("TRAP: saib_queue_task_status_update failed (BUSY)\n");