From 00769fad7cacccb04f66d3f2973b0634ee6bae4d Mon Sep 17 00:00:00 2001 From: gsxdsm Date: Fri, 31 Jul 2026 18:13:32 -0700 Subject: [PATCH] =?UTF-8?q?fix(engine):=20run=20merges=20outside=20the=20a?= =?UTF-8?q?dmission=20drain=20=E2=80=94=20every=20merge=20froze=20planning?= =?UTF-8?q?=20admission?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Claude Fable 5 --- .changeset/merge-outside-admission-drain.md | 7 ++++++ packages/engine/src/project-engine.ts | 27 ++++++++++++++++++--- 2 files changed, 30 insertions(+), 4 deletions(-) create mode 100644 .changeset/merge-outside-admission-drain.md diff --git a/.changeset/merge-outside-admission-drain.md b/.changeset/merge-outside-admission-drain.md new file mode 100644 index 0000000000..24d8af8fb4 --- /dev/null +++ b/.changeset/merge-outside-admission-drain.md @@ -0,0 +1,7 @@ +--- +"@runfusion/fusion": patch +--- + +summary: Planning admission no longer freezes for the duration of every merge. +category: fix +dev: Root cause of the 5-10 min "Queued to plan" stalls — the merge lane ran the entire merge inside projectAdmissionCoordinator's single-flight drain, so triage's poll parked awaiting it (and its re-entrance guard then dropped every tick silently). The lane start now claims and returns; the merge body runs outside the drain. No automated regression test yet — the drain-blocking shape needs a project-engine harness; the triage poll watchdog (e51ebff381) is the backstop meanwhile. diff --git a/packages/engine/src/project-engine.ts b/packages/engine/src/project-engine.ts index c9b1da5600..c821a0bd07 100644 --- a/packages/engine/src/project-engine.ts +++ b/packages/engine/src/project-engine.ts @@ -3881,7 +3881,24 @@ export class ProjectEngine { } } let selected = false; - let value: T | undefined; + /* + FNXC:ConcurrencyAdmission 2026-08-01-01:50 (ROOT CAUSE — triage admission died during every merge): + This lane previously ran `value = await start()` INSIDE its admission `start()` callback — + i.e. the ENTIRE merge (git rebase, verification, landing: minutes, or forever when the + merge wedges) executed inside `admitOldest`'s single-flight drain. The coordinator is a + project-wide singleton and every caller awaits the previous drain, so triage's poll parked + at `await existing` for the whole merge window, its `polling` re-entrance guard stayed + closed, and every 15s tick + task:created wake dropped silently. Observed twice on the + live board as "queued to plan with open capacity" (5m50s behind FN-8627's merge; ~10min + behind FN-8635's adoption-paused landing), each ending in a batch admission the second + the merge finished. With merge pinned at 1, every merge was a planning outage. + + The lane start now only CLAIMS the admission and returns; the merge body runs after + `admitOldest` settles, outside the drain. Capacity stays honest: the merge row's own + merging/landing status is what `claimed()` counts, and at-most-once merging is enforced + by the merge lease, not by this drain. The transient admit→status-write gap is the same + one every other lane (triage `void specifyTask`, scheduler `void schedule`) already has. + */ await projectAdmissionCoordinator.admitOldest({ projectId: cwd, maxConcurrent: (await store.getSettings()).maxConcurrent ?? 2, @@ -3895,14 +3912,16 @@ export class ProjectEngine { createdAt: mergeCandidate?.createdAt, start: async () => { selected = true; - value = await start(); return true; }, }], }); if (!selected) return undefined; - projectAdmissionCoordinator.releaseReservation(taskId); - return value; + try { + return await start(); + } finally { + projectAdmissionCoordinator.releaseReservation(taskId); + } }; if (mergeStrategy === "pull-request" && this.options.processPullRequestMerge && !routeWorkspaceDirect) {