From cda9532c3b437ff1446ae1c52d40cae2e5aa9e3c Mon Sep 17 00:00:00 2001 From: gsxdsm Date: Thu, 9 Jul 2026 08:13:33 -0700 Subject: [PATCH] FN-7718: fix zombie heartbeat timers surviving agent stop/start Ensures stopping and restarting an agent durably clears its heartbeat timer instead of relying on the later FN-7645 watchdog repair. - HeartbeatTriggerScheduler.auditTimerRegistrations now unregisters lingering timers for non-eligible (stopped/paused/disabled) agents - syncTimerForAgent force-re-arms a stale present timer on a start transition so no orphaned timer entry lingers - Added 308 lines of new heartbeat-scheduler regression tests covering the stop/start zombie-timer scenarios - Added changeset (patch) documenting the fix - Updated docs/agents.md and docs/architecture.md to describe the new invariant Files changed: .changeset/fn-7718-zombie-timer-invalidate.md | 7 + docs/agents.md | 2 + docs/architecture.md | 1 + .../src/__tests__/heartbeat-scheduler.test.ts | 308 +++++++++++++++++++++ packages/engine/src/agent-heartbeat.ts | 49 +++- 5 files changed, 364 insertions(+), 3 deletions(-) Fusion-Task-Id: FN-7718 Fusion-Task-Lineage: fc834ccd-495e-4294-805d-325b4cb536a2 Co-authored-by: Fusion (runfusion.ai) --- .changeset/fn-7718-zombie-timer-invalidate.md | 7 + docs/agents.md | 2 + docs/architecture.md | 1 + .../src/__tests__/heartbeat-scheduler.test.ts | 308 ++++++++++++++++++ packages/engine/src/agent-heartbeat.ts | 49 ++- 5 files changed, 364 insertions(+), 3 deletions(-) create mode 100644 .changeset/fn-7718-zombie-timer-invalidate.md diff --git a/.changeset/fn-7718-zombie-timer-invalidate.md b/.changeset/fn-7718-zombie-timer-invalidate.md new file mode 100644 index 0000000000..c1be6f0580 --- /dev/null +++ b/.changeset/fn-7718-zombie-timer-invalidate.md @@ -0,0 +1,7 @@ +--- +"@runfusion/fusion": patch +--- + +summary: Fix agents needing repeated stop/start because a stopped agent's heartbeat timer was never fully cleared. +category: fix +dev: HeartbeatTriggerScheduler.auditTimerRegistrations now unregisters lingering timers for non-eligible (stopped/paused/disabled) agents, and syncTimerForAgent force-re-arms a stale present timer on a start transition, so a stop/start durably clears the zombie-timer condition instead of deferring to the FN-7645 watchdog repair (FN-7718). diff --git a/docs/agents.md b/docs/agents.md index d5040e1207..6a7192d1f8 100644 --- a/docs/agents.md +++ b/docs/agents.md @@ -1286,6 +1286,8 @@ Dashboard surfacing path: - `useAgents` already refreshes on `agent:updated`/`agent:stateChanged` (`packages/dashboard/app/hooks/useAgents.ts`) - No new SSE event is introduced; stale durable agents become dashboard-visible as `Unresponsive` through the existing refresh path +**Orphaned-timer invalidation on stop (FN-7718):** CLI-driven `fn agent stop`/`start` mutate the agent row from a separate process, so the in-process `agent:updated` listener never fires for those transitions — the 60s audit is the only cross-process reconciliation path. The audit now unregisters a lingering timer entry for any agent that fails eligibility (non-tickable state, `runtimeConfig.enabled === false`, or ephemeral/non-heartbeat-managed) instead of skipping past it, so a stopped agent never keeps an orphaned/"zombie" registration. `syncTimerForAgent`'s in-process start seam mirrors this: an eligible agent with a present-but-stale timer entry is force-cleared and re-armed rather than left as a no-op. This means a `stop`/`start` cycle is a durable fix — it no longer relies on the FN-7645 stale-repair path eventually catching the drift minutes later. + ### Timer State Lifecycle (FN-2289) Heartbeat timers are armed for agents in valid working states and remain armed across state transitions: diff --git a/docs/architecture.md b/docs/architecture.md index c07c9288fa..11eac9c829 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -1353,6 +1353,7 @@ Limits are controlled by project settings (`maxSpawnedAgentsPerParent`, `maxSpaw - on-demand runs - Assignment triggers skipped because a heartbeat run is already active are deferred and re-fired from `HeartbeatMonitor.onRunCompleted`, preserving the existing completion recovery path while avoiding timer-dependent stalls. - FN-7645: `HeartbeatTriggerScheduler.auditTimerRegistrations` (60s cadence) repairs both MISSING timer registrations and "zombie" ones — a timer map entry that stays present after its underlying `setInterval` silently stops firing. When a tickable, heartbeat-managed agent's `lastHeartbeatAt` exceeds the repair-stale threshold (`heartbeatIntervalMs * heartbeatRepairStaleMultiplier`, default 2x) even though a timer entry already exists, the audit clears and re-registers it (phase-aligned via `computeInitialDelayMs`), logging `reason=zombie-timer-rearmed`. This closes the gap where long-interval (~1h) agents could silently drift stale for hours because their sparse cadence meant a single lost tick was never re-armed by the previous missing-registration-only repair; short-interval agents were unaffected because their frequent ticks self-heal within minutes. All existing guards are preserved: pause suppression (`globalPause`/`enginePaused`) still gates dispatch in `onTimerTick`, FN-4119 stale active-run reaping still runs first for agents with a live run, ephemeral/task-worker agents stay excluded via `isTimerEligibleAgent`, and `registrationEpochs` staleness protection is untouched. +- FN-7718: CLI-driven `fn agent stop`/`start` mutate the agent row from a SEPARATE process, so the in-process `agent:updated` listener never fires for those transitions — the 60s audit is the ONLY cross-process reconciliation path. The audit now invalidates a stopped/non-eligible agent's lingering timer entry (state made non-tickable, `runtimeConfig.enabled === false`, or ephemeral/`!isHeartbeatManaged`) instead of bare-`continue`ing past it, so the entry never survives to become an orphaned/"zombie" registration. `syncTimerForAgent` mirrors this for the in-process start seam: an eligible agent whose present timer entry is already stale beyond the same repair threshold is force-cleared and re-armed rather than left in place by the "already ticking" no-op. Net effect: a `stop`/`start` cycle durably clears the zombie-timer condition in one audit cycle instead of deferring repair to the FN-7645 stale-repair path minutes later. ### Custom instructions `packages/engine/src/agent-instructions.ts` resolves per-agent instruction text/path with path-traversal and extension validation. diff --git a/packages/engine/src/__tests__/heartbeat-scheduler.test.ts b/packages/engine/src/__tests__/heartbeat-scheduler.test.ts index 0e762b184f..17ded011ac 100644 --- a/packages/engine/src/__tests__/heartbeat-scheduler.test.ts +++ b/packages/engine/src/__tests__/heartbeat-scheduler.test.ts @@ -228,6 +228,72 @@ describe("HeartbeatTriggerScheduler", () => { eventStore.emit("agent:created", afterStop); expect(scheduler.getRegisteredAgents()).not.toContain(afterStop.id); }); + + it("FN-7718: force re-arms a stale present timer entry on an in-process start transition instead of no-oping", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + // Long interval so the default 2x-multiplier stale threshold (7.2M ms) is + // easy to cross deterministically within the test. + const agent = baseAgent("agent-stale-start", { + runtimeConfig: { enabled: true, heartbeatIntervalMs: 3_600_000 }, + lastHeartbeatAt: "2026-01-01T00:00:00.000Z", + }); + const eventStore = createLifecycleStore([agent]); + scheduler = new HeartbeatTriggerScheduler(eventStore as unknown as AgentStore, callback); + scheduler.start(); + eventStore.emit("agent:created", agent); + expect(scheduler.getRegisteredAgents()).toContain(agent.id); + + // Advance well past the stale threshold with no lifecycle event firing — + // the timer entry stays present in `this.timers` (a live audit cycle + // would normally repair this, but this test isolates the in-process + // syncTimerForAgent seam specifically, independent of the audit). + await vi.advanceTimersByTimeAsync(8 * 60 * 60 * 1000); // 8 hours, no ticks fired (interval never elapses) + callback.mockClear(); + vi.mocked(heartbeatLog.warn).mockClear(); + + // Simulate an in-process start transition (e.g. agent:updated firing for + // an in-process-driven resume) while the stale entry from before is still + // present. Previously syncTimerForAgent's bare `this.timers.has(...)` + // check would no-op here and leave the stale entry untouched. + const started = { ...eventStore.agents.get(agent.id)!, state: "active" as const }; + eventStore.agents.set(agent.id, started); + eventStore.emit("agent:updated", started); + + expect(heartbeatLog.warn).toHaveBeenCalledWith(expect.stringContaining("Timer sync force re-armed stale present entry")); + expect(scheduler.getRegisteredAgents()).toContain(agent.id); + + // Exactly one entry results — registerAgent clears before re-arming. + const timers = (scheduler as unknown as { timers: Map }).timers; + expect(timers.has(agent.id)).toBe(true); + }); + + it("FN-7718: does not force re-arm a healthy fresh present timer entry on an unrelated in-process update", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + const agent = baseAgent("agent-fresh-update", { + runtimeConfig: { enabled: true, heartbeatIntervalMs: 3_600_000 }, + lastHeartbeatAt: "2026-01-01T00:00:00.000Z", + }); + const eventStore = createLifecycleStore([agent]); + scheduler = new HeartbeatTriggerScheduler(eventStore as unknown as AgentStore, callback); + scheduler.start(); + eventStore.emit("agent:created", agent); + expect(scheduler.getRegisteredAgents()).toContain(agent.id); + + // Well within the stale threshold — an unrelated update must not reset + // the interval or force a re-arm. + await vi.advanceTimersByTimeAsync(30 * 60_000); // 30 minutes + vi.mocked(heartbeatLog.warn).mockClear(); + + const renamed = { ...eventStore.agents.get(agent.id)!, name: "renamed-fresh" }; + eventStore.agents.set(agent.id, renamed); + eventStore.emit("agent:updated", renamed); + + expect(heartbeatLog.warn).not.toHaveBeenCalledWith(expect.stringContaining("Timer sync force re-armed stale present entry")); + }); }); describe("scheduler timer audit", () => { @@ -694,6 +760,248 @@ describe("HeartbeatTriggerScheduler", () => { expect(heartbeatLog.log).toHaveBeenCalledWith("Timer audit skipped re-arm for agent-long (active run)"); }); }); + + describe("FN-7718: orphaned/zombie timer invalidation on stop/start", () => { + /** + * FNXC:AgentHeartbeat 2026-07-09-00:00: + * Regression coverage for FN-7718: `fn agent stop`/`start` mutate the + * agent row from a SEPARATE process, so the in-process `agent:updated` + * listener never fires for CLI-driven transitions — the 60s audit is the + * ONLY cross-process reconciliation path. Before the fix, the audit loop + * opened with `if (!this.isTimerEligibleAgent(agent)) continue;`, which + * skipped stopped/paused/disabled agents WITHOUT clearing any timer entry + * armed while they were running — an orphaned/zombie entry that then sat + * until the FN-7645 stale-repair path eventually fired minutes later on + * the next start, producing the recurring `zombie-timer-rearmed` symptom. + */ + function buildAgent(overrides: Partial & { id: string; heartbeatIntervalMs: number }): Agent { + const { heartbeatIntervalMs, ...rest } = overrides; + return { + name: rest.id, + role: "executor", + state: "active", + lastHeartbeatAt: "2026-01-01T00:00:00.000Z", + runtimeConfig: { enabled: true, heartbeatIntervalMs }, + createdAt: "2026-01-01T00:00:00.000Z", + updatedAt: "2026-01-01T00:00:00.000Z", + metadata: {}, + ...rest, + } as Agent; + } + + it("clears an orphaned timer entry within one audit cycle when an agent is stopped out-of-process, and the subsequent start arms exactly one fresh timer with no zombie-timer-rearmed repair", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + // listAgents is mutated in-place to simulate a CLI `fn agent stop`/`start` + // cycle mutating the DB out-of-process — no `agent:updated` event fires. + let agent = buildAgent({ id: "agent-cli", heartbeatIntervalMs: 300_000, state: "active" }); + vi.mocked(store.listAgents).mockImplementation(async () => [agent]); + vi.mocked(store.getActiveHeartbeatRun).mockResolvedValue(null); + + scheduler = new HeartbeatTriggerScheduler(store, callback); + scheduler.start(); + await vi.advanceTimersByTimeAsync(0); + + expect(scheduler.getRegisteredAgents()).toContain("agent-cli"); + + // Simulate `fn agent stop`: DB now reports a non-tickable state, but no + // in-process event fires — the timer entry the scheduler armed earlier + // is still present until the next audit cycle reconciles it. + agent = { ...agent, state: "paused" }; + vi.mocked(heartbeatLog.warn).mockClear(); + + // Advance across several 60s audit cycles. + await vi.advanceTimersByTimeAsync(3 * 60_000); + + // Assertion it is gone: the orphaned timer must be cleared within one + // audit cycle — no lingering registration for the stopped agent. + expect(scheduler.getRegisteredAgents()).not.toContain("agent-cli"); + + // Simulate `fn agent start` with a stale lastHeartbeatAt left over from + // before the stop (no heartbeat happened while paused). + agent = { ...agent, state: "active", lastHeartbeatAt: "2026-01-01T00:00:00.000Z" }; + callback.mockClear(); + + await vi.advanceTimersByTimeAsync(60_000); // one more audit cycle picks up the start + + // Exactly one fresh timer entry results — not zero, not two. + expect(scheduler.getRegisteredAgents()).toContain("agent-cli"); + const timers = (scheduler as unknown as { timers: Map }).timers; + expect(timers.has("agent-cli")).toBe(true); + + // No zombie-timer-rearmed repair should ever have been needed for this + // agent — the orphaned entry was invalidated at stop time, so the start + // begins clean instead of requiring the FN-7645 stale-repair path. + expect(heartbeatLog.warn).not.toHaveBeenCalledWith(expect.stringContaining("zombie-timer-rearmed")); + }); + + it("clears a present timer for each ineligibility cause: non-tickable state, runtimeConfig.enabled:false, and ephemeral/non-heartbeat-managed", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + const agents: Record = { + "agent-paused": buildAgent({ id: "agent-paused", heartbeatIntervalMs: 300_000 }), + "agent-disabled": buildAgent({ id: "agent-disabled", heartbeatIntervalMs: 300_000 }), + "agent-ephemeral": buildAgent({ id: "agent-ephemeral", heartbeatIntervalMs: 300_000, metadata: { agentKind: "task-worker" } }), + }; + vi.mocked(store.listAgents).mockImplementation(async () => Object.values(agents)); + vi.mocked(store.getActiveHeartbeatRun).mockResolvedValue(null); + + scheduler = new HeartbeatTriggerScheduler(store, callback); + scheduler.start(); + await vi.advanceTimersByTimeAsync(0); + + // ephemeral/task-worker agents are never timer-eligible in the first + // place, so only the two heartbeat-managed agents get an initial entry. + expect(scheduler.getRegisteredAgents()).toContain("agent-paused"); + expect(scheduler.getRegisteredAgents()).toContain("agent-disabled"); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-ephemeral"); + + // Flip each to a non-eligible condition out-of-process. + agents["agent-paused"] = { ...agents["agent-paused"], state: "paused" }; + agents["agent-disabled"] = { + ...agents["agent-disabled"], + runtimeConfig: { ...(agents["agent-disabled"].runtimeConfig as Record), enabled: false }, + }; + + await vi.advanceTimersByTimeAsync(60_000); // one audit cycle + + expect(scheduler.getRegisteredAgents()).not.toContain("agent-paused"); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-disabled"); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-ephemeral"); + }); + + it("is a safe no-op for a non-eligible agent that never had a timer entry", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + const agent = buildAgent({ id: "agent-never-armed", heartbeatIntervalMs: 300_000, state: "paused" }); + vi.mocked(store.listAgents).mockResolvedValue([agent]); + vi.mocked(store.getActiveHeartbeatRun).mockResolvedValue(null); + + scheduler = new HeartbeatTriggerScheduler(store, callback); + scheduler.start(); + await vi.advanceTimersByTimeAsync(0); + + expect(scheduler.getRegisteredAgents()).not.toContain("agent-never-armed"); + + await expect(vi.advanceTimersByTimeAsync(3 * 60_000)).resolves.not.toThrow(); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-never-armed"); + }); + + it("clears an orphaned entry for both short (300_000) and long (3_600_000) interval buckets, and does not regress the FN-7645 long-interval zombie re-arm for a still-eligible stale agent", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + const agents: Record = { + "agent-short-stopped": buildAgent({ id: "agent-short-stopped", heartbeatIntervalMs: 300_000 }), + "agent-long-stopped": buildAgent({ id: "agent-long-stopped", heartbeatIntervalMs: 3_600_000 }), + "agent-long-zombie": buildAgent({ id: "agent-long-zombie", heartbeatIntervalMs: 3_600_000 }), + }; + vi.mocked(store.listAgents).mockImplementation(async () => Object.values(agents)); + vi.mocked(store.getActiveHeartbeatRun).mockResolvedValue(null); + + scheduler = new HeartbeatTriggerScheduler(store, callback); + scheduler.start(); + await vi.advanceTimersByTimeAsync(0); + + for (const id of Object.keys(agents)) { + expect(scheduler.getRegisteredAgents()).toContain(id); + } + + // Stop the short and long agents out-of-process; leave agent-long-zombie + // eligible but kill its live interval handle to reproduce the FN-7645 + // present-but-non-advancing case, which must still self-heal. + agents["agent-short-stopped"] = { ...agents["agent-short-stopped"], state: "paused" }; + agents["agent-long-stopped"] = { ...agents["agent-long-stopped"], state: "paused" }; + const timers = (scheduler as unknown as { timers: Map }).timers; + clearInterval(timers.get("agent-long-zombie")!.handle as ReturnType); + + vi.mocked(heartbeatLog.warn).mockClear(); + + // 4 hours covers both the 60s stop-side audit reconciliation and the + // FN-7645 stale threshold (2x interval = 7.2M ms) for the still-eligible + // zombie agent. + await vi.advanceTimersByTimeAsync(4 * 60 * 60 * 1000); + + // Both stopped agents' orphaned timers were invalidated. + expect(scheduler.getRegisteredAgents()).not.toContain("agent-short-stopped"); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-long-stopped"); + + // The still-eligible long-interval zombie agent is unaffected by the + // stop-path fix and still gets FN-7645's stale re-arm + dispatch. + expect(scheduler.getRegisteredAgents()).toContain("agent-long-zombie"); + expect(callback).toHaveBeenCalledWith("agent-long-zombie", "timer", expect.anything()); + expect(heartbeatLog.warn).toHaveBeenCalledWith(expect.stringContaining("zombie-timer-rearmed")); + }); + + it("guards remain intact: globalPause suppresses dispatch, a healthy active run is untouched, and a healthy fresh short-interval agent is never force-re-armed", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + const agents: Record = { + "agent-stopped": buildAgent({ id: "agent-stopped", heartbeatIntervalMs: 300_000 }), + "agent-healthy-run": buildAgent({ id: "agent-healthy-run", heartbeatIntervalMs: 300_000 }), + "agent-healthy-fresh": buildAgent({ id: "agent-healthy-fresh", heartbeatIntervalMs: 300_000 }), + }; + (agents["agent-healthy-run"].runtimeConfig as Record).heartbeatTimeoutMs = 24 * 60 * 60 * 1000; + vi.mocked(store.listAgents).mockImplementation(async () => Object.values(agents)); + vi.mocked(store.getActiveHeartbeatRun).mockImplementation(async (agentId: string) => + agentId === "agent-healthy-run" ? ({ id: "run-healthy", status: "active" } as any) : null, + ); + + const taskStore = { + getSettings: vi.fn().mockResolvedValue({ globalPause: false, enginePaused: false }), + } as unknown as TaskStore; + + scheduler = new HeartbeatTriggerScheduler(store, callback, taskStore); + scheduler.start(); + await vi.advanceTimersByTimeAsync(0); + + // agent-healthy-run never gets a bare timer armed while its run is live. + expect(scheduler.getRegisteredAgents()).not.toContain("agent-healthy-run"); + expect(scheduler.getRegisteredAgents()).toContain("agent-healthy-fresh"); + + agents["agent-stopped"] = { ...agents["agent-stopped"], state: "paused" }; + callback.mockClear(); + vi.mocked(store.endHeartbeatRun).mockClear(); + + // Flip globalPause on for this cycle to confirm dispatch stays suppressed. + vi.mocked(taskStore.getSettings).mockResolvedValue({ globalPause: true, enginePaused: false } as any); + + await vi.advanceTimersByTimeAsync(60_000); + + expect(scheduler.getRegisteredAgents()).not.toContain("agent-stopped"); + expect(store.endHeartbeatRun).not.toHaveBeenCalled(); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-healthy-run"); + // Healthy fresh short-interval agent keeps its original entry — not + // force-cleared or re-armed just because an unrelated agent was stopped. + expect(scheduler.getRegisteredAgents()).toContain("agent-healthy-fresh"); + }); + + it("a repeat stop (already-cleared orphaned timer) is a safe idempotent no-op", async () => { + vi.useFakeTimers(); + vi.setSystemTime(new Date("2026-01-01T00:00:00.000Z")); + + let agent = buildAgent({ id: "agent-repeat-stop", heartbeatIntervalMs: 300_000 }); + vi.mocked(store.listAgents).mockImplementation(async () => [agent]); + vi.mocked(store.getActiveHeartbeatRun).mockResolvedValue(null); + + scheduler = new HeartbeatTriggerScheduler(store, callback); + scheduler.start(); + await vi.advanceTimersByTimeAsync(0); + + agent = { ...agent, state: "paused" }; + await vi.advanceTimersByTimeAsync(60_000); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-repeat-stop"); + + // Multiple further audit cycles while still stopped must remain a + // cheap no-op — no throw, no re-entry into the timer map. + await expect(vi.advanceTimersByTimeAsync(3 * 60_000)).resolves.not.toThrow(); + expect(scheduler.getRegisteredAgents()).not.toContain("agent-repeat-stop"); + }); + }); }); describe("registerAgent", () => { diff --git a/packages/engine/src/agent-heartbeat.ts b/packages/engine/src/agent-heartbeat.ts index eb3807b6e9..73d4bb85d7 100644 --- a/packages/engine/src/agent-heartbeat.ts +++ b/packages/engine/src/agent-heartbeat.ts @@ -4438,8 +4438,28 @@ export class HeartbeatTriggerScheduler { } if (this.timers.has(agent.id)) { - // Already ticking — non-config updates should not reset the interval. - return; + /* + * FNXC:AgentHeartbeat 2026-07-09-00:00: + * FN-7718 — a bare "already ticking" return here used to no-op even when + * the present timer entry was a stale/orphaned leftover (e.g. one that + * survived a stop the audit had not yet reconciled, or a start transition + * racing an in-flight registration). Reuse the same repair-stale gate the + * audit uses (default multiplier, since this sync path has no access to + * per-project settings) so a start transition force-clears+re-arms a + * present-but-stale entry instead of inheriting it, while a healthy fresh + * entry is still left alone — unrelated agent:updated events must never + * reset a healthy cadence. + */ + const staleThresholdMs = this.getRepairStaleThresholdMs(agent, HeartbeatTriggerScheduler.DEFAULT_REPAIR_STALE_MULTIPLIER); + const elapsedMs = getHeartbeatAgeMs(agent); + const staleAtSync = Number.isFinite(elapsedMs) && elapsedMs > staleThresholdMs; + if (!staleAtSync) { + // Already ticking and fresh — non-config updates should not reset the interval. + return; + } + heartbeatLog.warn( + `Timer sync force re-armed stale present entry for ${agent.id} (${reason}): no heartbeat for ${Math.round(elapsedMs / 1000)}s (threshold ${Math.round(staleThresholdMs / 1000)}s)`, + ); } this.registerAgent(agent.id, this.getAgentTimerConfig(agent), { @@ -4622,7 +4642,30 @@ export class HeartbeatTriggerScheduler { let rearmedCount = 0; let zombieRearmedCount = 0; for (const agent of agents) { - if (!this.isTimerEligibleAgent(agent)) continue; + /* + * FNXC:AgentHeartbeat 2026-07-09-00:00: + * FN-7718 — CLI-driven `fn agent stop`/`start` mutate the agent row from + * a SEPARATE process, so the in-process `agent:updated` listener + * (syncTimerForAgent -> unregisterAgent) never fires for those + * transitions. This 60s audit is therefore the ONLY cross-process + * reconciliation path. Previously this loop bare-`continue`d past every + * non-eligible agent (stopped/paused, runtimeConfig.enabled===false, or + * ephemeral/!isHeartbeatManaged), which meant a timer entry armed while + * the agent was still running/eligible was never cleared — an orphaned + * "zombie" registration that lingered until the FN-7645 stale-repair + * path eventually fired minutes after a subsequent start (the recurring + * `zombie-timer-rearmed` symptom). Fix: when an agent is no longer + * timer-eligible but still has a present timer entry, unregister it here + * so a later start begins from a completely clean scheduling state + * instead of inheriting a stale/orphaned timer. + */ + if (!this.isTimerEligibleAgent(agent)) { + if (this.timers.has(agent.id)) { + this.unregisterAgent(agent.id); + heartbeatLog.log(`Timer audit cleared orphaned timer for non-eligible agent ${agent.id} (audit:${reason})`); + } + continue; + } const hasTimerEntry = this.timers.has(agent.id); const staleThresholdMs = this.getRepairStaleThresholdMs(agent, staleMultiplier);