perf(gateway): prewarm first Control UI requests while idle (#158555)

* perf(gateway): prewarm first-use chat paths while idle

* fix(gateway): validate idle prewarm handler lookup

* test(gateway): normalize idle prewarm fixture paths

* test(gateway): narrow prewarm call order assertion
This commit is contained in:
Peter Steinberger
2026-09-25 22:04:51 -07:00
committed by GitHub
parent ccaeab1686
commit 430ea3c671
10 changed files with 300 additions and 177 deletions
+9
View File
@@ -285,6 +285,15 @@ Output includes first process output, `/healthz`, `/readyz`, HTTP listen log tim
Use JSON output or `--output` when comparing changes. Use `--cpu-prof-dir` only after trace output points at import, compile, or CPU-bound work that phase timings alone cannot explain.
After readiness, the Gateway uses idle turns to prepare common Control UI handler
modules and configured local agent skill discovery. Each item yields to admitted
foreground work, and shutdown joins preparation that has already started. This
work does not execute chat requests, create connection state, or fetch live provider
catalogs. Context-window cache preparation shares this sequence and starts no
earlier than five seconds after scheduling. Compare immediate first requests with
requests after an idle interval; readiness alone does not guarantee every optional
cache is warm.
</Accordion>
<Accordion title="Workspace computation (scripts/bench-workspace-computation.ts)">
+11 -23
View File
@@ -8,7 +8,6 @@ import {
} from "../process/gateway-work-admission.js";
import { scheduleGatewayIdleTask } from "./server-idle-task.js";
import { createGatewaySidecarStopOwner } from "./server-sidecar-owners.js";
import { scheduleContextCachePrewarm } from "./server-startup-context-cache-prewarm.js";
import { scheduleGatewayHandlerPrewarm } from "./server-startup-handler-prewarm.js";
afterEach(() => {
@@ -16,7 +15,7 @@ afterEach(() => {
resetGatewayWorkAdmission();
});
it.each(["idle", "handler", "context"] as const)(
it.each(["idle", "handler"] as const)(
"joins started %s work before the Gateway sidecar owner closes",
async (kind) => {
vi.useFakeTimers();
@@ -40,30 +39,19 @@ it.each(["idle", "handler", "context"] as const)(
log,
errorMessage: "idle lifecycle test failed",
})
: kind === "handler"
? scheduleGatewayHandlerPrewarm({
cfgAtStart: {},
log,
items: [
{ name: "first", load: run },
{ name: "later", load: later },
],
})
: scheduleContextCachePrewarm({
getConfig: () => ({}),
log,
startupTrace: {
measure: async (_name, warm) => {
await run();
return warm();
},
},
});
: scheduleGatewayHandlerPrewarm({
getConfig: () => ({}),
log,
items: [
{ name: "first", load: run },
{ name: "later", load: later },
],
});
const owner = createGatewaySidecarStopOwner();
owner.publish(handle);
let stopping: Promise<void> | undefined;
try {
await vi.advanceTimersByTimeAsync(kind === "context" ? 5_000 : 0);
await vi.advanceTimersByTimeAsync(0);
expect([...events]).toEqual(["started"]);
owner.beginClose();
stopping = owner.stop().then(() => {
@@ -138,7 +126,7 @@ it("joins the outgoing handler when shutdown begins in its warning callback", as
});
});
const sidecar = scheduleGatewayHandlerPrewarm({
cfgAtStart: {},
getConfig: () => ({}),
log: { warn },
items: [
{
@@ -23,9 +23,6 @@ import {
} from "../test-helpers.e2e.js";
// Optional startup prewarming must not compete with the catalog request drain.
vi.mock("../server-startup-context-cache-prewarm.js", () => ({
scheduleContextCachePrewarm: () => ({ stop() {} }),
}));
vi.mock("../server-startup-handler-prewarm.js", () => ({
scheduleGatewayHandlerPrewarm: () => ({ stop() {} }),
}));
@@ -97,9 +97,6 @@ vi.mock("../plugins/plugin-lookup-table.js", async (importOriginal) => ({
}));
// These independent startup tasks do not participate in plugin replacement.
vi.mock("./server-startup-context-cache-prewarm.js", () => ({
scheduleContextCachePrewarm: () => ({ stop() {} }),
}));
vi.mock("./server-startup-handler-prewarm.js", () => ({
scheduleGatewayHandlerPrewarm: () => ({ stop() {} }),
}));
@@ -1,52 +0,0 @@
import type { OpenClawConfig } from "../config/types.openclaw.js";
import { getActiveGatewayRootWorkCount } from "../process/gateway-work-admission.js";
import { scheduleGatewayIdleTask, type GatewayIdleTaskHandle } from "./server-idle-task.js";
const CONTEXT_CACHE_PREWARM_START_DELAY_MS = 5_000;
const CONTEXT_CACHE_PREWARM_RETRY_DELAY_MS = 250;
type StartupTrace = {
measure: <T>(name: string, run: () => T | Promise<T>) => Promise<T>;
};
export function scheduleContextCachePrewarm(params: {
getConfig: () => OpenClawConfig;
startupTrace?: StartupTrace;
log: { warn: (msg: string) => void };
}): GatewayIdleTaskHandle {
let stopped = false;
const warm = async () => {
if (stopped) {
return;
}
const { prewarmContextWindowCacheAfterReady } = await import("../agents/context.js");
if (!stopped) {
await prewarmContextWindowCacheAfterReady({
config: params.getConfig(),
isCancelled: () => stopped,
});
}
};
// Source-backed provider discovery can consume the main thread. Give
// readiness probes and immediate client work a clean event-loop window.
const idleTask = scheduleGatewayIdleTask({
delayMs: CONTEXT_CACHE_PREWARM_START_DELAY_MS,
retryDelayMs: CONTEXT_CACHE_PREWARM_RETRY_DELAY_MS,
isClosing: () => stopped,
isBusy: () => getActiveGatewayRootWorkCount({ excludeCurrent: true }) > 0,
run: () =>
params.startupTrace
? params.startupTrace.measure("post-ready.context-window-cache", warm)
: warm(),
log: params.log,
errorMessage: "post-ready.context-window-cache failed after gateway ready",
});
return {
stop: () => {
stopped = true;
return idleTask.stop();
},
};
}
@@ -1,12 +1,22 @@
import path from "node:path";
import { expectDefined } from "@openclaw/normalization-core/expect";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { createDeferred } from "../../test/helpers/promise.js";
import type { OpenClawConfig } from "../config/types.openclaw.js";
import {
resetGatewayWorkAdmission,
tryBeginGatewayIndependentRootWorkAdmission,
tryBeginGatewayRootWorkAdmission,
} from "../process/gateway-work-admission.js";
const mocks = vi.hoisted(() => ({
events: [] as string[],
executeRequest: vi.fn(),
ensureSkillsWatcher: vi.fn(),
prepareWorkspaceSkillEntries: vi.fn<
typeof import("../skills/loading/workspace-skill-loader.js").prepareWorkspaceSkillEntries
>(async () => ({ entries: [] })),
prewarmContextWindowCacheAfterReady: vi.fn(async () => {}),
getMemoryCapabilityRegistration: vi.fn<() => { pluginId: string } | undefined>(),
prewarmMemorySearchWorker: vi.fn(async () => {
mocks.events.push("memory-search");
@@ -36,6 +46,40 @@ vi.mock("../plugins/management-service.js", () => ({
listManagedPlugins: mocks.listManagedPlugins,
}));
vi.mock("./server/ws-connection/message-handler.js", () => {
mocks.events.push("connection");
return { attachGatewayWsMessageHandler: mocks.executeRequest };
});
vi.mock("./server-chat.js", () => {
mocks.events.push("agent-events");
return { createAgentEventHandler: mocks.executeRequest };
});
vi.mock("./server-session-key.js", () => ({ resolveSessionKeyForRun: mocks.executeRequest }));
vi.mock("./server-methods/core-handlers.js", async () => {
const { createLazyCoreHandlers } = await import("./server-methods/lazy-core-handlers.js");
return {
coreGatewayHandlers: createLazyCoreHandlers({
methods: ["chat.history", "chat.send", "sessions.list"],
loadHandlers: async () => {
mocks.events.push("handlers");
return {
"chat.history": mocks.executeRequest,
"chat.send": mocks.executeRequest,
"sessions.list": mocks.executeRequest,
};
},
}),
};
});
vi.mock("../skills/loading/workspace-skill-loader.js", () => ({
prepareWorkspaceSkillEntries: mocks.prepareWorkspaceSkillEntries,
}));
vi.mock("../agents/workspace-access.js", () => ({ getAgentWorkspaceAccess: () => undefined }));
vi.mock("../skills/runtime/refresh.js", () => ({ ensureSkillsWatcher: mocks.ensureSkillsWatcher }));
vi.mock("../agents/context.js", () => ({
prewarmContextWindowCacheAfterReady: mocks.prewarmContextWindowCacheAfterReady,
}));
vi.mock("../plugins/memory-state.js", () => ({
getMemoryCapabilityRegistration: mocks.getMemoryCapabilityRegistration,
}));
@@ -45,9 +89,17 @@ vi.mock("../plugins/public-surface-loader.js", () => ({
}));
const { scheduleGatewayHandlerPrewarm } = await import("./server-startup-handler-prewarm.js");
const workspaces = {
main: path.resolve("prewarm-main"),
research: path.resolve("prewarm-research"),
};
beforeEach(() => {
mocks.events.length = 0;
mocks.executeRequest.mockClear();
mocks.ensureSkillsWatcher.mockClear();
mocks.prepareWorkspaceSkillEntries.mockClear();
mocks.prewarmContextWindowCacheAfterReady.mockClear();
mocks.loadCombinedSessionStoreForGatewayCore.mockClear();
mocks.listManagedPlugins.mockClear();
mocks.getMemoryCapabilityRegistration.mockReset();
@@ -61,32 +113,92 @@ afterEach(() => {
});
describe("scheduleGatewayHandlerPrewarm", () => {
it.each([undefined, "memory-lancedb", "memory-core"])(
"warms retrieval only for active Memory Core (memory plugin: %s)",
it("prepares first-use modules, primary skills, and Memory Core in sequence without executing requests", async () => {
vi.useFakeTimers();
mocks.getMemoryCapabilityRegistration.mockReturnValue({ pluginId: "memory-core" });
const cfg: OpenClawConfig = {
agents: {
entries: {
main: { workspace: workspaces.main },
research: { workspace: workspaces.research },
},
},
};
const sidecar = scheduleGatewayHandlerPrewarm({
getConfig: () => cfg,
log: { warn: vi.fn() },
});
try {
expect(mocks.events).toEqual([]);
// Dynamic imports can enqueue the next idle timer after the current timer drain.
do {
await vi.runAllTimersAsync();
await vi.dynamicImportSettled();
} while (vi.getTimerCount() > 0);
expect(mocks.events).toContain("connection");
expect(mocks.events).toContain("agent-events");
expect(mocks.events.filter((event) => event === "handlers")).toHaveLength(3);
expect(mocks.executeRequest).not.toHaveBeenCalled();
expect(mocks.prepareWorkspaceSkillEntries.mock.calls).toEqual([
[workspaces.main, { config: cfg, agentId: "main" }],
[workspaces.research, { config: cfg, agentId: "research" }],
]);
expect(mocks.ensureSkillsWatcher.mock.calls).toEqual([
[{ workspaceDir: workspaces.main, config: cfg, agentId: "main" }],
[{ workspaceDir: workspaces.research, config: cfg, agentId: "research" }],
]);
expect(mocks.ensureSkillsWatcher.mock.invocationCallOrder[0]).toBeLessThan(
expectDefined(
mocks.prepareWorkspaceSkillEntries.mock.invocationCallOrder[0],
"skill preparation call",
),
);
expect(mocks.prewarmContextWindowCacheAfterReady).toHaveBeenCalledOnce();
expect(mocks.loadCombinedSessionStoreForGatewayCore).not.toHaveBeenCalled();
expect(mocks.loadBundledPluginPublicArtifactModuleSync).toHaveBeenCalledOnce();
expect(mocks.prewarmMemorySearchWorker).toHaveBeenCalledOnce();
const memoryCall = expectDefined(
mocks.prewarmMemorySearchWorker.mock.invocationCallOrder[0],
"memory retrieval preparation call",
);
expect(mocks.prewarmContextWindowCacheAfterReady.mock.invocationCallOrder[0]).toBeLessThan(
memoryCall,
);
expect(memoryCall).toBeLessThan(
expectDefined(
mocks.listManagedPlugins.mock.invocationCallOrder[0],
"plugin preparation call",
),
);
expect(mocks.listManagedPlugins).toHaveBeenCalledWith({ config: cfg });
} finally {
await sidecar.stop();
}
});
it.each([undefined, "memory-lancedb"])(
"skips retrieval preparation when Memory Core is inactive (%s)",
async (pluginId) => {
vi.useFakeTimers();
mocks.getMemoryCapabilityRegistration.mockReturnValue(pluginId ? { pluginId } : undefined);
const cfg = {
agents: { list: [{ id: "main", default: true }, { id: "research" }] },
} as never;
const sidecar = scheduleGatewayHandlerPrewarm({
cfgAtStart: cfg,
getConfig: () => ({ agents: { entries: {} } }),
log: { warn: vi.fn() },
});
expect(mocks.events).toEqual([]);
await vi.runAllTimersAsync();
expect(mocks.events).toEqual(
pluginId === "memory-core" ? ["memory-search", "plugins"] : ["plugins"],
);
expect(mocks.loadBundledPluginPublicArtifactModuleSync).toHaveBeenCalledTimes(
pluginId === "memory-core" ? 1 : 0,
);
expect(mocks.loadCombinedSessionStoreForGatewayCore).not.toHaveBeenCalled();
expect(mocks.listManagedPlugins).toHaveBeenCalledWith({ config: cfg });
await sidecar.stop();
try {
do {
await vi.runAllTimersAsync();
await vi.dynamicImportSettled();
} while (vi.getTimerCount() > 0);
expect(mocks.loadBundledPluginPublicArtifactModuleSync).not.toHaveBeenCalled();
expect(mocks.prewarmMemorySearchWorker).not.toHaveBeenCalled();
expect(mocks.listManagedPlugins).toHaveBeenCalledOnce();
} finally {
await sidecar.stop();
}
},
);
@@ -96,7 +208,7 @@ describe("scheduleGatewayHandlerPrewarm", () => {
const load = vi.fn(async () => {});
const sidecar = scheduleGatewayHandlerPrewarm({
cfgAtStart: {} as never,
getConfig: () => ({}),
log: { warn: vi.fn() },
items: [{ name: "sessions", load }],
waitForPostReadyWork: () => gatewayReady,
@@ -119,7 +231,7 @@ describe("scheduleGatewayHandlerPrewarm", () => {
}
const load = vi.fn(async () => {});
const sidecar = scheduleGatewayHandlerPrewarm({
cfgAtStart: {} as never,
getConfig: () => ({}),
log: { warn: vi.fn() },
items: [{ name: "sessions", load }],
});
@@ -131,7 +243,7 @@ describe("scheduleGatewayHandlerPrewarm", () => {
await vi.advanceTimersByTimeAsync(249);
expect(load).not.toHaveBeenCalled();
await vi.advanceTimersByTimeAsync(1);
await vi.waitFor(() => expect(load).toHaveBeenCalledOnce());
expect(load).toHaveBeenCalledOnce();
await sidecar.stop();
});
@@ -141,7 +253,7 @@ describe("scheduleGatewayHandlerPrewarm", () => {
const load = vi.fn(async () => {});
const sidecar = scheduleGatewayHandlerPrewarm({
cfgAtStart: {} as never,
getConfig: () => ({}),
log: { warn: vi.fn() },
items: [{ name: "sessions", load }],
waitForPostReadyWork: () => gatewayReady,
@@ -165,7 +277,7 @@ describe("scheduleGatewayHandlerPrewarm", () => {
.mockResolvedValue("request result");
scheduleGatewayHandlerPrewarm({
cfgAtStart: {} as never,
getConfig: () => ({}),
log: { warn },
items: [
{
@@ -197,7 +309,7 @@ describe("scheduleGatewayHandlerPrewarm", () => {
);
const second = vi.fn(async () => {});
const sidecar = scheduleGatewayHandlerPrewarm({
cfgAtStart: {} as never,
getConfig: () => ({}),
log: { warn: vi.fn() },
items: [
{ name: "first", load: first },
@@ -215,3 +327,66 @@ describe("scheduleGatewayHandlerPrewarm", () => {
expect(second).not.toHaveBeenCalled();
});
});
it("keeps the context cache delayed and uses current config after foreground work", async () => {
vi.useFakeTimers();
const initial: OpenClawConfig = { agents: { entries: {} } };
let current = initial;
const handle = scheduleGatewayHandlerPrewarm({
getConfig: () => current,
log: { warn: vi.fn() },
});
await vi.advanceTimersByTimeAsync(4_999);
expect(mocks.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
const request = tryBeginGatewayRootWorkAdmission();
if (!request) {
throw new Error("Expected foreground admission");
}
try {
await vi.advanceTimersByTimeAsync(1);
expect(mocks.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
current = { agents: { entries: {} }, skills: { load: { watch: false } } };
request.release();
await vi.advanceTimersByTimeAsync(250);
expect(mocks.prewarmContextWindowCacheAfterReady).toHaveBeenCalledWith({
config: current,
isCancelled: expect.any(Function),
});
} finally {
request.release();
await handle.stop();
}
});
it("skips optional discovery when foreground work arrives after idle admission", async () => {
vi.useFakeTimers();
mocks.getMemoryCapabilityRegistration.mockReturnValue({ pluginId: "memory-core" });
const handle = scheduleGatewayHandlerPrewarm({
getConfig: () => ({ agents: { entries: { main: { workspace: workspaces.main } } } }),
log: { warn: vi.fn() },
startupTrace: {
measure: async (_name, load) => {
const request = tryBeginGatewayIndependentRootWorkAdmission("test-request");
if (!request) {
throw new Error("Expected foreground admission");
}
try {
return await load();
} finally {
request.release();
}
},
},
});
try {
do {
await vi.runAllTimersAsync();
await vi.dynamicImportSettled();
} while (vi.getTimerCount() > 0);
expect(mocks.prepareWorkspaceSkillEntries).not.toHaveBeenCalled();
expect(mocks.ensureSkillsWatcher).not.toHaveBeenCalled();
expect(mocks.prewarmMemorySearchWorker).not.toHaveBeenCalled();
} finally {
await handle.stop();
}
});
+77 -9
View File
@@ -1,3 +1,4 @@
import { listAgentIds, resolveAgentWorkspaceDir } from "../agents/agent-scope-config.js";
import type { OpenClawConfig } from "../config/types.openclaw.js";
import { getActiveGatewayRootWorkCount } from "../process/gateway-work-admission.js";
import { scheduleGatewayIdleTask, type GatewayIdleTaskHandle } from "./server-idle-task.js";
@@ -10,20 +11,80 @@ type StartupTrace = {
type GatewayHandlerPrewarmItem = {
name: string;
notBeforeMs?: number;
load: () => Promise<unknown>;
};
function dashboardDataPrewarmItems(cfg: OpenClawConfig): GatewayHandlerPrewarmItem[] {
function gatewayPrewarmItems(
getConfig: () => OpenClawConfig,
isCancelled: () => boolean,
): GatewayHandlerPrewarmItem[] {
return [
{ name: "connection", load: () => import("./server/ws-connection/message-handler.js") },
...["chat.history", "chat.send", "sessions.list"].map((method) => ({
name: method,
load: async () => {
const [{ coreGatewayHandlers }, { prepareGatewayRequestHandler }] = await Promise.all([
import("./server-methods/core-handlers.js"),
import("./server-methods/lazy-core-handlers.js"),
]);
if (!isCancelled()) {
const handler = coreGatewayHandlers[method];
if (!handler) {
throw new Error(`Gateway prewarm handler not found: ${method}`);
}
await prepareGatewayRequestHandler(handler);
}
},
})),
{ name: "agent-events", load: () => import("./server-chat.js") },
{ name: "session-key", load: () => import("./server-session-key.js") },
...listAgentIds(getConfig()).map((agentId) => ({
name: `skills.${agentId}`,
load: async () => {
const [
{ prepareWorkspaceSkillEntries },
{ getAgentWorkspaceAccess },
{ ensureSkillsWatcher },
] = await Promise.all([
import("../skills/loading/workspace-skill-loader.js"),
import("../agents/workspace-access.js"),
import("../skills/runtime/refresh.js"),
]);
const config = getConfig();
if (isCancelled() || !listAgentIds(config).includes(agentId)) {
return;
}
const workspaceDir = resolveAgentWorkspaceDir(config, agentId);
// Remote workspaces retain request-owned discovery and connection lifetimes.
if (!getAgentWorkspaceAccess(workspaceDir, "loadSkills")) {
ensureSkillsWatcher({ workspaceDir, config, agentId });
await prepareWorkspaceSkillEntries(workspaceDir, { config, agentId });
}
},
})),
{
name: "context-window-cache",
notBeforeMs: 5_000,
load: async () => {
const { prewarmContextWindowCacheAfterReady } = await import("../agents/context.js");
if (!isCancelled()) {
await prewarmContextWindowCacheAfterReady({ config: getConfig(), isCancelled });
}
},
},
{
name: "memory-search",
load: async () => {
const { getMemoryCapabilityRegistration } = await import("../plugins/memory-state.js");
if (getMemoryCapabilityRegistration()?.pluginId !== "memory-core") {
if (isCancelled() || getMemoryCapabilityRegistration()?.pluginId !== "memory-core") {
return;
}
const { loadBundledPluginPublicArtifactModuleSync } =
await import("../plugins/public-surface-loader.js");
if (isCancelled()) {
return;
}
const { prewarmMemorySearchWorker } = loadBundledPluginPublicArtifactModuleSync<{
prewarmMemorySearchWorker: () => Promise<void>;
}>({ dirName: "memory-core", artifactBasename: "prewarm-api.js" });
@@ -34,23 +95,30 @@ function dashboardDataPrewarmItems(cfg: OpenClawConfig): GatewayHandlerPrewarmIt
name: "plugins",
load: async () => {
const { listManagedPlugins } = await import("../plugins/management-service.js");
await listManagedPlugins({ config: cfg });
if (!isCancelled()) {
await listManagedPlugins({ config: getConfig() });
}
},
},
];
}
export function scheduleGatewayHandlerPrewarm(params: {
cfgAtStart: OpenClawConfig;
getConfig: () => OpenClawConfig;
startupTrace?: StartupTrace;
log: { info?: (msg: string) => void; warn: (msg: string) => void };
log: { warn: (msg: string) => void };
items?: readonly GatewayHandlerPrewarmItem[];
waitForPostReadyWork?: () => Promise<void>;
}): GatewayIdleTaskHandle {
// Session rows are resident; warm only process-stable plugin data and retrieval code.
// Provider catalogs stay request-driven because their adapters may do unbounded external work.
const items = params.items ?? dashboardDataPrewarmItems(params.cfgAtStart);
let stopped = false;
const startedAt = Date.now();
// Warm code and local facts without executing requests or acquiring live provider catalogs.
const items =
params.items ??
gatewayPrewarmItems(
params.getConfig,
() => stopped || getActiveGatewayRootWorkCount({ excludeCurrent: true }) > 0,
);
let nextIndex = 0;
let currentItemName = "unknown";
let idleTask: GatewayIdleTaskHandle | undefined;
@@ -71,7 +139,7 @@ export function scheduleGatewayHandlerPrewarm(params: {
currentItemName = item.name;
const load = () => item.load();
idleTask = scheduleGatewayIdleTask({
delayMs: 0,
delayMs: Math.max(0, (item.notBeforeMs ?? 0) - (Date.now() - startedAt)),
retryDelayMs: GATEWAY_HANDLER_PREWARM_RETRY_DELAY_MS,
isClosing: () => stopped,
isBusy: () => getActiveGatewayRootWorkCount({ excludeCurrent: true }) > 0,
+1 -58
View File
@@ -118,7 +118,6 @@ const hoisted = vi.hoisted(() => {
async (_cfg?: unknown, _options?: unknown) => {},
);
const prewarmConfigDrivenReplyRuntime = vi.fn(async () => {});
const prewarmContextWindowCacheAfterReady = vi.fn(async () => {});
const scheduleGatewayHandlerPrewarm = vi.fn(() => ({ stop: vi.fn() }));
return {
startPluginServices,
@@ -148,7 +147,6 @@ const hoisted = vi.hoisted(() => {
prepareModelRuntimeSnapshot,
refreshPreparedModelRuntimeSnapshots,
prewarmConfigDrivenReplyRuntime,
prewarmContextWindowCacheAfterReady,
scheduleGatewayHandlerPrewarm,
};
});
@@ -249,9 +247,6 @@ vi.mock("../auto-reply/reply/get-reply-from-config.runtime.js", () => ({
getReplyFromConfig: vi.fn(),
prewarmConfigDrivenReplyRuntime: hoisted.prewarmConfigDrivenReplyRuntime,
}));
vi.mock("../agents/context.js", () => ({
prewarmContextWindowCacheAfterReady: hoisted.prewarmContextWindowCacheAfterReady,
}));
vi.mock("./server-startup-handler-prewarm.js", () => ({
scheduleGatewayHandlerPrewarm: hoisted.scheduleGatewayHandlerPrewarm,
@@ -262,7 +257,6 @@ const {
startGatewaySidecars: startGatewaySidecarsImpl,
} = await import("./server-startup-post-attach.js");
const sentinelStartup = await import("./server-startup-restart-sentinel.js");
const { scheduleContextCachePrewarm } = await import("./server-startup-context-cache-prewarm.js");
const { STARTUP_UNAVAILABLE_GATEWAY_METHODS } = await import("./methods/core-method-policy.js");
type PostAttachParams = Parameters<typeof startGatewayPostAttachRuntimeImpl>[0];
@@ -521,8 +515,6 @@ describe("startGatewayPostAttachRuntime", () => {
hoisted.refreshPreparedModelRuntimeSnapshots.mockResolvedValue(undefined);
hoisted.prewarmConfigDrivenReplyRuntime.mockReset();
hoisted.prewarmConfigDrivenReplyRuntime.mockResolvedValue(undefined);
hoisted.prewarmContextWindowCacheAfterReady.mockReset();
hoisted.prewarmContextWindowCacheAfterReady.mockResolvedValue(undefined);
hoisted.scheduleGatewayHandlerPrewarm.mockClear();
});
@@ -1746,55 +1738,6 @@ describe("startGatewayPostAttachRuntime", () => {
expect(returned).toBe(true);
});
it("defers context-window cache prewarm to a post-ready sidecar", async () => {
vi.useFakeTimers();
const startupConfig = { agents: { defaults: { model: "openai/gpt-5.5" } } };
const currentConfig = { ...startupConfig };
const admission = tryBeginGatewayRootWorkAdmission();
if (!admission) {
throw new Error("Expected request work admission");
}
const sidecar = scheduleContextCachePrewarm({
getConfig: () => currentConfig,
log: { warn: vi.fn() },
});
try {
expect(hoisted.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
await vi.advanceTimersByTimeAsync(4_999);
expect(hoisted.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
await vi.advanceTimersByTimeAsync(1);
expect(hoisted.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
admission.release();
await vi.advanceTimersByTimeAsync(249);
expect(hoisted.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
await vi.advanceTimersByTimeAsync(1);
await vi.dynamicImportSettled();
await waitForGatewayTestState(() => {
expect(hoisted.prewarmContextWindowCacheAfterReady).toHaveBeenCalledWith({
config: currentConfig,
isCancelled: expect.any(Function),
});
});
} finally {
admission.release();
await stopTrackedSidecar(sidecar);
}
});
it("cancels context-window cache prewarm when the gateway stops first", async () => {
vi.useFakeTimers();
const sidecar = scheduleContextCachePrewarm({
getConfig: () => ({}) as never,
log: { warn: vi.fn() },
});
await stopTrackedSidecar(sidecar);
await vi.runAllTimersAsync();
expect(hoisted.prewarmContextWindowCacheAfterReady).not.toHaveBeenCalled();
});
it("keeps transcripts auto-start alive when Gmail post-ready sidecars stop", async () => {
const onPostReadySidecars = vi.fn<SidecarPublisher>();
const started = createDeferred();
@@ -1825,7 +1768,7 @@ describe("startGatewayPostAttachRuntime", () => {
const gmailSidecars = onPostReadySidecars.mock.calls[0];
const lifetimeSidecars = [...publishedGatewayLifetimeSidecars];
expect(gmailSidecars).toHaveLength(2);
expect(lifetimeSidecars).toHaveLength(4);
expect(lifetimeSidecars).toHaveLength(3);
await started.promise;
@@ -36,7 +36,6 @@ import type { GatewayClient, GatewayContextResolver } from "./server-methods/sha
import type { GatewayPluginRuntimeClaim } from "./server-plugin-runtime-generation.js";
import type { refreshLatestUpdateRestartSentinel } from "./server-restart-sentinel.js";
import type { GatewaySidecarStartupMode } from "./server-sidecar-startup-mode.js";
import { scheduleContextCachePrewarm } from "./server-startup-context-cache-prewarm.js";
import { scheduleGatewayHandlerPrewarm } from "./server-startup-handler-prewarm.js";
import type { logGatewayStartup } from "./server-startup-log.js";
import {
@@ -999,7 +998,6 @@ export async function startGatewayPostAttachRuntime(
// work can create sessions that the recovery scan must leave alone.
params.unlockStartupMethods();
const newGatewayLifetimeSidecars = [
scheduleContextCachePrewarm(params),
scheduleGatewayHandlerPrewarm(params),
...(mainSessionRecoverySidecar ? [mainSessionRecoverySidecar] : []),
];
@@ -47,7 +47,7 @@ export function registerGatewayStartupReadinessTests(params: {
expect(trace.mark).toHaveBeenCalledWith("sidecars.ready");
expect(trace.detail).toHaveBeenCalledWith("sidecars.ready", [
["loadedPluginCount", 2],
["postReadySidecarCount", 4],
["postReadySidecarCount", 3],
]);
});
}