diff options
| author | Adam Malczewski <[email protected]> | 2026-06-03 13:02:15 +0900 |
|---|---|---|
| committer | Adam Malczewski <[email protected]> | 2026-06-03 13:02:15 +0900 |
| commit | e87e6b39285c8001045d1ebdac873b182c0f7868 (patch) | |
| tree | 27003852f7b182fd65c6ad762784aa5fcf839ebc /packages/frontend/src/lib/components/ModelSelector.svelte | |
| parent | ae672fd4f5542a2c217cf97657bf81eeebdaabbd (diff) | |
| download | dispatch-e87e6b39285c8001045d1ebdac873b182c0f7868.tar.gz dispatch-e87e6b39285c8001045d1ebdac873b182c0f7868.zip | |
feat: prompt cache warming for idle tabs
Keep a tab's provider prompt-cache warm while idle by periodically replaying
the exact cached conversation prefix plus a single trivial throwaway turn,
resetting the provider's ~5-min cache TTL so the user's next real message hits
a warm cache.
Backend:
- Agent.warmCache(history): extracts buildLlmContext() shared with run(), then
re-sends the identical system+tools+history prefix (same Anthropic
cache_control breakpoints) plus a 'reply with just a .' probe turn via
toolChoice:none. Returns the request usage; mutates no history, emits/persists
nothing.
- AgentManager.warmCacheForTab(): resolves the same agent the next real turn
would use, replays the FULL genuine history, refuses while a turn is running.
- POST /chat/warm: returns ONLY the warming request's usage (never persisted,
never folded into the real usage aggregate).
Frontend:
- cache-warming.svelte.ts store: per-tab 4-min repeating idle timer with
countdown, warming-specific last-request cache %, and error capture. Arms on
turn end, pauses during a turn, disables+resets on a real user message.
- cache-warm-storage.ts: per-tab localStorage persistence of the toggle.
- Lifecycle hooks wired into tabs.svelte.ts (status/statuses/sendMessage/
hydrate/create/open/close).
- ModelSelector: bottom-of-panel checkbox + debug strip (last-% / countdown /
error), shown only when enabled. Warming cache data never touches the real
Cache Rate view.
Tests: core warmCache (5), api warm route (3) + warmCacheForTab (3), frontend
store (12) + storage (10). check / test (779) / frontend build / typecheck all
green.
Diffstat (limited to 'packages/frontend/src/lib/components/ModelSelector.svelte')
| -rw-r--r-- | packages/frontend/src/lib/components/ModelSelector.svelte | 89 |
1 files changed, 89 insertions, 0 deletions
diff --git a/packages/frontend/src/lib/components/ModelSelector.svelte b/packages/frontend/src/lib/components/ModelSelector.svelte index 8601795..1e6cdf0 100644 --- a/packages/frontend/src/lib/components/ModelSelector.svelte +++ b/packages/frontend/src/lib/components/ModelSelector.svelte @@ -10,8 +10,13 @@ const modelCache = new Map<string, string[]>(); REASONING_EFFORT_LABELS, } from "@dispatch/core/src/types/index.js"; import type { KeyInfo } from "../types.js"; + import { + cacheWarming, + WARM_INTERVAL_MS, + } from "../cache-warming.svelte.js"; import { config } from "../config.js"; import { router } from "../router.svelte.js"; + import { tabStore } from "../tabs.svelte.js"; interface AgentInfo { name: string; @@ -51,6 +56,7 @@ const modelCache = new Map<string, string[]>(); const { keys = [], + activeTabId = null, activeKeyId = null, activeModelId = null, reasoningEffort = "max", @@ -65,6 +71,7 @@ const modelCache = new Map<string, string[]>(); onWorkingDirectoryChange = (_dir: string | null) => {}, }: { keys?: KeyInfo[]; + activeTabId?: string | null; activeKeyId?: string | null; activeModelId?: string | null; reasoningEffort?: string; @@ -87,6 +94,26 @@ const modelCache = new Map<string, string[]>(); let sliderDragging = $state<number | null>(null); let modelSearch = $state(""); + // ─── Prompt-cache warming (debug strip lives at the bottom) ────── + // Reactive per-tab warming state from the singleton store. `warm.now` is a + // 1s ticking clock so the countdown re-renders while a fire is pending. + const warm = $derived(cacheWarming.stateFor(activeTabId)); + const warmCountdown = $derived.by(() => { + const next = warm.nextFireAt; + if (next === null) return null; + const ms = Math.max(0, next - cacheWarming.now); + const total = Math.round(ms / 1000); + const m = Math.floor(total / 60); + const s = total % 60; + return `${m}:${s.toString().padStart(2, "0")}`; + }); + const warmIntervalLabel = `${Math.round(WARM_INTERVAL_MS / 60000)} min`; + + function toggleCacheWarming(enabled: boolean): void { + if (!activeTabId) return; + tabStore.setCacheWarmingEnabled(activeTabId, enabled); + } + let cwdExists = $state<boolean | null>(null); let cwdCheckTimer: ReturnType<typeof setTimeout> | null = null; @@ -434,6 +461,68 @@ const modelCache = new Map<string, string[]>(); Agent Settings </button> {/if} + + <!-- Prompt-cache warming (bottom of the Chat Settings panel) --> + <div class="mt-3 pt-3 border-t border-base-300"> + <label class="flex items-center gap-2 cursor-pointer"> + <input + type="checkbox" + class="checkbox checkbox-sm rounded-sm" + checked={warm.enabled} + disabled={!activeTabId} + onchange={(e) => toggleCacheWarming(e.currentTarget.checked)} + /> + <span class="text-xs font-semibold">Keep prompt cache warm</span> + </label> + <p class="text-[10px] text-base-content/40 mt-1 leading-snug"> + While this tab is idle, replays the cached conversation every {warmIntervalLabel} + so the provider cache stays warm for your next message. Warming traffic is + debug-only — it never touches history, the Cache Rate metric, or context size. + </p> + + {#if warm.enabled} + <div class="mt-2 flex flex-col gap-2 bg-base-300/40 rounded-lg p-2"> + <!-- Warming "last request" cache rate (separate from the real metric) --> + <div class="flex flex-col gap-0.5"> + <div class="flex items-center justify-between"> + <span class="text-xs text-base-content/50">Last request (warming)</span> + <span class="text-xs font-mono">{warm.lastPct === null ? "-%" : `${warm.lastPct}%`}</span> + </div> + <progress + class="progress w-full h-2 {warm.lastPct === null + ? '' + : warm.lastPct >= 70 + ? 'progress-success' + : warm.lastPct >= 30 + ? 'progress-warning' + : 'progress-error'}" + value={warm.lastPct ?? 0} + max="100" + ></progress> + </div> + + <!-- Countdown to the next warming fire --> + <div class="flex items-center justify-between"> + <span class="text-xs text-base-content/50">Next warm in</span> + <span class="text-xs font-mono"> + {#if warm.firing} + warming… + {:else if warmCountdown !== null} + {warmCountdown} + {:else} + — + {/if} + </span> + </div> + + {#if warm.error} + <div class="text-[10px] text-error break-words"> + {warm.error} + </div> + {/if} + </div> + {/if} + </div> </div> {#if showKeyModal} |
