From f6fcbedd9fd33530478b341ff8102778b2cb08bb Mon Sep 17 00:00:00 2001 From: goodylili Date: Tue, 23 Jun 2026 14:13:22 +0100 Subject: [PATCH] perf(app): isolate answer streaming from the global store The top-level app component subscribes to the entire store, so the answer typewriter (which wrote each tick into the main store) re-rendered the whole ~8500-line tree up to 24 times per reply, plus on every keystroke elsewhere. Park the in-flight partial text in a dedicated stream store read only by a tiny node; the chat message still holds the empty placeholder while streaming and the final text is committed back to the store ONCE at the end. Net: the app tree re-renders twice per answer (push + commit) instead of ~24 times. Behavior is unchanged: same typewriter cadence (24 ticks / 55ms), same final Markdown render, all text paths (server, BYOK, error fallback) and the media path are untouched. --- frontend/src/components/cortex/cortex-app.tsx | 7 ++++++- .../components/cortex/streaming-answer.tsx | 12 +++++++++++ frontend/src/lib/cortex/store.ts | 21 ++++++++----------- frontend/src/lib/cortex/stream-store.ts | 17 +++++++++++++++ 4 files changed, 44 insertions(+), 13 deletions(-) create mode 100644 frontend/src/components/cortex/streaming-answer.tsx create mode 100644 frontend/src/lib/cortex/stream-store.ts diff --git a/frontend/src/components/cortex/cortex-app.tsx b/frontend/src/components/cortex/cortex-app.tsx index 2712b8c..e2925ea 100644 --- a/frontend/src/components/cortex/cortex-app.tsx +++ b/frontend/src/components/cortex/cortex-app.tsx @@ -150,6 +150,7 @@ import { import { Onboarding } from "./onboarding"; import { CaptureModal } from "./capture"; import { Markdown } from "./markdown"; +import { StreamingAnswer } from "./streaming-answer"; import { MediaBlock } from "./media-block"; import { SourceChips, type SourceItem } from "./sources"; @@ -3885,7 +3886,11 @@ export function CortexApp({ ) : (
- {m.streaming ? m.a : } + {m.streaming ? ( + + ) : ( + + )}
)} {!m.streaming && !m.media && ( diff --git a/frontend/src/components/cortex/streaming-answer.tsx b/frontend/src/components/cortex/streaming-answer.tsx new file mode 100644 index 0000000..8ed8693 --- /dev/null +++ b/frontend/src/components/cortex/streaming-answer.tsx @@ -0,0 +1,12 @@ +"use client"; + +import { useStreamStore } from "@/lib/cortex/stream-store"; + +// Renders the in-flight answer text while it streams. It subscribes ONLY to the +// stream store, so each typewriter tick re-renders this node alone rather than the +// whole app. Once streaming finishes the chat message holds the final text and is +// rendered as Markdown instead. +export function StreamingAnswer() { + const text = useStreamStore((s) => s.text); + return <>{text}; +} diff --git a/frontend/src/lib/cortex/store.ts b/frontend/src/lib/cortex/store.ts index 7649575..4be252b 100644 --- a/frontend/src/lib/cortex/store.ts +++ b/frontend/src/lib/cortex/store.ts @@ -1,5 +1,6 @@ "use client"; import { create } from "zustand"; +import { useStreamStore } from "./stream-store"; import { type Memory, type CortexEvent, @@ -999,16 +1000,17 @@ export const useCortex = create((set, get) => ({ activeId: get().activeId, }); - // Reveal the answer progressively, but cap the work: each store write - // re-renders the whole app tree (the top-level view subscribes to the store), - // so a per-word/30ms tick meant ~33 full re-renders per second. Reveal several - // words per tick and bound the run to STREAM_MAX_TICKS updates regardless of - // length, keeping the typewriter feel at a fraction of the render cost. + // Reveal the answer progressively. Each per-tick partial goes into the dedicated + // stream store (see stream-store.ts), NOT the main store, so only the small + // node re-renders per tick instead of the whole app tree + // (the top-level view subscribes to the entire main store). The final text is + // committed back to the chat ONCE, which is the only main-store write of the run. const STREAM_MAX_TICKS = 24; const STREAM_INTERVAL_MS = 55; const stream = (text: string) => { const words = text.split(" "); const perTick = Math.max(1, Math.ceil(words.length / STREAM_MAX_TICKS)); + useStreamStore.getState().setText(""); let i = 0; const tick = setInterval(() => { if (i >= words.length) { @@ -1022,6 +1024,7 @@ export const useCortex = create((set, get) => ({ } return { chat }; }); + useStreamStore.getState().setText(""); persist({ memories: get().memories, events: get().events, @@ -1033,13 +1036,7 @@ export const useCortex = create((set, get) => ({ return; } i = Math.min(words.length, i + perTick); - const partial = words.slice(0, i).join(" "); - set((s) => { - const chat = [...s.chat]; - const last = chat[chat.length - 1]; - if (last) last.a = partial; - return { chat }; - }); + useStreamStore.getState().setText(words.slice(0, i).join(" ")); }, STREAM_INTERVAL_MS); }; diff --git a/frontend/src/lib/cortex/stream-store.ts b/frontend/src/lib/cortex/stream-store.ts new file mode 100644 index 0000000..2a8d75d --- /dev/null +++ b/frontend/src/lib/cortex/stream-store.ts @@ -0,0 +1,17 @@ +import { create } from "zustand"; + +// The answer typewriter reveals text in ~24 ticks. Writing each tick into the main +// store re-rendered the whole app tree (the top-level view subscribes to the entire +// store), which is the biggest source of jank during a reply. Park the in-flight +// partial text here instead: only the small reads it, so a tick +// re-renders that node alone. The final text is committed back to the chat once, +// streaming is done. +interface StreamState { + text: string; + setText: (text: string) => void; +} + +export const useStreamStore = create((set) => ({ + text: "", + setText: (text) => set({ text }), +}));