From f7330d47c59a57ab2a5d93a0ddaac7ac1202ff9a Mon Sep 17 00:00:00 2001 From: withkarann <31102185+withkarann@users.noreply.github.com> Date: Thu, 3 Sep 2026 04:01:49 +0200 Subject: [PATCH] perf: lex block tokens only and reuse parsed blocks while streaming parseMarkdownIntoBlocks only needs the raw text of each block token, but Lexer.lex also runs the inline tokenizer over every block. Call blockTokens directly instead. While a document streams, text is only appended, so blocks that ended before the last two blocks cannot change. Keep the previous parse and re-lex just the tail of the document when the new input extends the previous one. Co-Authored-By: Claude Fable 5.1 --- .changeset/quick-blocks-stream.md | 5 + .../__benchmarks__/parse-blocks.bench.ts | 21 ++ .../parse-blocks-incremental.test.tsx | 253 ++++++++++++++++++ packages/streamdown/lib/parse-blocks.tsx | 146 +++++++++- 4 files changed, 411 insertions(+), 14 deletions(-) create mode 100644 .changeset/quick-blocks-stream.md create mode 100644 packages/streamdown/__tests__/parse-blocks-incremental.test.tsx diff --git a/.changeset/quick-blocks-stream.md b/.changeset/quick-blocks-stream.md new file mode 100644 index 00000000..03e362c2 --- /dev/null +++ b/.changeset/quick-blocks-stream.md @@ -0,0 +1,5 @@ +--- +"streamdown": patch +--- + +Speed up block parsing by lexing only block tokens and reusing already parsed blocks while a document streams. diff --git a/packages/streamdown/__benchmarks__/parse-blocks.bench.ts b/packages/streamdown/__benchmarks__/parse-blocks.bench.ts index 8cf3f1ae..e02861a6 100644 --- a/packages/streamdown/__benchmarks__/parse-blocks.bench.ts +++ b/packages/streamdown/__benchmarks__/parse-blocks.bench.ts @@ -299,6 +299,27 @@ describe("parseMarkdownIntoBlocks - Streaming Simulation", () => { { iterations: 1000 } ); + // A long document that keeps growing at the end, which is what a streamed + // response looks like once it is a few hundred lines in. + const longDocument = Array.from( + { length: 100 }, + (_, i) => `## Section ${i}\n\nParagraph ${i} with some text.` + ).join("\n\n"); + const longStreamingSteps = Array.from( + { length: 30 }, + (_, i) => `${longDocument}\n\n${"More streamed text. ".repeat(i + 1)}` + ); + + bench( + "streaming text after a long document (30 incremental steps)", + () => { + for (const step of longStreamingSteps) { + parseMarkdownIntoBlocks(step); + } + }, + { iterations: 100 } + ); + const codeStreamingSteps = [ "```javascript", "```javascript\n", diff --git a/packages/streamdown/__tests__/parse-blocks-incremental.test.tsx b/packages/streamdown/__tests__/parse-blocks-incremental.test.tsx new file mode 100644 index 00000000..1bbc592e --- /dev/null +++ b/packages/streamdown/__tests__/parse-blocks-incremental.test.tsx @@ -0,0 +1,253 @@ +import { Lexer } from "marked"; +import { describe, expect, it } from "vitest"; +import { parseMarkdownIntoBlocks } from "../lib/parse-blocks"; + +// Documents chosen so that appended text can change how the tail is lexed: +// setext underlines, lazy continuation lines, blank lines inside lists and +// blockquotes, unclosed fences, HTML that spans blank lines, split math +// blocks, tables that only become tables once the delimiter row arrives, and +// so on. Each one is streamed below and every prefix must parse the same way +// as a full parse of that prefix. +const documents = [ + "# Heading\n\nParagraph one.\n\nParagraph two with **bold** and _em_.\n", + "Setext heading\n===\n\nAnother\n---\n\nNot a heading\n\n===\n", + "- item one\n- item two\n\n continued paragraph\n\n- item three\n - nested\n\n nested paragraph\n", + "1. first\n2. second\nlazy line\n\n3) third\n", + "> quote\nlazy quote line\n\n> another\n>\n> - list in quote\n", + "Paragraph\n not code, lazy\n\n indented code\n more code\n\nafter\n", + "```ts\nconst a = 1;\n\nconst b = 2;\n```\n\ntext\n\n~~~\nunclosed tilde fence\n", + "```js\nconst x = 1;\n```\n\n```python\ny = 2\n```\n", + "Some text\n\n```\nnever closed\n\nstill code\n", + "
\n\ntext inside\n\n
nested
\n\n
\n\nafter div\n", + "
\nTitle\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n
\n", + "\n\nthinking text\n\n\n\nreply\n", + "\n\n
\n\n
\n\nparagraph\n", + "$$\nx = 1\n$$\n\ntext\n\n$$\ny = 2\n=\n$$\n", + "Math with split delimiters\n\n$$\na = b\n\n= c\n$$\n\nafter\n", + "```sh\necho $$\n```\n\n$$\nz\n$$\n", + "| a | b |\n| --- | --- |\n| 1 | 2 |\n| 3 | 4 |\n\ntext after table\n", + "| a | b |\ntext that makes it a paragraph\n\n| c | d |\n|---|---|\n", + "* * *\n\n---\n\n___\n\n- - -\n", + "[ref]: https://example.com\n\nSee [ref] and [^1].\n\n[^1]: footnote\n", + "Line with trailing spaces \nhard break\n\n\n\nmany blank lines\n\n\ntext\n", + "\n\n\nleading blank lines\n\ntext\n", + "Tabs\there\n\n\tindented with tab\n\n- item\n\n\ttab continued\n", + "Windows\r\nline endings\r\n\r\n- list\r\n- items\r\n\r\n```\r\ncode\r\n```\r\n", + "Old mac\rline endings\r\rparagraph\r", + "# H1\n## H2\n### H3\nparagraph\n#not heading\n", + "Term\n: not a definition list\n\n\n\ntext\n", + "\n\n
\npre text\n
\n", + "Intro\n\n1. step\n\n ```\n code in list\n ```\n\n2. step two\n", + "Emoji and unicode: 日本語のテキスト\n\n中文段落\n\n한국어 문단\n", + // Chains of blocks that end only because the next line interrupts them. + "para\n#x\n\npara\n# heading\n\npara\n=\n\npara\n-x\n\npara\n- item\n", + "para\n> quote\n#x\n\npara\n- item\n#x\n\n- item\n> quote\n#x\n", + "para\n```\nfence after para\n```\npara\n***\npara\n
\nhtml\n", + "- a\n- b\n1. c\n2) d\n+ e\n* f\n\n- g\n\n h\n- i\n", + "| a | b |\n|---|---|\n| 1 | 2 |\n#x\n| 3 | 4 |\n\n# real\n", + "> a\n> b\n\n> c\nd\n=\n\n> e\n- f\n", + // A setext underline reaches back across lines that are not blank. + "para\nfoo\n***\nbar\n=\n", + "a\n***\nb\n***\nc\n***\nd\n=\n", + "see [x]\n***\n[x]: /u\nz\n=\n", + "- \n\ntext\n", + // The stable region shrinks when a setext underline pulls blocks together. + "a\n\nb\n\npara\n***\nfoo\n=\nx\n\ny\n\nz\n", +]; + +// Lines that interact in awkward ways when they follow each other without a +// blank line in between. Combined at random below to cover cases nobody +// thought to write down. +const lineSnippets = [ + "para", + "***", + "=", + "-", + "--", + "# h", + "#x", + "- item", + "1. one", + "> q", + " ind", + " code", + "\tx", + "", + "```", + "~~~", + "$$", + "
", + "
", + "
", + "[x]: /u", + "see [x]", + "| a | b |", + "|---|---|", + "", +]; + +// Small deterministic generator so failures are reproducible. +const randomDocuments = (count: number): string[] => { + let seed = 12_345; + const next = () => { + seed = (seed * 1_103_515_245 + 12_345) % 2_147_483_648; + return seed; + }; + const result: string[] = []; + for (let i = 0; i < count; i += 1) { + const lineCount = 2 + (next() % 8); + const lines: string[] = []; + for (let j = 0; j < lineCount; j += 1) { + lines.push(lineSnippets[next() % lineSnippets.length]); + } + result.push(`${lines.join("\n")}\n`); + } + return result; +}; + +const footnotePattern = /\[\^[\w-]+\]/; + +const chunkings: [string, (doc: string) => number[]][] = [ + ["one character", (doc) => Array.from({ length: doc.length }, () => 1)], + [ + "fixed size", + (doc) => Array.from({ length: Math.ceil(doc.length / 5) }, () => 5), + ], + [ + "varying size", + (doc) => { + const sizes: number[] = []; + let total = 0; + let i = 0; + while (total < doc.length) { + const size = 1 + ((i * 7) % 11); + sizes.push(size); + total += size; + i += 1; + } + return sizes; + }, + ], +]; + +// The incremental path only applies when the input extends the previous +// input, so parsing an unrelated document first guarantees a full parse. +const parseFresh = (markdown: string): string[] => { + parseMarkdownIntoBlocks("unrelated\n\ndocument\n"); + return parseMarkdownIntoBlocks(markdown); +}; + +const prefixes = (doc: string, sizes: number[]): string[] => { + const result: string[] = []; + let end = 0; + for (const size of sizes) { + end = Math.min(doc.length, end + size); + result.push(doc.slice(0, end)); + if (end === doc.length) { + break; + } + } + return result; +}; + +describe("parseMarkdownIntoBlocks incremental parsing", () => { + for (const [name, chunk] of chunkings) { + it(`matches a full parse at every prefix when streamed ${name} at a time`, () => { + for (const doc of documents) { + // Make sure the first prefix of this document starts from a cold cache. + parseMarkdownIntoBlocks("unrelated\n\ndocument\n"); + + for (const prefix of prefixes(doc, chunk(doc))) { + const streamed = parseMarkdownIntoBlocks(prefix); + const expected = parseFresh(prefix); + + expect( + streamed, + `prefix of length ${prefix.length} of ${JSON.stringify(doc)}` + ).toEqual(expected); + + // Put the streamed result back so the next prefix extends it. + parseMarkdownIntoBlocks("unrelated\n\ndocument\n"); + parseMarkdownIntoBlocks(prefix); + } + } + }); + } + + it("matches a full parse for randomly combined lines streamed one character at a time", () => { + for (const doc of randomDocuments(400)) { + parseMarkdownIntoBlocks("unrelated\n\ndocument\n"); + + for (let i = 1; i <= doc.length; i += 1) { + const prefix = doc.slice(0, i); + const streamed = parseMarkdownIntoBlocks(prefix); + const expected = parseFresh(prefix); + + expect( + streamed, + `prefix of length ${i} of ${JSON.stringify(doc)}` + ).toEqual(expected); + + parseMarkdownIntoBlocks("unrelated\n\ndocument\n"); + parseMarkdownIntoBlocks(prefix); + } + } + }); + + it("keeps parsing correctly when two documents stream at the same time", () => { + const a = documents[2]; + const b = documents[9]; + const maxLength = Math.max(a.length, b.length); + + for (let i = 1; i <= maxLength; i += 1) { + const prefixA = a.slice(0, i); + const prefixB = b.slice(0, i); + + expect(parseMarkdownIntoBlocks(prefixA)).toEqual(parseFresh(prefixA)); + expect(parseMarkdownIntoBlocks(prefixB)).toEqual(parseFresh(prefixB)); + } + }); + + it("returns the same blocks for the same input on repeated calls", () => { + for (const doc of documents) { + const first = parseMarkdownIntoBlocks(doc); + const second = parseMarkdownIntoBlocks(doc); + expect(second).toEqual(first); + } + }); + + it("joins the blocks back into the normalized input", () => { + for (const doc of documents) { + if (footnotePattern.test(doc)) { + continue; + } + const blocks = parseFresh(doc); + expect(blocks.join("")).toBe(doc.replace(/\r\n|\r/g, "\n")); + } + }); + + it("produces the same raw blocks as marked's full lexer", () => { + for (const doc of documents) { + const normalized = doc.replace(/\r\n|\r/g, "\n"); + const fromLex = Lexer.lex(normalized, { gfm: true }).map( + (token) => token.raw + ); + const fromBlockTokens = new Lexer({ gfm: true }) + .blockTokens(normalized) + .map((token) => token.raw); + expect(fromBlockTokens).toEqual(fromLex); + } + }); + + it("handles a document that grows past the footnote threshold", () => { + const doc = + "Intro paragraph\n\nSecond paragraph with a ref [^1].\n\n[^1]: note\n"; + parseMarkdownIntoBlocks("unrelated\n\ndocument\n"); + + for (let i = 1; i <= doc.length; i += 1) { + const prefix = doc.slice(0, i); + expect(parseMarkdownIntoBlocks(prefix)).toEqual(parseFresh(prefix)); + } + }); +}); diff --git a/packages/streamdown/lib/parse-blocks.tsx b/packages/streamdown/lib/parse-blocks.tsx index 14061bff..3ed0650e 100644 --- a/packages/streamdown/lib/parse-blocks.tsx +++ b/packages/streamdown/lib/parse-blocks.tsx @@ -1,4 +1,4 @@ -import { Lexer } from "marked"; +import { Lexer, type Token } from "marked"; // Regex patterns moved to top level for performance // Footnote identifiers must be alphanumeric, underscore, or hyphen (e.g., [^1], [^note], [^my-note]) @@ -93,23 +93,60 @@ const countDoubleDollars = (str: string): number => { return count; }; -// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: "Complex parsing logic that handles multiple markdown edge cases" -export const parseMarkdownIntoBlocks = (markdown: string): string[] => { - // Check if the markdown contains footnotes (references or definitions) - // Footnote references: [^1], [^label], etc. - // Footnote definitions: [^1]: text, [^label]: text, etc. - // Use atomic groups or possessive quantifiers to prevent backtracking - const hasFootnoteReference = footnoteReferencePattern.test(markdown); - const hasFootnoteDefinition = footnoteDefinitionPattern.test(markdown); +// marked's `Lexer.lex` normalizes line endings before tokenizing. Do the same +// here so that the `raw` text of the block tokens joins back into the input. +const lineEndingPattern = /\r\n|\r/g; - // If footnotes are present, return the entire document as a single block - // This ensures footnote references and definitions remain in the same mdast tree - if (hasFootnoteReference || hasFootnoteDefinition) { - return [markdown]; +// Only the block-level tokens are needed here: each block is rendered from its +// `raw` text by its own remark pipeline later. `Lexer.lex` would also run the +// inline tokenizer over every block, which is wasted work, so call the block +// tokenizer directly. +const lexBlocks = (markdown: string): Token[] => + new Lexer({ gfm: true }).blockTokens(markdown); + +// Streaming appends text to the end of the document. Text before the tail can +// still change meaning: a lone "#" is a heading that ends the paragraph above +// it, while "#x" continues that paragraph; "2" after a list is a paragraph, +// while "2." is another item of that list. A block is only final once it ends +// with a blank line and the block after it is complete, that is, followed by +// another block. Blocks before the last such boundary are reused and only the +// rest of the document is lexed again. +// +// A single cached entry covers one document streaming at a time; anything +// else falls back to a full parse. The entry keeps the last document in +// memory for the lifetime of the module. +interface ParseCache { + blocks: string[]; + input: string; + // How many leading blocks have been checked to sit at their expected + // offsets in `input`. marked trims a few raws (a bare "- " lexes to "-\n"), + // so the offsets used below are checked before they are trusted, but only + // for blocks that are about to be reused. + verifiedCount: number; + verifiedLength: number; +} + +let lastParse: ParseCache | null = null; + +const blankLineEnding = "\n\n"; + +// Number of leading blocks that cannot change when text is appended. +const countStableBlocks = (blocks: string[]): number => { + for (let i = blocks.length - 3; i >= 0; i -= 1) { + if (blocks[i].endsWith(blankLineEnding)) { + return i + 1; + } } + return 0; +}; - const tokens = Lexer.lex(markdown, { gfm: true }); +// A block is a slice of the input it was lexed from, and V8 keeps that whole +// input alive while the slice exists. Copy the blocks lexed from the tail so +// the cache does not hold on to every intermediate document of a stream. +const copyString = (value: string): string => ` ${value}`.slice(1); +// biome-ignore lint/complexity/noExcessiveCognitiveComplexity: "Complex parsing logic that handles multiple markdown edge cases" +const mergeTokensIntoBlocks = (tokens: Token[]): string[] => { // Post-process to merge consecutive blocks that belong together const mergedBlocks: string[] = []; const htmlStack: string[] = []; // Track opening HTML tags @@ -192,3 +229,84 @@ export const parseMarkdownIntoBlocks = (markdown: string): string[] => { return mergedBlocks; }; + +// Reuses the blocks of the previous parse that cannot have changed and lexes +// only the rest of the input. Returns null when nothing can be reused. +const reuseParsedBlocks = ( + previous: ParseCache, + input: string +): ParseCache | null => { + if (input.length <= previous.input.length) { + return null; + } + + const stableCount = countStableBlocks(previous.blocks); + if (stableCount === 0 || !input.startsWith(previous.input)) { + return null; + } + + let verifiedCount = previous.verifiedCount; + let verifiedLength = previous.verifiedLength; + + // The tail can shrink the stable region (a setext underline can pull + // several blocks into one), so never trust more blocks than are stable. + if (verifiedCount > stableCount) { + verifiedCount = stableCount; + verifiedLength = 0; + for (let i = 0; i < verifiedCount; i += 1) { + verifiedLength += previous.blocks[i].length; + } + } + + while ( + verifiedCount < stableCount && + input.startsWith(previous.blocks[verifiedCount], verifiedLength) + ) { + verifiedLength += previous.blocks[verifiedCount].length; + verifiedCount += 1; + } + + if (verifiedCount !== stableCount) { + return null; + } + + const tailBlocks = mergeTokensIntoBlocks( + lexBlocks(input.slice(verifiedLength)) + ).map(copyString); + + return { + input, + blocks: previous.blocks.slice(0, stableCount).concat(tailBlocks), + verifiedCount, + verifiedLength, + }; +}; + +export const parseMarkdownIntoBlocks = (markdown: string): string[] => { + // Check if the markdown contains footnotes (references or definitions) + // Footnote references: [^1], [^label], etc. + // Footnote definitions: [^1]: text, [^label]: text, etc. + // Use atomic groups or possessive quantifiers to prevent backtracking + const hasFootnoteReference = footnoteReferencePattern.test(markdown); + const hasFootnoteDefinition = footnoteDefinitionPattern.test(markdown); + + // If footnotes are present, return the entire document as a single block + // This ensures footnote references and definitions remain in the same mdast tree + if (hasFootnoteReference || hasFootnoteDefinition) { + return [markdown]; + } + + const input = markdown.includes("\r") + ? markdown.replace(lineEndingPattern, "\n") + : markdown; + const reused = lastParse ? reuseParsedBlocks(lastParse, input) : null; + const { blocks, verifiedCount, verifiedLength } = reused ?? { + blocks: mergeTokensIntoBlocks(lexBlocks(input)), + verifiedCount: 0, + verifiedLength: 0, + }; + + lastParse = { input, blocks, verifiedCount, verifiedLength }; + + return blocks; +};