Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Original file line number Diff line number Diff line change
Expand Up @@ -5174,4 +5174,59 @@ describe("splitBufferedAssistantText", () => {
rest: "```\n- one\n- two\n",
});
});

it("holds a heading until the block under it is done", () => {
expect(splitBufferedAssistantText("intro\n\n## Setup\n\nInstall it")).toEqual({
ready: "intro\n\n",
rest: "## Setup\n\nInstall it",
});
expect(
splitBufferedAssistantText("intro\n\n# Plan\n\n## Setup\n\nInstall it.\n\nNext"),
).toEqual({
ready: "intro\n\n# Plan\n\n## Setup\n\nInstall it.\n\n",
rest: "Next",
});
});

it("delivers the paragraph above a heading with no blank line between them", () => {
expect(splitBufferedAssistantText("para\n## Setup\n\nInstall")).toEqual({
ready: "para\n",
rest: "## Setup\n\nInstall",
});
// A bold line there continues the paragraph, so both stay buffered.
expect(splitBufferedAssistantText("para\n**Setup**\n\nInstall")).toEqual({
ready: "",
rest: "para\n**Setup**\n\nInstall",
});
});

it("holds a line of only bold text like a heading", () => {
expect(splitBufferedAssistantText("**Risk by area:**\n\n| a |\n|---|\n")).toEqual({
ready: "",
rest: "**Risk by area:**\n\n| a |\n|---|\n",
});
expect(splitBufferedAssistantText("**Use *npm* now**\n\nInstall it")).toEqual({
ready: "",
rest: "**Use *npm* now**\n\nInstall it",
});
expect(splitBufferedAssistantText("**Note:** read this.\n\nNext")).toEqual({
ready: "**Note:** read this.\n\n",
rest: "Next",
});
});

it("delivers a held heading with its first list item or its whole code block", () => {
expect(splitBufferedAssistantText("## Steps\n\n- one\n- tw")).toEqual({
ready: "## Steps\n\n- one\n",
rest: "- tw",
});
expect(splitBufferedAssistantText("## Code\n\n```ts\na\n\nb\n")).toEqual({
ready: "",
rest: "## Code\n\n```ts\na\n\nb\n",
});
expect(splitBufferedAssistantText("## Code\n\n```ts\na\n```\nafter")).toEqual({
ready: "## Code\n\n```ts\na\n```\n",
rest: "after",
});
});
});
27 changes: 25 additions & 2 deletions apps/server/src/orchestration/Layers/ProviderRuntimeIngestion.ts
Original file line number Diff line number Diff line change
Expand Up @@ -211,6 +211,12 @@ const BLANK_LINE_PATTERN = /^[ \t]*$/;
// nested items count. The trailing space is required, so a partial `-` or
// `1.` never matches before the model finishes the marker.
const LIST_ITEM_START_PATTERN = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]/;
// A section title: an ATX heading, or a line of only bold text, which models
// often use as a heading.
const SECTION_TITLE_PATTERN = /^ {0,3}(?:#{1,6}(?:[ \t]|$)|\*\*(?:[^*]|\*(?!\*))+\*\*:?$)/;
// An unindented ATX heading ends the paragraph or list above it, even with no
// blank line between them. A bold line would continue the paragraph instead.
const TOP_LEVEL_HEADING_PATTERN = /^#{1,6}(?:[ \t]|$)/;

/**
* Splits buffered assistant text at the last blank line, closing code fence,
Expand All @@ -221,17 +227,26 @@ const LIST_ITEM_START_PATTERN = /^[ \t]*(?:[-*+]|\d{1,9}[.)])[ \t]/;
* never leaks; a list item start is the one lookahead that may sit on the
* partial line, since tight lists have no blank lines between items and would
* otherwise land all at once.
*
* A section title holds the boundary until a content line follows it, so a
* title never lands alone and waits above a block that is still streaming.
*/
export function splitBufferedAssistantText(text: string): { ready: string; rest: string } {
let openFence: { marker: string; indent: number } | null = null;
let boundary = -1;
let lineStart = 0;
let titleAwaitingContent = false;
for (;;) {
const newline = text.indexOf("\n", lineStart);
const line = text
.slice(lineStart, newline === -1 ? text.length : newline)
.replace(/[ \t\r]+$/, "");
if (openFence === null && lineStart > 0 && LIST_ITEM_START_PATTERN.test(line)) {
if (
openFence === null &&
lineStart > 0 &&
!titleAwaitingContent &&
LIST_ITEM_START_PATTERN.test(line)
) {
boundary = lineStart;
}
if (newline === -1) {
Expand All @@ -243,6 +258,7 @@ export function splitBufferedAssistantText(text: string): { ready: string; rest:
const marker = fenceMatch[2]!;
if (openFence === null) {
openFence = { marker, indent };
titleAwaitingContent = false;
} else if (
marker[0] === openFence.marker[0] &&
marker.length >= openFence.marker.length &&
Expand All @@ -254,7 +270,14 @@ export function splitBufferedAssistantText(text: string): { ready: string; rest:
boundary = newline + 1;
}
} else if (openFence === null && BLANK_LINE_PATTERN.test(line) && lineStart > 0) {
boundary = newline + 1;
if (!titleAwaitingContent) {
boundary = newline + 1;
}
} else if (openFence === null) {
if (lineStart > 0 && !titleAwaitingContent && TOP_LEVEL_HEADING_PATTERN.test(line)) {
boundary = lineStart;
}
titleAwaitingContent = SECTION_TITLE_PATTERN.test(line);
Comment thread
t3dotgg marked this conversation as resolved.
}
lineStart = newline + 1;
}
Expand Down
Loading