diff --git a/apps/mobile/modules/t3-markdown-text/package.json b/apps/mobile/modules/t3-markdown-text/package.json index 376befbeb152..ce2991cd592e 100644 --- a/apps/mobile/modules/t3-markdown-text/package.json +++ b/apps/mobile/modules/t3-markdown-text/package.json @@ -26,7 +26,8 @@ "./markdown": "./src/nativeMarkdownText.ts", "./primitive": "./src/MarkdownTextPrimitive.tsx", "./renderer": "./src/SelectableMarkdownText.tsx", - "./types": "./src/SelectableMarkdownText.types.ts" + "./types": "./src/SelectableMarkdownText.types.ts", + "./fence-key": "./src/stableFenceKey.ts" }, "peerDependencies": { "@t3tools/client-runtime": "*", diff --git a/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx b/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx index b57a182dbc56..800acb473707 100644 --- a/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx +++ b/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx @@ -17,6 +17,7 @@ import type { NativeMarkdownTextStyle, SelectableMarkdownSkill, } from "./SelectableMarkdownText.types"; +import { useStableFenceKey } from "./stableFenceKey"; import { useHighlightedCode, type HighlightedCode } from "./useHighlightedCode"; /** Set by SelectableMarkdownText so images anywhere in the block tree can use it. */ @@ -156,6 +157,7 @@ function NativeCodeBlock(props: { const theme = colorScheme === "dark" ? "dark" : "light"; const highlighted = useHighlightedCode(content, props.node.language, theme, props.highlightCode); const languageLabel = props.node.language?.toUpperCase() ?? "CODE"; + const fenceKey = useStableFenceKey(languageLabel, content); return ( { + const language = "ts"; + const shared = `const value = 1;\n${"x".repeat(24)}`; + const first = `${shared}${"a".repeat(40)}`; + const second = `${shared}${"b".repeat(40)}`; + expect(first.length).toBe(second.length); + expect(first.slice(0, 32)).toBe(second.slice(0, 32)); + expect(first).not.toBe(second); + + const afterFirst = nextFenceKey(null, language, first); + const afterSecond = nextFenceKey(afterFirst, language, second); + + expect(afterFirst.key).toBe(`${language}:${first}`); + expect(afterSecond.key).toBe(`${language}:${second}`); + expect(afterSecond.key).not.toBe(afterFirst.key); +}); + +it("keeps the key while the same fence streams longer", () => { + let state = nextFenceKey(null, "ts", "const"); + const initialKey = state.key; + state = nextFenceKey(state, "ts", "const value"); + state = nextFenceKey(state, "ts", "const value = 1;"); + expect(state.key).toBe(initialKey); +}); + +it("replaces the key when a streamed fence is swapped for another", () => { + let streamed = nextFenceKey(null, "ts", "const"); + streamed = nextFenceKey(streamed, "ts", "const value = 1;\nreturn value;"); + const swapped = nextFenceKey(streamed, "ts", "let other = 2;\nreturn other;"); + expect(swapped.key).toBe("ts:let other = 2;\nreturn other;"); + expect(swapped.key).not.toBe(streamed.key); +}); diff --git a/apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts b/apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts new file mode 100644 index 000000000000..860399880b6b --- /dev/null +++ b/apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts @@ -0,0 +1,45 @@ +import { useState } from "react"; + +export interface FenceKeyState { + readonly languageLabel: string; + readonly content: string; + readonly key: string; +} + +/** + * Identity for a fence's horizontal scroller and copy button. + * + * The key includes the full fence text when the fence changes, so a recycled + * row cannot keep another block's scroll offset or copied state just because + * the language, length, and opening characters match. Streaming appends keep + * the previous key; otherwise every chunk would remount the token tree. + */ +export function nextFenceKey( + previous: FenceKeyState | null, + languageLabel: string, + content: string, +): FenceKeyState { + if ( + previous !== null && + previous.languageLabel === languageLabel && + content.startsWith(previous.content) + ) { + return { languageLabel, content, key: previous.key }; + } + return { languageLabel, content, key: `${languageLabel}:${content}` }; +} + +/** Remembers the last fence drawn in this container. */ +export function useStableFenceKey(languageLabel: string, content: string): string { + const [stored, setStored] = useState(null); + const next = nextFenceKey(stored, languageLabel, content); + if ( + stored === null || + stored.languageLabel !== languageLabel || + stored.content !== content || + stored.key !== next.key + ) { + setStored(next); + } + return next.key; +} diff --git a/apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx b/apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx new file mode 100644 index 000000000000..6b90550396c2 --- /dev/null +++ b/apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx @@ -0,0 +1,160 @@ +import { memo } from "react"; +import { Platform, ScrollView, StyleSheet, Text, View } from "react-native"; + +import type { NativeMarkdownTextStyle } from "@t3tools/mobile-markdown-text/types"; + +import { CopyTextButton } from "../../components/CopyTextButton"; +import type { AndroidTranscriptCodePart } from "./androidTranscriptSlices"; + +const MONO_FONT_FAMILY = Platform.select({ + ios: "ui-monospace", + android: "monospace", + default: "monospace", +}); + +/** + * Code inside a transcript slice scales with the body size, matching the + * highlighted fence (12pt at the default 15pt body). + */ +function codeSliceFontSize(textStyle: NativeMarkdownTextStyle): number { + return Math.max(10, Math.round(textStyle.fontSize * 0.8)); +} + +/** + * Line height for a plain code window. Kept identical to the highlighted fence + * so a windowed block does not jump when the reader stops on it. + */ +function codeSliceLineHeight(textStyle: NativeMarkdownTextStyle): number { + return codeSliceFontSize(textStyle) + 6; +} + +/** + * Corner radii for one window of a fence. Middle and trailing windows stay + * square on the joined edge so the windows read as a single card. + */ +function codeSliceRadius(part: AndroidTranscriptCodePart): { + readonly borderTopLeftRadius: number; + readonly borderTopRightRadius: number; + readonly borderBottomLeftRadius: number; + readonly borderBottomRightRadius: number; +} { + const radius = part === "start" || part === "end" ? 10 : 0; + return { + borderTopLeftRadius: part === "start" ? radius : 0, + borderTopRightRadius: part === "start" ? radius : 0, + borderBottomLeftRadius: part === "end" ? radius : 0, + borderBottomRightRadius: part === "end" ? radius : 0, + }; +} + +/** + * One plain-text window of a long fenced block. + * + * Highlighted fences mount a `Text` per Shiki token. On Android that tree is + * created on the UI thread in the frame the row enters, which is the settled- + * thread hitch. A window is a single non-selectable `Text`; the header copies + * the whole fence, not just the lines in view. + */ +export const AndroidTranscriptCodeSlice = memo(function AndroidTranscriptCodeSlice(props: { + readonly text: string; + readonly language: string | null; + readonly part: Exclude; + readonly fullCode: string; + readonly textStyle: NativeMarkdownTextStyle; +}) { + const fontSize = codeSliceFontSize(props.textStyle); + const lineHeight = codeSliceLineHeight(props.textStyle); + const showHeader = props.part === "start"; + const languageLabel = props.language?.toUpperCase() ?? "CODE"; + return ( + + {showHeader ? ( + + + {languageLabel} + + + + ) : null} + + + {props.text} + + + + ); +}); + +const styles = StyleSheet.create({ + card: { + borderCurve: "continuous", + borderWidth: 1, + overflow: "hidden", + }, + header: { + minHeight: 42, + borderBottomWidth: 1, + paddingLeft: 14, + paddingRight: 6, + flexDirection: "row", + alignItems: "center", + justifyContent: "space-between", + }, + body: { + paddingHorizontal: 14, + paddingVertical: 12, + }, +}); diff --git a/apps/mobile/src/features/threads/ThreadFeed.tsx b/apps/mobile/src/features/threads/ThreadFeed.tsx index 83b0bef9c022..1197d4e0bac0 100644 --- a/apps/mobile/src/features/threads/ThreadFeed.tsx +++ b/apps/mobile/src/features/threads/ThreadFeed.tsx @@ -5,7 +5,12 @@ import { } from "./worktree-setup-card"; import * as Haptics from "expo-haptics"; import { KeyboardAwareLegendList } from "@legendapp/list/keyboard"; -import { useViewabilityAmount, type LegendListRef } from "@legendapp/list/react-native"; +import { + useRecyclingEffect, + useViewabilityAmount, + type LegendListRecyclingState, + type LegendListRef, +} from "@legendapp/list/react-native"; import type { ChatAttachment, ChatFileAttachment, @@ -138,6 +143,7 @@ import { resolveNativeMarkdownTypography, } from "../../lib/appearancePreferences"; import { useAppearancePreferences } from "../settings/appearance/AppearancePreferencesProvider"; +import { useStableFenceKey } from "@t3tools/mobile-markdown-text/fence-key"; import { markdownFileIconSource } from "@t3tools/mobile-markdown-text/file-icons"; import { PierreEntryIcon } from "../../components/PierreEntryIcon"; import { markdownLinkIconSource } from "@t3tools/mobile-markdown-text/link-icons"; @@ -172,6 +178,14 @@ import { WORK_GROUP_TOGGLE_HEIGHT, } from "./thread-work-log"; import { appendPendingThreadMessages, type PendingThreadFeedEntry } from "./pending-thread-feed"; +import { AndroidTranscriptCodeSlice } from "./AndroidTranscriptCodeSlice"; +import { + androidTranscriptItemType, + assistantSliceGap, + assistantSliceMarkdown, + expandAndroidAssistantTranscriptRows, + type AndroidAssistantSliceEntry, +} from "./androidTranscriptSlices"; import type { QueuedThreadMessage } from "../../state/thread-outbox-model"; import { useMarkdownCodeHighlight } from "./markdownCodeHighlightState"; import { @@ -235,15 +249,53 @@ const THREAD_FEED_DISCLOSURE_ENTER_TRANSITION = FadeIn.delay( THREAD_DISCLOSURE_TRANSITION_MS, ).duration(140); -// Entering animations must only play for rows born just now — LegendList -// remounts rows when they scroll back into view, and replaying an entrance for -// old content would be its own kind of jank. +// Entering animations must only play for rows born just now. Replaying one when +// a settled row scrolls back into view is its own kind of jank. const FRESH_ENTRY_WINDOW_MS = 3_000; +/** True when a row was created moments ago and may fade in as it mounts. */ function isFreshTimestamp(input: string): boolean { const timestamp = Date.parse(input); return Number.isFinite(timestamp) && Date.now() - timestamp < FRESH_ENTRY_WINDOW_MS; } +/** + * Android reuses list containers. iOS remounts them; UITextView is cheap, and + * recycling would reuse its selection state. + */ +const RECYCLE_ANDROID_TRANSCRIPT_ROWS = Platform.OS === "android"; + +/** + * Row id LegendList last painted in this container, when the item has one. + * Streaming updates replace the object but keep the id; a recycle does not. + */ +function rowRecycleId(item: unknown): string | undefined { + if (typeof item !== "object" || item === null || !("id" in item)) { + return undefined; + } + const id = item.id; + return typeof id === "string" ? id : undefined; +} + +/** + * Runs `reset` when this container is reused for a different feed row. + * Same-id updates, including a streaming append, do not count as a recycle. + * Outside a list, or on iOS where rows remount, the effect never fires. + */ +function useResetOnRowRecycle(reset: () => void) { + const resetRef = useRef(reset); + resetRef.current = reset; + useRecyclingEffect( + useCallback((info: LegendListRecyclingState) => { + const previousId = rowRecycleId(info.prevItem); + const nextId = rowRecycleId(info.item); + if (previousId !== undefined && previousId === nextId) { + return; + } + resetRef.current(); + }, []), + ); +} + export interface ThreadFeedProps { readonly worktreeSetup?: WorktreeSetupCardProps | null; readonly setupWorkingStartedAt?: string | null; @@ -279,6 +331,10 @@ export interface ThreadFeedProps { } | null; } +/** + * Image attachment thumbnail. A failed load refreshes the URL once. Recycling + * the row clears that retry so the next message can load its own image. + */ function MessageAttachmentImage(props: { readonly environmentId: EnvironmentId; readonly attachmentId: string; @@ -300,6 +356,9 @@ function MessageAttachmentImage(props: { const uri = useAssetUrl(props.environmentId, resource); const refreshAssetUrl = useRefreshAssetUrl(props.environmentId, resource); const retriedImage = useRef(false); + useResetOnRowRecycle(() => { + retriedImage.current = false; + }); if (uri === null) { return ( @@ -362,6 +421,10 @@ function isFileAttachment(attachment: ChatAttachment): attachment is ChatFileAtt return attachment.type === "file"; } +/** + * File or video attachment row. An in-flight open is aborted when the row is + * recycled, so a reused container cannot finish the previous file's open. + */ function MessageAttachmentFile(props: { readonly environmentId: EnvironmentId; readonly attachment: ChatFileAttachment; @@ -399,6 +462,11 @@ function MessageAttachmentFile(props: { : null; const openingRef = useRef(null); const [opening, setOpening] = useState(false); + useResetOnRowRecycle(() => { + openingRef.current?.abort(); + openingRef.current = null; + setOpening(false); + }); useFocusEffect( useCallback(() => { @@ -562,8 +630,13 @@ const ThreadMediaVisibleContext = createContext(false); // LegendList only computes hook visibility when the list has a viewability config. const THREAD_MEDIA_VIEWABILITY_CONFIG = { itemVisiblePercentThreshold: 0 }; +/** + * Gates video thumbnails on viewability. Visibility belongs to the row in this + * container, so a recycled container starts hidden until the new row reports. + */ function ThreadMediaVisibility(props: { readonly children: ReactNode }) { const [visible, setVisible] = useState(false); + useResetOnRowRecycle(() => setVisible(false)); useViewabilityAmount( useCallback((token) => setVisible(token.sizeVisible > 0), []), ); @@ -634,6 +707,10 @@ const markdownLinkStyles = StyleSheet.create({ }, }); +/** + * External link with a favicon. A host that failed to load its icon is + * remembered on the row and cleared when that row is recycled. + */ const MarkdownExternalLink = memo(function MarkdownExternalLink(props: { readonly children: ReactNode; readonly color: string; @@ -642,6 +719,7 @@ const MarkdownExternalLink = memo(function MarkdownExternalLink(props: { readonly onPress: (href: string) => void; }) { const [failedHost, setFailedHost] = useState(null); + useResetOnRowRecycle(() => setFailedHost(null)); const linkIcon = resolveMarkdownLinkIcon(props.host); const faviconUrl = linkIcon ? null : faviconUrlForOrigin(`https://${props.host}`); @@ -828,6 +906,10 @@ const AssistantMarkdownContent = memo(function AssistantMarkdownContent(props: { }); }); +/** + * Highlighted fenced block. The copy control and horizontal scroller remount + * when the fence changes so a recycled row does not reuse the previous one. + */ function MarkdownCodeBlock(props: { readonly backgroundColor: string; readonly borderColor: string; @@ -843,6 +925,10 @@ function MarkdownCodeBlock(props: { }) { const content = props.content.replace(/\n$/, ""); const languageLabel = props.language?.trim() || "text"; + // Full fence text, kept stable while the same fence streams. A prefix of the + // body is not enough: recycled rows with the same language and opening + // characters would keep the previous scroll offset and copy state. + const fenceKey = useStableFenceKey(languageLabel, content); const highlighted = useMarkdownCodeHighlight({ code: content, enabled: props.highlightCode && Boolean(props.language?.trim()), @@ -872,6 +958,7 @@ function MarkdownCodeBlock(props: { {languageLabel} >; + +/** + * Draws one window of an assistant message that was split so Android can mount + * it across frames. Chrome (attachments, copy, timestamp) stays on the last + * window; the copy still covers the whole message. + */ +function renderAssistantTranscriptSlice( + entry: AndroidAssistantSliceEntry>, + props: Parameters[1], +) { + const { message } = entry.source; + const { markdownStyles, iconSubtleColor } = props; + const styles = markdownStyles.assistant; + const renderedFullText = renderAssistantCitationsAsText(message.text); + const sliceMarkdown = assistantSliceMarkdown(entry.slice); + const timestampLabel = formatMessageTime(message.updatedAt); + const attachments = message.attachments ?? []; + const assistantTurnStillInProgress = + props.unsettledTurnId !== null && message.turnId === props.unsettledTurnId; + const showAssistantMeta = + entry.isLast && + props.terminalAssistantMessageIds.has(message.id) && + !assistantTurnStillInProgress && + !message.streaming; + const hasWideBlock = + entry.slice.kind === "code" || + hasWideMarkdownBlock(sliceMarkdown ?? "", WIDE_MARKDOWN_BLOCK_OPTIONS); + const enterAnimated = entry.isFirst && isFreshTimestamp(message.createdAt); + const gap = assistantSliceGap(entry.slice, entry.isLast); + + return ( + 0 ? { marginBottom: gap } : undefined} + {...(enterAnimated ? { entering: FadeIn.duration(220) } : {})} + > + {sliceMarkdown && sliceMarkdown.trim().length > 0 ? ( + + + + ) : entry.slice.kind === "code" && entry.slice.codePart !== "only" ? ( + + ) : null} + {entry.isLast + ? attachments.map((attachment) => { + return isImageAttachment(attachment) ? ( + + ) : isFileAttachment(attachment) ? ( + + ) : ( + + ); + }) + : null} + {showAssistantMeta ? ( + + + + {timestampLabel} + + + ) : null} + + ); +} + +/** + * Renders one transcript row. Long assistant messages arrive already sliced on + * Android; every other row is unchanged. + */ function renderFeedEntry( - info: { item: PendingThreadFeedEntry; index: number }, + info: { item: PresentedThreadFeedEntry; index: number }, props: Pick< ThreadFeedProps, | "environmentId" @@ -1391,6 +1590,9 @@ function renderFeedEntry( }, ) { const entry = info.item; + if (entry.type === "assistant-slice") { + return renderAssistantTranscriptSlice(entry, props); + } const { markdownStyles, iconSubtleColor, userBubbleColor } = props; if (entry.type === "turn-fold") { @@ -1623,6 +1825,7 @@ function renderFeedEntry( value={props.userBubbleMaxWidth - USER_BUBBLE_HORIZONTAL_PADDING * 2} > 0 ? ( - appendPendingThreadMessages( - deriveThreadFeedPresentation( + expandAndroidAssistantTranscriptRows( + appendPendingThreadMessages( + deriveThreadFeedPresentation( + props.feed, + props.latestTurn, + expandedTurnIds, + expandedWorkGroupIds, + props.activeWorkStartedAt, + ), props.feed, - props.latestTurn, - expandedTurnIds, - expandedWorkGroupIds, - props.activeWorkStartedAt, + props.queuedMessages, ), - props.feed, - props.queuedMessages, + Platform.OS, ), [ props.queuedMessages, @@ -2600,7 +2815,7 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { } }, [settleDisclosureAfterLayout]); - const shouldRestoreVisibleContentPosition = useCallback((entry: ThreadFeedEntry) => { + const shouldRestoreVisibleContentPosition = useCallback((entry: { readonly id: string }) => { const disclosureAnchorKey = disclosureAnchorKeyRef.current; return disclosureAnchorKey === null || entry.id === disclosureAnchorKey; }, []); @@ -2713,11 +2928,13 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { // exact; message rows stay undefined and use LegendList's per-type running // average once one of their type has been measured. const getFixedItemSize = useCallback( - (entry: ThreadFeedEntry) => { + (entry: PresentedThreadFeedEntry) => { if (workRowSizing.fixedRowHeight === undefined) { return undefined; } switch (entry.type) { + case "assistant-slice": + return undefined; case "message": // A collapsed reasoning row is the same chrome as a work toggle. return entry.message.role === "reasoning" && !expandedReasoningMessageIds.has(entry.id) @@ -2747,11 +2964,14 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { // Disclosures can mount existing offscreen rows as well as new work rows. // Fade those in after movement; never retain removed rows over replacements. const renderItem = useCallback( - (info: { item: PendingThreadFeedEntry; index: number }) => ( - + (info: { item: PresentedThreadFeedEntry; index: number }) => { + const entering = disclosureToggleSettling + ? THREAD_FEED_DISCLOSURE_ENTER_TRANSITION + : undefined; + // Android recycles this container. A key would destroy the native text + // tree every time a settled row comes back. iOS keeps the key because + // those rows remount. + const row = ( {renderFeedEntry(info, { environmentId: props.environmentId, @@ -2793,8 +3013,16 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { ) : null} - - ), + ); + if (RECYCLE_ANDROID_TRANSCRIPT_ROWS) { + return {row}; + } + return ( + + {row} + + ); + }, [ props.worktreeSetup, props.setupWorkingStartedAt, @@ -2932,8 +3160,13 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { viewabilityConfig={THREAD_MEDIA_VIEWABILITY_CONFIG} keyExtractor={(entry) => entry.id} getItemType={(entry) => - entry.type === "message" ? `message:${entry.message.role}` : entry.type + androidTranscriptItemType(entry) ?? + (entry.type === "message" ? `message:${entry.message.role}` : entry.type) } + // Recycling is list-wide, so stateful children are keyed by row id + // and reset when the container is reused. Assistant text is not + // keyed: a small slice updates in place instead of remounting. + recycleItems={RECYCLE_ANDROID_TRANSCRIPT_ROWS} getFixedItemSize={getFixedItemSize} // Virtualized rows must move with their measurements. Native layout // transitions can retain stale positions during sync, even at duration 0. diff --git a/apps/mobile/src/features/threads/androidTranscriptSlices.test.ts b/apps/mobile/src/features/threads/androidTranscriptSlices.test.ts new file mode 100644 index 000000000000..60a9b00fc6bc --- /dev/null +++ b/apps/mobile/src/features/threads/androidTranscriptSlices.test.ts @@ -0,0 +1,463 @@ +import { describe, expect, it } from "vite-plus/test"; + +import { + ANDROID_TRANSCRIPT_CODE_LINE_BUDGET, + ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET, + ANDROID_TRANSCRIPT_SLICE_GAP, + androidTranscriptItemType, + assistantSliceGap, + assistantSliceMarkdown, + expandAndroidAssistantTranscriptRows, + fencedCodeMarkdown, + splitAssistantTranscriptSlices, + type AndroidTranscriptSlice, +} from "./androidTranscriptSlices"; + +/** Minimal feed message for slice tests. */ +function message(id: string, role: "assistant" | "user", text: string) { + return { + type: "message" as const, + id, + createdAt: "2026-01-01T00:00:00.000Z", + message: { role, text }, + }; +} + +/** Assistant message fixture. */ +function assistantMessage(id: string, text: string) { + return message(id, "assistant", text); +} + +/** Fenced block with a numbered line per row, so windows are easy to count. */ +function codeFence(language: string, lineCount: number): string { + const body = Array.from({ length: lineCount }, (_, index) => `line ${index + 1}`).join("\n"); + return `\`\`\`${language}\n${body}\n\`\`\``; +} + +/** Compact kind/part/line-count label for slice assertions. */ +function sliceKinds(slices: readonly AndroidTranscriptSlice[]): string[] { + return slices.map((slice) => + slice.kind === "code" ? `code:${slice.codePart}:${slice.text.split("\n").length}` : "markdown", + ); +} + +/** Markdown slices only. Code windows are checked separately. */ +function markdownTexts(slices: readonly AndroidTranscriptSlice[]): string[] { + return slices.filter((slice) => slice.kind === "markdown").map((slice) => slice.text); +} + +describe("splitAssistantTranscriptSlices", () => { + it("leaves a short message with one fence as a single row", () => { + const markdown = `See this.\n\n${codeFence("ts", 4)}`; + expect(splitAssistantTranscriptSlices(markdown)).toBeNull(); + }); + + it("splits prose that would mount as one oversized selectable text", () => { + const paragraph = "word ".repeat(200).trim(); + const markdown = `${paragraph}\n\n${paragraph}`; + const slices = splitAssistantTranscriptSlices(markdown); + expect(slices).not.toBeNull(); + expect(slices!.every((slice) => slice.kind === "markdown")).toBe(true); + for (const slice of slices!) { + expect(slice.text.length).toBeLessThanOrEqual(ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET); + } + const words = (value: string) => value.replaceAll(/\s+/g, " ").trim(); + expect(words(slices!.map((slice) => slice.text).join(" "))).toBe(words(markdown)); + }); + + it("puts each fence on its own row when a message has more than one", () => { + const markdown = `${codeFence("ts", 3)}\n\nBetween.\n\n${codeFence("go", 2)}`; + const slices = splitAssistantTranscriptSlices(markdown); + expect(sliceKinds(slices!)).toEqual(["code:only:3", "markdown", "code:only:2"]); + expect(assistantSliceMarkdown(slices![0]!)).toContain("```ts"); + expect(assistantSliceMarkdown(slices![2]!)).toContain("```go"); + }); + + it("windows a long fence and keeps earlier windows stable as it grows", () => { + const before = codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 1); + const after = codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET * 2 + 3); + const first = splitAssistantTranscriptSlices(before); + const grown = splitAssistantTranscriptSlices(after); + expect(sliceKinds(first!)).toEqual([ + `code:start:${ANDROID_TRANSCRIPT_CODE_LINE_BUDGET}`, + "code:end:1", + ]); + expect(sliceKinds(grown!)).toEqual([ + `code:start:${ANDROID_TRANSCRIPT_CODE_LINE_BUDGET}`, + `code:middle:${ANDROID_TRANSCRIPT_CODE_LINE_BUDGET}`, + "code:end:3", + ]); + const grownHead = grown![0]!; + const firstHead = first![0]!; + const grownNext = grown![1]!; + const firstNext = first![1]!; + expect(grownHead).toMatchObject({ + key: firstHead.key, + text: firstHead.text, + }); + expect(grownNext.key).toBe(firstNext.key); + expect(grownHead.kind === "code" && grownHead.text.startsWith("line 1")).toBe(true); + expect(grownHead.kind === "code" && grownHead.text.includes("line 17")).toBe(false); + expect(assistantSliceMarkdown(grown![0]!)).toBeNull(); + expect( + grown!.every( + (slice) => + slice.kind !== "code" || + slice.fullCode.split("\n").length === ANDROID_TRANSCRIPT_CODE_LINE_BUDGET * 2 + 3, + ), + ).toBe(true); + }); + + it("treats an unclosed streaming fence as code and does not renumber finished windows", () => { + const opened = `Intro.\n\n\`\`\`ts\n${Array.from({ length: 10 }, (_, index) => `line ${index + 1}`).join("\n")}`; + const longer = `${opened}\n${Array.from({ length: 12 }, (_, index) => `line ${index + 11}`).join("\n")}`; + const before = splitAssistantTranscriptSlices(opened); + const after = splitAssistantTranscriptSlices(longer); + expect(before).toBeNull(); + const intro = after![0]!; + const codeHead = after![1]!; + expect(intro).toMatchObject({ kind: "markdown", text: "Intro." }); + expect(codeHead).toMatchObject({ kind: "code", codePart: "start" }); + expect(codeHead.kind === "code" && codeHead.text.split("\n")).toHaveLength( + ANDROID_TRANSCRIPT_CODE_LINE_BUDGET, + ); + }); + + it("keeps a GFM table intact when the surrounding message is split", () => { + const cell = "c".repeat(80); + const row = `| ${cell} | ${cell} |`; + const table = [row, "| --- | --- |", row, row].join("\n"); + const slices = splitAssistantTranscriptSlices(`${table}\n\n${"word ".repeat(200).trim()}`); + const tableSlices = slices!.filter((slice) => slice.text.includes("| --- |")); + expect(tableSlices).toHaveLength(1); + expect(tableSlices[0]!.text).toBe(table); + }); + + it("ignores a four-space indented fence and still splits long prose", () => { + const indented = ` \`\`\`\n${"x".repeat(ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET + 40)}`; + const slices = splitAssistantTranscriptSlices(indented); + expect(slices!.every((slice) => slice.kind === "markdown")).toBe(true); + }); + + it("does not treat a backtick fence as code when the info string contains a backtick", () => { + const prose = `\`\`\`js const tick = \`oops\`\n${"word ".repeat(200).trim()}`; + const slices = splitAssistantTranscriptSlices(prose); + expect(slices?.some((slice) => slice.kind === "code") ?? false).toBe(false); + }); + + it("splits an ordered list only between complete items", () => { + const items = Array.from( + { length: 30 }, + (_, index) => `${index + 1}. ${"entry ".repeat(8).trim()}`, + ); + const slices = splitAssistantTranscriptSlices(items.join("\n")); + expect(slices).not.toBeNull(); + expect(slices!.length).toBeGreaterThan(1); + const lines = slices!.flatMap((slice) => + slice.text.split("\n").filter((line) => line.trim().length > 0), + ); + expect(lines).toEqual(items); + for (const slice of slices!) { + expect(slice.kind).toBe("markdown"); + expect(slice.text).toMatch(/^\d+\. /); + } + }); + + it("keeps a long ordered-list item whole instead of dropping its marker", () => { + const item = `12. ${"word ".repeat(200).trim()}`; + const tail = "after ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${item}\n\n${tail}`)!; + const owner = slices.find((slice) => slice.text.includes("12. ")); + expect(owner?.text.startsWith("12. ")).toBe(true); + expect(owner?.text).toContain("word"); + expect(owner?.text.includes(tail.slice(0, 12))).toBe(false); + for (const slice of slices) { + if (slice === owner) continue; + expect(slice.text.startsWith("word")).toBe(false); + } + }); + + it("keeps nested list items and loose continuation with their parent", () => { + const parent = `1. parent item\n - child stays\n - child two\n\n continuation stays`; + const siblings = Array.from( + { length: 4 }, + (_, index) => `${index + 2}. ${"entry ".repeat(160).trim()}`, + ); + const slices = splitAssistantTranscriptSlices(`${parent}\n\n${siblings.join("\n")}`)!; + const first = slices.find((slice) => slice.text.includes("parent item")); + expect(first?.text).toContain("- child stays"); + expect(first?.text).toContain("- child two"); + expect(first?.text).toContain("continuation stays"); + expect(first?.text).not.toContain("2. "); + }); + + it("keeps a lazy list continuation with its marker", () => { + const item = `1. short\n${"lazy ".repeat(40).trim()}`; + const tail = "after ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${item}\n\n${tail}`)!; + const owner = slices.find((slice) => slice.text.includes("1. short")); + expect(owner?.text.startsWith("1. short")).toBe(true); + expect(owner?.text).toContain("lazy"); + expect(owner?.text.includes("after")).toBe(false); + }); + + it("keeps lazy blockquote continuation with the quote marker", () => { + const quote = `> quoted start\n${"still quoted ".repeat(30).trim()}`; + const after = "outside ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${quote}\n\n${after}`)!; + const quoteSlice = slices.find((slice) => slice.text.includes("> quoted start")); + expect(quoteSlice?.text.startsWith(">")).toBe(true); + expect(quoteSlice?.text).toContain("still quoted"); + expect(quoteSlice?.text.includes("outside")).toBe(false); + }); + + it("keeps a multi-line setext heading with its underline", () => { + const heading = `${"Title words ".repeat(40).trim()}\n${"still the title ".repeat(20).trim()}\n---`; + const after = "body ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${heading}\n\n${after}`)!; + const headingSlice = slices.find((slice) => slice.text.includes("Title words")); + expect(headingSlice?.text).toContain("still the title"); + expect(headingSlice?.text.trimEnd().endsWith("---")).toBe(true); + expect(headingSlice?.text.includes("body")).toBe(false); + expect(slices.some((slice) => slice.text.trim() === "---")).toBe(false); + }); + + it("keeps a blockquote together when the message around it is split", () => { + const quote = ["> " + "alpha ".repeat(80).trim(), ">", "> " + "beta ".repeat(40).trim()].join( + "\n", + ); + const after = "gamma ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${quote}\n\n${after}`)!; + const quoteSlice = slices.find((slice) => slice.text.includes("> alpha")); + expect(quoteSlice?.text).toContain("> beta"); + expect(quoteSlice?.text.startsWith(">")).toBe(true); + expect(slices.some((slice) => slice.text.includes("gamma") && !slice.text.includes(">"))).toBe( + true, + ); + }); + + it("does not turn a wrapped ordered line into its own list", () => { + const paragraph = `${"word ".repeat(80).trim()}\n2. ${"cont ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(paragraph)!; + const owner = slices.find((slice) => slice.text.includes("2. ")); + expect(owner?.text.startsWith("2.")).toBe(false); + expect(owner?.text).toMatch(/\S\n2\. /); + }); + + it("keeps an inline link in one slice, including a long destination", () => { + const url = `https://example.com/${"a".repeat(200)}`; + const label = `docs ${"label ".repeat(10).trim()}`; + const link = `[${label}](${url})`; + const before = "before ".repeat(80).trim(); + const after = "after ".repeat(80).trim(); + const slices = splitAssistantTranscriptSlices(`${before} ${link} ${after}`)!; + expect(markdownTexts(slices).filter((text) => text.includes(link))).toHaveLength(1); + for (const text of markdownTexts(slices)) { + if (text.includes(link)) continue; + expect(text.includes(url)).toBe(false); + expect(text.includes(`](${url.slice(0, 24)}`)).toBe(false); + } + }); + + it("keeps emphasis and inline code intact when the paragraph is split", () => { + const bold = `**${"bold ".repeat(30).trim()}**`; + const code = `\`${"c".repeat(180)}\``; + const text = `${"word ".repeat(80).trim()} ${bold} ${code} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(text)!; + expect(markdownTexts(slices).filter((slice) => slice.includes(bold))).toHaveLength(1); + expect(markdownTexts(slices).filter((slice) => slice.includes(code))).toHaveLength(1); + for (const slice of markdownTexts(slices)) { + const markers = slice.match(/\*\*/g)?.length ?? 0; + expect(markers % 2).toBe(0); + } + }); + + it("keeps a raw HTML element in one slice", () => { + const html = `${"x".repeat(240)}`; + const text = `${"word ".repeat(80).trim()} ${html} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(text)!; + expect(markdownTexts(slices).filter((slice) => slice.includes(html))).toHaveLength(1); + }); + + it("copies link reference definitions onto the slice that uses them", () => { + const body = `${"word ".repeat(200).trim()}\n\nSee [the docs][ref].\n\n${"word ".repeat(200).trim()}`; + const definition = "[ref]: https://example.com/docs"; + const slices = splitAssistantTranscriptSlices(`${body}\n\n${definition}`)!; + const linkSlice = slices.find((slice) => slice.text.includes("[the docs][ref]")); + expect(linkSlice?.text).toContain(definition); + }); + + it("keeps an empty fenced block when the message is split", () => { + const markdown = `${codeFence("ts", 3)}\n\n\`\`\`ts\n\`\`\`\n\n${codeFence("go", 2)}`; + const slices = splitAssistantTranscriptSlices(markdown)!; + const empty = slices.find((slice) => slice.kind === "code" && slice.text === ""); + expect(empty).toMatchObject({ + kind: "code", + codePart: "only", + language: "ts", + text: "", + fullCode: "", + }); + expect(assistantSliceMarkdown(empty!)).toContain("```ts"); + }); + + it("strips the opening fence indent from code that is split out", () => { + const fence = " ```ts\n const value = 1;\n still indented\n ```"; + const slices = splitAssistantTranscriptSlices(`${fence}\n\n${"word ".repeat(200).trim()}`)!; + const code = slices.find((slice) => slice.kind === "code"); + expect(code).toMatchObject({ + kind: "code", + text: "const value = 1;\n still indented", + fullCode: "const value = 1;\n still indented", + }); + }); + + it("copies a reference definition whose destination is on the next line", () => { + const body = `${"word ".repeat(200).trim()}\n\nSee [the docs][docs].\n\n${"word ".repeat(200).trim()}`; + const definition = "[docs]:\n https://example.com/docs"; + const slices = splitAssistantTranscriptSlices(`${body}\n\n${definition}`)!; + const linkSlice = slices.find((slice) => slice.text.includes("[the docs][docs]")); + expect(linkSlice?.text).toContain("[docs]:"); + expect(linkSlice?.text).toContain("https://example.com/docs"); + }); + + it("copies a reference definition whose title is on the next line", () => { + const body = `${"word ".repeat(200).trim()}\n\nSee [the docs][docs].\n\n${"word ".repeat(200).trim()}`; + const definition = '[docs]: https://example.com/docs\n"API docs"'; + const slices = splitAssistantTranscriptSlices(`${body}\n\n${definition}`)!; + const linkSlice = slices.find((slice) => slice.text.includes("[the docs][docs]")); + expect(linkSlice?.text).toContain("https://example.com/docs"); + expect(linkSlice?.text).toContain('"API docs"'); + }); + + it("keeps nested elements of the same name in one slice", () => { + const html = `outer inner ${"still ".repeat(80).trim()}`; + const text = `${"word ".repeat(80).trim()} ${html} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(text)!; + const owners = markdownTexts(slices).filter((slice) => slice.includes("")); + expect(owners).toHaveLength(1); + expect(owners[0]).toContain(html); + }); + + it("keeps a textarea block intact through its closing tag", () => { + const block = ``; + const slices = splitAssistantTranscriptSlices(`${block}\n\n${"word ".repeat(200).trim()}`)!; + const owners = slices.filter((slice) => slice.text.toLowerCase().includes("textarea")); + expect(owners).toHaveLength(1); + expect(owners[0]!.text).toContain(block); + }); + + it("does not start a slice on a mid-line list, heading, or quote marker", () => { + const markdown = `${"word ".repeat(150)}- dash ${"word ".repeat(40)}# title ${"word ".repeat(40)}> quote ${"word ".repeat(80)}`; + const slices = splitAssistantTranscriptSlices(markdown)!; + expect(slices.length).toBeGreaterThan(1); + for (const slice of slices) { + expect(slice.text).not.toMatch(/^[-+*] |^#{1,6} |^>/); + } + }); + + it("keeps an artifact-template directive inside one slice", () => { + const directive = `::artifact-template{skill_name="artifact-template-hello-world" skill_directory="${"a".repeat(500)}" display_name="Hello World" artifact_kind="document"}`; + const markdown = `${"word ".repeat(80).trim()} ${directive} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(markdown)!; + const owners = slices.filter((slice) => slice.text.includes("::artifact-template")); + expect(owners).toHaveLength(1); + expect(owners[0]!.text).toContain(directive); + }); + + it("leaves a fence inside a list item in that item", () => { + const item = `1. run this\n\n \`\`\`ts\n const value = 1;\n \`\`\``; + const rest = "tail ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${item}\n\n${rest}`)!; + const owner = slices.find((slice) => slice.text.includes("run this")); + expect(owner?.kind).toBe("markdown"); + expect(owner?.text).toContain("```ts"); + expect(owner?.text).toContain("const value = 1;"); + expect(slices.some((slice) => slice.kind === "code")).toBe(false); + }); +}); + +describe("expandAndroidAssistantTranscriptRows", () => { + it("returns the same array off Android and for rows that are already small", () => { + const feed = [ + message("user-1", "user", "hello"), + { type: "thinking" as const, id: "thinking" }, + assistantMessage("short", `ok\n\n${codeFence("ts", 2)}`), + ]; + expect(expandAndroidAssistantTranscriptRows(feed, "ios")).toBe(feed); + expect(expandAndroidAssistantTranscriptRows(feed, "android")).toBe(feed); + }); + + it("does not slice a long user message", () => { + const feed = [message("user-1", "user", "word ".repeat(400).trim())]; + expect(expandAndroidAssistantTranscriptRows(feed, "android")).toBe(feed); + }); + + it("keeps the message id on the first slice and appends the rest", () => { + const fence = codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 2); + const feed = [ + message("user-1", "user", "ship it"), + { type: "work-toggle" as const, id: "work-1" }, + assistantMessage("assistant-1", fence), + ]; + const rows = expandAndroidAssistantTranscriptRows(feed, "android"); + expect(rows.map((row) => row.type)).toEqual([ + "message", + "work-toggle", + "assistant-slice", + "assistant-slice", + ]); + expect(rows[0]).toBe(feed[0]); + expect(rows[1]).toBe(feed[1]); + const head = rows[2]; + const tail = rows[3]; + if (head?.type !== "assistant-slice" || tail?.type !== "assistant-slice") { + throw new Error("expected assistant slices"); + } + expect(head.id).toBe("assistant-1"); + expect(head.isFirst).toBe(true); + expect(head.isLast).toBe(false); + expect(tail.id).toBe(`assistant-1:${tail.slice.key}`); + expect(tail.isLast).toBe(true); + expect(tail.source).toBe(feed[2]); + expect(androidTranscriptItemType(head)).toBe("assistant-code-head"); + expect(androidTranscriptItemType(tail)).toBe("assistant-code-body"); + expect(androidTranscriptItemType(feed[0]!)).toBeNull(); + }); + + it("keeps earlier slice ids when the settled message later grows at the end", () => { + const before = expandAndroidAssistantTranscriptRows( + [assistantMessage("assistant-1", codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 1))], + "android", + ); + const after = expandAndroidAssistantTranscriptRows( + [assistantMessage("assistant-1", codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 4))], + "android", + ); + expect(before.map((row) => row.id)).toEqual([after[0]!.id, after[1]!.id]); + expect(after).toHaveLength(2); + }); +}); + +describe("assistant slice presentation", () => { + it("closes a fence that contains backticks and leaves plain windows without markdown", () => { + const fenced = fencedCodeMarkdown("ts", "const tick = ```;"); + expect(fenced.startsWith("````")).toBe(true); + expect(fenced).toContain("const tick = ```;"); + expect(fenced.trimEnd().endsWith("````")).toBe(true); + }); + + it("does not gap code windows that belong to the same fence", () => { + const slices = splitAssistantTranscriptSlices( + codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET * 2 + 1), + )!; + expect(assistantSliceGap(slices[0]!, false)).toBe(0); + expect(assistantSliceGap(slices[1]!, false)).toBe(0); + expect(assistantSliceGap(slices[2]!, true)).toBe(0); + const prose = splitAssistantTranscriptSlices( + `${"word ".repeat(200).trim()}\n\n${"word ".repeat(200).trim()}`, + )!; + expect(assistantSliceGap(prose[0]!, false)).toBe(ANDROID_TRANSCRIPT_SLICE_GAP); + expect(assistantSliceGap(prose[0]!, true)).toBe(0); + }); +}); diff --git a/apps/mobile/src/features/threads/androidTranscriptSlices.ts b/apps/mobile/src/features/threads/androidTranscriptSlices.ts new file mode 100644 index 000000000000..0d43d2a968c2 --- /dev/null +++ b/apps/mobile/src/features/threads/androidTranscriptSlices.ts @@ -0,0 +1,1926 @@ +import { renderAssistantCitationsAsText } from "@t3tools/shared/assistantCitations"; + +/** + * Prose mounted as one Android text view. Larger selectable paragraphs are the + * stalls measured when a settled thread scrolls back into view. + */ +export const ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET = 720; + +/** + * Code lines mounted as one Android row. A longer fence is windowed so a fling + * never builds that fence's whole text tree in a single frame. + */ +export const ANDROID_TRANSCRIPT_CODE_LINE_BUDGET = 16; + +/** Gap between slices that are not a continuation of the same code fence. */ +export const ANDROID_TRANSCRIPT_SLICE_GAP = 14; + +export type AndroidTranscriptCodePart = "only" | "start" | "middle" | "end"; + +export type AndroidTranscriptSlice = + | { + readonly kind: "markdown"; + readonly key: string; + readonly text: string; + } + | { + readonly kind: "code"; + readonly key: string; + readonly text: string; + readonly language: string | null; + readonly codePart: AndroidTranscriptCodePart; + readonly fullCode: string; + }; + +interface TranscriptMessageEntry { + readonly type: "message"; + readonly id: string; + readonly createdAt: string; + readonly message: { + readonly role: string; + readonly text: string; + }; +} + +export interface AndroidAssistantSliceEntry { + readonly type: "assistant-slice"; + readonly id: string; + readonly createdAt: string; + readonly source: TSource; + readonly slice: AndroidTranscriptSlice; + readonly isFirst: boolean; + readonly isLast: boolean; +} + +interface SourceLine { + readonly text: string; + readonly start: number; + readonly end: number; +} + +interface FenceOpener { + readonly char: "`" | "~"; + readonly length: number; + readonly info: string; + /** Spaces before the opening fence. Content lines lose up to this many. */ + readonly indent: number; +} + +interface ListMarkerInfo { + readonly indent: number; + readonly ordered: boolean; + readonly number: number | null; +} + +type HtmlBlockKind = + | { readonly kind: "comment" } + | { readonly kind: "processing" } + | { readonly kind: "declaration" } + | { readonly kind: "pre"; readonly tag: string } + | { readonly kind: "block"; readonly tag: string }; + +interface MarkdownBlock { + readonly kind: "markdown"; + readonly start: number; + readonly end: number; + /** + * True only for a plain paragraph. Lists, quotes, tables, and HTML stay one + * piece because each slice is parsed as its own Markdown document. + */ + readonly inlineSplittable: boolean; +} + +interface CodeBlock { + readonly kind: "code"; + readonly start: number; + readonly body: string; + readonly language: string | null; + readonly lineCount: number; +} + +type TranscriptBlock = MarkdownBlock | CodeBlock; + +interface TextRange { + readonly start: number; + readonly end: number; +} + +interface EmphasisDelimiter { + readonly char: "*" | "_" | "~"; + readonly pos: number; + origLen: number; + len: number; + readonly canOpen: boolean; + readonly canClose: boolean; +} + +const HTML_BLOCK_TAGS = new Set([ + "address", + "article", + "aside", + "base", + "basefont", + "blockquote", + "body", + "caption", + "center", + "col", + "colgroup", + "dd", + "details", + "dialog", + "dir", + "div", + "dl", + "dt", + "fieldset", + "figcaption", + "figure", + "footer", + "form", + "frame", + "frameset", + "h1", + "h2", + "h3", + "h4", + "h5", + "h6", + "head", + "header", + "hr", + "html", + "iframe", + "legend", + "li", + "link", + "main", + "menu", + "menuitem", + "nav", + "noframes", + "ol", + "optgroup", + "option", + "p", + "param", + "search", + "section", + "summary", + "table", + "tbody", + "td", + "tfoot", + "th", + "thead", + "title", + "tr", + "track", + "ul", +]); + +/** + * Splits an expensive assistant message into bounded Android list rows. + * + * Each Markdown slice is a complete document: links, emphasis, inline code, + * lists, blockquotes, tables, and HTML are never cut in half. A construct + * longer than the budget stays one row. Short messages return null so they + * keep the highlighted renderer. Keys come from source offsets, so a streaming + * append does not renumber earlier slices. + */ +export function splitAssistantTranscriptSlices( + markdown: string, +): readonly AndroidTranscriptSlice[] | null { + if ( + markdown.length <= ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET && + !markdown.includes("```") && + !markdown.includes("~~~") + ) { + return null; + } + const normalized = markdown.replaceAll("\r\n", "\n"); + const blocks = parseTranscriptBlocks(normalized); + if (!shouldSplitAssistantTranscript(normalized, blocks)) { + return null; + } + const definitions = collectLinkReferenceDefinitions(normalized); + const slices = assembleTranscriptSlices(normalized, blocks, definitions); + return slices.length > 1 ? slices : null; +} + +/** + * Expands assistant messages that would mount an unbounded text tree. + * + * Non-Android feeds are returned unchanged, including the same array, so iOS + * keeps today's row identity. The first slice reuses the message id; later + * slices append, which is the direction live-follow already scrolls. + */ +export function expandAndroidAssistantTranscriptRows< + TEntry extends { readonly type: string; readonly id: string }, +>( + entries: readonly TEntry[], + platform: string, +): readonly (TEntry | AndroidAssistantSliceEntry>)[] { + if (platform !== "android") { + return entries; + } + + let changed = false; + const rows: (TEntry | AndroidAssistantSliceEntry>)[] = []; + for (const entry of entries) { + const slices = slicesForEntry(entry); + if (!slices) { + rows.push(entry); + continue; + } + changed = true; + const source = entry as Extract; + for (let index = 0; index < slices.length; index += 1) { + const slice = slices[index]; + if (!slice) continue; + rows.push({ + type: "assistant-slice", + id: index === 0 ? source.id : `${source.id}:${slice.key}`, + createdAt: source.createdAt, + source, + slice, + isFirst: index === 0, + isLast: index === slices.length - 1, + }); + } + } + + return changed ? rows : entries; +} + +/** + * LegendList item type for a slice. The list prefers a same-type container, so + * a plain code window is reused for another code window before a prose row. + */ +export function androidTranscriptItemType(entry: { + readonly type: string; + readonly slice?: AndroidTranscriptSlice; +}): string | null { + if (entry.type !== "assistant-slice" || !entry.slice) { + return null; + } + if (entry.slice.kind === "markdown") { + return "assistant-markdown-slice"; + } + if (entry.slice.codePart === "only") { + return "assistant-code-block"; + } + if (entry.slice.codePart === "start") { + return "assistant-code-head"; + } + return "assistant-code-body"; +} + +/** + * Markdown for slices the shared renderer can draw. Plain windows of a long + * fence return null; those are one `Text`, not a token per span. + */ +export function assistantSliceMarkdown(slice: AndroidTranscriptSlice): string | null { + if (slice.kind === "markdown") { + return slice.text; + } + if (slice.codePart !== "only") { + return null; + } + return fencedCodeMarkdown(slice.language, slice.text); +} + +/** + * Space after a slice row. Continued code windows share one card, so they do + * not take the gap that separate blocks use. The last slice uses the message + * row's own bottom margin instead. + */ +export function assistantSliceGap(slice: AndroidTranscriptSlice, isLast: boolean): number { + if (isLast) { + return 0; + } + if (slice.kind === "code" && (slice.codePart === "start" || slice.codePart === "middle")) { + return 0; + } + return ANDROID_TRANSCRIPT_SLICE_GAP; +} + +/** + * Wraps a short code window in a fence the markdown renderer already knows how + * to draw. The fence is longer than any backtick run in the body so the body + * cannot close it early. + */ +export function fencedCodeMarkdown(language: string | null, code: string): string { + let longestRun = 0; + let run = 0; + for (const character of code) { + if (character === "`") { + run += 1; + longestRun = Math.max(longestRun, run); + } else { + run = 0; + } + } + const fence = "`".repeat(Math.max(3, longestRun + 1)); + const info = language ? language.replace(/[\r\n`]/g, "") : ""; + return `${fence}${info}\n${code}\n${fence}`; +} + +const assistantSliceCache = new Map(); +const ASSISTANT_SLICE_CACHE_LIMIT = 200; + +/** + * Returns slice rows for an assistant message, or null when the entry should + * stay as it is. User, reasoning, and non-message rows are never split. + * Results are cached by the raw message text so a streaming tail does not + * re-scan every earlier message. + */ +function slicesForEntry(entry: { + readonly type: string; + readonly id: string; +}): readonly AndroidTranscriptSlice[] | null { + if (!isAssistantMessageEntry(entry)) { + return null; + } + const raw = entry.message.text; + const cached = assistantSliceCache.get(raw); + if (cached !== undefined) { + return cached; + } + const text = renderAssistantCitationsAsText(raw); + const slices = text.trim().length === 0 ? null : splitAssistantTranscriptSlices(text); + assistantSliceCache.delete(raw); + assistantSliceCache.set(raw, slices); + while (assistantSliceCache.size > ASSISTANT_SLICE_CACHE_LIMIT) { + const oldest = assistantSliceCache.keys().next().value; + if (oldest === undefined) { + break; + } + assistantSliceCache.delete(oldest); + } + return slices; +} + +/** + * Narrows a feed entry to an assistant message with the fields slicing reads. + */ +function isAssistantMessageEntry(entry: { + readonly type: string; + readonly id: string; +}): entry is TranscriptMessageEntry { + if ( + entry.type !== "message" || + !("createdAt" in entry) || + typeof entry.createdAt !== "string" || + !("message" in entry) + ) { + return false; + } + const message = entry.message; + if ( + typeof message !== "object" || + message === null || + !("role" in message) || + !("text" in message) + ) { + return false; + } + return message.role === "assistant" && typeof message.text === "string"; +} + +/** + * True when one list row would mount more text than a frame can afford. + * One short fence stays on the highlighted path; a second fence or a long + * fence is enough to split. + */ +function shouldSplitAssistantTranscript( + markdown: string, + blocks: readonly TranscriptBlock[], +): boolean { + if (markdown.length > ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + return true; + } + let codeBlocks = 0; + for (const block of blocks) { + if (block.kind !== "code") { + continue; + } + codeBlocks += 1; + if (block.lineCount > ANDROID_TRANSCRIPT_CODE_LINE_BUDGET) { + return true; + } + } + return codeBlocks >= 2; +} + +/** + * Walks top-level Markdown blocks. Fences become code windows. Lists, quotes, + * tables, and HTML stay intact. Only plain paragraphs may be cut later. + */ +function parseTranscriptBlocks(markdown: string): readonly TranscriptBlock[] { + const lines = sourceLines(markdown); + const lineTexts = lines.map((sourceLine) => sourceLine.text); + const blocks: TranscriptBlock[] = []; + let index = 0; + while (index < lines.length) { + const line = lines[index]; + if (!line || isBlankLine(line.text)) { + index += 1; + continue; + } + const fence = openingFence(line.text); + if (fence) { + const consumed = consumeFence(lines, index, fence); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + if (isAtxHeading(line.text) || isThematicBreak(line.text)) { + blocks.push(markdownSpan(line, line, false)); + index += 1; + continue; + } + if (isBlockquoteLine(line.text)) { + const consumed = consumeBlockquote(lines, index); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + if (listMarker(line.text)) { + const consumed = consumeList(lines, index); + blocks.push(...consumed.blocks); + index = consumed.next; + continue; + } + const definitionLines = parseLinkReferenceDefinition(lineTexts, index); + if (definitionLines !== null) { + const definitionEnd = index + definitionLines; + const definitionLast = lines[definitionEnd - 1] ?? line; + blocks.push(markdownSpan(line, definitionLast, false)); + index = definitionEnd; + continue; + } + const html = htmlBlockKind(line.text); + if (html) { + const consumed = consumeHtmlBlock(lines, index, html); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + const next = lines[index + 1]?.text ?? null; + if (next !== null && isTableStart(line.text, next)) { + const consumed = consumeTable(lines, index); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + if (isIndentedCodeLine(line.text)) { + const consumed = consumeIndentedCode(lines, index); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + const consumed = consumeParagraph(lines, index); + blocks.push(consumed.block); + index = consumed.next; + } + return blocks; +} + +/** + * Turns parsed blocks into row slices. Markdown on either side of a fence is + * packed separately so a fence cannot be swallowed by a prose range. + */ +function assembleTranscriptSlices( + markdown: string, + blocks: readonly TranscriptBlock[], + definitions: string, +): AndroidTranscriptSlice[] { + const slices: AndroidTranscriptSlice[] = []; + let markdownSpans: TextRange[] = []; + const flushMarkdown = () => { + if (markdownSpans.length === 0) return; + slices.push(...packMarkdownSpans(markdown, markdownSpans, definitions)); + markdownSpans = []; + }; + for (const block of blocks) { + if (block.kind === "code") { + flushMarkdown(); + slices.push(...sliceCodeBlock(block)); + continue; + } + markdownSpans.push(...expandMarkdownBlock(markdown, block)); + } + flushMarkdown(); + return slices; +} + +/** + * Breaks a plain paragraph on inline-safe whitespace. Every other block is + * one span, even when it is longer than the budget. + */ +function expandMarkdownBlock(markdown: string, block: MarkdownBlock): readonly TextRange[] { + if (!block.inlineSplittable) { + return [{ start: block.start, end: block.end }]; + } + const text = markdown.slice(block.start, block.end); + return splitPlainParagraph(text).map((range) => ({ + start: block.start + range.start, + end: block.start + range.end, + })); +} + +/** + * Packs neighboring Markdown spans up to the char budget. An oversized span + * is emitted alone so a long list item or quote is not joined to more text. + */ +function packMarkdownSpans( + markdown: string, + spans: readonly TextRange[], + definitions: string, +): AndroidTranscriptSlice[] { + const slices: AndroidTranscriptSlice[] = []; + let groupStart = -1; + let groupEnd = -1; + const flush = () => { + if (groupStart < 0) return; + pushMarkdownSlice(slices, markdown, groupStart, groupEnd, definitions); + groupStart = -1; + groupEnd = -1; + }; + for (const span of spans) { + const spanLength = emittedLength(markdown, span.start, span.end); + if (spanLength > ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + flush(); + pushMarkdownSlice(slices, markdown, span.start, span.end, definitions); + continue; + } + if (groupStart < 0) { + groupStart = span.start; + groupEnd = span.end; + continue; + } + const combined = emittedLength(markdown, groupStart, span.end); + if (combined > ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + flush(); + groupStart = span.start; + groupEnd = span.end; + continue; + } + groupEnd = span.end; + } + flush(); + return slices; +} + +/** + * Emits one Markdown slice, appending link reference definitions the slice + * does not already contain so `[text][id]` still resolves after a split. + */ +function pushMarkdownSlice( + slices: AndroidTranscriptSlice[], + markdown: string, + start: number, + end: number, + definitions: string, +): void { + const text = withLinkDefinitions(trimEdgeNewlines(markdown.slice(start, end)), definitions); + if (text.trim().length === 0) return; + slices.push({ + kind: "markdown", + key: `md:${start}`, + text, + }); +} + +/** + * Windows a fence into fixed line ranges. The first window stops changing once + * it fills, so scrolling back reuses that row instead of the rest of the file. + * An empty body is still one window so the fence header is not dropped. + */ +function sliceCodeBlock(block: CodeBlock): readonly AndroidTranscriptSlice[] { + const lines = block.body.split("\n"); + const windows: string[] = []; + for (let index = 0; index < lines.length; index += ANDROID_TRANSCRIPT_CODE_LINE_BUDGET) { + windows.push(lines.slice(index, index + ANDROID_TRANSCRIPT_CODE_LINE_BUDGET).join("\n")); + } + return windows.map((text, index) => ({ + kind: "code" as const, + key: `code:${block.start}:${index}`, + text, + language: block.language, + codePart: codeWindowPart(index, windows.length), + fullCode: block.body, + })); +} + +/** + * Names a code window so the first piece can show the header and the last + * piece can close the card. A fence that fits in one window stays "only". + */ +function codeWindowPart(index: number, count: number): AndroidTranscriptCodePart { + if (count <= 1) { + return "only"; + } + if (index === 0) { + return "start"; + } + if (index === count - 1) { + return "end"; + } + return "middle"; +} + +/** + * Splits one paragraph into ranges that each parse as their own paragraph. + * Points inside links, emphasis, inline code, or raw HTML are not used. When + * no safe point exists, the open construct stays whole. + */ +function splitPlainParagraph(text: string): readonly TextRange[] { + if (text.length <= ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + return [{ start: 0, end: text.length }]; + } + const points = inlineSafeBreaks(text); + const ranges: TextRange[] = []; + let cursor = 0; + while (cursor < text.length) { + if (text.length - cursor <= ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + ranges.push({ start: cursor, end: text.length }); + break; + } + let best = -1; + for (const point of points) { + if (point <= cursor) continue; + if (point > cursor + ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) break; + best = point; + } + const next = best === -1 ? nextSafeBreak(points, cursor, text.length) : best; + const end = next <= cursor || next > text.length ? text.length : next; + ranges.push({ start: cursor, end }); + if (end >= text.length) break; + cursor = end; + } + return ranges; +} + +/** + * First safe break after the budget, or the end of the paragraph when the + * remainder is one unbreakable construct. + */ +function nextSafeBreak(points: readonly number[], cursor: number, length: number): number { + for (const point of points) { + if (point > cursor + ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + return point; + } + } + return length; +} + +/** + * Whitespace indexes where the following text can start a new Markdown document + * without opening a block the original paragraph did not have. + */ +function inlineSafeBreaks(text: string): readonly number[] { + const blocked = protectedIntervals(text); + const points: number[] = []; + let index = 0; + while (index < text.length) { + if (!isInlineWhitespace(text[index] ?? "")) { + index += 1; + continue; + } + let next = index + 1; + while (next < text.length && isInlineWhitespace(text[next] ?? "")) { + next += 1; + } + if ( + next < text.length && + !insideProtected(blocked, next) && + !newlineWouldStartBlock(text, next) + ) { + points.push(next); + } + index = next; + } + return points; +} + +/** + * True when `index` sits strictly inside a link, code span, emphasis run, or tag. + * The edges themselves are safe: the construct stays entirely on one side. + */ +function insideProtected(intervals: readonly TextRange[], index: number): boolean { + for (const interval of intervals) { + if (interval.start >= index) break; + if (index > interval.start && index < interval.end) return true; + } + return false; +} + +/** + * True when a slice starting at `index` would turn a wrapped line into a list, + * quote, heading, fence, or other block. Those breaks stay with the line above. + */ +function newlineWouldStartBlock(text: string, index: number): boolean { + // A slice is its own document, so a break after a space is also a line start. + if (index === 0) return false; + const lineEnd = text.indexOf("\n", index); + const line = text.slice(index, lineEnd === -1 ? text.length : lineEnd); + if (isBlankLine(line)) return false; + if (openingFence(line) || isAtxHeading(line) || isThematicBreak(line)) return true; + if (isBlockquoteLine(line) || htmlBlockKind(line) !== null) return true; + if (listMarker(line) || isIndentedCodeLine(line)) return true; + if (!/^( {0,3})\[[^\]\n]+\]:/.test(line)) return false; + return parseLinkReferenceDefinition(text.slice(index).split("\n"), 0) !== null; +} + +/** + * Regions that must stay inside one slice. Later scanners skip earlier regions + * so a bracket inside inline code is not treated as a link. + */ +function protectedIntervals(text: string): readonly TextRange[] { + const intervals: TextRange[] = []; + collectCodeSpans(text, intervals); + collectInlineLinkDefinitions(text, intervals); + collectArtifactTemplates(text, intervals); + collectLinks(text, intervals); + collectHtmlAndAutolinks(text, intervals); + collectEmphasis(text, intervals); + intervals.sort((left, right) => left.start - right.start || left.end - right.end); + return intervals; +} + +/** + * Keeps a `::artifact-template{...}` directive in one slice. Paragraph packing + * must not cut its attributes, or the card fails to parse. + */ +function collectArtifactTemplates(text: string, intervals: TextRange[]): void { + const marker = "::artifact-template"; + let searchFrom = 0; + while (searchFrom < text.length) { + const at = text.indexOf(marker, searchFrom); + if (at === -1) return; + const previous = at === 0 ? "" : (text[at - 1] ?? ""); + if (at > 0 && !isInlineWhitespace(previous)) { + searchFrom = at + marker.length; + continue; + } + const skip = coveringEnd(intervals, at); + if (skip !== -1) { + searchFrom = skip; + continue; + } + const end = endOfArtifactTemplate(text, at + marker.length); + if (end === -1) { + searchFrom = at + marker.length; + continue; + } + intervals.push({ start: at, end }); + searchFrom = end; + } +} + +/** Offset just after the directive's closing `}`, or -1 when the braces never close. */ +function endOfArtifactTemplate(text: string, afterName: number): number { + let index = afterName; + while (index < text.length && (text[index] === " " || text[index] === "\t")) index += 1; + if (text[index] !== "{") return -1; + index += 1; + let quote: '"' | "'" | null = null; + while (index < text.length) { + const character = text[index] ?? ""; + if (quote) { + if (character === "\\") { + index += 2; + continue; + } + if (character === quote) quote = null; + index += 1; + continue; + } + if (character === '"' || character === "'") { + quote = character; + index += 1; + continue; + } + if (character === "}") return index + 1; + if (character === "\n" && text[index + 1] === "\n") return -1; + index += 1; + } + return -1; +} + +/** + * Protects a link reference definition that shares a paragraph with other text, + * including a destination or title that begins on the next line. + */ +function collectInlineLinkDefinitions(text: string, intervals: TextRange[]): void { + const lines = text.split("\n"); + const starts: number[] = []; + let offset = 0; + for (const line of lines) { + starts.push(offset); + offset += line.length + 1; + } + for (let index = 0; index < lines.length; index += 1) { + const start = starts[index] ?? 0; + if (coveringEnd(intervals, start) !== -1) continue; + const count = parseLinkReferenceDefinition(lines, index); + if (count === null) continue; + const last = index + count - 1; + const end = (starts[last] ?? start) + (lines[last]?.length ?? 0); + if (end > start) intervals.push({ start, end }); + index += count - 1; + } +} + +/** + * Records CommonMark code spans. An unclosed span protects the rest of the + * paragraph so the next slice cannot start between the backticks. + */ +function collectCodeSpans(text: string, intervals: TextRange[]): void { + let index = 0; + while (index < text.length) { + if (text[index] !== "`") { + index += 1; + continue; + } + let length = 0; + while (text[index + length] === "`") length += 1; + let cursor = index + length; + let closed = false; + while (cursor < text.length) { + if (text[cursor] !== "`") { + cursor += 1; + continue; + } + let run = 0; + while (text[cursor + run] === "`") run += 1; + if (run === length) { + intervals.push({ start: index, end: cursor + run }); + index = cursor + run; + closed = true; + break; + } + cursor += run; + } + if (!closed) { + intervals.push({ start: index, end: text.length }); + return; + } + } +} + +/** + * Records inline links and images, including the destination through the + * closing `)`. An unclosed `](` protects the tail so the URL is not cut. + */ +function collectLinks(text: string, intervals: TextRange[]): void { + let index = 0; + while (index < text.length) { + const skip = coveringEnd(intervals, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + const image = text[index] === "!" && text[index + 1] === "["; + if (image || text[index] === "[") { + const open = image ? index + 1 : index; + const end = endOfLink(text, open, intervals); + if (end > open + 1) { + intervals.push({ start: image ? index : open, end }); + index = end; + continue; + } + } + index += 1; + } +} + +/** + * End offset of the link that opens at `openIndex`, or `openIndex` when the + * brackets never close. Reference and shortcut links stop at their last `]`. + */ +function endOfLink(text: string, openIndex: number, blocked: readonly TextRange[]): number { + let depth = 1; + let index = openIndex + 1; + while (index < text.length) { + const skip = coveringEnd(blocked, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + if (text[index] === "[") depth += 1; + else if (text[index] === "]") { + depth -= 1; + if (depth === 0) break; + } + index += 1; + } + if (index >= text.length || text[index] !== "]") return openIndex; + const after = index + 1; + if (text[after] === "(") { + const destinationEnd = endOfLinkDestination(text, after); + return destinationEnd === -1 ? text.length : destinationEnd; + } + if (text[after] === "[") { + let cursor = after + 1; + while (cursor < text.length && text[cursor] !== "]" && text[cursor] !== "\n") { + cursor += 1; + } + return cursor >= text.length || text[cursor] !== "]" ? text.length : cursor + 1; + } + return after; +} + +/** + * Offset just after the `)` that closes a link destination, or -1 when the + * parenthesis never closes. Quoted titles may contain parentheses. + */ +function endOfLinkDestination(text: string, parenIndex: number): number { + let depth = 1; + let index = parenIndex + 1; + let quote: '"' | "'" | null = null; + while (index < text.length && depth > 0) { + const character = text[index]; + if (quote) { + if (character === "\\") { + index += 2; + continue; + } + if (character === quote) quote = null; + index += 1; + continue; + } + if (character === "\\") { + index += 2; + continue; + } + if ((character === '"' || character === "'") && depth === 1) { + quote = character; + index += 1; + continue; + } + if (character === "(") depth += 1; + else if (character === ")") depth -= 1; + index += 1; + } + return depth === 0 ? index : -1; +} + +/** + * Offset just after the matching close tag, or -1 when the element never closes. + * Nested elements of the same name are counted so `a b c` + * stays one span instead of ending at the inner ``. + */ +function endOfBalancedHtmlElement( + text: string, + openTagEnd: number, + name: string, + blocked: readonly TextRange[], +): number { + let depth = 1; + let cursor = openTagEnd; + while (cursor < text.length && depth > 0) { + const skip = coveringEnd(blocked, cursor); + if (skip !== -1) { + cursor = skip; + continue; + } + if (text[cursor] !== "<") { + cursor += 1; + continue; + } + const tag = /^<\/?([A-Za-z][A-Za-z0-9-]*)\b[^>\n]*\/?>/.exec(text.slice(cursor)); + if (!tag) { + cursor += 1; + continue; + } + const tagName = (tag[1] ?? "").toLowerCase(); + const isClose = text.startsWith(""); + cursor += tag[0].length; + if (tagName !== name) continue; + if (isClose) { + depth -= 1; + if (depth === 0) return cursor; + continue; + } + if (!isSelfClosing) depth += 1; + } + return -1; +} + +/** + * Records autolinks and raw HTML tags. A paired element stays together so + * `` is not left in a different slice from ``. + */ +function collectHtmlAndAutolinks(text: string, intervals: TextRange[]): void { + let index = 0; + while (index < text.length) { + const skip = coveringEnd(intervals, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + if (text[index] !== "<") { + index += 1; + continue; + } + const rest = text.slice(index); + const autolink = + /^<[A-Za-z][A-Za-z0-9+.-]*:[^>\s]+>/.exec(rest) ?? /^<[^>\s]+@[^>\s]+>/.exec(rest); + if (autolink) { + intervals.push({ start: index, end: index + autolink[0].length }); + index += autolink[0].length; + continue; + } + const tag = /^<\/?([A-Za-z][A-Za-z0-9-]*)\b[^>\n]*\/?>/.exec(rest); + if (!tag) { + index += 1; + continue; + } + const name = (tag[1] ?? "").toLowerCase(); + const closingOrEmpty = rest.startsWith(""); + if (closingOrEmpty) { + intervals.push({ start: index, end: index + tag[0].length }); + index += tag[0].length; + continue; + } + const closeEnd = endOfBalancedHtmlElement(text, index + tag[0].length, name, intervals); + const end = closeEnd === -1 ? index + tag[0].length : closeEnd; + intervals.push({ start: index, end }); + index = end; + } +} + +/** + * Records matched emphasis and strikethrough. Unmatched `*` or `_` stays + * literal and does not block a split. + */ +function collectEmphasis(text: string, intervals: TextRange[]): void { + const delimiters: EmphasisDelimiter[] = []; + let index = 0; + while (index < text.length) { + const skip = coveringEnd(intervals, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + const character = text[index]; + if (character !== "*" && character !== "_" && character !== "~") { + index += 1; + continue; + } + let length = 0; + while (text[index + length] === character) length += 1; + if (character === "~" && length < 2) { + index += length; + continue; + } + const flanking = delimiterFlanking(character, text[index - 1], text[index + length]); + if (flanking.canOpen || flanking.canClose) { + delimiters.push({ + char: character, + pos: index, + origLen: length, + len: length, + canOpen: flanking.canOpen, + canClose: flanking.canClose, + }); + } + index += length; + } + + const stack: EmphasisDelimiter[] = []; + for (const delimiter of delimiters) { + if (delimiter.canClose) { + let stackIndex = stack.length - 1; + while (stackIndex >= 0 && delimiter.len > 0) { + const opener = stack[stackIndex]; + if ( + !opener || + opener.char !== delimiter.char || + !opener.canOpen || + opener.len === 0 || + !emphasisCanMatch(opener, delimiter) + ) { + stackIndex -= 1; + continue; + } + const use = delimiter.char === "~" ? 2 : Math.min(opener.len, delimiter.len); + if (opener.len < use || delimiter.len < use) { + stackIndex -= 1; + continue; + } + const openStart = opener.pos + opener.len - use; + const closeEnd = delimiter.pos + (delimiter.origLen - delimiter.len) + use; + intervals.push({ start: openStart, end: closeEnd }); + opener.len -= use; + delimiter.len -= use; + stack.splice(stackIndex + 1); + if (opener.len === 0) stack.splice(stackIndex, 1); + else stackIndex -= 1; + } + } + if (delimiter.len > 0 && delimiter.canOpen) stack.push(delimiter); + } +} + +/** + * CommonMark flanking rules. Underscores inside words are not emphasis. + * `before` and `after` are the characters just outside the delimiter run. + */ +function delimiterFlanking( + character: "*" | "_" | "~", + before: string | undefined, + after: string | undefined, +): { readonly canOpen: boolean; readonly canClose: boolean } { + const beforeSpace = isUnicodeSpace(before); + const afterSpace = isUnicodeSpace(after); + const beforePunctuation = !beforeSpace && isAsciiPunctuation(before); + const afterPunctuation = !afterSpace && isAsciiPunctuation(after); + const left = !afterSpace && (!afterPunctuation || beforeSpace || beforePunctuation); + const right = !beforeSpace && (!beforePunctuation || afterSpace || afterPunctuation); + if (character === "_") { + return { + canOpen: left && (!right || beforePunctuation), + canClose: right && (!left || afterPunctuation), + }; + } + return { canOpen: left, canClose: right }; +} + +/** + * Applies CommonMark's multiple-of-three rule so `***` is not paired with a + * delimiter that would leave an odd unmatched run. + */ +function emphasisCanMatch(opener: EmphasisDelimiter, closer: EmphasisDelimiter): boolean { + if (opener.char === "~") return opener.origLen >= 2 && closer.origLen >= 2; + if (!(opener.canOpen && opener.canClose && closer.canOpen && closer.canClose)) { + return true; + } + const sum = opener.origLen + closer.origLen; + return !(sum % 3 === 0 && opener.origLen % 3 !== 0 && closer.origLen % 3 !== 0); +} + +/** + * End of the protected region containing `index`, or -1 when `index` is free. + */ +function coveringEnd(intervals: readonly TextRange[], index: number): number { + let end = -1; + for (const interval of intervals) { + if (index >= interval.start && index < interval.end && interval.end > end) { + end = interval.end; + } + } + return end; +} + +/** + * Lines of `markdown` with offsets. The trailing newline belongs to the line + * so a later slice can include the break that separated two blocks. + */ +function sourceLines(markdown: string): SourceLine[] { + const lines: SourceLine[] = []; + let start = 0; + while (start < markdown.length) { + const newline = markdown.indexOf("\n", start); + if (newline === -1) { + lines.push({ text: markdown.slice(start), start, end: markdown.length }); + break; + } + lines.push({ text: markdown.slice(start, newline), start, end: newline + 1 }); + start = newline + 1; + } + return lines; +} + +/** + * One Markdown span covering `from` through `to`, inclusive of those lines. + */ +function markdownSpan(from: SourceLine, to: SourceLine, inlineSplittable: boolean): MarkdownBlock { + return { + kind: "markdown", + start: from.start, + end: to.end, + inlineSplittable, + }; +} + +/** + * Reads a fence through its closer, or through the end while a reply is still + * streaming. The body excludes the fence markers. + */ +function consumeFence( + lines: readonly SourceLine[], + start: number, + opener: FenceOpener, +): { readonly block: CodeBlock; readonly next: number } { + const body: string[] = []; + let index = start + 1; + while (index < lines.length && !isClosingFence(lines[index]?.text ?? "", opener)) { + body.push(stripFenceIndent(lines[index]?.text ?? "", opener.indent)); + index += 1; + } + if (index < lines.length) index += 1; + return { + block: { + kind: "code", + start: lines[start]?.start ?? 0, + body: body.join("\n"), + language: fenceLanguage(opener), + lineCount: body.length, + }, + next: index, + }; +} + +/** + * Reads a blockquote through its last `>` line, including lazy continuation + * that would still belong to the quote in one document. + */ +function consumeBlockquote( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isBlockquoteLine(line)) { + index += 1; + continue; + } + if (isBlankLine(line)) { + const next = nextNonBlank(lines, index + 1); + if (next !== null && isBlockquoteLine(lines[next]?.text ?? "")) { + index += 1; + continue; + } + break; + } + const following = lines[index + 1]?.text ?? null; + if (interruptsParagraph(line, following)) break; + index += 1; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Reads one top-level list as a sequence of items. Items stay whole, including + * nested lists, indented continuation, and lazy lines, and may be packed later. + */ +function consumeList( + lines: readonly SourceLine[], + start: number, +): { readonly blocks: readonly MarkdownBlock[]; readonly next: number } { + const base = listMarker(lines[start]?.text ?? ""); + if (!base) return { blocks: [], next: start + 1 }; + const blocks: MarkdownBlock[] = []; + let index = start; + while (index < lines.length) { + while (index < lines.length && isBlankLine(lines[index]?.text ?? "")) { + const next = nextNonBlank(lines, index + 1); + const nextMarker = next === null ? null : listMarker(lines[next]?.text ?? ""); + if (nextMarker && nextMarker.indent === base.indent) { + index += 1; + continue; + } + return { blocks, next: index }; + } + const marker = listMarker(lines[index]?.text ?? ""); + if (!marker || marker.indent !== base.indent) break; + const itemStart = index; + index += 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isBlankLine(line)) { + const next = nextNonBlank(lines, index + 1); + if (next !== null && leadingIndent(lines[next]?.text ?? "") > base.indent) { + index += 1; + continue; + } + break; + } + if (leadingIndent(line) > base.indent) { + index += 1; + continue; + } + // Same-indent markers start the next item. Other lines that would not + // interrupt a paragraph are lazy continuation and stay with this item. + const sibling = listMarker(line); + if (sibling && sibling.indent <= base.indent) break; + const following = lines[index + 1]?.text ?? null; + if (interruptsParagraph(line, following)) break; + index += 1; + } + const last = lines[index - 1] ?? lines[itemStart]; + const first = lines[itemStart] ?? last; + if (first && last) blocks.push(markdownSpan(first, last, false)); + } + return { blocks, next: index }; +} + +/** + * Drops up to `indent` leading spaces. CommonMark removes the opening fence's + * indentation from each content line, and leaves a shorter indent empty. + */ +function stripFenceIndent(line: string, indent: number): string { + if (indent <= 0) return line; + let index = 0; + while (index < line.length && index < indent && line[index] === " ") { + index += 1; + } + return line.slice(index); +} + +/** + * Reads one HTML block. Textarea, pre, script, and style run through their + * closing tag, including blank lines inside. Other block tags stop at the + * closer or at the next blank line. + */ +function consumeHtmlBlock( + lines: readonly SourceLine[], + start: number, + kind: HtmlBlockKind, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start; + if (kind.kind === "comment") { + index = consumeUntilIncludes(lines, start, "-->"); + } else if (kind.kind === "processing") { + index = consumeUntilIncludes(lines, start, "?>"); + } else if (kind.kind === "declaration") { + index = consumeUntilIncludes(lines, start, ">"); + } else if (kind.kind === "pre") { + index = consumeUntilIncludes(lines, start, ``); + } else { + const close = ``; + if ((lines[start]?.text ?? "").toLowerCase().includes(close)) { + index = start + 1; + } else { + index = start + 1; + while (index < lines.length && !isBlankLine(lines[index]?.text ?? "")) { + const includes = (lines[index]?.text ?? "").toLowerCase().includes(close); + index += 1; + if (includes) break; + } + } + } + const last = lines[Math.max(start, index - 1)] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Advances past the line that contains `needle`, or to the end of the input. + */ +function consumeUntilIncludes(lines: readonly SourceLine[], start: number, needle: string): number { + let index = start; + while (index < lines.length) { + const includes = (lines[index]?.text ?? "").toLowerCase().includes(needle.toLowerCase()); + index += 1; + if (includes) break; + } + return index; +} + +/** + * Reads a GFM table from its header through the last pipe row. The delimiter + * row stays with the header so the table still parses. + */ +function consumeTable( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 2; + while ( + index < lines.length && + !isBlankLine(lines[index]?.text ?? "") && + (lines[index]?.text ?? "").includes("|") + ) { + index += 1; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Reads an indented code block. It stays Markdown, not a fenced window, so the + * shared renderer can still show it as code. + */ +function consumeIndentedCode( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isIndentedCodeLine(line)) { + index += 1; + continue; + } + if (isBlankLine(line)) { + const next = nextNonBlank(lines, index + 1); + if (next !== null && isIndentedCodeLine(lines[next]?.text ?? "")) { + index += 1; + continue; + } + } + break; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Reads a paragraph. A setext underline is kept with its text line. A following + * block that would interrupt the paragraph starts the next span instead. + */ +function consumeParagraph( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isBlankLine(line)) break; + // Setext wins over a thematic break. The underline stays on the heading + // so a later slice cannot turn `---` into a horizontal rule. + if (isSetextUnderline(line)) { + index += 1; + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; + } + const following = lines[index + 1]?.text ?? null; + if (interruptsParagraph(line, following)) break; + index += 1; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, true), + next: index, + }; +} + +/** + * True when `line` starts a new block in the middle of a paragraph. Ordered + * lists other than `1` do not interrupt, matching CommonMark. + */ +function interruptsParagraph(line: string, nextLine: string | null): boolean { + if (openingFence(line) || isAtxHeading(line) || isThematicBreak(line)) return true; + if (isBlockquoteLine(line) || htmlBlockKind(line) !== null) return true; + if (nextLine !== null && isTableStart(line, nextLine)) return true; + const marker = listMarker(line); + if (!marker) return false; + return !marker.ordered || marker.number === 1; +} + +/** + * Link reference definitions in the message. Copies are appended to slices that + * do not already hold them, so a reference link stays clickable after a split. + * Lines inside fences are skipped. + */ +function collectLinkReferenceDefinitions(markdown: string): string { + const lines = sourceLines(markdown); + const texts = lines.map((line) => line.text); + const definitions: string[] = []; + let fence: FenceOpener | null = null; + for (let index = 0; index < texts.length; index += 1) { + const line = texts[index] ?? ""; + if (fence) { + if (isClosingFence(line, fence)) fence = null; + continue; + } + const opener = openingFence(line); + if (opener) { + fence = opener; + continue; + } + const count = parseLinkReferenceDefinition(texts, index); + if (count === null) continue; + definitions.push(texts.slice(index, index + count).join("\n")); + index += count - 1; + } + return definitions.join("\n"); +} + +/** + * Appends `definitions` when the slice does not already contain them. + * Definitions are not rendered, so the extra lines do not show up in the row. + */ +function withLinkDefinitions(text: string, definitions: string): string { + if (definitions.length === 0 || text.includes(definitions)) return text; + return `${text}\n\n${definitions}`; +} + +/** + * Opening fence (`\`\`\`` or `~~~`) with at most three spaces of indent. + * A backtick fence whose info string contains a backtick is prose, not a fence. + */ +function openingFence(line: string): FenceOpener | null { + const match = /^( {0,3})(`{3,}|~{3,})(.*)$/.exec(line); + if (!match) return null; + const marker = match[2] ?? ""; + const info = match[3] ?? ""; + const character = marker[0]; + if (character !== "`" && character !== "~") return null; + if (character === "`" && info.includes("`")) return null; + return { char: character, length: marker.length, info, indent: (match[1] ?? "").length }; +} + +/** + * True when `line` closes `opener`. The closing line is only the marker, and + * it must be at least as long as the opener of the same character. + */ +function isClosingFence(line: string, opener: FenceOpener): boolean { + const match = /^( {0,3})(`{3,}|~{3,})[ \t]*$/.exec(line); + const marker = match?.[2] ?? ""; + return marker.length >= opener.length && marker[0] === opener.char; +} + +/** + * Language info word from an opening fence. Empty info stays null so the + * header can fall back to a generic code label. + */ +function fenceLanguage(opener: FenceOpener): string | null { + const info = opener.info.trim(); + if (info.length === 0) return null; + const word = info.split(/[ \t]+/)[0] ?? ""; + if (word.length === 0 || word.includes(opener.char)) return null; + return word; +} + +/** + * Top-level list marker, or null for thematic breaks and ordinary prose. + * The marker's indent is how continuation lines are recognized. + */ +function listMarker(line: string): ListMarkerInfo | null { + if (isThematicBreak(line)) return null; + const match = /^( {0,3})([-+*]|\d{1,9}[.)])(?:[ \t]+(.*)|[ \t]*)$/.exec(line); + if (!match) return null; + const marker = match[2] ?? ""; + const ordered = /^\d/.test(marker); + return { + indent: (match[1] ?? "").length, + ordered, + number: ordered ? Number.parseInt(marker, 10) : null, + }; +} + +/** + * Count of leading spaces, with a tab counted as four. Blank lines are not + * continuation; callers handle those separately. + */ +function leadingIndent(line: string): number { + let count = 0; + for (const character of line) { + if (character === " ") count += 1; + else if (character === "\t") count += 4; + else break; + } + return count; +} + +/** + * True for a GFM header row followed by a delimiter row. + */ +function isTableStart(line: string, nextLine: string): boolean { + return line.includes("|") && isTableDelimiter(nextLine); +} + +/** + * True for a GFM delimiter row of one or more dashed cells. + */ +function isTableDelimiter(line: string): boolean { + const trimmed = line.trim(); + if (!trimmed.includes("-")) return false; + const cells = trimmed.replace(/^\|/, "").replace(/\|$/, "").split("|"); + return cells.length > 0 && cells.every((cell) => /^\s*:?-{3,}:?\s*$/.test(cell)); +} + +/** + * True for an ATX heading. A `#` glued to the next word is not a heading. + */ +function isAtxHeading(line: string): boolean { + return /^( {0,3})#{1,6}(?:[ \t]+.*|[ \t]*)$/.test(line); +} + +/** + * True for a setext underline. Used on the lines after heading text, where it + * is a heading rather than a thematic break. + */ +function isSetextUnderline(line: string): boolean { + return /^( {0,3})(?:=+|-+)[ \t]*$/.test(line); +} + +/** + * True for a thematic break of three or more `-`, `*`, or `_`. + */ +function isThematicBreak(line: string): boolean { + return /^( {0,3})([-*_])(?:\s*\2){2,}\s*$/.test(line); +} + +/** + * True when `line` opens or continues a blockquote. + */ +function isBlockquoteLine(line: string): boolean { + return /^( {0,3})>/.test(line); +} + +/** + * True for a four-space indented code line. Fence detection allows only three. + */ +function isIndentedCodeLine(line: string): boolean { + return /^(?: {4}|\t)\S/.test(line); +} + +/** + * Lines consumed by a link reference definition starting at `start`, or null. + * The destination and title may each sit on the following line (`[docs]:` / + * ` https://example.com`). A line that only looks like a label stays prose. + */ +function parseLinkReferenceDefinition(lines: readonly string[], start: number): number | null { + if (start < 0 || start >= lines.length) return null; + + let lineIndex = start; + let column = 0; + const peek = (): string => { + const line = lines[lineIndex] ?? ""; + if (column < line.length) return line[column] ?? ""; + return lineIndex + 1 < lines.length ? "\n" : ""; + }; + const advance = (): string => { + const line = lines[lineIndex] ?? ""; + if (column < line.length) { + const character = line[column] ?? ""; + column += 1; + return character; + } + if (lineIndex + 1 < lines.length) { + lineIndex += 1; + column = 0; + return "\n"; + } + return ""; + }; + const skipSpaces = () => { + while (peek() === " " || peek() === "\t") advance(); + }; + + let indent = 0; + while (peek() === " " && indent < 3) { + advance(); + indent += 1; + } + if (peek() !== "[") return null; + advance(); + + let labelLength = 0; + while (labelLength <= 999) { + const character = peek(); + if (character === "" || character === "\n" || character === "[") return null; + if (character === "\\") { + advance(); + if (peek() === "" || peek() === "\n") return null; + advance(); + labelLength += 1; + continue; + } + if (character === "]") break; + advance(); + labelLength += 1; + } + if (labelLength === 0 || labelLength > 999 || peek() !== "]") return null; + advance(); + if (peek() !== ":") return null; + advance(); + + skipSpaces(); + if (peek() === "\n") { + advance(); + skipSpaces(); + } + if (peek() === "" || peek() === "\n") return null; + if (!readLinkDestination(peek, advance)) return null; + + skipSpaces(); + if (peek() === "\n") { + const nextLine = (lines[lineIndex + 1] ?? "").trimStart(); + const opener = nextLine[0]; + if (opener === '"' || opener === "'" || opener === "(") { + advance(); + skipSpaces(); + } + } + const titleOpener = peek(); + if (titleOpener === '"' || titleOpener === "'" || titleOpener === "(") { + if (!readLinkTitle(titleOpener, peek, advance, lines, () => lineIndex)) return null; + skipSpaces(); + } + if (peek() !== "" && peek() !== "\n") return null; + return lineIndex - start + 1; +} + +/** Reads an angle or bare link destination. The cursor sits on its first character. */ +function readLinkDestination(peek: () => string, advance: () => string): boolean { + if (peek() === "<") { + advance(); + while (true) { + const character = peek(); + if (character === "" || character === "\n" || character === " " || character === "<") { + return false; + } + if (character === "\\") { + advance(); + if (peek() === "" || peek() === "\n") return false; + advance(); + continue; + } + if (character === ">") { + advance(); + return true; + } + advance(); + } + } + + let length = 0; + let depth = 0; + while (true) { + const character = peek(); + if (character === "" || character === "\n" || character === " " || character === "\t") break; + if (character.charCodeAt(0) < 32) return false; + if (character === "\\") { + advance(); + const escaped = peek(); + if (escaped === "" || escaped === "\n" || escaped === " " || escaped === "\t") return false; + advance(); + length += 1; + continue; + } + if (character === "(") { + depth += 1; + advance(); + length += 1; + continue; + } + if (character === ")") { + if (depth === 0) break; + depth -= 1; + advance(); + length += 1; + continue; + } + advance(); + length += 1; + } + return length > 0 && depth === 0; +} + +/** Reads a quoted title that may continue onto later non-blank lines. */ +function readLinkTitle( + opener: string, + peek: () => string, + advance: () => string, + lines: readonly string[], + lineIndex: () => number, +): boolean { + const closer = opener === "(" ? ")" : opener; + advance(); + while (true) { + const character = peek(); + if (character === "") return false; + if (character === "\n") { + const next = lines[lineIndex() + 1]; + if (next === undefined || next.trim().length === 0) return false; + advance(); + continue; + } + if (character === "\\") { + advance(); + if (peek() === "" || peek() === "\n") return false; + advance(); + continue; + } + if (character === closer) { + advance(); + return true; + } + advance(); + } +} + +/** + * Classifies a CommonMark HTML block opener, or null for inline tags. + */ +function htmlBlockKind(line: string): HtmlBlockKind | null { + if (!/^( {0,3})?@[\\\]^_`{|}~]/.test(character); +} + +/** + * Length of the slice text after edge newlines are removed. That is the text + * the row actually mounts. + */ +function emittedLength(markdown: string, start: number, end: number): number { + return trimEdgeNewlines(markdown.slice(start, end)).length; +} + +/** + * Removes blank lines from the edges of a slice without stripping the indent + * a list item or indented code block needs. + */ +function trimEdgeNewlines(text: string): string { + let start = 0; + let end = text.length; + while (start < end && text[start] === "\n") start += 1; + while (end > start && text[end - 1] === "\n") end -= 1; + return text.slice(start, end); +}