From b0c9bd81113e6f329a02dd093cb16aa23a0794c1 Mon Sep 17 00:00:00 2001 From: macodev00 <273427913+macodev00@users.noreply.github.com> Date: Mon, 28 Sep 2026 09:49:38 +0000 Subject: [PATCH] fix(mobile): smooth Android scrolling through long settled threads Long settled assistant messages mounted as one selectable text tree, so scrolling them back into view stalled the Android UI thread. On Android, split those messages only at markdown-safe boundaries, window long code fences, and recycle list rows. Empty fences, multiline link definitions, nested same-name HTML, indented fence bodies, and textarea blocks stay intact. Recycled code rows key off the full fence so scroll position and copy state do not leak across blocks. Fixes #13925 --- .../modules/t3-markdown-text/package.json | 3 +- .../src/NativeMarkdownBlock.tsx | 4 + .../src/stableFenceKey.test.ts | 36 + .../t3-markdown-text/src/stableFenceKey.ts | 45 + .../threads/AndroidTranscriptCodeSlice.tsx | 160 ++ .../src/features/threads/ThreadFeed.tsx | 279 ++- .../threads/androidTranscriptSlices.test.ts | 463 ++++ .../threads/androidTranscriptSlices.ts | 1926 +++++++++++++++++ 8 files changed, 2892 insertions(+), 24 deletions(-) create mode 100644 apps/mobile/modules/t3-markdown-text/src/stableFenceKey.test.ts create mode 100644 apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts create mode 100644 apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx create mode 100644 apps/mobile/src/features/threads/androidTranscriptSlices.test.ts create mode 100644 apps/mobile/src/features/threads/androidTranscriptSlices.ts diff --git a/apps/mobile/modules/t3-markdown-text/package.json b/apps/mobile/modules/t3-markdown-text/package.json index 376befbeb152..ce2991cd592e 100644 --- a/apps/mobile/modules/t3-markdown-text/package.json +++ b/apps/mobile/modules/t3-markdown-text/package.json @@ -26,7 +26,8 @@ "./markdown": "./src/nativeMarkdownText.ts", "./primitive": "./src/MarkdownTextPrimitive.tsx", "./renderer": "./src/SelectableMarkdownText.tsx", - "./types": "./src/SelectableMarkdownText.types.ts" + "./types": "./src/SelectableMarkdownText.types.ts", + "./fence-key": "./src/stableFenceKey.ts" }, "peerDependencies": { "@t3tools/client-runtime": "*", diff --git a/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx b/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx index b57a182dbc56..800acb473707 100644 --- a/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx +++ b/apps/mobile/modules/t3-markdown-text/src/NativeMarkdownBlock.tsx @@ -17,6 +17,7 @@ import type { NativeMarkdownTextStyle, SelectableMarkdownSkill, } from "./SelectableMarkdownText.types"; +import { useStableFenceKey } from "./stableFenceKey"; import { useHighlightedCode, type HighlightedCode } from "./useHighlightedCode"; /** Set by SelectableMarkdownText so images anywhere in the block tree can use it. */ @@ -156,6 +157,7 @@ function NativeCodeBlock(props: { const theme = colorScheme === "dark" ? "dark" : "light"; const highlighted = useHighlightedCode(content, props.node.language, theme, props.highlightCode); const languageLabel = props.node.language?.toUpperCase() ?? "CODE"; + const fenceKey = useStableFenceKey(languageLabel, content); return ( { + const language = "ts"; + const shared = `const value = 1;\n${"x".repeat(24)}`; + const first = `${shared}${"a".repeat(40)}`; + const second = `${shared}${"b".repeat(40)}`; + expect(first.length).toBe(second.length); + expect(first.slice(0, 32)).toBe(second.slice(0, 32)); + expect(first).not.toBe(second); + + const afterFirst = nextFenceKey(null, language, first); + const afterSecond = nextFenceKey(afterFirst, language, second); + + expect(afterFirst.key).toBe(`${language}:${first}`); + expect(afterSecond.key).toBe(`${language}:${second}`); + expect(afterSecond.key).not.toBe(afterFirst.key); +}); + +it("keeps the key while the same fence streams longer", () => { + let state = nextFenceKey(null, "ts", "const"); + const initialKey = state.key; + state = nextFenceKey(state, "ts", "const value"); + state = nextFenceKey(state, "ts", "const value = 1;"); + expect(state.key).toBe(initialKey); +}); + +it("replaces the key when a streamed fence is swapped for another", () => { + let streamed = nextFenceKey(null, "ts", "const"); + streamed = nextFenceKey(streamed, "ts", "const value = 1;\nreturn value;"); + const swapped = nextFenceKey(streamed, "ts", "let other = 2;\nreturn other;"); + expect(swapped.key).toBe("ts:let other = 2;\nreturn other;"); + expect(swapped.key).not.toBe(streamed.key); +}); diff --git a/apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts b/apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts new file mode 100644 index 000000000000..860399880b6b --- /dev/null +++ b/apps/mobile/modules/t3-markdown-text/src/stableFenceKey.ts @@ -0,0 +1,45 @@ +import { useState } from "react"; + +export interface FenceKeyState { + readonly languageLabel: string; + readonly content: string; + readonly key: string; +} + +/** + * Identity for a fence's horizontal scroller and copy button. + * + * The key includes the full fence text when the fence changes, so a recycled + * row cannot keep another block's scroll offset or copied state just because + * the language, length, and opening characters match. Streaming appends keep + * the previous key; otherwise every chunk would remount the token tree. + */ +export function nextFenceKey( + previous: FenceKeyState | null, + languageLabel: string, + content: string, +): FenceKeyState { + if ( + previous !== null && + previous.languageLabel === languageLabel && + content.startsWith(previous.content) + ) { + return { languageLabel, content, key: previous.key }; + } + return { languageLabel, content, key: `${languageLabel}:${content}` }; +} + +/** Remembers the last fence drawn in this container. */ +export function useStableFenceKey(languageLabel: string, content: string): string { + const [stored, setStored] = useState(null); + const next = nextFenceKey(stored, languageLabel, content); + if ( + stored === null || + stored.languageLabel !== languageLabel || + stored.content !== content || + stored.key !== next.key + ) { + setStored(next); + } + return next.key; +} diff --git a/apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx b/apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx new file mode 100644 index 000000000000..6b90550396c2 --- /dev/null +++ b/apps/mobile/src/features/threads/AndroidTranscriptCodeSlice.tsx @@ -0,0 +1,160 @@ +import { memo } from "react"; +import { Platform, ScrollView, StyleSheet, Text, View } from "react-native"; + +import type { NativeMarkdownTextStyle } from "@t3tools/mobile-markdown-text/types"; + +import { CopyTextButton } from "../../components/CopyTextButton"; +import type { AndroidTranscriptCodePart } from "./androidTranscriptSlices"; + +const MONO_FONT_FAMILY = Platform.select({ + ios: "ui-monospace", + android: "monospace", + default: "monospace", +}); + +/** + * Code inside a transcript slice scales with the body size, matching the + * highlighted fence (12pt at the default 15pt body). + */ +function codeSliceFontSize(textStyle: NativeMarkdownTextStyle): number { + return Math.max(10, Math.round(textStyle.fontSize * 0.8)); +} + +/** + * Line height for a plain code window. Kept identical to the highlighted fence + * so a windowed block does not jump when the reader stops on it. + */ +function codeSliceLineHeight(textStyle: NativeMarkdownTextStyle): number { + return codeSliceFontSize(textStyle) + 6; +} + +/** + * Corner radii for one window of a fence. Middle and trailing windows stay + * square on the joined edge so the windows read as a single card. + */ +function codeSliceRadius(part: AndroidTranscriptCodePart): { + readonly borderTopLeftRadius: number; + readonly borderTopRightRadius: number; + readonly borderBottomLeftRadius: number; + readonly borderBottomRightRadius: number; +} { + const radius = part === "start" || part === "end" ? 10 : 0; + return { + borderTopLeftRadius: part === "start" ? radius : 0, + borderTopRightRadius: part === "start" ? radius : 0, + borderBottomLeftRadius: part === "end" ? radius : 0, + borderBottomRightRadius: part === "end" ? radius : 0, + }; +} + +/** + * One plain-text window of a long fenced block. + * + * Highlighted fences mount a `Text` per Shiki token. On Android that tree is + * created on the UI thread in the frame the row enters, which is the settled- + * thread hitch. A window is a single non-selectable `Text`; the header copies + * the whole fence, not just the lines in view. + */ +export const AndroidTranscriptCodeSlice = memo(function AndroidTranscriptCodeSlice(props: { + readonly text: string; + readonly language: string | null; + readonly part: Exclude; + readonly fullCode: string; + readonly textStyle: NativeMarkdownTextStyle; +}) { + const fontSize = codeSliceFontSize(props.textStyle); + const lineHeight = codeSliceLineHeight(props.textStyle); + const showHeader = props.part === "start"; + const languageLabel = props.language?.toUpperCase() ?? "CODE"; + return ( + + {showHeader ? ( + + + {languageLabel} + + + + ) : null} + + + {props.text} + + + + ); +}); + +const styles = StyleSheet.create({ + card: { + borderCurve: "continuous", + borderWidth: 1, + overflow: "hidden", + }, + header: { + minHeight: 42, + borderBottomWidth: 1, + paddingLeft: 14, + paddingRight: 6, + flexDirection: "row", + alignItems: "center", + justifyContent: "space-between", + }, + body: { + paddingHorizontal: 14, + paddingVertical: 12, + }, +}); diff --git a/apps/mobile/src/features/threads/ThreadFeed.tsx b/apps/mobile/src/features/threads/ThreadFeed.tsx index 83b0bef9c022..1197d4e0bac0 100644 --- a/apps/mobile/src/features/threads/ThreadFeed.tsx +++ b/apps/mobile/src/features/threads/ThreadFeed.tsx @@ -5,7 +5,12 @@ import { } from "./worktree-setup-card"; import * as Haptics from "expo-haptics"; import { KeyboardAwareLegendList } from "@legendapp/list/keyboard"; -import { useViewabilityAmount, type LegendListRef } from "@legendapp/list/react-native"; +import { + useRecyclingEffect, + useViewabilityAmount, + type LegendListRecyclingState, + type LegendListRef, +} from "@legendapp/list/react-native"; import type { ChatAttachment, ChatFileAttachment, @@ -138,6 +143,7 @@ import { resolveNativeMarkdownTypography, } from "../../lib/appearancePreferences"; import { useAppearancePreferences } from "../settings/appearance/AppearancePreferencesProvider"; +import { useStableFenceKey } from "@t3tools/mobile-markdown-text/fence-key"; import { markdownFileIconSource } from "@t3tools/mobile-markdown-text/file-icons"; import { PierreEntryIcon } from "../../components/PierreEntryIcon"; import { markdownLinkIconSource } from "@t3tools/mobile-markdown-text/link-icons"; @@ -172,6 +178,14 @@ import { WORK_GROUP_TOGGLE_HEIGHT, } from "./thread-work-log"; import { appendPendingThreadMessages, type PendingThreadFeedEntry } from "./pending-thread-feed"; +import { AndroidTranscriptCodeSlice } from "./AndroidTranscriptCodeSlice"; +import { + androidTranscriptItemType, + assistantSliceGap, + assistantSliceMarkdown, + expandAndroidAssistantTranscriptRows, + type AndroidAssistantSliceEntry, +} from "./androidTranscriptSlices"; import type { QueuedThreadMessage } from "../../state/thread-outbox-model"; import { useMarkdownCodeHighlight } from "./markdownCodeHighlightState"; import { @@ -235,15 +249,53 @@ const THREAD_FEED_DISCLOSURE_ENTER_TRANSITION = FadeIn.delay( THREAD_DISCLOSURE_TRANSITION_MS, ).duration(140); -// Entering animations must only play for rows born just now — LegendList -// remounts rows when they scroll back into view, and replaying an entrance for -// old content would be its own kind of jank. +// Entering animations must only play for rows born just now. Replaying one when +// a settled row scrolls back into view is its own kind of jank. const FRESH_ENTRY_WINDOW_MS = 3_000; +/** True when a row was created moments ago and may fade in as it mounts. */ function isFreshTimestamp(input: string): boolean { const timestamp = Date.parse(input); return Number.isFinite(timestamp) && Date.now() - timestamp < FRESH_ENTRY_WINDOW_MS; } +/** + * Android reuses list containers. iOS remounts them; UITextView is cheap, and + * recycling would reuse its selection state. + */ +const RECYCLE_ANDROID_TRANSCRIPT_ROWS = Platform.OS === "android"; + +/** + * Row id LegendList last painted in this container, when the item has one. + * Streaming updates replace the object but keep the id; a recycle does not. + */ +function rowRecycleId(item: unknown): string | undefined { + if (typeof item !== "object" || item === null || !("id" in item)) { + return undefined; + } + const id = item.id; + return typeof id === "string" ? id : undefined; +} + +/** + * Runs `reset` when this container is reused for a different feed row. + * Same-id updates, including a streaming append, do not count as a recycle. + * Outside a list, or on iOS where rows remount, the effect never fires. + */ +function useResetOnRowRecycle(reset: () => void) { + const resetRef = useRef(reset); + resetRef.current = reset; + useRecyclingEffect( + useCallback((info: LegendListRecyclingState) => { + const previousId = rowRecycleId(info.prevItem); + const nextId = rowRecycleId(info.item); + if (previousId !== undefined && previousId === nextId) { + return; + } + resetRef.current(); + }, []), + ); +} + export interface ThreadFeedProps { readonly worktreeSetup?: WorktreeSetupCardProps | null; readonly setupWorkingStartedAt?: string | null; @@ -279,6 +331,10 @@ export interface ThreadFeedProps { } | null; } +/** + * Image attachment thumbnail. A failed load refreshes the URL once. Recycling + * the row clears that retry so the next message can load its own image. + */ function MessageAttachmentImage(props: { readonly environmentId: EnvironmentId; readonly attachmentId: string; @@ -300,6 +356,9 @@ function MessageAttachmentImage(props: { const uri = useAssetUrl(props.environmentId, resource); const refreshAssetUrl = useRefreshAssetUrl(props.environmentId, resource); const retriedImage = useRef(false); + useResetOnRowRecycle(() => { + retriedImage.current = false; + }); if (uri === null) { return ( @@ -362,6 +421,10 @@ function isFileAttachment(attachment: ChatAttachment): attachment is ChatFileAtt return attachment.type === "file"; } +/** + * File or video attachment row. An in-flight open is aborted when the row is + * recycled, so a reused container cannot finish the previous file's open. + */ function MessageAttachmentFile(props: { readonly environmentId: EnvironmentId; readonly attachment: ChatFileAttachment; @@ -399,6 +462,11 @@ function MessageAttachmentFile(props: { : null; const openingRef = useRef(null); const [opening, setOpening] = useState(false); + useResetOnRowRecycle(() => { + openingRef.current?.abort(); + openingRef.current = null; + setOpening(false); + }); useFocusEffect( useCallback(() => { @@ -562,8 +630,13 @@ const ThreadMediaVisibleContext = createContext(false); // LegendList only computes hook visibility when the list has a viewability config. const THREAD_MEDIA_VIEWABILITY_CONFIG = { itemVisiblePercentThreshold: 0 }; +/** + * Gates video thumbnails on viewability. Visibility belongs to the row in this + * container, so a recycled container starts hidden until the new row reports. + */ function ThreadMediaVisibility(props: { readonly children: ReactNode }) { const [visible, setVisible] = useState(false); + useResetOnRowRecycle(() => setVisible(false)); useViewabilityAmount( useCallback((token) => setVisible(token.sizeVisible > 0), []), ); @@ -634,6 +707,10 @@ const markdownLinkStyles = StyleSheet.create({ }, }); +/** + * External link with a favicon. A host that failed to load its icon is + * remembered on the row and cleared when that row is recycled. + */ const MarkdownExternalLink = memo(function MarkdownExternalLink(props: { readonly children: ReactNode; readonly color: string; @@ -642,6 +719,7 @@ const MarkdownExternalLink = memo(function MarkdownExternalLink(props: { readonly onPress: (href: string) => void; }) { const [failedHost, setFailedHost] = useState(null); + useResetOnRowRecycle(() => setFailedHost(null)); const linkIcon = resolveMarkdownLinkIcon(props.host); const faviconUrl = linkIcon ? null : faviconUrlForOrigin(`https://${props.host}`); @@ -828,6 +906,10 @@ const AssistantMarkdownContent = memo(function AssistantMarkdownContent(props: { }); }); +/** + * Highlighted fenced block. The copy control and horizontal scroller remount + * when the fence changes so a recycled row does not reuse the previous one. + */ function MarkdownCodeBlock(props: { readonly backgroundColor: string; readonly borderColor: string; @@ -843,6 +925,10 @@ function MarkdownCodeBlock(props: { }) { const content = props.content.replace(/\n$/, ""); const languageLabel = props.language?.trim() || "text"; + // Full fence text, kept stable while the same fence streams. A prefix of the + // body is not enough: recycled rows with the same language and opening + // characters would keep the previous scroll offset and copy state. + const fenceKey = useStableFenceKey(languageLabel, content); const highlighted = useMarkdownCodeHighlight({ code: content, enabled: props.highlightCode && Boolean(props.language?.trim()), @@ -872,6 +958,7 @@ function MarkdownCodeBlock(props: { {languageLabel} >; + +/** + * Draws one window of an assistant message that was split so Android can mount + * it across frames. Chrome (attachments, copy, timestamp) stays on the last + * window; the copy still covers the whole message. + */ +function renderAssistantTranscriptSlice( + entry: AndroidAssistantSliceEntry>, + props: Parameters[1], +) { + const { message } = entry.source; + const { markdownStyles, iconSubtleColor } = props; + const styles = markdownStyles.assistant; + const renderedFullText = renderAssistantCitationsAsText(message.text); + const sliceMarkdown = assistantSliceMarkdown(entry.slice); + const timestampLabel = formatMessageTime(message.updatedAt); + const attachments = message.attachments ?? []; + const assistantTurnStillInProgress = + props.unsettledTurnId !== null && message.turnId === props.unsettledTurnId; + const showAssistantMeta = + entry.isLast && + props.terminalAssistantMessageIds.has(message.id) && + !assistantTurnStillInProgress && + !message.streaming; + const hasWideBlock = + entry.slice.kind === "code" || + hasWideMarkdownBlock(sliceMarkdown ?? "", WIDE_MARKDOWN_BLOCK_OPTIONS); + const enterAnimated = entry.isFirst && isFreshTimestamp(message.createdAt); + const gap = assistantSliceGap(entry.slice, entry.isLast); + + return ( + 0 ? { marginBottom: gap } : undefined} + {...(enterAnimated ? { entering: FadeIn.duration(220) } : {})} + > + {sliceMarkdown && sliceMarkdown.trim().length > 0 ? ( + + + + ) : entry.slice.kind === "code" && entry.slice.codePart !== "only" ? ( + + ) : null} + {entry.isLast + ? attachments.map((attachment) => { + return isImageAttachment(attachment) ? ( + + ) : isFileAttachment(attachment) ? ( + + ) : ( + + ); + }) + : null} + {showAssistantMeta ? ( + + + + {timestampLabel} + + + ) : null} + + ); +} + +/** + * Renders one transcript row. Long assistant messages arrive already sliced on + * Android; every other row is unchanged. + */ function renderFeedEntry( - info: { item: PendingThreadFeedEntry; index: number }, + info: { item: PresentedThreadFeedEntry; index: number }, props: Pick< ThreadFeedProps, | "environmentId" @@ -1391,6 +1590,9 @@ function renderFeedEntry( }, ) { const entry = info.item; + if (entry.type === "assistant-slice") { + return renderAssistantTranscriptSlice(entry, props); + } const { markdownStyles, iconSubtleColor, userBubbleColor } = props; if (entry.type === "turn-fold") { @@ -1623,6 +1825,7 @@ function renderFeedEntry( value={props.userBubbleMaxWidth - USER_BUBBLE_HORIZONTAL_PADDING * 2} > 0 ? ( - appendPendingThreadMessages( - deriveThreadFeedPresentation( + expandAndroidAssistantTranscriptRows( + appendPendingThreadMessages( + deriveThreadFeedPresentation( + props.feed, + props.latestTurn, + expandedTurnIds, + expandedWorkGroupIds, + props.activeWorkStartedAt, + ), props.feed, - props.latestTurn, - expandedTurnIds, - expandedWorkGroupIds, - props.activeWorkStartedAt, + props.queuedMessages, ), - props.feed, - props.queuedMessages, + Platform.OS, ), [ props.queuedMessages, @@ -2600,7 +2815,7 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { } }, [settleDisclosureAfterLayout]); - const shouldRestoreVisibleContentPosition = useCallback((entry: ThreadFeedEntry) => { + const shouldRestoreVisibleContentPosition = useCallback((entry: { readonly id: string }) => { const disclosureAnchorKey = disclosureAnchorKeyRef.current; return disclosureAnchorKey === null || entry.id === disclosureAnchorKey; }, []); @@ -2713,11 +2928,13 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { // exact; message rows stay undefined and use LegendList's per-type running // average once one of their type has been measured. const getFixedItemSize = useCallback( - (entry: ThreadFeedEntry) => { + (entry: PresentedThreadFeedEntry) => { if (workRowSizing.fixedRowHeight === undefined) { return undefined; } switch (entry.type) { + case "assistant-slice": + return undefined; case "message": // A collapsed reasoning row is the same chrome as a work toggle. return entry.message.role === "reasoning" && !expandedReasoningMessageIds.has(entry.id) @@ -2747,11 +2964,14 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { // Disclosures can mount existing offscreen rows as well as new work rows. // Fade those in after movement; never retain removed rows over replacements. const renderItem = useCallback( - (info: { item: PendingThreadFeedEntry; index: number }) => ( - + (info: { item: PresentedThreadFeedEntry; index: number }) => { + const entering = disclosureToggleSettling + ? THREAD_FEED_DISCLOSURE_ENTER_TRANSITION + : undefined; + // Android recycles this container. A key would destroy the native text + // tree every time a settled row comes back. iOS keeps the key because + // those rows remount. + const row = ( {renderFeedEntry(info, { environmentId: props.environmentId, @@ -2793,8 +3013,16 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { ) : null} - - ), + ); + if (RECYCLE_ANDROID_TRANSCRIPT_ROWS) { + return {row}; + } + return ( + + {row} + + ); + }, [ props.worktreeSetup, props.setupWorkingStartedAt, @@ -2932,8 +3160,13 @@ export const ThreadFeed = memo(function ThreadFeed(props: ThreadFeedProps) { viewabilityConfig={THREAD_MEDIA_VIEWABILITY_CONFIG} keyExtractor={(entry) => entry.id} getItemType={(entry) => - entry.type === "message" ? `message:${entry.message.role}` : entry.type + androidTranscriptItemType(entry) ?? + (entry.type === "message" ? `message:${entry.message.role}` : entry.type) } + // Recycling is list-wide, so stateful children are keyed by row id + // and reset when the container is reused. Assistant text is not + // keyed: a small slice updates in place instead of remounting. + recycleItems={RECYCLE_ANDROID_TRANSCRIPT_ROWS} getFixedItemSize={getFixedItemSize} // Virtualized rows must move with their measurements. Native layout // transitions can retain stale positions during sync, even at duration 0. diff --git a/apps/mobile/src/features/threads/androidTranscriptSlices.test.ts b/apps/mobile/src/features/threads/androidTranscriptSlices.test.ts new file mode 100644 index 000000000000..60a9b00fc6bc --- /dev/null +++ b/apps/mobile/src/features/threads/androidTranscriptSlices.test.ts @@ -0,0 +1,463 @@ +import { describe, expect, it } from "vite-plus/test"; + +import { + ANDROID_TRANSCRIPT_CODE_LINE_BUDGET, + ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET, + ANDROID_TRANSCRIPT_SLICE_GAP, + androidTranscriptItemType, + assistantSliceGap, + assistantSliceMarkdown, + expandAndroidAssistantTranscriptRows, + fencedCodeMarkdown, + splitAssistantTranscriptSlices, + type AndroidTranscriptSlice, +} from "./androidTranscriptSlices"; + +/** Minimal feed message for slice tests. */ +function message(id: string, role: "assistant" | "user", text: string) { + return { + type: "message" as const, + id, + createdAt: "2026-01-01T00:00:00.000Z", + message: { role, text }, + }; +} + +/** Assistant message fixture. */ +function assistantMessage(id: string, text: string) { + return message(id, "assistant", text); +} + +/** Fenced block with a numbered line per row, so windows are easy to count. */ +function codeFence(language: string, lineCount: number): string { + const body = Array.from({ length: lineCount }, (_, index) => `line ${index + 1}`).join("\n"); + return `\`\`\`${language}\n${body}\n\`\`\``; +} + +/** Compact kind/part/line-count label for slice assertions. */ +function sliceKinds(slices: readonly AndroidTranscriptSlice[]): string[] { + return slices.map((slice) => + slice.kind === "code" ? `code:${slice.codePart}:${slice.text.split("\n").length}` : "markdown", + ); +} + +/** Markdown slices only. Code windows are checked separately. */ +function markdownTexts(slices: readonly AndroidTranscriptSlice[]): string[] { + return slices.filter((slice) => slice.kind === "markdown").map((slice) => slice.text); +} + +describe("splitAssistantTranscriptSlices", () => { + it("leaves a short message with one fence as a single row", () => { + const markdown = `See this.\n\n${codeFence("ts", 4)}`; + expect(splitAssistantTranscriptSlices(markdown)).toBeNull(); + }); + + it("splits prose that would mount as one oversized selectable text", () => { + const paragraph = "word ".repeat(200).trim(); + const markdown = `${paragraph}\n\n${paragraph}`; + const slices = splitAssistantTranscriptSlices(markdown); + expect(slices).not.toBeNull(); + expect(slices!.every((slice) => slice.kind === "markdown")).toBe(true); + for (const slice of slices!) { + expect(slice.text.length).toBeLessThanOrEqual(ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET); + } + const words = (value: string) => value.replaceAll(/\s+/g, " ").trim(); + expect(words(slices!.map((slice) => slice.text).join(" "))).toBe(words(markdown)); + }); + + it("puts each fence on its own row when a message has more than one", () => { + const markdown = `${codeFence("ts", 3)}\n\nBetween.\n\n${codeFence("go", 2)}`; + const slices = splitAssistantTranscriptSlices(markdown); + expect(sliceKinds(slices!)).toEqual(["code:only:3", "markdown", "code:only:2"]); + expect(assistantSliceMarkdown(slices![0]!)).toContain("```ts"); + expect(assistantSliceMarkdown(slices![2]!)).toContain("```go"); + }); + + it("windows a long fence and keeps earlier windows stable as it grows", () => { + const before = codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 1); + const after = codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET * 2 + 3); + const first = splitAssistantTranscriptSlices(before); + const grown = splitAssistantTranscriptSlices(after); + expect(sliceKinds(first!)).toEqual([ + `code:start:${ANDROID_TRANSCRIPT_CODE_LINE_BUDGET}`, + "code:end:1", + ]); + expect(sliceKinds(grown!)).toEqual([ + `code:start:${ANDROID_TRANSCRIPT_CODE_LINE_BUDGET}`, + `code:middle:${ANDROID_TRANSCRIPT_CODE_LINE_BUDGET}`, + "code:end:3", + ]); + const grownHead = grown![0]!; + const firstHead = first![0]!; + const grownNext = grown![1]!; + const firstNext = first![1]!; + expect(grownHead).toMatchObject({ + key: firstHead.key, + text: firstHead.text, + }); + expect(grownNext.key).toBe(firstNext.key); + expect(grownHead.kind === "code" && grownHead.text.startsWith("line 1")).toBe(true); + expect(grownHead.kind === "code" && grownHead.text.includes("line 17")).toBe(false); + expect(assistantSliceMarkdown(grown![0]!)).toBeNull(); + expect( + grown!.every( + (slice) => + slice.kind !== "code" || + slice.fullCode.split("\n").length === ANDROID_TRANSCRIPT_CODE_LINE_BUDGET * 2 + 3, + ), + ).toBe(true); + }); + + it("treats an unclosed streaming fence as code and does not renumber finished windows", () => { + const opened = `Intro.\n\n\`\`\`ts\n${Array.from({ length: 10 }, (_, index) => `line ${index + 1}`).join("\n")}`; + const longer = `${opened}\n${Array.from({ length: 12 }, (_, index) => `line ${index + 11}`).join("\n")}`; + const before = splitAssistantTranscriptSlices(opened); + const after = splitAssistantTranscriptSlices(longer); + expect(before).toBeNull(); + const intro = after![0]!; + const codeHead = after![1]!; + expect(intro).toMatchObject({ kind: "markdown", text: "Intro." }); + expect(codeHead).toMatchObject({ kind: "code", codePart: "start" }); + expect(codeHead.kind === "code" && codeHead.text.split("\n")).toHaveLength( + ANDROID_TRANSCRIPT_CODE_LINE_BUDGET, + ); + }); + + it("keeps a GFM table intact when the surrounding message is split", () => { + const cell = "c".repeat(80); + const row = `| ${cell} | ${cell} |`; + const table = [row, "| --- | --- |", row, row].join("\n"); + const slices = splitAssistantTranscriptSlices(`${table}\n\n${"word ".repeat(200).trim()}`); + const tableSlices = slices!.filter((slice) => slice.text.includes("| --- |")); + expect(tableSlices).toHaveLength(1); + expect(tableSlices[0]!.text).toBe(table); + }); + + it("ignores a four-space indented fence and still splits long prose", () => { + const indented = ` \`\`\`\n${"x".repeat(ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET + 40)}`; + const slices = splitAssistantTranscriptSlices(indented); + expect(slices!.every((slice) => slice.kind === "markdown")).toBe(true); + }); + + it("does not treat a backtick fence as code when the info string contains a backtick", () => { + const prose = `\`\`\`js const tick = \`oops\`\n${"word ".repeat(200).trim()}`; + const slices = splitAssistantTranscriptSlices(prose); + expect(slices?.some((slice) => slice.kind === "code") ?? false).toBe(false); + }); + + it("splits an ordered list only between complete items", () => { + const items = Array.from( + { length: 30 }, + (_, index) => `${index + 1}. ${"entry ".repeat(8).trim()}`, + ); + const slices = splitAssistantTranscriptSlices(items.join("\n")); + expect(slices).not.toBeNull(); + expect(slices!.length).toBeGreaterThan(1); + const lines = slices!.flatMap((slice) => + slice.text.split("\n").filter((line) => line.trim().length > 0), + ); + expect(lines).toEqual(items); + for (const slice of slices!) { + expect(slice.kind).toBe("markdown"); + expect(slice.text).toMatch(/^\d+\. /); + } + }); + + it("keeps a long ordered-list item whole instead of dropping its marker", () => { + const item = `12. ${"word ".repeat(200).trim()}`; + const tail = "after ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${item}\n\n${tail}`)!; + const owner = slices.find((slice) => slice.text.includes("12. ")); + expect(owner?.text.startsWith("12. ")).toBe(true); + expect(owner?.text).toContain("word"); + expect(owner?.text.includes(tail.slice(0, 12))).toBe(false); + for (const slice of slices) { + if (slice === owner) continue; + expect(slice.text.startsWith("word")).toBe(false); + } + }); + + it("keeps nested list items and loose continuation with their parent", () => { + const parent = `1. parent item\n - child stays\n - child two\n\n continuation stays`; + const siblings = Array.from( + { length: 4 }, + (_, index) => `${index + 2}. ${"entry ".repeat(160).trim()}`, + ); + const slices = splitAssistantTranscriptSlices(`${parent}\n\n${siblings.join("\n")}`)!; + const first = slices.find((slice) => slice.text.includes("parent item")); + expect(first?.text).toContain("- child stays"); + expect(first?.text).toContain("- child two"); + expect(first?.text).toContain("continuation stays"); + expect(first?.text).not.toContain("2. "); + }); + + it("keeps a lazy list continuation with its marker", () => { + const item = `1. short\n${"lazy ".repeat(40).trim()}`; + const tail = "after ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${item}\n\n${tail}`)!; + const owner = slices.find((slice) => slice.text.includes("1. short")); + expect(owner?.text.startsWith("1. short")).toBe(true); + expect(owner?.text).toContain("lazy"); + expect(owner?.text.includes("after")).toBe(false); + }); + + it("keeps lazy blockquote continuation with the quote marker", () => { + const quote = `> quoted start\n${"still quoted ".repeat(30).trim()}`; + const after = "outside ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${quote}\n\n${after}`)!; + const quoteSlice = slices.find((slice) => slice.text.includes("> quoted start")); + expect(quoteSlice?.text.startsWith(">")).toBe(true); + expect(quoteSlice?.text).toContain("still quoted"); + expect(quoteSlice?.text.includes("outside")).toBe(false); + }); + + it("keeps a multi-line setext heading with its underline", () => { + const heading = `${"Title words ".repeat(40).trim()}\n${"still the title ".repeat(20).trim()}\n---`; + const after = "body ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${heading}\n\n${after}`)!; + const headingSlice = slices.find((slice) => slice.text.includes("Title words")); + expect(headingSlice?.text).toContain("still the title"); + expect(headingSlice?.text.trimEnd().endsWith("---")).toBe(true); + expect(headingSlice?.text.includes("body")).toBe(false); + expect(slices.some((slice) => slice.text.trim() === "---")).toBe(false); + }); + + it("keeps a blockquote together when the message around it is split", () => { + const quote = ["> " + "alpha ".repeat(80).trim(), ">", "> " + "beta ".repeat(40).trim()].join( + "\n", + ); + const after = "gamma ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${quote}\n\n${after}`)!; + const quoteSlice = slices.find((slice) => slice.text.includes("> alpha")); + expect(quoteSlice?.text).toContain("> beta"); + expect(quoteSlice?.text.startsWith(">")).toBe(true); + expect(slices.some((slice) => slice.text.includes("gamma") && !slice.text.includes(">"))).toBe( + true, + ); + }); + + it("does not turn a wrapped ordered line into its own list", () => { + const paragraph = `${"word ".repeat(80).trim()}\n2. ${"cont ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(paragraph)!; + const owner = slices.find((slice) => slice.text.includes("2. ")); + expect(owner?.text.startsWith("2.")).toBe(false); + expect(owner?.text).toMatch(/\S\n2\. /); + }); + + it("keeps an inline link in one slice, including a long destination", () => { + const url = `https://example.com/${"a".repeat(200)}`; + const label = `docs ${"label ".repeat(10).trim()}`; + const link = `[${label}](${url})`; + const before = "before ".repeat(80).trim(); + const after = "after ".repeat(80).trim(); + const slices = splitAssistantTranscriptSlices(`${before} ${link} ${after}`)!; + expect(markdownTexts(slices).filter((text) => text.includes(link))).toHaveLength(1); + for (const text of markdownTexts(slices)) { + if (text.includes(link)) continue; + expect(text.includes(url)).toBe(false); + expect(text.includes(`](${url.slice(0, 24)}`)).toBe(false); + } + }); + + it("keeps emphasis and inline code intact when the paragraph is split", () => { + const bold = `**${"bold ".repeat(30).trim()}**`; + const code = `\`${"c".repeat(180)}\``; + const text = `${"word ".repeat(80).trim()} ${bold} ${code} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(text)!; + expect(markdownTexts(slices).filter((slice) => slice.includes(bold))).toHaveLength(1); + expect(markdownTexts(slices).filter((slice) => slice.includes(code))).toHaveLength(1); + for (const slice of markdownTexts(slices)) { + const markers = slice.match(/\*\*/g)?.length ?? 0; + expect(markers % 2).toBe(0); + } + }); + + it("keeps a raw HTML element in one slice", () => { + const html = `${"x".repeat(240)}`; + const text = `${"word ".repeat(80).trim()} ${html} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(text)!; + expect(markdownTexts(slices).filter((slice) => slice.includes(html))).toHaveLength(1); + }); + + it("copies link reference definitions onto the slice that uses them", () => { + const body = `${"word ".repeat(200).trim()}\n\nSee [the docs][ref].\n\n${"word ".repeat(200).trim()}`; + const definition = "[ref]: https://example.com/docs"; + const slices = splitAssistantTranscriptSlices(`${body}\n\n${definition}`)!; + const linkSlice = slices.find((slice) => slice.text.includes("[the docs][ref]")); + expect(linkSlice?.text).toContain(definition); + }); + + it("keeps an empty fenced block when the message is split", () => { + const markdown = `${codeFence("ts", 3)}\n\n\`\`\`ts\n\`\`\`\n\n${codeFence("go", 2)}`; + const slices = splitAssistantTranscriptSlices(markdown)!; + const empty = slices.find((slice) => slice.kind === "code" && slice.text === ""); + expect(empty).toMatchObject({ + kind: "code", + codePart: "only", + language: "ts", + text: "", + fullCode: "", + }); + expect(assistantSliceMarkdown(empty!)).toContain("```ts"); + }); + + it("strips the opening fence indent from code that is split out", () => { + const fence = " ```ts\n const value = 1;\n still indented\n ```"; + const slices = splitAssistantTranscriptSlices(`${fence}\n\n${"word ".repeat(200).trim()}`)!; + const code = slices.find((slice) => slice.kind === "code"); + expect(code).toMatchObject({ + kind: "code", + text: "const value = 1;\n still indented", + fullCode: "const value = 1;\n still indented", + }); + }); + + it("copies a reference definition whose destination is on the next line", () => { + const body = `${"word ".repeat(200).trim()}\n\nSee [the docs][docs].\n\n${"word ".repeat(200).trim()}`; + const definition = "[docs]:\n https://example.com/docs"; + const slices = splitAssistantTranscriptSlices(`${body}\n\n${definition}`)!; + const linkSlice = slices.find((slice) => slice.text.includes("[the docs][docs]")); + expect(linkSlice?.text).toContain("[docs]:"); + expect(linkSlice?.text).toContain("https://example.com/docs"); + }); + + it("copies a reference definition whose title is on the next line", () => { + const body = `${"word ".repeat(200).trim()}\n\nSee [the docs][docs].\n\n${"word ".repeat(200).trim()}`; + const definition = '[docs]: https://example.com/docs\n"API docs"'; + const slices = splitAssistantTranscriptSlices(`${body}\n\n${definition}`)!; + const linkSlice = slices.find((slice) => slice.text.includes("[the docs][docs]")); + expect(linkSlice?.text).toContain("https://example.com/docs"); + expect(linkSlice?.text).toContain('"API docs"'); + }); + + it("keeps nested elements of the same name in one slice", () => { + const html = `outer inner ${"still ".repeat(80).trim()}`; + const text = `${"word ".repeat(80).trim()} ${html} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(text)!; + const owners = markdownTexts(slices).filter((slice) => slice.includes("")); + expect(owners).toHaveLength(1); + expect(owners[0]).toContain(html); + }); + + it("keeps a textarea block intact through its closing tag", () => { + const block = ``; + const slices = splitAssistantTranscriptSlices(`${block}\n\n${"word ".repeat(200).trim()}`)!; + const owners = slices.filter((slice) => slice.text.toLowerCase().includes("textarea")); + expect(owners).toHaveLength(1); + expect(owners[0]!.text).toContain(block); + }); + + it("does not start a slice on a mid-line list, heading, or quote marker", () => { + const markdown = `${"word ".repeat(150)}- dash ${"word ".repeat(40)}# title ${"word ".repeat(40)}> quote ${"word ".repeat(80)}`; + const slices = splitAssistantTranscriptSlices(markdown)!; + expect(slices.length).toBeGreaterThan(1); + for (const slice of slices) { + expect(slice.text).not.toMatch(/^[-+*] |^#{1,6} |^>/); + } + }); + + it("keeps an artifact-template directive inside one slice", () => { + const directive = `::artifact-template{skill_name="artifact-template-hello-world" skill_directory="${"a".repeat(500)}" display_name="Hello World" artifact_kind="document"}`; + const markdown = `${"word ".repeat(80).trim()} ${directive} ${"word ".repeat(80).trim()}`; + const slices = splitAssistantTranscriptSlices(markdown)!; + const owners = slices.filter((slice) => slice.text.includes("::artifact-template")); + expect(owners).toHaveLength(1); + expect(owners[0]!.text).toContain(directive); + }); + + it("leaves a fence inside a list item in that item", () => { + const item = `1. run this\n\n \`\`\`ts\n const value = 1;\n \`\`\``; + const rest = "tail ".repeat(200).trim(); + const slices = splitAssistantTranscriptSlices(`${item}\n\n${rest}`)!; + const owner = slices.find((slice) => slice.text.includes("run this")); + expect(owner?.kind).toBe("markdown"); + expect(owner?.text).toContain("```ts"); + expect(owner?.text).toContain("const value = 1;"); + expect(slices.some((slice) => slice.kind === "code")).toBe(false); + }); +}); + +describe("expandAndroidAssistantTranscriptRows", () => { + it("returns the same array off Android and for rows that are already small", () => { + const feed = [ + message("user-1", "user", "hello"), + { type: "thinking" as const, id: "thinking" }, + assistantMessage("short", `ok\n\n${codeFence("ts", 2)}`), + ]; + expect(expandAndroidAssistantTranscriptRows(feed, "ios")).toBe(feed); + expect(expandAndroidAssistantTranscriptRows(feed, "android")).toBe(feed); + }); + + it("does not slice a long user message", () => { + const feed = [message("user-1", "user", "word ".repeat(400).trim())]; + expect(expandAndroidAssistantTranscriptRows(feed, "android")).toBe(feed); + }); + + it("keeps the message id on the first slice and appends the rest", () => { + const fence = codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 2); + const feed = [ + message("user-1", "user", "ship it"), + { type: "work-toggle" as const, id: "work-1" }, + assistantMessage("assistant-1", fence), + ]; + const rows = expandAndroidAssistantTranscriptRows(feed, "android"); + expect(rows.map((row) => row.type)).toEqual([ + "message", + "work-toggle", + "assistant-slice", + "assistant-slice", + ]); + expect(rows[0]).toBe(feed[0]); + expect(rows[1]).toBe(feed[1]); + const head = rows[2]; + const tail = rows[3]; + if (head?.type !== "assistant-slice" || tail?.type !== "assistant-slice") { + throw new Error("expected assistant slices"); + } + expect(head.id).toBe("assistant-1"); + expect(head.isFirst).toBe(true); + expect(head.isLast).toBe(false); + expect(tail.id).toBe(`assistant-1:${tail.slice.key}`); + expect(tail.isLast).toBe(true); + expect(tail.source).toBe(feed[2]); + expect(androidTranscriptItemType(head)).toBe("assistant-code-head"); + expect(androidTranscriptItemType(tail)).toBe("assistant-code-body"); + expect(androidTranscriptItemType(feed[0]!)).toBeNull(); + }); + + it("keeps earlier slice ids when the settled message later grows at the end", () => { + const before = expandAndroidAssistantTranscriptRows( + [assistantMessage("assistant-1", codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 1))], + "android", + ); + const after = expandAndroidAssistantTranscriptRows( + [assistantMessage("assistant-1", codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET + 4))], + "android", + ); + expect(before.map((row) => row.id)).toEqual([after[0]!.id, after[1]!.id]); + expect(after).toHaveLength(2); + }); +}); + +describe("assistant slice presentation", () => { + it("closes a fence that contains backticks and leaves plain windows without markdown", () => { + const fenced = fencedCodeMarkdown("ts", "const tick = ```;"); + expect(fenced.startsWith("````")).toBe(true); + expect(fenced).toContain("const tick = ```;"); + expect(fenced.trimEnd().endsWith("````")).toBe(true); + }); + + it("does not gap code windows that belong to the same fence", () => { + const slices = splitAssistantTranscriptSlices( + codeFence("ts", ANDROID_TRANSCRIPT_CODE_LINE_BUDGET * 2 + 1), + )!; + expect(assistantSliceGap(slices[0]!, false)).toBe(0); + expect(assistantSliceGap(slices[1]!, false)).toBe(0); + expect(assistantSliceGap(slices[2]!, true)).toBe(0); + const prose = splitAssistantTranscriptSlices( + `${"word ".repeat(200).trim()}\n\n${"word ".repeat(200).trim()}`, + )!; + expect(assistantSliceGap(prose[0]!, false)).toBe(ANDROID_TRANSCRIPT_SLICE_GAP); + expect(assistantSliceGap(prose[0]!, true)).toBe(0); + }); +}); diff --git a/apps/mobile/src/features/threads/androidTranscriptSlices.ts b/apps/mobile/src/features/threads/androidTranscriptSlices.ts new file mode 100644 index 000000000000..0d43d2a968c2 --- /dev/null +++ b/apps/mobile/src/features/threads/androidTranscriptSlices.ts @@ -0,0 +1,1926 @@ +import { renderAssistantCitationsAsText } from "@t3tools/shared/assistantCitations"; + +/** + * Prose mounted as one Android text view. Larger selectable paragraphs are the + * stalls measured when a settled thread scrolls back into view. + */ +export const ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET = 720; + +/** + * Code lines mounted as one Android row. A longer fence is windowed so a fling + * never builds that fence's whole text tree in a single frame. + */ +export const ANDROID_TRANSCRIPT_CODE_LINE_BUDGET = 16; + +/** Gap between slices that are not a continuation of the same code fence. */ +export const ANDROID_TRANSCRIPT_SLICE_GAP = 14; + +export type AndroidTranscriptCodePart = "only" | "start" | "middle" | "end"; + +export type AndroidTranscriptSlice = + | { + readonly kind: "markdown"; + readonly key: string; + readonly text: string; + } + | { + readonly kind: "code"; + readonly key: string; + readonly text: string; + readonly language: string | null; + readonly codePart: AndroidTranscriptCodePart; + readonly fullCode: string; + }; + +interface TranscriptMessageEntry { + readonly type: "message"; + readonly id: string; + readonly createdAt: string; + readonly message: { + readonly role: string; + readonly text: string; + }; +} + +export interface AndroidAssistantSliceEntry { + readonly type: "assistant-slice"; + readonly id: string; + readonly createdAt: string; + readonly source: TSource; + readonly slice: AndroidTranscriptSlice; + readonly isFirst: boolean; + readonly isLast: boolean; +} + +interface SourceLine { + readonly text: string; + readonly start: number; + readonly end: number; +} + +interface FenceOpener { + readonly char: "`" | "~"; + readonly length: number; + readonly info: string; + /** Spaces before the opening fence. Content lines lose up to this many. */ + readonly indent: number; +} + +interface ListMarkerInfo { + readonly indent: number; + readonly ordered: boolean; + readonly number: number | null; +} + +type HtmlBlockKind = + | { readonly kind: "comment" } + | { readonly kind: "processing" } + | { readonly kind: "declaration" } + | { readonly kind: "pre"; readonly tag: string } + | { readonly kind: "block"; readonly tag: string }; + +interface MarkdownBlock { + readonly kind: "markdown"; + readonly start: number; + readonly end: number; + /** + * True only for a plain paragraph. Lists, quotes, tables, and HTML stay one + * piece because each slice is parsed as its own Markdown document. + */ + readonly inlineSplittable: boolean; +} + +interface CodeBlock { + readonly kind: "code"; + readonly start: number; + readonly body: string; + readonly language: string | null; + readonly lineCount: number; +} + +type TranscriptBlock = MarkdownBlock | CodeBlock; + +interface TextRange { + readonly start: number; + readonly end: number; +} + +interface EmphasisDelimiter { + readonly char: "*" | "_" | "~"; + readonly pos: number; + origLen: number; + len: number; + readonly canOpen: boolean; + readonly canClose: boolean; +} + +const HTML_BLOCK_TAGS = new Set([ + "address", + "article", + "aside", + "base", + "basefont", + "blockquote", + "body", + "caption", + "center", + "col", + "colgroup", + "dd", + "details", + "dialog", + "dir", + "div", + "dl", + "dt", + "fieldset", + "figcaption", + "figure", + "footer", + "form", + "frame", + "frameset", + "h1", + "h2", + "h3", + "h4", + "h5", + "h6", + "head", + "header", + "hr", + "html", + "iframe", + "legend", + "li", + "link", + "main", + "menu", + "menuitem", + "nav", + "noframes", + "ol", + "optgroup", + "option", + "p", + "param", + "search", + "section", + "summary", + "table", + "tbody", + "td", + "tfoot", + "th", + "thead", + "title", + "tr", + "track", + "ul", +]); + +/** + * Splits an expensive assistant message into bounded Android list rows. + * + * Each Markdown slice is a complete document: links, emphasis, inline code, + * lists, blockquotes, tables, and HTML are never cut in half. A construct + * longer than the budget stays one row. Short messages return null so they + * keep the highlighted renderer. Keys come from source offsets, so a streaming + * append does not renumber earlier slices. + */ +export function splitAssistantTranscriptSlices( + markdown: string, +): readonly AndroidTranscriptSlice[] | null { + if ( + markdown.length <= ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET && + !markdown.includes("```") && + !markdown.includes("~~~") + ) { + return null; + } + const normalized = markdown.replaceAll("\r\n", "\n"); + const blocks = parseTranscriptBlocks(normalized); + if (!shouldSplitAssistantTranscript(normalized, blocks)) { + return null; + } + const definitions = collectLinkReferenceDefinitions(normalized); + const slices = assembleTranscriptSlices(normalized, blocks, definitions); + return slices.length > 1 ? slices : null; +} + +/** + * Expands assistant messages that would mount an unbounded text tree. + * + * Non-Android feeds are returned unchanged, including the same array, so iOS + * keeps today's row identity. The first slice reuses the message id; later + * slices append, which is the direction live-follow already scrolls. + */ +export function expandAndroidAssistantTranscriptRows< + TEntry extends { readonly type: string; readonly id: string }, +>( + entries: readonly TEntry[], + platform: string, +): readonly (TEntry | AndroidAssistantSliceEntry>)[] { + if (platform !== "android") { + return entries; + } + + let changed = false; + const rows: (TEntry | AndroidAssistantSliceEntry>)[] = []; + for (const entry of entries) { + const slices = slicesForEntry(entry); + if (!slices) { + rows.push(entry); + continue; + } + changed = true; + const source = entry as Extract; + for (let index = 0; index < slices.length; index += 1) { + const slice = slices[index]; + if (!slice) continue; + rows.push({ + type: "assistant-slice", + id: index === 0 ? source.id : `${source.id}:${slice.key}`, + createdAt: source.createdAt, + source, + slice, + isFirst: index === 0, + isLast: index === slices.length - 1, + }); + } + } + + return changed ? rows : entries; +} + +/** + * LegendList item type for a slice. The list prefers a same-type container, so + * a plain code window is reused for another code window before a prose row. + */ +export function androidTranscriptItemType(entry: { + readonly type: string; + readonly slice?: AndroidTranscriptSlice; +}): string | null { + if (entry.type !== "assistant-slice" || !entry.slice) { + return null; + } + if (entry.slice.kind === "markdown") { + return "assistant-markdown-slice"; + } + if (entry.slice.codePart === "only") { + return "assistant-code-block"; + } + if (entry.slice.codePart === "start") { + return "assistant-code-head"; + } + return "assistant-code-body"; +} + +/** + * Markdown for slices the shared renderer can draw. Plain windows of a long + * fence return null; those are one `Text`, not a token per span. + */ +export function assistantSliceMarkdown(slice: AndroidTranscriptSlice): string | null { + if (slice.kind === "markdown") { + return slice.text; + } + if (slice.codePart !== "only") { + return null; + } + return fencedCodeMarkdown(slice.language, slice.text); +} + +/** + * Space after a slice row. Continued code windows share one card, so they do + * not take the gap that separate blocks use. The last slice uses the message + * row's own bottom margin instead. + */ +export function assistantSliceGap(slice: AndroidTranscriptSlice, isLast: boolean): number { + if (isLast) { + return 0; + } + if (slice.kind === "code" && (slice.codePart === "start" || slice.codePart === "middle")) { + return 0; + } + return ANDROID_TRANSCRIPT_SLICE_GAP; +} + +/** + * Wraps a short code window in a fence the markdown renderer already knows how + * to draw. The fence is longer than any backtick run in the body so the body + * cannot close it early. + */ +export function fencedCodeMarkdown(language: string | null, code: string): string { + let longestRun = 0; + let run = 0; + for (const character of code) { + if (character === "`") { + run += 1; + longestRun = Math.max(longestRun, run); + } else { + run = 0; + } + } + const fence = "`".repeat(Math.max(3, longestRun + 1)); + const info = language ? language.replace(/[\r\n`]/g, "") : ""; + return `${fence}${info}\n${code}\n${fence}`; +} + +const assistantSliceCache = new Map(); +const ASSISTANT_SLICE_CACHE_LIMIT = 200; + +/** + * Returns slice rows for an assistant message, or null when the entry should + * stay as it is. User, reasoning, and non-message rows are never split. + * Results are cached by the raw message text so a streaming tail does not + * re-scan every earlier message. + */ +function slicesForEntry(entry: { + readonly type: string; + readonly id: string; +}): readonly AndroidTranscriptSlice[] | null { + if (!isAssistantMessageEntry(entry)) { + return null; + } + const raw = entry.message.text; + const cached = assistantSliceCache.get(raw); + if (cached !== undefined) { + return cached; + } + const text = renderAssistantCitationsAsText(raw); + const slices = text.trim().length === 0 ? null : splitAssistantTranscriptSlices(text); + assistantSliceCache.delete(raw); + assistantSliceCache.set(raw, slices); + while (assistantSliceCache.size > ASSISTANT_SLICE_CACHE_LIMIT) { + const oldest = assistantSliceCache.keys().next().value; + if (oldest === undefined) { + break; + } + assistantSliceCache.delete(oldest); + } + return slices; +} + +/** + * Narrows a feed entry to an assistant message with the fields slicing reads. + */ +function isAssistantMessageEntry(entry: { + readonly type: string; + readonly id: string; +}): entry is TranscriptMessageEntry { + if ( + entry.type !== "message" || + !("createdAt" in entry) || + typeof entry.createdAt !== "string" || + !("message" in entry) + ) { + return false; + } + const message = entry.message; + if ( + typeof message !== "object" || + message === null || + !("role" in message) || + !("text" in message) + ) { + return false; + } + return message.role === "assistant" && typeof message.text === "string"; +} + +/** + * True when one list row would mount more text than a frame can afford. + * One short fence stays on the highlighted path; a second fence or a long + * fence is enough to split. + */ +function shouldSplitAssistantTranscript( + markdown: string, + blocks: readonly TranscriptBlock[], +): boolean { + if (markdown.length > ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + return true; + } + let codeBlocks = 0; + for (const block of blocks) { + if (block.kind !== "code") { + continue; + } + codeBlocks += 1; + if (block.lineCount > ANDROID_TRANSCRIPT_CODE_LINE_BUDGET) { + return true; + } + } + return codeBlocks >= 2; +} + +/** + * Walks top-level Markdown blocks. Fences become code windows. Lists, quotes, + * tables, and HTML stay intact. Only plain paragraphs may be cut later. + */ +function parseTranscriptBlocks(markdown: string): readonly TranscriptBlock[] { + const lines = sourceLines(markdown); + const lineTexts = lines.map((sourceLine) => sourceLine.text); + const blocks: TranscriptBlock[] = []; + let index = 0; + while (index < lines.length) { + const line = lines[index]; + if (!line || isBlankLine(line.text)) { + index += 1; + continue; + } + const fence = openingFence(line.text); + if (fence) { + const consumed = consumeFence(lines, index, fence); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + if (isAtxHeading(line.text) || isThematicBreak(line.text)) { + blocks.push(markdownSpan(line, line, false)); + index += 1; + continue; + } + if (isBlockquoteLine(line.text)) { + const consumed = consumeBlockquote(lines, index); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + if (listMarker(line.text)) { + const consumed = consumeList(lines, index); + blocks.push(...consumed.blocks); + index = consumed.next; + continue; + } + const definitionLines = parseLinkReferenceDefinition(lineTexts, index); + if (definitionLines !== null) { + const definitionEnd = index + definitionLines; + const definitionLast = lines[definitionEnd - 1] ?? line; + blocks.push(markdownSpan(line, definitionLast, false)); + index = definitionEnd; + continue; + } + const html = htmlBlockKind(line.text); + if (html) { + const consumed = consumeHtmlBlock(lines, index, html); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + const next = lines[index + 1]?.text ?? null; + if (next !== null && isTableStart(line.text, next)) { + const consumed = consumeTable(lines, index); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + if (isIndentedCodeLine(line.text)) { + const consumed = consumeIndentedCode(lines, index); + blocks.push(consumed.block); + index = consumed.next; + continue; + } + const consumed = consumeParagraph(lines, index); + blocks.push(consumed.block); + index = consumed.next; + } + return blocks; +} + +/** + * Turns parsed blocks into row slices. Markdown on either side of a fence is + * packed separately so a fence cannot be swallowed by a prose range. + */ +function assembleTranscriptSlices( + markdown: string, + blocks: readonly TranscriptBlock[], + definitions: string, +): AndroidTranscriptSlice[] { + const slices: AndroidTranscriptSlice[] = []; + let markdownSpans: TextRange[] = []; + const flushMarkdown = () => { + if (markdownSpans.length === 0) return; + slices.push(...packMarkdownSpans(markdown, markdownSpans, definitions)); + markdownSpans = []; + }; + for (const block of blocks) { + if (block.kind === "code") { + flushMarkdown(); + slices.push(...sliceCodeBlock(block)); + continue; + } + markdownSpans.push(...expandMarkdownBlock(markdown, block)); + } + flushMarkdown(); + return slices; +} + +/** + * Breaks a plain paragraph on inline-safe whitespace. Every other block is + * one span, even when it is longer than the budget. + */ +function expandMarkdownBlock(markdown: string, block: MarkdownBlock): readonly TextRange[] { + if (!block.inlineSplittable) { + return [{ start: block.start, end: block.end }]; + } + const text = markdown.slice(block.start, block.end); + return splitPlainParagraph(text).map((range) => ({ + start: block.start + range.start, + end: block.start + range.end, + })); +} + +/** + * Packs neighboring Markdown spans up to the char budget. An oversized span + * is emitted alone so a long list item or quote is not joined to more text. + */ +function packMarkdownSpans( + markdown: string, + spans: readonly TextRange[], + definitions: string, +): AndroidTranscriptSlice[] { + const slices: AndroidTranscriptSlice[] = []; + let groupStart = -1; + let groupEnd = -1; + const flush = () => { + if (groupStart < 0) return; + pushMarkdownSlice(slices, markdown, groupStart, groupEnd, definitions); + groupStart = -1; + groupEnd = -1; + }; + for (const span of spans) { + const spanLength = emittedLength(markdown, span.start, span.end); + if (spanLength > ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + flush(); + pushMarkdownSlice(slices, markdown, span.start, span.end, definitions); + continue; + } + if (groupStart < 0) { + groupStart = span.start; + groupEnd = span.end; + continue; + } + const combined = emittedLength(markdown, groupStart, span.end); + if (combined > ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + flush(); + groupStart = span.start; + groupEnd = span.end; + continue; + } + groupEnd = span.end; + } + flush(); + return slices; +} + +/** + * Emits one Markdown slice, appending link reference definitions the slice + * does not already contain so `[text][id]` still resolves after a split. + */ +function pushMarkdownSlice( + slices: AndroidTranscriptSlice[], + markdown: string, + start: number, + end: number, + definitions: string, +): void { + const text = withLinkDefinitions(trimEdgeNewlines(markdown.slice(start, end)), definitions); + if (text.trim().length === 0) return; + slices.push({ + kind: "markdown", + key: `md:${start}`, + text, + }); +} + +/** + * Windows a fence into fixed line ranges. The first window stops changing once + * it fills, so scrolling back reuses that row instead of the rest of the file. + * An empty body is still one window so the fence header is not dropped. + */ +function sliceCodeBlock(block: CodeBlock): readonly AndroidTranscriptSlice[] { + const lines = block.body.split("\n"); + const windows: string[] = []; + for (let index = 0; index < lines.length; index += ANDROID_TRANSCRIPT_CODE_LINE_BUDGET) { + windows.push(lines.slice(index, index + ANDROID_TRANSCRIPT_CODE_LINE_BUDGET).join("\n")); + } + return windows.map((text, index) => ({ + kind: "code" as const, + key: `code:${block.start}:${index}`, + text, + language: block.language, + codePart: codeWindowPart(index, windows.length), + fullCode: block.body, + })); +} + +/** + * Names a code window so the first piece can show the header and the last + * piece can close the card. A fence that fits in one window stays "only". + */ +function codeWindowPart(index: number, count: number): AndroidTranscriptCodePart { + if (count <= 1) { + return "only"; + } + if (index === 0) { + return "start"; + } + if (index === count - 1) { + return "end"; + } + return "middle"; +} + +/** + * Splits one paragraph into ranges that each parse as their own paragraph. + * Points inside links, emphasis, inline code, or raw HTML are not used. When + * no safe point exists, the open construct stays whole. + */ +function splitPlainParagraph(text: string): readonly TextRange[] { + if (text.length <= ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + return [{ start: 0, end: text.length }]; + } + const points = inlineSafeBreaks(text); + const ranges: TextRange[] = []; + let cursor = 0; + while (cursor < text.length) { + if (text.length - cursor <= ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + ranges.push({ start: cursor, end: text.length }); + break; + } + let best = -1; + for (const point of points) { + if (point <= cursor) continue; + if (point > cursor + ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) break; + best = point; + } + const next = best === -1 ? nextSafeBreak(points, cursor, text.length) : best; + const end = next <= cursor || next > text.length ? text.length : next; + ranges.push({ start: cursor, end }); + if (end >= text.length) break; + cursor = end; + } + return ranges; +} + +/** + * First safe break after the budget, or the end of the paragraph when the + * remainder is one unbreakable construct. + */ +function nextSafeBreak(points: readonly number[], cursor: number, length: number): number { + for (const point of points) { + if (point > cursor + ANDROID_TRANSCRIPT_MARKDOWN_CHAR_BUDGET) { + return point; + } + } + return length; +} + +/** + * Whitespace indexes where the following text can start a new Markdown document + * without opening a block the original paragraph did not have. + */ +function inlineSafeBreaks(text: string): readonly number[] { + const blocked = protectedIntervals(text); + const points: number[] = []; + let index = 0; + while (index < text.length) { + if (!isInlineWhitespace(text[index] ?? "")) { + index += 1; + continue; + } + let next = index + 1; + while (next < text.length && isInlineWhitespace(text[next] ?? "")) { + next += 1; + } + if ( + next < text.length && + !insideProtected(blocked, next) && + !newlineWouldStartBlock(text, next) + ) { + points.push(next); + } + index = next; + } + return points; +} + +/** + * True when `index` sits strictly inside a link, code span, emphasis run, or tag. + * The edges themselves are safe: the construct stays entirely on one side. + */ +function insideProtected(intervals: readonly TextRange[], index: number): boolean { + for (const interval of intervals) { + if (interval.start >= index) break; + if (index > interval.start && index < interval.end) return true; + } + return false; +} + +/** + * True when a slice starting at `index` would turn a wrapped line into a list, + * quote, heading, fence, or other block. Those breaks stay with the line above. + */ +function newlineWouldStartBlock(text: string, index: number): boolean { + // A slice is its own document, so a break after a space is also a line start. + if (index === 0) return false; + const lineEnd = text.indexOf("\n", index); + const line = text.slice(index, lineEnd === -1 ? text.length : lineEnd); + if (isBlankLine(line)) return false; + if (openingFence(line) || isAtxHeading(line) || isThematicBreak(line)) return true; + if (isBlockquoteLine(line) || htmlBlockKind(line) !== null) return true; + if (listMarker(line) || isIndentedCodeLine(line)) return true; + if (!/^( {0,3})\[[^\]\n]+\]:/.test(line)) return false; + return parseLinkReferenceDefinition(text.slice(index).split("\n"), 0) !== null; +} + +/** + * Regions that must stay inside one slice. Later scanners skip earlier regions + * so a bracket inside inline code is not treated as a link. + */ +function protectedIntervals(text: string): readonly TextRange[] { + const intervals: TextRange[] = []; + collectCodeSpans(text, intervals); + collectInlineLinkDefinitions(text, intervals); + collectArtifactTemplates(text, intervals); + collectLinks(text, intervals); + collectHtmlAndAutolinks(text, intervals); + collectEmphasis(text, intervals); + intervals.sort((left, right) => left.start - right.start || left.end - right.end); + return intervals; +} + +/** + * Keeps a `::artifact-template{...}` directive in one slice. Paragraph packing + * must not cut its attributes, or the card fails to parse. + */ +function collectArtifactTemplates(text: string, intervals: TextRange[]): void { + const marker = "::artifact-template"; + let searchFrom = 0; + while (searchFrom < text.length) { + const at = text.indexOf(marker, searchFrom); + if (at === -1) return; + const previous = at === 0 ? "" : (text[at - 1] ?? ""); + if (at > 0 && !isInlineWhitespace(previous)) { + searchFrom = at + marker.length; + continue; + } + const skip = coveringEnd(intervals, at); + if (skip !== -1) { + searchFrom = skip; + continue; + } + const end = endOfArtifactTemplate(text, at + marker.length); + if (end === -1) { + searchFrom = at + marker.length; + continue; + } + intervals.push({ start: at, end }); + searchFrom = end; + } +} + +/** Offset just after the directive's closing `}`, or -1 when the braces never close. */ +function endOfArtifactTemplate(text: string, afterName: number): number { + let index = afterName; + while (index < text.length && (text[index] === " " || text[index] === "\t")) index += 1; + if (text[index] !== "{") return -1; + index += 1; + let quote: '"' | "'" | null = null; + while (index < text.length) { + const character = text[index] ?? ""; + if (quote) { + if (character === "\\") { + index += 2; + continue; + } + if (character === quote) quote = null; + index += 1; + continue; + } + if (character === '"' || character === "'") { + quote = character; + index += 1; + continue; + } + if (character === "}") return index + 1; + if (character === "\n" && text[index + 1] === "\n") return -1; + index += 1; + } + return -1; +} + +/** + * Protects a link reference definition that shares a paragraph with other text, + * including a destination or title that begins on the next line. + */ +function collectInlineLinkDefinitions(text: string, intervals: TextRange[]): void { + const lines = text.split("\n"); + const starts: number[] = []; + let offset = 0; + for (const line of lines) { + starts.push(offset); + offset += line.length + 1; + } + for (let index = 0; index < lines.length; index += 1) { + const start = starts[index] ?? 0; + if (coveringEnd(intervals, start) !== -1) continue; + const count = parseLinkReferenceDefinition(lines, index); + if (count === null) continue; + const last = index + count - 1; + const end = (starts[last] ?? start) + (lines[last]?.length ?? 0); + if (end > start) intervals.push({ start, end }); + index += count - 1; + } +} + +/** + * Records CommonMark code spans. An unclosed span protects the rest of the + * paragraph so the next slice cannot start between the backticks. + */ +function collectCodeSpans(text: string, intervals: TextRange[]): void { + let index = 0; + while (index < text.length) { + if (text[index] !== "`") { + index += 1; + continue; + } + let length = 0; + while (text[index + length] === "`") length += 1; + let cursor = index + length; + let closed = false; + while (cursor < text.length) { + if (text[cursor] !== "`") { + cursor += 1; + continue; + } + let run = 0; + while (text[cursor + run] === "`") run += 1; + if (run === length) { + intervals.push({ start: index, end: cursor + run }); + index = cursor + run; + closed = true; + break; + } + cursor += run; + } + if (!closed) { + intervals.push({ start: index, end: text.length }); + return; + } + } +} + +/** + * Records inline links and images, including the destination through the + * closing `)`. An unclosed `](` protects the tail so the URL is not cut. + */ +function collectLinks(text: string, intervals: TextRange[]): void { + let index = 0; + while (index < text.length) { + const skip = coveringEnd(intervals, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + const image = text[index] === "!" && text[index + 1] === "["; + if (image || text[index] === "[") { + const open = image ? index + 1 : index; + const end = endOfLink(text, open, intervals); + if (end > open + 1) { + intervals.push({ start: image ? index : open, end }); + index = end; + continue; + } + } + index += 1; + } +} + +/** + * End offset of the link that opens at `openIndex`, or `openIndex` when the + * brackets never close. Reference and shortcut links stop at their last `]`. + */ +function endOfLink(text: string, openIndex: number, blocked: readonly TextRange[]): number { + let depth = 1; + let index = openIndex + 1; + while (index < text.length) { + const skip = coveringEnd(blocked, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + if (text[index] === "[") depth += 1; + else if (text[index] === "]") { + depth -= 1; + if (depth === 0) break; + } + index += 1; + } + if (index >= text.length || text[index] !== "]") return openIndex; + const after = index + 1; + if (text[after] === "(") { + const destinationEnd = endOfLinkDestination(text, after); + return destinationEnd === -1 ? text.length : destinationEnd; + } + if (text[after] === "[") { + let cursor = after + 1; + while (cursor < text.length && text[cursor] !== "]" && text[cursor] !== "\n") { + cursor += 1; + } + return cursor >= text.length || text[cursor] !== "]" ? text.length : cursor + 1; + } + return after; +} + +/** + * Offset just after the `)` that closes a link destination, or -1 when the + * parenthesis never closes. Quoted titles may contain parentheses. + */ +function endOfLinkDestination(text: string, parenIndex: number): number { + let depth = 1; + let index = parenIndex + 1; + let quote: '"' | "'" | null = null; + while (index < text.length && depth > 0) { + const character = text[index]; + if (quote) { + if (character === "\\") { + index += 2; + continue; + } + if (character === quote) quote = null; + index += 1; + continue; + } + if (character === "\\") { + index += 2; + continue; + } + if ((character === '"' || character === "'") && depth === 1) { + quote = character; + index += 1; + continue; + } + if (character === "(") depth += 1; + else if (character === ")") depth -= 1; + index += 1; + } + return depth === 0 ? index : -1; +} + +/** + * Offset just after the matching close tag, or -1 when the element never closes. + * Nested elements of the same name are counted so `a b c` + * stays one span instead of ending at the inner ``. + */ +function endOfBalancedHtmlElement( + text: string, + openTagEnd: number, + name: string, + blocked: readonly TextRange[], +): number { + let depth = 1; + let cursor = openTagEnd; + while (cursor < text.length && depth > 0) { + const skip = coveringEnd(blocked, cursor); + if (skip !== -1) { + cursor = skip; + continue; + } + if (text[cursor] !== "<") { + cursor += 1; + continue; + } + const tag = /^<\/?([A-Za-z][A-Za-z0-9-]*)\b[^>\n]*\/?>/.exec(text.slice(cursor)); + if (!tag) { + cursor += 1; + continue; + } + const tagName = (tag[1] ?? "").toLowerCase(); + const isClose = text.startsWith(""); + cursor += tag[0].length; + if (tagName !== name) continue; + if (isClose) { + depth -= 1; + if (depth === 0) return cursor; + continue; + } + if (!isSelfClosing) depth += 1; + } + return -1; +} + +/** + * Records autolinks and raw HTML tags. A paired element stays together so + * `` is not left in a different slice from ``. + */ +function collectHtmlAndAutolinks(text: string, intervals: TextRange[]): void { + let index = 0; + while (index < text.length) { + const skip = coveringEnd(intervals, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + if (text[index] !== "<") { + index += 1; + continue; + } + const rest = text.slice(index); + const autolink = + /^<[A-Za-z][A-Za-z0-9+.-]*:[^>\s]+>/.exec(rest) ?? /^<[^>\s]+@[^>\s]+>/.exec(rest); + if (autolink) { + intervals.push({ start: index, end: index + autolink[0].length }); + index += autolink[0].length; + continue; + } + const tag = /^<\/?([A-Za-z][A-Za-z0-9-]*)\b[^>\n]*\/?>/.exec(rest); + if (!tag) { + index += 1; + continue; + } + const name = (tag[1] ?? "").toLowerCase(); + const closingOrEmpty = rest.startsWith(""); + if (closingOrEmpty) { + intervals.push({ start: index, end: index + tag[0].length }); + index += tag[0].length; + continue; + } + const closeEnd = endOfBalancedHtmlElement(text, index + tag[0].length, name, intervals); + const end = closeEnd === -1 ? index + tag[0].length : closeEnd; + intervals.push({ start: index, end }); + index = end; + } +} + +/** + * Records matched emphasis and strikethrough. Unmatched `*` or `_` stays + * literal and does not block a split. + */ +function collectEmphasis(text: string, intervals: TextRange[]): void { + const delimiters: EmphasisDelimiter[] = []; + let index = 0; + while (index < text.length) { + const skip = coveringEnd(intervals, index); + if (skip !== -1) { + index = skip; + continue; + } + if (text[index] === "\\") { + index += 2; + continue; + } + const character = text[index]; + if (character !== "*" && character !== "_" && character !== "~") { + index += 1; + continue; + } + let length = 0; + while (text[index + length] === character) length += 1; + if (character === "~" && length < 2) { + index += length; + continue; + } + const flanking = delimiterFlanking(character, text[index - 1], text[index + length]); + if (flanking.canOpen || flanking.canClose) { + delimiters.push({ + char: character, + pos: index, + origLen: length, + len: length, + canOpen: flanking.canOpen, + canClose: flanking.canClose, + }); + } + index += length; + } + + const stack: EmphasisDelimiter[] = []; + for (const delimiter of delimiters) { + if (delimiter.canClose) { + let stackIndex = stack.length - 1; + while (stackIndex >= 0 && delimiter.len > 0) { + const opener = stack[stackIndex]; + if ( + !opener || + opener.char !== delimiter.char || + !opener.canOpen || + opener.len === 0 || + !emphasisCanMatch(opener, delimiter) + ) { + stackIndex -= 1; + continue; + } + const use = delimiter.char === "~" ? 2 : Math.min(opener.len, delimiter.len); + if (opener.len < use || delimiter.len < use) { + stackIndex -= 1; + continue; + } + const openStart = opener.pos + opener.len - use; + const closeEnd = delimiter.pos + (delimiter.origLen - delimiter.len) + use; + intervals.push({ start: openStart, end: closeEnd }); + opener.len -= use; + delimiter.len -= use; + stack.splice(stackIndex + 1); + if (opener.len === 0) stack.splice(stackIndex, 1); + else stackIndex -= 1; + } + } + if (delimiter.len > 0 && delimiter.canOpen) stack.push(delimiter); + } +} + +/** + * CommonMark flanking rules. Underscores inside words are not emphasis. + * `before` and `after` are the characters just outside the delimiter run. + */ +function delimiterFlanking( + character: "*" | "_" | "~", + before: string | undefined, + after: string | undefined, +): { readonly canOpen: boolean; readonly canClose: boolean } { + const beforeSpace = isUnicodeSpace(before); + const afterSpace = isUnicodeSpace(after); + const beforePunctuation = !beforeSpace && isAsciiPunctuation(before); + const afterPunctuation = !afterSpace && isAsciiPunctuation(after); + const left = !afterSpace && (!afterPunctuation || beforeSpace || beforePunctuation); + const right = !beforeSpace && (!beforePunctuation || afterSpace || afterPunctuation); + if (character === "_") { + return { + canOpen: left && (!right || beforePunctuation), + canClose: right && (!left || afterPunctuation), + }; + } + return { canOpen: left, canClose: right }; +} + +/** + * Applies CommonMark's multiple-of-three rule so `***` is not paired with a + * delimiter that would leave an odd unmatched run. + */ +function emphasisCanMatch(opener: EmphasisDelimiter, closer: EmphasisDelimiter): boolean { + if (opener.char === "~") return opener.origLen >= 2 && closer.origLen >= 2; + if (!(opener.canOpen && opener.canClose && closer.canOpen && closer.canClose)) { + return true; + } + const sum = opener.origLen + closer.origLen; + return !(sum % 3 === 0 && opener.origLen % 3 !== 0 && closer.origLen % 3 !== 0); +} + +/** + * End of the protected region containing `index`, or -1 when `index` is free. + */ +function coveringEnd(intervals: readonly TextRange[], index: number): number { + let end = -1; + for (const interval of intervals) { + if (index >= interval.start && index < interval.end && interval.end > end) { + end = interval.end; + } + } + return end; +} + +/** + * Lines of `markdown` with offsets. The trailing newline belongs to the line + * so a later slice can include the break that separated two blocks. + */ +function sourceLines(markdown: string): SourceLine[] { + const lines: SourceLine[] = []; + let start = 0; + while (start < markdown.length) { + const newline = markdown.indexOf("\n", start); + if (newline === -1) { + lines.push({ text: markdown.slice(start), start, end: markdown.length }); + break; + } + lines.push({ text: markdown.slice(start, newline), start, end: newline + 1 }); + start = newline + 1; + } + return lines; +} + +/** + * One Markdown span covering `from` through `to`, inclusive of those lines. + */ +function markdownSpan(from: SourceLine, to: SourceLine, inlineSplittable: boolean): MarkdownBlock { + return { + kind: "markdown", + start: from.start, + end: to.end, + inlineSplittable, + }; +} + +/** + * Reads a fence through its closer, or through the end while a reply is still + * streaming. The body excludes the fence markers. + */ +function consumeFence( + lines: readonly SourceLine[], + start: number, + opener: FenceOpener, +): { readonly block: CodeBlock; readonly next: number } { + const body: string[] = []; + let index = start + 1; + while (index < lines.length && !isClosingFence(lines[index]?.text ?? "", opener)) { + body.push(stripFenceIndent(lines[index]?.text ?? "", opener.indent)); + index += 1; + } + if (index < lines.length) index += 1; + return { + block: { + kind: "code", + start: lines[start]?.start ?? 0, + body: body.join("\n"), + language: fenceLanguage(opener), + lineCount: body.length, + }, + next: index, + }; +} + +/** + * Reads a blockquote through its last `>` line, including lazy continuation + * that would still belong to the quote in one document. + */ +function consumeBlockquote( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isBlockquoteLine(line)) { + index += 1; + continue; + } + if (isBlankLine(line)) { + const next = nextNonBlank(lines, index + 1); + if (next !== null && isBlockquoteLine(lines[next]?.text ?? "")) { + index += 1; + continue; + } + break; + } + const following = lines[index + 1]?.text ?? null; + if (interruptsParagraph(line, following)) break; + index += 1; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Reads one top-level list as a sequence of items. Items stay whole, including + * nested lists, indented continuation, and lazy lines, and may be packed later. + */ +function consumeList( + lines: readonly SourceLine[], + start: number, +): { readonly blocks: readonly MarkdownBlock[]; readonly next: number } { + const base = listMarker(lines[start]?.text ?? ""); + if (!base) return { blocks: [], next: start + 1 }; + const blocks: MarkdownBlock[] = []; + let index = start; + while (index < lines.length) { + while (index < lines.length && isBlankLine(lines[index]?.text ?? "")) { + const next = nextNonBlank(lines, index + 1); + const nextMarker = next === null ? null : listMarker(lines[next]?.text ?? ""); + if (nextMarker && nextMarker.indent === base.indent) { + index += 1; + continue; + } + return { blocks, next: index }; + } + const marker = listMarker(lines[index]?.text ?? ""); + if (!marker || marker.indent !== base.indent) break; + const itemStart = index; + index += 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isBlankLine(line)) { + const next = nextNonBlank(lines, index + 1); + if (next !== null && leadingIndent(lines[next]?.text ?? "") > base.indent) { + index += 1; + continue; + } + break; + } + if (leadingIndent(line) > base.indent) { + index += 1; + continue; + } + // Same-indent markers start the next item. Other lines that would not + // interrupt a paragraph are lazy continuation and stay with this item. + const sibling = listMarker(line); + if (sibling && sibling.indent <= base.indent) break; + const following = lines[index + 1]?.text ?? null; + if (interruptsParagraph(line, following)) break; + index += 1; + } + const last = lines[index - 1] ?? lines[itemStart]; + const first = lines[itemStart] ?? last; + if (first && last) blocks.push(markdownSpan(first, last, false)); + } + return { blocks, next: index }; +} + +/** + * Drops up to `indent` leading spaces. CommonMark removes the opening fence's + * indentation from each content line, and leaves a shorter indent empty. + */ +function stripFenceIndent(line: string, indent: number): string { + if (indent <= 0) return line; + let index = 0; + while (index < line.length && index < indent && line[index] === " ") { + index += 1; + } + return line.slice(index); +} + +/** + * Reads one HTML block. Textarea, pre, script, and style run through their + * closing tag, including blank lines inside. Other block tags stop at the + * closer or at the next blank line. + */ +function consumeHtmlBlock( + lines: readonly SourceLine[], + start: number, + kind: HtmlBlockKind, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start; + if (kind.kind === "comment") { + index = consumeUntilIncludes(lines, start, "-->"); + } else if (kind.kind === "processing") { + index = consumeUntilIncludes(lines, start, "?>"); + } else if (kind.kind === "declaration") { + index = consumeUntilIncludes(lines, start, ">"); + } else if (kind.kind === "pre") { + index = consumeUntilIncludes(lines, start, ``); + } else { + const close = ``; + if ((lines[start]?.text ?? "").toLowerCase().includes(close)) { + index = start + 1; + } else { + index = start + 1; + while (index < lines.length && !isBlankLine(lines[index]?.text ?? "")) { + const includes = (lines[index]?.text ?? "").toLowerCase().includes(close); + index += 1; + if (includes) break; + } + } + } + const last = lines[Math.max(start, index - 1)] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Advances past the line that contains `needle`, or to the end of the input. + */ +function consumeUntilIncludes(lines: readonly SourceLine[], start: number, needle: string): number { + let index = start; + while (index < lines.length) { + const includes = (lines[index]?.text ?? "").toLowerCase().includes(needle.toLowerCase()); + index += 1; + if (includes) break; + } + return index; +} + +/** + * Reads a GFM table from its header through the last pipe row. The delimiter + * row stays with the header so the table still parses. + */ +function consumeTable( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 2; + while ( + index < lines.length && + !isBlankLine(lines[index]?.text ?? "") && + (lines[index]?.text ?? "").includes("|") + ) { + index += 1; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Reads an indented code block. It stays Markdown, not a fenced window, so the + * shared renderer can still show it as code. + */ +function consumeIndentedCode( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isIndentedCodeLine(line)) { + index += 1; + continue; + } + if (isBlankLine(line)) { + const next = nextNonBlank(lines, index + 1); + if (next !== null && isIndentedCodeLine(lines[next]?.text ?? "")) { + index += 1; + continue; + } + } + break; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; +} + +/** + * Reads a paragraph. A setext underline is kept with its text line. A following + * block that would interrupt the paragraph starts the next span instead. + */ +function consumeParagraph( + lines: readonly SourceLine[], + start: number, +): { readonly block: MarkdownBlock; readonly next: number } { + let index = start + 1; + while (index < lines.length) { + const line = lines[index]?.text ?? ""; + if (isBlankLine(line)) break; + // Setext wins over a thematic break. The underline stays on the heading + // so a later slice cannot turn `---` into a horizontal rule. + if (isSetextUnderline(line)) { + index += 1; + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, false), + next: index, + }; + } + const following = lines[index + 1]?.text ?? null; + if (interruptsParagraph(line, following)) break; + index += 1; + } + const last = lines[index - 1] ?? lines[start]; + return { + block: markdownSpan(lines[start] ?? last!, last!, true), + next: index, + }; +} + +/** + * True when `line` starts a new block in the middle of a paragraph. Ordered + * lists other than `1` do not interrupt, matching CommonMark. + */ +function interruptsParagraph(line: string, nextLine: string | null): boolean { + if (openingFence(line) || isAtxHeading(line) || isThematicBreak(line)) return true; + if (isBlockquoteLine(line) || htmlBlockKind(line) !== null) return true; + if (nextLine !== null && isTableStart(line, nextLine)) return true; + const marker = listMarker(line); + if (!marker) return false; + return !marker.ordered || marker.number === 1; +} + +/** + * Link reference definitions in the message. Copies are appended to slices that + * do not already hold them, so a reference link stays clickable after a split. + * Lines inside fences are skipped. + */ +function collectLinkReferenceDefinitions(markdown: string): string { + const lines = sourceLines(markdown); + const texts = lines.map((line) => line.text); + const definitions: string[] = []; + let fence: FenceOpener | null = null; + for (let index = 0; index < texts.length; index += 1) { + const line = texts[index] ?? ""; + if (fence) { + if (isClosingFence(line, fence)) fence = null; + continue; + } + const opener = openingFence(line); + if (opener) { + fence = opener; + continue; + } + const count = parseLinkReferenceDefinition(texts, index); + if (count === null) continue; + definitions.push(texts.slice(index, index + count).join("\n")); + index += count - 1; + } + return definitions.join("\n"); +} + +/** + * Appends `definitions` when the slice does not already contain them. + * Definitions are not rendered, so the extra lines do not show up in the row. + */ +function withLinkDefinitions(text: string, definitions: string): string { + if (definitions.length === 0 || text.includes(definitions)) return text; + return `${text}\n\n${definitions}`; +} + +/** + * Opening fence (`\`\`\`` or `~~~`) with at most three spaces of indent. + * A backtick fence whose info string contains a backtick is prose, not a fence. + */ +function openingFence(line: string): FenceOpener | null { + const match = /^( {0,3})(`{3,}|~{3,})(.*)$/.exec(line); + if (!match) return null; + const marker = match[2] ?? ""; + const info = match[3] ?? ""; + const character = marker[0]; + if (character !== "`" && character !== "~") return null; + if (character === "`" && info.includes("`")) return null; + return { char: character, length: marker.length, info, indent: (match[1] ?? "").length }; +} + +/** + * True when `line` closes `opener`. The closing line is only the marker, and + * it must be at least as long as the opener of the same character. + */ +function isClosingFence(line: string, opener: FenceOpener): boolean { + const match = /^( {0,3})(`{3,}|~{3,})[ \t]*$/.exec(line); + const marker = match?.[2] ?? ""; + return marker.length >= opener.length && marker[0] === opener.char; +} + +/** + * Language info word from an opening fence. Empty info stays null so the + * header can fall back to a generic code label. + */ +function fenceLanguage(opener: FenceOpener): string | null { + const info = opener.info.trim(); + if (info.length === 0) return null; + const word = info.split(/[ \t]+/)[0] ?? ""; + if (word.length === 0 || word.includes(opener.char)) return null; + return word; +} + +/** + * Top-level list marker, or null for thematic breaks and ordinary prose. + * The marker's indent is how continuation lines are recognized. + */ +function listMarker(line: string): ListMarkerInfo | null { + if (isThematicBreak(line)) return null; + const match = /^( {0,3})([-+*]|\d{1,9}[.)])(?:[ \t]+(.*)|[ \t]*)$/.exec(line); + if (!match) return null; + const marker = match[2] ?? ""; + const ordered = /^\d/.test(marker); + return { + indent: (match[1] ?? "").length, + ordered, + number: ordered ? Number.parseInt(marker, 10) : null, + }; +} + +/** + * Count of leading spaces, with a tab counted as four. Blank lines are not + * continuation; callers handle those separately. + */ +function leadingIndent(line: string): number { + let count = 0; + for (const character of line) { + if (character === " ") count += 1; + else if (character === "\t") count += 4; + else break; + } + return count; +} + +/** + * True for a GFM header row followed by a delimiter row. + */ +function isTableStart(line: string, nextLine: string): boolean { + return line.includes("|") && isTableDelimiter(nextLine); +} + +/** + * True for a GFM delimiter row of one or more dashed cells. + */ +function isTableDelimiter(line: string): boolean { + const trimmed = line.trim(); + if (!trimmed.includes("-")) return false; + const cells = trimmed.replace(/^\|/, "").replace(/\|$/, "").split("|"); + return cells.length > 0 && cells.every((cell) => /^\s*:?-{3,}:?\s*$/.test(cell)); +} + +/** + * True for an ATX heading. A `#` glued to the next word is not a heading. + */ +function isAtxHeading(line: string): boolean { + return /^( {0,3})#{1,6}(?:[ \t]+.*|[ \t]*)$/.test(line); +} + +/** + * True for a setext underline. Used on the lines after heading text, where it + * is a heading rather than a thematic break. + */ +function isSetextUnderline(line: string): boolean { + return /^( {0,3})(?:=+|-+)[ \t]*$/.test(line); +} + +/** + * True for a thematic break of three or more `-`, `*`, or `_`. + */ +function isThematicBreak(line: string): boolean { + return /^( {0,3})([-*_])(?:\s*\2){2,}\s*$/.test(line); +} + +/** + * True when `line` opens or continues a blockquote. + */ +function isBlockquoteLine(line: string): boolean { + return /^( {0,3})>/.test(line); +} + +/** + * True for a four-space indented code line. Fence detection allows only three. + */ +function isIndentedCodeLine(line: string): boolean { + return /^(?: {4}|\t)\S/.test(line); +} + +/** + * Lines consumed by a link reference definition starting at `start`, or null. + * The destination and title may each sit on the following line (`[docs]:` / + * ` https://example.com`). A line that only looks like a label stays prose. + */ +function parseLinkReferenceDefinition(lines: readonly string[], start: number): number | null { + if (start < 0 || start >= lines.length) return null; + + let lineIndex = start; + let column = 0; + const peek = (): string => { + const line = lines[lineIndex] ?? ""; + if (column < line.length) return line[column] ?? ""; + return lineIndex + 1 < lines.length ? "\n" : ""; + }; + const advance = (): string => { + const line = lines[lineIndex] ?? ""; + if (column < line.length) { + const character = line[column] ?? ""; + column += 1; + return character; + } + if (lineIndex + 1 < lines.length) { + lineIndex += 1; + column = 0; + return "\n"; + } + return ""; + }; + const skipSpaces = () => { + while (peek() === " " || peek() === "\t") advance(); + }; + + let indent = 0; + while (peek() === " " && indent < 3) { + advance(); + indent += 1; + } + if (peek() !== "[") return null; + advance(); + + let labelLength = 0; + while (labelLength <= 999) { + const character = peek(); + if (character === "" || character === "\n" || character === "[") return null; + if (character === "\\") { + advance(); + if (peek() === "" || peek() === "\n") return null; + advance(); + labelLength += 1; + continue; + } + if (character === "]") break; + advance(); + labelLength += 1; + } + if (labelLength === 0 || labelLength > 999 || peek() !== "]") return null; + advance(); + if (peek() !== ":") return null; + advance(); + + skipSpaces(); + if (peek() === "\n") { + advance(); + skipSpaces(); + } + if (peek() === "" || peek() === "\n") return null; + if (!readLinkDestination(peek, advance)) return null; + + skipSpaces(); + if (peek() === "\n") { + const nextLine = (lines[lineIndex + 1] ?? "").trimStart(); + const opener = nextLine[0]; + if (opener === '"' || opener === "'" || opener === "(") { + advance(); + skipSpaces(); + } + } + const titleOpener = peek(); + if (titleOpener === '"' || titleOpener === "'" || titleOpener === "(") { + if (!readLinkTitle(titleOpener, peek, advance, lines, () => lineIndex)) return null; + skipSpaces(); + } + if (peek() !== "" && peek() !== "\n") return null; + return lineIndex - start + 1; +} + +/** Reads an angle or bare link destination. The cursor sits on its first character. */ +function readLinkDestination(peek: () => string, advance: () => string): boolean { + if (peek() === "<") { + advance(); + while (true) { + const character = peek(); + if (character === "" || character === "\n" || character === " " || character === "<") { + return false; + } + if (character === "\\") { + advance(); + if (peek() === "" || peek() === "\n") return false; + advance(); + continue; + } + if (character === ">") { + advance(); + return true; + } + advance(); + } + } + + let length = 0; + let depth = 0; + while (true) { + const character = peek(); + if (character === "" || character === "\n" || character === " " || character === "\t") break; + if (character.charCodeAt(0) < 32) return false; + if (character === "\\") { + advance(); + const escaped = peek(); + if (escaped === "" || escaped === "\n" || escaped === " " || escaped === "\t") return false; + advance(); + length += 1; + continue; + } + if (character === "(") { + depth += 1; + advance(); + length += 1; + continue; + } + if (character === ")") { + if (depth === 0) break; + depth -= 1; + advance(); + length += 1; + continue; + } + advance(); + length += 1; + } + return length > 0 && depth === 0; +} + +/** Reads a quoted title that may continue onto later non-blank lines. */ +function readLinkTitle( + opener: string, + peek: () => string, + advance: () => string, + lines: readonly string[], + lineIndex: () => number, +): boolean { + const closer = opener === "(" ? ")" : opener; + advance(); + while (true) { + const character = peek(); + if (character === "") return false; + if (character === "\n") { + const next = lines[lineIndex() + 1]; + if (next === undefined || next.trim().length === 0) return false; + advance(); + continue; + } + if (character === "\\") { + advance(); + if (peek() === "" || peek() === "\n") return false; + advance(); + continue; + } + if (character === closer) { + advance(); + return true; + } + advance(); + } +} + +/** + * Classifies a CommonMark HTML block opener, or null for inline tags. + */ +function htmlBlockKind(line: string): HtmlBlockKind | null { + if (!/^( {0,3})?@[\\\]^_`{|}~]/.test(character); +} + +/** + * Length of the slice text after edge newlines are removed. That is the text + * the row actually mounts. + */ +function emittedLength(markdown: string, start: number, end: number): number { + return trimEdgeNewlines(markdown.slice(start, end)).length; +} + +/** + * Removes blank lines from the edges of a slice without stripping the indent + * a list item or indented code block needs. + */ +function trimEdgeNewlines(text: string): string { + let start = 0; + let end = text.length; + while (start < end && text[start] === "\n") start += 1; + while (end > start && text[end - 1] === "\n") end -= 1; + return text.slice(start, end); +}