diff --git a/apps/desktop/electron/main/index.ts b/apps/desktop/electron/main/index.ts index 6c1db01290..54592f13b9 100644 --- a/apps/desktop/electron/main/index.ts +++ b/apps/desktop/electron/main/index.ts @@ -52,7 +52,7 @@ import { import { PersistenceOutbox } from "./persistence-outbox"; import { AgentSidecar } from "./agent-sidecar"; import { Logger, ignoreBrokenStdio } from "./logger"; -import { installMainProcessErrorHandlers } from "./main-process-errors"; +import { describeError, installMainProcessErrorHandlers } from "./main-process-errors"; import { isDbSchemaTooNewError, } from "./host-boot-diagnostics"; @@ -791,12 +791,6 @@ function setCurrentWorkspacePath(path: string | null): void { } } -/** One-line message for an error of unknown shape, for user-facing lists. */ -function describeError(error: unknown): string { - if (error instanceof Error) return error.message.slice(0, 300); - return String(error).slice(0, 300); -} - /** * True when a rejection only says the host transport is gone (D080): the call * lost a race with shutdown, a crash, or a supervised restart. Every such diff --git a/apps/desktop/electron/main/main-process-errors.ts b/apps/desktop/electron/main/main-process-errors.ts index e2822be696..1d6d53457d 100644 --- a/apps/desktop/electron/main/main-process-errors.ts +++ b/apps/desktop/electron/main/main-process-errors.ts @@ -46,6 +46,12 @@ export function isNonAsciiHttpHeaderError(error: unknown): boolean { ); } +/** One-line message for an error of unknown shape, for user-facing lists. */ +export function describeError(error: unknown): string { + if (error instanceof Error) return error.message.slice(0, 300); + return String(error).slice(0, 300); +} + export function describeMainProcessError(error: unknown): string { if (error instanceof Error) { return error.stack ?? `${error.name}: ${error.message}`; diff --git a/apps/desktop/src/components/settings/RemoteHostsPage.tsx b/apps/desktop/src/components/settings/RemoteHostsPage.tsx index c91ad1ab7f..37887b84aa 100644 --- a/apps/desktop/src/components/settings/RemoteHostsPage.tsx +++ b/apps/desktop/src/components/settings/RemoteHostsPage.tsx @@ -245,8 +245,8 @@ export function RemoteHostsPage() { value={addMode} onChange={(mode) => setAddMode(mode)} options={[ - { value: "ssh", label: t("settings.remoteHosts.addSsh") }, - { value: "pair", label: t("settings.remoteHosts.addPair") }, + { value: "ssh", id: "remote-host-add-ssh", controls: "remote-host-add-panel-ssh", label: t("settings.remoteHosts.addSsh") }, + { value: "pair", id: "remote-host-add-pair", controls: "remote-host-add-panel-pair", label: t("settings.remoteHosts.addPair") }, ]} label={t("settings.remoteHosts.addTitle")} role="tablist" diff --git a/apps/desktop/src/components/ui.tsx b/apps/desktop/src/components/ui.tsx index 260d8059c3..3a5c4a9301 100644 --- a/apps/desktop/src/components/ui.tsx +++ b/apps/desktop/src/components/ui.tsx @@ -695,7 +695,12 @@ export function SegmentedControl({ }: { value: T; onChange: (value: T) => void; - options: readonly { readonly value: T; readonly label: ReactNode }[]; + options: readonly { + readonly value: T; + readonly label: ReactNode; + readonly id?: string; + readonly controls?: string; + }[]; label: string; role?: "group" | "radiogroup" | "tablist"; className?: string; @@ -714,7 +719,12 @@ export function SegmentedControl({ key={option.value} type="button" {...(itemRole === "tab" - ? { role: "tab", id: `${label}-tab-${option.value}`, "aria-selected": value === option.value } + ? { + role: "tab", + id: option.id ?? `${label}-tab-${option.value}`, + "aria-selected": value === option.value, + "aria-controls": option.controls, + } : itemRole === "radio" ? { role: "radio", "aria-checked": value === option.value } : { "aria-pressed": value === option.value })} diff --git a/apps/desktop/src/features/chat/transcript/AssistantTurn.tsx b/apps/desktop/src/features/chat/transcript/AssistantTurn.tsx index 6b1fae5584..95eadbabb4 100644 --- a/apps/desktop/src/features/chat/transcript/AssistantTurn.tsx +++ b/apps/desktop/src/features/chat/transcript/AssistantTurn.tsx @@ -34,6 +34,7 @@ import { shouldGroupTurnProcess, } from "../../../lib/turn-process"; import { useAppStore } from "../../../stores/app-store"; +import { latestGenerationMessage } from "../../../lib/live-throughput"; import { Markdown } from "../../../components/Markdown"; import { IconBranch, IconReview } from "../../../components/icons"; import { TooltipButton } from "../../../components/ui"; @@ -41,6 +42,7 @@ import { AssistantErrorMessage, CopyButton, MessageMeta, + LiveMessageMeta, MessageTimestamp, } from "./shared"; import { activityItemsEqual, ActivityGroup } from "./ActivityGroup"; @@ -295,6 +297,9 @@ export const AssistantTurn = memo(function AssistantTurn({ const responseDurationMs = assistantTurnResponseDuration(entry); const responseOutputTokens = assistantTurnResponseOutputTokens(entry); const modelId = metaMessage?.modelId ?? latestUsageMessage?.modelId; + // The tail message is the one still growing; the live rate is estimated from + // it because the provider only reports usage at message_end. + const latestMessage = latestGenerationMessage(entry); const hasError = messages.some((message) => Boolean(message.error)); const complete = !isActive && !hasError && Boolean(content) && Boolean(actionMessage); @@ -413,12 +418,23 @@ export const AssistantTurn = memo(function AssistantTurn({ {turnAllActivityItems.filter((item) => item.kind === "tool" && item.message.toolName === "GenerateImages").map((item) => ( ))} - {!isActive && metaMessage ? ( + {!isActive && (metaMessage || latestMessage?.timeToFirstTokenMs !== undefined) ? ( + ) : null} + {isActive ? ( + item.kind === "tool" && item.message.toolStatus === "running", + )} /> ) : null} {complete && actionMessage ? ( diff --git a/apps/desktop/src/features/chat/transcript/FirstOutputLatency.tsx b/apps/desktop/src/features/chat/transcript/FirstOutputLatency.tsx new file mode 100644 index 0000000000..bc41205c72 --- /dev/null +++ b/apps/desktop/src/features/chat/transcript/FirstOutputLatency.tsx @@ -0,0 +1,14 @@ +import { useTranslation } from "react-i18next"; + +/** A measured request latency; absent on old records and before the first output. */ +export function FirstOutputLatency({ milliseconds }: { milliseconds?: number }) { + const { t } = useTranslation(); + if (milliseconds === undefined || !Number.isFinite(milliseconds) || milliseconds < 0) { + return null; + } + return ( + + {t("chat.firstOutputLatency", { seconds: (milliseconds / 1000).toFixed(1) })} + + ); +} diff --git a/apps/desktop/src/features/chat/transcript/shared.tsx b/apps/desktop/src/features/chat/transcript/shared.tsx index e3aac58539..56b233c793 100644 --- a/apps/desktop/src/features/chat/transcript/shared.tsx +++ b/apps/desktop/src/features/chat/transcript/shared.tsx @@ -1,3 +1,4 @@ +import { FirstOutputLatency } from "./FirstOutputLatency"; import { memo, useCallback, @@ -19,6 +20,9 @@ import { type ThinkingLevel, } from "@pi-desktop/shared"; import { useOpenChatFileRef, useOpenPreviewTarget } from "../../../hooks/use-preview-target"; +import { generationPhase, type ThroughputMessage } from "../../../lib/live-throughput"; +import { useLiveThroughput } from "./use-live-throughput"; + import { useDisclosureAnchorNotifier } from "../../../lib/disclosure-anchor-context"; import { isThinkingActive, resolveThinkingDisplayMode } from "../../../lib/turn-process"; import { TranscriptSearchContext } from "../../../lib/transcript-search-context"; @@ -155,11 +159,13 @@ export function MessageMeta({ usage, responseDurationMs, responseOutputTokens, + timeToFirstTokenMs, }: { modelId?: string; usage?: MessageUsage; responseDurationMs?: number; responseOutputTokens?: number; + timeToFirstTokenMs?: number; }) { const { t } = useTranslation(); const throughput = calculateTokenRate( @@ -167,7 +173,7 @@ export function MessageMeta({ responseDurationMs, ); const showThroughput = !usage && throughput !== undefined; - if (!modelId && !showThroughput) { + if (!modelId && !showThroughput && timeToFirstTokenMs === undefined) { return null; } return ( @@ -177,6 +183,7 @@ export function MessageMeta({ {modelId} ) : null} + {showThroughput ? ( {t("chat.usageThroughputEstimated", { @@ -188,6 +195,61 @@ export function MessageMeta({ ); } +/** + * Meta row for the turn that is still streaming. + * + * Mounted only for the active tail turn, which keeps the sampler and its + * interval off every history row and keeps per-token work out of the store. + * The figure is always an estimate — the provider reports usage once, at + * `message_end` — so it carries the "≈" copy (ADR 0073 §4), and `MessageMeta` + * takes over with the exact value once the turn completes. + */ +export function LiveMessageMeta({ + modelId, + message, + toolRunning = false, +}: { + modelId?: string; + message?: ThroughputMessage; + toolRunning?: boolean; +}) { + const { t } = useTranslation(); + const phase = generationPhase(message, toolRunning); + const generating = phase === "thinking" || phase === "generating"; + const { rate, stale } = useLiveThroughput(message, generating); + const phaseLabel = t({ + waiting: "chat.liveWaiting", + thinking: "chat.thinking", + generating: "chat.liveGenerating", + tool: "chat.liveToolRunning", + }[phase]); + const rateLabel = rate === undefined ? undefined : t("chat.usageThroughputEstimated", { + count: Math.round(rate), + }); + return ( +
+ {modelId ? ( + + {modelId} + + ) : null} + + + {phaseLabel} + + {rate === undefined ? null : ( + + {stale ? t("chat.liveLastRate", { rate: rateLabel }) : rateLabel} + + )} +
+ ); +} + export function AssistantErrorMessage({ message }: { message: UiMessage }) { const { t } = useTranslation(); const [open, setOpen] = useState(true); diff --git a/apps/desktop/src/features/chat/transcript/use-live-throughput.ts b/apps/desktop/src/features/chat/transcript/use-live-throughput.ts new file mode 100644 index 0000000000..844c3abdab --- /dev/null +++ b/apps/desktop/src/features/chat/transcript/use-live-throughput.ts @@ -0,0 +1,46 @@ +import { useEffect, useRef, useState } from "react"; +import { + LIVE_THROUGHPUT_SAMPLE_MS, + advanceThroughput, + type LiveTokenRate, + type ThroughputMessage, + type ThroughputTracker, +} from "../../../lib/live-throughput"; + +/** Only the active turn mounts this sampler. Committed props feed a fixed cadence. */ +export function useLiveThroughput( + message: ThroughputMessage | undefined, + generating: boolean, +): LiveTokenRate { + const inputRef = useRef({ message, generating }); + const trackerRef = useRef({ samples: [] }); + const [view, setView] = useState({ stale: false }); + + useEffect(() => { + inputRef.current = { message, generating }; + }, [message, generating]); + + useEffect(() => { + const sample = () => { + const input = inputRef.current; + const next = advanceThroughput( + trackerRef.current, input.message, input.generating, performance.now(), + ); + trackerRef.current = next.tracker; + setView((previous) => + previous.rate === next.view.rate && previous.stale === next.view.stale && + previous.messageId === input.message?.id + ? previous + : { ...next.view, messageId: input.message?.id }, + ); + }; + sample(); + const timer = window.setInterval(sample, LIVE_THROUGHPUT_SAMPLE_MS); + return () => window.clearInterval(timer); + }, []); + + return { + rate: view.rate, + stale: view.stale || !generating || view.messageId !== message?.id, + }; +} diff --git a/apps/desktop/src/features/settings/import-page.tsx b/apps/desktop/src/features/settings/import-page.tsx index aecf4cdcb5..4a13ef9721 100644 --- a/apps/desktop/src/features/settings/import-page.tsx +++ b/apps/desktop/src/features/settings/import-page.tsx @@ -61,6 +61,8 @@ export function ImportSection() { onChange={(value) => setKind(value)} options={IMPORT_KINDS.map((entry) => ({ value: entry.id, + id: `import-tab-${entry.id}`, + controls: `import-panel-${entry.id}`, label: t(entry.labelKey), }))} label={t("settings.import")} diff --git a/apps/desktop/src/lib/live-throughput.ts b/apps/desktop/src/lib/live-throughput.ts new file mode 100644 index 0000000000..14c68cc86b --- /dev/null +++ b/apps/desktop/src/lib/live-throughput.ts @@ -0,0 +1,166 @@ +import { calculateTokenRate, estimateResponseOutputTokens } from "./context-usage"; +import type { AssistantTurnEntry } from "./assistant-turns"; +import type { UiMessage } from "@pi-desktop/shared"; + +/** Renderer estimate of visible output; provider usage arrives at message_end. */ +/** Span the rate is measured over. */ +export const LIVE_THROUGHPUT_WINDOW_MS = 3_000; +/** Minimum spacing between samples; the estimate walks the whole message. */ +export const LIVE_THROUGHPUT_SAMPLE_MS = 250; +/** Silence past this point dims the figure instead of replacing it. */ +export const LIVE_THROUGHPUT_STALE_MS = 1_500; +/** Below this span the sample set is too thin to put a number on screen. */ +export const LIVE_THROUGHPUT_MIN_SPAN_MS = 600; +/** Backstop on the ring; the window normally bounds it well below this. */ +export const LIVE_THROUGHPUT_MAX_SAMPLES = 120; + +export type ThroughputSample = { + ts: number; + /** Cumulative estimated output tokens for the message, not a delta. */ + tokens: number; +}; + +export type LiveTokenRate = { + /** Absent until the samples span `LIVE_THROUGHPUT_MIN_SPAN_MS` and grow. */ + rate?: number; + /** True during tool/waiting phases or after output stops for the stale interval. */ + stale: boolean; +}; + +/** Uses ADR 0073’s visible thinking/text estimate, shared with stopped turns. */ +export function sampleTokensForMessage( + message: Pick | undefined, +): number { + if (!message) return 0; + return estimateResponseOutputTokens(message) ?? 0; +} + +/** Prunes the window, retaining one baseline when every prior sample is older. */ +export function pushThroughputSample( + samples: readonly ThroughputSample[], + sample: ThroughputSample, +): ThroughputSample[] { + const cutoff = sample.ts - LIVE_THROUGHPUT_WINDOW_MS; + const firstInWindow = samples.findIndex((entry) => entry.ts >= cutoff); + // Keep the newest pre-window sample only when the window holds nothing else. + const start = firstInWindow === -1 ? Math.max(0, samples.length - 1) : firstInWindow; + const next = [...samples.slice(start), sample]; + return next.length > LIVE_THROUGHPUT_MAX_SAMPLES + ? next.slice(next.length - LIVE_THROUGHPUT_MAX_SAMPLES) + : next; +} + +/** Only growth updates the rate; idle ticks must not dilute the retained value. */ +export function sampleDidGrow(samples: readonly ThroughputSample[]): boolean { + if (samples.length < 2) return false; + const newest = samples[samples.length - 1]; + const previous = samples[samples.length - 2]; + return newest.tokens > previous.tokens; +} + +/** Endpoint deltas keep irregular sample spacing from biasing the window rate. */ +export function windowedTokenRate( + samples: readonly ThroughputSample[], +): number | undefined { + const newest = samples.at(-1); + if (!newest) return undefined; + const cutoff = newest.ts - LIVE_THROUGHPUT_WINDOW_MS; + const oldest = samples.find((entry) => entry.ts >= cutoff) ?? samples[0]; + const spanMs = newest.ts - oldest.ts; + if (spanMs < LIVE_THROUGHPUT_MIN_SPAN_MS) return undefined; + return calculateTokenRate(newest.tokens - oldest.tokens, spanMs); +} + +/** Retains the last measured rate through silence and dims it after the threshold. */ +export function retainLiveRate( + fresh: number | undefined, + remembered: { rate: number; at: number } | undefined, + now: number, +): LiveTokenRate { + if (fresh !== undefined) return { rate: fresh, stale: false }; + if (!remembered) return { stale: false }; + return { + rate: remembered.rate, + stale: now - remembered.at > LIVE_THROUGHPUT_STALE_MS, + }; +} + +export type GenerationPhase = "waiting" | "thinking" | "generating" | "tool"; +export type ThroughputMessage = Pick; + +export function generationPhase( + message: ThroughputMessage | undefined, + toolRunning: boolean, +): GenerationPhase { + if (toolRunning) return "tool"; + if (message?.status !== "streaming") return "waiting"; + if (message.content) return "generating"; + if (message.thinking) return "thinking"; + return "waiting"; +} + +/** Time-based smoothing gives the same response at different sampling cadences. */ +export function smoothTokenRate( + previous: number | undefined, + next: number, + elapsedMs: number, +): number { + if (previous === undefined) return next; + const weight = 1 - Math.exp(-Math.max(0, elapsedMs) / 750); + return previous + weight * (next - previous); +} + +export type ThroughputTracker = { + messageId?: string; + samples: ThroughputSample[]; + remembered?: { rate: number; at: number }; + smoothed?: { rate: number; at: number }; +}; + +/** A new message or resumed generation starts a new window; history is display-only. */ +export function advanceThroughput( + previous: ThroughputTracker, + message: ThroughputMessage | undefined, + generating: boolean, + now: number, +): { tracker: ThroughputTracker; view: LiveTokenRate } { + let tracker = { ...previous }; + if (message?.id !== tracker.messageId || !generating) { + tracker = { messageId: message?.id, samples: [], remembered: tracker.remembered }; + } + let fresh: number | undefined; + if (generating) { + const tokens = sampleTokensForMessage(message); + const last = tracker.samples.at(-1); + // A replaced/truncated message cannot share a baseline with the old text. + if (last && (tokens < last.tokens || now < last.ts)) { + tracker.samples = []; + tracker.smoothed = undefined; + } + tracker.samples = pushThroughputSample(tracker.samples, { ts: now, tokens }); + const raw = sampleDidGrow(tracker.samples) ? windowedTokenRate(tracker.samples) : undefined; + if (raw !== undefined) { + fresh = smoothTokenRate( + tracker.smoothed?.rate, raw, now - (tracker.smoothed?.at ?? now), + ); + tracker.smoothed = { rate: fresh, at: now }; + tracker.remembered = { rate: fresh, at: now }; + } + } + const view = retainLiveRate(fresh, tracker.remembered, now); + // Outside generation the remembered number is explicitly historical immediately. + if ((!generating || !tracker.smoothed) && view.rate !== undefined) view.stale = true; + return { tracker, view }; +} + +/** Thinking-only messages live in activity parts, before an answer row exists. */ +export function latestGenerationMessage(entry: AssistantTurnEntry): UiMessage | undefined { + for (let index = entry.parts.length - 1; index >= 0; index--) { + const part = entry.parts[index]; + if (part.kind === "message") return part.message; + for (let item = part.items.length - 1; item >= 0; item--) { + if (part.items[item].kind === "thinking") return part.items[item].message; + } + } + return undefined; +} diff --git a/apps/desktop/src/styles/messages.css b/apps/desktop/src/styles/messages.css index 8323949b6f..c4f72625cf 100644 --- a/apps/desktop/src/styles/messages.css +++ b/apps/desktop/src/styles/messages.css @@ -1105,6 +1105,18 @@ text-overflow: ellipsis; } +/* The live figure changes while streaming, so proportional digits would make + the chip's width twitch on every update. */ +.message-meta-chip.throughput { + font-variant-numeric: tabular-nums; +} + +/* Tool execution generates no tokens. The last measured rate is kept and dimmed + rather than blanked, so a long tool call does not read as a stalled model. */ +.message-meta-chip.throughput[data-stale="true"] { + color: var(--ds-text-muted); +} + .context-inspector { position: relative; display: inline-flex; diff --git a/apps/desktop/src/styles/sidebar-threads.css b/apps/desktop/src/styles/sidebar-threads.css index 42e5646bb6..2e7a3c0213 100644 --- a/apps/desktop/src/styles/sidebar-threads.css +++ b/apps/desktop/src/styles/sidebar-threads.css @@ -331,5 +331,3 @@ .thread-item.selected.active { background: var(--ds-bg-active); } - - diff --git a/apps/desktop/src/styles/ui-kit.css b/apps/desktop/src/styles/ui-kit.css index 4dd24eca62..ed2899906b 100644 --- a/apps/desktop/src/styles/ui-kit.css +++ b/apps/desktop/src/styles/ui-kit.css @@ -346,7 +346,7 @@ width: 16px; height: 16px; flex: 0 0 auto; - border-radius: 4px; + border-radius: var(--radius-3xs); border: 1.5px solid var(--ds-text-secondary); background: transparent; cursor: pointer; diff --git a/apps/desktop/src/styles/voice.css b/apps/desktop/src/styles/voice.css index c4e4cafd76..b1d99c00c7 100644 --- a/apps/desktop/src/styles/voice.css +++ b/apps/desktop/src/styles/voice.css @@ -112,5 +112,3 @@ .voice-settings .settings-row-control { max-width: none; } - - diff --git a/apps/desktop/test/agent-capability-settings.test.mjs b/apps/desktop/test/agent-capability-settings.test.mjs index 7db21f9fdb..5b306a9944 100644 --- a/apps/desktop/test/agent-capability-settings.test.mjs +++ b/apps/desktop/test/agent-capability-settings.test.mjs @@ -55,7 +55,7 @@ test("skills and MCP filter one list by level instead of stacking two sections", } assert.doesNotMatch(layout, /AgentCapabilitySection|AgentCapabilityColumn/); assert.match(layout, /agent-capability-list/); - assert.match(layout, /role="radiogroup"/); + assert.match(layout, / { }); test("the workbench reuses the shared segmented control instead of a third copy", () => { - assert.match(layout, /"settings-segment", "agent-capability-segment"|settings-segment agent-capability-segment/); - assert.match(layout, /"settings-segment-item"/); + assert.match(layout, / { + // ADR 0073 §3 fixes the estimate at four Unicode code points per token, so a + // surrogate pair counts once rather than twice. + assert.equal(sampleTokensForMessage({ content: "abcd", thinking: "" }), 1); + assert.equal(sampleTokensForMessage({ content: "", thinking: "abcd" }), 1); + assert.equal(sampleTokensForMessage({ content: "ab", thinking: "cd" }), 1); + assert.equal(sampleTokensForMessage({ content: "😀😀😀😀", thinking: "" }), 1); + assert.equal(sampleTokensForMessage({ content: "", thinking: "" }), 0); + assert.equal(sampleTokensForMessage(undefined), 0); +}); + +test("pushing prunes samples outside the window and never mutates the input", () => { + const first = pushThroughputSample([], { ts: 1_000, tokens: 10 }); + const second = pushThroughputSample(first, { ts: 2_000, tokens: 40 }); + assert.deepEqual(first, [{ ts: 1_000, tokens: 10 }]); + + // The 1_000 sample falls outside a window that ends at 4_500. + const pruned = pushThroughputSample(second, { + ts: 1_000 + LIVE_THROUGHPUT_WINDOW_MS + 500, + tokens: 90, + }); + assert.deepEqual( + pruned.map((sample) => sample.ts), + [2_000, 4_500], + ); +}); + +test("pushing keeps one sample already older than the window as the baseline", () => { + // Pruning must not drop every earlier sample, or a slow stream would never + // span enough time to produce a rate. + const samples = pushThroughputSample([{ ts: 0, tokens: 5 }], { + ts: LIVE_THROUGHPUT_WINDOW_MS * 2, + tokens: 25, + }); + assert.equal(samples.length, 2); + assert.equal(samples[0].ts, 0); +}); + +test("a measurement needs a span and a positive token delta", () => { + assert.equal(windowedTokenRate([]), undefined); + assert.equal(windowedTokenRate([{ ts: 1_000, tokens: 10 }]), undefined); + + const tooShort = [ + { ts: 1_000, tokens: 10 }, + { ts: 1_000 + LIVE_THROUGHPUT_MIN_SPAN_MS - 100, tokens: 60 }, + ]; + assert.equal(windowedTokenRate(tooShort), undefined); + + // A stalled window reports nothing rather than dividing a zero delta. + assert.equal( + windowedTokenRate([ + { ts: 1_000, tokens: 40 }, + { ts: 3_000, tokens: 40 }, + ]), + undefined, + ); +}); + +test("the measurement divides the windowed delta by the windowed span", () => { + assert.equal( + windowedTokenRate([ + { ts: 1_000, tokens: 100 }, + { ts: 3_000, tokens: 260 }, + ]), + 80, + ); +}); + +test("the measurement ignores growth before the window and stays integral", () => { + // Only the last two samples sit inside the window, so the early burst must + // not inflate the live figure. + const rate = windowedTokenRate([ + { ts: 1_000, tokens: 0 }, + { ts: 7_500, tokens: 5_000 }, + { ts: 10_000, tokens: 5_050 }, + ]); + assert.equal(rate, 20); + assert.equal(Number.isInteger(rate), true); +}); + +test("a fresh measurement wins; nothing shows before the first one", () => { + assert.deepEqual(retainLiveRate(80, undefined, 5_000), { + rate: 80, + stale: false, + }); + assert.deepEqual(retainLiveRate(80, { rate: 20, at: 0 }, 5_000), { + rate: 80, + stale: false, + }); + assert.deepEqual(retainLiveRate(undefined, undefined, 5_000), { + stale: false, + }); +}); + +test("a brief gap holds the figure steady before dimming it", () => { + const remembered = { rate: 80, at: 1_000 }; + assert.deepEqual( + retainLiveRate(undefined, remembered, 1_000 + LIVE_THROUGHPUT_STALE_MS - 100), + { rate: 80, stale: false }, + ); + assert.deepEqual( + retainLiveRate(undefined, remembered, 1_000 + LIVE_THROUGHPUT_STALE_MS + 100), + { rate: 80, stale: true }, + ); +}); + +test("only a growing sample counts as generation", () => { + assert.equal(sampleDidGrow([]), false); + assert.equal(sampleDidGrow([{ ts: 0, tokens: 10 }]), false); + assert.equal( + sampleDidGrow([ + { ts: 0, tokens: 10 }, + { ts: 250, tokens: 10 }, + ]), + false, + ); + assert.equal( + sampleDidGrow([ + { ts: 0, tokens: 10 }, + { ts: 250, tokens: 11 }, + ]), + true, + ); +}); + +test("a long tool call retains the rate and resumed generation gets a fresh baseline", () => { + let tracker = { samples: [] }; + let view; + const tick = (now, tokens, generating) => { + ({ tracker, view } = advanceThroughput(tracker, { + id: "a", content: "x".repeat(tokens * 4), status: "streaming", + }, generating, now)); + }; + for (let now = 0; now <= 2000; now += LIVE_THROUGHPUT_SAMPLE_MS) tick(now, now / 25, true); + assert.deepEqual(view, { rate: 40, stale: false }); + for (let now = 2250; now <= 12000; now += LIVE_THROUGHPUT_SAMPLE_MS) { + tick(now, 80, false); + assert.deepEqual(view, { rate: 40, stale: true }); + } + tick(12250, 80, true); + assert.deepEqual(view, { rate: 40, stale: true }); + for (let now = 12500; now <= 14000; now += LIVE_THROUGHPUT_SAMPLE_MS) { + tick(now, 80 + (now - 12250) / 10, true); + } + assert.deepEqual(view, { rate: 100, stale: false }); +}); + +test("smoothing is cadence-independent and damps a speed jump", () => { + const once = smoothTokenRate(40, 100, 500); + const twice = smoothTokenRate(smoothTokenRate(40, 100, 250), 100, 250); + assert.ok(Math.abs(once - twice) < 1e-9); + assert.ok(once > 40 && once < 100); +}); + +test("tool gaps and message replacement never enter the next generation window", () => { + let tracker = { samples: [] }; + const msg = (id, tokens) => ({ id, content: "x".repeat(tokens * 4), status: "streaming" }); + for (let now = 0; now <= 1000; now += 250) { + tracker = advanceThroughput(tracker, msg("a", now / 25), true, now).tracker; + } + assert.equal(tracker.remembered.rate, 40); + const tools = advanceThroughput(tracker, msg("a", 40), false, 1250); + assert.deepEqual(tools.view, { rate: 40, stale: true }); + const restart = advanceThroughput(tools.tracker, msg("b", 400), true, 20000); + assert.equal(restart.tracker.samples.length, 1); + assert.equal(restart.tracker.smoothed, undefined); + tracker = restart.tracker; + for (let now = 20250; now <= 21000; now += 250) { + tracker = advanceThroughput(tracker, msg("b", 400 + (now - 20000) / 50), true, now).tracker; + } + assert.equal(tracker.remembered.rate, 20); +}); + +test("thinking to text keeps the same token and time baseline", () => { + let tracker = { samples: [] }; + for (let now = 0; now <= 1000; now += 250) { + const message = { id: "a", thinking: "x".repeat(Math.min(now, 500) / 25 * 4), content: "x".repeat(Math.max(now - 500, 0) / 25 * 4), status: "streaming" }; + tracker = advanceThroughput(tracker, message, true, now).tracker; + } + assert.equal(tracker.remembered.rate, 40); +}); + +test("generation labels distinguish empty, thinking, text and running tools", () => { + const message = { id: "a", content: "", status: "streaming" }; + assert.equal(generationPhase(message, false), "waiting"); + assert.equal(generationPhase({ ...message, thinking: "reason" }, false), "thinking"); + assert.equal(generationPhase({ ...message, content: "answer" }, false), "generating"); + assert.equal(generationPhase({ ...message, content: "answer", status: "complete" }, false), "waiting"); + assert.equal(generationPhase(message, true), "tool"); +}); diff --git a/apps/desktop/test/main-process-errors.test.mjs b/apps/desktop/test/main-process-errors.test.mjs index e7deaf2344..bebf64f5d7 100644 --- a/apps/desktop/test/main-process-errors.test.mjs +++ b/apps/desktop/test/main-process-errors.test.mjs @@ -4,6 +4,7 @@ import test from "node:test"; import { classifyMainProcessError, describeMainProcessError, + describeError, installMainProcessErrorHandlers, isNonAsciiHttpHeaderError, reportMainProcessError, @@ -113,3 +114,9 @@ test("Electron main installs handlers and does not use Electron's default dialog /process\.on\("uncaughtException"/, ); }); + +test("user-facing error descriptions preserve the 300-character limit", () => { + assert.equal(describeError(new Error("x".repeat(400))), "x".repeat(300)); + assert.equal(describeError("y".repeat(400)), "y".repeat(300)); + assert.equal(describeError(null), "null"); +}); diff --git a/apps/desktop/test/plan-mode-source-contract.test.mjs b/apps/desktop/test/plan-mode-source-contract.test.mjs index c23b3af9ed..40113e8604 100644 --- a/apps/desktop/test/plan-mode-source-contract.test.mjs +++ b/apps/desktop/test/plan-mode-source-contract.test.mjs @@ -44,8 +44,8 @@ test("renderer exposes Agent, Plan, and Goal as the only operating modes", () => assert.match(composerSource, /settings\.modeGoal/); assert.match(composerSource, /IconListChecks/); assert.match(composerSource, /IconTarget/); - assert.match(settingsSource, /\["plan", "settings\.modePlan"\]/); - assert.match(settingsSource, /\["goal", "settings\.modeGoal"\]/); + assert.match(settingsSource, /value: "plan", label: t\("settings\.modePlan"\)/); + assert.match(settingsSource, /value: "goal", label: t\("settings\.modeGoal"\)/); assert.match(commandsSource, /case "builtin\.mode\.plan"/); assert.match(commandsSource, /case "builtin\.mode\.goal"/); for (const source of [composerSource, settingsSource, commandsSource]) { diff --git a/apps/desktop/test/segmented-control-rendering.test.mjs b/apps/desktop/test/segmented-control-rendering.test.mjs new file mode 100644 index 0000000000..a1defd8593 --- /dev/null +++ b/apps/desktop/test/segmented-control-rendering.test.mjs @@ -0,0 +1,33 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { fileURLToPath } from "node:url"; +import { createElement } from "react"; +import { renderToStaticMarkup } from "react-dom/server"; +import { createServer } from "vite"; + +test("tab controls preserve panel links independently of translated labels", async () => { + const server = await createServer({ + root: fileURLToPath(new URL("..", import.meta.url)), + configFile: false, + server: { middlewareMode: true, hmr: false, ws: false }, + esbuild: { jsx: "automatic" }, + appType: "custom", + optimizeDeps: { noDiscovery: true, include: [] }, + }); + try { + const { SegmentedControl } = await server.ssrLoadModule("/src/components/ui.tsx"); + for (const label of ["Import settings", "导入设置"]) { + const html = renderToStaticMarkup(createElement(SegmentedControl, { + value: "sessions", onChange() {}, label, role: "tablist", + options: [ + { value: "sessions", label: "Sessions", id: "import-tab-sessions", controls: "import-panel-sessions" }, + { value: "models", label: "Models", id: "import-tab-models", controls: "import-panel-models" }, + ], + })); + assert.match(html, /id="import-tab-sessions"[^>]*aria-selected="true"[^>]*aria-controls="import-panel-sessions"/); + assert.match(html, /id="import-tab-models"[^>]*aria-selected="false"[^>]*aria-controls="import-panel-models"/); + } + } finally { + await server.close(); + } +}); diff --git a/apps/desktop/test/settings-general.test.mjs b/apps/desktop/test/settings-general.test.mjs index dc784da570..e6395bcecd 100644 --- a/apps/desktop/test/settings-general.test.mjs +++ b/apps/desktop/test/settings-general.test.mjs @@ -105,7 +105,9 @@ test("Basics and AI tabs expose their respective app and AI controls", () => { '{tab === "shortcuts" && settings && (', ); const generalSource = settingsPageSource.slice(generalStart, aiStart); - const aiSource = settingsPageSource.slice(aiStart, shortcutsStart); + const voiceStart = settingsPageSource.indexOf('tab === "voice"', aiStart); + assert.ok(voiceStart > aiStart && voiceStart < shortcutsStart); + const aiSource = settingsPageSource.slice(aiStart, voiceStart); assert.match(generalSource, / { // The AI tab keeps the Settings picker control: a native { assert.match(sharedTypesSource, /developerMode\?: boolean/); assert.match(settingsPageSource, /function DeveloperSection/); - assert.match(settingsPageSource, /role="switch"/); + assert.match(settingsPageSource, / { // One tab strip owns the four kinds; each kind gets one panel behind it. assert.match(page, /role="tablist"/); - assert.match(page, /aria-controls={`import-panel-\$\{entry\.id\}`}/); + assert.match(page, /id: `import-tab-\$\{entry\.id\}`/); + assert.match(page, /controls: `import-panel-\$\{entry\.id\}`/); assert.match(page, /aria-labelledby={`import-tab-\$\{entry\.id\}`}/); // One toolbar and one idle state per kind — not per section or per step. diff --git a/apps/desktop/test/settings-project-archive.test.mjs b/apps/desktop/test/settings-project-archive.test.mjs index fc18489382..f8cc19565e 100644 --- a/apps/desktop/test/settings-project-archive.test.mjs +++ b/apps/desktop/test/settings-project-archive.test.mjs @@ -104,8 +104,8 @@ test("project archive is a toolbar over a list, with no page-level prose", () => assert.match(projectsPageSource, /project\.clearSearch/); assert.match(projectsPageSource, /projects-result-count[^]*aria-live="polite"/); assert.match(projectsPageSource, /project\.resultCount/); - assert.match(projectsPageSource, /"settings-segment projects-sort"/); - assert.match(projectsPageSource, /aria-pressed=\{sort === mode\}/); + assert.match(projectsPageSource, / { assert.match(page, /role="tablist"/); - assert.match(page, /aria-controls={`remote-host-add-panel-\$\{mode\}`}/); + for (const mode of ["ssh", "pair"]) { + assert.ok(page.includes(`id: "remote-host-add-${mode}"`)); + assert.ok(page.includes(`controls: "remote-host-add-panel-${mode}"`)); + assert.ok(page.includes(`aria-labelledby="remote-host-add-${mode}"`)); + } assert.match(page, /id={`remote-host-add-panel-\$\{mode\}`}|id="remote-host-add-panel-ssh"/); assert.match(page, /hidden=\{addMode !== "ssh"\}/); assert.match(page, /hidden=\{addMode !== "pair"\}/); diff --git a/apps/desktop/test/transcript-style.test.mjs b/apps/desktop/test/transcript-style.test.mjs index ba69fb0646..763e84344b 100644 --- a/apps/desktop/test/transcript-style.test.mjs +++ b/apps/desktop/test/transcript-style.test.mjs @@ -487,13 +487,42 @@ test("assistant context inspector keeps a compact summary and retry action wired assert.match(stylesSource, /\.context-inspector-popover\.is-open/); }); -test("context inspector keeps generation speed completion-only", () => { - assert.doesNotMatch(inspectorSource, /useLiveElapsedMs|usageThroughputLive/); +test("the composer inspector keeps generation speed completion-only", () => { + // The popover reports the provider's exact figure, so it stays a + // completed-turn surface; only the transcript meta row estimates live. + assert.doesNotMatch(inspectorSource, /useLiveElapsedMs|useLiveThroughput/); + assert.doesNotMatch(inspectorSource, /usageThroughputLive/); + // The reverted runtime-stamped attempt must not come back (ADR 0258). assert.doesNotMatch(transcriptSource, /useLiveElapsedMs|usageThroughputLive/); - assert.doesNotMatch(transcriptSource, /assistantTurnStreamingMessage/); assert.doesNotMatch(stylesSource, /message-meta-live-rate|live-rate-pulse/); }); +test("the transcript meta row estimates throughput while streaming (#93)", () => { + // Mounted only for the active tail turn, so the sampler and its interval stay + // off history rows and no per-token state reaches the store (ADR 0242). + assert.match(transcriptSource, /export function LiveMessageMeta\(/); + assert.match(transcriptSource, /useLiveThroughput\(message, generating\)/); + // The displayed figure is gated on growth, so a window straddling the moment + // output stopped cannot make the number sag across a long tool call. + assert.match(transcriptSource, /advanceThroughput\(\s*trackerRef\.current,/); + assert.match( + transcriptSource, + /isActive \? \(\s* { // The trigger toggles; pointer enter/leave and focus/blur no longer open or // close the panel, so no hover-grace timer is needed. diff --git a/crates/host-core/src/plugin_sessions.rs b/crates/host-core/src/plugin_sessions.rs index fd09e435a4..799cb06fa6 100644 --- a/crates/host-core/src/plugin_sessions.rs +++ b/crates/host-core/src/plugin_sessions.rs @@ -240,6 +240,7 @@ fn parse_message( provider_id, usage: None, response_duration_ms: None, + time_to_first_token_ms: None, response_output_tokens: None, error: None, revision_root_id: None, diff --git a/crates/host-core/src/sessions.rs b/crates/host-core/src/sessions.rs index d0b9c5d1e4..c364970b9e 100644 --- a/crates/host-core/src/sessions.rs +++ b/crates/host-core/src/sessions.rs @@ -170,6 +170,9 @@ pub struct UiMessage { /// Elapsed model streaming time for the response throughput statistic. #[serde(skip_serializing_if = "Option::is_none")] pub response_duration_ms: Option, + /// Delay until the first visible model output, captured by the runtime. + #[serde(skip_serializing_if = "Option::is_none")] + pub time_to_first_token_ms: Option, /// Partial output estimate used when a user stops before final usage arrives. #[serde(skip_serializing_if = "Option::is_none")] pub response_output_tokens: Option, @@ -315,6 +318,9 @@ pub(crate) fn ui_to_record(message: &UiMessage) -> (MessageRecord, Option= 0) { + meta_obj.insert("timeToFirstTokenMs".into(), json!(latency)); + } if let Some(tokens) = message.response_output_tokens { meta_obj.insert("responseOutputTokens".into(), json!(tokens)); } @@ -464,6 +470,10 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { let error = meta.get("error").cloned(); let response_duration_ms = meta.get("responseDurationMs").and_then(|v| v.as_i64()); let response_output_tokens = meta.get("responseOutputTokens").and_then(|v| v.as_i64()); + let time_to_first_token_ms = meta + .get("timeToFirstTokenMs") + .and_then(|v| v.as_i64()) + .filter(|value| *value >= 0); let revision_root_id = meta .get("revisionRootId") .and_then(|v| v.as_str()) @@ -543,6 +553,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { provider_id, usage, response_duration_ms, + time_to_first_token_ms, response_output_tokens, error, revision_root_id, @@ -592,6 +603,7 @@ pub(crate) fn record_to_ui(record: MessageRecord) -> UiMessage { provider_id, usage, response_duration_ms, + time_to_first_token_ms, response_output_tokens, error, revision_root_id, @@ -3792,6 +3804,7 @@ mod tests { provider_id: None, usage: None, response_duration_ms: None, + time_to_first_token_ms: None, response_output_tokens: None, error: None, revision_root_id: None, @@ -4436,6 +4449,7 @@ mod tests { provider_id: None, usage: None, response_duration_ms: None, + time_to_first_token_ms: None, response_output_tokens: None, error: None, revision_root_id: None, @@ -4872,6 +4886,7 @@ mod tests { total_tokens: 48, }), response_duration_ms: Some(2_000), + time_to_first_token_ms: Some(1_250), response_output_tokens: Some(34), error: None, revision_root_id: None, @@ -4928,6 +4943,7 @@ mod tests { assert_eq!(usage.reasoning_tokens, Some(5)); assert_eq!(usage.total_tokens, 48); assert_eq!(detail.messages[0].response_duration_ms, Some(2_000)); + assert_eq!(detail.messages[0].time_to_first_token_ms, Some(1_250)); assert_eq!(detail.messages[0].response_output_tokens, Some(34)); } @@ -4937,6 +4953,7 @@ mod tests { let session = create_session(&db, None, None, None, None, None).unwrap(); let assistant = UiMessage { id: "assistant-search-1".into(), + time_to_first_token_ms: None, role: "assistant".into(), content: "answer with sources".into(), attachments: None, diff --git a/docs/adr/README.md b/docs/adr/README.md index a1729f3fa7..596fc8c4aa 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -340,4 +340,7 @@ Each ADR includes: | provider-display-order | [Provider display order](provider-display-order.md) | Accepted | | registry-header-variable-spelling | [Remote header variables accept the registry's `{name}` spelling](registry-header-variable-spelling.md) | Proposed | | provider-system-certificates | [Desktop sidecar uses OS-trusted certificates](provider-system-certificates.md) | Accepted | +| live-turn-throughput-estimate | [Live turn throughput is a renderer-side windowed estimate](live-turn-throughput-estimate.md) | Accepted | +| first-output-latency | [Runtime-owned first-output latency](first-output-latency.md) | Accepted | + | image-generation-capability | [Image generation as a configured Agent capability](image-generation-capability.md) | Accepted | diff --git a/docs/adr/first-output-latency.md b/docs/adr/first-output-latency.md new file mode 100644 index 0000000000..22749b931b --- /dev/null +++ b/docs/adr/first-output-latency.md @@ -0,0 +1,45 @@ +# ADR: Runtime-owned first-output latency + +- Status: Accepted +- Date: 2026-09-15 +- Related: ADR 0073, ADR 0242, ADR `live-turn-throughput-estimate` + +## Context + +Completed throughput does not explain the wait before the first visible output. +Renderer mount times cannot measure that wait consistently across reloads or +tool-loop continuations, so the request owner must provide the timing. + +## Decision + +The parent agent runtime measures one logical model request from entering its +stream function until the first non-empty text or visible thinking output. +It uses a monotonic clock, not renderer mount time or the assistant start event. +Transport retries inside the same stream are included; a subsequent logical +request (including a tool-loop continuation or runtime recovery) resets the +anchor. Retained content from a previous recovery attempt is not first output. +A tool-call-only response has no visible first-output metric. The number is +client-observed latency, including transport and queuing, not server-only TTFT. + +An optional nonnegative integer `UiMessage.timeToFirstTokenMs` carries the +measurement through the constant-size delta identity and completed snapshot. +Rust stores it in existing message metadata alongside response duration. +As in ADR 0073, this is an additive optional JSON field: neither table layout +nor schema/protocol version changes, no migration is needed, and older records +omit the readout. No historical timing is inferred from transcript timestamps. + +The active and completed meta rows show seconds to one decimal place. A logical +turn with several model calls displays the latest assistant message's value, +never a sum or average. Subagent-native timing is outside this initial scope. +Existing TPS sampling and provider diagnostic timing keep their meaning. + +## Consequences + +The readout survives reloads without changing existing database layouts. It +measures client-observed latency; provider-internal timing remains unknown. + +## Validation + +Timer, runtime event, delta-coalescing, and durable metadata round-trip tests; +E2E-CHAT-first-output-latency plus transcript and protocol smoke suites after +main integration. A label without a measured value stays absent. diff --git a/docs/adr/live-turn-throughput-estimate.md b/docs/adr/live-turn-throughput-estimate.md new file mode 100644 index 0000000000..223396566e --- /dev/null +++ b/docs/adr/live-turn-throughput-estimate.md @@ -0,0 +1,108 @@ +# ADR `live-turn-throughput-estimate`: Live turn throughput is a renderer-side windowed estimate + +- Status: Accepted +- Date: 2026-09-15 +- Deciders: PI-Desktop desktop UI maintainers +- Amends: 0073 +- Related: ADR 0047 · ADR 0242 (D412) · + [04-ux/08-component-spec](../spec/04-ux/08-component-spec.md) · + [04-ux/09-interaction-patterns](../spec/04-ux/09-interaction-patterns.md) · + E2E-CHAT-live-generation-throughput · issue #93 (duplicate #394) + +## Context + +Generation speed exists only after a turn settles. ADR 0073 made the stopped +case durable by preserving `responseDurationMs` and an optional +`responseOutputTokens` estimate, and the composer inspector divides one by the +other. While a turn runs there is no readout at all, so a user cannot tell a +model that is streaming slowly from one that has stalled, or from a long tool +call during which the model is not running. + +The runtime offers no incremental token source. Provider usage is read exactly +once, at `message_end`; the `message_update` event carries text and thinking +deltas only. Any figure shown during the turn is therefore an estimate, not a +measurement. + +A previous attempt shipped and was reverted the same day (`d3beca92`, +`2c8ad1ff`, 2026-08-02). It stamped `responseDurationMs` onto every streaming +`message_update` inside the runtime and displayed a cumulative average taken +from the start of the stream. That predates ADR 0242, which replaced broadcast +message snapshots with coalesced deltas and established that per-token data +must not reach store subscribers. + +## Decision + +1. The live figure is computed in the renderer. No runtime, protocol, IPC, + storage, or store-schema change: `PROTOCOL_VERSION` stays at 11. +2. Estimated output tokens reuse ADR 0073 §3 — visible thinking plus answer + text at four Unicode code points per token — through the same + `estimateResponseOutputTokens` helper the stopped path uses. Sampling is + throttled rather than run per delta, because the estimate walks the message. +3. The rate is measured across a recent window from its endpoints, not + cumulatively from the start of the stream. A cumulative average folds + tool-execution wall clock into the denominator, so it decays after every + long tool call and misreports a model that was never running. +4. Silence is reported as staleness, not as zero. Once no tokens have arrived + for the stale interval the last measured rate is retained and dimmed. A rate + appears only after the samples span a minimum interval. +5. The figure always uses the estimated copy (`chat.usageThroughputEstimated`), + honouring ADR 0073 §4: exact provider usage wins, and an estimate is + labelled as one. When the turn settles the meta row shows the completed-turn + values and the live chip is gone. +6. It renders in the transcript meta row of the active turn, and is mounted + only for that turn. The composer inspector stays a completed-turn surface. + The sample window lives in that component's ref. + +## Consequences + +- A glanceable speed readout during thinking, streaming, and the generation + around tool calls, without waiting for the turn to end. +- The number is approximate and labelled as such. It tracks the recent window, + so it reacts to a slowdown instead of averaging it away. +- ADR 0242's boundaries hold: because the window is a ref inside the active + turn and no store state is added, token growth cannot re-render the sidebar, + and mounting only for the active turn keeps the sampler and its interval off + history rows. +- Mount and unmount coincide with turn start and end, so the window needs no + explicit lifecycle and cannot leak across turns or sessions. +- Phase and historical-rate labels are localized in all eight shipped catalogs. +- Native rendering of the chip is a visual behaviour that unit tests cannot + prove; E2E-CHAT-live-generation-throughput owns that acceptance. + +## Rejected alternatives + +### Stamp streaming duration in the runtime (the reverted approach) + +Rejected. It couples a presentation concern to the agent runtime and writes +per-delta metadata that every peer must carry, and its cumulative average is +the behaviour this ADR exists to avoid. ADR 0242's delta pipeline makes the +renderer-side estimate feasible without touching the runtime at all. + +### Keep one live rate in the store + +Rejected. A sample per coalesced flush would notify every store subscriber, +which is precisely the re-render ADR 0242 removed. A ref confines the churn to +the component that displays it. + +### Cumulative average from stream start + +Rejected. It is stable but answers the wrong question: it cannot distinguish a +model that slowed down from a turn that spent a minute inside `Bash`. + +### Blank the chip during tool execution + +Rejected. A chip that disappears and returns changes the row height and moves +the content the user is reading, which is the class of jitter issue #323 tracks. +Dimming keeps the layout stable and still signals that nothing is generating. + + +## Refinement: phase-aware display and smoothing + +Sampling runs in a component-owned effect at 250 ms intervals using +`performance.now()`. Committed message props feed the timer; render no longer +mutates the sampler. The pure tracker resets on message identity changes and +non-generation phases. A 750 ms time-based exponential filter damps window-rate +jitter without depending on delta arrival frequency. Thinking and text share +the same baseline. Existing tool lifecycle state makes the retained rate +historical immediately; waiting, thinking, generating, and tool labels explain +the current phase. No provider TTFT is claimed from renderer timing. diff --git a/docs/spec/03-runtime/01-ipc-protocol.md b/docs/spec/03-runtime/01-ipc-protocol.md index bfc1d5f8cb..45cf4417da 100644 --- a/docs/spec/03-runtime/01-ipc-protocol.md +++ b/docs/spec/03-runtime/01-ipc-protocol.md @@ -2289,6 +2289,12 @@ semantics are unchanged. Native terminal completion follows SDK settlement, not intermediate retry/compaction loop ends. Native abort never invokes `replaceSessionMessages` and reloads durable detail after abort returns. +### First-output latency + +`UiMessage.timeToFirstTokenMs` is optional runtime-measured milliseconds. +Stream snapshots and coalesced deltas preserve it; legacy messages omit it. +See ADR `first-output-latency.md` for measurement semantics. + ### Provider ordering `pi-desktop/providers/reorder({ id, targetId, placement: "before" | "after" })` diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index 829b7efe28..6ceb1b3659 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -1587,6 +1587,14 @@ format, repair-needing newline, unavailable provider/auth, active lease, or external byte change makes continuation fail closed while detail remains browseable. +### First-output latency + +The runtime measures `UiMessage.timeToFirstTokenMs` from logical model request +start to the first visible text/thinking output using a monotonic clock. The +value latches once per request, includes transport retries and waiting, and +excludes preceding tool execution. Tool-only output leaves it absent. +See ADR `first-output-latency.md`. + ### Provider certificate trust (issue #714) The desktop sidecar starts with Node's `--use-system-ca`, retaining bundled diff --git a/docs/spec/03-runtime/04-data-storage.md b/docs/spec/03-runtime/04-data-storage.md index 85caae73f3..3533f58548 100644 --- a/docs/spec/03-runtime/04-data-storage.md +++ b/docs/spec/03-runtime/04-data-storage.md @@ -1594,6 +1594,12 @@ The first slice has no projection cache or async scan bound; every list still reads/parses complete files. Caching by canonical path/file identity/size/mtime and bounded asynchronous scanning remain deferred performance work. +### First-output latency + +Rust host-core persists optional `UiMessage.timeToFirstTokenMs` in existing +message JSON metadata and restores nonnegative integer values. Legacy rows +omit it; no schema migration is required. See ADR `first-output-latency.md`. + ### Provider display order `kv(ns="app", key="providers.order")` stores an ordered array of provider IDs. diff --git a/docs/spec/04-ux/07-ui-design-system.md b/docs/spec/04-ux/07-ui-design-system.md index 30832c084c..5c2d87a43c 100644 --- a/docs/spec/04-ux/07-ui-design-system.md +++ b/docs/spec/04-ux/07-ui-design-system.md @@ -1236,7 +1236,7 @@ Implementation: `components/ui.tsx → SegmentedControl`. | Roles | `radiogroup` (default), `group`, or `tablist` | | Item roles | `radio` / none / `tab` — derived from container role | | Generic | `` for type-safe value/onChange | -| Options | `readonly { value: T; label: ReactNode }[]` — label accepts JSX (e.g. count badge) | +| Options | `readonly { value: T; label: ReactNode; id?: string; controls?: string }[]` — label accepts JSX (e.g. count badge) | Every multi-option selector rendered as a row of equal buttons **must** use `SegmentedControl`. Inline `
` with manual diff --git a/docs/spec/04-ux/08-component-spec.md b/docs/spec/04-ux/08-component-spec.md index 6b686916e9..59073d41ae 100644 --- a/docs/spec/04-ux/08-component-spec.md +++ b/docs/spec/04-ux/08-component-spec.md @@ -1471,6 +1471,27 @@ storage but compose into one assistant turn until the next user message. default scale). User messages retain their real hover-action row and normal spacing, including before the first assistant output. Opacity hides those buttons without removing their space, so hover does not shift content. +- While the turn is still streaming, that meta row shows the model chip beside + a live generation-speed chip in tokens per second (ADR `live-turn-throughput-estimate`). The figure is + always the estimate form, because the provider reports usage only when the + message ends, and it is measured over a recent window rather than from the + start of the turn. Tool execution produces no tokens, so the last measured + rate is retained and dimmed instead of blanked; a chip appears only once the + samples span enough time to be meaningful. When the turn settles, the meta + row switches to the completed-turn values. +- The live meta row labels waiting, thinking, generating, and tool execution. + A retained rate is labelled "Last" immediately during tools or waiting, and + after 1.5 seconds without output. Rates are approximate whole tokens/s. + Sampling runs every 250 ms on a monotonic clock over a three-second window, + with a 750 ms exponential smoothing time constant. Each new message resets + the window and smoothing baseline while retaining the prior rate for display; + thinking-to-answer transitions share one baseline. TPS is renderer-only. +- First-output latency shows optional runtime-measured `timeToFirstTokenMs` in + seconds to one decimal place. It includes request waiting and transport + retries, excludes preceding tool time, and survives completion and reload. + Stopped thinking-only responses retain this value. During tools and waiting + for the next request, the live row hides prior latency. Old messages show no + guessed value (ADR `first-output-latency.md`). - Toggle Thinking disclosure: expand/collapse reasoning independently from the final answer. The latest reasoning row opens while it streams and closes when the turn settles only if the user has not interacted with it. The expanded @@ -1813,8 +1834,9 @@ Single message render — either user (plaintext) or assistant (markdown streami estimate note are intentionally omitted from the default view. Rows below the heading share one muted-label / tabular-value rhythm separated by spacing; the popover keeps its floating-layer edge and draws no inner - section rules (D297). Generation speed is a completed-turn value in tokens - per second and is not updated while a response is streaming. The + section rules (D297). The popover's generation speed is a completed-turn + value in tokens per second and is not updated while a response is streaming; + the transcript meta row carries the live estimate instead (ADR `live-turn-throughput-estimate`). The context-window total uses the same effective model window as the agent sidecar: a published models.dev `limit.context` replaces a legacy 128k generic binding seed, while a non-default per-model Advanced value remains diff --git a/docs/spec/04-ux/09-interaction-patterns.md b/docs/spec/04-ux/09-interaction-patterns.md index a46ad44f42..8fd283edea 100644 --- a/docs/spec/04-ux/09-interaction-patterns.md +++ b/docs/spec/04-ux/09-interaction-patterns.md @@ -730,8 +730,13 @@ may be retained while exactly one workspace supplies the visible shell context. with aborted status and restore no draft. Preserve the measured stream duration and use provider output usage when available; otherwise store a visibly estimated output count so the conversation still shows throughput -6. Composer re-activates (unblocked) -7. Abort is idempotent — pressing abort when already aborting does nothing +6. The live generation-speed chip stops at abort and the settled turn's own + values take over; no live figure survives into history (ADR `live-turn-throughput-estimate`) +7. Composer re-activates (unblocked) +8. Abort is idempotent — pressing abort when already aborting does nothing + +Live phase labels and retained-rate behavior follow the transcript meta row +contract in `08-component-spec.md` (ADR `live-turn-throughput-estimate`). ### 3.3 Abort UX diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 9ad5f6648e..42d01e5cdf 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -8345,7 +8345,7 @@ must keep splitting are covered by `markdown-blocks.test.mjs`. | A — App startup | E2E-001, E2E-002, E2E-003, E2E-004, E2E-067, E2E-076, E2E-079, E2E-092, E2E-097, E2E-143, E2E-150, E2E-168, E2E-204 | | A / C / F / Quality — Tray session navigation | E2E-TRAY-bounded-session-navigation | | B — Model config | E2E-005, E2E-006, E2E-007, E2E-038, E2E-050, E2E-052, E2E-055, E2E-066, E2E-080, E2E-082, E2E-102c, E2E-102d, E2E-102e, E2E-151, E2E-154, E2E-163, E2E-166, E2E-172, E2E-174, E2E-197, E2E-005G, E2E-005J, E2E-199, E2E-201, E2E-202, E2E-203, E2E-205, E2E-206, E2E-209 | -| C — Conversation & stream | E2E-CHAT-running-status-survives-output-pauses, E2E-008, E2E-008d, E2E-008e, E2E-008a, E2E-009, E2E-010, E2E-011, E2E-011a, E2E-011b, E2E-011d, E2E-011e, E2E-011g, E2E-031, E2E-040, E2E-047, E2E-048, E2E-048A, E2E-049, E2E-052, E2E-053, E2E-054, E2E-055, E2E-059, E2E-059a, E2E-060c, E2E-060d, E2E-061, E2E-061a, E2E-062, E2E-064, E2E-065, E2E-068, E2E-071, E2E-073, E2E-074, E2E-075, E2E-081, E2E-083, E2E-084, E2E-086, E2E-087, E2E-088, E2E-088b, E2E-089, E2E-090, E2E-COMPOSER-narrow-controls, E2E-094, E2E-095, E2E-096, E2E-097, E2E-098, E2E-099, E2E-102, E2E-102a, E2E-102b, E2E-102c, E2E-102d, E2E-102g, E2E-106, E2E-109, E2E-111, E2E-114, E2E-116, E2E-117, E2E-118, E2E-119, E2E-120, E2E-121, E2E-218, E2E-259, E2E-219, E2E-AGENTS-001, E2E-142, E2E-144, E2E-145, E2E-146, E2E-146a, E2E-147, E2E-151, E2E-154, E2E-155, E2E-158, E2E-159, E2E-161, E2E-162, E2E-166, E2E-172, E2E-173, E2E-174, E2E-177, E2E-178, E2E-179, E2E-180, E2E-182, E2E-183, E2E-187, E2E-198, E2E-199, E2E-202, E2E-203, E2E-207, E2E-208, E2E-CHAT-content-width-handles, E2E-250, E2E-102i, E2E-PLUGIN-session-orchestrator-real-workers, E2E-SUBAGENT-settlement-updates-before-parent-poll, E2E-SUBAGENT-resume-a-settled-delegation | +| C — Conversation & stream | E2E-CHAT-running-status-survives-output-pauses, E2E-008, E2E-008d, E2E-008e, E2E-008a, E2E-009, E2E-010, E2E-011, E2E-011a, E2E-011b, E2E-011d, E2E-011e, E2E-011g, E2E-031, E2E-040, E2E-047, E2E-048, E2E-048A, E2E-049, E2E-052, E2E-053, E2E-054, E2E-055, E2E-059, E2E-059a, E2E-060c, E2E-060d, E2E-061, E2E-061a, E2E-062, E2E-064, E2E-065, E2E-068, E2E-071, E2E-073, E2E-074, E2E-075, E2E-081, E2E-083, E2E-084, E2E-086, E2E-087, E2E-088, E2E-088b, E2E-089, E2E-090, E2E-COMPOSER-narrow-controls, E2E-094, E2E-095, E2E-096, E2E-097, E2E-098, E2E-099, E2E-102, E2E-102a, E2E-102b, E2E-102c, E2E-102d, E2E-102g, E2E-106, E2E-109, E2E-111, E2E-114, E2E-116, E2E-117, E2E-118, E2E-119, E2E-120, E2E-121, E2E-218, E2E-259, E2E-219, E2E-AGENTS-001, E2E-142, E2E-144, E2E-145, E2E-146, E2E-146a, E2E-147, E2E-151, E2E-154, E2E-155, E2E-158, E2E-159, E2E-161, E2E-162, E2E-166, E2E-172, E2E-173, E2E-174, E2E-177, E2E-178, E2E-179, E2E-180, E2E-182, E2E-183, E2E-187, E2E-198, E2E-199, E2E-202, E2E-203, E2E-207, E2E-208, E2E-CHAT-content-width-handles, E2E-250, E2E-102i, E2E-PLUGIN-session-orchestrator-real-workers, E2E-SUBAGENT-settlement-updates-before-parent-poll, E2E-SUBAGENT-resume-a-settled-delegation, E2E-CHAT-live-generation-throughput, E2E-CHAT-first-output-latency | | C — Conversation & stream (composer drafts) | E2E-011c, E2E-011c-1 | | D — Workspace | E2E-012, E2E-013, E2E-022B, E2E-024I, E2E-047, E2E-049, E2E-057, E2E-058, E2E-060, E2E-068, E2E-075, E2E-078, E2E-153, E2E-158, E2E-182, E2E-187, E2E-252 | | D — Workspace (project ordering) | E2E-253 | @@ -8355,7 +8355,7 @@ must keep splitting are covered by `markdown-blocks.test.mjs`. | G — Plugins | E2E-022, E2E-022A, E2E-022B, E2E-022C, E2E-023, E2E-024, E2E-024B, E2E-024C, E2E-024D, E2E-024AA, E2E-024E, E2E-024W, E2E-024F, E2E-024G, E2E-024H, E2E-024I, E2E-024J, E2E-024K, E2E-024L, E2E-024M, E2E-024N, E2E-024O, E2E-024P, E2E-025, E2E-026, E2E-105, E2E-117, E2E-120, E2E-122, E2E-123, E2E-024Q, E2E-148, E2E-152, E2E-153, E2E-PLUGIN-imported-pi-package-skills, E2E-PLUGIN-imported-pi-package-wrapper, E2E-PLUGIN-import-extension-installs-dependencies, E2E-PLUGIN-import-extension-reports-missing-dependency, E2E-PLUGIN-global-shortcut-owns-only-its-own-command, E2E-PLUGIN-permission-gate-for-real-time-capabilities, E2E-PLUGIN-background-audio-and-realtime-connection, E2E-PLUGIN-fs-root-follows-the-calling-session | | H — Diagnostics | E2E-027, E2E-031, E2E-034, E2E-042, E2E-096, E2E-098, E2E-104, E2E-107, E2E-108, E2E-109, E2E-110, E2E-113, E2E-115, E2E-116, E2E-118, E2E-121, E2E-146, E2E-146a, E2E-155, E2E-159, E2E-176, E2E-194, E2E-195 | | Security | E2E-028, E2E-029, E2E-030, E2E-024J, E2E-024K, E2E-024M, E2E-049, E2E-068, E2E-086, E2E-102c, E2E-102d, E2E-102e, E2E-105, E2E-106, E2E-107, E2E-108, E2E-109, E2E-110, E2E-112, E2E-113, E2E-115, E2E-116, E2E-117, E2E-119, E2E-121, E2E-122, E2E-123, E2E-142, E2E-148, E2E-151, E2E-153, E2E-158, E2E-187, E2E-196c, E2E-196b, E2E-196, E2E-PLUGIN-fs-root-follows-the-calling-session | -| Quality | E2E-CHAT-running-status-survives-output-pauses, E2E-032, E2E-033, E2E-039, E2E-043, E2E-044, E2E-045, E2E-046, E2E-047, E2E-048, E2E-048A, E2E-049, E2E-050, E2E-053, E2E-055, E2E-056, E2E-057, E2E-058, E2E-059, E2E-060, E2E-061, E2E-062, E2E-063, E2E-064, E2E-065, E2E-066, E2E-067, E2E-068, E2E-069, E2E-070, E2E-071, E2E-072, E2E-073, E2E-074, E2E-075, E2E-076, E2E-077, E2E-078, E2E-079, E2E-080, E2E-081, E2E-082, E2E-083, E2E-084, E2E-085, E2E-086, E2E-092, E2E-093, E2E-094, E2E-095, E2E-096, E2E-097, E2E-098, E2E-099, E2E-100, E2E-101, E2E-102, E2E-102a, E2E-102b, E2E-102c, E2E-102d, E2E-102e, E2E-103, E2E-AGENTS-001, E2E-021a, E2E-024N, E2E-059a, E2E-060b, E2E-060c, E2E-061a, E2E-073a, E2E-111, E2E-114, E2E-117, E2E-118, E2E-119, E2E-120, E2E-122, E2E-123, E2E-142, E2E-143, E2E-144, E2E-145, E2E-146, E2E-147, E2E-148, E2E-150, E2E-151, E2E-153, E2E-155, E2E-158, E2E-159, E2E-160, E2E-161, E2E-162, E2E-163, E2E-168, E2E-172, E2E-173, E2E-174, E2E-011g, E2E-176, E2E-177, E2E-178, E2E-179, E2E-180, E2E-181, E2E-182, E2E-183, E2E-186, E2E-187, E2E-194, E2E-195, E2E-196a, E2E-196b, E2E-196c, E2E-198, E2E-199, E2E-200, E2E-196, E2E-201, E2E-204, E2E-202, E2E-203, E2E-205, E2E-206, E2E-207, E2E-208, E2E-209, E2E-210, E2E-218, E2E-259, E2E-219, E2E-250, E2E-252, E2E-102i, E2E-SUBAGENT-settlement-updates-before-parent-poll, E2E-PLUGIN-imported-pi-package-skills, E2E-PLUGIN-fs-root-follows-the-calling-session, E2E-SUBAGENT-resume-a-settled-delegation | +| Quality | E2E-CHAT-running-status-survives-output-pauses, E2E-032, E2E-033, E2E-039, E2E-043, E2E-044, E2E-045, E2E-046, E2E-047, E2E-048, E2E-048A, E2E-049, E2E-050, E2E-053, E2E-055, E2E-056, E2E-057, E2E-058, E2E-059, E2E-060, E2E-061, E2E-062, E2E-063, E2E-064, E2E-065, E2E-066, E2E-067, E2E-068, E2E-069, E2E-070, E2E-071, E2E-072, E2E-073, E2E-074, E2E-075, E2E-076, E2E-077, E2E-078, E2E-079, E2E-080, E2E-081, E2E-082, E2E-083, E2E-084, E2E-085, E2E-086, E2E-092, E2E-093, E2E-094, E2E-095, E2E-096, E2E-097, E2E-098, E2E-099, E2E-100, E2E-101, E2E-102, E2E-102a, E2E-102b, E2E-102c, E2E-102d, E2E-102e, E2E-103, E2E-AGENTS-001, E2E-021a, E2E-024N, E2E-059a, E2E-060b, E2E-060c, E2E-061a, E2E-073a, E2E-111, E2E-114, E2E-117, E2E-118, E2E-119, E2E-120, E2E-122, E2E-123, E2E-142, E2E-143, E2E-144, E2E-145, E2E-146, E2E-147, E2E-148, E2E-150, E2E-151, E2E-153, E2E-155, E2E-158, E2E-159, E2E-160, E2E-161, E2E-162, E2E-163, E2E-168, E2E-172, E2E-173, E2E-174, E2E-011g, E2E-176, E2E-177, E2E-178, E2E-179, E2E-180, E2E-181, E2E-182, E2E-183, E2E-186, E2E-187, E2E-194, E2E-195, E2E-196a, E2E-196b, E2E-196c, E2E-198, E2E-199, E2E-200, E2E-196, E2E-201, E2E-204, E2E-202, E2E-203, E2E-205, E2E-206, E2E-207, E2E-208, E2E-209, E2E-210, E2E-218, E2E-259, E2E-219, E2E-250, E2E-252, E2E-102i, E2E-SUBAGENT-settlement-updates-before-parent-poll, E2E-PLUGIN-imported-pi-package-skills, E2E-PLUGIN-fs-root-follows-the-calling-session, E2E-SUBAGENT-resume-a-settled-delegation, E2E-CHAT-live-generation-throughput, E2E-CHAT-first-output-latency | | Quality (project ordering) | E2E-253 | | C — Conversation & stream (IME slash alias) | E2E-255 | | E — Tools & permissions (Skill residency) | E2E-254 | @@ -8406,7 +8406,7 @@ must keep splitting are covered by `markdown-blocks.test.mjs`. | M2 (IME slash alias) | E2E-255 | | M5 (Skill residency) | E2E-254 | | M6 | E2E-104, E2E-105, E2E-106, E2E-107, E2E-108, E2E-109, E2E-110, E2E-111, E2E-112, E2E-113, E2E-114, E2E-115, E2E-116, E2E-117, E2E-118, E2E-119, E2E-120, E2E-103, E2E-172 | -| M6+ | E2E-121, E2E-122, E2E-148, E2E-150, E2E-151, E2E-154, E2E-155, E2E-158, E2E-159, E2E-160, E2E-161, E2E-162, E2E-163, E2E-166, E2E-168, E2E-173, E2E-174, E2E-176, E2E-179, E2E-196a, E2E-196b, E2E-196c, E2E-198, E2E-199, E2E-200, E2E-202, E2E-203, E2E-205, E2E-209, E2E-210, E2E-212, E2E-213, E2E-214, E2E-215, E2E-216, E2E-217, E2E-218, E2E-259, E2E-219, E2E-257, E2E-SUBAGENT-settlement-updates-before-parent-poll, E2E-PLUGIN-fs-root-follows-the-calling-session, E2E-SUBAGENT-resume-a-settled-delegation | +| M6+ | E2E-121, E2E-122, E2E-148, E2E-150, E2E-151, E2E-154, E2E-155, E2E-158, E2E-159, E2E-160, E2E-161, E2E-162, E2E-163, E2E-166, E2E-168, E2E-173, E2E-174, E2E-176, E2E-179, E2E-196a, E2E-196b, E2E-196c, E2E-198, E2E-199, E2E-200, E2E-202, E2E-203, E2E-205, E2E-209, E2E-210, E2E-212, E2E-213, E2E-214, E2E-215, E2E-216, E2E-217, E2E-218, E2E-259, E2E-219, E2E-257, E2E-SUBAGENT-settlement-updates-before-parent-poll, E2E-PLUGIN-fs-root-follows-the-calling-session, E2E-SUBAGENT-resume-a-settled-delegation, E2E-CHAT-live-generation-throughput, E2E-CHAT-first-output-latency | | M6+ (Session Orchestrator) | E2E-PLUGIN-session-orchestrator-real-workers | | M6+ (Selected model order) | E2E-MODEL-selected-order-persists | | M6+ (Session list responsiveness) | E2E-SESSION-list-refresh-keeps-desktop-responsive | @@ -14142,6 +14142,44 @@ plugin-form fixtures in an isolated temporary directory at runtime. `apps/desktop/test/queued-turn-finalization.test.mjs`); desktop journey Draft (run only in a capable environment when this surface changes) +#### E2E-CHAT-live-generation-throughput: The streaming meta row estimates tokens/s and dims through tool calls + +- **Preconditions**: A provider that streams slowly enough to read the chip + while it updates. One prompt that produces a long answer with visible + reasoning, and one that runs a tool for at least ten seconds before the model + resumes. Run once on macOS and once on Windows/Linux. +- **Steps**: + 1. Submit the long-answer prompt. Watch the active turn's meta row from the + first reasoning tokens through the end of the answer. + 2. Note whether the chip's width stays fixed as the number changes. + 3. Submit the tool prompt. Watch the chip across the tool call and after the + model resumes streaming. + 4. Let the turn settle, then open the composer context-usage popover. + 5. Submit another prompt and stop it mid-stream with `Cmd/Ctrl + .`. + 6. Scroll back through earlier turns in the same session. +- **Additional checks**: TPS displays whole tokens/s without fractional digits. + Waiting, thinking, generating, and running-tool labels + follow lifecycle state. Tool execution immediately marks the retained speed + as "Last". A new message resets the window and filter; thinking-to-text + transitions do not reset the denominator. Repeat with Chinese localization. +- **Expected**: The chip appears once enough output has streamed, always in the + estimated form, and updates while reasoning and answer text arrive. Its width + does not change as digits change. During the tool call the last rate is + retained and dimmed rather than blanked or removed, and it brightens when + generation resumes. After the turn settles the meta row shows the + completed-turn values with no live chip, and the popover's generation speed is + still the completed-turn figure. A stopped turn keeps its stopped-turn + metrics. History rows show no live chip. The sidebar does not re-render while + tokens arrive. +- **Specs linked**: `04-ux/08-component-spec.md`, + `04-ux/09-interaction-patterns.md` §3.2, ADR `live-turn-throughput-estimate`, ADR 0073, ADR 0242 +- **Acceptance**: C (conversation and stream), Quality (glanceable telemetry) +- **Milestone**: M6+ +- **Status**: Renderer automation in `pnpm test:e2e:transcript` covers phase + transitions and historical-rate labels. Module-covered + (`apps/desktop/test/live-throughput.test.mjs`, + `apps/desktop/test/transcript-style.test.mjs`); native journey Draft + ### E2E-SESSION-native-pi-continue-appends-original-jsonl - **Preconditions:** A synthetic Pi v3 session contains branches, compaction @@ -14288,6 +14326,30 @@ plugin-form fixtures in an isolated temporary directory at runtime. - **Acceptance**: B (model config), E (tools & permissions), F (persistence), G (plugins), Security, Quality - **Milestone**: Post-MVP (R7 v1) +- **Status**: Documented; automation pending + + + + +### E2E-CHAT-first-output-latency: Measured first output survives completion + +- **Preconditions:** A deterministic delayed stream with visible thinking, text, + and a tool-loop continuation; a legacy message without timing metadata. +- **Steps:** Wait before the first delta, stream thinking then text, finish the + message, run a tool, and start the next request. Reload the completed session. +- **Expected:** No number before first output. The first thinking/text event + fixes the latency; later text cannot overwrite it. The next request measures + its own delay. Completed/reloaded messages retain the recorded value and old + messages show no guessed value. Stopped thinking-only output retains latency, + including zero. A waiting tool-loop continuation hides prior response latency. + Display uses seconds with one decimal place. +- **Specs:** 03-runtime/01-ipc-protocol, 03-runtime/04-data-storage, + 04-ux/08-component-spec, ADR first-output-latency. +- **Acceptance:** C (conversation), data compatibility. +- **Milestone:** M6+. +- **Status:** Runtime/delta/storage tests plus `test:e2e:transcript` cover the + deterministic boundaries. The provider journey remains manual. + - **Status**: Partially automated (`pnpm test:e2e:trusted-extensions`): the declared row materializing as `plugin::` in the native provider list with its `ownerPluginId`, endpoint, models, and thinking diff --git a/docs/spec/08-meta/decisions-log.md b/docs/spec/08-meta/decisions-log.md index 62fb7b250a..bea3292102 100644 --- a/docs/spec/08-meta/decisions-log.md +++ b/docs/spec/08-meta/decisions-log.md @@ -5447,6 +5447,33 @@ It deliberately does not include plugin OAuth: the `provider.oauth` permission and a Host-owned plugin login flow are future work, so a declared provider has no OAuth login, token refresh, or account label today. +### Live generation throughput (issue #93) + +[ADR `live-turn-throughput-estimate`](/adr/live-turn-throughput-estimate) amends ADR 0073: the active +turn's meta row shows a live tokens-per-second estimate beside the model chip. +The runtime reports provider usage only at `message_end`, so the figure is +computed in the renderer from visible thinking plus answer text at four Unicode +code points per token — ADR 0073 §3's convention — and always uses the +estimated copy. It is measured across a recent window rather than cumulatively, +so a long tool call cannot make a running model look slow; silence retains and +dims the last rate instead of reporting zero. The sample window is a ref inside +the active turn, so no per-token state reaches the store and ADR 0242's +memoization boundaries hold. Renderer-only: protocol stays at 11. The completed-turn values and the composer inspector are +unchanged. Validation contract: E2E-CHAT-live-generation-throughput. + + +### Live generation feedback refinement + +ADR `live-turn-throughput-estimate` now specifies explicit generation phases, historical-rate labels, +time-based smoothing, and per-message sampling baselines. All eight locales +carry the new labels. E2E-CHAT-live-generation-throughput covers the rendered +phase transitions; no IPC, persistence, or plugin contract changes. + +### Runtime-owned first-output latency + +ADR `first-output-latency.md` adds optional per-response timing, measured before +IPC and retained in existing message metadata. TPS semantics are unchanged. + ## 2026-09-15 — Side chats materialize on first Send (#421) *(Retired by ADR 0268 on 2026-09-16: the feature and its surfaces were removed, and these IDs stay retired and must not be reused. The record is kept for history.)* diff --git a/docs/zh-CN/spec/04-ux/07-ui-design-system.md b/docs/zh-CN/spec/04-ux/07-ui-design-system.md index 90e638c261..8ace3f3bbe 100644 --- a/docs/zh-CN/spec/04-ux/07-ui-design-system.md +++ b/docs/zh-CN/spec/04-ux/07-ui-design-system.md @@ -1066,6 +1066,78 @@ Linux 保留淡入淡出和滑动退出。 | 运动 | 进入200ms缓出slide-down/fade,退出150ms缓入淡入淡出;减少运动 → 接近零持续时间(不是 `none`,移除监听 `animationend`) | | Z 指数 | z-Toast (50) | +### 11.9 SettingsToggle + +实现: `components/ui.tsx → SettingsToggle`. + +| 属性 | 值 | +|---|---| +| Size | 32×20, thumb 16px | +| CSS class | `.settings-toggle` / `.settings-toggle.on` | +| Role | `role="switch"` with `aria-checked` | +| Variants | default, `busy` (`.is-busy`, `aria-busy`, disabled) | +| Background | neutral accent when on (not green); theme-specific override in `theme-overrides.css` | + +设置页和编辑面板的布尔开关**必须**使用 `SettingsToggle`,禁止手写 +`