Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- Added per-step token cost tracking and estimated tool call token usage to Ask Sourcebot chat history. [#1353](https://github.com/sourcebot-dev/sourcebot/pull/1353)

## [5.0.4] - 2026-06-18

### Changed
Expand Down
3 changes: 2 additions & 1 deletion packages/web/src/ee/features/chat/agent.test.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -137,7 +137,8 @@ const createAssistantMessage = (parts: SBChatMessagePart[]): SBChatMessage => ({
});

const createFakeStreamResult = () => ({
response: Promise.resolve(new Response()),
response: Promise.resolve({ messages: [] }),
steps: Promise.resolve([]),
totalUsage: Promise.resolve({
inputTokens: 1,
outputTokens: 1,
Expand Down
69 changes: 67 additions & 2 deletions packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
@@ -1,4 +1,5 @@
import { SBChatMessage, SBChatMessageMetadata } from "@/features/chat/types";
import { SBChatMessage, SBChatMessageMetadata, StepTokenUsageEntry, ToolTokenUsageEntry } from "@/features/chat/types";
import { estimateModelToolOutputTokens } from "@/ee/features/chat/tokenEstimation";
import { getFileSource } from '@/features/git';
import { isServiceError } from "@/lib/utils";
import { LanguageModelV3 as AISDKLanguageModelV3 } from "@ai-sdk/provider";
Expand DownExpand Up@@ -190,19 +191,76 @@ export const createMessageStream = async ({
});

const totalUsage = await researchStream.totalUsage;
const steps = await researchStream.steps;
const response = await researchStream.response;

// Tool output estimates are derived from `response.messages` rather
// than per-step `toolResults` because the response messages cover
// tool calls that never run inside a step — approval-gated tools
// execute before the step loop, and thrown tool errors are recorded
// as `tool-error` parts that `toolResults` excludes. Their
// `tool-result` parts also carry the output in model-visible form
// (`toModelOutput` already applied), which is exactly the payload
// whose token footprint we want to estimate.
const toolUsageByToolCallId = new Map<string, ToolTokenUsageEntry>(
response.messages.flatMap((message) =>
message.role !== 'tool' ? [] : message.content.flatMap((part) =>
part.type !== 'tool-result' ? [] : [[part.toolCallId, {
toolCallId: part.toolCallId,
toolName: part.toolName,
estimatedOutputTokens: estimateModelToolOutputTokens(part.output),
}] as const]
)
)
);

// One entry per step, in step order. The UI joins its step groups
// to these entries by array position, so the order and count must
// mirror the stream's steps exactly. Tool calls nest under the
// step they ran in; `content` is matched rather than `toolResults`
// so that thrown tool errors (`tool-error` parts, which
// `toolResults` excludes) are still attributed to their step.
const stepTokenUsage: StepTokenUsageEntry[] = steps.map(({ usage, content }) => ({
inputTokens: usage.inputTokens,
outputTokens: usage.outputTokens,
cacheReadTokens: usage.inputTokenDetails?.cacheReadTokens,
tools: content.flatMap((part) => {
if (part.type !== 'tool-result' && part.type !== 'tool-error') {
return [];
}
const entry = toolUsageByToolCallId.get(part.toolCallId);
if (!entry) {
return [];
}
toolUsageByToolCallId.delete(part.toolCallId);
return [entry];
}),
}));

// Any estimates left unclaimed belong to tool calls that executed
// before the step loop (approval continuations). Their output
// enters the context as input to this phase's first step, so nest
// them under it.
if (toolUsageByToolCallId.size > 0 && stepTokenUsage.length > 0) {
stepTokenUsage[0].tools.unshift(...toolUsageByToolCallId.values());
}

writer.write({
type: 'message-metadata',
messageMetadata: {
// Spread first so the derived fields below can't be overwritten by caller metadata.
...metadata,
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
// Concatenated (not summed) across approval-continuation
// phases so earlier phases' steps are preserved in order.
stepTokenUsage: [...(priorMetadata?.stepTokenUsage ?? []), ...stepTokenUsage],
modelName,
traceId,
...metadata,
}
});

Expand DownExpand Up@@ -430,6 +488,13 @@ const createAgentStream = async ({
logger.warn(`Tool call repair failed for "${toolCall.toolName}": ${error.message}`);
return null;
},
// Token usage collection deliberately does NOT happen here: the SDK
// awaits this callback before starting the next step, so it must
// stay cheap, and `toolResults` misses tool calls that never run
// inside a step (approval-gated tools execute before the step loop)
// as well as thrown tool errors (recorded as `tool-error` parts).
// Both are instead derived post-stream in `createMessageStream`
// from `steps` and `response.messages`.
onStepFinish: ({ toolResults }) => {
toolResults.forEach(({ output, dynamic }) => {
if (dynamic || isServiceError(output)) {
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -91,33 +91,57 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
// should be visible to the user. By "steps", we mean parts that originated
// from the same LLM invocation. By "visibile", we mean parts that have some
// visual representation in the UI (e.g., text, reasoning, tool calls, etc.).
const uiVisibleThinkingSteps = useMemo(() => {
const steps = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Filter out the answerPart and empty steps
return steps
.map(
(step) => step
// First, filter out any parts that are not text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
})
)
//
// Each step is tagged with its stepIndex — the invocation's position in
// the turn, which indexes into `metadata.stepTokenUsage`. Indices are
// assigned by counting 'step-start' markers (one per invocation) BEFORE
// any filtering, so dropping empty or answer-only steps below cannot
// shift the indices of the steps that remain.
const { uiVisibleThinkingSteps, answerStepIndex } = useMemo(() => {
const groupedParts = groupMessageIntoSteps(assistantMessage?.parts ?? []);

// Parts written before the first step-start (e.g. data parts) don't
// belong to any step; they get stepIndex -1 and never survive the
// visibility filters below.
let stepIndex = -1;
let answerStepIndex: number | undefined = undefined;

const steps = groupedParts
.map((stepParts) => {
if (stepParts[0]?.type === 'step-start') {
stepIndex++;
}

if (stepParts.some((part) => part.type === 'text' && part.text.includes(ANSWER_TAG))) {
answerStepIndex = stepIndex;
}

return {
stepIndex,
parts: stepParts
// First, filter out the answer text
.filter((part) => {
if (part.type === 'text') {
return !part.text.includes(ANSWER_TAG);
}

return true;
})
.filter((part) => {
// Only include text, reasoning, and tool parts
return (
part.type === 'text' ||
part.type === 'reasoning' ||
part.type.startsWith('tool-') ||
part.type === 'dynamic-tool'
)
}),
};
})
// Then, filter out any steps that are empty
.filter(step => step.length > 0);
.filter((step) => step.parts.length > 0);

return { uiVisibleThinkingSteps: steps, answerStepIndex };
}, [assistantMessage?.parts]);

// "thinking" is when the agent is generating output that is not the answer.
Expand DownExpand Up@@ -379,6 +403,7 @@ const ChatThreadListItemComponent = forwardRef<HTMLDivElement, ChatThreadListIte
isNetworkActive={isNetworkActive}
isAwaitingToolApproval={isAwaitingToolApproval}
thinkingSteps={uiVisibleThinkingSteps}
answerStepIndex={answerStepIndex}
metadata={assistantMessage?.metadata}
/>

Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -111,7 +111,7 @@ describe('DetailsCard', () => {
isTurnInProgress={true}
isNetworkActive={false}
isAwaitingToolApproval={false}
thinkingSteps={[[failedActivationPart]]}
thinkingSteps={[{ stepIndex: 0, parts: [failedActivationPart] }]}
/>
</TooltipProvider>
);
Expand Down
Loading
Loading