Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added
- [EE] Added prompt caching for Ask Sourcebot. For Anthropic models, the static prompt prefix (tool definitions, system prompt, and conversation history) is marked with a cache breakpoint so it is billed at the provider's discounted cache-read rate on subsequent agent steps and follow-up turns. Toggle with `SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED` (default `true`). [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)
- [EE] Added a cached-token breakdown to the Ask Sourcebot message details, showing what share of the input tokens were served from the model provider's prompt cache. [#1278](https://github.com/sourcebot-dev/sourcebot/pull/1278)

### Fixed
- Upgraded `protobufjs` to `^7.6.2`. [#1281](https://github.com/sourcebot-dev/sourcebot/pull/1281)

Expand Down
1 change: 1 addition & 0 deletions packages/shared/src/env.server.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -283,6 +283,7 @@ const options = {
*/
SOURCEBOT_CHAT_MODEL_TEMPERATURE: numberSchema.optional(),
SOURCEBOT_CHAT_MAX_STEP_COUNT: numberSchema.default(100),
SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED: booleanSchema.default('true'),
SOURCEBOT_MCP_TOOL_CALL_TIMEOUT_MS: numberSchema.int().positive().max(maxTimerDelayMs).default(60000),

DEBUG_WRITE_CHAT_MESSAGES_TO_FILE: booleanSchema.default('false'),
Expand Down
35 changes: 34 additions & 1 deletion packages/web/src/ee/features/chat/agent.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -197,6 +197,8 @@ export const createMessageStream = async ({
totalTokens: (priorMetadata?.totalTokens ?? 0) + (totalUsage.totalTokens ?? 0),
totalInputTokens: (priorMetadata?.totalInputTokens ?? 0) + (totalUsage.inputTokens ?? 0),
totalOutputTokens: (priorMetadata?.totalOutputTokens ?? 0) + (totalUsage.outputTokens ?? 0),
totalCacheReadTokens: (priorMetadata?.totalCacheReadTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheReadTokens ?? 0),
totalCacheWriteTokens: (priorMetadata?.totalCacheWriteTokens ?? 0) + (totalUsage.inputTokenDetails?.cacheWriteTokens ?? 0),
totalResponseTimeMs: (priorMetadata?.totalResponseTimeMs ?? 0) + (new Date().getTime() - startTime.getTime()),
modelName,
traceId,
Expand DownExpand Up@@ -343,11 +345,42 @@ const createAgentStream = async ({
...(hasMcpTools ? { tool_request_activation: toolRequestActivation, ...mcpToolSetsObj.tools } : {}),
};

// Anthropic prompt caching: mark the end of the prompt's static prefix —
// tool definitions, the system prompt (including any resolved file sources),
// and the conversation history — with an ephemeral (5m) cache breakpoint on
// the last input message. Anthropic caches everything up to and including
// this point, so the large prefix is written once (~1.25x) and read back at
// ~0.1x on every subsequent agent step and follow-up turn instead of being
// reprocessed in full. The `anthropic` provider-options namespace is ignored
// by non-Anthropic providers, so this is safe to apply unconditionally.
//
// Caveat: when MCP tools are lazily activated mid-run via prepareStep, the
// tools section (which precedes everything else in the prefix) grows and
// invalidates the cache for that step; the cache re-warms on subsequent
// steps once the active tool set is stable.
const isPromptCachingEnabled = env.SOURCEBOT_CHAT_PROMPT_CACHING_ENABLED === 'true';
const messagesWithCachedPrefix: ModelMessage[] = inputMessages.map((message, index) => {
if (!isPromptCachingEnabled || index !== inputMessages.length - 1) {
return message;
}

return {
...message,
providerOptions: {
...message.providerOptions,
anthropic: {
...message.providerOptions?.anthropic,
cacheControl: { type: 'ephemeral' },
Comment thread
brendan-kellam marked this conversation as resolved.
},
},
};
});

try {
const stream = streamText({
model,
providerOptions,
messages: inputMessages,
messages: messagesWithCachedPrefix,
system: systemPrompt,
tools: allTools,
activeTools: [
Expand Down
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,12 @@ const DetailsCardComponent = ({
(part.type === 'dynamic-tool' && part.toolName.startsWith('mcp_'))
).length, [thinkingSteps]);

const cacheReadTokens = metadata?.totalCacheReadTokens ?? 0;
const inputTokens = metadata?.totalInputTokens ?? 0;
const cachedInputPercent = inputTokens > 0
? Math.round((cacheReadTokens / inputTokens) * 100)
: 0;

const handleExpandedChanged = useCallback((next: boolean) => {
captureEvent('wa_chat_details_card_toggled', { chatId, isExpanded: next });
onExpandedChanged(next);
Expand DownExpand Up@@ -127,30 +133,44 @@ const DetailsCardComponent = ({
</div>
)}
{metadata?.totalTokens && (
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center text-xs cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
<div className="flex items-center gap-1.5 text-xs">
<Tooltip>
<TooltipTrigger asChild>
<div className="flex items-center cursor-help">
<Zap className="w-3 h-3 mr-1 flex-shrink-0" />
{getShortenedNumberDisplayString(metadata.totalTokens, 0)} tokens
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="space-y-1 text-xs">
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Input</span>
<span>{metadata.totalInputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4">
<span className="text-muted-foreground">Output</span>
<span>{metadata.totalOutputTokens?.toLocaleString() ?? '—'}</span>
</div>
<div className="flex justify-between gap-4 border-t border-border pt-1">
<span className="text-muted-foreground">Total</span>
<span>{metadata.totalTokens.toLocaleString()}</span>
</div>
</div>
</div>
</TooltipContent>
</Tooltip>
</TooltipContent>
</Tooltip>
{cachedInputPercent > 0 && (
<Tooltip>
<TooltipTrigger asChild>
<span className="text-muted-foreground cursor-help">({cachedInputPercent}% cached)</span>
</TooltipTrigger>
<TooltipContent side="bottom">
<div className="max-w-xs text-xs">
{cacheReadTokens.toLocaleString()} of {inputTokens.toLocaleString()} input tokens were read from the model provider prompt cache. Cached tokens are typically billed at a fraction of the cost of regular input tokens, so the real cost is lower than the token count suggests.
</div>
</TooltipContent>
</Tooltip>
)}
</div>
)}
{metadata?.totalResponseTimeMs && (
<div className="flex items-center text-xs">
Expand Down
3 changes: 3 additions & 0 deletions packages/web/src/features/chat/types.ts
Original file line numberDiff line numberDiff line change
Expand Up@@ -55,6 +55,9 @@ export const sbChatMessageMetadataSchema = z.object({
totalInputTokens: z.number().optional(),
totalOutputTokens: z.number().optional(),
totalTokens: z.number().optional(),
// Portion of input tokens served from / written to the prompt cache.
totalCacheReadTokens: z.number().optional(),
totalCacheWriteTokens: z.number().optional(),
totalResponseTimeMs: z.number().optional(),
feedback: z.array(z.object({
type: z.enum(['like', 'dislike']),
Expand Down
Loading