Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions examples/agno/README.md
Original file line numberDiff line numberDiff line change
Expand Up@@ -9,6 +9,8 @@ Each script calls `braintrust.auto_instrument()` before importing `agno`, so all
| `async_simple_agent_stream.py` | one agent, async + streamed |
| `team_agent.py` | research + advisor team, sync |
| `async_team_agent.py` | research + advisor team, async + streamed |
| `accuracy_eval.py` | `AccuracyEval` over one agent, scored in Braintrust |
| `eval_suite.py` | an `agno.eval` suite; each `Case` becomes an experiment row |

## Run

Expand Down
33 changes: 33 additions & 0 deletions examples/agno/accuracy_eval.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,33 @@
import braintrust


braintrust.auto_instrument()

# An eval run logs to whatever is current. Swap init_logger for
# braintrust.init(project=..., experiment=...) to score it as an experiment row instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval.accuracy import AccuracyEval
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="You are a stock price agent. Answer with the ticker and nothing else.",
)

evaluation = AccuracyEval(
name="Ticker Lookup",
model=OpenAIChat(id="gpt-4o-mini"),
agent=agent,
input="Which ticker does Figma trade under?",
expected_output="FIG",
num_iterations=2,
)

result = evaluation.run(print_summary=True)
print(f"average score: {result.avg_score}/10")
51 changes: 51 additions & 0 deletions examples/agno/eval_suite.py
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,51 @@
# agno.eval lazy-imports its submodules through a module-level __getattr__, which
# static analysis cannot see.
# pylint: disable=no-name-in-module

import sys

import braintrust


braintrust.auto_instrument()

# A suite run opens a Braintrust experiment of its own, so each Case lands as a
# scored experiment row. Pass eval_experiments=False to setup_agno() (or set
# BRAINTRUST_AGNO_EVAL_EXPERIMENTS=false) to keep suite runs in logs instead.
braintrust.init_logger(project="agno-evals-project")

from agno.agent import Agent
from agno.eval import Case, cli
from agno.models.openai import OpenAIChat
from agno.tools.yfinance import YFinanceTools


agent = Agent(
id="stock-agent",
name="Stock Price Agent",
model=OpenAIChat(id="gpt-4o-mini"),
tools=[YFinanceTools()],
instructions="Use your tools for any market data question.",
)

CASES = (
Case(
name="looks_up_current_price",
agent=agent,
input="What is the current price of FIG?",
tags=("smoke",),
criteria="Reports a current share price for Figma.",
expected_tool_calls=("get_current_stock_price",),
),
Case(
name="explains_pe_ratio",
agent=agent,
input="Explain the P/E ratio in one sentence.",
criteria="Explains that the P/E ratio compares share price to earnings per share.",
),
)

if __name__ == "__main__":
# python eval_suite.py --tag smoke # run a tagged subset
# python eval_suite.py --list # list cases without running them
sys.exit(cli(CASES))
31 changes: 26 additions & 5 deletions py/src/braintrust/integrations/agno/__init__.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -2,13 +2,19 @@

import logging

from braintrust.logger import NOOP_SPAN, current_span, init_logger
from braintrust.logger import NOOP_SPAN, current_experiment, current_span, init_logger

from .eval_experiments import configure as _configure_eval_experiments
from .integration import AgnoIntegration
from .patchers import (
wrap_accuracy_eval,
wrap_agent,
wrap_agent_as_judge_eval,
wrap_eval_suite,
wrap_function_call,
wrap_model,
wrap_performance_eval,
wrap_reliability_eval,
wrap_team,
wrap_workflow,
)
Expand All@@ -19,9 +25,14 @@
__all__ = [
"AgnoIntegration",
"setup_agno",
"wrap_accuracy_eval",
"wrap_agent",
"wrap_agent_as_judge_eval",
"wrap_eval_suite",
"wrap_function_call",
"wrap_model",
"wrap_performance_eval",
"wrap_reliability_eval",
"wrap_team",
"wrap_workflow",
]
Expand All@@ -31,20 +42,30 @@ def setup_agno(
api_key: str | None = None,
project_id: str | None = None,
project_name: str | None = None,
eval_experiments: bool | None = None,
) -> bool:
"""
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models, and function calls for tracing.
Setup Braintrust integration with Agno. Will automatically patch Agno agents, models,
function calls, and evals (``agno.eval``) for tracing.

Args:
api_key: Braintrust API key (optional, can use env var BRAINTRUST_API_KEY)
project_id: Braintrust project ID (optional)
project_name: Braintrust project name (optional, can use env var BRAINTRUST_PROJECT)
project_name: Braintrust project name (optional; defaults to the Global project)
eval_experiments: Whether an eval suite run should open a Braintrust experiment,
so its cases land as experiment rows rather than logs. Defaults to the
BRAINTRUST_AGNO_EVAL_EXPERIMENTS env var, which itself defaults to true.
Individual evals (AccuracyEval and friends) always log to whatever is
current, so pass eval_experiments=False to keep suite runs in logs too.

Returns:
True if setup was successful, False otherwise
"""
span = current_span()
if span == NOOP_SPAN:
_configure_eval_experiments(eval_experiments)

# An experiment opened by the caller is the destination for eval rows, so don't
# install a logger that would only shadow it for non-eval tracing.
if current_span() == NOOP_SPAN and current_experiment() is None:
init_logger(project=project_name, api_key=api_key, project_id=project_id)

return AgnoIntegration.setup()
Loading