Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions py/src/braintrust/cli/eval.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -138,6 +138,7 @@ async def run_evaluator_task(evaluator, position, opts: EvaluatorOpts):
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand Down
14 changes: 14 additions & 0 deletions py/src/braintrust/framework.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -359,6 +359,11 @@ class Evaluator(Generic[Input, Output]):
JSON-serializable type, but its keys must be strings.
"""

tags: list[str] | None = None
"""
Optional list of tags for the experiment
"""

trial_count: int = 1
"""
The number of times to run the evaluator per input. This is useful for evaluating applications that
Expand DownExpand Up@@ -654,6 +659,7 @@ def _EvalCommon(
experiment_name: str | None,
trial_count: int,
metadata: Metadata | None,
tags: list[str] | None,
is_public: bool,
update: bool,
reporter: ReporterDef[Input, Output, EvalReport] | None,
Expand DownExpand Up@@ -695,6 +701,7 @@ def _EvalCommon(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
timeout=timeout,
Expand DownExpand Up@@ -743,6 +750,7 @@ async def make_empty_summary():
experiment_name=evaluator.experiment_name,
description=evaluator.description,
metadata=evaluator.metadata,
tags=evaluator.tags,
is_public=evaluator.is_public,
update=evaluator.update,
base_experiment=base_experiment_name,
Expand DownExpand Up@@ -780,6 +788,7 @@ async def EvalAsync(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -833,6 +842,7 @@ async def EvalAsync(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -869,6 +879,7 @@ async def EvalAsync(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand DownExpand Up@@ -904,6 +915,7 @@ def Eval(
experiment_name: str | None = None,
trial_count: int = 1,
metadata: Metadata | None = None,
tags: list[str] | None = None,
is_public: bool = False,
update: bool = False,
reporter: ReporterDef[Input, Output, EvalReport] | None = None,
Expand DownExpand Up@@ -957,6 +969,7 @@ def Eval(
anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log
the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata`
can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) A list of tags to associate with the experiment.
:param is_public: (Optional) Whether the experiment should be public. Defaults to false.
:param reporter: (Optional) A reporter that takes an evaluator and its result and returns a report.
:param timeout: (Optional) The duration, in seconds, after which to time out the evaluation.
Expand DownExpand Up@@ -994,6 +1007,7 @@ def Eval(
experiment_name=experiment_name,
trial_count=trial_count,
metadata=metadata,
tags=tags,
is_public=is_public,
update=update,
reporter=reporter,
Expand Down
10 changes: 10 additions & 0 deletions py/src/braintrust/logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -1506,6 +1506,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1529,6 +1530,7 @@ def init(
api_key: str | None = ...,
org_name: str | None = ...,
metadata: Metadata | None = ...,
tags: list[str] | None = ...,
git_metadata_settings: GitMetadataSettings | None = ...,
set_current: bool = ...,
update: bool | None = ...,
Expand All@@ -1551,6 +1553,7 @@ def init(
api_key: str | None = None,
org_name: str | None = None,
metadata: Metadata | None = None,
tags: list[str] | None = None,
git_metadata_settings: GitMetadataSettings | None = None,
set_current: bool = True,
update: bool | None = None,
Expand All@@ -1575,6 +1578,7 @@ def init(
key is specified, will prompt the user to login.
:param org_name: (Optional) The name of a specific organization to connect to. This is useful if you belong to multiple.
:param metadata: (Optional) a dictionary with additional data about the test example, model outputs, or just about anything else that's relevant, that you can use to help find and analyze examples later. For example, you could log the `prompt`, example's `id`, or anything else that would be useful to slice/dice later. The values in `metadata` can be any JSON-serializable type, but its keys must be strings.
:param tags: (Optional) a list of strings to tag the experiment with. Tags can be used to filter and organize experiments.
:param git_metadata_settings: (Optional) Settings for collecting git metadata. By default, will collect all git metadata fields allowed in org-level settings.
:param set_current: If true (the default), set the global current-experiment to the newly-created one.
:param open: If the experiment already exists, open it in read-only mode. Throws an error if the experiment does not already exist.
Expand DownExpand Up@@ -1623,6 +1627,9 @@ def compute_metadata():
lazy_metadata = LazyValue(compute_metadata, use_mutex=True)
return ReadonlyExperiment(lazy_metadata=lazy_metadata, state=state)

if tags is not None:
validate_tags(tags)

# pylint: disable=function-redefined
def compute_metadata():
state.login(org_name=org_name, api_key=api_key, app_url=app_url)
Expand DownExpand Up@@ -1676,6 +1683,9 @@ def compute_metadata():
if metadata is not None:
args["metadata"] = metadata

if tags is not None:
args["tags"] = tags

while True:
try:
response = state.app_conn().post_json("api/experiment/register", args)
Expand Down
8 changes: 8 additions & 0 deletions py/src/braintrust/test_logger.py
Original file line numberDiff line numberDiff line change
Expand Up@@ -59,6 +59,14 @@ def test_init_validation(self):

assert str(cm.exception) == "Cannot open an experiment without specifying its name"

# duplicate tags
tag = "Exp1"
with self.assertRaises(ValueError) as cm:
braintrust.init(project="project", tags=[tag, tag])

assert str(cm.exception) == f"duplicate tag: {tag}"


def test_init_with_dataset_id_only(self):
"""Test that init accepts dataset={'id': '...'} parameter"""
# Test the logic that extracts dataset_id from the dict
Expand Down
Loading